From 66d064b07f3e7bc91ea29a1f9954eae9701884ec Mon Sep 17 00:00:00 2001 From: exveria1015 <42349717+exveria1015@users.noreply.github.com> Date: Sat, 5 Sep 2026 23:28:12 +0900 Subject: [PATCH 1/4] Configure Astra-only task-aware agent mode Route all nine child roles to GPT-6 Astra: D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max. Add bounded D4 synthesis roles while preserving read-only boundaries, one-level delegation, and the three-child limit. Update installers, validators, CI, and documentation; add opt-in Astra parent defaults and retire the Sol default option. Verified Bash and PowerShell-on-Linux installation, migration, strict config loading, and rejection checks. Live child model execution remains unverified. --- .github/workflows/ci.yml | 120 ++++++++++++++++++- CHANGELOG.md | 25 ++-- README.md | 180 ++++++++++++++++------------ RELEASE_CHECKLIST.md | 8 +- agents/astra-architect-max.toml | 27 +++++ agents/astra-architect.toml | 25 ++++ agents/luna-task-max.toml | 8 +- agents/luna-task-medium.toml | 23 ++++ agents/luna-task.toml | 7 +- agents/sol-specialist-max.toml | 11 +- agents/sol-specialist.toml | 4 +- agents/terra-worker-max.toml | 8 +- agents/terra-worker.toml | 4 +- config/AGENTS.task-aware.md | 92 ++++++++++---- config/config.task-aware.toml | 5 + scripts/Install-TaskAwareAgent.ps1 | 14 ++- scripts/Test-TaskAwareAgent.ps1 | 59 ++++++--- scripts/install-task-aware-agent.sh | 17 ++- scripts/test-task-aware-agent.sh | 91 +++++++++++--- 19 files changed, 555 insertions(+), 173 deletions(-) create mode 100644 agents/astra-architect-max.toml create mode 100644 agents/astra-architect.toml create mode 100644 agents/luna-task-medium.toml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 004a427..1c09084 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,7 +15,7 @@ permissions: contents: read env: - CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.145.0' }} + CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.149.1' }} jobs: windows: @@ -63,6 +63,66 @@ jobs: $backupCount = @(Get-ChildItem -Directory -LiteralPath (Join-Path $cleanHome 'task-aware-backups')).Count if ($backupCount -lt 2) { throw 'Repeated installation reused a backup directory.' } + $standardHome = Join-Path $env:RUNNER_TEMP 'task-aware-standard-defaults' + New-Item -ItemType Directory -Force -Path $standardHome | Out-Null + @( + 'model = "parent-model"' + 'model_reasoning_effort = "medium"' + 'approval_policy = "on-request"' + 'sandbox_mode = "workspace-write"' + ) | Set-Content -LiteralPath (Join-Path $standardHome 'config.toml') -Encoding utf8 + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $standardHome + $standard = Get-Content -Raw -LiteralPath (Join-Path $standardHome 'config.toml') + foreach ($expected in @( + 'model = "parent-model"', + 'model_reasoning_effort = "medium"', + 'approval_policy = "on-request"', + 'sandbox_mode = "workspace-write"' + )) { + $key = $expected.Split('=')[0].Trim() + if ([regex]::Matches($standard, "(?m)^\s*" + [regex]::Escape($key) + '\s*=').Count -ne 1 -or + $standard -notmatch [regex]::Escape($expected)) { + throw "Standard install changed $expected." + } + } + + $astraHome = Join-Path $env:RUNNER_TEMP 'task-aware-astra-default' + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $astraHome -SetAstraDefault + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $astraHome -SetAstraDefault + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $astraHome -ConfigOnlyRuntime + $astra = Get-Content -Raw -LiteralPath (Join-Path $astraHome 'config.toml') + foreach ($expected in @('model = "gpt-6-astra"', 'model_reasoning_effort = "xhigh"')) { + if ([regex]::Matches($astra, "(?m)^\s*" + [regex]::Escape($expected.Split('=')[0].Trim()) + '\s*=').Count -ne 1 -or + $astra -notmatch [regex]::Escape($expected)) { + throw "Astra default was not set exactly once: $expected." + } + } + + $retiredSolHome = Join-Path $env:RUNNER_TEMP 'task-aware-retired-sol' + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $retiredSolHome -SetSolDefault -ErrorAction Stop + } + catch { + $stopped = $_.Exception.Message -like '*was retired*' + } + if (-not $stopped -or (Test-Path -LiteralPath $retiredSolHome)) { + throw 'Retired Sol option did not fail before filesystem mutation.' + } + + $astraSolConflictHome = Join-Path $env:RUNNER_TEMP 'task-aware-astra-sol-conflict' + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $astraSolConflictHome ` + -SetAstraDefault -SetSolDefault -WhatIf -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped -or (Test-Path -LiteralPath $astraSolConflictHome)) { + throw 'Conflicting Astra and Sol options did not fail before filesystem mutation.' + } + $migrationHome = Join-Path $env:RUNNER_TEMP 'task-aware-migration' New-Item -ItemType Directory -Force -Path $migrationHome | Out-Null $migrationConfig = @( @@ -164,6 +224,64 @@ jobs: exit 1 fi + standard_home="$RUNNER_TEMP/task-aware-standard-defaults" + mkdir -p -- "$standard_home" + cat > "$standard_home/config.toml" <<'EOF' + model = "parent-model" + model_reasoning_effort = "medium" + approval_policy = "on-request" + sandbox_mode = "workspace-write" + EOF + scripts/install-task-aware-agent.sh --codex-home "$standard_home" + for expected in \ + 'model = "parent-model"' \ + 'model_reasoning_effort = "medium"' \ + 'approval_policy = "on-request"' \ + 'sandbox_mode = "workspace-write"'; do + key=${expected%% =*} + if [[ $(grep -Ec "^[[:space:]]*$key[[:space:]]*=" "$standard_home/config.toml") -ne 1 ]] || + ! grep -Fqx "$expected" "$standard_home/config.toml"; then + printf 'Standard install changed %s.\n' "$expected" >&2 + exit 1 + fi + done + + astra_home="$RUNNER_TEMP/task-aware-astra-default" + scripts/install-task-aware-agent.sh --codex-home "$astra_home" --set-astra-default + scripts/install-task-aware-agent.sh --codex-home "$astra_home" --set-astra-default + scripts/test-task-aware-agent.sh --codex-home "$astra_home" --config-only-runtime + for expected in \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "xhigh"'; do + key=${expected%% =*} + if [[ $(grep -Ec "^[[:space:]]*$key[[:space:]]*=" "$astra_home/config.toml") -ne 1 ]] || + ! grep -Fqx "$expected" "$astra_home/config.toml"; then + printf 'Astra default was not set exactly once: %s.\n' "$expected" >&2 + exit 1 + fi + done + + retired_sol_home="$RUNNER_TEMP/task-aware-retired-sol" + if scripts/install-task-aware-agent.sh --codex-home "$retired_sol_home" --set-sol-default; then + printf '%s\n' 'Retired Sol option was not rejected.' >&2 + exit 1 + fi + if [[ -e "$retired_sol_home" ]]; then + printf '%s\n' 'Retired Sol option mutated the target home.' >&2 + exit 1 + fi + + astra_sol_conflict_home="$RUNNER_TEMP/task-aware-astra-sol-conflict" + if scripts/install-task-aware-agent.sh --codex-home "$astra_sol_conflict_home" \ + --set-astra-default --set-sol-default; then + printf '%s\n' 'Conflicting Astra and Sol options were not rejected.' >&2 + exit 1 + fi + if [[ -e "$astra_sol_conflict_home" ]]; then + printf '%s\n' 'Conflicting options mutated the target home.' >&2 + exit 1 + fi + migration_home="$RUNNER_TEMP/task-aware-migration" mkdir -p -- "$migration_home" cat > "$migration_home/config.toml" <<'EOF' diff --git a/CHANGELOG.md b/CHANGELOG.md index bb861ae..46b74f7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,20 +6,23 @@ ### 変更 -- 現行の価格差を踏まえ、Luna Low の D1 を固定入力、明示的な出力契約、客観的完了条件を持つ限定的な読み取り専用調査・検証まで拡張。 -- Terra Medium の D2 を、状態変更を伴う実装、tool-heavy な複数工程、通常判断が必要な調査・検証として明確化。 -- D1 に Luna Max、D2 に Terra Max、D3 に Sol Max の上位 variant を追加。 -- 中間の xhigh role は設けず、標準とMaxの二段階に統一。 -- 能力クラスを先に決め、同じクラス内で標準またはMax枠を選ぶ二段階ルーティングへ変更。 -- 価格低下を、Maxによる完全性向上または手戻り回避を選びやすくする根拠として反映。 -- 最小十分な役割を選びつつ、D0 の細分化、明白な D2/D3 の意図的な過小ルーティング、不要な microtask fan-out を禁止。 -- 価格低下後も、統合負荷と競合を抑えるため同時に開く子スレッドの上限を3つに維持。 +- 親と全九役の子を GPT-6 Astra に統一。D1 は Low/Medium/High、D2 は Medium/High、D3 は High/xhigh、D4 は xhigh/Max を使用。 +- 既存六役の名前とファイル名を維持。`_max` は上位枠の識別名とし、D1・D2 は High、D3 は xhigh、D4 は Max に対応。 +- D1 の軽い突き合わせ向けに `luna_task_medium` を追加。標準 Low・Medium と上位 High の選択基準を明確化。 +- D4 の所見統合用に `astra_architect` / `astra_architect_max` を追加。2件以上の独立した D3 所見を入力とし、読み取り専用・再委譲禁止・最大3子を維持。 +- Astra xhigh を親へ設定する `--set-astra-default` / `-SetAstraDefault` を追加。通常導入は既存の親モデル、effort、権限を維持。 +- 旧 Sol 既定値オプションを廃止。指定時は Astra オプションへの案内を表示し、ファイル変更前に停止。 +- 能力クラスを先に決め、同じクラス内で標準または上位枠を選ぶ二段階ルーティングへ変更。D1 の限定的な読み取り専用調査と、D2 の通常判断を伴う実装・調査の境界を明確化。 +- 委譲の条件を満たす場合の実行、承認済み作業の継続、変更範囲に応じた検証をポリシーへ追加。 +- D0 の細分化、明白な D2/D3 の意図的な過小ルーティング、不要な小タスクへの分割を禁止。子は最大3つ、再委譲禁止を維持。 ### 検証 -- Windows/Linux validator に、価格対応後の D1/D2 境界と過剰委譲防止規則の検査を追加。 -- Windows/Linux installer と validator を六 role の配置、model、effort、sandbox 検査へ拡張し、旧Luna/Terra High roleをbackup後に除去する移行を追加。 -- release 前の live probe を、六 role の model/effort 確認と D0 から D3 までの標準・Maxルーティング確認へ拡張。 +- CI の固定バージョンを、Astra 設定の厳密な読込を確認した Codex CLI 0.149.1 へ更新。 +- Windows/Linux validator を九役の model、effort、sandbox と Astra 向けポリシーの検査へ更新。 +- Windows/Linux installer に、旧 Luna/Terra High role を backup 後に除去する移行を追加。 +- CI に親の設定維持、Astra opt-in、廃止した Sol オプションの停止と再導入の検査を追加。 +- リリース前の実動作確認を、九役の model/effort と D0 から D4 の役割選択へ拡張。静的検査とモデル起動の検証を区別。 ## [0.1.0] - 2026-07-26 diff --git a/README.md b/README.md index ddcbe89..759990b 100644 --- a/README.md +++ b/README.md @@ -6,44 +6,59 @@ Codex の親エージェントがタスクを難易度別に分類し、必要な場合だけ役割別のカスタムエージェントへ委譲するための設定一式です。 OpenAI の公式製品ではなく、Codex の公開仕様に基づくコミュニティプロジェクトです。 -この構成が想定する親は、`gpt-5.6-sol` をクライアントの Ultra モードで動かす **Sol Ultra** です。 -親は難易度判定、タスクの分割、子の選択、結果の統合を担当します。 -子には Luna Low/Max、Terra Medium/Max、Sol High/Max を使い分けます。 +この構成では、親に **GPT-6 Astra** を使い、難易度の判定、タスクの分割、子の選択、結果の統合を担当させます。 +親を切り替える導入オプションは `gpt-6-astra` と `xhigh` を設定します。 +子もすべて Astra とし、D1 は Low・Medium・High、D2 は Medium/High、D3 は High/xhigh、D4 は xhigh/Max を使います。 -これにより、すべての子が親の高い推論労力を継承して消費量が膨らむことを避けつつ、メインスレッドへ途中経過が流れ込む量を抑えます。 -現行の価格差は、固定契約の読み取り専用作業を Luna へ寄せ、各難度内で必要な場合に高い推論労力を選ぶ根拠にします。 -安価であることだけを理由に子を細分化はしません。 -各 subagent は独自に token と調整時間を使うため、D0 の直接処理と最大3子の上限は維持します。 +子のモデルと推論労力を役割ごとに指定することで、親の設定をすべての子が継承することを避けます。 +ただし、子もそれぞれトークンと調整時間を使うため、分割する利点がある作業だけを委譲します。 +単純な D0 は親が処理し、同時に開く子は最大3つとします。 リポジトリを取得しただけでは Codex の動作は変わりません。 導入スクリプトを実行し、新しいタスクで設定を読み込む必要があります。 ## 想定する実行構成 -Sol はモデル、Ultra は対応するモデルと環境で最大推論と能動的な委譲を利用する実行モードです。 -現行 Codex は、対応モデルの `model_reasoning_effort = "ultra"` も受け付けます。 -このリポジトリでは、親の統合と委譲に Ultra を残し、子の上限を Max にします。 - -対応するアカウントとクライアントで Sol Ultra を選ぶと、親は分割可能な作業をサブエージェントへ能動的に委譲します。 -このリポジトリは、その委譲に D0 から D4 までの能力分類と、同じ難度内で推論労力を選ぶ六つの子エージェントを追加します。 +親の基準は Astra xhigh です。既に選択している親のモデルと推論労力は、標準導入では変更しません。 +Ultra を利用できる環境では、並列に分割できる大きな作業に Astra Ultra を選べます。 +このポリシーは Ultra の有無にかかわらず、委譲の条件を満たす作業を親へ明示します。 | 難易度 | 対象 | 実行役 | モデルと推論労力 | sandbox | | --- | --- | --- | --- | --- | -| D0 | 単純で明確な1工程 | 親が直接処理 | Sol Ultra | 親の設定 | -| D1 標準 | 小さく均質で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Luna Low | read-only | -| D1 Max枠 | D1 のまま、異種入力、密な突合、coverage 重視、多数の edge case がある作業 | `luna_task_max` | Luna Max | read-only | -| D2 標準 | 状態変更を伴う実装、tool-heavy な複数工程、通常判断が必要な調査・検証 | `terra_worker` | Terra Medium | 親から継承 | -| D2 Max枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Terra Max | 親から継承 | -| D3 標準 | 一つの難しい判断、曖昧性、高リスク、複数領域、設計判断 | `sol_specialist` | Sol High | read-only | -| D3 Max枠 | 不確実性と結果の重大性がともに高く、証拠競合、不可逆設計、security、敵対的 edge case を含む作業 | `sol_specialist_max` | Sol Max | read-only | -| D4 | 独立した D3 タスクが複数 | 親が分割して統合 | Sol Ultra | 親と選択した子の設定 | - -能力クラスを D1/D2/D3 から先に決め、その後で標準またはMax枠を選びます。 -Max枠を選んでも権限や能力境界は広がりません。 -Luna の二役と Sol の二役は読み取り専用で、書き込みを伴う通常実装は、親の権限を継承する Terra の二役か親が担当します。 -Ultra は親だけに残し、複数の判断をまたぐ推論と最終統合を親が担当します。 -上位枠は三モデルとも Max に統一します。 -`xhigh` は標準とMaxの間に独立した能力境界を作らないため、別roleにはしません。 +| D0 | 単純で明確な1工程 | 親が直接処理 | Astra xhigh(必要に応じて変更) | 親の設定 | +| D1 標準 Low | 少量・同形式で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Astra Low | read-only | +| D1 標準 Medium | 複数ファイルや異なる形式を扱い、客観的な条件に沿った軽い突き合わせが必要な作業 | `luna_task_medium` | Astra Medium | read-only | +| D1 上位枠 | D1 のまま、異種入力、密な突合、網羅性の確認、多数の境界条件がある作業 | `luna_task_max` | Astra High | read-only | +| D2 標準 | 状態変更を伴う実装、ツールを使う複数工程、通常判断が必要な調査・検証 | `terra_worker` | Astra Medium | 親から継承 | +| D2 上位枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Astra High | 親から継承 | +| D3 標準 | 一つの難しい判断、曖昧性、高リスク、複数領域、設計判断 | `sol_specialist` | Astra High | read-only | +| D3 上位枠 | 不確実性と結果の重大性がともに高く、証拠の競合、不可逆な設計、セキュリティ上の重大性などを含む作業 | `sol_specialist_max` | Astra xhigh | read-only | +| D4 標準 | 2件以上の独立した D3 作業の所見を、制約や依存関係を踏まえて全体の判断へまとめる作業 | `astra_architect` | Astra xhigh | read-only | +| D4 上位枠 | D3 作業間で証拠や推奨が競合し、全体の判断が不可逆な選択や重大な影響を伴う作業 | `astra_architect_max` | Astra Max | read-only | + +能力クラスを D1/D2/D3/D4 から先に決め、その後で必要な推論労力を選びます。 +上位枠を選んでも権限や能力境界は広がりません。 +D1・D3・D4 の子は読み取り専用です。書き込みを伴う通常実装は、親の権限を継承する D2 の子か親が担当します。 +子には再委譲させず、最終統合は親が担当します。 + +既存六つの役割名とファイル名は互換性のため維持し、D1 Medium を1役、D4 を2役追加して計九役にしています。 +モデルは九役とも `gpt-6-astra` です。 +`_max` は上位枠の互換名です。D1・D2 の `_max` は High、D3 の `_max` は xhigh、D4 の `_max` は Max に対応します。 +役割名からモデルや effort を推測せず、上の表と TOML の設定値を確認してください。 + +この割り当ては、役割と権限の境界を保ちながら、すべての子を Astra に統一するためのプロジェクトの選択です。 +Low・Medium・High・xhigh・Max の品質や消費量を比較したベンチマーク結果ではありません。 +モデルの用途と利用可能な設定は [Codex のモデル案内](https://learn.chatgpt.com/docs/models) を参照してください。 + +D1 の標準枠では、少量・同形式の入力なら Low、複数ファイルや異なる形式を軽く突き合わせるなら Medium を選びます。 +網羅性の確認や密な突き合わせ、多数の例外処理が必要なら High を使います。 +上位の条件が明らかな場合は、Low や Medium で失敗するのを待たず、適切な役割へ直接渡します。 +いずれも客観的な完了条件を持つ読み取り専用作業に限り、実装や設計上の判断は D2 以上へ分けます。 + +D4 では、親が独立した D3 作業を分割し、その所見が揃ってから、必要に応じて D4 の子へ全体の判断を依頼します。 +子へ渡す資料には、少なくとも2件の独立した D3 所見と、その根拠を含めます。 +D4 の子は所見の統合を担当し、再委譲や状態変更は行いません。最終判断と実行は親が担当します。 +D4 も同時に開く最大3子に数えるため、入力が揃い、空き枠ができてから起動します。 ## ルーティングの仕組み @@ -58,9 +73,10 @@ Ultra は親だけに残し、複数の判断をまたぐ推論と最終統合 4. 委譲の調整コストが、親による直接処理より小さい。 条件を満たした後、親は能力クラスを選び、そのクラス内で必要十分な推論労力を選びます。 -標準 effort が基本ですが、完全性の向上または手戻りの回避が見込める場合は、価格低下を踏まえてMax枠を積極的に選べます。 -Max枠の task packet には、標準 effort では誤りや手戻りが増える具体的な理由を含めます。 -原子的な D0 を Luna が安価であるという理由だけで分割しません。 +標準 effort を基本とし、追加の時間とトークンを使っても、網羅性の向上や手戻りの回避が見込める場合に上位枠を選びます。 +D1 Medium の task packet には、Low では不足する突き合わせの内容を記載します。 +上位枠を選ぶ場合は、各クラスの標準 effort では不足する具体的な理由も含めます。 +原子的な D0 を、推論労力を下げるためだけに分割しません。 同じ入力と完了条件を共有する小さな作業は、分離によって待ち時間、コンテキスト分離、証拠の独立性が改善しない限り、一つの task packet にまとめます。 子は別の子を起動しません。 @@ -72,24 +88,29 @@ Max枠の task packet には、標準 effort では誤りや手戻りが増え `NEEDS_ESCALATION` は Codex ランタイムの自動判定ではなく、子が能力不足の根拠を親へ返すための応答規約です。 合理的に選んだ下位役割が返した場合だけ、親はその根拠を確認し、必要な上位役割へ再委譲します。 -最初から D2 または D3 と明らかな作業を、昇格結果を得るためだけに Luna へ渡しません。 +最初から D2 または D3 と明らかな作業を、昇格結果を得るためだけに D1 の役割へ渡しません。 -子を起動するときは、`spawn_agent` の `agent_type` に `luna_task`、`luna_task_max`、`terra_worker`、`terra_worker_max`、`sol_specialist`、`sol_specialist_max` のいずれかを明示します。 +子を起動するときは、`spawn_agent` の `agent_type` に `luna_task`、`luna_task_medium`、`luna_task_max`、`terra_worker`、`terra_worker_max`、`sol_specialist`、`sol_specialist_max`、`astra_architect`、`astra_architect_max` のいずれかを明示します。 `task_name` は子タスクの表示名とパスを付ける項目であり、custom agent の選択には使いません。 `task_name = "luna_task"` だけを指定すると、子が親のモデルと推論労力を継承するため、想定したコスト制御になりません。 -D1 から D3 までの委譲では、標準とMax枠のどちらでも `agent_type` を必須とし、まず必ず引数付きで起動します。 +D1 から D4 までの委譲では、標準と上位枠のどちらでも `agent_type` を必須とし、まず必ず引数付きで起動します。 tool が `agent_type` または custom agent を明示的に拒否した場合だけ、既定の子を起動せず、親で処理して不一致を報告します。 各 spawn は `fork_turns = "none"` を指定し、親の全会話履歴ではなく task packet だけを子へ渡します。 task packet 自体にも再委譲禁止を明記します。 +Astra 向けの実行規則として、承認済みの作業を実装と必要な検証まで進めることも明記します。 +通常の可逆的な選択は会話の文脈から判断し、結果や権限に影響する不足情報がある場合に質問します。 +回答待ちの間も独立した作業は進めます。検証は変更範囲に合わせ、必要な確認が通った後は、追加変更や未解決の問題がある場合に再実行します。 +この調整は [Astra の公式ガイド](https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-astra) を参考にし、このリポジトリの最大3子・一段階委譲へ限定しています。 + ## 前提条件 - Windows では PowerShell 7 以降を使用できること。 - Linux では Bash、`awk`、`grep` を使用できること。 - カスタムエージェントと subagent workflow に対応した現行 Codex を使用していること。 -- この公開候補の検証基準である Codex CLI 0.145.0 以降を使用すること。 -- 使用するアカウントで `gpt-5.6-luna`、`gpt-5.6-terra`、`gpt-5.6-sol` を利用できること。 -- Sol Ultra を使う場合は、対応するアカウントとクライアントで Ultra が有効であること。 +- CI の設定互換性検証基準は Codex CLI 0.149.1。Astra の利用には、アカウントとクライアントのモデル選択欄で対応を確認すること。 +- 使用するアカウントで `gpt-6-astra` を利用できること。 +- Astra Ultra を使う場合は、対応するアカウントとクライアントで Ultra が有効であること。 - コマンドをこのリポジトリのルートで実行すること。 導入前に、Codex の起動と設定読込が正常であることを確認してください。 @@ -99,10 +120,10 @@ codex --version codex doctor --summary --no-color --ascii ``` -モデルと推論の選択欄では、三つの子モデル、High/Max を含む必要な effort、親に使う Sol Ultra が表示されることも確認します。 - -Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 から D4 までのルーティング規則を使えます。 -ただし、それはこのリポジトリが想定する Sol Ultra と同じ実行構成ではありません。 +モデルと推論の選択欄では、Astra と各役割に必要な Low・Medium・High・xhigh・Max が表示されることも確認します。 +2026-09-05 のローカルモデルカタログ(client version 0.153.0)では、Astra の `low`、`medium`、`high`、`xhigh`、`max`、`ultra` を確認しました。 +設定の厳密な読込は、別途 Codex CLI 0.149.1 で検証しました。 +カタログへの掲載や設定の読込成功だけでは、実際のモデル呼び出し成功は保証されません。 ## 導入で変更するもの @@ -113,12 +134,15 @@ Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 か | --- | --- | | `config.toml` | `[agents] enabled = true`、`max_concurrent_threads_per_session = 3` を設定し、旧 key を除去 | | `AGENTS.md` | マーカーで囲んだ task-aware delegation policy を追加または更新 | -| `agents/luna-task.toml` | Luna Low の読み取り専用エージェントを配置 | -| `agents/luna-task-max.toml` | Luna Max の読み取り専用エージェントを配置 | -| `agents/terra-worker.toml` | Terra Medium の作業エージェントを配置 | -| `agents/terra-worker-max.toml` | Terra Max の作業エージェントを配置 | -| `agents/sol-specialist.toml` | Sol High の読み取り専用エージェントを配置 | -| `agents/sol-specialist-max.toml` | Sol Max の読み取り専用エージェントを配置 | +| `agents/luna-task.toml` | Astra Low の読み取り専用エージェントを配置 | +| `agents/luna-task-medium.toml` | Astra Medium の読み取り専用エージェントを配置 | +| `agents/luna-task-max.toml` | Astra High の読み取り専用エージェントを配置 | +| `agents/terra-worker.toml` | Astra Medium の作業エージェントを配置 | +| `agents/terra-worker-max.toml` | Astra High の作業エージェントを配置 | +| `agents/sol-specialist.toml` | Astra High の読み取り専用エージェントを配置 | +| `agents/sol-specialist-max.toml` | Astra xhigh の読み取り専用エージェントを配置(役割名は互換性維持) | +| `agents/astra-architect.toml` | Astra xhigh の読み取り専用D4エージェントを配置 | +| `agents/astra-architect-max.toml` | Astra Max の読み取り専用D4エージェントを配置 | 既存ファイルは、変更前に `$CODEX_HOME/task-aware-backups//` へ退避します。 同名のカスタムエージェントファイルは上書きされます。 @@ -136,16 +160,16 @@ Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 か pwsh -File .\scripts\Install-TaskAwareAgent.ps1 ``` -親の既定値を Sol xhigh にする場合は、`-SetSolDefault` を指定します。 +親の既定値を Astra xhigh にする場合は、`-SetAstraDefault` を指定します。 ```powershell -pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault +pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetAstraDefault ``` フルアクセスも設定する場合は、`-EnableFullAccess` を追加します。 ```powershell -pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault -EnableFullAccess +pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetAstraDefault -EnableFullAccess ``` 別の Codex 環境へ試験導入する場合は `-CodexHome `、変更予定だけを確認する場合は `-WhatIf` を指定できます。 @@ -160,16 +184,16 @@ PowerShell 版も、`config.toml` がない空の `CODEX_HOME` を初期化で ./scripts/install-task-aware-agent.sh ``` -親の既定値を Sol xhigh にする場合は、`--set-sol-default` を指定します。 +親の既定値を Astra xhigh にする場合は、`--set-astra-default` を指定します。 ```bash -./scripts/install-task-aware-agent.sh --set-sol-default +./scripts/install-task-aware-agent.sh --set-astra-default ``` フルアクセスも設定する場合は、`--enable-full-access` を追加します。 ```bash -./scripts/install-task-aware-agent.sh --set-sol-default --enable-full-access +./scripts/install-task-aware-agent.sh --set-astra-default --enable-full-access ``` 別の Codex 環境へ導入する場合は、`CODEX_HOME` 環境変数または `--codex-home ` を指定できます。 @@ -177,14 +201,15 @@ Linux 版は、空の `CODEX_HOME` に必要なファイルを新規作成でき Windows 側の checkout を WSL から使う場合は、`.sh` の改行が LF であることを確認してください。 CRLF のまま実行すると、shebang の `bash` を解決できず起動に失敗します。 -### Sol Ultra の選択 +### 親の effort と旧オプション -`-SetSolDefault` と `--set-sol-default` が設定する親の既定値は、`gpt-5.6-sol` と `xhigh` です。 -どちらのオプションだけでも Sol Ultra にはなりません。 +`-SetAstraDefault` と `--set-astra-default` は `gpt-6-astra` と `xhigh` を設定します。 +既に親へ Ultra などを選択していて、その effort を保つ場合は標準導入を使ってください。 +Ultra を新たに使う場合は、対応クライアントのモデル選択で Astra と Ultra を選びます。 -想定構成どおりに使う場合は、導入後に対応クライアントのモデル選択で Sol と Ultra を選んでください。 -インストーラーは、アカウントやクライアントごとの Ultra 対応を暗黙に仮定しないため、`model_reasoning_effort = "ultra"` を自動では書き込みません。 -対応を確認できた環境では、クライアントの選択または明示的な設定で親を Ultra にしてください。 +旧 `-SetSolDefault` と `--set-sol-default` は廃止しました。 +指定すると、ファイルを変更する前に Astra オプションへの移行案内を表示して停止します。 +親も Astra に切り替える場合は `-SetAstraDefault` または `--set-astra-default` を使ってください。 ### フルアクセスの影響 @@ -217,22 +242,25 @@ PowerShell 版も対象の `CODEX_HOME` を明示し、`codex --strict-config do 認証情報のない clean home や CI では `--config-only-runtime` を加えます。 Linux 版は対象の `CODEX_HOME` を明示し、`codex --strict-config doctor --summary` を実行します。 -どちらの検証スクリプトも、配置したファイル、主要な設定値、六つの子のモデル ID、推論労力、宣言した sandbox を静的に確認します。 +どちらの検証スクリプトも、配置したファイル、主要な設定値、九つの子のモデル ID、推論労力、宣言した sandbox を静的に確認します。 `codex` コマンドが見つかる場合は、続けて `codex doctor` を実行します。 この検証は、runtime が子へ適用した実効 sandbox、モデルの利用権限、実際の子の起動、D0 から D4 までの分類結果までは確認しません。 導入後は Codex を再起動するか、新しいタスクを開始し、次の手順で実動作も確認してください。 -1. 親のモデルと推論の選択欄が Sol Ultra になっていることを確認する。 +1. 親のモデルが Astra、effort が選択した値(導入オプションなら xhigh)になっていることを確認する。 2. 一つのファイルから既知の文字列を読むだけの D0 を依頼し、子を起動しないことを確認する。 -3. 小さく均質で固定契約を持つ読み取り専用 D1 を依頼し、`luna_task` を確認する。 -4. 異種入力の密な突合と coverage 判定を伴うが、客観的に完了判定できる D1 を依頼し、`luna_task_max` を確認する。 -5. 専用の空ディレクトリに一つのファイルを作成して検証する D2 を依頼し、`terra_worker` を確認する。 -6. 複数ファイルの結合制約と長い検証経路を持つ D2 を依頼し、`terra_worker_max` を確認する。 -7. 一つの明確な設計トレードオフを判断する D3 を依頼し、`sol_specialist` を確認する。 -8. 証拠が競合し、不可逆性または security 上の重大性も高い D3 を依頼し、`sol_specialist_max` を確認する。 -9. 親が D0 では spawn せず、D1 から D3 では対応する標準またはMax枠の `agent_type` を渡すことを確認する。 -10. 各子の詳細を開き、Luna Low/Max、Terra Medium/Max、Sol High/Max が実際に適用されていることを確認する。 +3. 少量・同形式の固定入力を持つ読み取り専用 D1 を依頼し、`luna_task` が Low で動くことを確認する。 +4. 複数ファイルや異なる形式の軽い突き合わせを行う D1 を依頼し、`luna_task_medium` が Medium で動くことを確認する。 +5. 密な突き合わせや網羅性の確認が必要な D1 を依頼し、`luna_task_max` が High で動くことを確認する。 +6. 専用の空ディレクトリに一つのファイルを作成して検証する D2 を依頼し、`terra_worker` を確認する。 +7. 複数ファイルの結合制約と長い検証経路を持つ D2 を依頼し、`terra_worker_max` を確認する。 +8. 一つの明確な設計トレードオフを判断する D3 を依頼し、`sol_specialist` が High で動くことを確認する。 +9. 証拠の競合と判断の重大性がともに高い D3 を依頼し、`sol_specialist_max` が xhigh で動くことを確認する。 +10. 2件以上の独立した D3 所見をまとめる D4 を依頼し、`astra_architect` が xhigh で動くことを確認する。 +11. D3 所見間で重大な推奨や証拠が競合する D4 を依頼し、`astra_architect_max` が Max で動くことを確認する。 +12. 親が D0 では spawn せず、D1 から D4 では対応する `agent_type` を渡すこと、子が再委譲せず最大3子を守ることを確認する。 +13. 各子の詳細で、model がすべて `gpt-6-astra`、effort が上の表と一致することを確認する。 `AGENTS.md` の指示チェーンは新しい実行の開始時に構築されるため、導入前から開いているタスクでは確認できません。 @@ -246,13 +274,13 @@ Linux 版は対象の `CODEX_HOME` を明示し、`codex --strict-config doctor 導入後に `config.toml` や `AGENTS.md` を変更している場合は、バックアップをそのまま上書きせず、差分を確認して task-aware 関連の設定だけを手動で統合してください。 導入前に存在しなかったファイルはバックアップへ含まれません。 -その場合は、追加された六つのエージェントファイルと、`AGENTS.md` の `BEGIN CODEX TASK-AWARE AGENT` から `END CODEX TASK-AWARE AGENT` までのブロックを手動で取り除く必要があります。 +その場合は、追加された九つのエージェントファイルと、`AGENTS.md` の `BEGIN CODEX TASK-AWARE AGENT` から `END CODEX TASK-AWARE AGENT` までのブロックを手動で取り除く必要があります。 ## ファイル構成 - `config/AGENTS.task-aware.md`:グローバル指示へ追加するルーティング規則 - `config/config.task-aware.toml`:`config.toml` へ統合する設定例 -- `agents/*.toml`:Luna、Terra、Sol の役割別エージェント +- `agents/*.toml`:Astra の役割別エージェント(旧役割名を維持) - `scripts/Install-TaskAwareAgent.ps1`:バックアップ付き導入スクリプト - `scripts/Test-TaskAwareAgent.ps1`:配置と Codex 設定の検証スクリプト - `scripts/install-task-aware-agent.sh`:Linux 向けのバックアップ付き導入スクリプト @@ -275,17 +303,17 @@ Apache License 2.0 です。SPDX identifier は `Apache-2.0` です。全文は ## 設計上の注意 - サブエージェントはメインスレッドのノイズを減らしますが、総トークン量や待ち時間が必ず減るわけではありません。 -- 価格低下は同じ難度内でMax枠を選ぶ閾値を下げますが、必要な能力、書き込み、曖昧さ、リスクより優先しません。 -- Max は応答時間と token 使用量を増やすため、標準 effort で十分な作業には使いません。 -- Ultra 自体も分割可能な作業へサブエージェントを使うため、独立した成果がある作業だけに委譲を絞ります。 +- 上位枠は必要な能力、書き込み、曖昧さ、リスクを先に判定して選びます。価格だけを理由に選びません。 +- High や Max は追加の推論時間とトークンを使うため、各クラスの標準 effort で十分な作業には使いません。 +- Ultra を選んだ場合も、独立した成果があり、調整コストを上回る利点がある作業を委譲します。 - モデルが利用できることと、ルーティングが適切であることは静的テストだけでは保証できません。 - ルーティングは親の判断に依存するため、D0 から D4 までの境界は決定的ではありません。 -- Codex Desktop や CLI の実効 permission profile が親の権限を子へ強制する runtime では、role TOML の `sandbox_mode` より親の実効権限が優先される場合があります。Luna と Sol の developer instructions も書き込みを禁止しますが、これは OS sandbox と同じ強制境界ではありません。 +- Codex Desktop や CLI の実効 permission profile が親の権限を子へ強制する runtime では、role TOML の `sandbox_mode` より親の実効権限が優先される場合があります。D1・D3・D4 の developer instructions も書き込みを禁止しますが、これは OS sandbox と同じ強制境界ではありません。 ## 参考資料 - [Models](https://learn.chatgpt.com/docs/models) -- [Codex rate card](https://help.openai.com/en/articles/20001106-codex-rate-card) +- [GPT-6 Astra guide](https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-astra) - [Subagents](https://learn.chatgpt.com/docs/agent-configuration/subagents) - [Configuration Reference](https://learn.chatgpt.com/docs/config-file/config-reference) - [Configuration Schema](https://developers.openai.com/codex/config-schema.json) diff --git a/RELEASE_CHECKLIST.md b/RELEASE_CHECKLIST.md index f9641fd..4460eff 100644 --- a/RELEASE_CHECKLIST.md +++ b/RELEASE_CHECKLIST.md @@ -4,12 +4,14 @@ - [ ] `main` が最新の `origin/main` と一致し、working tree に意図しない変更がない。 - [ ] [Codex changelog](https://learn.chatgpt.com/docs/changelog) で最新版を確認し、`.github/workflows/ci.yml` と README の検証基準を更新する。 -- [ ] [Codex rate card](https://help.openai.com/en/articles/20001106-codex-rate-card) でモデル間の価格差を確認し、ルーティング根拠が現行レートと矛盾しない。 +- [ ] [Codex Models](https://learn.chatgpt.com/docs/models) と [Astra guide](https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-astra) で、モデル、effort、委譲方針の互換性を確認する。 - [ ] `CHANGELOG.md` の version、日付、release link を確定する。 - [ ] `LICENSE`、`SECURITY.md`、`CONTRIBUTING.md` が release archive に含まれる。 - [ ] GitHub Actions の Windows/Linux job が成功する。 -- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1/D2/D3 の標準・Max条件を実行し、Luna Low/Max、Terra Medium/Max、Sol High/Max の model と reasoning effort を live probe する。 -- [ ] Max枠が能力境界や sandbox を広げず、task packet に昇格理由が含まれることを確認する。 +- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1 の Low/Medium/High 条件と D2/D3/D4 の標準・上位条件を実行する。子の実起動を通じて、model がすべて `gpt-6-astra`、effort が D1 の Low/Medium/High、D2 の Medium/High、D3 の High/xhigh、D4 の xhigh/Max になっていることを確認する。 +- [ ] D4 の入力に2件以上の独立した D3 所見が含まれ、D4 も読み取り専用・再委譲禁止・同時に最大3子を守ることを確認する。 +- [ ] 親の Astra xhigh 設定、標準導入での既存親設定の維持、廃止した Sol オプションの案内と変更前の停止を確認する。 +- [ ] 上位枠が能力境界や sandbox を広げず、task packet に昇格理由が含まれることを確認する。 - [ ] `git diff --check` と秘密情報 scan を通す。 ## 公開 diff --git a/agents/astra-architect-max.toml b/agents/astra-architect-max.toml new file mode 100644 index 0000000..3369d64 --- /dev/null +++ b/agents/astra-architect-max.toml @@ -0,0 +1,27 @@ +name = "astra_architect_max" +description = """ +Use for a bounded D4 synthesis when findings from at least two independent D3 +work items conflict and the combined decision has major consequences, including +irreversible choices or security-sensitive trade-offs across work items. +Use Astra Max when xhigh is insufficient for reconciling the combined evidence. +This is a read-only synthesis role; the parent owns orchestration. +""" +model = "gpt-6-astra" +model_reasoning_effort = "max" +sandbox_mode = "read-only" + +developer_instructions = """ +Resolve exactly one exceptionally demanding D4 synthesis from supplied findings. +Require evidence from at least two independent D3 work items. If findings are +missing, return NEEDS_INPUT with the missing evidence; do not invent it. +Use Max reasoning to compare competing interpretations and counterevidence, +trace dependencies across decisions, and identify assumptions that change the +overall recommendation. Preserve uncertainty when evidence remains conflicting. +Use the supplied evidence and inspect references only to resolve specific gaps. +Do not repeat completed investigations or act as another orchestrator. +Do not delegate. +Do not modify files or external state, even if the runtime grants broader access. +Return the overall recommendation, evidence provenance, alternatives and their +consequences, unresolved conflicts, and validation needed before implementation. +The parent makes the final decision and owns any state-changing follow-through. +""" diff --git a/agents/astra-architect.toml b/agents/astra-architect.toml new file mode 100644 index 0000000..75453aa --- /dev/null +++ b/agents/astra-architect.toml @@ -0,0 +1,25 @@ +name = "astra_architect" +description = """ +Use for one bounded D4 synthesis of findings from at least two independent D3 +work items. Reconcile their constraints, dependencies, and recommendations into +an overall decision. Use Astra xhigh by default; prefer astra_architect_max +when conflicts across the findings and their consequences require deeper review. +This is a read-only synthesis role; the parent owns orchestration. +""" +model = "gpt-6-astra" +model_reasoning_effort = "xhigh" +sandbox_mode = "read-only" + +developer_instructions = """ +Resolve exactly one bounded D4 synthesis from the supplied D3 findings. +Require evidence from at least two independent D3 work items. If findings are +missing, return NEEDS_INPUT with the missing evidence; do not invent it. +Compare assumptions, constraints, dependencies, and conflicting recommendations. +Use the supplied evidence and inspect references only to resolve specific gaps. +Do not repeat completed investigations or act as another orchestrator. +Do not delegate. +Do not modify files or external state, even if the runtime grants broader access. +Return the overall recommendation, evidence provenance, resolved conflicts, +remaining uncertainty, and validation needed before implementation. +The parent makes the final decision and owns any state-changing follow-through. +""" diff --git a/agents/luna-task-max.toml b/agents/luna-task-max.toml index 3d1c396..42bf246 100644 --- a/agents/luna-task-max.toml +++ b/agents/luna-task-max.toml @@ -2,17 +2,17 @@ name = "luna_task_max" description = """ Use for D1 work that remains deterministic, read-only, and objectively verifiable, but needs dense cross-checking across heterogeneous inputs, -coverage-sensitive validation, or many edge cases where Luna Low would be +coverage-sensitive validation, or many edge cases where Astra Medium would be materially more error-prone. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-5.6-luna" -model_reasoning_effort = "max" +model = "gpt-6-astra" +model_reasoning_effort = "high" sandbox_mode = "read-only" developer_instructions = """ Handle exactly one bounded D1 task. -Use Max reasoning for completeness and cross-checking, not to broaden the task's +Use High reasoning for completeness and cross-checking, not to broaden the task's capability boundary. Do not broaden scope or delegate. Do not modify files or external state, even if the runtime grants broader access. diff --git a/agents/luna-task-medium.toml b/agents/luna-task-medium.toml new file mode 100644 index 0000000..e5112a4 --- /dev/null +++ b/agents/luna-task-medium.toml @@ -0,0 +1,23 @@ +name = "luna_task_medium" +description = """ +Use for bounded D1 work with fixed inputs, an explicit output contract, and an +objective success condition that needs modest reconciliation across files or +formats. Use Astra Medium when compact, homogeneous Low work is insufficient, +but dense cross-checking, coverage-sensitive validation, and many edge cases +do not justify luna_task_max. This role remains deterministic and read-only. +Do not use for material judgment, broad investigation, or state changes. +""" +model = "gpt-6-astra" +model_reasoning_effort = "medium" +sandbox_mode = "read-only" + +developer_instructions = """ +Handle exactly one bounded D1 task. +Use Medium reasoning to reconcile the supplied files or formats against the +explicit output contract and objective success condition. +Do not broaden scope or delegate. +Do not modify files or external state, even if the runtime grants broader access. +Return only the requested result and essential evidence. +If state changes, material judgment, or architectural reasoning are required, +return NEEDS_ESCALATION with a one-sentence reason. +""" diff --git a/agents/luna-task.toml b/agents/luna-task.toml index 0df4637..0c74aaf 100644 --- a/agents/luna-task.toml +++ b/agents/luna-task.toml @@ -3,11 +3,12 @@ description = """ Use as the default for compact, homogeneous D1 extraction, classification, format conversion, repetitive checks, and bounded read-only investigation or validation with fixed inputs, an explicit output contract, and an explicit -success condition. Prefer luna_task_max when dense cross-checking, coverage, -or numerous edge cases make Low materially more error-prone. +success condition. Use luna_task_medium for modest reconciliation across files +or formats; prefer luna_task_max for dense cross-checking, coverage-sensitive +validation, or numerous edge cases. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-5.6-luna" +model = "gpt-6-astra" model_reasoning_effort = "low" sandbox_mode = "read-only" diff --git a/agents/sol-specialist-max.toml b/agents/sol-specialist-max.toml index 4530471..61d264e 100644 --- a/agents/sol-specialist-max.toml +++ b/agents/sol-specialist-max.toml @@ -3,15 +3,18 @@ description = """ Use for D3 work when both uncertainty and consequence are high, such as conflicting evidence, security-sensitive trade-offs, irreversible architecture, adversarial edge cases, or a strong need to reduce reasoning variance. +Use Astra xhigh for this higher-risk D3 contract when Astra High is insufficient. """ -model = "gpt-5.6-sol" -model_reasoning_effort = "max" +model = "gpt-6-astra" +model_reasoning_effort = "xhigh" sandbox_mode = "read-only" developer_instructions = """ Resolve one exceptionally demanding D3 decision or evidence lane. -Use Max reasoning to reconcile uncertainty, consequences, and edge cases. -Do not repeat broad exploration already completed by cheaper agents. +Use Extra High reasoning to reconcile uncertainty, consequences, and edge cases. +Compare alternatives and counterevidence; identify assumptions that change the +recommendation. +Do not repeat broad exploration already completed by other agents. Do not delegate. Do not modify files or external state, even if the runtime grants broader access. Return a concise recommendation, supporting evidence, risks, and validation plan. diff --git a/agents/sol-specialist.toml b/agents/sol-specialist.toml index f2d23fd..efe271c 100644 --- a/agents/sol-specialist.toml +++ b/agents/sol-specialist.toml @@ -5,13 +5,13 @@ architecture-heavy decision requiring substantial judgment or trade-off analysis. Prefer sol_specialist_max when uncertainty and consequence are both high. """ -model = "gpt-5.6-sol" +model = "gpt-6-astra" model_reasoning_effort = "high" sandbox_mode = "read-only" developer_instructions = """ Resolve one difficult decision or evidence lane. -Do not repeat broad exploration already completed by cheaper agents. +Do not repeat broad exploration already completed by other agents. Do not delegate. Do not modify files or external state, even if the runtime grants broader access. Return a concise recommendation, supporting evidence, risks, and validation plan. diff --git a/agents/terra-worker-max.toml b/agents/terra-worker-max.toml index 2bfdb1b..b0df81d 100644 --- a/agents/terra-worker-max.toml +++ b/agents/terra-worker-max.toml @@ -2,16 +2,16 @@ name = "terra_worker_max" description = """ Use for D2 work that stays within ordinary engineering judgment but has many coupled constraints, a long tool or verification chain, difficult debugging, -or expensive rework that justifies deeper reasoning than Terra Medium. +or expensive rework that justifies deeper reasoning than Astra Medium. Do not use for unresolved architectural trade-offs, exceptional risk, or D3 ambiguity. """ -model = "gpt-5.6-terra" -model_reasoning_effort = "max" +model = "gpt-6-astra" +model_reasoning_effort = "high" developer_instructions = """ Own one independently verifiable D2 work item. -Use Max reasoning for coupled constraints, edge cases, and verification, not to +Use High reasoning for coupled constraints, edge cases, and verification, not to broaden the task's capability boundary. Stay within the supplied file and subsystem scope. Do not spawn subagents. diff --git a/agents/terra-worker.toml b/agents/terra-worker.toml index 830c31e..a2c062f 100644 --- a/agents/terra-worker.toml +++ b/agents/terra-worker.toml @@ -3,9 +3,9 @@ description = """ Use as the default for bounded D2 state-changing implementation, tool-heavy multi-step work, or investigation and verification that requires ordinary judgment while keeping clear success criteria. Prefer terra_worker_max when -coupled constraints, difficult debugging, or expensive rework justify Max. +coupled constraints, difficult debugging, or expensive rework justify High. """ -model = "gpt-5.6-terra" +model = "gpt-6-astra" model_reasoning_effort = "medium" developer_instructions = """ diff --git a/config/AGENTS.task-aware.md b/config/AGENTS.task-aware.md index ceee959..84d83a0 100644 --- a/config/AGENTS.task-aware.md +++ b/config/AGENTS.task-aware.md @@ -4,6 +4,23 @@ Before spawning any subagent, decompose the request into independently verifiable work items and classify each item separately. +### Parent execution + +The target parent is GPT-6 Astra. Apply the delegation gates below actively +when independent work can run alongside useful parent work. Ultra is optional; +follow the same gates at any supported parent effort. Children never delegate. + +Finish authorized work through implementation and relevant verification. +Resolve routine, reversible choices from context. Ask only when missing input +would materially affect correctness, scope, authorization, or irreversible +outcomes; continue independent work while an answer is pending. +If local guidance causes a pause, identify the file and exact instruction and +explain why existing user authorization does not resolve it. + +Keep verification proportional to the change. Complete required checks; repeat +or broaden them only for changed code, failures, or unresolved risks. Report +the result, essential evidence, and remaining limits concisely. + ### Difficulty routing Classify capability first, then choose reasoning effort inside that class. @@ -15,8 +32,8 @@ difficulty class. - D1: Deterministic extraction, classification, transformation, repetitive checking, or bounded read-only investigation or verification. Inputs, the output contract, and the success condition must be explicit, and the work - must not require material judgment. Use `luna_task` or `luna_task_max` as - defined under effort routing. + must not require material judgment. Use `luna_task`, `luna_task_medium`, or + `luna_task_max` as defined under effort routing. - D2: State-changing implementation, tool-heavy multi-step work, or bounded investigation and verification that requires ordinary judgment. Completion criteria must still be clear. Use `terra_worker` or `terra_worker_max` as @@ -24,17 +41,22 @@ difficulty class. - D3: Ambiguous, cross-system, high-risk, security-sensitive, or architectural work requiring trade-off judgment. Use `sol_specialist` or `sol_specialist_max` as defined under effort routing. -- D4: Two or more independent D3 work items. Orchestrate them from the root, - but keep each child bounded and non-recursive. +- D4: Two or more independent D3 work items whose findings need an overall + decision. The root orchestrates the work items. After their findings are + available, use `astra_architect` or `astra_architect_max` for one bounded + synthesis when delegation gates are met. Keep every child non-recursive. ### Effort routing - D1 default: call `spawn_agent` with `agent_type = "luna_task"` for compact, - homogeneous, deterministic work. + homogeneous, deterministic work with Astra Low. +- D1 standard Medium: call `spawn_agent` with `agent_type = "luna_task_medium"` + for modest reconciliation across fixed files or formats with an objective + success condition, when Low is insufficient and High is not justified. - D1 elevated: call `spawn_agent` with `agent_type = "luna_task_max"` when the work remains deterministic and read-only but heterogeneous inputs, dense cross-checking, coverage-sensitive validation, or numerous edge cases make - Low materially more error-prone. + Medium materially more error-prone. - D2 default: call `spawn_agent` with `agent_type = "terra_worker"` for bounded implementation or investigation requiring ordinary judgment. - D2 elevated: call `spawn_agent` with `agent_type = "terra_worker_max"` when @@ -45,21 +67,34 @@ difficulty class. - D3 elevated: call `spawn_agent` with `agent_type = "sol_specialist_max"` when both uncertainty and consequence are high, including conflicting evidence, security-sensitive trade-offs, irreversible architecture, adversarial edge - cases, or a strong need to reduce reasoning variance. - -Use the base effort when it is sufficient. Lower model prices reduce the -threshold for elevated effort when it is likely to improve completeness or -avoid rework, but price and task size alone are not sufficient reasons. + cases, or a strong need to reduce reasoning variance. This compatibility + role uses Astra xhigh; the standard D3 route uses Astra High. +- D4 default: call `spawn_agent` with `agent_type = "astra_architect"` for + bounded synthesis of findings from at least two independent D3 work items. +- D4 elevated: call `spawn_agent` with `agent_type = "astra_architect_max"` + when findings conflict across work items and the combined decision has major + consequences, so xhigh is insufficient for reconciling the evidence. + +Use the least effort that meets the role's completion contract. D1 standard +work uses Astra Low for compact, homogeneous inputs or Astra Medium for modest +reconciliation across files or formats. Its upper role uses Astra High for +dense cross-checking, coverage-sensitive validation, or many edge cases. +D2 uses Astra Medium by default and Astra High when additional reasoning +justifies the time and token cost. +D3 uses Astra High by default and Astra xhigh when uncertainty and +consequence justify deeper comparison of alternatives and counterevidence. +D4 uses Astra xhigh by default and Astra Max for exceptional synthesis needs. Capability, mutation, ambiguity, and risk determine the difficulty class before -cost is considered. Never substitute an elevated lower-class role for a higher -class. +effort is considered. Never substitute a lower-class role for a higher class. -Use Max as the single elevated effort for D1-D3. Do not add an xhigh middle lane -unless it gains a distinct routing criterion; otherwise it increases routing -ambiguity without changing the capability boundary. +All nine child roles use gpt-6-astra with low, medium, high, xhigh, or max reasoning. +The mapping is D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max. +The six existing role identifiers and filenames remain stable for compatibility. +The luna, terra, sol, and _max names no longer specify the model or effort. +Select by the role contract and mapping, not the model name in the identifier. `task_name` labels the child task; it does not select a custom agent. -For every D1-D3 spawn, `agent_type` is mandatory. Never omit it, and never +For every D1-D4 spawn, `agent_type` is mandatory. Never omit it, and never encode the role only in `task_name`. Before calling `spawn_agent`, verify that the request includes the exact base or elevated `agent_type` selected above. Always attempt the @@ -80,7 +115,7 @@ Spawn a subagent only when all of the following are true: Never spawn an agent merely to restate the request, create a generic plan, or duplicate another agent's investigation. -Do not split an atomic D0 item solely because Luna is inexpensive. Combine +Do not split an atomic D0 item solely to use a lower effort. Combine adjacent microtasks that share inputs and a success contract when separate children would not improve elapsed time, context isolation, or evidence independence. @@ -93,6 +128,13 @@ descendants. Do not rely on `agents.max_depth` for this boundary because Codex V2 ignores that legacy setting. Use only one writing agent for overlapping files or state. +D4 synthesis requires findings from at least two independent D3 work items in +the task packet. Reuse finished findings; do not restart D3 work for a second +opinion without a specific unresolved issue. Start synthesis only after its +inputs are available and a child slot is free. The same three-child limit +includes D4 roles. D4 children are read-only; the parent owns orchestration, +final integration, and any authorized state changes. + ### Task packet Give every child only the minimum task packet required: @@ -103,13 +145,19 @@ Give every child only the minimum task packet required: - completion condition; - required output shape. -When selecting an elevated effort variant, also include the concrete reason the -base effort is likely to be materially more error-prone or expensive to rework. +For D1 Medium, state what reconciliation makes Low insufficient. Select D1 +High directly when its conditions are already clear; do not require a failed +Low or Medium attempt. +When selecting an upper role, include the concrete reason the standard role +is insufficient. For D1/D2, explain why Medium is more error-prone or expensive +to rework; for D3, explain why High is insufficient given the uncertainty and +consequence of the decision. For D4, explain why xhigh is insufficient to +resolve conflicts across the supplied D3 findings and their combined impact. The packet must explicitly tell the child not to delegate. Require distilled findings instead of raw logs. If a reasonably selected -lower-cost role returns `NEEDS_ESCALATION`, escalate only with concrete -evidence. Do not route an obvious D2 or D3 item through a cheaper role merely +lower-effort role returns `NEEDS_ESCALATION`, escalate only with concrete +evidence. Do not route an obvious D2 or D3 item through a lower-class role merely to obtain an escalation result. After a spawn succeeds, the parent must not perform the same assigned work in parallel. Wait for the child and limit parent-side checks to validating the diff --git a/config/config.task-aware.toml b/config/config.task-aware.toml index 5f046b8..69b12d2 100644 --- a/config/config.task-aware.toml +++ b/config/config.task-aware.toml @@ -1,4 +1,9 @@ # Merge this section into ~/.codex/config.toml. +# Optional parent defaults (place before any table, or use --set-astra-default): +# model = "gpt-6-astra" +# model_reasoning_effort = "xhigh" +# Standard installation preserves the existing parent model and effort. + [agents] enabled = true max_concurrent_threads_per_session = 3 diff --git a/scripts/Install-TaskAwareAgent.ps1 b/scripts/Install-TaskAwareAgent.ps1 index e1ef56b..82c9d6d 100644 --- a/scripts/Install-TaskAwareAgent.ps1 +++ b/scripts/Install-TaskAwareAgent.ps1 @@ -4,12 +4,17 @@ param( if ($env:CODEX_HOME) { $env:CODEX_HOME } else { Join-Path $HOME '.codex' } ), + [switch]$SetAstraDefault, [switch]$SetSolDefault, [switch]$EnableFullAccess ) $ErrorActionPreference = 'Stop' +if ($SetSolDefault) { + throw '-SetSolDefault was retired; use -SetAstraDefault instead.' +} + $RepositoryRoot = Split-Path -Parent $PSScriptRoot $AgentsSource = Join-Path $RepositoryRoot 'agents' $PolicySource = Join-Path $RepositoryRoot 'config/AGENTS.task-aware.md' @@ -134,11 +139,14 @@ function Set-TopLevelTomlValue { $expectedAgentFiles = @( 'luna-task.toml', + 'luna-task-medium.toml', 'luna-task-max.toml', 'terra-worker.toml', 'terra-worker-max.toml', 'sol-specialist.toml', - 'sol-specialist-max.toml' + 'sol-specialist-max.toml', + 'astra-architect.toml', + 'astra-architect-max.toml' ) $retiredAgentFiles = @('luna-task-high.toml', 'terra-worker-high.toml') if (-not (Test-Path -LiteralPath $PolicySource -PathType Leaf)) { @@ -202,8 +210,8 @@ $config = Remove-TomlSectionKeys -Content $config -Section 'features' -Keys @( 'multi_agent' ) -if ($SetSolDefault) { - $config = Set-TopLevelTomlValue -Content $config -Key 'model' -Value '"gpt-5.6-sol"' +if ($SetAstraDefault) { + $config = Set-TopLevelTomlValue -Content $config -Key 'model' -Value '"gpt-6-astra"' $config = Set-TopLevelTomlValue -Content $config -Key 'model_reasoning_effort' -Value '"xhigh"' } diff --git a/scripts/Test-TaskAwareAgent.ps1 b/scripts/Test-TaskAwareAgent.ps1 index 6c9a6e4..919bf16 100644 --- a/scripts/Test-TaskAwareAgent.ps1 +++ b/scripts/Test-TaskAwareAgent.ps1 @@ -53,35 +53,50 @@ Assert-FileContains -Path $agentsMdPath -Patterns @( 'Task-aware delegation policy', 'agent_type\s*=\s*"luna_task"', 'agent_type\s*=\s*"luna_task_max"', + 'agent_type\s*=\s*"luna_task_medium"', + 'agent_type\s*=\s*"astra_architect"', + 'agent_type\s*=\s*"astra_architect_max"', 'agent_type\s*=\s*"terra_worker"', 'agent_type\s*=\s*"terra_worker_max"', 'agent_type\s*=\s*"sol_specialist"', 'agent_type\s*=\s*"sol_specialist_max"', 'Classify capability first, then choose reasoning effort', 'Higher effort never expands a role''s permissions', - 'Lower model prices reduce the\s+threshold for elevated effort', - 'Use Max as the single elevated effort for D1-D3', - 'Do not add an xhigh middle lane', - 'concrete reason the\s+base effort is likely to be materially more error-prone', + 'The target parent is GPT-6 Astra', + 'Children never delegate', + 'D4 synthesis requires findings from at least two independent D3 work items', + 'same three-child limit', + 'For D1 Medium, state what reconciliation makes Low insufficient', + 'Keep verification proportional to the change', + 'All nine child roles use gpt-6-astra with low, medium, high, xhigh, or max reasoning', + 'D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max', + 'D3 uses Astra High by default and Astra xhigh', + 'The luna, terra, sol, and _max names no longer specify the model or effort', + 'include the concrete reason the standard role', + 'explain why Medium is more error-prone', + 'for D3, explain why High is insufficient', 'bounded read-only investigation or verification', 'Inputs, the\s+output contract, and the success condition must be explicit', 'State-changing implementation', 'tool-heavy multi-step work', 'requires ordinary judgment', - 'Do not split an atomic D0 item solely because Luna is inexpensive', - 'Do not route an obvious D2 or D3 item through a cheaper role', + 'Do not split an atomic D0 item solely to use a lower effort', + 'Do not route an obvious D2 or D3 item through a lower-class role', 'fork_turns\s*=\s*"none"', 'packet must explicitly tell the child not to delegate', '' ) $expectedAgents = [ordered]@{ - 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-5.6-luna'; Effort = 'low'; Sandbox = 'read-only' } - 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-5.6-luna'; Effort = 'max'; Sandbox = 'read-only' } - 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-5.6-terra'; Effort = 'medium'; Sandbox = $null } - 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-5.6-terra'; Effort = 'max'; Sandbox = $null } - 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-5.6-sol'; Effort = 'high'; Sandbox = 'read-only' } - 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-5.6-sol'; Effort = 'max'; Sandbox = 'read-only' } + 'luna-task-medium.toml' = [ordered]@{ Name = 'luna_task_medium'; Model = 'gpt-6-astra'; Effort = 'medium'; Sandbox = 'read-only' } + 'astra-architect.toml' = [ordered]@{ Name = 'astra_architect'; Model = 'gpt-6-astra'; Effort = 'xhigh'; Sandbox = 'read-only' } + 'astra-architect-max.toml' = [ordered]@{ Name = 'astra_architect_max'; Model = 'gpt-6-astra'; Effort = 'max'; Sandbox = 'read-only' } + 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-6-astra'; Effort = 'low'; Sandbox = 'read-only' } + 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = 'read-only' } + 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-6-astra'; Effort = 'medium'; Sandbox = $null } + 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = $null } + 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = 'read-only' } + 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-6-astra'; Effort = 'xhigh'; Sandbox = 'read-only' } } foreach ($entry in $expectedAgents.GetEnumerator()) { @@ -106,11 +121,18 @@ foreach ($entry in $expectedAgents.GetEnumerator()) { $patterns += 'success condition' $patterns += 'Do not use for material judgment, broad investigation, or state changes' } + elseif ($entry.Key -eq 'luna-task-medium.toml') { + $patterns += 'bounded D1 work with fixed inputs' + $patterns += 'modest reconciliation across files or' + $patterns += 'objective success condition' + $patterns += 'Do not use for material judgment, broad investigation, or state changes' + $patterns += 'Do not broaden scope or delegate' + } elseif ($entry.Key -eq 'luna-task-max.toml') { $patterns += 'D1 work that remains deterministic, read-only, and objectively' $patterns += 'dense cross-checking across heterogeneous inputs' $patterns += 'Do not use for material judgment, broad investigation, or state changes' - $patterns += 'Use Max reasoning for completeness and cross-checking' + $patterns += 'Use High reasoning for completeness and cross-checking' $patterns += 'not to broaden the task''s\s+capability boundary' } elseif ($entry.Key -eq 'terra-worker.toml') { @@ -122,7 +144,7 @@ foreach ($entry in $expectedAgents.GetEnumerator()) { $patterns += 'D2 work that stays within ordinary engineering judgment' $patterns += 'many\s+coupled constraints' $patterns += 'Do not use for unresolved architectural trade-offs' - $patterns += 'Use Max reasoning for coupled constraints, edge cases, and verification' + $patterns += 'Use High reasoning for coupled constraints, edge cases, and verification' $patterns += 'not to\s+broaden the task''s capability boundary' } elseif ($entry.Key -eq 'sol-specialist-max.toml') { @@ -134,6 +156,15 @@ foreach ($entry in $expectedAgents.GetEnumerator()) { $patterns += 'Use as the default for one bounded D3' $patterns += 'Prefer sol_specialist_max when uncertainty and consequence are both' } + elseif ($entry.Key -in @('astra-architect.toml', 'astra-architect-max.toml')) { + $patterns += 'bounded D4 synthesis' + $patterns += 'at least two independent D3' + $patterns += 'read-only synthesis role; the parent owns orchestration' + $patterns += 'NEEDS_INPUT' + $patterns += 'Do not repeat completed investigations' + $patterns += 'Do not delegate' + $patterns += 'Do not modify files or external state' + } Assert-FileContains -Path (Join-Path $agentsPath $entry.Key) -Patterns $patterns } diff --git a/scripts/install-task-aware-agent.sh b/scripts/install-task-aware-agent.sh index 41b2c9b..d7eac1d 100755 --- a/scripts/install-task-aware-agent.sh +++ b/scripts/install-task-aware-agent.sh @@ -10,7 +10,8 @@ Install the Codex Task-Aware Agent configuration globally. Options: --codex-home PATH Target Codex home (default: $CODEX_HOME or ~/.codex) - --set-sol-default Set gpt-5.6-sol with xhigh reasoning as the default + --set-astra-default Set gpt-6-astra with xhigh reasoning as the default + --set-sol-default Retired; use --set-astra-default instead --enable-full-access Set approval_policy=never and danger-full-access -h, --help Show this help EOF @@ -22,7 +23,7 @@ die() { } codex_home="${CODEX_HOME:-$HOME/.codex}" -set_sol_default=false +set_astra_default=false enable_full_access=false while (($# > 0)); do @@ -33,7 +34,10 @@ while (($# > 0)); do shift 2 ;; --set-sol-default) - set_sol_default=true + die '--set-sol-default was retired; use --set-astra-default instead' + ;; + --set-astra-default) + set_astra_default=true shift ;; --enable-full-access) @@ -72,11 +76,14 @@ command -v awk >/dev/null || die 'awk is required' expected_agent_files=( luna-task.toml + luna-task-medium.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml sol-specialist.toml sol-specialist-max.toml + astra-architect.toml + astra-architect-max.toml ) retired_agent_files=(luna-task-high.toml terra-worker-high.toml) for agent_file in "${expected_agent_files[@]}"; do @@ -315,8 +322,8 @@ remove_toml_section_key "$config_path" agents max_threads remove_toml_section_key "$config_path" agents max_depth remove_toml_section_key "$config_path" features multi_agent -if [[ "$set_sol_default" == true ]]; then - set_top_level_toml_value "$config_path" model '"gpt-5.6-sol"' +if [[ "$set_astra_default" == true ]]; then + set_top_level_toml_value "$config_path" model '"gpt-6-astra"' set_top_level_toml_value "$config_path" model_reasoning_effort '"xhigh"' fi diff --git a/scripts/test-task-aware-agent.sh b/scripts/test-task-aware-agent.sh index 1061e09..4916be4 100755 --- a/scripts/test-task-aware-agent.sh +++ b/scripts/test-task-aware-agent.sh @@ -94,26 +94,36 @@ assert_file_contains "$agents_md_path" \ 'Task-aware delegation policy' \ 'agent_type[[:space:]]*=[[:space:]]*"luna_task"' \ 'agent_type[[:space:]]*=[[:space:]]*"luna_task_max"' \ + 'agent_type[[:space:]]*=[[:space:]]*"luna_task_medium"' \ + 'agent_type[[:space:]]*=[[:space:]]*"astra_architect"' \ + 'agent_type[[:space:]]*=[[:space:]]*"astra_architect_max"' \ 'agent_type[[:space:]]*=[[:space:]]*"terra_worker"' \ 'agent_type[[:space:]]*=[[:space:]]*"terra_worker_max"' \ 'agent_type[[:space:]]*=[[:space:]]*"sol_specialist"' \ 'agent_type[[:space:]]*=[[:space:]]*"sol_specialist_max"' \ 'Classify capability first, then choose reasoning effort' \ "Higher effort never expands a role's permissions" \ - 'Lower model prices reduce the' \ - 'threshold for elevated effort' \ - 'Use Max as the single elevated effort for D1-D3' \ - 'Do not add an xhigh middle lane' \ - 'concrete reason the' \ - 'base effort is likely to be materially more error-prone' \ + 'The target parent is GPT-6 Astra' \ + 'Children never delegate' \ + 'D4 synthesis requires findings from at least two independent D3 work items' \ + 'same three-child limit' \ + 'For D1 Medium, state what reconciliation makes Low insufficient' \ + 'Keep verification proportional to the change' \ + 'All nine child roles use gpt-6-astra with low, medium, high, xhigh, or max reasoning' \ + 'D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max' \ + 'D3 uses Astra High by default and Astra xhigh' \ + 'The luna, terra, sol, and _max names no longer specify the model or effort' \ + 'include the concrete reason the standard role' \ + 'explain why Medium is more error-prone' \ + 'for D3, explain why High is insufficient' \ 'bounded read-only investigation or verification' \ 'Inputs, the' \ 'output contract, and the success condition must be explicit' \ 'State-changing implementation' \ 'tool-heavy multi-step work' \ 'requires ordinary judgment' \ - 'Do not split an atomic D0 item solely because Luna is inexpensive' \ - 'Do not route an obvious D2 or D3 item through a cheaper role' \ + 'Do not split an atomic D0 item solely to use a lower effort' \ + 'Do not route an obvious D2 or D3 item through a lower-class role' \ 'fork_turns[[:space:]]*=[[:space:]]*"none"' \ 'packet must explicitly tell the child not to delegate' \ '' @@ -122,7 +132,7 @@ assert_file_contains "$agents_path/luna-task.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-luna"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"low"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'Use as the default for compact, homogeneous D1' \ @@ -131,17 +141,30 @@ assert_file_contains "$agents_path/luna-task.toml" \ 'success condition' \ 'Do not use for material judgment, broad investigation, or state changes' +assert_file_contains "$agents_path/luna-task-medium.toml" \ + '^name[[:space:]]*=[[:space:]]*"luna_task_medium"[[:space:]]*$' \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' \ + '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ + 'bounded D1 work with fixed inputs' \ + 'modest reconciliation across files or' \ + 'objective success condition' \ + 'Do not use for material judgment, broad investigation, or state changes' \ + 'Do not broaden scope or delegate' + assert_file_contains "$agents_path/luna-task-max.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task_max"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-luna"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'D1 work that remains deterministic, read-only, and objectively' \ 'dense cross-checking across heterogeneous inputs' \ 'Do not use for material judgment, broad investigation, or state changes' \ - 'Use Max reasoning for completeness and cross-checking' \ + 'Use High reasoning for completeness and cross-checking' \ "not to broaden the task's" \ 'capability boundary' @@ -154,7 +177,7 @@ assert_file_contains "$agents_path/terra-worker.toml" \ 'multi-step work' \ 'requires ordinary' \ 'judgment while keeping clear success criteria' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-terra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' assert_file_contains "$agents_path/terra-worker-max.toml" \ @@ -165,17 +188,17 @@ assert_file_contains "$agents_path/terra-worker-max.toml" \ 'many' \ 'coupled constraints' \ 'Do not use for unresolved architectural trade-offs' \ - 'Use Max reasoning for coupled constraints, edge cases, and verification' \ + 'Use High reasoning for coupled constraints, edge cases, and verification' \ 'not to' \ "broaden the task's capability boundary" \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-terra"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' assert_file_contains "$agents_path/sol-specialist.toml" \ '^name[[:space:]]*=[[:space:]]*"sol_specialist"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-sol"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'Use as the default for one bounded D3' \ @@ -188,10 +211,40 @@ assert_file_contains "$agents_path/sol-specialist-max.toml" \ 'D3 work when both uncertainty and consequence are high' \ 'security-sensitive trade-offs' \ 'reasoning variance' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-sol"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"xhigh"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' +assert_file_contains "$agents_path/astra-architect.toml" \ + '^name[[:space:]]*=[[:space:]]*"astra_architect"[[:space:]]*$' \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"xhigh"[[:space:]]*$' \ + '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ + 'bounded D4 synthesis' \ + 'at least two independent D3' \ + 'read-only synthesis role; the parent owns orchestration' \ + 'NEEDS_INPUT' \ + 'Do not repeat completed investigations' \ + 'Do not delegate' \ + 'Do not modify files or external state' + +assert_file_contains "$agents_path/astra-architect-max.toml" \ + '^name[[:space:]]*=[[:space:]]*"astra_architect_max"[[:space:]]*$' \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ + '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ + 'bounded D4 synthesis' \ + 'at least two independent D3' \ + 'read-only synthesis role; the parent owns orchestration' \ + 'NEEDS_INPUT' \ + 'Do not repeat completed investigations' \ + 'Do not delegate' \ + 'Do not modify files or external state' + assert_file_absent "$agents_path/luna-task-high.toml" assert_file_absent "$agents_path/terra-worker-high.toml" From c3214473482c72e24a51b5b679c4dc126a041a28 Mon Sep 17 00:00:00 2001 From: Codex Date: Tue, 8 Sep 2026 11:01:10 +0900 Subject: [PATCH 2/4] Preserve local task-aware changes before Astra merge --- .github/workflows/ci.yml | 334 +++++++++++++++++++- CHANGELOG.md | 34 +- README.md | 163 ++++++---- RELEASE_CHECKLIST.md | 13 +- agents/luna-task-max.toml | 4 + agents/luna-task.toml | 4 + agents/sol-admin-max.toml | 43 +++ agents/sol-specialist-max.toml | 4 + agents/sol-specialist.toml | 4 + agents/terra-worker-max.toml | 6 + agents/terra-worker.toml | 6 + config/AGENTS.task-aware.md | 282 ++++++++++------- rules/full-admin.rules | 84 +++++ scripts/Install-TaskAwareAgent.ps1 | 42 ++- scripts/Test-TaskAwareAgent.ps1 | 424 +++++++++++++++++++------ scripts/install-task-aware-agent.sh | 63 +++- scripts/test-task-aware-agent.sh | 463 +++++++++++++++++++++------- 17 files changed, 1546 insertions(+), 427 deletions(-) create mode 100644 agents/sol-admin-max.toml create mode 100644 rules/full-admin.rules diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 004a427..0b0e44d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,7 +15,7 @@ permissions: contents: read env: - CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.145.0' }} + CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.149.0' }} jobs: windows: @@ -125,6 +125,187 @@ jobs: } if (-not $stopped) { throw 'Malformed AGENTS.md markers were not rejected.' } + $literalMarkerHome = Join-Path $env:RUNNER_TEMP 'task-aware-literal-markers' + New-Item -ItemType Directory -Force -Path $literalMarkerHome | Out-Null + $literalMarkerLine = 'User text: and are examples.' + [IO.File]::WriteAllText( + (Join-Path $literalMarkerHome 'AGENTS.md'), + $literalMarkerLine + "`n", + [Text.UTF8Encoding]::new($false) + ) + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $literalMarkerHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $literalMarkerHome -SkipRuntime + if ((Get-Content -Raw -LiteralPath (Join-Path $literalMarkerHome 'AGENTS.md')) -notmatch [regex]::Escape($literalMarkerLine)) { + throw 'Inline marker examples were not preserved.' + } + + $preserveHome = Join-Path $env:RUNNER_TEMP 'task-aware-preserve' + New-Item -ItemType Directory -Force -Path $preserveHome | Out-Null + $preserveConfig = @( + 'model = "gpt-6-astra"' + 'model_reasoning_effort = "max"' + 'approval_policy = "never"' + 'sandbox_mode = "workspace-write"' + 'unmanaged_setting = "retain-me"' + ) -join "`n" + [IO.File]::WriteAllText( + (Join-Path $preserveHome 'config.toml'), + $preserveConfig, + [Text.UTF8Encoding]::new($false) + ) + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $preserveHome + $preserved = Get-Content -Raw -LiteralPath (Join-Path $preserveHome 'config.toml') + foreach ($line in @( + 'model = "gpt-6-astra"' + 'model_reasoning_effort = "max"' + 'approval_policy = "never"' + 'sandbox_mode = "workspace-write"' + 'unmanaged_setting = "retain-me"' + )) { + if ($preserved -notmatch [regex]::Escape($line)) { + throw "Default installation did not preserve: $line" + } + } + + $astraHome = Join-Path $env:RUNNER_TEMP 'task-aware-astra' + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $astraHome -SetAstraDefault + $astraConfig = Get-Content -Raw -LiteralPath (Join-Path $astraHome 'config.toml') + if ($astraConfig -notmatch '(?m)^model = "gpt-6-astra"$') { + throw 'Astra parent model was not selected.' + } + if ($astraConfig -notmatch '(?m)^model_reasoning_effort = "xhigh"$') { + throw 'Astra parent reasoning effort was not selected.' + } + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $astraHome -ConfigOnlyRuntime + + $solHome = Join-Path $env:RUNNER_TEMP 'task-aware-sol' + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $solHome -SetSolDefault + $solConfig = Get-Content -Raw -LiteralPath (Join-Path $solHome 'config.toml') + if ($solConfig -notmatch '(?m)^model = "gpt-5.6-sol"$') { + throw 'Legacy Sol parent model selection regressed.' + } + if ($solConfig -notmatch '(?m)^model_reasoning_effort = "xhigh"$') { + throw 'Legacy Sol parent reasoning effort regressed.' + } + + $selectionConflictHome = Join-Path $env:RUNNER_TEMP 'task-aware-selection-conflict' + New-Item -ItemType Directory -Force -Path $selectionConflictHome | Out-Null + $selectionConfigPath = Join-Path $selectionConflictHome 'config.toml' + [IO.File]::WriteAllText( + $selectionConfigPath, + "model = ""preserve""`n", + [Text.UTF8Encoding]::new($false) + ) + $beforeHash = (Get-FileHash -Algorithm SHA256 -LiteralPath $selectionConfigPath).Hash + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $selectionConflictHome -SetSolDefault -SetAstraDefault -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped) { throw 'Conflicting parent model switches were accepted.' } + $afterHash = (Get-FileHash -Algorithm SHA256 -LiteralPath $selectionConfigPath).Hash + if ($beforeHash -cne $afterHash) { throw 'Conflicting parent model switches mutated config.toml.' } + foreach ($path in @('task-aware-backups', 'agents', 'rules', 'AGENTS.md')) { + if (Test-Path -LiteralPath (Join-Path $selectionConflictHome $path)) { + throw "Conflicting parent model switches created $path." + } + } + + foreach ($markerCase in @('duplicate', 'reversed')) { + $markerHome = Join-Path $env:RUNNER_TEMP "task-aware-markers-$markerCase" + New-Item -ItemType Directory -Force -Path $markerHome | Out-Null + $markerContent = switch ($markerCase) { + 'duplicate' { + "`n`n`n`n" + } + default { + "`n`n" + } + } + [IO.File]::WriteAllText((Join-Path $markerHome 'AGENTS.md'), $markerContent, [Text.UTF8Encoding]::new($false)) + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $markerHome -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped) { throw "Marker case unexpectedly succeeded: $markerCase" } + } + + $crlfHome = Join-Path $env:RUNNER_TEMP 'task-aware-crlf' + New-Item -ItemType Directory -Force -Path $crlfHome | Out-Null + $policy = Get-Content -Raw -LiteralPath ./config/AGENTS.task-aware.md + $crlfPolicy = [regex]::Replace($policy, "`r?`n", "`r`n") + [IO.File]::WriteAllText((Join-Path $crlfHome 'AGENTS.md'), $crlfPolicy, [Text.UTF8Encoding]::new($false)) + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $crlfHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $crlfHome -SkipRuntime + + foreach ($drift in @('policy', 'role', 'rule')) { + $driftHome = Join-Path $env:RUNNER_TEMP "task-aware-drift-$drift" + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $driftHome + switch ($drift) { + 'policy' { + $driftPath = Join-Path $driftHome 'AGENTS.md' + $driftContent = Get-Content -Raw -LiteralPath $driftPath + $driftContent = $driftContent.Replace( + '', + "# drift`n" + ) + [IO.File]::WriteAllText($driftPath, $driftContent, [Text.UTF8Encoding]::new($false)) + } + 'role' { + [IO.File]::AppendAllText( + (Join-Path $driftHome 'agents/luna-task.toml'), + "# drift`n", + [Text.UTF8Encoding]::new($false) + ) + } + 'rule' { + [IO.File]::AppendAllText( + (Join-Path $driftHome 'rules/task-aware-full-admin.rules'), + "# drift`n", + [Text.UTF8Encoding]::new($false) + ) + } + } + $stopped = $false + try { + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $driftHome -SkipRuntime -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped) { throw "Drift case unexpectedly passed: $drift" } + } + + - name: Reject oversized managed policy + shell: pwsh + run: | + $budgetRepo = Join-Path $env:RUNNER_TEMP 'task-aware-budget-repo' + $budgetHome = Join-Path $env:RUNNER_TEMP 'task-aware-budget-home' + New-Item -ItemType Directory -Path $budgetRepo | Out-Null + foreach ($directory in @('scripts', 'config', 'agents', 'rules')) { + Copy-Item -LiteralPath $directory -Destination $budgetRepo -Recurse + } + $budgetSource = Join-Path $budgetRepo 'config/AGENTS.task-aware.md' + $policy = Get-Content -Raw -LiteralPath $budgetSource + $end = '' + $policy = $policy.Replace($end, "`n$end") + [IO.File]::WriteAllText($budgetSource, $policy, [Text.UTF8Encoding]::new($false)) + & (Join-Path $budgetRepo 'scripts/Install-TaskAwareAgent.ps1') -CodexHome $budgetHome + $stopped = $false + try { + & (Join-Path $budgetRepo 'scripts/Test-TaskAwareAgent.ps1') -CodexHome $budgetHome -SkipRuntime + } + catch { + if ($_.Exception.Message -notmatch 'Managed policy exceeds the 10 KiB maintenance budget') { throw } + $stopped = $true + } + if (-not $stopped) { throw 'An oversized managed policy passed validation.' } + linux: name: Linux clean-home round trip runs-on: ubuntu-latest @@ -215,3 +396,154 @@ jobs: printf '%s\n' 'Malformed AGENTS.md markers were not rejected.' >&2 exit 1 fi + + literal_marker_home="$RUNNER_TEMP/task-aware-literal-markers" + mkdir -p -- "$literal_marker_home" + literal_marker_line='User text: and are examples.' + printf '%s\n' "$literal_marker_line" > "$literal_marker_home/AGENTS.md" + scripts/install-task-aware-agent.sh --codex-home "$literal_marker_home" + scripts/test-task-aware-agent.sh \ + --codex-home "$literal_marker_home" \ + --skip-runtime + if ! grep -Fqx -- "$literal_marker_line" "$literal_marker_home/AGENTS.md"; then + printf '%s\n' 'Inline marker examples were not preserved.' >&2 + exit 1 + fi + + preserve_home="$RUNNER_TEMP/task-aware-preserve" + mkdir -p -- "$preserve_home" + printf '%s\n' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"' \ + 'unmanaged_setting = "retain-me"' > "$preserve_home/config.toml" + scripts/install-task-aware-agent.sh --codex-home "$preserve_home" + for preserved_line in \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"' \ + 'unmanaged_setting = "retain-me"'; do + if ! grep -Fqx -- "$preserved_line" "$preserve_home/config.toml"; then + printf 'Default installation did not preserve: %s\n' "$preserved_line" >&2 + exit 1 + fi + done + + astra_home="$RUNNER_TEMP/task-aware-astra" + scripts/install-task-aware-agent.sh --codex-home "$astra_home" --set-astra-default + if ! grep -Fxq 'model = "gpt-6-astra"' "$astra_home/config.toml"; then + printf '%s\n' 'Astra parent model was not selected.' >&2 + exit 1 + fi + if ! grep -Fxq 'model_reasoning_effort = "xhigh"' "$astra_home/config.toml"; then + printf '%s\n' 'Astra parent reasoning effort was not selected.' >&2 + exit 1 + fi + scripts/test-task-aware-agent.sh \ + --codex-home "$astra_home" \ + --config-only-runtime + + sol_home="$RUNNER_TEMP/task-aware-sol" + scripts/install-task-aware-agent.sh --codex-home "$sol_home" --set-sol-default + if ! grep -Fxq 'model = "gpt-5.6-sol"' "$sol_home/config.toml"; then + printf '%s\n' 'Legacy Sol parent model selection regressed.' >&2 + exit 1 + fi + if ! grep -Fxq 'model_reasoning_effort = "xhigh"' "$sol_home/config.toml"; then + printf '%s\n' 'Legacy Sol parent reasoning effort regressed.' >&2 + exit 1 + fi + + selection_conflict_home="$RUNNER_TEMP/task-aware-selection-conflict" + mkdir -p -- "$selection_conflict_home" + printf '%s\n' 'model = "preserve"' > "$selection_conflict_home/config.toml" + before_hash=$(sha256sum "$selection_conflict_home/config.toml" | awk '{print $1}') + if scripts/install-task-aware-agent.sh \ + --codex-home "$selection_conflict_home" \ + --set-sol-default \ + --set-astra-default; then + printf '%s\n' 'Conflicting parent model switches were accepted.' >&2 + exit 1 + fi + after_hash=$(sha256sum "$selection_conflict_home/config.toml" | awk '{print $1}') + if [[ "$before_hash" != "$after_hash" ]]; then + printf '%s\n' 'Conflicting parent model switches mutated config.toml.' >&2 + exit 1 + fi + for path in task-aware-backups agents rules AGENTS.md; do + if [[ -e "$selection_conflict_home/$path" ]]; then + printf 'Conflicting parent model switches created %s.\n' "$path" >&2 + exit 1 + fi + done + + for marker_case in duplicate reversed; do + marker_home="$RUNNER_TEMP/task-aware-markers-$marker_case" + mkdir -p -- "$marker_home" + if [[ "$marker_case" == duplicate ]]; then + printf '%s\n' \ + '' \ + '' \ + '' \ + '' > "$marker_home/AGENTS.md" + else + printf '%s\n' \ + '' \ + '' > "$marker_home/AGENTS.md" + fi + if scripts/install-task-aware-agent.sh --codex-home "$marker_home"; then + printf 'Marker case unexpectedly succeeded: %s\n' "$marker_case" >&2 + exit 1 + fi + done + + crlf_home="$RUNNER_TEMP/task-aware-crlf" + mkdir -p -- "$crlf_home" + awk '{ printf "%s\r\n", $0 }' config/AGENTS.task-aware.md > "$crlf_home/AGENTS.md" + scripts/install-task-aware-agent.sh --codex-home "$crlf_home" + scripts/test-task-aware-agent.sh --codex-home "$crlf_home" --skip-runtime + + for drift in policy role rule; do + drift_home="$RUNNER_TEMP/task-aware-drift-$drift" + scripts/install-task-aware-agent.sh --codex-home "$drift_home" + case "$drift" in + policy) + sed -i '//i# drift' "$drift_home/AGENTS.md" + ;; + role) + printf '%s\n' '# drift' >> "$drift_home/agents/luna-task.toml" + ;; + rule) + printf '%s\n' '# drift' >> "$drift_home/rules/task-aware-full-admin.rules" + ;; + esac + if scripts/test-task-aware-agent.sh --codex-home "$drift_home" --skip-runtime; then + printf 'Drift case unexpectedly passed: %s\n' "$drift" >&2 + exit 1 + fi + done + + - name: Reject oversized managed policy + run: | + budget_repo="$RUNNER_TEMP/task-aware-budget-repo" + budget_home="$RUNNER_TEMP/task-aware-budget-home" + mkdir -p -- "$budget_repo" + cp -R -- scripts config agents rules "$budget_repo/" + budget_source="$budget_repo/config/AGENTS.task-aware.md" + awk ' + // { + printf "" + } + { print } + ' "$budget_source" > "$budget_source.tmp" + mv -- "$budget_source.tmp" "$budget_source" + "$budget_repo/scripts/install-task-aware-agent.sh" --codex-home "$budget_home" + if "$budget_repo/scripts/test-task-aware-agent.sh" --codex-home "$budget_home" --skip-runtime > "$RUNNER_TEMP/budget-error.txt" 2>&1; then + printf '%s\n' 'An oversized managed policy passed validation.' >&2 + exit 1 + fi + grep -Fq 'Managed policy exceeds the 10 KiB maintenance budget' "$RUNNER_TEMP/budget-error.txt" diff --git a/CHANGELOG.md b/CHANGELOG.md index bb861ae..17c3686 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,22 +4,36 @@ ## [Unreleased] +### 追加 + +- 親モデルとして Astra(`gpt-6-astra`)に対応。Windows の `-SetAstraDefault` と Linux の `--set-astra-default` で Astra xhigh を明示的に選択できる。 +- 現行運用の管理者操作用 `sol_admin_max` と承認規則を配布。操作・対象・権限範囲の明示許可を必須とし、通常役からの自動昇格を禁止する。 +- 子の作業指示に有限の `NO_PROGRESS_LIMIT`、`HARD_DEADLINE`、`SAFE_CANCELLATION` と最終返却の形式を定義。 +- 複数工程の必須確認、任意の証拠、終了条件を先に固定し、検証範囲を追加できる条件を定義。 + ### 変更 -- 現行の価格差を踏まえ、Luna Low の D1 を固定入力、明示的な出力契約、客観的完了条件を持つ限定的な読み取り専用調査・検証まで拡張。 -- Terra Medium の D2 を、状態変更を伴う実装、tool-heavy な複数工程、通常判断が必要な調査・検証として明確化。 +- 常時読み込む委任ポリシーの重複を統合し、判断順序、役割、権限、検証の終了条件、有限の待機を維持したまま短縮。管理ブロックに10 KiBの保守上限を設け、追記による再肥大化を検証で検出する。 +- 配布ポリシーを現行の v3.1 に更新。権限範囲を固定し、能力分類、委譲可否の判定、役割・推論労力の選択の順に処理する。 +- D1 は固定入力、明示的な出力契約、客観的完了条件を持つ限定的な読み取り専用作業とする。 +- D2 は範囲と完了条件が明確な状態変更を伴う実装・修復・統合とし、通常判断が必要という理由だけで読み取り専用調査を Terra へ送らない。 +- 単独の D3 は原則として親が担当し、Sol への委譲には独立した判断・証拠の価値と通常の委譲条件の両方を求める。 - D1 に Luna Max、D2 に Terra Max、D3 に Sol Max の上位 variant を追加。 -- 中間の xhigh role は設けず、標準とMaxの二段階に統一。 -- 能力クラスを先に決め、同じクラス内で標準またはMax枠を選ぶ二段階ルーティングへ変更。 -- 価格低下を、Maxによる完全性向上または手戻り回避を選びやすくする根拠として反映。 -- 最小十分な役割を選びつつ、D0 の細分化、明白な D2/D3 の意図的な過小ルーティング、不要な microtask fan-out を禁止。 -- 価格低下後も、統合負荷と競合を抑えるため同時に開く子スレッドの上限を3つに維持。 +- 標準とMaxの二段階を維持し、検証範囲、制約の結合、手戻りなどの具体的な必要性に基づいて Max を選ぶ。中間の xhigh role は設けない。 +- 通常の六役は管理者昇格を拒否し、`approval_policy = "never"` を明示。Terra の二役には `sandbox_mode = "workspace-write"` を明示。 +- D0 の細分化、明白な D2/D3 の意図的な過小ルーティング、不要な小作業への分割を禁止。 +- ポリシーから価格、固定スレッド数、旧ランタイム版に依存する説明を外し、実行環境の同時起動上限を尊重する。配布設定の上限は子3つを維持。 +- 委譲が親の権限を広げないよう、作業指示に権限と変更を許す範囲を追加。 +- 標準導入で親モデル・推論労力・権限を保持し、従来の Sol xhigh 選択も維持。Astra と Sol の同時選択は変更前に拒否する。 +- README を Astra の導入手順と七役に更新し、Ultra を前提にした説明を改める。 ### 検証 -- Windows/Linux validator に、価格対応後の D1/D2 境界と過剰委譲防止規則の検査を追加。 -- Windows/Linux installer と validator を六 role の配置、model、effort、sandbox 検査へ拡張し、旧Luna/Terra High roleをbackup後に除去する移行を追加。 -- release 前の live probe を、六 role の model/effort 確認と D0 から D3 までの標準・Maxルーティング確認へ拡張。 +- Windows/Linux の導入・検証処理を七役と管理者操作用の規則へ拡張。旧 Luna/Terra High role はバックアップ後に除去する。 +- マーカーの一意性・順序、配置されたポリシーと配布元の一致、七役と規則のバイト単位での一致を検査する。 +- v3.1 の判断順序、権限境界、有限の待機・終了条件を、管理対象のポリシーブロック内で検査する。 +- 隔離した `CODEX_HOME` で Astra/Sol の選択、既存設定保持、相互排他、再導入、バックアップ、不正なマーカーやファイルの差異の検出を確認する回帰テストを追加。 +- リリース時の実動作確認を、Astra 親、委譲条件を満たした標準・Max役、管理者許可がない場合の拒否へ更新。設定の一致とモデルの実行確認を分けて記録する。 ## [0.1.0] - 2026-07-26 diff --git a/README.md b/README.md index ddcbe89..750f55a 100644 --- a/README.md +++ b/README.md @@ -6,43 +6,45 @@ Codex の親エージェントがタスクを難易度別に分類し、必要な場合だけ役割別のカスタムエージェントへ委譲するための設定一式です。 OpenAI の公式製品ではなく、Codex の公開仕様に基づくコミュニティプロジェクトです。 -この構成が想定する親は、`gpt-5.6-sol` をクライアントの Ultra モードで動かす **Sol Ultra** です。 +この配布構成では、親に **Astra(`gpt-6-astra`)** を使えます。 親は難易度判定、タスクの分割、子の選択、結果の統合を担当します。 -子には Luna Low/Max、Terra Medium/Max、Sol High/Max を使い分けます。 +子には Luna Low/Max、Terra Medium/Max、Sol High/Max と、明示的に許可された管理者操作用の Sol Max を使い分けます。 +従来の Sol を親にする構成も引き続き利用できます。 -これにより、すべての子が親の高い推論労力を継承して消費量が膨らむことを避けつつ、メインスレッドへ途中経過が流れ込む量を抑えます。 -現行の価格差は、固定契約の読み取り専用作業を Luna へ寄せ、各難度内で必要な場合に高い推論労力を選ぶ根拠にします。 -安価であることだけを理由に子を細分化はしません。 -各 subagent は独自に token と調整時間を使うため、D0 の直接処理と最大3子の上限は維持します。 +親を Astra にしても、子は各役割のモデルと推論労力を使います。 +すべての子が親の設定を継承して消費量が膨らむことを避けつつ、メインスレッドへ途中経過が流れ込む量を抑えるための構成です。 +子も個別にトークンと調整時間を使うため、独立した成果を返せる作業だけを委譲します。 リポジトリを取得しただけでは Codex の動作は変わりません。 導入スクリプトを実行し、新しいタスクで設定を読み込む必要があります。 ## 想定する実行構成 -Sol はモデル、Ultra は対応するモデルと環境で最大推論と能動的な委譲を利用する実行モードです。 -現行 Codex は、対応モデルの `model_reasoning_effort = "ultra"` も受け付けます。 -このリポジトリでは、親の統合と委譲に Ultra を残し、子の上限を Max にします。 +Astra の導入オプションは、親の既定値を `gpt-6-astra` と `model_reasoning_effort = "xhigh"`(Extra High)に設定します。 +標準導入では親の設定を保持するので、すでに選択した Astra の推論労力もそのまま使えます。 +子の推論労力は、役割ごとの標準値または Max に固定します。 -対応するアカウントとクライアントで Sol Ultra を選ぶと、親は分割可能な作業をサブエージェントへ能動的に委譲します。 -このリポジトリは、その委譲に D0 から D4 までの能力分類と、同じ難度内で推論労力を選ぶ六つの子エージェントを追加します。 +ローカルの Codex は、利用者からの明示依頼、または適用されるプロジェクト・スキルの指示に基づいて子を起動します。 +この配布ポリシーは、後述の委譲条件をすべて満たした場合に限って委譲を求めます。 +Astra や Ultra を選ぶこと自体を、子を起動する条件にはしません。詳しくは [公式 Subagents 資料](https://learn.chatgpt.com/docs/agent-configuration/subagents)を参照してください。 | 難易度 | 対象 | 実行役 | モデルと推論労力 | sandbox | | --- | --- | --- | --- | --- | -| D0 | 単純で明確な1工程 | 親が直接処理 | Sol Ultra | 親の設定 | +| D0 | 単純で明確な1工程 | 親が直接処理 | 選択した親(Astra など) | 親の設定 | | D1 標準 | 小さく均質で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Luna Low | read-only | | D1 Max枠 | D1 のまま、異種入力、密な突合、coverage 重視、多数の edge case がある作業 | `luna_task_max` | Luna Max | read-only | -| D2 標準 | 状態変更を伴う実装、tool-heavy な複数工程、通常判断が必要な調査・検証 | `terra_worker` | Terra Medium | 親から継承 | -| D2 Max枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Terra Max | 親から継承 | -| D3 標準 | 一つの難しい判断、曖昧性、高リスク、複数領域、設計判断 | `sol_specialist` | Sol High | read-only | +| D2 標準 | 範囲と完了条件が明確な、状態変更を伴う実装・修復・統合 | `terra_worker` | Terra Medium | workspace-write | +| D2 Max枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Terra Max | workspace-write | +| D3 標準 | 独立した判断や証拠が必要で、委譲条件を満たす難しい読み取り専用作業 | `sol_specialist` | Sol High | read-only | | D3 Max枠 | 不確実性と結果の重大性がともに高く、証拠競合、不可逆設計、security、敵対的 edge case を含む作業 | `sol_specialist_max` | Sol Max | read-only | -| D4 | 独立した D3 タスクが複数 | 親が分割して統合 | Sol Ultra | 親と選択した子の設定 | +| D4 | D3 の候補が複数あり、各候補の委譲条件を個別に判定 | 親が分割して統合 | 選択した親(Astra など) | 親と選択した子の設定 | +| 管理者操作 | 操作・対象・昇格方法について利用者の明示許可がある単一作業 | `sol_admin_max` | Sol Max | danger-full-access | -能力クラスを D1/D2/D3 から先に決め、その後で標準またはMax枠を選びます。 +作業を能力クラスに分類し、委譲するかを判定した後で、役割と標準またはMax枠を選びます。 Max枠を選んでも権限や能力境界は広がりません。 -Luna の二役と Sol の二役は読み取り専用で、書き込みを伴う通常実装は、親の権限を継承する Terra の二役か親が担当します。 -Ultra は親だけに残し、複数の判断をまたぐ推論と最終統合を親が担当します。 -上位枠は三モデルとも Max に統一します。 +単独の D3 は原則として親が担当し、独立した検証や判断に価値があり、すべての委譲条件を満たす場合だけ Sol へ渡します。 +読み取り専用の調査を、通常の判断が必要という理由だけで、書き込み役の Terra へ送りません。 +表の sandbox は役割ファイルの宣言値です。実効権限については[設計上の注意](#設計上の注意)も確認してください。 `xhigh` は標準とMaxの間に独立した能力境界を作らないため、別roleにはしません。 ## ルーティングの仕組み @@ -57,39 +59,54 @@ Ultra は親だけに残し、複数の判断をまたぐ推論と最終統合 3. 親のコンテキスト消費か経過時間を減らせる見込みがある。 4. 委譲の調整コストが、親による直接処理より小さい。 -条件を満たした後、親は能力クラスを選び、そのクラス内で必要十分な推論労力を選びます。 -標準 effort が基本ですが、完全性の向上または手戻りの回避が見込める場合は、価格低下を踏まえてMax枠を積極的に選べます。 -Max枠の task packet には、標準 effort では誤りや手戻りが増える具体的な理由を含めます。 +分類しただけでは委譲は決まりません。条件を満たさない作業は親が担当します。 +委譲は親に与えられた権限を広げません。 +回答、レビュー、診断、監視の作業指示(task packet)では、子の能力にかかわらずファイルと外部状態の変更を明示的に禁止します。 +標準の推論労力を基本とし、検証範囲、制約の結合、手戻りのリスクなど、必要性を示せる場合に Max を選びます。 +Max枠の task packet には、標準の推論労力では誤りや手戻りが増える具体的な理由を含めます。 原子的な D0 を Luna が安価であるという理由だけで分割しません。 同じ入力と完了条件を共有する小さな作業は、分離によって待ち時間、コンテキスト分離、証拠の独立性が改善しない限り、一つの task packet にまとめます。 子は別の子を起動しません。 -`max_concurrent_threads_per_session = 3` とポリシー上の上限により、親から同時に開く子スレッドは最大3つです。 +配布設定は `max_concurrent_threads_per_session = 3` とし、ポリシーも runtime が設定した上限を超える起動を禁止します。 +この配布設定を使う場合、親から同時に開ける子スレッドは最大3つです。 同じファイルや状態を更新するエージェントは1つに限定します。 子からの再委譲は、agent TOML と `AGENTS.md` の指示で禁止します。 -旧 `agents.max_depth` は Codex V2 で無視されるため、実効的な強制境界としては使用しません。 +導入時に除去する旧 `agents.max_depth` は、実効的な強制境界として使用しません。 `NEEDS_ESCALATION` は Codex ランタイムの自動判定ではなく、子が能力不足の根拠を親へ返すための応答規約です。 合理的に選んだ下位役割が返した場合だけ、親はその根拠を確認し、必要な上位役割へ再委譲します。 最初から D2 または D3 と明らかな作業を、昇格結果を得るためだけに Luna へ渡しません。 -子を起動するときは、`spawn_agent` の `agent_type` に `luna_task`、`luna_task_max`、`terra_worker`、`terra_worker_max`、`sol_specialist`、`sol_specialist_max` のいずれかを明示します。 +子を起動するときは、`spawn_agent` の `agent_type` に表の役割名を明示します。`sol_admin_max` には、通常の委譲条件に加えて管理者操作の明示許可が必要です。 `task_name` は子タスクの表示名とパスを付ける項目であり、custom agent の選択には使いません。 `task_name = "luna_task"` だけを指定すると、子が親のモデルと推論労力を継承するため、想定したコスト制御になりません。 D1 から D3 までの委譲では、標準とMax枠のどちらでも `agent_type` を必須とし、まず必ず引数付きで起動します。 -tool が `agent_type` または custom agent を明示的に拒否した場合だけ、既定の子を起動せず、親で処理して不一致を報告します。 +委譲条件を満たさない作業は親が直接処理します。 +委譲条件を満たすのにtoolが`agent_type`またはcustom agentを明示的に拒否した場合は、親の権限と能力に収まるときだけ親で処理し、runtimeの不一致を報告します。 各 spawn は `fork_turns = "none"` を指定し、親の全会話履歴ではなく task packet だけを子へ渡します。 -task packet 自体にも再委譲禁止を明記します。 +task packet には権限と変更を許す範囲を含め、再委譲禁止も明記します。 + +### 待機と作業の終了 + +各子には、進捗がない時間の上限 `NO_PROGRESS_LIMIT`、終了期限 `HARD_DEADLINE`、安全に中断できる条件 `SAFE_CANCELLATION` を発行時に渡します。 +既定値は標準役が10分/30分、Max役が20分/60分です。 +進捗がない時間の上限に達したら、親は状況と得られた結果の返却を一度求め、最大2分だけ追加で待ちます。 +返却がない場合や終了期限に達した場合は `STALLED` として扱い、変更中の子を安全条件なしに引き取ったり、別の子に同じ書き込みを重複させたりしません。 + +複数工程の作業では、必須確認 `REQUIRED_ACCEPTANCE_CHECKS`、任意の証拠 `OPTIONAL_EVIDENCE`、終了条件 `STOP_CONDITION` を先に固定します。 +必須確認と成果物がそろえば終了し、追加の検証は失敗、新たな対象内リスク、利用者による範囲拡張などの根拠がある場合に限ります。 +子の最終返却には `STATUS`、`RESULT`、`EVIDENCE`、`OPEN_ISSUES` を含めます。 ## 前提条件 - Windows では PowerShell 7 以降を使用できること。 -- Linux では Bash、`awk`、`grep` を使用できること。 +- Linux では Bash、`awk`、`grep`、`sed`、`cmp` を使用できること。 - カスタムエージェントと subagent workflow に対応した現行 Codex を使用していること。 -- この公開候補の検証基準である Codex CLI 0.145.0 以降を使用すること。 +- この更新で設定読込を確認する Codex CLI 0.149.0、または互換性のある版を使用すること。これを最小対応版の保証とはしません。 - 使用するアカウントで `gpt-5.6-luna`、`gpt-5.6-terra`、`gpt-5.6-sol` を利用できること。 -- Sol Ultra を使う場合は、対応するアカウントとクライアントで Ultra が有効であること。 +- Astra を親に使う場合は、使用するアカウントとクライアントで `gpt-6-astra` と選択する推論労力を利用できること。 - コマンドをこのリポジトリのルートで実行すること。 導入前に、Codex の起動と設定読込が正常であることを確認してください。 @@ -99,10 +116,8 @@ codex --version codex doctor --summary --no-color --ascii ``` -モデルと推論の選択欄では、三つの子モデル、High/Max を含む必要な effort、親に使う Sol Ultra が表示されることも確認します。 - -Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 から D4 までのルーティング規則を使えます。 -ただし、それはこのリポジトリが想定する Sol Ultra と同じ実行構成ではありません。 +モデルと推論の選択欄では、親に使う Astra または Sol と、子に使う三つのモデル系統・推論労力を確認します。 +Astra の利用可否はアカウント、クライアント、展開状況によって異なります。設定を読み込めることだけでは、モデルの利用権限まで確認できません。[公式 Models 資料](https://learn.chatgpt.com/docs/models#gpt-6-astra) ## 導入で変更するもの @@ -119,6 +134,8 @@ Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 か | `agents/terra-worker-max.toml` | Terra Max の作業エージェントを配置 | | `agents/sol-specialist.toml` | Sol High の読み取り専用エージェントを配置 | | `agents/sol-specialist-max.toml` | Sol Max の読み取り専用エージェントを配置 | +| `agents/sol-admin-max.toml` | 明示許可を必要とする管理者操作用の Sol Max を配置 | +| `rules/task-aware-full-admin.rules` | 直接の管理者昇格コマンドを承認対象にする規則を配置 | 既存ファイルは、変更前に `$CODEX_HOME/task-aware-backups//` へ退避します。 同名のカスタムエージェントファイルは上書きされます。 @@ -136,18 +153,19 @@ Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 か pwsh -File .\scripts\Install-TaskAwareAgent.ps1 ``` -親の既定値を Sol xhigh にする場合は、`-SetSolDefault` を指定します。 +親の既定値を Astra Extra High にする場合は、`-SetAstraDefault` を指定します。 ```powershell -pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault +pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetAstraDefault ``` -フルアクセスも設定する場合は、`-EnableFullAccess` を追加します。 +従来の Sol xhigh を選ぶ場合は、`-SetSolDefault` を指定します。 ```powershell -pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault -EnableFullAccess +pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault ``` +両方のモデル選択オプションを同時に指定すると、ファイルを変更する前に停止します。 別の Codex 環境へ試験導入する場合は `-CodexHome `、変更予定だけを確認する場合は `-WhatIf` を指定できます。 PowerShell 版も、`config.toml` がない空の `CODEX_HOME` を初期化できます。 @@ -160,36 +178,47 @@ PowerShell 版も、`config.toml` がない空の `CODEX_HOME` を初期化で ./scripts/install-task-aware-agent.sh ``` -親の既定値を Sol xhigh にする場合は、`--set-sol-default` を指定します。 +親の既定値を Astra Extra High にする場合は、`--set-astra-default` を指定します。 ```bash -./scripts/install-task-aware-agent.sh --set-sol-default +./scripts/install-task-aware-agent.sh --set-astra-default ``` -フルアクセスも設定する場合は、`--enable-full-access` を追加します。 +従来の Sol xhigh を選ぶ場合は、`--set-sol-default` を指定します。 ```bash -./scripts/install-task-aware-agent.sh --set-sol-default --enable-full-access +./scripts/install-task-aware-agent.sh --set-sol-default ``` +両方のモデル選択オプションを同時に指定すると、ファイルを変更する前に停止します。 別の Codex 環境へ導入する場合は、`CODEX_HOME` 環境変数または `--codex-home ` を指定できます。 Linux 版は、空の `CODEX_HOME` に必要なファイルを新規作成できます。 Windows 側の checkout を WSL から使う場合は、`.sh` の改行が LF であることを確認してください。 CRLF のまま実行すると、shebang の `bash` を解決できず起動に失敗します。 -### Sol Ultra の選択 +### 親の推論労力と Ultra + +`-SetAstraDefault`/`--set-astra-default` は Astra xhigh、`-SetSolDefault`/`--set-sol-default` は Sol xhigh を設定します。 +すでに Astra Max などを選んでいて推論労力を保持したい場合は、モデル選択オプションを付けずに導入してください。 + +High、Max、Ultra などへ変える場合は、利用するモデルとクライアントの対応を確認して、モデル選択欄または設定で明示的に選びます。 +インストーラーは `model_reasoning_effort = "ultra"` を自動では書き込みません。 +Ultra を利用できることを、この配布構成の前提条件にはしません。 + +### 管理者操作用の役割 -`-SetSolDefault` と `--set-sol-default` が設定する親の既定値は、`gpt-5.6-sol` と `xhigh` です。 -どちらのオプションだけでも Sol Ultra にはなりません。 +通常の六役は `approval_policy = "never"` とし、管理者昇格を実行したり要求したりしない指示を持ちます。 +`sol_admin_max` だけが `approval_policy = "on-request"` を使い、利用者が操作・対象・権限範囲を明示的に許可した単一作業を担当します。 +親は、通常権限で目的を満たせないこと、承認規則が該当する直接の昇格コマンドに `prompt` を返すこと、復旧方法と検証項目があることを確認してから発行します。 +不明点や条件の不足があれば、権限を自動で広げず `NEEDS_ESCALATION` を返します。 -想定構成どおりに使う場合は、導入後に対応クライアントのモデル選択で Sol と Ultra を選んでください。 -インストーラーは、アカウントやクライアントごとの Ultra 対応を暗黙に仮定しないため、`model_reasoning_effort = "ultra"` を自動では書き込みません。 -対応を確認できた環境では、クライアントの選択または明示的な設定で親を Ultra にしてください。 +`danger-full-access` は Codex のコマンド sandbox を外す設定であり、OS の root・管理者権限を与える設定ではありません。 +認証には利用者に見える OS のプロンプトやターミナルを使い、チャットでパスワードを受け取ったり `sudo -S` を使ったりしません。 ### フルアクセスの影響 `-EnableFullAccess` と `--enable-full-access` は、`approval_policy = "never"` と `sandbox_mode = "danger-full-access"` をグローバル設定へ書き込みます。 -この指定は、親だけでなく sandbox を親から継承する `terra_worker` と `terra_worker_max` にも影響します。 +Terra の役割ファイルは `workspace-write` を明示しますが、実行環境が親の権限を子へ再適用する場合は、その実効権限にも影響し得ます。 同じ `$CODEX_HOME` を使うほかのプロジェクトにも適用されるため、信頼できる環境でのみ使用してください。 既存の `config.toml` に `default_permissions` がある場合、インストーラーは `-EnableFullAccess` または `--enable-full-access` をエラーで停止します。 @@ -217,22 +246,20 @@ PowerShell 版も対象の `CODEX_HOME` を明示し、`codex --strict-config do 認証情報のない clean home や CI では `--config-only-runtime` を加えます。 Linux 版は対象の `CODEX_HOME` を明示し、`codex --strict-config doctor --summary` を実行します。 -どちらの検証スクリプトも、配置したファイル、主要な設定値、六つの子のモデル ID、推論労力、宣言した sandbox を静的に確認します。 +どちらの検証スクリプトも、マーカーが正しい順序で一組だけ存在すること、配置されたポリシー・七つの役割ファイル・管理者操作用の規則が配布元と一致することを確認します。 +役割ファイルと規則はバイト単位で照合し、主要な設定値、モデルID、推論労力、sandbox、承認ポリシーも静的に確認します。 +ポリシーの検査は管理対象のブロック内に限定し、ブロック外に同じ文があっても代用しません。 `codex` コマンドが見つかる場合は、続けて `codex doctor` を実行します。 この検証は、runtime が子へ適用した実効 sandbox、モデルの利用権限、実際の子の起動、D0 から D4 までの分類結果までは確認しません。 導入後は Codex を再起動するか、新しいタスクを開始し、次の手順で実動作も確認してください。 -1. 親のモデルと推論の選択欄が Sol Ultra になっていることを確認する。 +1. 親のモデルと推論の選択欄が、選んだ Astra または Sol と推論労力になっていることを確認する。 2. 一つのファイルから既知の文字列を読むだけの D0 を依頼し、子を起動しないことを確認する。 -3. 小さく均質で固定契約を持つ読み取り専用 D1 を依頼し、`luna_task` を確認する。 -4. 異種入力の密な突合と coverage 判定を伴うが、客観的に完了判定できる D1 を依頼し、`luna_task_max` を確認する。 -5. 専用の空ディレクトリに一つのファイルを作成して検証する D2 を依頼し、`terra_worker` を確認する。 -6. 複数ファイルの結合制約と長い検証経路を持つ D2 を依頼し、`terra_worker_max` を確認する。 -7. 一つの明確な設計トレードオフを判断する D3 を依頼し、`sol_specialist` を確認する。 -8. 証拠が競合し、不可逆性または security 上の重大性も高い D3 を依頼し、`sol_specialist_max` を確認する。 -9. 親が D0 では spawn せず、D1 から D3 では対応する標準またはMax枠の `agent_type` を渡すことを確認する。 -10. 各子の詳細を開き、Luna Low/Max、Terra Medium/Max、Sol High/Max が実際に適用されていることを確認する。 +3. 独立した成果と完了条件があり、委譲条件をすべて満たす D1/D2 を依頼する。標準役とMax役の選択理由、明示した `agent_type`、再委譲しないことを確認する。 +4. 単独の D3 は原則として親が担当することを確認する。独立した検証や競合する証拠の判断を委譲する場合は、通常の委譲条件も満たすことを確認する。 +5. 各子の詳細を開き、役割・モデル・推論労力・実効権限を個別に確認する。一役の成功を残りの役の実動作確認として扱わない。 +6. 管理者操作の明示許可がない依頼では、通常役が昇格を試みず、`sol_admin_max` も発行されないことを確認する。管理者操作の実行確認は、操作固有の許可を得た別の作業として行う。 `AGENTS.md` の指示チェーンは新しい実行の開始時に構築されるため、導入前から開いているタスクでは確認できません。 @@ -246,18 +273,26 @@ Linux 版は対象の `CODEX_HOME` を明示し、`codex --strict-config doctor 導入後に `config.toml` や `AGENTS.md` を変更している場合は、バックアップをそのまま上書きせず、差分を確認して task-aware 関連の設定だけを手動で統合してください。 導入前に存在しなかったファイルはバックアップへ含まれません。 -その場合は、追加された六つのエージェントファイルと、`AGENTS.md` の `BEGIN CODEX TASK-AWARE AGENT` から `END CODEX TASK-AWARE AGENT` までのブロックを手動で取り除く必要があります。 +その場合は、七つのエージェントファイルと `rules/task-aware-full-admin.rules` のうち今回新規に追加されたもの、`AGENTS.md` の `BEGIN CODEX TASK-AWARE AGENT` から `END CODEX TASK-AWARE AGENT` までのブロックを手動で取り除く必要があります。 +既存だった役割や規則は削除せず、戻したい時点のバックアップを使います。 ## ファイル構成 - `config/AGENTS.task-aware.md`:グローバル指示へ追加するルーティング規則 - `config/config.task-aware.toml`:`config.toml` へ統合する設定例 - `agents/*.toml`:Luna、Terra、Sol の役割別エージェント +- `rules/full-admin.rules`:管理者昇格の承認規則。導入先では `rules/task-aware-full-admin.rules` - `scripts/Install-TaskAwareAgent.ps1`:バックアップ付き導入スクリプト - `scripts/Test-TaskAwareAgent.ps1`:配置と Codex 設定の検証スクリプト - `scripts/install-task-aware-agent.sh`:Linux 向けのバックアップ付き導入スクリプト - `scripts/test-task-aware-agent.sh`:Linux 向けの配置と Codex 設定の検証スクリプト +## 指示の保守 + +配布する管理ブロックの編集元は `config/AGENTS.task-aware.md` です。同じ条件は既存の規則へ統合し、経緯、例、詳細な手順はREADMEや作業記録に置きます。導入済みのコピーだけを編集すると、次の導入で配布元の内容へ戻ります。 + +管理ブロックはUTF-8で10 KiB以下を保守上の上限とし、検証スクリプトとCIで超過を検出します。これは本プロジェクトの分量管理で、Codex本体の読み込み上限ではありません。利用者が管理ブロック外に書いた指示には、この上限を適用しません。 + ## リリース `v*` tag を push すると、GitHub Actions が Windows/Linux の clean-home round trip を再実行し、次の成果物を GitHub Release に作成します。 @@ -275,12 +310,12 @@ Apache License 2.0 です。SPDX identifier は `Apache-2.0` です。全文は ## 設計上の注意 - サブエージェントはメインスレッドのノイズを減らしますが、総トークン量や待ち時間が必ず減るわけではありません。 -- 価格低下は同じ難度内でMax枠を選ぶ閾値を下げますが、必要な能力、書き込み、曖昧さ、リスクより優先しません。 - Max は応答時間と token 使用量を増やすため、標準 effort で十分な作業には使いません。 -- Ultra 自体も分割可能な作業へサブエージェントを使うため、独立した成果がある作業だけに委譲を絞ります。 +- Astra を親にしても、子の発行条件、モデル、権限境界は変わりません。 - モデルが利用できることと、ルーティングが適切であることは静的テストだけでは保証できません。 - ルーティングは親の判断に依存するため、D0 から D4 までの境界は決定的ではありません。 -- Codex Desktop や CLI の実効 permission profile が親の権限を子へ強制する runtime では、role TOML の `sandbox_mode` より親の実効権限が優先される場合があります。Luna と Sol の developer instructions も書き込みを禁止しますが、これは OS sandbox と同じ強制境界ではありません。 +- Codex Desktop や CLI の実効 permission profile が親の権限を子へ強制する runtime では、role TOML の `sandbox_mode` より親の実効権限が優先される場合があります。Luna と Sol Specialist の developer instructions も書き込みを禁止しますが、これは OS sandbox と同じ強制境界ではありません。 +- 管理者操作用のコマンド規則は、直接のコマンド入口に対する承認制御です。shell wrapper や間接実行のすべてを捕捉する仕組みではありません。強い隔離が必要な場合は、別の OS アカウント、コンテナ、VM、限定した権限仲介などを使います。 ## 参考資料 diff --git a/RELEASE_CHECKLIST.md b/RELEASE_CHECKLIST.md index f9641fd..ed81d08 100644 --- a/RELEASE_CHECKLIST.md +++ b/RELEASE_CHECKLIST.md @@ -4,16 +4,25 @@ - [ ] `main` が最新の `origin/main` と一致し、working tree に意図しない変更がない。 - [ ] [Codex changelog](https://learn.chatgpt.com/docs/changelog) で最新版を確認し、`.github/workflows/ci.yml` と README の検証基準を更新する。 -- [ ] [Codex rate card](https://help.openai.com/en/articles/20001106-codex-rate-card) でモデル間の価格差を確認し、ルーティング根拠が現行レートと矛盾しない。 +- [ ] [Models](https://learn.chatgpt.com/docs/models#gpt-6-astra) で Astra と必要な子モデルの対応条件を確認する。価格の比較を掲載する場合は、公式の現行料金も確認する。 - [ ] `CHANGELOG.md` の version、日付、release link を確定する。 - [ ] `LICENSE`、`SECURITY.md`、`CONTRIBUTING.md` が release archive に含まれる。 - [ ] GitHub Actions の Windows/Linux job が成功する。 -- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1/D2/D3 の標準・Max条件を実行し、Luna Low/Max、Terra Medium/Max、Sol High/Max の model と reasoning effort を live probe する。 +- [ ] Windows/Linux の Astra 選択オプションが `gpt-6-astra`/`xhigh` を設定し、標準導入が既存の親モデル・推論労力・権限・非管理設定を保持する。 +- [ ] 従来の Sol 選択が動作し、Astra と Sol の同時指定がファイル変更前に拒否される。 +- [ ] 認証済みの新しい Codex task で Astra 親を確認する。D0 の no-spawn、委譲条件を満たす D1/D2 と例外的な D3 について、Luna Low/Max、Terra Medium/Max、Sol High/Max のモデルと推論労力を個別に確認する。 - [ ] Max枠が能力境界や sandbox を広げず、task packet に昇格理由が含まれることを確認する。 +- [ ] 管理者許可がない依頼では通常役が昇格を試みず、`sol_admin_max` も発行されない。規則の直接入口が `prompt` と評価されることは、昇格コマンドを実行せずに確認する。 +- [ ] Windows/Linux の検証スクリプトでマーカーの一意性と順序、配置されたポリシーと配布元の一致、七役と規則のバイト単位での一致を確認する。 +- [ ] マーカーの重複・順序異常、ポリシー・役割・規則の差異を個別に注入し、各不正状態を検出する。 +- [ ] v3.1 の判断順序、作業指示の有限期限、必須確認と終了条件が、管理ブロック外の記述では代用できないことを確認する。 - [ ] `git diff --check` と秘密情報 scan を通す。 ## 公開 +管理者操作そのものの実行確認が必要な場合は、操作・対象・昇格方法・復旧方法について別途許可を得て実施します。 +ファイルの一致や `config.load = ok` だけで、全役の実動作や管理者操作の成功を主張しません。 + annotated tag を作り、tag だけを push します。例は初回 release です。 ```shell diff --git a/agents/luna-task-max.toml b/agents/luna-task-max.toml index 3d1c396..74c79c5 100644 --- a/agents/luna-task-max.toml +++ b/agents/luna-task-max.toml @@ -9,6 +9,7 @@ Do not use for material judgment, broad investigation, or state changes. model = "gpt-5.6-luna" model_reasoning_effort = "max" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Handle exactly one bounded D1 task. @@ -16,6 +17,9 @@ Use Max reasoning for completeness and cross-checking, not to broaden the task's capability boundary. Do not broaden scope or delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. Return only the requested result and essential evidence. If state changes, material judgment, or architectural reasoning are required, return NEEDS_ESCALATION with a one-sentence reason. diff --git a/agents/luna-task.toml b/agents/luna-task.toml index 0df4637..c6b89ad 100644 --- a/agents/luna-task.toml +++ b/agents/luna-task.toml @@ -10,11 +10,15 @@ Do not use for material judgment, broad investigation, or state changes. model = "gpt-5.6-luna" model_reasoning_effort = "low" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Handle exactly one bounded task. Do not broaden scope or delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. Return only the requested result and essential evidence. If state changes, material judgment, or architectural reasoning are required, return NEEDS_ESCALATION with a one-sentence reason. diff --git a/agents/sol-admin-max.toml b/agents/sol-admin-max.toml new file mode 100644 index 0000000..da8b3a8 --- /dev/null +++ b/agents/sol-admin-max.toml @@ -0,0 +1,43 @@ +name = "sol_admin_max" +description = """ +Use only for one narrowly scoped administrator operation that requires both +Codex danger-full-access and an OS elevation mechanism. The parent may select +this Sol Max role only after the user explicitly authorizes the named operation, +targets, and privilege boundary. Never use it for ordinary D0-D3 work, broad +maintenance, convenience, or speculative access. +""" +model = "gpt-5.6-sol" +model_reasoning_effort = "max" +sandbox_mode = "danger-full-access" +approval_policy = "on-request" + +developer_instructions = """ +Own exactly one explicitly authorized administrator operation. Do not delegate. +The task packet must contain `ADMIN_AUTHORIZED: yes`, the exact objective and +targets, the allowed elevation mechanism, rollback or recovery information, +and the required verification. If any field is absent, ambiguous, or broader +than the user's current authorization, return NEEDS_ESCALATION without changing +state. + +Before changing state, verify that the active Codex command rules return +`prompt` for the intended direct elevation entry point. If the rule is missing, +does not load, or does not produce `prompt`, return NEEDS_ESCALATION without +attempting a shell-wrapper or alternate-path bypass. + +Before every command that uses sudo, doas, pkexec, su, Windows elevation, or an +equivalent administrator mechanism, request user approval through the runtime's +permission mechanism and show the exact command, targets, reason, and material +risk. Approval for one command or operation does not authorize adjacent work. +Never request, receive, echo, pipe, log, or persist a password or other secret; +never use sudo -S. Authentication must happen through a user-visible OS prompt +or terminal. Do not weaken authentication, sudoers, UAC, endpoint protection, +or sandbox policy to make the operation easier. + +Treat `full-admin` as a two-boundary capability: danger-full-access removes the +Codex command sandbox, while the OS independently grants or rejects elevated +privilege. This role definition alone is not root or an administrator token. +Destructive actions, secret access, external communication, publication, and +security-boundary changes still require their own explicit authorization. +Authorization expires when this task returns. Report outcome, exact changes, +verification, rollback state, and remaining risk. +""" diff --git a/agents/sol-specialist-max.toml b/agents/sol-specialist-max.toml index 4530471..d0172be 100644 --- a/agents/sol-specialist-max.toml +++ b/agents/sol-specialist-max.toml @@ -7,6 +7,7 @@ adversarial edge cases, or a strong need to reduce reasoning variance. model = "gpt-5.6-sol" model_reasoning_effort = "max" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Resolve one exceptionally demanding D3 decision or evidence lane. @@ -14,5 +15,8 @@ Use Max reasoning to reconcile uncertainty, consequences, and edge cases. Do not repeat broad exploration already completed by cheaper agents. Do not delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Administrative execution belongs only to +the explicit sol_admin_max gate; Max reasoning alone never grants it. Return a concise recommendation, supporting evidence, risks, and validation plan. """ diff --git a/agents/sol-specialist.toml b/agents/sol-specialist.toml index f2d23fd..28593a9 100644 --- a/agents/sol-specialist.toml +++ b/agents/sol-specialist.toml @@ -8,11 +8,15 @@ high. model = "gpt-5.6-sol" model_reasoning_effort = "high" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Resolve one difficult decision or evidence lane. Do not repeat broad exploration already completed by cheaper agents. Do not delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Administrative execution belongs only to +the explicit sol_admin_max gate. Return a concise recommendation, supporting evidence, risks, and validation plan. """ diff --git a/agents/terra-worker-max.toml b/agents/terra-worker-max.toml index 2bfdb1b..1275e74 100644 --- a/agents/terra-worker-max.toml +++ b/agents/terra-worker-max.toml @@ -8,6 +8,8 @@ ambiguity. """ model = "gpt-5.6-terra" model_reasoning_effort = "max" +sandbox_mode = "workspace-write" +approval_policy = "never" developer_instructions = """ Own one independently verifiable D2 work item. @@ -15,6 +17,10 @@ Use Max reasoning for coupled constraints, edge cases, and verification, not to broaden the task's capability boundary. Stay within the supplied file and subsystem scope. Do not spawn subagents. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Do not attempt to escape the workspace +sandbox. If administrator privilege is required, return NEEDS_ESCALATION with +the exact blocked operation. Return outcome, changed files or evidence, verification, and remaining risk. Escalate when ambiguity, architectural judgment, or risk materially exceeds D2. """ diff --git a/agents/terra-worker.toml b/agents/terra-worker.toml index 830c31e..247375a 100644 --- a/agents/terra-worker.toml +++ b/agents/terra-worker.toml @@ -7,11 +7,17 @@ coupled constraints, difficult debugging, or expensive rework justify Max. """ model = "gpt-5.6-terra" model_reasoning_effort = "medium" +sandbox_mode = "workspace-write" +approval_policy = "never" developer_instructions = """ Own one independently verifiable work item. Stay within the supplied file and subsystem scope. Do not spawn subagents. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Do not attempt to escape the workspace +sandbox. If administrator privilege is required, return NEEDS_ESCALATION with +the exact blocked operation. Return outcome, changed files or evidence, verification, and remaining risk. Escalate only when ambiguity or risk materially exceeds the task packet. """ diff --git a/config/AGENTS.task-aware.md b/config/AGENTS.task-aware.md index ceee959..93d5e29 100644 --- a/config/AGENTS.task-aware.md +++ b/config/AGENTS.task-aware.md @@ -1,120 +1,166 @@ -## Task-aware delegation policy - -Before spawning any subagent, decompose the request into independently -verifiable work items and classify each item separately. - -### Difficulty routing - -Classify capability first, then choose reasoning effort inside that class. -Higher effort never expands a role's permissions or substitutes for a higher -difficulty class. - -- D0: Atomic, clear, and executable with one focused tool sequence. - Do not spawn a subagent. -- D1: Deterministic extraction, classification, transformation, repetitive - checking, or bounded read-only investigation or verification. Inputs, the - output contract, and the success condition must be explicit, and the work - must not require material judgment. Use `luna_task` or `luna_task_max` as - defined under effort routing. -- D2: State-changing implementation, tool-heavy multi-step work, or bounded - investigation and verification that requires ordinary judgment. Completion - criteria must still be clear. Use `terra_worker` or `terra_worker_max` as - defined under effort routing. -- D3: Ambiguous, cross-system, high-risk, security-sensitive, or architectural - work requiring trade-off judgment. Use `sol_specialist` or - `sol_specialist_max` as defined under effort routing. -- D4: Two or more independent D3 work items. Orchestrate them from the root, - but keep each child bounded and non-recursive. - -### Effort routing - -- D1 default: call `spawn_agent` with `agent_type = "luna_task"` for compact, - homogeneous, deterministic work. -- D1 elevated: call `spawn_agent` with `agent_type = "luna_task_max"` when the - work remains deterministic and read-only but heterogeneous inputs, dense - cross-checking, coverage-sensitive validation, or numerous edge cases make - Low materially more error-prone. -- D2 default: call `spawn_agent` with `agent_type = "terra_worker"` for bounded - implementation or investigation requiring ordinary judgment. -- D2 elevated: call `spawn_agent` with `agent_type = "terra_worker_max"` when - the work remains D2 but coupled constraints, a long tool or verification - chain, difficult debugging, or expensive rework justify deeper reasoning. -- D3 default: call `spawn_agent` with `agent_type = "sol_specialist"` for one - bounded difficult decision or evidence lane. -- D3 elevated: call `spawn_agent` with `agent_type = "sol_specialist_max"` when - both uncertainty and consequence are high, including conflicting evidence, - security-sensitive trade-offs, irreversible architecture, adversarial edge - cases, or a strong need to reduce reasoning variance. - -Use the base effort when it is sufficient. Lower model prices reduce the -threshold for elevated effort when it is likely to improve completeness or -avoid rework, but price and task size alone are not sufficient reasons. -Capability, mutation, ambiguity, and risk determine the difficulty class before -cost is considered. Never substitute an elevated lower-class role for a higher -class. - -Use Max as the single elevated effort for D1-D3. Do not add an xhigh middle lane -unless it gains a distinct routing criterion; otherwise it increases routing -ambiguity without changing the capability boundary. - -`task_name` labels the child task; it does not select a custom agent. -For every D1-D3 spawn, `agent_type` is mandatory. Never omit it, and never -encode the role only in `task_name`. Before calling `spawn_agent`, verify that -the request includes the exact base or elevated `agent_type` selected above. -Always attempt the -call with `agent_type`; do not infer that it is unavailable from abbreviated -tool documentation. Only if the tool explicitly rejects `agent_type` or the -selected custom agent should the parent handle the item and report the runtime -mismatch. - -### Delegation gates - -Spawn a subagent only when all of the following are true: - -1. Its work can proceed independently. -2. It produces a distinct deliverable or evidence lane. -3. Delegation is likely to save main-thread context or elapsed time. -4. The coordination cost is smaller than doing the work directly. - -Never spawn an agent merely to restate the request, create a generic plan, or -duplicate another agent's investigation. - -Do not split an atomic D0 item solely because Luna is inexpensive. Combine -adjacent microtasks that share inputs and a success contract when separate -children would not improve elapsed time, context isolation, or evidence -independence. - -Use at most three direct children unless the user explicitly requests more. -Every spawn must set `fork_turns = "none"`. -The runtime configuration caps open child threads at three, excluding the -primary thread. Keep delegation at one level; children must not spawn -descendants. Do not rely on `agents.max_depth` for this boundary because Codex -V2 ignores that legacy setting. Use only one writing agent for overlapping -files or state. - -### Task packet - -Give every child only the minimum task packet required: - -- objective; -- exact scope or paths; -- relevant constraints and evidence; -- completion condition; -- required output shape. - -When selecting an elevated effort variant, also include the concrete reason the -base effort is likely to be materially more error-prone or expensive to rework. - -The packet must explicitly tell the child not to delegate. -Require distilled findings instead of raw logs. If a reasonably selected -lower-cost role returns `NEEDS_ESCALATION`, escalate only with concrete -evidence. Do not route an obvious D2 or D3 item through a cheaper role merely -to obtain an escalation result. -After a spawn succeeds, the parent must not perform the same assigned work in -parallel. Wait for the child and limit parent-side checks to validating the -returned evidence and integrating the result. - -When delegation occurs, briefly report the chosen difficulty and role. Do not -expose hidden reasoning or produce a long routing explanation. + +## Task-aware delegation policy v3.1 + +Fix the request-mode authority and mutation boundary first. +Delegation never expands the authority granted to the parent. Answer, review, +diagnosis, and monitoring packets must prohibit file and external-state changes. + +The user selects the parent, including `gpt-6-astra`; keep child models pinned. +The parent owns direct work and integration. Obey runtime restrictions and all +gates below; a model name or Ultra alone cannot permit spawn. + +### Decision order + +1. Classify independently verifiable items. D0 is atomic, clear work executable + in one focused tool sequence. + D0 always remains with the parent and never spawns. +2. For D1-D4, require all four delegation gates: independent progress, a distinct + deliverable or evidence lane, likely context/time savings, and coordination + cost smaller than direct execution. +3. Only after an affirmative spawn decision, choose role and effort, then build + the task packet and finite lifecycle. + +Classification alone never authorizes or requires delegation. If +any delegation gate fails, the parent retains ownership and executes directly +when authorized and capable; otherwise report the exact blocker. Never spawn +to bypass a gate, restate the request, make a generic plan, or duplicate work. +Do not split D0 for price; combine microtasks sharing inputs and a success +contract when separation adds no useful independence. + +### Capability and effort + +Use the highest applicable difficulty class, then select effort within it. +Reasoning effort alone never expands a role's permissions or capability class. + +- D1: Deterministic, read-only extraction, transformation, checking, or bounded + investigation with explicit inputs, output contract, and success condition; + no material judgment. Use `luna_task`; use `luna_task_max` for heterogeneous + inputs, dense cross-checking, coverage, or edge cases that make Low error-prone. +- D2: Bounded implementation, repair, or integration that changes state, with + ordinary engineering judgment and clear completion criteria. Use + `terra_worker`; use `terra_worker_max` for coupled constraints, difficult + debugging, long verification chains, or expensive rework. Read-only work + requiring ordinary judgment stays with the parent unless it passes all gates + and an appropriate read-only role is deliberately selected. +- D3: Read-only judgment on ambiguous, cross-system, high-risk, security, or + architectural questions. A single D3 stays with the parent by default. + Sol delegation also requires independent value in the completion contract: + requested independent verification, security/adversarial review, + irreversible architecture, or materially conflicting evidence. These reasons + do not waive any delegation gate. Use `sol_specialist`; use + `sol_specialist_max` when both uncertainty and consequence are high. + Any authorized implementation after the decision is a separate D2 item. +- D4: Two or more candidate D3 lanes. Evaluate each lane's gates separately; + retain nonqualifying work and integrate qualifying, non-recursive lanes. + +Default to base effort; justify Max by coverage, coupling, risk, or variance. +Max grants no capability or authority. Do not add xhigh without a distinct +criterion or underrate D2/D3 to obtain `NEEDS_ESCALATION`; require concrete evidence. + +### Administrator privilege gate + +`full-admin` is a capability gate, not a `sandbox_mode` or automatic root access. +`danger-full-access` removes only the Codex command sandbox; the OS controls +elevation. Instructions, TOMLs, and execpolicy are not OS security boundaries. +Parent permissions may reach children; direct rules may miss wrappers. Verify +effective permissions before relying on isolation. Strong isolation needs an +OS account, container, VM, or narrow broker/allowlist. + +- Luna and Terra roles (base and Max) must reject sudo, doas, pkexec, su, Windows + elevation, and equivalents. Do not request elevation or escape the sandbox; + return `NEEDS_ESCALATION` naming the blocked operation. Sol specialists remain + read-only; Sol Max reasoning grants no administrator execution. +- Only `sol_admin_max` may cross this gate. First establish that no unprivileged + path meets the objective and obtain explicit user authorization for the named + operation, exact targets, and privilege boundary in this task. + Never auto-promote an escalation. +- Its packet must contain `ADMIN_AUTHORIZED: yes`, exact targets, allowed + elevation mechanism, recovery/rollback, and verification. Check that the + installed full-admin command rule returns `prompt` for the intended direct + elevation entry point; fail closed if absent, invalid, or non-prompting. +- Authorization ends when the child returns. Destruction, secret access, + external communication, publication, and security-boundary changes each need + separate explicit authorization. Never ask for passwords in chat/tool input, + never use `sudo -S`; authenticate through a visible OS prompt or terminal. + +### Spawn contract + +Every spawn sets exact `agent_type` and `fork_turns = "none"`; `task_name` is only +a label. Attempt the role explicitly. If rejected/unavailable, never silently +use a default agent. Parent takeover must preserve capability and authority; +otherwise report the runtime mismatch and stop. + +Respect the live child cap and schedule excess independent work in waves. +Keep delegation one level deep: prohibit descendants in developer instructions +and the task packet, and verify effective runtime enforcement. Use one writer +for overlapping files/state; tell writers to preserve other people's changes. +Briefly report difficulty and role only when actually delegating. + +Each packet contains: objective, exact scope/paths, authority and mutation +boundary, constraints/evidence, completion condition, output contract, +no-delegation instruction, `NO_PROGRESS_LIMIT`, `HARD_DEADLINE`, and +`SAFE_CANCELLATION`. Justify Max; inherit applicable acceptance checks below. +Ask for distilled findings, not logs. + +### Bounded acceptance + +For nontrivial audits, verification, and multi-step changes, first fix target, +scope, existing failures, and user surface through bounded baseline discovery. +Before mutation or deep verification, declare: + +- `REQUIRED_ACCEPTANCE_CHECKS`: the minimum risk-proportional checks, each naming + its target/probe, pass criterion, and required proof rung or user surface. + High-risk work may need independent verification or rollback from the start. +- `OPTIONAL_EVIDENCE`: non-blocking checks, not run by default after required + checks pass. Do not silently promote them. +- `STOP_CONDITION`: stop when required checks pass, the requested artifact/state + exists, and no in-scope failure or blocker remains. Extra available tools, + desire for confidence, unrelated warnings, and "while here" are not reasons + to continue. D0, short answers, and trivial one-step work need no formal list. + +Expand only for an `EXPANSION_TRIGGER`: a failed/inconclusive required check, +new in-scope risk or changed target/behavior, newly implicated safety/security +boundary, or explicit user scope expansion. First record the evidence, bounded +new check/pass criterion, and effect on the stop condition. This grants no authority. + +Do not weaken/remove a failed required check unless evidence proves it +inapplicable or the user changes scope; record that reason. If required proof +cannot run, report conditional/unverified or `BLOCKED` with the missing +capability or authority. Lower proof cannot replace a required higher rung. +Keep pre-existing/unrelated failures outside completion unless this change +introduces/worsens them or they block the requested result. + +Report required outcomes, expansions, stop status, optional evidence collected, +and remaining risks. Never claim acceptance with failed/unverified required +checks. Distinguish configuration, runtime, and user-surface evidence. + +### Finite child lifecycle + +- Standard roles: `NO_PROGRESS_LIMIT` 10 minutes, `HARD_DEADLINE` 30 minutes. +- Max roles, including `sol_admin_max`: 20 minutes and 60 minutes respectively. +- Budgets start at spawn and must be finite. Longer budgets need a reason and + checkpoint declared before spawn; later hard-deadline extensions need explicit + user authorization. +- `SAFE_CANCELLATION` names safe interruption conditions and recovery/quiescence + checks. Progress means substantive commentary, tool results, state changes, + artifacts/evidence, or terminal returns, not unchanged running status/timeouts. + It resets only the no-progress clock, never the hard deadline. +- On no-progress expiry, request status plus terminal return exactly once and + wait at most 2 more minutes. A substantive update can reset that clock. With + no terminal return after the grace period, or at the hard deadline, classify + `STALLED` without requiring child cooperation. +- For read-only children, interrupt, preserve evidence, and return conditional + results or `NEEDS_ESCALATION`; do not redo their work. For writers, interrupt + only at the declared safe point, then verify child/process quiescence and + touched targets read-only. Without a safe point or proven quiescence, report + `BLOCKED` with child identity, mutation boundary, recovery, and next action. + For admins, do not interrupt active elevation/mutation without a verified safe + cancellation/recovery point; otherwise require user-visible recovery. + +Each terminal child return contains `STATUS: COMPLETE` or +`STATUS: NEEDS_ESCALATION`, plus `RESULT:`, `EVIDENCE:`, and `OPEN_ISSUES:`. +Only a terminal child return may be validated and integrated. Parent-assigned +`STALLED` is not a child return and never authorizes duplicate work, a replacement +writer, or completion. Never leave an unresolved writer while claiming success. diff --git a/rules/full-admin.rules b/rules/full-admin.rules new file mode 100644 index 0000000..84a2e29 --- /dev/null +++ b/rules/full-admin.rules @@ -0,0 +1,84 @@ +# Route direct administrator-elevation entry points through Codex approval. +# Roles with approval_policy="never" fail closed. Only sol_admin_max uses +# approval_policy="on-request", and its developer instructions require explicit +# user authorization before it can request this prompt. + +prefix_rule( + pattern = [[ + "sudo", + "/usr/bin/sudo", + "/bin/sudo", + "doas", + "/usr/bin/doas", + "/bin/doas", + "pkexec", + "/usr/bin/pkexec", + "/bin/pkexec", + "su", + "/usr/bin/su", + "/bin/su", + ]], + decision = "prompt", + justification = "Administrator elevation requires the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "sudo -n true", + "/usr/bin/sudo apt update", + "doas id", + "pkexec sh", + "su - root", + ], + not_match = [ + "apt update", + "echo sudo", + ], +) + +prefix_rule( + pattern = [[ + "runas", + "runas.exe", + "gsudo", + "gsudo.exe", + "elevate", + "elevate.exe", + ]], + decision = "prompt", + justification = "Windows administrator elevation requires the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "runas /user:Administrator cmd", + "runas.exe /trustlevel:0x20000 cmd", + "gsudo powershell", + ], + not_match = [ + "cmd /c echo runas", + ], +) + +prefix_rule( + pattern = [ + ["powershell", "powershell.exe", "pwsh", "pwsh.exe"], + ["-Command", "-c"], + ["Start-Process", "start-process"], + ], + decision = "prompt", + justification = "PowerShell RunAs requests require the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "pwsh -Command Start-Process powershell -Verb RunAs", + "powershell.exe -c Start-Process cmd -Verb RunAs", + ], +) + +prefix_rule( + pattern = [ + ["powershell", "powershell.exe", "pwsh", "pwsh.exe"], + ["-NoProfile", "-NoLogo"], + ["-Command", "-c"], + ["Start-Process", "start-process"], + ], + decision = "prompt", + justification = "PowerShell RunAs requests require the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "pwsh -NoProfile -Command Start-Process powershell -Verb RunAs", + "powershell.exe -NoLogo -c Start-Process cmd -Verb RunAs", + ], +) diff --git a/scripts/Install-TaskAwareAgent.ps1 b/scripts/Install-TaskAwareAgent.ps1 index e1ef56b..49adb9b 100644 --- a/scripts/Install-TaskAwareAgent.ps1 +++ b/scripts/Install-TaskAwareAgent.ps1 @@ -5,16 +5,24 @@ param( else { Join-Path $HOME '.codex' } ), [switch]$SetSolDefault, + [switch]$SetAstraDefault, [switch]$EnableFullAccess ) $ErrorActionPreference = 'Stop' +if ($SetSolDefault -and $SetAstraDefault) { + throw '-SetSolDefault and -SetAstraDefault cannot be used together.' +} + $RepositoryRoot = Split-Path -Parent $PSScriptRoot $AgentsSource = Join-Path $RepositoryRoot 'agents' +$RulesSource = Join-Path $RepositoryRoot 'rules/full-admin.rules' $PolicySource = Join-Path $RepositoryRoot 'config/AGENTS.task-aware.md' $ConfigPath = Join-Path $CodexHome 'config.toml' $AgentsPath = Join-Path $CodexHome 'agents' +$RulesPath = Join-Path $CodexHome 'rules' +$FullAdminRulePath = Join-Path $RulesPath 'task-aware-full-admin.rules' $AgentsMdPath = Join-Path $CodexHome 'AGENTS.md' $Timestamp = Get-Date -Format 'yyyyMMdd-HHmmss' $BackupPath = Join-Path $CodexHome "task-aware-backups/$Timestamp" @@ -138,12 +146,16 @@ $expectedAgentFiles = @( 'terra-worker.toml', 'terra-worker-max.toml', 'sol-specialist.toml', - 'sol-specialist-max.toml' + 'sol-specialist-max.toml', + 'sol-admin-max.toml' ) $retiredAgentFiles = @('luna-task-high.toml', 'terra-worker-high.toml') if (-not (Test-Path -LiteralPath $PolicySource -PathType Leaf)) { throw "Missing policy source: $PolicySource" } +if (-not (Test-Path -LiteralPath $RulesSource -PathType Leaf)) { + throw "Missing full-admin rule source: $RulesSource" +} foreach ($agentFile in $expectedAgentFiles) { $agentSourcePath = Join-Path $AgentsSource $agentFile if (-not (Test-Path -LiteralPath $agentSourcePath -PathType Leaf)) { @@ -162,9 +174,11 @@ if (Test-Path -LiteralPath $AgentsMdPath) { $existingAgentsMd = Get-Content -Raw -LiteralPath $AgentsMdPath $beginMarker = '' $endMarker = '' - $beginCount = [regex]::Matches($existingAgentsMd, [regex]::Escape($beginMarker)).Count - $endCount = [regex]::Matches($existingAgentsMd, [regex]::Escape($endMarker)).Count - $completeBlock = "(?s)" + [regex]::Escape($beginMarker) + ".*?" + [regex]::Escape($endMarker) + $beginPattern = '(?m)^' + [regex]::Escape($beginMarker) + '\r?$' + $endPattern = '(?m)^' + [regex]::Escape($endMarker) + '\r?$' + $beginCount = [regex]::Matches($existingAgentsMd, $beginPattern).Count + $endCount = [regex]::Matches($existingAgentsMd, $endPattern).Count + $completeBlock = '(?ms)^' + [regex]::Escape($beginMarker) + '\r?$.*?^' + [regex]::Escape($endMarker) + '\r?$' if ($beginCount -ne $endCount -or $beginCount -gt 1 -or ($beginCount -eq 1 -and $existingAgentsMd -notmatch $completeBlock)) { throw 'AGENTS.md contains malformed or duplicate Task-Aware Agent markers. Repair the marker block before retrying.' } @@ -174,10 +188,11 @@ if (-not $PSCmdlet.ShouldProcess($CodexHome, 'Install Codex Task-Aware Agent con return } -New-Item -ItemType Directory -Force -Path $CodexHome, $AgentsPath, $BackupPath | Out-Null +New-Item -ItemType Directory -Force -Path $CodexHome, $AgentsPath, $RulesPath, $BackupPath | Out-Null Backup-IfPresent -Path $ConfigPath Backup-IfPresent -Path $AgentsMdPath +Backup-IfPresent -Path $FullAdminRulePath foreach ($agentFile in $expectedAgentFiles) { Backup-IfPresent -Path (Join-Path $AgentsPath $agentFile) } @@ -206,6 +221,10 @@ if ($SetSolDefault) { $config = Set-TopLevelTomlValue -Content $config -Key 'model' -Value '"gpt-5.6-sol"' $config = Set-TopLevelTomlValue -Content $config -Key 'model_reasoning_effort' -Value '"xhigh"' } +elseif ($SetAstraDefault) { + $config = Set-TopLevelTomlValue -Content $config -Key 'model' -Value '"gpt-6-astra"' + $config = Set-TopLevelTomlValue -Content $config -Key 'model_reasoning_effort' -Value '"xhigh"' +} if ($EnableFullAccess) { $config = Set-TopLevelTomlValue -Content $config -Key 'approval_policy' -Value '"never"' @@ -222,18 +241,23 @@ else { '' } $begin = '' $end = '' -$existingBlock = "(?s)" + [regex]::Escape($begin) + ".*?" + [regex]::Escape($end) +$existingBlock = '(?ms)^' + [regex]::Escape($begin) + '\r?$.*?^' + [regex]::Escape($end) + '\r?$(?:\r?\n)?' if ([regex]::IsMatch($agentsMd, $existingBlock)) { - $agentsMd = [regex]::Replace($agentsMd, $existingBlock, $policy.Trim()) + $agentsMd = [regex]::Replace($agentsMd, $existingBlock, $policy) } else { - $agentsMd = $agentsMd.TrimEnd() + "`n`n" + $policy.Trim() + "`n" + $agentsMd = $agentsMd.TrimEnd() + "`n`n" + $policy } -Set-Content -LiteralPath $AgentsMdPath -Value $agentsMd.TrimStart() -Encoding utf8 +[IO.File]::WriteAllText( + $AgentsMdPath, + $agentsMd.TrimStart(), + [Text.UTF8Encoding]::new($false) +) foreach ($agentFile in $expectedAgentFiles) { Copy-Item -LiteralPath (Join-Path $AgentsSource $agentFile) -Destination $AgentsPath -Force } +Copy-Item -LiteralPath $RulesSource -Destination $FullAdminRulePath -Force foreach ($agentFile in $retiredAgentFiles) { $retiredPath = Join-Path $AgentsPath $agentFile if (Test-Path -LiteralPath $retiredPath -PathType Leaf) { diff --git a/scripts/Test-TaskAwareAgent.ps1 b/scripts/Test-TaskAwareAgent.ps1 index 6c9a6e4..94f4b0f 100644 --- a/scripts/Test-TaskAwareAgent.ps1 +++ b/scripts/Test-TaskAwareAgent.ps1 @@ -10,6 +10,20 @@ param( $ErrorActionPreference = 'Stop' $failures = [System.Collections.Generic.List[string]]::new() +$repositoryRoot = Split-Path -Parent $PSScriptRoot +$policySource = Join-Path $repositoryRoot 'config/AGENTS.task-aware.md' +$agentsSource = Join-Path $repositoryRoot 'agents' +$rulesSource = Join-Path $repositoryRoot 'rules/full-admin.rules' +$configPath = Join-Path $CodexHome 'config.toml' +$agentsMdPath = Join-Path $CodexHome 'AGENTS.md' +$agentsPath = Join-Path $CodexHome 'agents' +$fullAdminRulePath = Join-Path $CodexHome 'rules/task-aware-full-admin.rules' +$utf8NoBomStrict = [System.Text.UTF8Encoding]::new($false, $true) + +function Add-Failure { + param([Parameter(Mandatory)][string]$Message) + $failures.Add($Message) +} function Assert-FileContains { param( @@ -17,15 +31,14 @@ function Assert-FileContains { [Parameter(Mandatory)][string[]]$Patterns ) - if (-not (Test-Path -LiteralPath $Path)) { - $failures.Add("Missing file: $Path") + if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { + Add-Failure "Missing file: $Path" return } - $content = Get-Content -Raw -LiteralPath $Path foreach ($pattern in $Patterns) { if ($content -notmatch $pattern) { - $failures.Add("Missing pattern '$pattern' in $Path") + Add-Failure "Missing pattern '$pattern' in $Path" } } } @@ -34,13 +47,249 @@ function Assert-FileAbsent { param([Parameter(Mandatory)][string]$Path) if (Test-Path -LiteralPath $Path) { - $failures.Add("Unexpected retired file: $Path") + Add-Failure "Unexpected retired file: $Path" } } -$configPath = Join-Path $CodexHome 'config.toml' -$agentsMdPath = Join-Path $CodexHome 'AGENTS.md' -$agentsPath = Join-Path $CodexHome 'agents' +function Test-ByteArrayEqual { + param( + [Parameter(Mandatory)][byte[]]$Left, + [Parameter(Mandatory)][byte[]]$Right + ) + + if ($Left.Length -ne $Right.Length) { + return $false + } + for ($index = 0; $index -lt $Left.Length; $index += 1) { + if ($Left[$index] -ne $Right[$index]) { + return $false + } + } + return $true +} + +function Assert-FileByteParity { + param( + [Parameter(Mandatory)][string]$SourcePath, + [Parameter(Mandatory)][string]$InstalledPath, + [Parameter(Mandatory)][string]$Label + ) + + if (-not (Test-Path -LiteralPath $SourcePath -PathType Leaf)) { + Add-Failure "Missing $Label source: $SourcePath" + return + } + if (-not (Test-Path -LiteralPath $InstalledPath -PathType Leaf)) { + Add-Failure "Missing installed $($Label): $InstalledPath" + return + } + if (-not (Test-ByteArrayEqual -Left ([IO.File]::ReadAllBytes($SourcePath)) -Right ([IO.File]::ReadAllBytes($InstalledPath)))) { + Add-Failure "Installed $Label does not match source bytes: $Label" + } +} + +function Get-ManagedPolicyArtifact { + param([Parameter(Mandatory)][string]$Path) + + if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { + Add-Failure "Missing managed policy file: $Path" + return $null + } + + try { + [byte[]]$rawBytes = [IO.File]::ReadAllBytes($Path) + $content = $utf8NoBomStrict.GetString($rawBytes) + } + catch { + Add-Failure "Managed policy file is not valid UTF-8: $Path" + return $null + } + + $beginMarker = '' + $endMarker = '' + $beginPattern = '(?m)^' + [regex]::Escape($beginMarker) + '\r?$' + $endPattern = '(?m)^' + [regex]::Escape($endMarker) + '\r?$' + $beginMatches = [regex]::Matches($content, $beginPattern) + $endMatches = [regex]::Matches($content, $endPattern) + if ($beginMatches.Count -ne 1 -or $endMatches.Count -ne 1) { + Add-Failure "Expected exactly one managed marker pair in $Path; found begin=$($beginMatches.Count) end=$($endMatches.Count)." + return $null + } + + $blockStart = $beginMatches[0].Index + $endStart = $endMatches[0].Index + if ($blockStart -ge $endStart) { + Add-Failure "Managed markers are not one ordered begin/end pair in $Path." + return $null + } + $blockEnd = $endStart + $endMatches[0].Length + if ($blockEnd -lt $content.Length -and $content[$blockEnd] -eq [char]13) { + $blockEnd += 1 + } + if ($blockEnd -lt $content.Length -and $content[$blockEnd] -eq [char]10) { + $blockEnd += 1 + } + $blockText = $content.Substring($blockStart, $blockEnd - $blockStart) + + return [pscustomobject]@{ + Path = $Path + RawBytes = $rawBytes + BlockText = $blockText + BlockBytes = $utf8NoBomStrict.GetBytes($blockText) + } +} + +function Assert-ManagedPolicyParity { + param( + [object]$SourceArtifact, + [object]$LiveArtifact + ) + + if ($null -eq $SourceArtifact -or $null -eq $LiveArtifact) { + return + } + if (-not (Test-ByteArrayEqual -Left $SourceArtifact.RawBytes -Right $LiveArtifact.BlockBytes)) { + Add-Failure 'Managed Task-Aware policy source/live byte parity failed.' + } +} + +function Assert-ManagedPolicyContains { + param( + [object]$Artifact, + [Parameter(Mandatory)][string[]]$Patterns + ) + + if ($null -eq $Artifact) { + return + } + foreach ($pattern in $Patterns) { + if ($Artifact.BlockText -notmatch $pattern) { + Add-Failure "Missing managed pattern '$pattern' in $($Artifact.Path)" + } + } +} + +function Assert-ManagedPolicyExcludes { + param( + [object]$Artifact, + [Parameter(Mandatory)][string[]]$Patterns + ) + + if ($null -eq $Artifact) { + return + } + foreach ($pattern in $Patterns) { + if ($Artifact.BlockText -match $pattern) { + Add-Failure "Forbidden managed pattern '$pattern' found in $($Artifact.Path)" + } + } +} + +function Test-ManagedBlockMatches { + param( + [Parameter(Mandatory)][string]$BlockText, + [Parameter(Mandatory)][string[]]$Patterns + ) + + foreach ($pattern in $Patterns) { + if ($BlockText -notmatch $pattern) { + return $false + } + } + return $true +} + +function Test-ManagedBlockExcludes { + param( + [Parameter(Mandatory)][string]$BlockText, + [Parameter(Mandatory)][string[]]$Patterns + ) + + foreach ($pattern in $Patterns) { + if ($BlockText -match $pattern) { + return $false + } + } + return $true +} + +function Invoke-PolicyFaultInjection { + param([object]$SourceArtifact) + + if ($null -eq $SourceArtifact) { + Add-Failure 'Could not extract source policy for fault injection.' + return + } + if (-not (Test-ManagedBlockMatches -BlockText $SourceArtifact.BlockText -Patterns $managedPolicyPatterns) -or + -not (Test-ManagedBlockExcludes -BlockText $SourceArtifact.BlockText -Patterns $forbiddenPolicyPatterns)) { + Add-Failure 'Source policy does not satisfy the managed policy contract before fault injection.' + return + } + + $withoutDeadline = $SourceArtifact.BlockText.Replace('HARD_DEADLINE', 'DEADLINE_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutDeadline -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed HARD_DEADLINE clauses.' + } + $withoutStalled = $SourceArtifact.BlockText.Replace('STALLED', 'LIFECYCLE_STOPPED') + if (Test-ManagedBlockMatches -BlockText $withoutStalled -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed STALLED clauses.' + } + $withoutClassification = $SourceArtifact.BlockText.Replace( + 'Classification alone never authorizes or requires delegation', + 'Classification authorization removed' + ) + if (Test-ManagedBlockMatches -BlockText $withoutClassification -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed classification-not-authorization clause.' + } + $withoutRequired = $SourceArtifact.BlockText.Replace('REQUIRED_ACCEPTANCE_CHECKS', 'REQUIRED_CHECKS_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutRequired -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed REQUIRED_ACCEPTANCE_CHECKS clauses.' + } + $withoutOptional = $SourceArtifact.BlockText.Replace('OPTIONAL_EVIDENCE', 'OPTIONAL_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutOptional -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed OPTIONAL_EVIDENCE clauses.' + } + $withoutStop = $SourceArtifact.BlockText.Replace('STOP_CONDITION', 'STOP_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutStop -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed STOP_CONDITION clauses.' + } + $withLegacyWait = $SourceArtifact.BlockText + [Environment]::NewLine + 'a tool-wait timeout is nonterminal: re-wait and do not interrupt' + if (Test-ManagedBlockExcludes -BlockText $withLegacyWait -Patterns $forbiddenPolicyPatterns) { + Add-Failure 'Fault injection did not reject the unbounded wait clause.' + } +} + +$managedPolicyPatterns = @( + 'Managed source: config/AGENTS\.task-aware\.md', + 'Task-aware delegation policy v3\.1', + 'Fix the request-mode authority and mutation boundary', + 'Delegation never expands the authority granted to the parent', + 'gpt-6-astra', + '### Decision order', + 'D0 always remains with the parent and never spawns', + 'Only after an affirmative spawn decision, choose role and effort', + 'Classification alone never authorizes or requires delegation', + 'any delegation gate fails, the parent retains ownership and executes directly', + 'Reasoning effort alone never expands a role''s permissions', + 'Administrator privilege gate', + 'ADMIN_AUTHORIZED: yes', + 'Never auto-promote an escalation', + 'REQUIRED_ACCEPTANCE_CHECKS', + 'OPTIONAL_EVIDENCE', + 'STOP_CONDITION', + 'EXPANSION_TRIGGER', + 'NO_PROGRESS_LIMIT', + 'HARD_DEADLINE', + 'SAFE_CANCELLATION', + 'Max roles', + '(?-i:\bSTALLED\b)', + 'Only a terminal child return may be validated and integrated' +) +$forbiddenPolicyPatterns = @( + 'tool-wait timeout is nonterminal: re-wait and do not interrupt', + 'After required checks pass, continue collecting any additional evidence available', + 'D1 default: call `spawn_agent`' +) Assert-FileContains -Path $configPath -Patterns @( '(?m)^[ \t]*\[agents\][ \t]*(?:#[^\r\n]*)?\r?$', @@ -48,95 +297,82 @@ Assert-FileContains -Path $configPath -Patterns @( '(?m)^max_concurrent_threads_per_session\s*=\s*3\s*$' ) -Assert-FileContains -Path $agentsMdPath -Patterns @( - '', - 'Task-aware delegation policy', - 'agent_type\s*=\s*"luna_task"', - 'agent_type\s*=\s*"luna_task_max"', - 'agent_type\s*=\s*"terra_worker"', - 'agent_type\s*=\s*"terra_worker_max"', - 'agent_type\s*=\s*"sol_specialist"', - 'agent_type\s*=\s*"sol_specialist_max"', - 'Classify capability first, then choose reasoning effort', - 'Higher effort never expands a role''s permissions', - 'Lower model prices reduce the\s+threshold for elevated effort', - 'Use Max as the single elevated effort for D1-D3', - 'Do not add an xhigh middle lane', - 'concrete reason the\s+base effort is likely to be materially more error-prone', - 'bounded read-only investigation or verification', - 'Inputs, the\s+output contract, and the success condition must be explicit', - 'State-changing implementation', - 'tool-heavy multi-step work', - 'requires ordinary judgment', - 'Do not split an atomic D0 item solely because Luna is inexpensive', - 'Do not route an obvious D2 or D3 item through a cheaper role', - 'fork_turns\s*=\s*"none"', - 'packet must explicitly tell the child not to delegate', - '' -) +$sourcePolicyArtifact = Get-ManagedPolicyArtifact -Path $policySource +$livePolicyArtifact = Get-ManagedPolicyArtifact -Path $agentsMdPath +Assert-ManagedPolicyParity -SourceArtifact $sourcePolicyArtifact -LiveArtifact $livePolicyArtifact +# Keep always-loaded guidance compact; do not cap the user's unmanaged text. +if ($null -ne $sourcePolicyArtifact -and $sourcePolicyArtifact.RawBytes.Length -gt 10240) { + Add-Failure 'Managed policy exceeds the 10 KiB maintenance budget; consolidate existing rules before adding more.' +} +Assert-ManagedPolicyContains -Artifact $sourcePolicyArtifact -Patterns $managedPolicyPatterns +Assert-ManagedPolicyContains -Artifact $livePolicyArtifact -Patterns $managedPolicyPatterns +Assert-ManagedPolicyExcludes -Artifact $sourcePolicyArtifact -Patterns $forbiddenPolicyPatterns +Assert-ManagedPolicyExcludes -Artifact $livePolicyArtifact -Patterns $forbiddenPolicyPatterns +Invoke-PolicyFaultInjection -SourceArtifact $sourcePolicyArtifact $expectedAgents = [ordered]@{ - 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-5.6-luna'; Effort = 'low'; Sandbox = 'read-only' } - 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-5.6-luna'; Effort = 'max'; Sandbox = 'read-only' } - 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-5.6-terra'; Effort = 'medium'; Sandbox = $null } - 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-5.6-terra'; Effort = 'max'; Sandbox = $null } - 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-5.6-sol'; Effort = 'high'; Sandbox = 'read-only' } - 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-5.6-sol'; Effort = 'max'; Sandbox = 'read-only' } + 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-5.6-luna'; Effort = 'low'; Sandbox = 'read-only'; Approval = 'never' } + 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-5.6-luna'; Effort = 'max'; Sandbox = 'read-only'; Approval = 'never' } + 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-5.6-terra'; Effort = 'medium'; Sandbox = 'workspace-write'; Approval = 'never' } + 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-5.6-terra'; Effort = 'max'; Sandbox = 'workspace-write'; Approval = 'never' } + 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-5.6-sol'; Effort = 'high'; Sandbox = 'read-only'; Approval = 'never' } + 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-5.6-sol'; Effort = 'max'; Sandbox = 'read-only'; Approval = 'never' } + 'sol-admin-max.toml' = [ordered]@{ Name = 'sol_admin_max'; Model = 'gpt-5.6-sol'; Effort = 'max'; Sandbox = 'danger-full-access'; Approval = 'on-request' } } foreach ($entry in $expectedAgents.GetEnumerator()) { - $escapedModel = [regex]::Escape([string]$entry.Value.Model) - $escapedEffort = [regex]::Escape([string]$entry.Value.Effort) - $escapedName = [regex]::Escape([string]$entry.Value.Name) + $agentFile = $entry.Key + $definition = $entry.Value + $escapedName = [regex]::Escape([string]$definition.Name) + $escapedModel = [regex]::Escape([string]$definition.Model) + $escapedEffort = [regex]::Escape([string]$definition.Effort) + $escapedSandbox = [regex]::Escape([string]$definition.Sandbox) + $escapedApproval = [regex]::Escape([string]$definition.Approval) + $installedPath = Join-Path $agentsPath $agentFile + $sourcePath = Join-Path $agentsSource $agentFile $patterns = @( - "(?m)^name\s*=\s*`"$escapedName`"\s*$", + "(?m)^name\s*=\s*""$escapedName""\s*$", '(?m)^description\s*=\s*"""', '(?m)^developer_instructions\s*=\s*"""', - "(?m)^model\s*=\s*`"$escapedModel`"\s*$", - "(?m)^model_reasoning_effort\s*=\s*`"$escapedEffort`"\s*$" + "(?m)^model\s*=\s*""$escapedModel""\s*$", + "(?m)^model_reasoning_effort\s*=\s*""$escapedEffort""\s*$", + "(?m)^sandbox_mode\s*=\s*""$escapedSandbox""\s*$", + "(?m)^approval_policy\s*=\s*""$escapedApproval""\s*$" ) - if ($entry.Value.Sandbox) { - $escapedSandbox = [regex]::Escape([string]$entry.Value.Sandbox) - $patterns += "(?m)^sandbox_mode\s*=\s*`"$escapedSandbox`"\s*$" - } - if ($entry.Key -eq 'luna-task.toml') { - $patterns += 'Use as the default for compact, homogeneous D1' - $patterns += 'bounded read-only investigation or' - $patterns += 'fixed inputs, an explicit output contract' - $patterns += 'success condition' - $patterns += 'Do not use for material judgment, broad investigation, or state changes' - } - elseif ($entry.Key -eq 'luna-task-max.toml') { - $patterns += 'D1 work that remains deterministic, read-only, and objectively' - $patterns += 'dense cross-checking across heterogeneous inputs' - $patterns += 'Do not use for material judgment, broad investigation, or state changes' - $patterns += 'Use Max reasoning for completeness and cross-checking' - $patterns += 'not to broaden the task''s\s+capability boundary' - } - elseif ($entry.Key -eq 'terra-worker.toml') { - $patterns += 'Use as the default for bounded D2 state-changing implementation' - $patterns += 'tool-heavy\s+multi-step work' - $patterns += 'requires\s+ordinary\s+judgment' - } - elseif ($entry.Key -eq 'terra-worker-max.toml') { - $patterns += 'D2 work that stays within ordinary engineering judgment' - $patterns += 'many\s+coupled constraints' - $patterns += 'Do not use for unresolved architectural trade-offs' - $patterns += 'Use Max reasoning for coupled constraints, edge cases, and verification' - $patterns += 'not to\s+broaden the task''s capability boundary' - } - elseif ($entry.Key -eq 'sol-specialist-max.toml') { - $patterns += 'D3 work when both uncertainty and consequence are high' - $patterns += 'security-sensitive trade-offs' - $patterns += 'reasoning variance' - } - elseif ($entry.Key -eq 'sol-specialist.toml') { - $patterns += 'Use as the default for one bounded D3' - $patterns += 'Prefer sol_specialist_max when uncertainty and consequence are both' - } - Assert-FileContains -Path (Join-Path $agentsPath $entry.Key) -Patterns $patterns + Assert-FileContains -Path $installedPath -Patterns $patterns + Assert-FileByteParity -SourcePath $sourcePath -InstalledPath $installedPath -Label "agent $agentFile" } +foreach ($agentFile in @( + 'luna-task.toml', + 'luna-task-max.toml', + 'terra-worker.toml', + 'terra-worker-max.toml', + 'sol-specialist.toml', + 'sol-specialist-max.toml' + )) { + Assert-FileContains -Path (Join-Path $agentsPath $agentFile) -Patterns @('Never invoke or request sudo') +} +Assert-FileContains -Path (Join-Path $agentsPath 'sol-admin-max.toml') -Patterns @( + 'ADMIN_AUTHORIZED: yes', + 'Before every command that uses sudo', + 'never use sudo -S', + 'This role definition alone is not root or an administrator token' +) + +Assert-FileContains -Path $fullAdminRulePath -Patterns @( + 'decision\s*=\s*"prompt"', + '"sudo"', + '"doas"', + '"pkexec"', + '"su"', + '"runas"', + '"gsudo"', + '"Start-Process"', + 'explicit sol_admin_max full-admin gate' +) +Assert-FileByteParity -SourcePath $rulesSource -InstalledPath $fullAdminRulePath -Label 'full-admin rule' + Assert-FileAbsent -Path (Join-Path $agentsPath 'luna-task-high.toml') Assert-FileAbsent -Path (Join-Path $agentsPath 'terra-worker-high.toml') @@ -150,6 +386,21 @@ if ($codex -and -not $SkipRuntime) { $previousCodexHome = $env:CODEX_HOME try { $env:CODEX_HOME = [IO.Path]::GetFullPath($CodexHome) + $execPolicyOutput = (& $codex.Source execpolicy check --pretty --rules $fullAdminRulePath -- sudo -n true | Out-String) + if ($LASTEXITCODE -ne 0) { + throw "codex execpolicy check failed with exit code $LASTEXITCODE for $fullAdminRulePath" + } + try { + $execPolicyReport = $execPolicyOutput | ConvertFrom-Json -Depth 20 + } + catch { + throw "codex execpolicy check did not return valid JSON: $($execPolicyOutput.Trim())" + } + if ($execPolicyReport.decision -ne 'prompt') { + throw "Full-admin execpolicy did not prompt for sudo: $($execPolicyOutput.Trim())" + } + Write-Host "Full-admin execpolicy prompt passed for CODEX_HOME=$($env:CODEX_HOME)." + if ($ConfigOnlyRuntime) { $doctorOutput = (& $codex.Source --strict-config doctor --json --no-color | Out-String) $doctorExitCode = $LASTEXITCODE @@ -159,7 +410,6 @@ if ($codex -and -not $SkipRuntime) { catch { throw "codex doctor did not return valid JSON for CODEX_HOME=$($env:CODEX_HOME): $($doctorOutput.Trim())" } - $configCheck = $doctorReport.checks.'config.load' if (-not $configCheck -or $configCheck.status -ne 'ok') { throw "Codex strict config load failed for CODEX_HOME=$($env:CODEX_HOME) (doctor exit $doctorExitCode)." diff --git a/scripts/install-task-aware-agent.sh b/scripts/install-task-aware-agent.sh index 41b2c9b..2be52f8 100755 --- a/scripts/install-task-aware-agent.sh +++ b/scripts/install-task-aware-agent.sh @@ -11,6 +11,7 @@ Install the Codex Task-Aware Agent configuration globally. Options: --codex-home PATH Target Codex home (default: $CODEX_HOME or ~/.codex) --set-sol-default Set gpt-5.6-sol with xhigh reasoning as the default + --set-astra-default Set gpt-6-astra with xhigh reasoning as the default --enable-full-access Set approval_policy=never and danger-full-access -h, --help Show this help EOF @@ -23,6 +24,7 @@ die() { codex_home="${CODEX_HOME:-$HOME/.codex}" set_sol_default=false +set_astra_default=false enable_full_access=false while (($# > 0)); do @@ -36,6 +38,10 @@ while (($# > 0)); do set_sol_default=true shift ;; + --set-astra-default) + set_astra_default=true + shift + ;; --enable-full-access) enable_full_access=true shift @@ -50,12 +56,19 @@ while (($# > 0)); do esac done +if [[ "$set_sol_default" == true && "$set_astra_default" == true ]]; then + die '--set-sol-default and --set-astra-default cannot be used together' +fi + script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd -P) repository_root=$(cd -- "$script_dir/.." && pwd -P) agents_source="$repository_root/agents" +rules_source="$repository_root/rules/full-admin.rules" policy_source="$repository_root/config/AGENTS.task-aware.md" config_path="$codex_home/config.toml" agents_path="$codex_home/agents" +rules_path="$codex_home/rules" +full_admin_rule_path="$rules_path/task-aware-full-admin.rules" agents_md_path="$codex_home/AGENTS.md" timestamp=$(date '+%Y%m%d-%H%M%S') backup_path="$codex_home/task-aware-backups/$timestamp" @@ -67,6 +80,7 @@ while [[ -e "$backup_path" ]]; do done [[ -d "$agents_source" ]] || die "missing agents directory: $agents_source" +[[ -f "$rules_source" ]] || die "missing full-admin rule file: $rules_source" [[ -f "$policy_source" ]] || die "missing policy file: $policy_source" command -v awk >/dev/null || die 'awk is required' @@ -77,6 +91,7 @@ expected_agent_files=( terra-worker-max.toml sol-specialist.toml sol-specialist-max.toml + sol-admin-max.toml ) retired_agent_files=(luna-task-high.toml terra-worker-high.toml) for agent_file in "${expected_agent_files[@]}"; do @@ -92,14 +107,17 @@ validate_policy_markers() { end_marker = "" } - $0 == begin_marker { - begin_count++ - if (begin_count > 1 || end_count > 0) invalid = 1 - } - - $0 == end_marker { - end_count++ - if (begin_count != 1 || end_count > 1) invalid = 1 + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + begin_count++ + if (begin_count > 1 || end_count > 0) invalid = 1 + } + if (line == end_marker) { + end_count++ + if (begin_count != 1 || end_count > 1) invalid = 1 + } } END { @@ -260,18 +278,24 @@ merge_policy_block() { for (i = 1; i <= policy_count; i++) print policy[i] } - $0 == begin_marker { - if (!policy_written) { - emit_policy() - policy_written = 1 + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + if (!policy_written) { + emit_policy() + policy_written = 1 + } + block_found = 1 + skipping = 1 + next } - block_found = 1 - skipping = 1 - next } skipping { - if ($0 == end_marker) skipping = 0 + line = $0 + sub(/\r$/, "", line) + if (line == end_marker) skipping = 0 next } @@ -296,10 +320,11 @@ if [[ "$enable_full_access" == true && -f "$config_path" ]] && die 'cannot use --enable-full-access while config.toml defines default_permissions; remove one permission system before retrying' fi -mkdir -p -- "$codex_home" "$agents_path" "$backup_path" +mkdir -p -- "$codex_home" "$agents_path" "$rules_path" "$backup_path" backup_if_present "$config_path" backup_if_present "$agents_md_path" +backup_if_present "$full_admin_rule_path" for agent_file in "${expected_agent_files[@]}"; do backup_if_present "$agents_path/$agent_file" done @@ -318,6 +343,9 @@ remove_toml_section_key "$config_path" features multi_agent if [[ "$set_sol_default" == true ]]; then set_top_level_toml_value "$config_path" model '"gpt-5.6-sol"' set_top_level_toml_value "$config_path" model_reasoning_effort '"xhigh"' +elif [[ "$set_astra_default" == true ]]; then + set_top_level_toml_value "$config_path" model '"gpt-6-astra"' + set_top_level_toml_value "$config_path" model_reasoning_effort '"xhigh"' fi if [[ "$enable_full_access" == true ]]; then @@ -329,6 +357,7 @@ merge_policy_block "$agents_md_path" for agent_file in "${expected_agent_files[@]}"; do cp -f -- "$agents_source/$agent_file" "$agents_path/$agent_file" done +cp -f -- "$rules_source" "$full_admin_rule_path" for agent_file in "${retired_agent_files[@]}"; do rm -f -- "$agents_path/$agent_file" done diff --git a/scripts/test-task-aware-agent.sh b/scripts/test-task-aware-agent.sh index 1061e09..5d4d934 100755 --- a/scripts/test-task-aware-agent.sh +++ b/scripts/test-task-aware-agent.sh @@ -10,7 +10,7 @@ Validate an installed Codex Task-Aware Agent configuration. Options: --codex-home PATH Target Codex home (default: $CODEX_HOME or ~/.codex) - --skip-runtime Skip the codex doctor runtime check + --skip-runtime Skip the codex execpolicy and doctor checks --config-only-runtime Require strict config loading but ignore unrelated doctor failures -h, --help Show this help @@ -24,10 +24,7 @@ config_only_runtime=false while (($# > 0)); do case "$1" in --codex-home) - if (($# < 2)); then - printf 'Error: --codex-home requires a path\n' >&2 - exit 1 - fi + (($# >= 2)) || { printf '%s\n' 'Error: --codex-home requires a path' >&2; exit 1; } codex_home=$2 shift 2 ;; @@ -51,146 +48,363 @@ while (($# > 0)); do done failures=0 +script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd -P) +repository_root=$(cd -- "$script_dir/.." && pwd -P) +source_policy_path="$repository_root/config/AGENTS.task-aware.md" +agents_source="$repository_root/agents" +rules_source="$repository_root/rules/full-admin.rules" +config_path="$codex_home/config.toml" +agents_md_path="$codex_home/AGENTS.md" +agents_path="$codex_home/agents" +full_admin_rule_path="$codex_home/rules/task-aware-full-admin.rules" + +record_failure() { + printf '%s\n' "$*" >&2 + failures=$((failures + 1)) +} assert_file_contains() { local path=$1 shift if [[ ! -f "$path" ]]; then - printf 'Missing file: %s\n' "$path" >&2 - failures=$((failures + 1)) + record_failure "Missing file: $path" return fi local pattern for pattern in "$@"; do if ! grep -Eq -- "$pattern" "$path"; then - printf "Missing pattern '%s' in %s\n" "$pattern" "$path" >&2 - failures=$((failures + 1)) + record_failure "Missing pattern '$pattern' in $path" fi done } assert_file_absent() { local path=$1 - if [[ -e "$path" ]]; then - printf 'Unexpected retired file: %s\n' "$path" >&2 - failures=$((failures + 1)) + record_failure "Unexpected retired file: $path" fi } -config_path="$codex_home/config.toml" -agents_md_path="$codex_home/AGENTS.md" -agents_path="$codex_home/agents" +assert_file_byte_parity() { + local source_path=$1 + local installed_path=$2 + local label=$3 + + if [[ ! -f "$source_path" ]]; then + record_failure "Missing $label source: $source_path" + elif [[ ! -f "$installed_path" ]]; then + record_failure "Missing installed $label: $installed_path" + elif ! cmp -s -- "$source_path" "$installed_path"; then + record_failure "Installed $label does not match source bytes: $label" + fi +} + +count_exact_marker() { + local path=$1 + local marker=$2 + + LC_ALL=C awk -v marker="$marker" ' + { + line = $0 + sub(/\r$/, "", line) + if (line == marker) { + count += 1 + } + } + END { print count + 0 } + ' "$path" +} + +has_one_ordered_marker_pair() { + local path=$1 + local begin_marker='' + local end_marker='' + + LC_ALL=C awk -v begin_marker="$begin_marker" -v end_marker="$end_marker" ' + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + begin_count += 1 + if (end_count > 0 || begin_count != 1) invalid = 1 + } else if (line == end_marker) { + end_count += 1 + if (begin_count != 1 || end_count != 1) invalid = 1 + } + } + END { exit !(begin_count == 1 && end_count == 1 && !invalid) } + ' "$path" +} + +extract_managed_policy_block() { + local path=$1 + local begin_marker='' + local end_marker='' + + LC_ALL=C awk -v begin_marker="$begin_marker" -v end_marker="$end_marker" ' + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + begin_count += 1 + if (end_count > 0 || begin_count != 1) invalid = 1 + if (begin_count == 1 && end_count == 0) emitting = 1 + } + if (emitting) print $0 + if (line == end_marker) { + end_count += 1 + if (begin_count != 1 || !emitting || end_count != 1) invalid = 1 + if (emitting) { + emitting = 0 + completed = 1 + } + } + } + END { exit !(begin_count == 1 && end_count == 1 && completed && !invalid) } + ' "$path" +} + +assert_managed_policy_parity() { + local source_path=$1 + local installed_path=$2 + local begin_marker='' + local end_marker='' + local path begin_count end_count valid=true + + for path in "$source_path" "$installed_path"; do + if [[ ! -f "$path" ]]; then + record_failure "Missing managed policy file: $path" + valid=false + continue + fi + begin_count=$(count_exact_marker "$path" "$begin_marker") + end_count=$(count_exact_marker "$path" "$end_marker") + if [[ "$begin_count" != 1 || "$end_count" != 1 ]]; then + record_failure "Expected exactly one managed marker pair in $path; found begin=$begin_count end=$end_count." + valid=false + fi + if ! has_one_ordered_marker_pair "$path"; then + record_failure "Managed markers are not one ordered begin/end pair in $path." + valid=false + fi + done + + if [[ "$valid" == true ]] && ! cmp -s -- "$source_path" <(extract_managed_policy_block "$installed_path"); then + record_failure 'Managed Task-Aware policy source/live byte parity failed.' + fi +} + +managed_block_matches_patterns() { + local block=$1 + shift + + local pattern + for pattern in "$@"; do + if ! grep -Eq -- "$pattern" <<< "$block"; then + return 1 + fi + done + return 0 +} + +managed_block_has_no_patterns() { + local block=$1 + shift + + local pattern + for pattern in "$@"; do + if grep -Eq -- "$pattern" <<< "$block"; then + return 1 + fi + done + return 0 +} + +assert_managed_policy_patterns() { + local path=$1 + shift + local block + + if [[ ! -f "$path" ]]; then + record_failure "Missing managed policy file: $path" + return + fi + if ! block=$(extract_managed_policy_block "$path"); then + record_failure "Could not extract one ordered managed marker pair from $path." + return + fi + + local pattern + for pattern in "$@"; do + if ! grep -Eq -- "$pattern" <<< "$block"; then + record_failure "Missing managed pattern '$pattern' in $path" + fi + done +} + +assert_managed_policy_excludes() { + local path=$1 + shift + local block + + if [[ ! -f "$path" ]]; then + record_failure "Missing managed policy file: $path" + return + fi + if ! block=$(extract_managed_policy_block "$path"); then + record_failure "Could not extract one ordered managed marker pair from $path." + return + fi + + local pattern + for pattern in "$@"; do + if grep -Eq -- "$pattern" <<< "$block"; then + record_failure "Forbidden managed pattern '$pattern' found in $path" + fi + done +} + +run_policy_fault_injection() { + local source_block without_deadline without_stalled without_classification + local without_required without_optional without_stop with_legacy_wait + + if ! source_block=$(extract_managed_policy_block "$source_policy_path"); then + record_failure 'Could not extract source policy for fault injection.' + return + fi + if ! managed_block_matches_patterns "$source_block" "${managed_policy_patterns[@]}" || + ! managed_block_has_no_patterns "$source_block" "${forbidden_policy_patterns[@]}"; then + record_failure 'Source policy does not satisfy the managed policy contract before fault injection.' + return + fi + + without_deadline=${source_block//HARD_DEADLINE/DEADLINE_REMOVED} + if managed_block_matches_patterns "$without_deadline" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed HARD_DEADLINE clauses.' + fi + without_stalled=${source_block//STALLED/LIFECYCLE_STOPPED} + if managed_block_matches_patterns "$without_stalled" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed STALLED clauses.' + fi + without_classification=${source_block//Classification alone never authorizes or requires delegation/Classification authorization removed} + if managed_block_matches_patterns "$without_classification" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed classification-not-authorization clause.' + fi + without_required=${source_block//REQUIRED_ACCEPTANCE_CHECKS/REQUIRED_CHECKS_REMOVED} + if managed_block_matches_patterns "$without_required" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed REQUIRED_ACCEPTANCE_CHECKS clauses.' + fi + without_optional=${source_block//OPTIONAL_EVIDENCE/OPTIONAL_REMOVED} + if managed_block_matches_patterns "$without_optional" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed OPTIONAL_EVIDENCE clauses.' + fi + without_stop=${source_block//STOP_CONDITION/STOP_REMOVED} + if managed_block_matches_patterns "$without_stop" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed STOP_CONDITION clauses.' + fi + with_legacy_wait="$source_block"$'\n''a tool-wait timeout is nonterminal: re-wait and do not interrupt' + if managed_block_has_no_patterns "$with_legacy_wait" "${forbidden_policy_patterns[@]}"; then + record_failure 'Fault injection did not reject the unbounded wait clause.' + fi +} assert_file_contains "$config_path" \ '^[[:space:]]*\[agents\][[:space:]]*(#.*)?$' \ '^enabled[[:space:]]*=[[:space:]]*true[[:space:]]*$' \ '^max_concurrent_threads_per_session[[:space:]]*=[[:space:]]*3[[:space:]]*$' -assert_file_contains "$agents_md_path" \ - '' \ - 'Task-aware delegation policy' \ - 'agent_type[[:space:]]*=[[:space:]]*"luna_task"' \ - 'agent_type[[:space:]]*=[[:space:]]*"luna_task_max"' \ - 'agent_type[[:space:]]*=[[:space:]]*"terra_worker"' \ - 'agent_type[[:space:]]*=[[:space:]]*"terra_worker_max"' \ - 'agent_type[[:space:]]*=[[:space:]]*"sol_specialist"' \ - 'agent_type[[:space:]]*=[[:space:]]*"sol_specialist_max"' \ - 'Classify capability first, then choose reasoning effort' \ - "Higher effort never expands a role's permissions" \ - 'Lower model prices reduce the' \ - 'threshold for elevated effort' \ - 'Use Max as the single elevated effort for D1-D3' \ - 'Do not add an xhigh middle lane' \ - 'concrete reason the' \ - 'base effort is likely to be materially more error-prone' \ - 'bounded read-only investigation or verification' \ - 'Inputs, the' \ - 'output contract, and the success condition must be explicit' \ - 'State-changing implementation' \ - 'tool-heavy multi-step work' \ - 'requires ordinary judgment' \ - 'Do not split an atomic D0 item solely because Luna is inexpensive' \ - 'Do not route an obvious D2 or D3 item through a cheaper role' \ - 'fork_turns[[:space:]]*=[[:space:]]*"none"' \ - 'packet must explicitly tell the child not to delegate' \ - '' - -assert_file_contains "$agents_path/luna-task.toml" \ - '^name[[:space:]]*=[[:space:]]*"luna_task"[[:space:]]*$' \ - '^description[[:space:]]*=[[:space:]]*"""' \ - '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-luna"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"low"[[:space:]]*$' \ - '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ - 'Use as the default for compact, homogeneous D1' \ - 'bounded read-only investigation or' \ - 'fixed inputs, an explicit output contract' \ - 'success condition' \ - 'Do not use for material judgment, broad investigation, or state changes' - -assert_file_contains "$agents_path/luna-task-max.toml" \ - '^name[[:space:]]*=[[:space:]]*"luna_task_max"[[:space:]]*$' \ - '^description[[:space:]]*=[[:space:]]*"""' \ - '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-luna"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ - '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ - 'D1 work that remains deterministic, read-only, and objectively' \ - 'dense cross-checking across heterogeneous inputs' \ - 'Do not use for material judgment, broad investigation, or state changes' \ - 'Use Max reasoning for completeness and cross-checking' \ - "not to broaden the task's" \ - 'capability boundary' - -assert_file_contains "$agents_path/terra-worker.toml" \ - '^name[[:space:]]*=[[:space:]]*"terra_worker"[[:space:]]*$' \ - '^description[[:space:]]*=[[:space:]]*"""' \ - '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - 'Use as the default for bounded D2 state-changing implementation' \ - 'tool-heavy' \ - 'multi-step work' \ - 'requires ordinary' \ - 'judgment while keeping clear success criteria' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-terra"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' - -assert_file_contains "$agents_path/terra-worker-max.toml" \ - '^name[[:space:]]*=[[:space:]]*"terra_worker_max"[[:space:]]*$' \ - '^description[[:space:]]*=[[:space:]]*"""' \ - '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - 'D2 work that stays within ordinary engineering judgment' \ - 'many' \ - 'coupled constraints' \ - 'Do not use for unresolved architectural trade-offs' \ - 'Use Max reasoning for coupled constraints, edge cases, and verification' \ - 'not to' \ - "broaden the task's capability boundary" \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-terra"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' - -assert_file_contains "$agents_path/sol-specialist.toml" \ - '^name[[:space:]]*=[[:space:]]*"sol_specialist"[[:space:]]*$' \ - '^description[[:space:]]*=[[:space:]]*"""' \ - '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-sol"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' \ - '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ - 'Use as the default for one bounded D3' \ - 'Prefer sol_specialist_max when uncertainty and consequence are both' - -assert_file_contains "$agents_path/sol-specialist-max.toml" \ - '^name[[:space:]]*=[[:space:]]*"sol_specialist_max"[[:space:]]*$' \ - '^description[[:space:]]*=[[:space:]]*"""' \ - '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - 'D3 work when both uncertainty and consequence are high' \ - 'security-sensitive trade-offs' \ - 'reasoning variance' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-sol"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ - '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' +assert_managed_policy_parity "$source_policy_path" "$agents_md_path" + +# Keep always-loaded guidance compact; do not cap the user's unmanaged text. +if [[ -f "$source_policy_path" ]] && (( $(wc -c < "$source_policy_path") > 10240 )); then + record_failure 'Managed policy exceeds the 10 KiB maintenance budget; consolidate existing rules before adding more.' +fi + +managed_policy_patterns=( + 'Managed source: config/AGENTS\.task-aware\.md' + 'Task-aware delegation policy v3\.1' + 'Fix the request-mode authority and mutation boundary' + 'Delegation never expands the authority granted to the parent' + 'gpt-6-astra' + '### Decision order' + 'D0 always remains with the parent and never spawns' + 'Only after an affirmative spawn decision, choose role and effort' + 'Classification alone never authorizes or requires delegation' + 'any delegation gate fails, the parent retains ownership and executes directly' + 'Reasoning effort alone never expands a role'\''s permissions' + 'Administrator privilege gate' + 'ADMIN_AUTHORIZED: yes' + 'Never auto-promote an escalation' + 'REQUIRED_ACCEPTANCE_CHECKS' + 'OPTIONAL_EVIDENCE' + 'STOP_CONDITION' + 'EXPANSION_TRIGGER' + 'NO_PROGRESS_LIMIT' + 'HARD_DEADLINE' + 'SAFE_CANCELLATION' + 'Max roles' + 'STALLED' + 'Only a terminal child return may be validated and integrated' +) +forbidden_policy_patterns=( + 'tool-wait timeout is nonterminal: re-wait and do not interrupt' + 'After required checks pass, continue collecting any additional evidence available' + 'D1 default: call `spawn_agent`' +) + +assert_managed_policy_patterns "$source_policy_path" "${managed_policy_patterns[@]}" +assert_managed_policy_patterns "$agents_md_path" "${managed_policy_patterns[@]}" +assert_managed_policy_excludes "$source_policy_path" "${forbidden_policy_patterns[@]}" +assert_managed_policy_excludes "$agents_md_path" "${forbidden_policy_patterns[@]}" +run_policy_fault_injection + +expected_agent_specs=( + 'luna-task.toml|luna_task|gpt-5.6-luna|low|read-only|never' + 'luna-task-max.toml|luna_task_max|gpt-5.6-luna|max|read-only|never' + 'terra-worker.toml|terra_worker|gpt-5.6-terra|medium|workspace-write|never' + 'terra-worker-max.toml|terra_worker_max|gpt-5.6-terra|max|workspace-write|never' + 'sol-specialist.toml|sol_specialist|gpt-5.6-sol|high|read-only|never' + 'sol-specialist-max.toml|sol_specialist_max|gpt-5.6-sol|max|read-only|never' + 'sol-admin-max.toml|sol_admin_max|gpt-5.6-sol|max|danger-full-access|on-request' +) + +for spec in "${expected_agent_specs[@]}"; do + IFS='|' read -r agent_file agent_name agent_model agent_effort agent_sandbox agent_approval <<< "$spec" + installed_agent_path="$agents_path/$agent_file" + source_agent_path="$agents_source/$agent_file" + assert_file_contains "$installed_agent_path" \ + "^name[[:space:]]*=[[:space:]]*\"$agent_name\"[[:space:]]*$" \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + "^model[[:space:]]*=[[:space:]]*\"$agent_model\"[[:space:]]*$" \ + "^model_reasoning_effort[[:space:]]*=[[:space:]]*\"$agent_effort\"[[:space:]]*$" \ + "^sandbox_mode[[:space:]]*=[[:space:]]*\"$agent_sandbox\"[[:space:]]*$" \ + "^approval_policy[[:space:]]*=[[:space:]]*\"$agent_approval\"[[:space:]]*$" + assert_file_byte_parity "$source_agent_path" "$installed_agent_path" "agent $agent_file" +done + +for agent_file in luna-task.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml sol-specialist.toml sol-specialist-max.toml; do + assert_file_contains "$agents_path/$agent_file" 'Never invoke or request sudo' +done +assert_file_contains "$agents_path/sol-admin-max.toml" \ + 'ADMIN_AUTHORIZED: yes' \ + 'Before every command that uses sudo' \ + 'never use sudo -S' \ + 'This role definition alone is not root or an administrator token' + +assert_file_contains "$full_admin_rule_path" \ + 'decision[[:space:]]*=[[:space:]]*"prompt"' \ + '"sudo"' \ + '"doas"' \ + '"pkexec"' \ + '"su"' \ + '"runas"' \ + '"gsudo"' \ + '"Start-Process"' \ + 'explicit sol_admin_max full-admin gate' +assert_file_byte_parity "$rules_source" "$full_admin_rule_path" 'full-admin rule' assert_file_absent "$agents_path/luna-task-high.toml" assert_file_absent "$agents_path/terra-worker-high.toml" @@ -202,12 +416,23 @@ fi if [[ "$skip_runtime" == false ]]; then if command -v codex >/dev/null 2>&1; then + set +e + execpolicy_output=$(CODEX_HOME="$codex_home" codex execpolicy check --pretty --rules "$full_admin_rule_path" -- sudo -n true 2>&1) + execpolicy_exit=$? + set -e + if ((execpolicy_exit != 0)) || ! grep -Eq '"decision"[[:space:]]*:[[:space:]]*"prompt"' <<< "$execpolicy_output"; then + printf '%s\n' "$execpolicy_output" >&2 + printf 'Full-admin execpolicy did not return prompt for harmless sudo text (exit %d).\n' "$execpolicy_exit" >&2 + exit 1 + fi + printf 'Full-admin execpolicy prompt passed for CODEX_HOME=%s.\n' "$codex_home" + if [[ "$config_only_runtime" == true ]]; then set +e doctor_output=$(CODEX_HOME="$codex_home" codex --strict-config doctor --json --no-color 2>&1) doctor_exit=$? set -e - config_status=$(printf '%s\n' "$doctor_output" | awk ' + config_status=$(awk ' /"config.load"[[:space:]]*:/ { in_config = 1 } in_config && /"status"[[:space:]]*:/ { status = $0 @@ -216,7 +441,7 @@ if [[ "$skip_runtime" == false ]]; then print status exit } - ') + ' <<< "$doctor_output") if [[ "$config_status" != ok ]]; then printf '%s\n' "$doctor_output" >&2 printf 'Codex strict config load failed with doctor exit %d for CODEX_HOME=%s.\n' "$doctor_exit" "$codex_home" >&2 From 097093299cbc896e85d0562e2e0a3bbea08a02a7 Mon Sep 17 00:00:00 2001 From: Codex Date: Thu, 24 Sep 2026 08:55:59 +0000 Subject: [PATCH 3/4] Route task-aware agents across GPT-6 Luna Sol and Astra --- .github/workflows/ci.yml | 119 ++++++++++++++++++++++++++++++- CHANGELOG.md | 7 +- README.md | 50 +++++++------ RELEASE_CHECKLIST.md | 3 +- agents/luna-task-max.toml | 4 +- agents/luna-task-medium.toml | 4 +- agents/luna-task.toml | 2 +- agents/terra-worker-max.toml | 4 +- agents/terra-worker.toml | 2 +- config/AGENTS.task-aware.md | 9 +-- docs/gpt6-family-validation.md | 36 ++++++++++ scripts/Test-TaskAwareAgent.ps1 | 14 ++-- scripts/test-task-aware-agent.sh | 26 +++---- 13 files changed, 221 insertions(+), 59 deletions(-) create mode 100644 docs/gpt6-family-validation.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index dcb9b5f..d4f24a7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,7 +15,7 @@ permissions: contents: read env: - CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.149.1' }} + CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.156.1' }} jobs: windows: @@ -144,6 +144,61 @@ jobs: } & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $migrationHome -ConfigOnlyRuntime + $familyMigrationHome = Join-Path $env:RUNNER_TEMP 'task-aware-family-migration' + $familyAgentsPath = Join-Path $familyMigrationHome 'agents' + New-Item -ItemType Directory -Force -Path $familyAgentsPath | Out-Null + @( + 'model = "gpt-6-astra"' + 'model_reasoning_effort = "max"' + 'approval_policy = "never"' + 'sandbox_mode = "workspace-write"' + ) | Set-Content -LiteralPath (Join-Path $familyMigrationHome 'config.toml') -Encoding utf8 + $familyRoles = @('luna-task.toml', 'luna-task-medium.toml', 'luna-task-max.toml', 'terra-worker.toml', 'terra-worker-max.toml') + foreach ($role in $familyRoles) { + @('model = "gpt-6-astra"', 'model_reasoning_effort = "high"') | + Set-Content -LiteralPath (Join-Path $familyAgentsPath $role) -Encoding utf8 + } + @('name = "custom_unrelated"', 'description = """Unrelated custom role."""', 'developer_instructions = """Keep this role."""', 'model = "gpt-6-astra"', 'model_reasoning_effort = "low"') | + Set-Content -LiteralPath (Join-Path $familyAgentsPath 'custom-unrelated.toml') -Encoding utf8 + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -ConfigOnlyRuntime + $familyConfig = Get-Content -Raw -LiteralPath (Join-Path $familyMigrationHome 'config.toml') + foreach ($expected in @('model = "gpt-6-astra"', 'model_reasoning_effort = "max"', 'approval_policy = "never"', 'sandbox_mode = "workspace-write"')) { + if ($familyConfig -notmatch [regex]::Escape($expected)) { throw "Family migration did not preserve parent setting: $expected" } + } + if ((Get-Content -Raw -LiteralPath (Join-Path $familyAgentsPath 'custom-unrelated.toml')) -notmatch 'custom_unrelated') { + throw 'Family migration overwrote an unrelated custom role.' + } + $backupRoles = Get-ChildItem -LiteralPath (Join-Path $familyMigrationHome 'task-aware-backups') -Recurse -Filter 'luna-task.toml' + if ($backupRoles.Count -eq 0 -or -not ($backupRoles | Where-Object { (Get-Content -Raw -LiteralPath $_.FullName) -match 'gpt-6-astra' })) { + throw 'Family migration did not back up the previous Astra-only D1 role.' + } + $backupConfigs = Get-ChildItem -LiteralPath (Join-Path $familyMigrationHome 'task-aware-backups') -Recurse -Filter 'config.toml' + if ($backupConfigs.Count -eq 0 -or -not ($backupConfigs | Where-Object { (Get-Content -Raw -LiteralPath $_.FullName) -match 'approval_policy = "never"' })) { + throw 'Family migration did not back up the prior parent configuration.' + } + $wrongRolePath = Join-Path $familyAgentsPath 'luna-task.toml' + $wrongRoleText = Get-Content -Raw -LiteralPath $wrongRolePath + $wrongRoleText = [regex]::Replace($wrongRoleText, '(?m)^model\s*=\s*"[^"]+"\s*$', 'model = "gpt-6-sol"') + [IO.File]::WriteAllText($wrongRolePath, $wrongRoleText, [Text.UTF8Encoding]::new($false)) + $stopped = $false + try { & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -SkipRuntime -ErrorAction Stop } + catch { $stopped = $true } + if (-not $stopped) { throw 'Validator accepted a wrong D1 model assignment.' } + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome + foreach ($role in $familyRoles) { + $rolePath = Join-Path $familyAgentsPath $role + $roleText = Get-Content -Raw -LiteralPath $rolePath + $roleText = [regex]::Replace($roleText, '(?m)^model\s*=\s*"[^"]+"\s*$', 'model = "gpt-6-astra"') + [IO.File]::WriteAllText($rolePath, $roleText, [Text.UTF8Encoding]::new($false)) + } + $stopped = $false + try { & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -SkipRuntime -ErrorAction Stop } + catch { $stopped = $true } + if (-not $stopped) { throw 'Validator accepted all-Astra D1/D2 role assignments.' } + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -ConfigOnlyRuntime + $conflictHome = Join-Path $env:RUNNER_TEMP 'task-aware-conflict' New-Item -ItemType Directory -Force -Path $conflictHome | Out-Null 'default_permissions = ":workspace"' | @@ -452,6 +507,68 @@ jobs: --codex-home "$migration_home" \ --config-only-runtime + family_migration_home="$RUNNER_TEMP/task-aware-family-migration" + family_agents_path="$family_migration_home/agents" + mkdir -p -- "$family_agents_path" + printf '%s\n' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"' > "$family_migration_home/config.toml" + family_roles=(luna-task.toml luna-task-medium.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml) + for role in "${family_roles[@]}"; do + printf '%s\n' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "high"' > "$family_agents_path/$role" + done + printf '%s\n' \ + 'name = "custom_unrelated"' \ + 'description = """Unrelated custom role."""' \ + 'developer_instructions = """Keep this role."""' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "low"' > "$family_agents_path/custom-unrelated.toml" + scripts/install-task-aware-agent.sh --codex-home "$family_migration_home" + scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --config-only-runtime + for expected in \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"'; do + grep -Fqx -- "$expected" "$family_migration_home/config.toml" || { + printf 'Family migration did not preserve parent setting: %s\n' "$expected" >&2 + exit 1 + } + done + grep -Fqx -- 'name = "custom_unrelated"' "$family_agents_path/custom-unrelated.toml" || { + printf '%s\n' 'Family migration overwrote an unrelated custom role.' >&2 + exit 1 + } + backup_luna=$(find "$family_migration_home/task-aware-backups" -path '*/agents/luna-task.toml' -type f -print -quit) + [[ -n "$backup_luna" ]] && grep -Fqx 'model = "gpt-6-astra"' "$backup_luna" || { + printf '%s\n' 'Family migration did not back up the previous Astra-only D1 role.' >&2 + exit 1 + } + backup_config=$(find "$family_migration_home/task-aware-backups" -name config.toml -type f -print -quit) + [[ -n "$backup_config" ]] && grep -Fqx 'approval_policy = "never"' "$backup_config" || { + printf '%s\n' 'Family migration did not back up the prior parent configuration.' >&2 + exit 1 + } + sed -i -E 's/^model[[:space:]]*=[[:space:]]*"[^"]+"[[:space:]]*$/model = "gpt-6-sol"/' "$family_agents_path/luna-task.toml" + if scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --skip-runtime; then + printf '%s\n' 'Validator accepted a wrong D1 model assignment.' >&2 + exit 1 + fi + scripts/install-task-aware-agent.sh --codex-home "$family_migration_home" + for role in "${family_roles[@]}"; do + sed -i -E 's/^model[[:space:]]*=[[:space:]]*"[^"]+"[[:space:]]*$/model = "gpt-6-astra"/' "$family_agents_path/$role" + done + if scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --skip-runtime; then + printf '%s\n' 'Validator accepted all-Astra D1/D2 role assignments.' >&2 + exit 1 + fi + scripts/install-task-aware-agent.sh --codex-home "$family_migration_home" + scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --config-only-runtime + conflict_home="$RUNNER_TEMP/task-aware-conflict" mkdir -p -- "$conflict_home" printf '%s\n' 'default_permissions = ":workspace"' > "$conflict_home/config.toml" diff --git a/CHANGELOG.md b/CHANGELOG.md index e8e9c83..c1bc2f6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,7 @@ ### 変更 - 配布ポリシーを v3.1 に更新し、判断順、権限、有限期限、必須確認と終了条件を統合。管理ブロックの 10 KiB 上限を維持。 -- 親の導入オプションと全十役の子を GPT-6 Astra に統一。管理者専用役も Astra Max を使用。D1 は Low/Medium/High、D2 は Medium/High、D3 は High/xhigh、D4 は xhigh/Max を使用。 +- 親の導入オプションは GPT-6 Astra xhigh。子は D1 を GPT-6 Luna Low/Medium/High、D2 を GPT-6 Sol Medium/High、D3 を GPT-6 Astra High/xhigh、D4 を GPT-6 Astra xhigh/Max に更新。管理者専用役は Astra Max を維持。 - 既存六役の名前とファイル名を維持。`_max` は上位枠の識別名とし、D1・D2 は High、D3 は xhigh、D4 は Max に対応。 - D1 の軽い突き合わせ向けに `luna_task_medium` を追加。標準 Low・Medium と上位 High の選択基準を明確化。 - D4 の所見統合用に `astra_architect` / `astra_architect_max` を追加。2件以上の独立した D3 所見を入力とし、読み取り専用・再委譲禁止・最大3子を維持。 @@ -28,10 +28,11 @@ - マーカーの完全一致・順序・一意性、配布ポリシー・十役・規則のバイト一致を検査。CRLF、ブロック外の記述、不正なマーカー、期限や終了条件の欠落、差異、10 KiB 超過を回帰検証する。 -- CI の固定バージョンを、Astra 設定の厳密な読込を確認した Codex CLI 0.149.1 へ更新。 -- Windows/Linux validator を十役の model、effort、sandbox、承認ポリシー と Astra 向けポリシーの検査へ更新。 +- CI の固定バージョンを、GPT-6 ファミリー設定を検証する Codex CLI 0.156.1 へ更新。 +- Windows/Linux validator を十役の model、effort、sandbox、承認ポリシーと GPT-6 ファミリーの割り当ての検査へ更新。 - Windows/Linux installer に、旧 Luna/Terra High role を backup 後に除去する移行を追加。 - CI に親の設定維持、Astra opt-in、廃止した Sol オプションの停止と再導入の検査を追加。 +- 全 Astra 構成からの再導入で D1/D2 のモデルを更新し、既存の親・権限・独自役割・バックアップを保持する移行と、誤ったモデル割り当ての拒否を検証。 - リリース前の実動作確認を、九役の model/effort と D0 から D4 の役割選択へ拡張。静的検査とモデル起動の検証を区別。 ## [0.1.0] - 2026-07-26 diff --git a/README.md b/README.md index 7505907..dabb745 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ OpenAI の公式製品ではなく、Codex の公開仕様に基づくコミュ この構成では、親に **GPT-6 Astra** を使い、難易度の判定、タスクの分割、子の選択、結果の統合を担当させます。 親を切り替える導入オプションは `gpt-6-astra` と `xhigh` を設定します。 -子もすべて Astra とし、D1 は Low・Medium・High、D2 は Medium/High、D3 は High/xhigh、D4 は xhigh/Max を使います。 +子は D1 に **GPT-6 Luna**(Low・Medium・High)、D2 に **GPT-6 Sol**(Medium/High)、D3 に **GPT-6 Astra**(High/xhigh)、D4 に **GPT-6 Astra**(xhigh/Max)を使います。 子のモデルと推論労力を役割ごとに指定することで、親の設定をすべての子が継承することを避けます。 ただし、子もそれぞれトークンと調整時間を使うため、分割する利点がある作業だけを委譲します。 @@ -26,11 +26,11 @@ Ultra を利用できる環境では、並列に分割できる大きな作業 | 難易度 | 対象 | 実行役 | モデルと推論労力 | sandbox | | --- | --- | --- | --- | --- | | D0 | 単純で明確な1工程 | 親が直接処理 | Astra xhigh(必要に応じて変更) | 親の設定 | -| D1 標準 Low | 少量・同形式で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Astra Low | read-only | -| D1 標準 Medium | 複数ファイルや異なる形式を扱い、客観的な条件に沿った軽い突き合わせが必要な作業 | `luna_task_medium` | Astra Medium | read-only | -| D1 上位枠 | D1 のまま、異種入力、密な突合、網羅性の確認、多数の境界条件がある作業 | `luna_task_max` | Astra High | read-only | -| D2 標準 | 状態変更を伴う実装、ツールを使う複数工程、通常判断が必要な調査・検証 | `terra_worker` | Astra Medium | 親から継承 | -| D2 上位枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Astra High | 親から継承 | +| D1 標準 Low | 少量・同形式で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Luna Low | read-only | +| D1 標準 Medium | 複数ファイルや異なる形式を扱い、客観的な条件に沿った軽い突き合わせが必要な作業 | `luna_task_medium` | Luna Medium | read-only | +| D1 上位枠 | D1 のまま、異種入力、密な突合、網羅性の確認、多数の境界条件がある作業 | `luna_task_max` | Luna High | read-only | +| D2 標準 | 状態変更を伴う実装、ツールを使う複数工程、通常判断が必要な調査・検証 | `terra_worker` | Sol Medium | 親から継承 | +| D2 上位枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Sol High | 親から継承 | | D3 標準 | 一つの難しい判断、曖昧性、高リスク、複数領域、設計判断 | `sol_specialist` | Astra High | read-only | | D3 上位枠 | 不確実性と結果の重大性がともに高く、証拠の競合、不可逆な設計、セキュリティ上の重大性などを含む作業 | `sol_specialist_max` | Astra xhigh | read-only | | D4 標準 | 2件以上の独立した D3 作業の所見を、制約や依存関係を踏まえて全体の判断へまとめる作業 | `astra_architect` | Astra xhigh | read-only | @@ -44,12 +44,14 @@ D1・D3・D4 の子は読み取り専用です。書き込みを伴う通常実 既存六つの役割名とファイル名は互換性のため維持し、D1 Medium を1役、D4 を2役追加して計九役にしています。 通常の九役に、管理者操作専用の `sol_admin_max` を加えた計十役を配布します。 -モデルは十役とも `gpt-6-astra` です。管理者役の名前も互換性のため維持します。 +D1 の三役は `gpt-6-luna`、D2 の二役は `gpt-6-sol`、D3・D4 と管理者役は `gpt-6-astra` です。 +`terra_worker` は GPT-6 Sol、`sol_specialist` は GPT-6 Astra を使う互換名です。管理者役の名前も維持します。 `_max` は上位枠の互換名です。D1・D2 の `_max` は High、D3 の `_max` は xhigh、D4 の `_max` は Max に対応します。 役割名からモデルや effort を推測せず、上の表と TOML の設定値を確認してください。 -この割り当ては、役割と権限の境界を保ちながら、すべての子を Astra に統一するためのプロジェクトの選択です。 -Low・Medium・High・xhigh・Max の品質や消費量を比較したベンチマーク結果ではありません。 +この割り当ては、定型処理に Luna、通常実装に Sol、重大な判断と統合に Astra を使うプロジェクトの選択です。 +既存の能力・権限境界と各役割の effort を維持し、モデルの移行だけを比較できるようにしています。 +モデル間や Low・Medium・High・xhigh・Max の品質・消費量を比較したベンチマーク結果ではありません。 モデルの用途と利用可能な設定は [Codex のモデル案内](https://learn.chatgpt.com/docs/models) を参照してください。 D1 の標準枠では、少量・同形式の入力なら Low、複数ファイルや異なる形式を軽く突き合わせるなら Medium を選びます。 @@ -125,8 +127,8 @@ Astra 向けの実行規則として、承認済みの作業を実装と必要 - Windows では PowerShell 7 以降を使用できること。 - Linux では Bash、`awk`、`grep`、`sed`、`cmp` を使用できること。 - カスタムエージェントと subagent workflow に対応した現行 Codex を使用していること。 -- CI の設定互換性検証基準は Codex CLI 0.149.1。Astra の利用には、アカウントとクライアントのモデル選択欄で対応を確認すること。 -- 使用するアカウントで `gpt-6-astra` を利用できること。 +- CI の設定互換性検証基準は Codex CLI 0.156.1。各モデルの利用には、アカウントとクライアントのモデル選択欄で対応を確認すること。 +- 使用するアカウントで `gpt-6-astra`、`gpt-6-sol`、`gpt-6-luna` を利用できること。 - Astra Ultra を使う場合は、対応するアカウントとクライアントで Ultra が有効であること。 - コマンドをこのリポジトリのルートで実行すること。 @@ -137,9 +139,12 @@ codex --version codex doctor --summary --no-color --ascii ``` -モデルと推論の選択欄では、Astra と各役割に必要な Low・Medium・High・xhigh・Max が表示されることも確認します。 -2026-09-05 のローカルモデルカタログ(client version 0.153.0)では、Astra の `low`、`medium`、`high`、`xhigh`、`max`、`ultra` を確認しました。 -設定の厳密な読込は、別途 Codex CLI 0.149.1 で検証しました。 +モデルと推論の選択欄では、上の表にあるモデルと各役割の effort が利用できることも確認します。 +2026-09-24 のローカルモデルカタログ(client version 0.155.0)では、Astra・Sol の `low`、`medium`、`high`、`xhigh`、`max`、`ultra` と、Luna の `low`、`medium`、`high`、`xhigh`、`max` を確認しました。 +Luna は Ultra に対応しません。子はすべて単独で処理し、Ultra は設定しません。 +設定の厳密な読込と変更した五役の実起動は Codex CLI 0.156.1 で確認しました。 +検証の条件と限界は [GPT-6 ファミリー移行の検証記録](docs/gpt6-family-validation.md) を参照してください。 +旧 CLI 0.154.0 では、この環境で GPT-6 Luna/Sol の実起動が拒否されました。モデルが表示されない場合や未対応エラーが出る場合は、クライアントを更新し、利用アカウントの対応も確認してください。 カタログへの掲載や設定の読込成功だけでは、実際のモデル呼び出し成功は保証されません。 ## 導入で変更するもの @@ -151,11 +156,11 @@ codex doctor --summary --no-color --ascii | --- | --- | | `config.toml` | `[agents] enabled = true`、`max_concurrent_threads_per_session = 3` を設定し、旧 key を除去 | | `AGENTS.md` | マーカーで囲んだ task-aware delegation policy を追加または更新 | -| `agents/luna-task.toml` | Astra Low の読み取り専用エージェントを配置 | -| `agents/luna-task-medium.toml` | Astra Medium の読み取り専用エージェントを配置 | -| `agents/luna-task-max.toml` | Astra High の読み取り専用エージェントを配置 | -| `agents/terra-worker.toml` | Astra Medium の作業エージェントを配置 | -| `agents/terra-worker-max.toml` | Astra High の作業エージェントを配置 | +| `agents/luna-task.toml` | Luna Low の読み取り専用エージェントを配置 | +| `agents/luna-task-medium.toml` | Luna Medium の読み取り専用エージェントを配置 | +| `agents/luna-task-max.toml` | Luna High の読み取り専用エージェントを配置 | +| `agents/terra-worker.toml` | Sol Medium の作業エージェントを配置 | +| `agents/terra-worker-max.toml` | Sol High の作業エージェントを配置 | | `agents/sol-specialist.toml` | Astra High の読み取り専用エージェントを配置 | | `agents/sol-specialist-max.toml` | Astra xhigh の読み取り専用エージェントを配置(役割名は互換性維持) | | `agents/astra-architect.toml` | Astra xhigh の読み取り専用D4エージェントを配置 | @@ -226,8 +231,9 @@ CRLF のまま実行すると、shebang の `bash` を解決できず起動に 既に親へ Ultra などを選択していて、その effort を保つ場合は標準導入を使ってください。 Ultra を新たに使う場合は、対応クライアントのモデル選択で Astra と Ultra を選びます。 -旧 `-SetSolDefault` と `--set-sol-default` は廃止しました。 +GPT-5.6 Sol 用だった旧 `-SetSolDefault` と `--set-sol-default` は廃止したままです。 指定すると、ファイルを変更する前に Astra オプションへの移行案内を表示して停止します。 +GPT-6 Sol を親に使う場合は、Codex のモデル選択で指定して標準導入を使います。 親も Astra に切り替える場合は `-SetAstraDefault` または `--set-astra-default` を使ってください。 ### 管理者操作用の役割 @@ -294,7 +300,7 @@ D1 以降の起動確認では、独立した成果など四つの委譲条件 10. 2件以上の独立した D3 所見をまとめる D4 を依頼し、`astra_architect` が xhigh で動くことを確認する。 11. D3 所見間で重大な推奨や証拠が競合する D4 を依頼し、`astra_architect_max` が Max で動くことを確認する。 12. 親が D0 では spawn せず、D1 から D4 では対応する `agent_type` を渡すこと、子が再委譲せず最大3子を守ることを確認する。 -13. 各子の詳細で、model がすべて `gpt-6-astra`、effort と実効権限が上の表と一致することを個別に確認する。 +13. 各子の詳細で、model が D1 は `gpt-6-luna`、D2 は `gpt-6-sol`、D3・D4 は `gpt-6-astra` となり、effort と実効権限が上の表と一致することを個別に確認する。 14. 管理者操作の明示許可がなければ、通常役が昇格を試みず `sol_admin_max` も発行されないことを確認する。実際の管理者操作は、操作固有の許可を得た別の作業で検証する。 `AGENTS.md` の指示チェーンは新しい実行の開始時に構築されるため、導入前から開いているタスクでは確認できません。 @@ -316,7 +322,7 @@ D1 以降の起動確認では、独立した成果など四つの委譲条件 - `config/AGENTS.task-aware.md`:グローバル指示へ追加するルーティング規則 - `config/config.task-aware.toml`:`config.toml` へ統合する設定例 -- `agents/*.toml`:Astra の役割別エージェント(旧役割名を維持) +- `agents/*.toml`:GPT-6 Luna・Sol・Astra の役割別エージェント(旧役割名を維持) - `rules/full-admin.rules`:管理者昇格の承認規則。導入先は `rules/task-aware-full-admin.rules` - `scripts/Install-TaskAwareAgent.ps1`:バックアップ付き導入スクリプト - `scripts/Test-TaskAwareAgent.ps1`:配置と Codex 設定の検証スクリプト diff --git a/RELEASE_CHECKLIST.md b/RELEASE_CHECKLIST.md index 739a602..85c8165 100644 --- a/RELEASE_CHECKLIST.md +++ b/RELEASE_CHECKLIST.md @@ -8,9 +8,10 @@ - [ ] `CHANGELOG.md` の version、日付、release link を確定する。 - [ ] `LICENSE`、`SECURITY.md`、`CONTRIBUTING.md` が release archive に含まれる。 - [ ] GitHub Actions の Windows/Linux job が成功する。 -- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1 の Low/Medium/High 条件と D2/D3/D4 の標準・上位条件を実行する。子の実起動を通じて、model がすべて `gpt-6-astra`、effort が D1 の Low/Medium/High、D2 の Medium/High、D3 の High/xhigh、D4 の xhigh/Max になっていることを確認する。 +- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1 の Low/Medium/High 条件と D2/D3/D4 の標準・上位条件を実行する。子の実起動を通じて、model が D1 は `gpt-6-luna`、D2 は `gpt-6-sol`、D3・D4 は `gpt-6-astra`、effort が D1 の Low/Medium/High、D2 の Medium/High、D3 の High/xhigh、D4 の xhigh/Max になっていることを確認する。 - [ ] D4 の入力に2件以上の独立した D3 所見が含まれ、D4 も読み取り専用・再委譲禁止・同時に最大3子を守ることを確認する。 - [ ] 親の Astra xhigh 設定、標準導入での既存親設定の維持、廃止した Sol オプションの案内と変更前の停止を確認する。 +- [ ] 全 Astra 構成から再導入し、D1/D2 のモデル更新と、親・effort・権限・独自役割・バックアップの保持を確認する。 - [ ] 上位枠が能力境界や sandbox を広げず、task packet に昇格理由が含まれることを確認する。 - [ ] 管理者許可がない場合は通常役が昇格を試みず、`sol_admin_max` も発行されない。規則の直接入口が `prompt` と評価されることを、昇格コマンドを実行せず確認する。 - [ ] 管理者役も `gpt-6-astra`/`max` とし、専用の承認条件を維持する。 diff --git a/agents/luna-task-max.toml b/agents/luna-task-max.toml index 8fee506..3dc889c 100644 --- a/agents/luna-task-max.toml +++ b/agents/luna-task-max.toml @@ -2,11 +2,11 @@ name = "luna_task_max" description = """ Use for D1 work that remains deterministic, read-only, and objectively verifiable, but needs dense cross-checking across heterogeneous inputs, -coverage-sensitive validation, or many edge cases where Astra Medium would be +coverage-sensitive validation, or many edge cases where Luna Medium would be materially more error-prone. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-6-astra" +model = "gpt-6-luna" model_reasoning_effort = "high" sandbox_mode = "read-only" approval_policy = "never" diff --git a/agents/luna-task-medium.toml b/agents/luna-task-medium.toml index a2a3de4..be7cff3 100644 --- a/agents/luna-task-medium.toml +++ b/agents/luna-task-medium.toml @@ -2,12 +2,12 @@ name = "luna_task_medium" description = """ Use for bounded D1 work with fixed inputs, an explicit output contract, and an objective success condition that needs modest reconciliation across files or -formats. Use Astra Medium when compact, homogeneous Low work is insufficient, +formats. Use Luna Medium when compact, homogeneous Low work is insufficient, but dense cross-checking, coverage-sensitive validation, and many edge cases do not justify luna_task_max. This role remains deterministic and read-only. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-6-astra" +model = "gpt-6-luna" model_reasoning_effort = "medium" sandbox_mode = "read-only" diff --git a/agents/luna-task.toml b/agents/luna-task.toml index ab8ae26..a0e27b5 100644 --- a/agents/luna-task.toml +++ b/agents/luna-task.toml @@ -8,7 +8,7 @@ or formats; prefer luna_task_max for dense cross-checking, coverage-sensitive validation, or numerous edge cases. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-6-astra" +model = "gpt-6-luna" model_reasoning_effort = "low" sandbox_mode = "read-only" approval_policy = "never" diff --git a/agents/terra-worker-max.toml b/agents/terra-worker-max.toml index 99255a0..c7d5821 100644 --- a/agents/terra-worker-max.toml +++ b/agents/terra-worker-max.toml @@ -2,11 +2,11 @@ name = "terra_worker_max" description = """ Use for D2 work that stays within ordinary engineering judgment but has many coupled constraints, a long tool or verification chain, difficult debugging, -or expensive rework that justifies deeper reasoning than Astra Medium. +or expensive rework that justifies deeper reasoning than Sol Medium. Do not use for unresolved architectural trade-offs, exceptional risk, or D3 ambiguity. """ -model = "gpt-6-astra" +model = "gpt-6-sol" model_reasoning_effort = "high" approval_policy = "never" diff --git a/agents/terra-worker.toml b/agents/terra-worker.toml index 0cef573..6081d5f 100644 --- a/agents/terra-worker.toml +++ b/agents/terra-worker.toml @@ -5,7 +5,7 @@ multi-step work, or investigation and verification that requires ordinary judgment while keeping clear success criteria. Prefer terra_worker_max when coupled constraints, difficult debugging, or expensive rework justify High. """ -model = "gpt-6-astra" +model = "gpt-6-sol" model_reasoning_effort = "medium" approval_policy = "never" diff --git a/config/AGENTS.task-aware.md b/config/AGENTS.task-aware.md index 75bfc0b..55e7901 100644 --- a/config/AGENTS.task-aware.md +++ b/config/AGENTS.task-aware.md @@ -6,7 +6,8 @@ Fix the request-mode authority and mutation boundary first. Delegation never expands the authority granted to the parent. Answer, review, diagnosis, and monitoring packets must prohibit file and external-state changes. -The target parent is GPT-6 Astra. All ten child roles use `gpt-6-astra`. +The target parent is GPT-6 Astra. D1 uses `gpt-6-luna`, D2 uses `gpt-6-sol`, +and D3/D4 use `gpt-6-astra`. Apply the gates when independent work can run alongside useful parent work; a model name or Ultra cannot permit spawn. Children never delegate. Finish authorized work and required checks; resolve reversible choices from @@ -55,9 +56,9 @@ Reasoning effort alone never expands a role's permissions or capability class. combined consequences make xhigh insufficient. Never repeat D3 investigations. The mapping is D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max. -The nine ordinary roles and separate `sol_admin_max` (Astra Max) retain distinct -capability/authority. The luna, terra, sol, and _max names are compatibility -identifiers, not model or effort declarations. Use the least sufficient effort. +Nine ordinary roles and separate `sol_admin_max` (Astra Max) retain distinct +capability/authority. Role names are compatibility identifiers, not model or +effort declarations. Use the least sufficient effort. For D1 Medium, state what reconciliation makes Low insufficient; select High directly when its conditions are clear, without waiting for Low/Medium failure. Justify upper roles against Medium for D1/D2, High for D3, or xhigh for D4. diff --git a/docs/gpt6-family-validation.md b/docs/gpt6-family-validation.md new file mode 100644 index 0000000..5e48b72 --- /dev/null +++ b/docs/gpt6-family-validation.md @@ -0,0 +1,36 @@ +# GPT-6 ファミリー移行の検証記録 + +検証日: 2026-09-24。対象は `codex/merge-astra-local` の `0ac314a` を基底にした GPT-6 ファミリー割り当てです。 + +## 設定と移行 + +- 全十役の TOML を解析し、モデルと effort の組を同日のモデルカタログで確認。 +- Codex CLI 0.156.1 で strict config load と管理者規則の `prompt` 評価を確認。昇格コマンド自体は実行していません。 +- Linux CI の clean-home、既存設定保持、再導入、旧設定移行、marker/CRLF、drift、fault injection、10 KiB 上限の検査を実行。 +- 旧全 Astra 構成から D1/D2 のモデルが更新され、親の model/effort、権限、独自役割、旧設定のバックアップが保持されることを確認。 +- 誤った D1 モデルと、D1/D2 を全 Astra に戻した構成が validator に拒否されることを確認。 +- Bash 構文と `git diff --check` を確認。Windows の PowerShell 構文・round trip は GitHub Actions の Windows job で確認します。 + +## 実起動 + +既存のグローバル設定を変更せず、一時 `CODEX_HOME` へ導入して認証済みの新しい CLI タスクを開始しました。 +親は Astra Low / read-only。変更対象の五役を `agent_type` と `fork_turns = "none"` で一つずつ起動し、モデル・effort の上書きや代替役へのフォールバックは行っていません。 +各子には再委譲・ツール・書き込みを禁止し、固定文字列 `FAMILY_PROBE_OK` の返却だけを依頼しました。 +返却結果に加え、各子の実行記録の `turn_context` でモデル・effort・sandbox を確認しました。 + +| 役割 | 実モデル | effort | 実効 sandbox | 結果 | +| --- | --- | --- | --- | --- | +| `luna_task` | `gpt-6-luna` | low | read-only | 応答成功 | +| `luna_task_medium` | `gpt-6-luna` | medium | read-only | 応答成功 | +| `luna_task_max` | `gpt-6-luna` | high | read-only | 応答成功 | +| `terra_worker` | `gpt-6-sol` | medium | read-only(親から継承) | 応答成功 | +| `terra_worker_max` | `gpt-6-sol` | high | read-only(親から継承) | 応答成功 | + +最初の CLI 0.154.0 では、五役とも GPT-6 Luna/Sol が ChatGPT アカウントで未対応という HTTP 400 を返しました。 +同じアカウントで CLI 0.156.1 を一時領域に導入すると、モデル一覧に Luna/Sol が現れ、上記五役が応答しました。 +この差を受け、CI の固定バージョンを 0.156.1 に更新しています。0.154.0 の設定読込成功は、モデル利用可能性の証拠にはなりません。 + +この probe は役割の読込・モデル選択・応答の確認です。D0-D4 の自律分類、実装品質、消費量、速度、書き込み時の sandbox、変更していない Astra 役の再評価は対象外です。 +管理者役は起動していません。全役の分類評価と管理者操作の実動作は、公開チェックリスト上の別確認です。 + +参照: [公式のモデル選択案内](https://learn.chatgpt.com/docs/models)、[GPT-6 ファミリーの案内](https://developers.openai.com/api/docs/guides/latest-model)。役割への割り当ては本プロジェクトの方針であり、モデル間の性能比較結果ではありません。 diff --git a/scripts/Test-TaskAwareAgent.ps1 b/scripts/Test-TaskAwareAgent.ps1 index 45926dd..f111b7d 100644 --- a/scripts/Test-TaskAwareAgent.ps1 +++ b/scripts/Test-TaskAwareAgent.ps1 @@ -264,9 +264,9 @@ $managedPolicyPatterns = @( 'Task-aware delegation policy v3\.1', 'Fix the request-mode authority and mutation boundary', 'Delegation never expands the authority granted to the parent', - 'gpt-6-astra', + 'D1 uses `gpt-6-luna`, D2 uses `gpt-6-sol`', 'The target parent is GPT-6 Astra', - 'All ten child roles use', + 'D3/D4 use `gpt-6-astra`', 'Children never delegate', 'Classify capability first, then choose reasoning effort', 'luna_task_medium', @@ -322,11 +322,11 @@ Assert-ManagedPolicyExcludes -Artifact $livePolicyArtifact -Patterns $forbiddenP Invoke-PolicyFaultInjection -SourceArtifact $sourcePolicyArtifact $expectedAgents = [ordered]@{ - 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-6-astra'; Effort = 'low'; Sandbox = 'read-only'; Approval = 'never' } - 'luna-task-medium.toml' = [ordered]@{ Name = 'luna_task_medium'; Model = 'gpt-6-astra'; Effort = 'medium'; Sandbox = 'read-only'; Approval = 'never' } - 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = 'read-only'; Approval = 'never' } - 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-6-astra'; Effort = 'medium'; Sandbox = $null; Approval = 'never' } - 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = $null; Approval = 'never' } + 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-6-luna'; Effort = 'low'; Sandbox = 'read-only'; Approval = 'never' } + 'luna-task-medium.toml' = [ordered]@{ Name = 'luna_task_medium'; Model = 'gpt-6-luna'; Effort = 'medium'; Sandbox = 'read-only'; Approval = 'never' } + 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-6-luna'; Effort = 'high'; Sandbox = 'read-only'; Approval = 'never' } + 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-6-sol'; Effort = 'medium'; Sandbox = $null; Approval = 'never' } + 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-6-sol'; Effort = 'high'; Sandbox = $null; Approval = 'never' } 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = 'read-only'; Approval = 'never' } 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-6-astra'; Effort = 'xhigh'; Sandbox = 'read-only'; Approval = 'never' } 'astra-architect.toml' = [ordered]@{ Name = 'astra_architect'; Model = 'gpt-6-astra'; Effort = 'xhigh'; Sandbox = 'read-only'; Approval = 'never' } diff --git a/scripts/test-task-aware-agent.sh b/scripts/test-task-aware-agent.sh index 1c50fdd..29231d1 100755 --- a/scripts/test-task-aware-agent.sh +++ b/scripts/test-task-aware-agent.sh @@ -327,9 +327,9 @@ managed_policy_patterns=( 'Task-aware delegation policy v3\.1' 'Fix the request-mode authority and mutation boundary' 'Delegation never expands the authority granted to the parent' - 'gpt-6-astra' + 'D1 uses `gpt-6-luna`, D2 uses `gpt-6-sol`' 'The target parent is GPT-6 Astra' - 'All ten child roles use' + 'D3/D4 use `gpt-6-astra`' 'Children never delegate' 'Classify capability first, then choose reasoning effort' 'luna_task_medium' @@ -372,11 +372,11 @@ assert_managed_policy_excludes "$agents_md_path" "${forbidden_policy_patterns[@] run_policy_fault_injection expected_agent_specs=( - 'luna-task.toml|luna_task|gpt-6-astra|low|read-only|never' - 'luna-task-medium.toml|luna_task_medium|gpt-6-astra|medium|read-only|never' - 'luna-task-max.toml|luna_task_max|gpt-6-astra|high|read-only|never' - 'terra-worker.toml|terra_worker|gpt-6-astra|medium||never' - 'terra-worker-max.toml|terra_worker_max|gpt-6-astra|high||never' + 'luna-task.toml|luna_task|gpt-6-luna|low|read-only|never' + 'luna-task-medium.toml|luna_task_medium|gpt-6-luna|medium|read-only|never' + 'luna-task-max.toml|luna_task_max|gpt-6-luna|high|read-only|never' + 'terra-worker.toml|terra_worker|gpt-6-sol|medium||never' + 'terra-worker-max.toml|terra_worker_max|gpt-6-sol|high||never' 'sol-specialist.toml|sol_specialist|gpt-6-astra|high|read-only|never' 'sol-specialist-max.toml|sol_specialist_max|gpt-6-astra|xhigh|read-only|never' 'astra-architect.toml|astra_architect|gpt-6-astra|xhigh|read-only|never' @@ -406,12 +406,12 @@ done for agent_file in luna-task.toml luna-task-medium.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml sol-specialist.toml sol-specialist-max.toml astra-architect.toml astra-architect-max.toml; do assert_file_contains "$agents_path/$agent_file" 'Never invoke or request sudo' done -# Retain the Astra branch role capability contracts. +# Retain the GPT-6 family role capability contracts. assert_file_contains "$agents_path/luna-task.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-luna"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"low"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'Use as the default for compact, homogeneous D1' \ @@ -424,7 +424,7 @@ assert_file_contains "$agents_path/luna-task-medium.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task_medium"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-luna"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'bounded D1 work with fixed inputs' \ @@ -437,7 +437,7 @@ assert_file_contains "$agents_path/luna-task-max.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task_max"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-luna"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'D1 work that remains deterministic, read-only, and objectively' \ @@ -456,7 +456,7 @@ assert_file_contains "$agents_path/terra-worker.toml" \ 'multi-step work' \ 'requires ordinary' \ 'judgment while keeping clear success criteria' \ - '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-sol"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' assert_file_contains "$agents_path/terra-worker-max.toml" \ @@ -470,7 +470,7 @@ assert_file_contains "$agents_path/terra-worker-max.toml" \ 'Use High reasoning for coupled constraints, edge cases, and verification' \ 'not to' \ "broaden the task's capability boundary" \ - '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-sol"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' assert_file_contains "$agents_path/sol-specialist.toml" \ From 79a896c2401a167c1030d20fa09f598519d0472d Mon Sep 17 00:00:00 2001 From: Codex Date: Thu, 24 Sep 2026 08:57:22 +0000 Subject: [PATCH 4/4] Record passing Windows and Linux migration checks --- docs/gpt6-family-validation.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/gpt6-family-validation.md b/docs/gpt6-family-validation.md index 5e48b72..db26b1f 100644 --- a/docs/gpt6-family-validation.md +++ b/docs/gpt6-family-validation.md @@ -9,7 +9,7 @@ - Linux CI の clean-home、既存設定保持、再導入、旧設定移行、marker/CRLF、drift、fault injection、10 KiB 上限の検査を実行。 - 旧全 Astra 構成から D1/D2 のモデルが更新され、親の model/effort、権限、独自役割、旧設定のバックアップが保持されることを確認。 - 誤った D1 モデルと、D1/D2 を全 Astra に戻した構成が validator に拒否されることを確認。 -- Bash 構文と `git diff --check` を確認。Windows の PowerShell 構文・round trip は GitHub Actions の Windows job で確認します。 +- Bash 構文と `git diff --check` を確認。Windows の PowerShell 構文・round trip と Linux の全検査も [GitHub Actions](https://github.com/Eonshore/Codex-Task-Aware-Agent/actions/runs/35978162954) で成功しました(実装コミット `0970932`)。 ## 実起動