diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 004a427..d4f24a7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,7 +15,7 @@ permissions: contents: read env: - CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.145.0' }} + CODEX_PACKAGE: ${{ github.event_name == 'schedule' && '@openai/codex@latest' || '@openai/codex@0.156.1' }} jobs: windows: @@ -63,6 +63,53 @@ jobs: $backupCount = @(Get-ChildItem -Directory -LiteralPath (Join-Path $cleanHome 'task-aware-backups')).Count if ($backupCount -lt 2) { throw 'Repeated installation reused a backup directory.' } + $standardHome = Join-Path $env:RUNNER_TEMP 'task-aware-standard-defaults' + New-Item -ItemType Directory -Force -Path $standardHome | Out-Null + @( + 'model = "parent-model"' + 'model_reasoning_effort = "medium"' + 'approval_policy = "on-request"' + 'sandbox_mode = "workspace-write"' + ) | Set-Content -LiteralPath (Join-Path $standardHome 'config.toml') -Encoding utf8 + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $standardHome + $standard = Get-Content -Raw -LiteralPath (Join-Path $standardHome 'config.toml') + foreach ($expected in @( + 'model = "parent-model"', + 'model_reasoning_effort = "medium"', + 'approval_policy = "on-request"', + 'sandbox_mode = "workspace-write"' + )) { + $key = $expected.Split('=')[0].Trim() + if ([regex]::Matches($standard, "(?m)^\s*" + [regex]::Escape($key) + '\s*=').Count -ne 1 -or + $standard -notmatch [regex]::Escape($expected)) { + throw "Standard install changed $expected." + } + } + + $astraHome = Join-Path $env:RUNNER_TEMP 'task-aware-astra-default' + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $astraHome -SetAstraDefault + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $astraHome -SetAstraDefault + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $astraHome -ConfigOnlyRuntime + $astra = Get-Content -Raw -LiteralPath (Join-Path $astraHome 'config.toml') + foreach ($expected in @('model = "gpt-6-astra"', 'model_reasoning_effort = "xhigh"')) { + if ([regex]::Matches($astra, "(?m)^\s*" + [regex]::Escape($expected.Split('=')[0].Trim()) + '\s*=').Count -ne 1 -or + $astra -notmatch [regex]::Escape($expected)) { + throw "Astra default was not set exactly once: $expected." + } + } + + $retiredSolHome = Join-Path $env:RUNNER_TEMP 'task-aware-retired-sol' + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $retiredSolHome -SetSolDefault -ErrorAction Stop + } + catch { + $stopped = $_.Exception.Message -like '*was retired*' + } + if (-not $stopped -or (Test-Path -LiteralPath $retiredSolHome)) { + throw 'Retired Sol option did not fail before filesystem mutation.' + } + $migrationHome = Join-Path $env:RUNNER_TEMP 'task-aware-migration' New-Item -ItemType Directory -Force -Path $migrationHome | Out-Null $migrationConfig = @( @@ -97,6 +144,61 @@ jobs: } & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $migrationHome -ConfigOnlyRuntime + $familyMigrationHome = Join-Path $env:RUNNER_TEMP 'task-aware-family-migration' + $familyAgentsPath = Join-Path $familyMigrationHome 'agents' + New-Item -ItemType Directory -Force -Path $familyAgentsPath | Out-Null + @( + 'model = "gpt-6-astra"' + 'model_reasoning_effort = "max"' + 'approval_policy = "never"' + 'sandbox_mode = "workspace-write"' + ) | Set-Content -LiteralPath (Join-Path $familyMigrationHome 'config.toml') -Encoding utf8 + $familyRoles = @('luna-task.toml', 'luna-task-medium.toml', 'luna-task-max.toml', 'terra-worker.toml', 'terra-worker-max.toml') + foreach ($role in $familyRoles) { + @('model = "gpt-6-astra"', 'model_reasoning_effort = "high"') | + Set-Content -LiteralPath (Join-Path $familyAgentsPath $role) -Encoding utf8 + } + @('name = "custom_unrelated"', 'description = """Unrelated custom role."""', 'developer_instructions = """Keep this role."""', 'model = "gpt-6-astra"', 'model_reasoning_effort = "low"') | + Set-Content -LiteralPath (Join-Path $familyAgentsPath 'custom-unrelated.toml') -Encoding utf8 + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -ConfigOnlyRuntime + $familyConfig = Get-Content -Raw -LiteralPath (Join-Path $familyMigrationHome 'config.toml') + foreach ($expected in @('model = "gpt-6-astra"', 'model_reasoning_effort = "max"', 'approval_policy = "never"', 'sandbox_mode = "workspace-write"')) { + if ($familyConfig -notmatch [regex]::Escape($expected)) { throw "Family migration did not preserve parent setting: $expected" } + } + if ((Get-Content -Raw -LiteralPath (Join-Path $familyAgentsPath 'custom-unrelated.toml')) -notmatch 'custom_unrelated') { + throw 'Family migration overwrote an unrelated custom role.' + } + $backupRoles = Get-ChildItem -LiteralPath (Join-Path $familyMigrationHome 'task-aware-backups') -Recurse -Filter 'luna-task.toml' + if ($backupRoles.Count -eq 0 -or -not ($backupRoles | Where-Object { (Get-Content -Raw -LiteralPath $_.FullName) -match 'gpt-6-astra' })) { + throw 'Family migration did not back up the previous Astra-only D1 role.' + } + $backupConfigs = Get-ChildItem -LiteralPath (Join-Path $familyMigrationHome 'task-aware-backups') -Recurse -Filter 'config.toml' + if ($backupConfigs.Count -eq 0 -or -not ($backupConfigs | Where-Object { (Get-Content -Raw -LiteralPath $_.FullName) -match 'approval_policy = "never"' })) { + throw 'Family migration did not back up the prior parent configuration.' + } + $wrongRolePath = Join-Path $familyAgentsPath 'luna-task.toml' + $wrongRoleText = Get-Content -Raw -LiteralPath $wrongRolePath + $wrongRoleText = [regex]::Replace($wrongRoleText, '(?m)^model\s*=\s*"[^"]+"\s*$', 'model = "gpt-6-sol"') + [IO.File]::WriteAllText($wrongRolePath, $wrongRoleText, [Text.UTF8Encoding]::new($false)) + $stopped = $false + try { & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -SkipRuntime -ErrorAction Stop } + catch { $stopped = $true } + if (-not $stopped) { throw 'Validator accepted a wrong D1 model assignment.' } + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome + foreach ($role in $familyRoles) { + $rolePath = Join-Path $familyAgentsPath $role + $roleText = Get-Content -Raw -LiteralPath $rolePath + $roleText = [regex]::Replace($roleText, '(?m)^model\s*=\s*"[^"]+"\s*$', 'model = "gpt-6-astra"') + [IO.File]::WriteAllText($rolePath, $roleText, [Text.UTF8Encoding]::new($false)) + } + $stopped = $false + try { & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -SkipRuntime -ErrorAction Stop } + catch { $stopped = $true } + if (-not $stopped) { throw 'Validator accepted all-Astra D1/D2 role assignments.' } + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $familyMigrationHome -ConfigOnlyRuntime + $conflictHome = Join-Path $env:RUNNER_TEMP 'task-aware-conflict' New-Item -ItemType Directory -Force -Path $conflictHome | Out-Null 'default_permissions = ":workspace"' | @@ -125,6 +227,166 @@ jobs: } if (-not $stopped) { throw 'Malformed AGENTS.md markers were not rejected.' } + $literalMarkerHome = Join-Path $env:RUNNER_TEMP 'task-aware-literal-markers' + New-Item -ItemType Directory -Force -Path $literalMarkerHome | Out-Null + $literalMarkerLine = 'User text: and are examples.' + [IO.File]::WriteAllText( + (Join-Path $literalMarkerHome 'AGENTS.md'), + $literalMarkerLine + "`n", + [Text.UTF8Encoding]::new($false) + ) + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $literalMarkerHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $literalMarkerHome -SkipRuntime + if ((Get-Content -Raw -LiteralPath (Join-Path $literalMarkerHome 'AGENTS.md')) -notmatch [regex]::Escape($literalMarkerLine)) { + throw 'Inline marker examples were not preserved.' + } + + $preserveHome = Join-Path $env:RUNNER_TEMP 'task-aware-preserve' + New-Item -ItemType Directory -Force -Path $preserveHome | Out-Null + $preserveConfig = @( + 'model = "gpt-6-astra"' + 'model_reasoning_effort = "max"' + 'approval_policy = "never"' + 'sandbox_mode = "workspace-write"' + 'unmanaged_setting = "retain-me"' + ) -join "`n" + [IO.File]::WriteAllText( + (Join-Path $preserveHome 'config.toml'), + $preserveConfig, + [Text.UTF8Encoding]::new($false) + ) + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $preserveHome + $preserved = Get-Content -Raw -LiteralPath (Join-Path $preserveHome 'config.toml') + foreach ($line in @( + 'model = "gpt-6-astra"' + 'model_reasoning_effort = "max"' + 'approval_policy = "never"' + 'sandbox_mode = "workspace-write"' + 'unmanaged_setting = "retain-me"' + )) { + if ($preserved -notmatch [regex]::Escape($line)) { + throw "Default installation did not preserve: $line" + } + } + + $selectionConflictHome = Join-Path $env:RUNNER_TEMP 'task-aware-selection-conflict' + New-Item -ItemType Directory -Force -Path $selectionConflictHome | Out-Null + $selectionConfigPath = Join-Path $selectionConflictHome 'config.toml' + [IO.File]::WriteAllText( + $selectionConfigPath, + "model = ""preserve""`n", + [Text.UTF8Encoding]::new($false) + ) + $beforeHash = (Get-FileHash -Algorithm SHA256 -LiteralPath $selectionConfigPath).Hash + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $selectionConflictHome -SetSolDefault -SetAstraDefault -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped) { throw 'Conflicting parent model switches were accepted.' } + $afterHash = (Get-FileHash -Algorithm SHA256 -LiteralPath $selectionConfigPath).Hash + if ($beforeHash -cne $afterHash) { throw 'Conflicting parent model switches mutated config.toml.' } + foreach ($path in @('task-aware-backups', 'agents', 'rules', 'AGENTS.md')) { + if (Test-Path -LiteralPath (Join-Path $selectionConflictHome $path)) { + throw "Conflicting parent model switches created $path." + } + } + + foreach ($markerCase in @('duplicate', 'reversed')) { + $markerHome = Join-Path $env:RUNNER_TEMP "task-aware-markers-$markerCase" + New-Item -ItemType Directory -Force -Path $markerHome | Out-Null + $markerContent = switch ($markerCase) { + 'duplicate' { + "`n`n`n`n" + } + default { + "`n`n" + } + } + [IO.File]::WriteAllText((Join-Path $markerHome 'AGENTS.md'), $markerContent, [Text.UTF8Encoding]::new($false)) + $stopped = $false + try { + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $markerHome -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped) { throw "Marker case unexpectedly succeeded: $markerCase" } + } + + $crlfHome = Join-Path $env:RUNNER_TEMP 'task-aware-crlf' + New-Item -ItemType Directory -Force -Path $crlfHome | Out-Null + $policy = Get-Content -Raw -LiteralPath ./config/AGENTS.task-aware.md + $crlfPolicy = [regex]::Replace($policy, "`r?`n", "`r`n") + [IO.File]::WriteAllText((Join-Path $crlfHome 'AGENTS.md'), $crlfPolicy, [Text.UTF8Encoding]::new($false)) + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $crlfHome + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $crlfHome -SkipRuntime + + foreach ($drift in @('policy', 'role', 'rule')) { + $driftHome = Join-Path $env:RUNNER_TEMP "task-aware-drift-$drift" + & ./scripts/Install-TaskAwareAgent.ps1 -CodexHome $driftHome + switch ($drift) { + 'policy' { + $driftPath = Join-Path $driftHome 'AGENTS.md' + $driftContent = Get-Content -Raw -LiteralPath $driftPath + $driftContent = $driftContent.Replace( + '', + "# drift`n" + ) + [IO.File]::WriteAllText($driftPath, $driftContent, [Text.UTF8Encoding]::new($false)) + } + 'role' { + [IO.File]::AppendAllText( + (Join-Path $driftHome 'agents/luna-task.toml'), + "# drift`n", + [Text.UTF8Encoding]::new($false) + ) + } + 'rule' { + [IO.File]::AppendAllText( + (Join-Path $driftHome 'rules/task-aware-full-admin.rules'), + "# drift`n", + [Text.UTF8Encoding]::new($false) + ) + } + } + $stopped = $false + try { + & ./scripts/Test-TaskAwareAgent.ps1 -CodexHome $driftHome -SkipRuntime -ErrorAction Stop + } + catch { + $stopped = $true + } + if (-not $stopped) { throw "Drift case unexpectedly passed: $drift" } + } + + - name: Reject oversized managed policy + shell: pwsh + run: | + $budgetRepo = Join-Path $env:RUNNER_TEMP 'task-aware-budget-repo' + $budgetHome = Join-Path $env:RUNNER_TEMP 'task-aware-budget-home' + New-Item -ItemType Directory -Path $budgetRepo | Out-Null + foreach ($directory in @('scripts', 'config', 'agents', 'rules')) { + Copy-Item -LiteralPath $directory -Destination $budgetRepo -Recurse + } + $budgetSource = Join-Path $budgetRepo 'config/AGENTS.task-aware.md' + $policy = Get-Content -Raw -LiteralPath $budgetSource + $end = '' + $policy = $policy.Replace($end, "`n$end") + [IO.File]::WriteAllText($budgetSource, $policy, [Text.UTF8Encoding]::new($false)) + & (Join-Path $budgetRepo 'scripts/Install-TaskAwareAgent.ps1') -CodexHome $budgetHome + $stopped = $false + try { + & (Join-Path $budgetRepo 'scripts/Test-TaskAwareAgent.ps1') -CodexHome $budgetHome -SkipRuntime + } + catch { + if ($_.Exception.Message -notmatch 'Managed policy exceeds the 10 KiB maintenance budget') { throw } + $stopped = $true + } + if (-not $stopped) { throw 'An oversized managed policy passed validation.' } + linux: name: Linux clean-home round trip runs-on: ubuntu-latest @@ -164,6 +426,53 @@ jobs: exit 1 fi + standard_home="$RUNNER_TEMP/task-aware-standard-defaults" + mkdir -p -- "$standard_home" + cat > "$standard_home/config.toml" <<'EOF' + model = "parent-model" + model_reasoning_effort = "medium" + approval_policy = "on-request" + sandbox_mode = "workspace-write" + EOF + scripts/install-task-aware-agent.sh --codex-home "$standard_home" + for expected in \ + 'model = "parent-model"' \ + 'model_reasoning_effort = "medium"' \ + 'approval_policy = "on-request"' \ + 'sandbox_mode = "workspace-write"'; do + key=${expected%% =*} + if [[ $(grep -Ec "^[[:space:]]*$key[[:space:]]*=" "$standard_home/config.toml") -ne 1 ]] || + ! grep -Fqx "$expected" "$standard_home/config.toml"; then + printf 'Standard install changed %s.\n' "$expected" >&2 + exit 1 + fi + done + + astra_home="$RUNNER_TEMP/task-aware-astra-default" + scripts/install-task-aware-agent.sh --codex-home "$astra_home" --set-astra-default + scripts/install-task-aware-agent.sh --codex-home "$astra_home" --set-astra-default + scripts/test-task-aware-agent.sh --codex-home "$astra_home" --config-only-runtime + for expected in \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "xhigh"'; do + key=${expected%% =*} + if [[ $(grep -Ec "^[[:space:]]*$key[[:space:]]*=" "$astra_home/config.toml") -ne 1 ]] || + ! grep -Fqx "$expected" "$astra_home/config.toml"; then + printf 'Astra default was not set exactly once: %s.\n' "$expected" >&2 + exit 1 + fi + done + + retired_sol_home="$RUNNER_TEMP/task-aware-retired-sol" + if scripts/install-task-aware-agent.sh --codex-home "$retired_sol_home" --set-sol-default; then + printf '%s\n' 'Retired Sol option was not rejected.' >&2 + exit 1 + fi + if [[ -e "$retired_sol_home" ]]; then + printf '%s\n' 'Retired Sol option mutated the target home.' >&2 + exit 1 + fi + migration_home="$RUNNER_TEMP/task-aware-migration" mkdir -p -- "$migration_home" cat > "$migration_home/config.toml" <<'EOF' @@ -198,6 +507,68 @@ jobs: --codex-home "$migration_home" \ --config-only-runtime + family_migration_home="$RUNNER_TEMP/task-aware-family-migration" + family_agents_path="$family_migration_home/agents" + mkdir -p -- "$family_agents_path" + printf '%s\n' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"' > "$family_migration_home/config.toml" + family_roles=(luna-task.toml luna-task-medium.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml) + for role in "${family_roles[@]}"; do + printf '%s\n' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "high"' > "$family_agents_path/$role" + done + printf '%s\n' \ + 'name = "custom_unrelated"' \ + 'description = """Unrelated custom role."""' \ + 'developer_instructions = """Keep this role."""' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "low"' > "$family_agents_path/custom-unrelated.toml" + scripts/install-task-aware-agent.sh --codex-home "$family_migration_home" + scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --config-only-runtime + for expected in \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"'; do + grep -Fqx -- "$expected" "$family_migration_home/config.toml" || { + printf 'Family migration did not preserve parent setting: %s\n' "$expected" >&2 + exit 1 + } + done + grep -Fqx -- 'name = "custom_unrelated"' "$family_agents_path/custom-unrelated.toml" || { + printf '%s\n' 'Family migration overwrote an unrelated custom role.' >&2 + exit 1 + } + backup_luna=$(find "$family_migration_home/task-aware-backups" -path '*/agents/luna-task.toml' -type f -print -quit) + [[ -n "$backup_luna" ]] && grep -Fqx 'model = "gpt-6-astra"' "$backup_luna" || { + printf '%s\n' 'Family migration did not back up the previous Astra-only D1 role.' >&2 + exit 1 + } + backup_config=$(find "$family_migration_home/task-aware-backups" -name config.toml -type f -print -quit) + [[ -n "$backup_config" ]] && grep -Fqx 'approval_policy = "never"' "$backup_config" || { + printf '%s\n' 'Family migration did not back up the prior parent configuration.' >&2 + exit 1 + } + sed -i -E 's/^model[[:space:]]*=[[:space:]]*"[^"]+"[[:space:]]*$/model = "gpt-6-sol"/' "$family_agents_path/luna-task.toml" + if scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --skip-runtime; then + printf '%s\n' 'Validator accepted a wrong D1 model assignment.' >&2 + exit 1 + fi + scripts/install-task-aware-agent.sh --codex-home "$family_migration_home" + for role in "${family_roles[@]}"; do + sed -i -E 's/^model[[:space:]]*=[[:space:]]*"[^"]+"[[:space:]]*$/model = "gpt-6-astra"/' "$family_agents_path/$role" + done + if scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --skip-runtime; then + printf '%s\n' 'Validator accepted all-Astra D1/D2 role assignments.' >&2 + exit 1 + fi + scripts/install-task-aware-agent.sh --codex-home "$family_migration_home" + scripts/test-task-aware-agent.sh --codex-home "$family_migration_home" --config-only-runtime + conflict_home="$RUNNER_TEMP/task-aware-conflict" mkdir -p -- "$conflict_home" printf '%s\n' 'default_permissions = ":workspace"' > "$conflict_home/config.toml" @@ -215,3 +586,129 @@ jobs: printf '%s\n' 'Malformed AGENTS.md markers were not rejected.' >&2 exit 1 fi + + literal_marker_home="$RUNNER_TEMP/task-aware-literal-markers" + mkdir -p -- "$literal_marker_home" + literal_marker_line='User text: and are examples.' + printf '%s\n' "$literal_marker_line" > "$literal_marker_home/AGENTS.md" + scripts/install-task-aware-agent.sh --codex-home "$literal_marker_home" + scripts/test-task-aware-agent.sh \ + --codex-home "$literal_marker_home" \ + --skip-runtime + if ! grep -Fqx -- "$literal_marker_line" "$literal_marker_home/AGENTS.md"; then + printf '%s\n' 'Inline marker examples were not preserved.' >&2 + exit 1 + fi + + preserve_home="$RUNNER_TEMP/task-aware-preserve" + mkdir -p -- "$preserve_home" + printf '%s\n' \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"' \ + 'unmanaged_setting = "retain-me"' > "$preserve_home/config.toml" + scripts/install-task-aware-agent.sh --codex-home "$preserve_home" + for preserved_line in \ + 'model = "gpt-6-astra"' \ + 'model_reasoning_effort = "max"' \ + 'approval_policy = "never"' \ + 'sandbox_mode = "workspace-write"' \ + 'unmanaged_setting = "retain-me"'; do + if ! grep -Fqx -- "$preserved_line" "$preserve_home/config.toml"; then + printf 'Default installation did not preserve: %s\n' "$preserved_line" >&2 + exit 1 + fi + done + + selection_conflict_home="$RUNNER_TEMP/task-aware-selection-conflict" + mkdir -p -- "$selection_conflict_home" + printf '%s\n' 'model = "preserve"' > "$selection_conflict_home/config.toml" + before_hash=$(sha256sum "$selection_conflict_home/config.toml" | awk '{print $1}') + if scripts/install-task-aware-agent.sh \ + --codex-home "$selection_conflict_home" \ + --set-sol-default \ + --set-astra-default; then + printf '%s\n' 'Conflicting parent model switches were accepted.' >&2 + exit 1 + fi + after_hash=$(sha256sum "$selection_conflict_home/config.toml" | awk '{print $1}') + if [[ "$before_hash" != "$after_hash" ]]; then + printf '%s\n' 'Conflicting parent model switches mutated config.toml.' >&2 + exit 1 + fi + for path in task-aware-backups agents rules AGENTS.md; do + if [[ -e "$selection_conflict_home/$path" ]]; then + printf 'Conflicting parent model switches created %s.\n' "$path" >&2 + exit 1 + fi + done + + for marker_case in duplicate reversed; do + marker_home="$RUNNER_TEMP/task-aware-markers-$marker_case" + mkdir -p -- "$marker_home" + if [[ "$marker_case" == duplicate ]]; then + printf '%s\n' \ + '' \ + '' \ + '' \ + '' > "$marker_home/AGENTS.md" + else + printf '%s\n' \ + '' \ + '' > "$marker_home/AGENTS.md" + fi + if scripts/install-task-aware-agent.sh --codex-home "$marker_home"; then + printf 'Marker case unexpectedly succeeded: %s\n' "$marker_case" >&2 + exit 1 + fi + done + + crlf_home="$RUNNER_TEMP/task-aware-crlf" + mkdir -p -- "$crlf_home" + awk '{ printf "%s\r\n", $0 }' config/AGENTS.task-aware.md > "$crlf_home/AGENTS.md" + scripts/install-task-aware-agent.sh --codex-home "$crlf_home" + scripts/test-task-aware-agent.sh --codex-home "$crlf_home" --skip-runtime + + for drift in policy role rule; do + drift_home="$RUNNER_TEMP/task-aware-drift-$drift" + scripts/install-task-aware-agent.sh --codex-home "$drift_home" + case "$drift" in + policy) + sed -i '//i# drift' "$drift_home/AGENTS.md" + ;; + role) + printf '%s\n' '# drift' >> "$drift_home/agents/luna-task.toml" + ;; + rule) + printf '%s\n' '# drift' >> "$drift_home/rules/task-aware-full-admin.rules" + ;; + esac + if scripts/test-task-aware-agent.sh --codex-home "$drift_home" --skip-runtime; then + printf 'Drift case unexpectedly passed: %s\n' "$drift" >&2 + exit 1 + fi + done + + - name: Reject oversized managed policy + run: | + budget_repo="$RUNNER_TEMP/task-aware-budget-repo" + budget_home="$RUNNER_TEMP/task-aware-budget-home" + mkdir -p -- "$budget_repo" + cp -R -- scripts config agents rules "$budget_repo/" + budget_source="$budget_repo/config/AGENTS.task-aware.md" + awk ' + // { + printf "" + } + { print } + ' "$budget_source" > "$budget_source.tmp" + mv -- "$budget_source.tmp" "$budget_source" + "$budget_repo/scripts/install-task-aware-agent.sh" --codex-home "$budget_home" + if "$budget_repo/scripts/test-task-aware-agent.sh" --codex-home "$budget_home" --skip-runtime > "$RUNNER_TEMP/budget-error.txt" 2>&1; then + printf '%s\n' 'An oversized managed policy passed validation.' >&2 + exit 1 + fi + grep -Fq 'Managed policy exceeds the 10 KiB maintenance budget' "$RUNNER_TEMP/budget-error.txt" diff --git a/CHANGELOG.md b/CHANGELOG.md index bb861ae..c1bc2f6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,22 +4,36 @@ ## [Unreleased] +### 追加 + +- 親モデルとして Astra(`gpt-6-astra`)に対応。Windows の `-SetAstraDefault` と Linux の `--set-astra-default` で Astra xhigh を明示的に選択できる。 +- 現行運用の管理者操作用 `sol_admin_max` と承認規則を配布。操作・対象・権限範囲の明示許可を必須とし、通常役からの自動昇格を禁止する。 +- 子の作業指示に有限の `NO_PROGRESS_LIMIT`、`HARD_DEADLINE`、`SAFE_CANCELLATION` と最終返却の形式を定義。 +- 複数工程の必須確認、任意の証拠、終了条件を先に固定し、検証範囲を追加できる条件を定義。 + ### 変更 -- 現行の価格差を踏まえ、Luna Low の D1 を固定入力、明示的な出力契約、客観的完了条件を持つ限定的な読み取り専用調査・検証まで拡張。 -- Terra Medium の D2 を、状態変更を伴う実装、tool-heavy な複数工程、通常判断が必要な調査・検証として明確化。 -- D1 に Luna Max、D2 に Terra Max、D3 に Sol Max の上位 variant を追加。 -- 中間の xhigh role は設けず、標準とMaxの二段階に統一。 -- 能力クラスを先に決め、同じクラス内で標準またはMax枠を選ぶ二段階ルーティングへ変更。 -- 価格低下を、Maxによる完全性向上または手戻り回避を選びやすくする根拠として反映。 -- 最小十分な役割を選びつつ、D0 の細分化、明白な D2/D3 の意図的な過小ルーティング、不要な microtask fan-out を禁止。 -- 価格低下後も、統合負荷と競合を抑えるため同時に開く子スレッドの上限を3つに維持。 +- 配布ポリシーを v3.1 に更新し、判断順、権限、有限期限、必須確認と終了条件を統合。管理ブロックの 10 KiB 上限を維持。 +- 親の導入オプションは GPT-6 Astra xhigh。子は D1 を GPT-6 Luna Low/Medium/High、D2 を GPT-6 Sol Medium/High、D3 を GPT-6 Astra High/xhigh、D4 を GPT-6 Astra xhigh/Max に更新。管理者専用役は Astra Max を維持。 +- 既存六役の名前とファイル名を維持。`_max` は上位枠の識別名とし、D1・D2 は High、D3 は xhigh、D4 は Max に対応。 +- D1 の軽い突き合わせ向けに `luna_task_medium` を追加。標準 Low・Medium と上位 High の選択基準を明確化。 +- D4 の所見統合用に `astra_architect` / `astra_architect_max` を追加。2件以上の独立した D3 所見を入力とし、読み取り専用・再委譲禁止・最大3子を維持。 +- Astra xhigh を親へ設定する `--set-astra-default` / `-SetAstraDefault` を追加。通常導入は既存の親モデル、effort、権限を維持。 +- 旧 Sol 既定値オプションを廃止。指定時は Astra オプションへの案内を表示し、ファイル変更前に停止。 +- 能力クラスを先に決め、同じクラス内で標準または上位枠を選ぶ二段階ルーティングへ変更。D1 の限定的な読み取り専用調査と、D2 の通常判断を伴う実装・調査の境界を明確化。 +- 委譲の条件を満たす場合の実行、承認済み作業の継続、変更範囲に応じた検証をポリシーへ追加。 +- D0 の細分化、明白な D2/D3 の意図的な過小ルーティング、不要な小タスクへの分割を禁止。子は最大3つ、再委譲禁止を維持。 ### 検証 -- Windows/Linux validator に、価格対応後の D1/D2 境界と過剰委譲防止規則の検査を追加。 -- Windows/Linux installer と validator を六 role の配置、model、effort、sandbox 検査へ拡張し、旧Luna/Terra High roleをbackup後に除去する移行を追加。 -- release 前の live probe を、六 role の model/effort 確認と D0 から D3 までの標準・Maxルーティング確認へ拡張。 +- マーカーの完全一致・順序・一意性、配布ポリシー・十役・規則のバイト一致を検査。CRLF、ブロック外の記述、不正なマーカー、期限や終了条件の欠落、差異、10 KiB 超過を回帰検証する。 + +- CI の固定バージョンを、GPT-6 ファミリー設定を検証する Codex CLI 0.156.1 へ更新。 +- Windows/Linux validator を十役の model、effort、sandbox、承認ポリシーと GPT-6 ファミリーの割り当ての検査へ更新。 +- Windows/Linux installer に、旧 Luna/Terra High role を backup 後に除去する移行を追加。 +- CI に親の設定維持、Astra opt-in、廃止した Sol オプションの停止と再導入の検査を追加。 +- 全 Astra 構成からの再導入で D1/D2 のモデルを更新し、既存の親・権限・独自役割・バックアップを保持する移行と、誤ったモデル割り当ての拒否を検証。 +- リリース前の実動作確認を、九役の model/effort と D0 から D4 の役割選択へ拡張。静的検査とモデル起動の検証を区別。 ## [0.1.0] - 2026-07-26 diff --git a/README.md b/README.md index ddcbe89..dabb745 100644 --- a/README.md +++ b/README.md @@ -6,44 +6,63 @@ Codex の親エージェントがタスクを難易度別に分類し、必要な場合だけ役割別のカスタムエージェントへ委譲するための設定一式です。 OpenAI の公式製品ではなく、Codex の公開仕様に基づくコミュニティプロジェクトです。 -この構成が想定する親は、`gpt-5.6-sol` をクライアントの Ultra モードで動かす **Sol Ultra** です。 -親は難易度判定、タスクの分割、子の選択、結果の統合を担当します。 -子には Luna Low/Max、Terra Medium/Max、Sol High/Max を使い分けます。 +この構成では、親に **GPT-6 Astra** を使い、難易度の判定、タスクの分割、子の選択、結果の統合を担当させます。 +親を切り替える導入オプションは `gpt-6-astra` と `xhigh` を設定します。 +子は D1 に **GPT-6 Luna**(Low・Medium・High)、D2 に **GPT-6 Sol**(Medium/High)、D3 に **GPT-6 Astra**(High/xhigh)、D4 に **GPT-6 Astra**(xhigh/Max)を使います。 -これにより、すべての子が親の高い推論労力を継承して消費量が膨らむことを避けつつ、メインスレッドへ途中経過が流れ込む量を抑えます。 -現行の価格差は、固定契約の読み取り専用作業を Luna へ寄せ、各難度内で必要な場合に高い推論労力を選ぶ根拠にします。 -安価であることだけを理由に子を細分化はしません。 -各 subagent は独自に token と調整時間を使うため、D0 の直接処理と最大3子の上限は維持します。 +子のモデルと推論労力を役割ごとに指定することで、親の設定をすべての子が継承することを避けます。 +ただし、子もそれぞれトークンと調整時間を使うため、分割する利点がある作業だけを委譲します。 +単純な D0 は親が処理し、同時に開く子は最大3つとします。 リポジトリを取得しただけでは Codex の動作は変わりません。 導入スクリプトを実行し、新しいタスクで設定を読み込む必要があります。 ## 想定する実行構成 -Sol はモデル、Ultra は対応するモデルと環境で最大推論と能動的な委譲を利用する実行モードです。 -現行 Codex は、対応モデルの `model_reasoning_effort = "ultra"` も受け付けます。 -このリポジトリでは、親の統合と委譲に Ultra を残し、子の上限を Max にします。 - -対応するアカウントとクライアントで Sol Ultra を選ぶと、親は分割可能な作業をサブエージェントへ能動的に委譲します。 -このリポジトリは、その委譲に D0 から D4 までの能力分類と、同じ難度内で推論労力を選ぶ六つの子エージェントを追加します。 +親の基準は Astra xhigh です。既に選択している親のモデルと推論労力は、標準導入では変更しません。 +Ultra を利用できる環境では、並列に分割できる大きな作業に Astra Ultra を選べます。 +このポリシーは Ultra の有無にかかわらず、委譲の条件を満たす作業を親へ明示します。 | 難易度 | 対象 | 実行役 | モデルと推論労力 | sandbox | | --- | --- | --- | --- | --- | -| D0 | 単純で明確な1工程 | 親が直接処理 | Sol Ultra | 親の設定 | -| D1 標準 | 小さく均質で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Luna Low | read-only | -| D1 Max枠 | D1 のまま、異種入力、密な突合、coverage 重視、多数の edge case がある作業 | `luna_task_max` | Luna Max | read-only | -| D2 標準 | 状態変更を伴う実装、tool-heavy な複数工程、通常判断が必要な調査・検証 | `terra_worker` | Terra Medium | 親から継承 | -| D2 Max枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Terra Max | 親から継承 | -| D3 標準 | 一つの難しい判断、曖昧性、高リスク、複数領域、設計判断 | `sol_specialist` | Sol High | read-only | -| D3 Max枠 | 不確実性と結果の重大性がともに高く、証拠競合、不可逆設計、security、敵対的 edge case を含む作業 | `sol_specialist_max` | Sol Max | read-only | -| D4 | 独立した D3 タスクが複数 | 親が分割して統合 | Sol Ultra | 親と選択した子の設定 | - -能力クラスを D1/D2/D3 から先に決め、その後で標準またはMax枠を選びます。 -Max枠を選んでも権限や能力境界は広がりません。 -Luna の二役と Sol の二役は読み取り専用で、書き込みを伴う通常実装は、親の権限を継承する Terra の二役か親が担当します。 -Ultra は親だけに残し、複数の判断をまたぐ推論と最終統合を親が担当します。 -上位枠は三モデルとも Max に統一します。 -`xhigh` は標準とMaxの間に独立した能力境界を作らないため、別roleにはしません。 +| D0 | 単純で明確な1工程 | 親が直接処理 | Astra xhigh(必要に応じて変更) | 親の設定 | +| D1 標準 Low | 少量・同形式で、固定入力と客観的完了条件がある読み取り専用作業 | `luna_task` | Luna Low | read-only | +| D1 標準 Medium | 複数ファイルや異なる形式を扱い、客観的な条件に沿った軽い突き合わせが必要な作業 | `luna_task_medium` | Luna Medium | read-only | +| D1 上位枠 | D1 のまま、異種入力、密な突合、網羅性の確認、多数の境界条件がある作業 | `luna_task_max` | Luna High | read-only | +| D2 標準 | 状態変更を伴う実装、ツールを使う複数工程、通常判断が必要な調査・検証 | `terra_worker` | Sol Medium | 親から継承 | +| D2 上位枠 | D2 のまま、制約の結合、長い検証経路、難しいデバッグ、手戻りコストが大きい作業 | `terra_worker_max` | Sol High | 親から継承 | +| D3 標準 | 一つの難しい判断、曖昧性、高リスク、複数領域、設計判断 | `sol_specialist` | Astra High | read-only | +| D3 上位枠 | 不確実性と結果の重大性がともに高く、証拠の競合、不可逆な設計、セキュリティ上の重大性などを含む作業 | `sol_specialist_max` | Astra xhigh | read-only | +| D4 標準 | 2件以上の独立した D3 作業の所見を、制約や依存関係を踏まえて全体の判断へまとめる作業 | `astra_architect` | Astra xhigh | read-only | +| D4 上位枠 | D3 作業間で証拠や推奨が競合し、全体の判断が不可逆な選択や重大な影響を伴う作業 | `astra_architect_max` | Astra Max | read-only | +| 管理者操作専用 | 操作・対象・昇格方法を明示的に許可された単一作業 | `sol_admin_max` | Astra Max | danger-full-access、on-request | + +能力クラスを D1/D2/D3/D4 から先に決め、その後で必要な推論労力を選びます。 +上位枠を選んでも権限や能力境界は広がりません。 +D1・D3・D4 の子は読み取り専用です。書き込みを伴う通常実装は、親の権限を継承する D2 の子か親が担当します。 +子には再委譲させず、最終統合は親が担当します。 + +既存六つの役割名とファイル名は互換性のため維持し、D1 Medium を1役、D4 を2役追加して計九役にしています。 +通常の九役に、管理者操作専用の `sol_admin_max` を加えた計十役を配布します。 +D1 の三役は `gpt-6-luna`、D2 の二役は `gpt-6-sol`、D3・D4 と管理者役は `gpt-6-astra` です。 +`terra_worker` は GPT-6 Sol、`sol_specialist` は GPT-6 Astra を使う互換名です。管理者役の名前も維持します。 +`_max` は上位枠の互換名です。D1・D2 の `_max` は High、D3 の `_max` は xhigh、D4 の `_max` は Max に対応します。 +役割名からモデルや effort を推測せず、上の表と TOML の設定値を確認してください。 + +この割り当ては、定型処理に Luna、通常実装に Sol、重大な判断と統合に Astra を使うプロジェクトの選択です。 +既存の能力・権限境界と各役割の effort を維持し、モデルの移行だけを比較できるようにしています。 +モデル間や Low・Medium・High・xhigh・Max の品質・消費量を比較したベンチマーク結果ではありません。 +モデルの用途と利用可能な設定は [Codex のモデル案内](https://learn.chatgpt.com/docs/models) を参照してください。 + +D1 の標準枠では、少量・同形式の入力なら Low、複数ファイルや異なる形式を軽く突き合わせるなら Medium を選びます。 +網羅性の確認や密な突き合わせ、多数の例外処理が必要なら High を使います。 +上位の条件が明らかな場合は、Low や Medium で失敗するのを待たず、適切な役割へ直接渡します。 +いずれも客観的な完了条件を持つ読み取り専用作業に限り、実装や設計上の判断は D2 以上へ分けます。 + +D4 では、親が独立した D3 作業を分割し、その所見が揃ってから、必要に応じて D4 の子へ全体の判断を依頼します。 +子へ渡す資料には、少なくとも2件の独立した D3 所見と、その根拠を含めます。 +D4 の子は所見の統合を担当し、再委譲や状態変更は行いません。最終判断と実行は親が担当します。 +D4 も同時に開く最大3子に数えるため、入力が揃い、空き枠ができてから起動します。 ## ルーティングの仕組み @@ -57,39 +76,60 @@ Ultra は親だけに残し、複数の判断をまたぐ推論と最終統合 3. 親のコンテキスト消費か経過時間を減らせる見込みがある。 4. 委譲の調整コストが、親による直接処理より小さい。 -条件を満たした後、親は能力クラスを選び、そのクラス内で必要十分な推論労力を選びます。 -標準 effort が基本ですが、完全性の向上または手戻りの回避が見込める場合は、価格低下を踏まえてMax枠を積極的に選べます。 -Max枠の task packet には、標準 effort では誤りや手戻りが増える具体的な理由を含めます。 -原子的な D0 を Luna が安価であるという理由だけで分割しません。 +親は各作業の能力クラスを分類し、委譲条件を満たした場合に、そのクラス内で必要十分な役割と推論労力を選びます。 +分類だけで子の起動を認めず、条件を満たさない作業は権限と能力の範囲内で親が処理します。 +標準 effort を基本とし、追加の時間とトークンを使っても、網羅性の向上や手戻りの回避が見込める場合に上位枠を選びます。 +D1 Medium の task packet には、Low では不足する突き合わせの内容を記載します。 +上位枠を選ぶ場合は、各クラスの標準 effort では不足する具体的な理由も含めます。 +原子的な D0 を、推論労力を下げるためだけに分割しません。 同じ入力と完了条件を共有する小さな作業は、分離によって待ち時間、コンテキスト分離、証拠の独立性が改善しない限り、一つの task packet にまとめます。 子は別の子を起動しません。 -`max_concurrent_threads_per_session = 3` とポリシー上の上限により、親から同時に開く子スレッドは最大3つです。 +配布設定は `max_concurrent_threads_per_session = 3` とし、ポリシーも runtime が設定した上限を超える起動を禁止します。 +この配布設定を使う場合、親から同時に開ける子スレッドは最大3つです。 同じファイルや状態を更新するエージェントは1つに限定します。 子からの再委譲は、agent TOML と `AGENTS.md` の指示で禁止します。 -旧 `agents.max_depth` は Codex V2 で無視されるため、実効的な強制境界としては使用しません。 +導入時に除去する旧 `agents.max_depth` は、実効的な強制境界として使用しません。 `NEEDS_ESCALATION` は Codex ランタイムの自動判定ではなく、子が能力不足の根拠を親へ返すための応答規約です。 合理的に選んだ下位役割が返した場合だけ、親はその根拠を確認し、必要な上位役割へ再委譲します。 -最初から D2 または D3 と明らかな作業を、昇格結果を得るためだけに Luna へ渡しません。 +最初から D2 または D3 と明らかな作業を、昇格結果を得るためだけに D1 の役割へ渡しません。 -子を起動するときは、`spawn_agent` の `agent_type` に `luna_task`、`luna_task_max`、`terra_worker`、`terra_worker_max`、`sol_specialist`、`sol_specialist_max` のいずれかを明示します。 +子を起動するときは、`spawn_agent` の `agent_type` に `luna_task`、`luna_task_medium`、`luna_task_max`、`terra_worker`、`terra_worker_max`、`sol_specialist`、`sol_specialist_max`、`astra_architect`、`astra_architect_max` のいずれかを明示します。 `task_name` は子タスクの表示名とパスを付ける項目であり、custom agent の選択には使いません。 `task_name = "luna_task"` だけを指定すると、子が親のモデルと推論労力を継承するため、想定したコスト制御になりません。 -D1 から D3 までの委譲では、標準とMax枠のどちらでも `agent_type` を必須とし、まず必ず引数付きで起動します。 +D1 から D4 までの委譲では、標準と上位枠のどちらでも `agent_type` を必須とし、まず必ず引数付きで起動します。 tool が `agent_type` または custom agent を明示的に拒否した場合だけ、既定の子を起動せず、親で処理して不一致を報告します。 各 spawn は `fork_turns = "none"` を指定し、親の全会話履歴ではなく task packet だけを子へ渡します。 -task packet 自体にも再委譲禁止を明記します。 +task packet には権限と変更を許す範囲を含め、再委譲禁止も明記します。 + +### 待機と作業の終了 + +各子には、進捗がない時間の上限 `NO_PROGRESS_LIMIT`、終了期限 `HARD_DEADLINE`、安全に中断できる条件 `SAFE_CANCELLATION` を発行時に渡します。 +既定値は標準役(D1 Medium を含む)が10分/30分、名前が `_max` で終わる上位役と管理者役が20分/60分です。 +この期限区分は役割の名前に対応し、実際の推論労力が Max かどうかでは決まりません。 +進捗がない時間の上限に達したら、親は状況と得られた結果の返却を一度求め、最大2分だけ追加で待ちます。 +返却がない場合や終了期限に達した場合は `STALLED` として扱い、変更中の子を安全条件なしに引き取ったり、別の子に同じ書き込みを重複させたりしません。 + +複数工程の作業では、必須確認 `REQUIRED_ACCEPTANCE_CHECKS`、任意の証拠 `OPTIONAL_EVIDENCE`、終了条件 `STOP_CONDITION` を先に固定します。 +必須確認と成果物がそろえば終了し、追加の検証は失敗、新たな対象内リスク、利用者による範囲拡張などの根拠がある場合に限ります。 +子の最終返却には `STATUS`、`RESULT`、`EVIDENCE`、`OPEN_ISSUES` を含めます。 +D4 の入力所見が不足する場合は `STATUS: NEEDS_INPUT` と不足する証拠を返し、親が次の対応を判断します。 + +Astra 向けの実行規則として、承認済みの作業を実装と必要な検証まで進めることも明記します。 +通常の可逆的な選択は会話の文脈から判断し、結果や権限に影響する不足情報がある場合に質問します。 +回答待ちの間も独立した作業は進めます。検証は変更範囲に合わせ、必要な確認が通った後は、追加変更や未解決の問題がある場合に再実行します。 +この調整は [Astra の公式ガイド](https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-astra) を参考にし、このリポジトリの最大3子・一段階委譲へ限定しています。 ## 前提条件 - Windows では PowerShell 7 以降を使用できること。 -- Linux では Bash、`awk`、`grep` を使用できること。 +- Linux では Bash、`awk`、`grep`、`sed`、`cmp` を使用できること。 - カスタムエージェントと subagent workflow に対応した現行 Codex を使用していること。 -- この公開候補の検証基準である Codex CLI 0.145.0 以降を使用すること。 -- 使用するアカウントで `gpt-5.6-luna`、`gpt-5.6-terra`、`gpt-5.6-sol` を利用できること。 -- Sol Ultra を使う場合は、対応するアカウントとクライアントで Ultra が有効であること。 +- CI の設定互換性検証基準は Codex CLI 0.156.1。各モデルの利用には、アカウントとクライアントのモデル選択欄で対応を確認すること。 +- 使用するアカウントで `gpt-6-astra`、`gpt-6-sol`、`gpt-6-luna` を利用できること。 +- Astra Ultra を使う場合は、対応するアカウントとクライアントで Ultra が有効であること。 - コマンドをこのリポジトリのルートで実行すること。 導入前に、Codex の起動と設定読込が正常であることを確認してください。 @@ -99,10 +139,13 @@ codex --version codex doctor --summary --no-color --ascii ``` -モデルと推論の選択欄では、三つの子モデル、High/Max を含む必要な effort、親に使う Sol Ultra が表示されることも確認します。 - -Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 から D4 までのルーティング規則を使えます。 -ただし、それはこのリポジトリが想定する Sol Ultra と同じ実行構成ではありません。 +モデルと推論の選択欄では、上の表にあるモデルと各役割の effort が利用できることも確認します。 +2026-09-24 のローカルモデルカタログ(client version 0.155.0)では、Astra・Sol の `low`、`medium`、`high`、`xhigh`、`max`、`ultra` と、Luna の `low`、`medium`、`high`、`xhigh`、`max` を確認しました。 +Luna は Ultra に対応しません。子はすべて単独で処理し、Ultra は設定しません。 +設定の厳密な読込と変更した五役の実起動は Codex CLI 0.156.1 で確認しました。 +検証の条件と限界は [GPT-6 ファミリー移行の検証記録](docs/gpt6-family-validation.md) を参照してください。 +旧 CLI 0.154.0 では、この環境で GPT-6 Luna/Sol の実起動が拒否されました。モデルが表示されない場合や未対応エラーが出る場合は、クライアントを更新し、利用アカウントの対応も確認してください。 +カタログへの掲載や設定の読込成功だけでは、実際のモデル呼び出し成功は保証されません。 ## 導入で変更するもの @@ -114,11 +157,16 @@ Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 か | `config.toml` | `[agents] enabled = true`、`max_concurrent_threads_per_session = 3` を設定し、旧 key を除去 | | `AGENTS.md` | マーカーで囲んだ task-aware delegation policy を追加または更新 | | `agents/luna-task.toml` | Luna Low の読み取り専用エージェントを配置 | -| `agents/luna-task-max.toml` | Luna Max の読み取り専用エージェントを配置 | -| `agents/terra-worker.toml` | Terra Medium の作業エージェントを配置 | -| `agents/terra-worker-max.toml` | Terra Max の作業エージェントを配置 | -| `agents/sol-specialist.toml` | Sol High の読み取り専用エージェントを配置 | -| `agents/sol-specialist-max.toml` | Sol Max の読み取り専用エージェントを配置 | +| `agents/luna-task-medium.toml` | Luna Medium の読み取り専用エージェントを配置 | +| `agents/luna-task-max.toml` | Luna High の読み取り専用エージェントを配置 | +| `agents/terra-worker.toml` | Sol Medium の作業エージェントを配置 | +| `agents/terra-worker-max.toml` | Sol High の作業エージェントを配置 | +| `agents/sol-specialist.toml` | Astra High の読み取り専用エージェントを配置 | +| `agents/sol-specialist-max.toml` | Astra xhigh の読み取り専用エージェントを配置(役割名は互換性維持) | +| `agents/astra-architect.toml` | Astra xhigh の読み取り専用D4エージェントを配置 | +| `agents/astra-architect-max.toml` | Astra Max の読み取り専用D4エージェントを配置 | +| `agents/sol-admin-max.toml` | 明示許可された管理者操作専用の Astra Max エージェントを配置 | +| `rules/task-aware-full-admin.rules` | 管理者昇格の直接コマンドを承認対象にする規則を配置 | 既存ファイルは、変更前に `$CODEX_HOME/task-aware-backups//` へ退避します。 同名のカスタムエージェントファイルは上書きされます。 @@ -136,16 +184,16 @@ Ultra を利用できない環境でも、Sol xhigh を親にして同じ D0 か pwsh -File .\scripts\Install-TaskAwareAgent.ps1 ``` -親の既定値を Sol xhigh にする場合は、`-SetSolDefault` を指定します。 +親の既定値を Astra xhigh にする場合は、`-SetAstraDefault` を指定します。 ```powershell -pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault +pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetAstraDefault ``` フルアクセスも設定する場合は、`-EnableFullAccess` を追加します。 ```powershell -pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetSolDefault -EnableFullAccess +pwsh -File .\scripts\Install-TaskAwareAgent.ps1 -SetAstraDefault -EnableFullAccess ``` 別の Codex 環境へ試験導入する場合は `-CodexHome `、変更予定だけを確認する場合は `-WhatIf` を指定できます。 @@ -160,16 +208,16 @@ PowerShell 版も、`config.toml` がない空の `CODEX_HOME` を初期化で ./scripts/install-task-aware-agent.sh ``` -親の既定値を Sol xhigh にする場合は、`--set-sol-default` を指定します。 +親の既定値を Astra xhigh にする場合は、`--set-astra-default` を指定します。 ```bash -./scripts/install-task-aware-agent.sh --set-sol-default +./scripts/install-task-aware-agent.sh --set-astra-default ``` フルアクセスも設定する場合は、`--enable-full-access` を追加します。 ```bash -./scripts/install-task-aware-agent.sh --set-sol-default --enable-full-access +./scripts/install-task-aware-agent.sh --set-astra-default --enable-full-access ``` 別の Codex 環境へ導入する場合は、`CODEX_HOME` 環境変数または `--codex-home ` を指定できます。 @@ -177,19 +225,32 @@ Linux 版は、空の `CODEX_HOME` に必要なファイルを新規作成でき Windows 側の checkout を WSL から使う場合は、`.sh` の改行が LF であることを確認してください。 CRLF のまま実行すると、shebang の `bash` を解決できず起動に失敗します。 -### Sol Ultra の選択 +### 親の effort と旧オプション + +`-SetAstraDefault` と `--set-astra-default` は `gpt-6-astra` と `xhigh` を設定します。 +既に親へ Ultra などを選択していて、その effort を保つ場合は標準導入を使ってください。 +Ultra を新たに使う場合は、対応クライアントのモデル選択で Astra と Ultra を選びます。 -`-SetSolDefault` と `--set-sol-default` が設定する親の既定値は、`gpt-5.6-sol` と `xhigh` です。 -どちらのオプションだけでも Sol Ultra にはなりません。 +GPT-5.6 Sol 用だった旧 `-SetSolDefault` と `--set-sol-default` は廃止したままです。 +指定すると、ファイルを変更する前に Astra オプションへの移行案内を表示して停止します。 +GPT-6 Sol を親に使う場合は、Codex のモデル選択で指定して標準導入を使います。 +親も Astra に切り替える場合は `-SetAstraDefault` または `--set-astra-default` を使ってください。 -想定構成どおりに使う場合は、導入後に対応クライアントのモデル選択で Sol と Ultra を選んでください。 -インストーラーは、アカウントやクライアントごとの Ultra 対応を暗黙に仮定しないため、`model_reasoning_effort = "ultra"` を自動では書き込みません。 -対応を確認できた環境では、クライアントの選択または明示的な設定で親を Ultra にしてください。 +### 管理者操作用の役割 + +通常の九役は `approval_policy = "never"` とし、管理者昇格を実行したり要求したりしない指示を持ちます。 +`sol_admin_max` だけが `approval_policy = "on-request"` を使い、利用者が操作・対象・権限範囲を明示的に許可した単一作業を担当します。 +親は、通常権限で目的を満たせないこと、承認規則が該当する直接の昇格コマンドに `prompt` を返すこと、復旧方法と検証項目があることを確認してから発行します。 +不明点や条件の不足があれば、権限を自動で広げず `NEEDS_ESCALATION` を返します。 + +`danger-full-access` は Codex のコマンド sandbox を外す設定であり、OS の root・管理者権限を与える設定ではありません。 +認証には利用者に見える OS のプロンプトやターミナルを使い、チャットでパスワードを受け取ったり `sudo -S` を使ったりしません。 ### フルアクセスの影響 `-EnableFullAccess` と `--enable-full-access` は、`approval_policy = "never"` と `sandbox_mode = "danger-full-access"` をグローバル設定へ書き込みます。 -この指定は、親だけでなく sandbox を親から継承する `terra_worker` と `terra_worker_max` にも影響します。 +この指定は、親と sandbox を親から継承する `terra_worker`/`terra_worker_max` に影響します。 +通常の九役では承認要求を無効とする `approval_policy = "never"` と、管理者昇格を禁止する指示を保ちます。 同じ `$CODEX_HOME` を使うほかのプロジェクトにも適用されるため、信頼できる環境でのみ使用してください。 既存の `config.toml` に `default_permissions` がある場合、インストーラーは `-EnableFullAccess` または `--enable-full-access` をエラーで停止します。 @@ -217,22 +278,30 @@ PowerShell 版も対象の `CODEX_HOME` を明示し、`codex --strict-config do 認証情報のない clean home や CI では `--config-only-runtime` を加えます。 Linux 版は対象の `CODEX_HOME` を明示し、`codex --strict-config doctor --summary` を実行します。 -どちらの検証スクリプトも、配置したファイル、主要な設定値、六つの子のモデル ID、推論労力、宣言した sandbox を静的に確認します。 -`codex` コマンドが見つかる場合は、続けて `codex doctor` を実行します。 +どちらの検証スクリプトも、マーカーが正しい順序で一組だけ存在すること、配置されたポリシー・十の役割ファイル・管理者操作用の規則が配布元と一致することを確認します。 +役割ファイルと規則はバイト単位で照合し、主要な設定値、モデルID、推論労力、sandbox、承認ポリシーも静的に確認します。 +ポリシーの検査は管理対象のブロック内に限定し、ブロック外に同じ文があっても代用しません。 +`codex` コマンドが見つかる場合は、続けて規則の `prompt` 評価と `codex doctor` を実行します。 +規則の検査は昇格コマンドを文字列として評価するだけで、管理者操作は実行しません。 この検証は、runtime が子へ適用した実効 sandbox、モデルの利用権限、実際の子の起動、D0 から D4 までの分類結果までは確認しません。 導入後は Codex を再起動するか、新しいタスクを開始し、次の手順で実動作も確認してください。 +D1 以降の起動確認では、独立した成果など四つの委譲条件を満たす作業を用意します。 -1. 親のモデルと推論の選択欄が Sol Ultra になっていることを確認する。 +1. 親のモデルが Astra、effort が選択した値(導入オプションなら xhigh)になっていることを確認する。 2. 一つのファイルから既知の文字列を読むだけの D0 を依頼し、子を起動しないことを確認する。 -3. 小さく均質で固定契約を持つ読み取り専用 D1 を依頼し、`luna_task` を確認する。 -4. 異種入力の密な突合と coverage 判定を伴うが、客観的に完了判定できる D1 を依頼し、`luna_task_max` を確認する。 -5. 専用の空ディレクトリに一つのファイルを作成して検証する D2 を依頼し、`terra_worker` を確認する。 -6. 複数ファイルの結合制約と長い検証経路を持つ D2 を依頼し、`terra_worker_max` を確認する。 -7. 一つの明確な設計トレードオフを判断する D3 を依頼し、`sol_specialist` を確認する。 -8. 証拠が競合し、不可逆性または security 上の重大性も高い D3 を依頼し、`sol_specialist_max` を確認する。 -9. 親が D0 では spawn せず、D1 から D3 では対応する標準またはMax枠の `agent_type` を渡すことを確認する。 -10. 各子の詳細を開き、Luna Low/Max、Terra Medium/Max、Sol High/Max が実際に適用されていることを確認する。 +3. 少量・同形式の固定入力を持つ読み取り専用 D1 を依頼し、`luna_task` が Low で動くことを確認する。 +4. 複数ファイルや異なる形式の軽い突き合わせを行う D1 を依頼し、`luna_task_medium` が Medium で動くことを確認する。 +5. 密な突き合わせや網羅性の確認が必要な D1 を依頼し、`luna_task_max` が High で動くことを確認する。 +6. 専用の空ディレクトリに一つのファイルを作成して検証する D2 を依頼し、`terra_worker` を確認する。 +7. 複数ファイルの結合制約と長い検証経路を持つ D2 を依頼し、`terra_worker_max` を確認する。 +8. 一つの明確な設計トレードオフを判断する D3 を依頼し、`sol_specialist` が High で動くことを確認する。 +9. 証拠の競合と判断の重大性がともに高い D3 を依頼し、`sol_specialist_max` が xhigh で動くことを確認する。 +10. 2件以上の独立した D3 所見をまとめる D4 を依頼し、`astra_architect` が xhigh で動くことを確認する。 +11. D3 所見間で重大な推奨や証拠が競合する D4 を依頼し、`astra_architect_max` が Max で動くことを確認する。 +12. 親が D0 では spawn せず、D1 から D4 では対応する `agent_type` を渡すこと、子が再委譲せず最大3子を守ることを確認する。 +13. 各子の詳細で、model が D1 は `gpt-6-luna`、D2 は `gpt-6-sol`、D3・D4 は `gpt-6-astra` となり、effort と実効権限が上の表と一致することを個別に確認する。 +14. 管理者操作の明示許可がなければ、通常役が昇格を試みず `sol_admin_max` も発行されないことを確認する。実際の管理者操作は、操作固有の許可を得た別の作業で検証する。 `AGENTS.md` の指示チェーンは新しい実行の開始時に構築されるため、導入前から開いているタスクでは確認できません。 @@ -246,18 +315,26 @@ Linux 版は対象の `CODEX_HOME` を明示し、`codex --strict-config doctor 導入後に `config.toml` や `AGENTS.md` を変更している場合は、バックアップをそのまま上書きせず、差分を確認して task-aware 関連の設定だけを手動で統合してください。 導入前に存在しなかったファイルはバックアップへ含まれません。 -その場合は、追加された六つのエージェントファイルと、`AGENTS.md` の `BEGIN CODEX TASK-AWARE AGENT` から `END CODEX TASK-AWARE AGENT` までのブロックを手動で取り除く必要があります。 +その場合は、十のエージェントファイルと `rules/task-aware-full-admin.rules` のうち今回新規に追加されたもの、および `AGENTS.md` の `BEGIN CODEX TASK-AWARE AGENT` から `END CODEX TASK-AWARE AGENT` までのブロックを手動で取り除きます。 +既存だった役割や規則は削除せず、戻したい時点のバックアップを使います。 ## ファイル構成 - `config/AGENTS.task-aware.md`:グローバル指示へ追加するルーティング規則 - `config/config.task-aware.toml`:`config.toml` へ統合する設定例 -- `agents/*.toml`:Luna、Terra、Sol の役割別エージェント +- `agents/*.toml`:GPT-6 Luna・Sol・Astra の役割別エージェント(旧役割名を維持) +- `rules/full-admin.rules`:管理者昇格の承認規則。導入先は `rules/task-aware-full-admin.rules` - `scripts/Install-TaskAwareAgent.ps1`:バックアップ付き導入スクリプト - `scripts/Test-TaskAwareAgent.ps1`:配置と Codex 設定の検証スクリプト - `scripts/install-task-aware-agent.sh`:Linux 向けのバックアップ付き導入スクリプト - `scripts/test-task-aware-agent.sh`:Linux 向けの配置と Codex 設定の検証スクリプト +## 指示の保守 + +配布する管理ブロックの編集元は `config/AGENTS.task-aware.md` です。同じ条件は既存の規則へ統合し、経緯、例、詳細な手順はREADMEや作業記録に置きます。導入済みのコピーだけを編集すると、次の導入で配布元の内容へ戻ります。 + +管理ブロックはUTF-8で10 KiB以下を保守上の上限とし、検証スクリプトとCIで超過を検出します。これは本プロジェクトの分量管理で、Codex本体の読み込み上限ではありません。利用者が管理ブロック外に書いた指示には、この上限を適用しません。 + ## リリース `v*` tag を push すると、GitHub Actions が Windows/Linux の clean-home round trip を再実行し、次の成果物を GitHub Release に作成します。 @@ -275,17 +352,19 @@ Apache License 2.0 です。SPDX identifier は `Apache-2.0` です。全文は ## 設計上の注意 - サブエージェントはメインスレッドのノイズを減らしますが、総トークン量や待ち時間が必ず減るわけではありません。 -- 価格低下は同じ難度内でMax枠を選ぶ閾値を下げますが、必要な能力、書き込み、曖昧さ、リスクより優先しません。 -- Max は応答時間と token 使用量を増やすため、標準 effort で十分な作業には使いません。 -- Ultra 自体も分割可能な作業へサブエージェントを使うため、独立した成果がある作業だけに委譲を絞ります。 +- 上位枠は必要な能力、書き込み、曖昧さ、リスクを先に判定して選びます。価格だけを理由に選びません。 +- High や Max は追加の推論時間とトークンを使うため、各クラスの標準 effort で十分な作業には使いません。 +- Ultra を選んだ場合も、独立した成果があり、調整コストを上回る利点がある作業を委譲します。 - モデルが利用できることと、ルーティングが適切であることは静的テストだけでは保証できません。 - ルーティングは親の判断に依存するため、D0 から D4 までの境界は決定的ではありません。 -- Codex Desktop や CLI の実効 permission profile が親の権限を子へ強制する runtime では、role TOML の `sandbox_mode` より親の実効権限が優先される場合があります。Luna と Sol の developer instructions も書き込みを禁止しますが、これは OS sandbox と同じ強制境界ではありません。 +- Codex Desktop や CLI の実効 permission profile が親の権限を子へ強制する runtime では、role TOML の `sandbox_mode` より親の実効権限が優先される場合があります。D1・D3・D4 の developer instructions も書き込みを禁止しますが、これは OS sandbox と同じ強制境界ではありません。 + +- 管理者操作用のコマンド規則は、直接のコマンド入口に対する承認制御です。shell wrapper や間接実行のすべてを捕捉する仕組みではありません。強い隔離が必要な場合は、別の OS アカウント、コンテナ、VM、限定した権限仲介などを使います。 ## 参考資料 - [Models](https://learn.chatgpt.com/docs/models) -- [Codex rate card](https://help.openai.com/en/articles/20001106-codex-rate-card) +- [GPT-6 Astra guide](https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-astra) - [Subagents](https://learn.chatgpt.com/docs/agent-configuration/subagents) - [Configuration Reference](https://learn.chatgpt.com/docs/config-file/config-reference) - [Configuration Schema](https://developers.openai.com/codex/config-schema.json) diff --git a/RELEASE_CHECKLIST.md b/RELEASE_CHECKLIST.md index f9641fd..85c8165 100644 --- a/RELEASE_CHECKLIST.md +++ b/RELEASE_CHECKLIST.md @@ -4,16 +4,26 @@ - [ ] `main` が最新の `origin/main` と一致し、working tree に意図しない変更がない。 - [ ] [Codex changelog](https://learn.chatgpt.com/docs/changelog) で最新版を確認し、`.github/workflows/ci.yml` と README の検証基準を更新する。 -- [ ] [Codex rate card](https://help.openai.com/en/articles/20001106-codex-rate-card) でモデル間の価格差を確認し、ルーティング根拠が現行レートと矛盾しない。 +- [ ] [Codex Models](https://learn.chatgpt.com/docs/models) と [Astra guide](https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-astra) で、モデル、effort、委譲方針の互換性を確認する。 - [ ] `CHANGELOG.md` の version、日付、release link を確定する。 - [ ] `LICENSE`、`SECURITY.md`、`CONTRIBUTING.md` が release archive に含まれる。 - [ ] GitHub Actions の Windows/Linux job が成功する。 -- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1/D2/D3 の標準・Max条件を実行し、Luna Low/Max、Terra Medium/Max、Sol High/Max の model と reasoning effort を live probe する。 -- [ ] Max枠が能力境界や sandbox を広げず、task packet に昇格理由が含まれることを確認する。 +- [ ] 認証済みの新しい Codex task で D0 の no-spawn、D1 の Low/Medium/High 条件と D2/D3/D4 の標準・上位条件を実行する。子の実起動を通じて、model が D1 は `gpt-6-luna`、D2 は `gpt-6-sol`、D3・D4 は `gpt-6-astra`、effort が D1 の Low/Medium/High、D2 の Medium/High、D3 の High/xhigh、D4 の xhigh/Max になっていることを確認する。 +- [ ] D4 の入力に2件以上の独立した D3 所見が含まれ、D4 も読み取り専用・再委譲禁止・同時に最大3子を守ることを確認する。 +- [ ] 親の Astra xhigh 設定、標準導入での既存親設定の維持、廃止した Sol オプションの案内と変更前の停止を確認する。 +- [ ] 全 Astra 構成から再導入し、D1/D2 のモデル更新と、親・effort・権限・独自役割・バックアップの保持を確認する。 +- [ ] 上位枠が能力境界や sandbox を広げず、task packet に昇格理由が含まれることを確認する。 +- [ ] 管理者許可がない場合は通常役が昇格を試みず、`sol_admin_max` も発行されない。規則の直接入口が `prompt` と評価されることを、昇格コマンドを実行せず確認する。 +- [ ] 管理者役も `gpt-6-astra`/`max` とし、専用の承認条件を維持する。 +- [ ] Windows/Linux でマーカーの一意性・順序、ポリシー・十役・規則のバイト一致、CRLF、差異、不正マーカーを確認する。 +- [ ] v3.1 の判断順、期限、必須・任意確認、終了条件が管理ブロック外の記述で代用できず、10 KiB 超過も検出される。 - [ ] `git diff --check` と秘密情報 scan を通す。 ## 公開 +管理者操作そのものの実行確認が必要な場合は、操作・対象・昇格方法・復旧方法について別途許可を得て実施します。 +ファイルの一致や `config.load = ok` だけで、全役の実動作や管理者操作の成功を主張しません。 + annotated tag を作り、tag だけを push します。例は初回 release です。 ```shell diff --git a/agents/astra-architect-max.toml b/agents/astra-architect-max.toml new file mode 100644 index 0000000..abc8249 --- /dev/null +++ b/agents/astra-architect-max.toml @@ -0,0 +1,32 @@ +name = "astra_architect_max" +description = """ +Use for a bounded D4 synthesis when findings from at least two independent D3 +work items conflict and the combined decision has major consequences, including +irreversible choices or security-sensitive trade-offs across work items. +Use Astra Max when xhigh is insufficient for reconciling the combined evidence. +This is a read-only synthesis role; the parent owns orchestration. +""" +model = "gpt-6-astra" +model_reasoning_effort = "max" +sandbox_mode = "read-only" + +approval_policy = "never" + +developer_instructions = """ +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. +Resolve exactly one exceptionally demanding D4 synthesis from supplied findings. +Require evidence from at least two independent D3 work items. If findings are +missing, return NEEDS_INPUT with the missing evidence; do not invent it. +Use Max reasoning to compare competing interpretations and counterevidence, +trace dependencies across decisions, and identify assumptions that change the +overall recommendation. Preserve uncertainty when evidence remains conflicting. +Use the supplied evidence and inspect references only to resolve specific gaps. +Do not repeat completed investigations or act as another orchestrator. +Do not delegate. +Do not modify files or external state, even if the runtime grants broader access. +Return the overall recommendation, evidence provenance, alternatives and their +consequences, unresolved conflicts, and validation needed before implementation. +The parent makes the final decision and owns any state-changing follow-through. +""" diff --git a/agents/astra-architect.toml b/agents/astra-architect.toml new file mode 100644 index 0000000..31fb122 --- /dev/null +++ b/agents/astra-architect.toml @@ -0,0 +1,30 @@ +name = "astra_architect" +description = """ +Use for one bounded D4 synthesis of findings from at least two independent D3 +work items. Reconcile their constraints, dependencies, and recommendations into +an overall decision. Use Astra xhigh by default; prefer astra_architect_max +when conflicts across the findings and their consequences require deeper review. +This is a read-only synthesis role; the parent owns orchestration. +""" +model = "gpt-6-astra" +model_reasoning_effort = "xhigh" +sandbox_mode = "read-only" + +approval_policy = "never" + +developer_instructions = """ +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. +Resolve exactly one bounded D4 synthesis from the supplied D3 findings. +Require evidence from at least two independent D3 work items. If findings are +missing, return NEEDS_INPUT with the missing evidence; do not invent it. +Compare assumptions, constraints, dependencies, and conflicting recommendations. +Use the supplied evidence and inspect references only to resolve specific gaps. +Do not repeat completed investigations or act as another orchestrator. +Do not delegate. +Do not modify files or external state, even if the runtime grants broader access. +Return the overall recommendation, evidence provenance, resolved conflicts, +remaining uncertainty, and validation needed before implementation. +The parent makes the final decision and owns any state-changing follow-through. +""" diff --git a/agents/luna-task-max.toml b/agents/luna-task-max.toml index 3d1c396..3dc889c 100644 --- a/agents/luna-task-max.toml +++ b/agents/luna-task-max.toml @@ -2,20 +2,24 @@ name = "luna_task_max" description = """ Use for D1 work that remains deterministic, read-only, and objectively verifiable, but needs dense cross-checking across heterogeneous inputs, -coverage-sensitive validation, or many edge cases where Luna Low would be +coverage-sensitive validation, or many edge cases where Luna Medium would be materially more error-prone. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-5.6-luna" -model_reasoning_effort = "max" +model = "gpt-6-luna" +model_reasoning_effort = "high" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Handle exactly one bounded D1 task. -Use Max reasoning for completeness and cross-checking, not to broaden the task's +Use High reasoning for completeness and cross-checking, not to broaden the task's capability boundary. Do not broaden scope or delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. Return only the requested result and essential evidence. If state changes, material judgment, or architectural reasoning are required, return NEEDS_ESCALATION with a one-sentence reason. diff --git a/agents/luna-task-medium.toml b/agents/luna-task-medium.toml new file mode 100644 index 0000000..be7cff3 --- /dev/null +++ b/agents/luna-task-medium.toml @@ -0,0 +1,28 @@ +name = "luna_task_medium" +description = """ +Use for bounded D1 work with fixed inputs, an explicit output contract, and an +objective success condition that needs modest reconciliation across files or +formats. Use Luna Medium when compact, homogeneous Low work is insufficient, +but dense cross-checking, coverage-sensitive validation, and many edge cases +do not justify luna_task_max. This role remains deterministic and read-only. +Do not use for material judgment, broad investigation, or state changes. +""" +model = "gpt-6-luna" +model_reasoning_effort = "medium" +sandbox_mode = "read-only" + +approval_policy = "never" + +developer_instructions = """ +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. +Handle exactly one bounded D1 task. +Use Medium reasoning to reconcile the supplied files or formats against the +explicit output contract and objective success condition. +Do not broaden scope or delegate. +Do not modify files or external state, even if the runtime grants broader access. +Return only the requested result and essential evidence. +If state changes, material judgment, or architectural reasoning are required, +return NEEDS_ESCALATION with a one-sentence reason. +""" diff --git a/agents/luna-task.toml b/agents/luna-task.toml index 0df4637..a0e27b5 100644 --- a/agents/luna-task.toml +++ b/agents/luna-task.toml @@ -3,18 +3,23 @@ description = """ Use as the default for compact, homogeneous D1 extraction, classification, format conversion, repetitive checks, and bounded read-only investigation or validation with fixed inputs, an explicit output contract, and an explicit -success condition. Prefer luna_task_max when dense cross-checking, coverage, -or numerous edge cases make Low materially more error-prone. +success condition. Use luna_task_medium for modest reconciliation across files +or formats; prefer luna_task_max for dense cross-checking, coverage-sensitive +validation, or numerous edge cases. Do not use for material judgment, broad investigation, or state changes. """ -model = "gpt-5.6-luna" +model = "gpt-6-luna" model_reasoning_effort = "low" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Handle exactly one bounded task. Do not broaden scope or delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. If administrator privilege is required, +return NEEDS_ESCALATION with the exact blocked operation. Return only the requested result and essential evidence. If state changes, material judgment, or architectural reasoning are required, return NEEDS_ESCALATION with a one-sentence reason. diff --git a/agents/sol-admin-max.toml b/agents/sol-admin-max.toml new file mode 100644 index 0000000..11854d7 --- /dev/null +++ b/agents/sol-admin-max.toml @@ -0,0 +1,43 @@ +name = "sol_admin_max" +description = """ +Use only for one narrowly scoped administrator operation that requires both +Codex danger-full-access and an OS elevation mechanism. The parent may select +this Astra Max role only after the user explicitly authorizes the named operation, +targets, and privilege boundary. Never use it for ordinary D0-D4 work, broad +maintenance, convenience, or speculative access. +""" +model = "gpt-6-astra" +model_reasoning_effort = "max" +sandbox_mode = "danger-full-access" +approval_policy = "on-request" + +developer_instructions = """ +Own exactly one explicitly authorized administrator operation. Do not delegate. +The task packet must contain `ADMIN_AUTHORIZED: yes`, the exact objective and +targets, the allowed elevation mechanism, rollback or recovery information, +and the required verification. If any field is absent, ambiguous, or broader +than the user's current authorization, return NEEDS_ESCALATION without changing +state. + +Before changing state, verify that the active Codex command rules return +`prompt` for the intended direct elevation entry point. If the rule is missing, +does not load, or does not produce `prompt`, return NEEDS_ESCALATION without +attempting a shell-wrapper or alternate-path bypass. + +Before every command that uses sudo, doas, pkexec, su, Windows elevation, or an +equivalent administrator mechanism, request user approval through the runtime's +permission mechanism and show the exact command, targets, reason, and material +risk. Approval for one command or operation does not authorize adjacent work. +Never request, receive, echo, pipe, log, or persist a password or other secret; +never use sudo -S. Authentication must happen through a user-visible OS prompt +or terminal. Do not weaken authentication, sudoers, UAC, endpoint protection, +or sandbox policy to make the operation easier. + +Treat `full-admin` as a two-boundary capability: danger-full-access removes the +Codex command sandbox, while the OS independently grants or rejects elevated +privilege. This role definition alone is not root or an administrator token. +Destructive actions, secret access, external communication, publication, and +security-boundary changes still require their own explicit authorization. +Authorization expires when this task returns. Report outcome, exact changes, +verification, rollback state, and remaining risk. +""" diff --git a/agents/sol-specialist-max.toml b/agents/sol-specialist-max.toml index 4530471..b1fe5fc 100644 --- a/agents/sol-specialist-max.toml +++ b/agents/sol-specialist-max.toml @@ -3,16 +3,23 @@ description = """ Use for D3 work when both uncertainty and consequence are high, such as conflicting evidence, security-sensitive trade-offs, irreversible architecture, adversarial edge cases, or a strong need to reduce reasoning variance. +Use Astra xhigh for this higher-risk D3 contract when Astra High is insufficient. """ -model = "gpt-5.6-sol" -model_reasoning_effort = "max" +model = "gpt-6-astra" +model_reasoning_effort = "xhigh" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Resolve one exceptionally demanding D3 decision or evidence lane. -Use Max reasoning to reconcile uncertainty, consequences, and edge cases. -Do not repeat broad exploration already completed by cheaper agents. +Use Extra High reasoning to reconcile uncertainty, consequences, and edge cases. +Compare alternatives and counterevidence; identify assumptions that change the +recommendation. +Do not repeat broad exploration already completed by other agents. Do not delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Administrative execution belongs only to +the explicit sol_admin_max gate; Max reasoning alone never grants it. Return a concise recommendation, supporting evidence, risks, and validation plan. """ diff --git a/agents/sol-specialist.toml b/agents/sol-specialist.toml index f2d23fd..fde45c1 100644 --- a/agents/sol-specialist.toml +++ b/agents/sol-specialist.toml @@ -5,14 +5,18 @@ architecture-heavy decision requiring substantial judgment or trade-off analysis. Prefer sol_specialist_max when uncertainty and consequence are both high. """ -model = "gpt-5.6-sol" +model = "gpt-6-astra" model_reasoning_effort = "high" sandbox_mode = "read-only" +approval_policy = "never" developer_instructions = """ Resolve one difficult decision or evidence lane. -Do not repeat broad exploration already completed by cheaper agents. +Do not repeat broad exploration already completed by other agents. Do not delegate. Do not modify files or external state, even if the runtime grants broader access. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Administrative execution belongs only to +the explicit sol_admin_max gate. Return a concise recommendation, supporting evidence, risks, and validation plan. """ diff --git a/agents/terra-worker-max.toml b/agents/terra-worker-max.toml index 2bfdb1b..c7d5821 100644 --- a/agents/terra-worker-max.toml +++ b/agents/terra-worker-max.toml @@ -2,19 +2,25 @@ name = "terra_worker_max" description = """ Use for D2 work that stays within ordinary engineering judgment but has many coupled constraints, a long tool or verification chain, difficult debugging, -or expensive rework that justifies deeper reasoning than Terra Medium. +or expensive rework that justifies deeper reasoning than Sol Medium. Do not use for unresolved architectural trade-offs, exceptional risk, or D3 ambiguity. """ -model = "gpt-5.6-terra" -model_reasoning_effort = "max" +model = "gpt-6-sol" +model_reasoning_effort = "high" + +approval_policy = "never" developer_instructions = """ Own one independently verifiable D2 work item. -Use Max reasoning for coupled constraints, edge cases, and verification, not to +Use High reasoning for coupled constraints, edge cases, and verification, not to broaden the task's capability boundary. Stay within the supplied file and subsystem scope. Do not spawn subagents. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Do not attempt to escape the workspace +sandbox. If administrator privilege is required, return NEEDS_ESCALATION with +the exact blocked operation. Return outcome, changed files or evidence, verification, and remaining risk. Escalate when ambiguity, architectural judgment, or risk materially exceeds D2. """ diff --git a/agents/terra-worker.toml b/agents/terra-worker.toml index 830c31e..6081d5f 100644 --- a/agents/terra-worker.toml +++ b/agents/terra-worker.toml @@ -3,15 +3,20 @@ description = """ Use as the default for bounded D2 state-changing implementation, tool-heavy multi-step work, or investigation and verification that requires ordinary judgment while keeping clear success criteria. Prefer terra_worker_max when -coupled constraints, difficult debugging, or expensive rework justify Max. +coupled constraints, difficult debugging, or expensive rework justify High. """ -model = "gpt-5.6-terra" +model = "gpt-6-sol" model_reasoning_effort = "medium" +approval_policy = "never" developer_instructions = """ Own one independently verifiable work item. Stay within the supplied file and subsystem scope. Do not spawn subagents. +Never invoke or request sudo, doas, pkexec, su, Windows elevation, or any +equivalent administrator mechanism. Do not attempt to escape the workspace +sandbox. If administrator privilege is required, return NEEDS_ESCALATION with +the exact blocked operation. Return outcome, changed files or evidence, verification, and remaining risk. Escalate only when ambiguity or risk materially exceeds the task packet. """ diff --git a/config/AGENTS.task-aware.md b/config/AGENTS.task-aware.md index ceee959..55e7901 100644 --- a/config/AGENTS.task-aware.md +++ b/config/AGENTS.task-aware.md @@ -1,120 +1,168 @@ -## Task-aware delegation policy - -Before spawning any subagent, decompose the request into independently -verifiable work items and classify each item separately. - -### Difficulty routing + +## Task-aware delegation policy v3.1 + +Fix the request-mode authority and mutation boundary first. +Delegation never expands the authority granted to the parent. Answer, review, +diagnosis, and monitoring packets must prohibit file and external-state changes. + +The target parent is GPT-6 Astra. D1 uses `gpt-6-luna`, D2 uses `gpt-6-sol`, +and D3/D4 use `gpt-6-astra`. +Apply the gates when independent work can run alongside useful parent work; +a model name or Ultra cannot permit spawn. Children never delegate. +Finish authorized work and required checks; resolve reversible choices from +context. Ask only for material gaps in correctness, scope, authority, or +irreversible outcomes; continue independent work while waiting. A pause must name its file/rule and why authorization is insufficient. Keep verification proportional to the change. + +### Decision order + +1. Classify verifiable items. D0 is atomic work in one focused tool sequence. + D0 always remains with the parent and never spawns. +2. D1-D4 need four gates: independent progress, distinct deliverable/evidence, + likely context/time savings, and coordination cheaper than direct execution. +3. Only after an affirmative spawn decision, choose role and effort, then set + the packet and finite lifecycle. + +Classification alone never authorizes or requires delegation. If +any delegation gate fails, the parent retains ownership and executes directly +within authority/capability. Never spawn to bypass gates, restate requests, make +generic plans, or duplicate work. Combine microtasks with shared inputs/contracts +unless separation adds useful independence; never split D0 for lower effort. + +### Capability and effort Classify capability first, then choose reasoning effort inside that class. -Higher effort never expands a role's permissions or substitutes for a higher -difficulty class. - -- D0: Atomic, clear, and executable with one focused tool sequence. - Do not spawn a subagent. -- D1: Deterministic extraction, classification, transformation, repetitive - checking, or bounded read-only investigation or verification. Inputs, the - output contract, and the success condition must be explicit, and the work - must not require material judgment. Use `luna_task` or `luna_task_max` as - defined under effort routing. +Reasoning effort alone never expands a role's permissions or capability class. + +- D1: Deterministic extraction, transformation, checking, or bounded read-only + investigation or verification with explicit inputs, output contract, and + objective success condition; no material judgment. Use `luna_task` (Low) for + compact homogeneous inputs, `luna_task_medium` (Medium) for modest reconciliation + across fixed files/formats, or `luna_task_max` (High) for dense cross-checking, + heterogeneous inputs, coverage, or many edge cases. - D2: State-changing implementation, tool-heavy multi-step work, or bounded - investigation and verification that requires ordinary judgment. Completion - criteria must still be clear. Use `terra_worker` or `terra_worker_max` as - defined under effort routing. -- D3: Ambiguous, cross-system, high-risk, security-sensitive, or architectural - work requiring trade-off judgment. Use `sol_specialist` or - `sol_specialist_max` as defined under effort routing. -- D4: Two or more independent D3 work items. Orchestrate them from the root, - but keep each child bounded and non-recursive. - -### Effort routing - -- D1 default: call `spawn_agent` with `agent_type = "luna_task"` for compact, - homogeneous, deterministic work. -- D1 elevated: call `spawn_agent` with `agent_type = "luna_task_max"` when the - work remains deterministic and read-only but heterogeneous inputs, dense - cross-checking, coverage-sensitive validation, or numerous edge cases make - Low materially more error-prone. -- D2 default: call `spawn_agent` with `agent_type = "terra_worker"` for bounded - implementation or investigation requiring ordinary judgment. -- D2 elevated: call `spawn_agent` with `agent_type = "terra_worker_max"` when - the work remains D2 but coupled constraints, a long tool or verification - chain, difficult debugging, or expensive rework justify deeper reasoning. -- D3 default: call `spawn_agent` with `agent_type = "sol_specialist"` for one - bounded difficult decision or evidence lane. -- D3 elevated: call `spawn_agent` with `agent_type = "sol_specialist_max"` when - both uncertainty and consequence are high, including conflicting evidence, - security-sensitive trade-offs, irreversible architecture, adversarial edge - cases, or a strong need to reduce reasoning variance. - -Use the base effort when it is sufficient. Lower model prices reduce the -threshold for elevated effort when it is likely to improve completeness or -avoid rework, but price and task size alone are not sufficient reasons. -Capability, mutation, ambiguity, and risk determine the difficulty class before -cost is considered. Never substitute an elevated lower-class role for a higher -class. - -Use Max as the single elevated effort for D1-D3. Do not add an xhigh middle lane -unless it gains a distinct routing criterion; otherwise it increases routing -ambiguity without changing the capability boundary. - -`task_name` labels the child task; it does not select a custom agent. -For every D1-D3 spawn, `agent_type` is mandatory. Never omit it, and never -encode the role only in `task_name`. Before calling `spawn_agent`, verify that -the request includes the exact base or elevated `agent_type` selected above. -Always attempt the -call with `agent_type`; do not infer that it is unavailable from abbreviated -tool documentation. Only if the tool explicitly rejects `agent_type` or the -selected custom agent should the parent handle the item and report the runtime -mismatch. - -### Delegation gates - -Spawn a subagent only when all of the following are true: - -1. Its work can proceed independently. -2. It produces a distinct deliverable or evidence lane. -3. Delegation is likely to save main-thread context or elapsed time. -4. The coordination cost is smaller than doing the work directly. - -Never spawn an agent merely to restate the request, create a generic plan, or -duplicate another agent's investigation. - -Do not split an atomic D0 item solely because Luna is inexpensive. Combine -adjacent microtasks that share inputs and a success contract when separate -children would not improve elapsed time, context isolation, or evidence -independence. - -Use at most three direct children unless the user explicitly requests more. -Every spawn must set `fork_turns = "none"`. -The runtime configuration caps open child threads at three, excluding the -primary thread. Keep delegation at one level; children must not spawn -descendants. Do not rely on `agents.max_depth` for this boundary because Codex -V2 ignores that legacy setting. Use only one writing agent for overlapping -files or state. - -### Task packet - -Give every child only the minimum task packet required: - -- objective; -- exact scope or paths; -- relevant constraints and evidence; -- completion condition; -- required output shape. - -When selecting an elevated effort variant, also include the concrete reason the -base effort is likely to be materially more error-prone or expensive to rework. - -The packet must explicitly tell the child not to delegate. -Require distilled findings instead of raw logs. If a reasonably selected -lower-cost role returns `NEEDS_ESCALATION`, escalate only with concrete -evidence. Do not route an obvious D2 or D3 item through a cheaper role merely -to obtain an escalation result. -After a spawn succeeds, the parent must not perform the same assigned work in -parallel. Wait for the child and limit parent-side checks to validating the -returned evidence and integrating the result. - -When delegation occurs, briefly report the chosen difficulty and role. Do not -expose hidden reasoning or produce a long routing explanation. + investigation and verification that requires ordinary judgment and clear + completion criteria. Use `terra_worker` (Medium) or `terra_worker_max` (High) + for coupled constraints, long verification, difficult debugging, or costly + rework. D2 inherits the parent's sandbox; unresolved architecture, exceptional + risk, and material ambiguity require D3. +- D3: Read-only ambiguous, cross-system, high-risk, security, or architectural + judgment. Use `sol_specialist` (High) for one bounded decision/evidence lane; + use `sol_specialist_max` (xhigh) when uncertainty and consequence are both + high, such as conflicting evidence, irreversible choices, or adversarial risk. +- D4: Overall judgment from at least two independent D3 work items orchestrated + by the root. Once findings exist, use `astra_architect` (xhigh) for bounded + read-only synthesis, or `astra_architect_max` (Max) when findings conflict and + combined consequences make xhigh insufficient. Never repeat D3 investigations. + +The mapping is D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max. +Nine ordinary roles and separate `sol_admin_max` (Astra Max) retain distinct +capability/authority. Role names are compatibility identifiers, not model or +effort declarations. Use the least sufficient effort. +For D1 Medium, state what reconciliation makes Low insufficient; select High +directly when its conditions are clear, without waiting for Low/Medium failure. +Justify upper roles against Medium for D1/D2, High for D3, or xhigh for D4. +Never underrate an obvious D2/D3 item to obtain `NEEDS_ESCALATION`. + +### Administrator privilege gate + +`full-admin` is a capability gate, not automatic root access. +`danger-full-access` removes the Codex sandbox; the OS controls elevation. +Instructions, TOMLs, and execpolicy are not OS boundaries. Parent permissions +may reach children; direct rules may miss wrappers. Verify effective permissions. +Strong isolation needs an OS account, container, VM, or narrow broker/allowlist. + +- All nine ordinary roles must reject sudo, doas, pkexec, su, Windows elevation, + and equivalents. Do not request elevation or escape the sandbox; return + `NEEDS_ESCALATION` naming the blocked operation. D1, D3, and D4 remain + read-only; higher reasoning effort grants no administrator execution. +- Only `sol_admin_max` may cross this gate. First establish that no unprivileged + path meets the objective and obtain explicit user authorization for the named + operation, exact targets, and privilege boundary in this task. + Never auto-promote an escalation. +- Its packet must contain `ADMIN_AUTHORIZED: yes`, exact targets, allowed + elevation mechanism, recovery/rollback, and verification. Check that the + installed full-admin command rule returns `prompt` for the intended direct + elevation entry point; fail closed if absent, invalid, or non-prompting. +- Authorization ends at child return. Destruction, secrets, external messages, + publication, and security-boundary changes need separate explicit permission. + Never request passwords in chat/tools or use `sudo -S`; use visible OS prompts. + +### Spawn contract + +Every spawn sets exact `agent_type` and `fork_turns = "none"`; `task_name` is a +label. Attempt the role explicitly. If rejected, never substitute a default +child. Parent takeover must preserve capability/authority; otherwise report the +runtime mismatch and stop. + +Respect the live child cap (at most three here); schedule excess work in waves. +Prohibit descendants in developer instructions and packets; verify runtime +enforcement. Use one writer per overlapping state and preserve others' changes. +Report difficulty and role only when delegating. + +D4 synthesis requires findings from at least two independent D3 work items in +the packet, with evidence. Reuse findings and inspect only specific gaps. Wait +for inputs and a free slot; the same three-child limit includes D4. Missing +findings require `STATUS: NEEDS_INPUT` naming the gap. D4 is read-only; the +parent owns orchestration, final judgment, and authorized changes. + +Packets contain objective, scope/paths, authority/mutation boundary, constraints, +evidence, completion/output contract, no-delegation instruction, `NO_PROGRESS_LIMIT`, +`HARD_DEADLINE`, and `SAFE_CANCELLATION`. Justify elevated effort, inherit required +checks, and request distilled findings. Do not duplicate active child work. + +### Bounded acceptance + +For nontrivial work, establish target, scope, existing failures, and user surface. +Before mutation or deep verification, declare: + +- `REQUIRED_ACCEPTANCE_CHECKS`: risk-proportional target/probe, pass criterion, + and required proof rung or user surface. Include independent verification or + rollback from the start when the risk requires it. +- `OPTIONAL_EVIDENCE`: non-blocking, not run by default after required checks + pass; never silently promote it. +- `STOP_CONDITION`: required checks pass, requested artifact/state exists, and + no in-scope blocker remains. Spare tools, unrelated warnings, or wanting more + confidence do not justify continuing. D0/short answers need no formal list. + +An `EXPANSION_TRIGGER` must be a failed/inconclusive required check, changed +scope/behavior, new in-scope risk or security boundary, or explicit user request. +Record evidence, the bounded new check/pass criterion, and stop-condition impact +before expanding. This grants no authority. Never weaken a failed check unless +proven inapplicable or the user changes scope; record the reason. Missing +required proof means conditional/unverified or `BLOCKED`, naming missing +capability/authority; lower proof cannot replace it. Exclude existing/unrelated +failures unless introduced/worsened here or blocking the requested result. +Report required outcomes, expansions, stop status, optional evidence, and risks. +Never claim acceptance with failed/unverified required checks. Distinguish +configuration, runtime, and user-surface evidence. + +### Finite child lifecycle + +- Standard roles: `NO_PROGRESS_LIMIT` 10 minutes, `HARD_DEADLINE` 30 minutes. +- Upper roles (identifiers ending in `_max`), including `sol_admin_max`: + 20 minutes and 60 minutes respectively; the identifier controls this budget, + not the model reasoning effort. +- Budgets start at spawn. Longer budgets require a reason/checkpoint before + spawn; later hard-deadline extensions require explicit user authorization. +- `SAFE_CANCELLATION` names safe interruption and recovery/quiescence checks. + Progress is substantive commentary, tools, changes, evidence, or terminal + returns, not unchanged status/timeouts. It resets only the no-progress clock. +- On no-progress expiry, request status and terminal return once; wait at most + 2 more minutes. Substantive progress resets that clock. No terminal return + after grace, or reaching the hard deadline, means `STALLED` regardless of + child cooperation. +- Read-only children: interrupt and preserve evidence; return conditional results + or `NEEDS_ESCALATION`, without redoing their work. Writers: interrupt only at + the safe point, then verify process quiescence and touched targets read-only. + Without either proof, report `BLOCKED` with child, mutation boundary, recovery, + and next action. Do not interrupt active admin elevation/mutation without a + verified recovery point; otherwise require user-visible recovery. + +Each terminal child return contains `STATUS: COMPLETE` or +`STATUS: NEEDS_ESCALATION` (or D4 `STATUS: NEEDS_INPUT`), plus `RESULT:`, +`EVIDENCE:`, and `OPEN_ISSUES:`. +Only a terminal child return may be validated and integrated. `STALLED` permits +neither duplicate work, replacement writers, nor completion. Never claim success +with an unresolved writer. diff --git a/config/config.task-aware.toml b/config/config.task-aware.toml index 5f046b8..69b12d2 100644 --- a/config/config.task-aware.toml +++ b/config/config.task-aware.toml @@ -1,4 +1,9 @@ # Merge this section into ~/.codex/config.toml. +# Optional parent defaults (place before any table, or use --set-astra-default): +# model = "gpt-6-astra" +# model_reasoning_effort = "xhigh" +# Standard installation preserves the existing parent model and effort. + [agents] enabled = true max_concurrent_threads_per_session = 3 diff --git a/docs/gpt6-family-validation.md b/docs/gpt6-family-validation.md new file mode 100644 index 0000000..db26b1f --- /dev/null +++ b/docs/gpt6-family-validation.md @@ -0,0 +1,36 @@ +# GPT-6 ファミリー移行の検証記録 + +検証日: 2026-09-24。対象は `codex/merge-astra-local` の `0ac314a` を基底にした GPT-6 ファミリー割り当てです。 + +## 設定と移行 + +- 全十役の TOML を解析し、モデルと effort の組を同日のモデルカタログで確認。 +- Codex CLI 0.156.1 で strict config load と管理者規則の `prompt` 評価を確認。昇格コマンド自体は実行していません。 +- Linux CI の clean-home、既存設定保持、再導入、旧設定移行、marker/CRLF、drift、fault injection、10 KiB 上限の検査を実行。 +- 旧全 Astra 構成から D1/D2 のモデルが更新され、親の model/effort、権限、独自役割、旧設定のバックアップが保持されることを確認。 +- 誤った D1 モデルと、D1/D2 を全 Astra に戻した構成が validator に拒否されることを確認。 +- Bash 構文と `git diff --check` を確認。Windows の PowerShell 構文・round trip と Linux の全検査も [GitHub Actions](https://github.com/Eonshore/Codex-Task-Aware-Agent/actions/runs/35978162954) で成功しました(実装コミット `0970932`)。 + +## 実起動 + +既存のグローバル設定を変更せず、一時 `CODEX_HOME` へ導入して認証済みの新しい CLI タスクを開始しました。 +親は Astra Low / read-only。変更対象の五役を `agent_type` と `fork_turns = "none"` で一つずつ起動し、モデル・effort の上書きや代替役へのフォールバックは行っていません。 +各子には再委譲・ツール・書き込みを禁止し、固定文字列 `FAMILY_PROBE_OK` の返却だけを依頼しました。 +返却結果に加え、各子の実行記録の `turn_context` でモデル・effort・sandbox を確認しました。 + +| 役割 | 実モデル | effort | 実効 sandbox | 結果 | +| --- | --- | --- | --- | --- | +| `luna_task` | `gpt-6-luna` | low | read-only | 応答成功 | +| `luna_task_medium` | `gpt-6-luna` | medium | read-only | 応答成功 | +| `luna_task_max` | `gpt-6-luna` | high | read-only | 応答成功 | +| `terra_worker` | `gpt-6-sol` | medium | read-only(親から継承) | 応答成功 | +| `terra_worker_max` | `gpt-6-sol` | high | read-only(親から継承) | 応答成功 | + +最初の CLI 0.154.0 では、五役とも GPT-6 Luna/Sol が ChatGPT アカウントで未対応という HTTP 400 を返しました。 +同じアカウントで CLI 0.156.1 を一時領域に導入すると、モデル一覧に Luna/Sol が現れ、上記五役が応答しました。 +この差を受け、CI の固定バージョンを 0.156.1 に更新しています。0.154.0 の設定読込成功は、モデル利用可能性の証拠にはなりません。 + +この probe は役割の読込・モデル選択・応答の確認です。D0-D4 の自律分類、実装品質、消費量、速度、書き込み時の sandbox、変更していない Astra 役の再評価は対象外です。 +管理者役は起動していません。全役の分類評価と管理者操作の実動作は、公開チェックリスト上の別確認です。 + +参照: [公式のモデル選択案内](https://learn.chatgpt.com/docs/models)、[GPT-6 ファミリーの案内](https://developers.openai.com/api/docs/guides/latest-model)。役割への割り当ては本プロジェクトの方針であり、モデル間の性能比較結果ではありません。 diff --git a/rules/full-admin.rules b/rules/full-admin.rules new file mode 100644 index 0000000..84a2e29 --- /dev/null +++ b/rules/full-admin.rules @@ -0,0 +1,84 @@ +# Route direct administrator-elevation entry points through Codex approval. +# Roles with approval_policy="never" fail closed. Only sol_admin_max uses +# approval_policy="on-request", and its developer instructions require explicit +# user authorization before it can request this prompt. + +prefix_rule( + pattern = [[ + "sudo", + "/usr/bin/sudo", + "/bin/sudo", + "doas", + "/usr/bin/doas", + "/bin/doas", + "pkexec", + "/usr/bin/pkexec", + "/bin/pkexec", + "su", + "/usr/bin/su", + "/bin/su", + ]], + decision = "prompt", + justification = "Administrator elevation requires the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "sudo -n true", + "/usr/bin/sudo apt update", + "doas id", + "pkexec sh", + "su - root", + ], + not_match = [ + "apt update", + "echo sudo", + ], +) + +prefix_rule( + pattern = [[ + "runas", + "runas.exe", + "gsudo", + "gsudo.exe", + "elevate", + "elevate.exe", + ]], + decision = "prompt", + justification = "Windows administrator elevation requires the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "runas /user:Administrator cmd", + "runas.exe /trustlevel:0x20000 cmd", + "gsudo powershell", + ], + not_match = [ + "cmd /c echo runas", + ], +) + +prefix_rule( + pattern = [ + ["powershell", "powershell.exe", "pwsh", "pwsh.exe"], + ["-Command", "-c"], + ["Start-Process", "start-process"], + ], + decision = "prompt", + justification = "PowerShell RunAs requests require the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "pwsh -Command Start-Process powershell -Verb RunAs", + "powershell.exe -c Start-Process cmd -Verb RunAs", + ], +) + +prefix_rule( + pattern = [ + ["powershell", "powershell.exe", "pwsh", "pwsh.exe"], + ["-NoProfile", "-NoLogo"], + ["-Command", "-c"], + ["Start-Process", "start-process"], + ], + decision = "prompt", + justification = "PowerShell RunAs requests require the explicit sol_admin_max full-admin gate and user approval.", + match = [ + "pwsh -NoProfile -Command Start-Process powershell -Verb RunAs", + "powershell.exe -NoLogo -c Start-Process cmd -Verb RunAs", + ], +) diff --git a/scripts/Install-TaskAwareAgent.ps1 b/scripts/Install-TaskAwareAgent.ps1 index e1ef56b..f710aed 100644 --- a/scripts/Install-TaskAwareAgent.ps1 +++ b/scripts/Install-TaskAwareAgent.ps1 @@ -4,17 +4,25 @@ param( if ($env:CODEX_HOME) { $env:CODEX_HOME } else { Join-Path $HOME '.codex' } ), + [switch]$SetAstraDefault, [switch]$SetSolDefault, [switch]$EnableFullAccess ) $ErrorActionPreference = 'Stop' +if ($SetSolDefault) { + throw '-SetSolDefault was retired; use -SetAstraDefault instead.' +} + $RepositoryRoot = Split-Path -Parent $PSScriptRoot $AgentsSource = Join-Path $RepositoryRoot 'agents' +$RulesSource = Join-Path $RepositoryRoot 'rules/full-admin.rules' $PolicySource = Join-Path $RepositoryRoot 'config/AGENTS.task-aware.md' $ConfigPath = Join-Path $CodexHome 'config.toml' $AgentsPath = Join-Path $CodexHome 'agents' +$RulesPath = Join-Path $CodexHome 'rules' +$FullAdminRulePath = Join-Path $RulesPath 'task-aware-full-admin.rules' $AgentsMdPath = Join-Path $CodexHome 'AGENTS.md' $Timestamp = Get-Date -Format 'yyyyMMdd-HHmmss' $BackupPath = Join-Path $CodexHome "task-aware-backups/$Timestamp" @@ -134,16 +142,23 @@ function Set-TopLevelTomlValue { $expectedAgentFiles = @( 'luna-task.toml', + 'luna-task-medium.toml', 'luna-task-max.toml', 'terra-worker.toml', 'terra-worker-max.toml', 'sol-specialist.toml', - 'sol-specialist-max.toml' + 'sol-specialist-max.toml', + 'astra-architect.toml', + 'astra-architect-max.toml', + 'sol-admin-max.toml' ) $retiredAgentFiles = @('luna-task-high.toml', 'terra-worker-high.toml') if (-not (Test-Path -LiteralPath $PolicySource -PathType Leaf)) { throw "Missing policy source: $PolicySource" } +if (-not (Test-Path -LiteralPath $RulesSource -PathType Leaf)) { + throw "Missing full-admin rule source: $RulesSource" +} foreach ($agentFile in $expectedAgentFiles) { $agentSourcePath = Join-Path $AgentsSource $agentFile if (-not (Test-Path -LiteralPath $agentSourcePath -PathType Leaf)) { @@ -162,9 +177,11 @@ if (Test-Path -LiteralPath $AgentsMdPath) { $existingAgentsMd = Get-Content -Raw -LiteralPath $AgentsMdPath $beginMarker = '' $endMarker = '' - $beginCount = [regex]::Matches($existingAgentsMd, [regex]::Escape($beginMarker)).Count - $endCount = [regex]::Matches($existingAgentsMd, [regex]::Escape($endMarker)).Count - $completeBlock = "(?s)" + [regex]::Escape($beginMarker) + ".*?" + [regex]::Escape($endMarker) + $beginPattern = '(?m)^' + [regex]::Escape($beginMarker) + '\r?$' + $endPattern = '(?m)^' + [regex]::Escape($endMarker) + '\r?$' + $beginCount = [regex]::Matches($existingAgentsMd, $beginPattern).Count + $endCount = [regex]::Matches($existingAgentsMd, $endPattern).Count + $completeBlock = '(?ms)^' + [regex]::Escape($beginMarker) + '\r?$.*?^' + [regex]::Escape($endMarker) + '\r?$' if ($beginCount -ne $endCount -or $beginCount -gt 1 -or ($beginCount -eq 1 -and $existingAgentsMd -notmatch $completeBlock)) { throw 'AGENTS.md contains malformed or duplicate Task-Aware Agent markers. Repair the marker block before retrying.' } @@ -174,10 +191,11 @@ if (-not $PSCmdlet.ShouldProcess($CodexHome, 'Install Codex Task-Aware Agent con return } -New-Item -ItemType Directory -Force -Path $CodexHome, $AgentsPath, $BackupPath | Out-Null +New-Item -ItemType Directory -Force -Path $CodexHome, $AgentsPath, $RulesPath, $BackupPath | Out-Null Backup-IfPresent -Path $ConfigPath Backup-IfPresent -Path $AgentsMdPath +Backup-IfPresent -Path $FullAdminRulePath foreach ($agentFile in $expectedAgentFiles) { Backup-IfPresent -Path (Join-Path $AgentsPath $agentFile) } @@ -202,8 +220,8 @@ $config = Remove-TomlSectionKeys -Content $config -Section 'features' -Keys @( 'multi_agent' ) -if ($SetSolDefault) { - $config = Set-TopLevelTomlValue -Content $config -Key 'model' -Value '"gpt-5.6-sol"' +if ($SetAstraDefault) { + $config = Set-TopLevelTomlValue -Content $config -Key 'model' -Value '"gpt-6-astra"' $config = Set-TopLevelTomlValue -Content $config -Key 'model_reasoning_effort' -Value '"xhigh"' } @@ -222,18 +240,23 @@ else { '' } $begin = '' $end = '' -$existingBlock = "(?s)" + [regex]::Escape($begin) + ".*?" + [regex]::Escape($end) +$existingBlock = '(?ms)^' + [regex]::Escape($begin) + '\r?$.*?^' + [regex]::Escape($end) + '\r?$(?:\r?\n)?' if ([regex]::IsMatch($agentsMd, $existingBlock)) { - $agentsMd = [regex]::Replace($agentsMd, $existingBlock, $policy.Trim()) + $agentsMd = [regex]::Replace($agentsMd, $existingBlock, $policy) } else { - $agentsMd = $agentsMd.TrimEnd() + "`n`n" + $policy.Trim() + "`n" + $agentsMd = $agentsMd.TrimEnd() + "`n`n" + $policy } -Set-Content -LiteralPath $AgentsMdPath -Value $agentsMd.TrimStart() -Encoding utf8 +[IO.File]::WriteAllText( + $AgentsMdPath, + $agentsMd.TrimStart(), + [Text.UTF8Encoding]::new($false) +) foreach ($agentFile in $expectedAgentFiles) { Copy-Item -LiteralPath (Join-Path $AgentsSource $agentFile) -Destination $AgentsPath -Force } +Copy-Item -LiteralPath $RulesSource -Destination $FullAdminRulePath -Force foreach ($agentFile in $retiredAgentFiles) { $retiredPath = Join-Path $AgentsPath $agentFile if (Test-Path -LiteralPath $retiredPath -PathType Leaf) { diff --git a/scripts/Test-TaskAwareAgent.ps1 b/scripts/Test-TaskAwareAgent.ps1 index 6c9a6e4..f111b7d 100644 --- a/scripts/Test-TaskAwareAgent.ps1 +++ b/scripts/Test-TaskAwareAgent.ps1 @@ -10,6 +10,20 @@ param( $ErrorActionPreference = 'Stop' $failures = [System.Collections.Generic.List[string]]::new() +$repositoryRoot = Split-Path -Parent $PSScriptRoot +$policySource = Join-Path $repositoryRoot 'config/AGENTS.task-aware.md' +$agentsSource = Join-Path $repositoryRoot 'agents' +$rulesSource = Join-Path $repositoryRoot 'rules/full-admin.rules' +$configPath = Join-Path $CodexHome 'config.toml' +$agentsMdPath = Join-Path $CodexHome 'AGENTS.md' +$agentsPath = Join-Path $CodexHome 'agents' +$fullAdminRulePath = Join-Path $CodexHome 'rules/task-aware-full-admin.rules' +$utf8NoBomStrict = [System.Text.UTF8Encoding]::new($false, $true) + +function Add-Failure { + param([Parameter(Mandatory)][string]$Message) + $failures.Add($Message) +} function Assert-FileContains { param( @@ -17,15 +31,14 @@ function Assert-FileContains { [Parameter(Mandatory)][string[]]$Patterns ) - if (-not (Test-Path -LiteralPath $Path)) { - $failures.Add("Missing file: $Path") + if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { + Add-Failure "Missing file: $Path" return } - $content = Get-Content -Raw -LiteralPath $Path foreach ($pattern in $Patterns) { if ($content -notmatch $pattern) { - $failures.Add("Missing pattern '$pattern' in $Path") + Add-Failure "Missing pattern '$pattern' in $Path" } } } @@ -34,13 +47,260 @@ function Assert-FileAbsent { param([Parameter(Mandatory)][string]$Path) if (Test-Path -LiteralPath $Path) { - $failures.Add("Unexpected retired file: $Path") + Add-Failure "Unexpected retired file: $Path" } } -$configPath = Join-Path $CodexHome 'config.toml' -$agentsMdPath = Join-Path $CodexHome 'AGENTS.md' -$agentsPath = Join-Path $CodexHome 'agents' +function Test-ByteArrayEqual { + param( + [Parameter(Mandatory)][byte[]]$Left, + [Parameter(Mandatory)][byte[]]$Right + ) + + if ($Left.Length -ne $Right.Length) { + return $false + } + for ($index = 0; $index -lt $Left.Length; $index += 1) { + if ($Left[$index] -ne $Right[$index]) { + return $false + } + } + return $true +} + +function Assert-FileByteParity { + param( + [Parameter(Mandatory)][string]$SourcePath, + [Parameter(Mandatory)][string]$InstalledPath, + [Parameter(Mandatory)][string]$Label + ) + + if (-not (Test-Path -LiteralPath $SourcePath -PathType Leaf)) { + Add-Failure "Missing $Label source: $SourcePath" + return + } + if (-not (Test-Path -LiteralPath $InstalledPath -PathType Leaf)) { + Add-Failure "Missing installed $($Label): $InstalledPath" + return + } + if (-not (Test-ByteArrayEqual -Left ([IO.File]::ReadAllBytes($SourcePath)) -Right ([IO.File]::ReadAllBytes($InstalledPath)))) { + Add-Failure "Installed $Label does not match source bytes: $Label" + } +} + +function Get-ManagedPolicyArtifact { + param([Parameter(Mandatory)][string]$Path) + + if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { + Add-Failure "Missing managed policy file: $Path" + return $null + } + + try { + [byte[]]$rawBytes = [IO.File]::ReadAllBytes($Path) + $content = $utf8NoBomStrict.GetString($rawBytes) + } + catch { + Add-Failure "Managed policy file is not valid UTF-8: $Path" + return $null + } + + $beginMarker = '' + $endMarker = '' + $beginPattern = '(?m)^' + [regex]::Escape($beginMarker) + '\r?$' + $endPattern = '(?m)^' + [regex]::Escape($endMarker) + '\r?$' + $beginMatches = [regex]::Matches($content, $beginPattern) + $endMatches = [regex]::Matches($content, $endPattern) + if ($beginMatches.Count -ne 1 -or $endMatches.Count -ne 1) { + Add-Failure "Expected exactly one managed marker pair in $Path; found begin=$($beginMatches.Count) end=$($endMatches.Count)." + return $null + } + + $blockStart = $beginMatches[0].Index + $endStart = $endMatches[0].Index + if ($blockStart -ge $endStart) { + Add-Failure "Managed markers are not one ordered begin/end pair in $Path." + return $null + } + $blockEnd = $endStart + $endMatches[0].Length + if ($blockEnd -lt $content.Length -and $content[$blockEnd] -eq [char]13) { + $blockEnd += 1 + } + if ($blockEnd -lt $content.Length -and $content[$blockEnd] -eq [char]10) { + $blockEnd += 1 + } + $blockText = $content.Substring($blockStart, $blockEnd - $blockStart) + + return [pscustomobject]@{ + Path = $Path + RawBytes = $rawBytes + BlockText = $blockText + BlockBytes = $utf8NoBomStrict.GetBytes($blockText) + } +} + +function Assert-ManagedPolicyParity { + param( + [object]$SourceArtifact, + [object]$LiveArtifact + ) + + if ($null -eq $SourceArtifact -or $null -eq $LiveArtifact) { + return + } + if (-not (Test-ByteArrayEqual -Left $SourceArtifact.RawBytes -Right $LiveArtifact.BlockBytes)) { + Add-Failure 'Managed Task-Aware policy source/live byte parity failed.' + } +} + +function Assert-ManagedPolicyContains { + param( + [object]$Artifact, + [Parameter(Mandatory)][string[]]$Patterns + ) + + if ($null -eq $Artifact) { + return + } + foreach ($pattern in $Patterns) { + if ($Artifact.BlockText -notmatch $pattern) { + Add-Failure "Missing managed pattern '$pattern' in $($Artifact.Path)" + } + } +} + +function Assert-ManagedPolicyExcludes { + param( + [object]$Artifact, + [Parameter(Mandatory)][string[]]$Patterns + ) + + if ($null -eq $Artifact) { + return + } + foreach ($pattern in $Patterns) { + if ($Artifact.BlockText -match $pattern) { + Add-Failure "Forbidden managed pattern '$pattern' found in $($Artifact.Path)" + } + } +} + +function Test-ManagedBlockMatches { + param( + [Parameter(Mandatory)][string]$BlockText, + [Parameter(Mandatory)][string[]]$Patterns + ) + + foreach ($pattern in $Patterns) { + if ($BlockText -notmatch $pattern) { + return $false + } + } + return $true +} + +function Test-ManagedBlockExcludes { + param( + [Parameter(Mandatory)][string]$BlockText, + [Parameter(Mandatory)][string[]]$Patterns + ) + + foreach ($pattern in $Patterns) { + if ($BlockText -match $pattern) { + return $false + } + } + return $true +} + +function Invoke-PolicyFaultInjection { + param([object]$SourceArtifact) + + if ($null -eq $SourceArtifact) { + Add-Failure 'Could not extract source policy for fault injection.' + return + } + if (-not (Test-ManagedBlockMatches -BlockText $SourceArtifact.BlockText -Patterns $managedPolicyPatterns) -or + -not (Test-ManagedBlockExcludes -BlockText $SourceArtifact.BlockText -Patterns $forbiddenPolicyPatterns)) { + Add-Failure 'Source policy does not satisfy the managed policy contract before fault injection.' + return + } + + $withoutDeadline = $SourceArtifact.BlockText.Replace('HARD_DEADLINE', 'DEADLINE_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutDeadline -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed HARD_DEADLINE clauses.' + } + $withoutStalled = $SourceArtifact.BlockText.Replace('STALLED', 'LIFECYCLE_STOPPED') + if (Test-ManagedBlockMatches -BlockText $withoutStalled -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed STALLED clauses.' + } + $withoutClassification = $SourceArtifact.BlockText.Replace( + 'Classification alone never authorizes or requires delegation', + 'Classification authorization removed' + ) + if (Test-ManagedBlockMatches -BlockText $withoutClassification -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed classification-not-authorization clause.' + } + $withoutRequired = $SourceArtifact.BlockText.Replace('REQUIRED_ACCEPTANCE_CHECKS', 'REQUIRED_CHECKS_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutRequired -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed REQUIRED_ACCEPTANCE_CHECKS clauses.' + } + $withoutOptional = $SourceArtifact.BlockText.Replace('OPTIONAL_EVIDENCE', 'OPTIONAL_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutOptional -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed OPTIONAL_EVIDENCE clauses.' + } + $withoutStop = $SourceArtifact.BlockText.Replace('STOP_CONDITION', 'STOP_REMOVED') + if (Test-ManagedBlockMatches -BlockText $withoutStop -Patterns $managedPolicyPatterns) { + Add-Failure 'Fault injection did not detect removed STOP_CONDITION clauses.' + } + $withLegacyWait = $SourceArtifact.BlockText + [Environment]::NewLine + 'a tool-wait timeout is nonterminal: re-wait and do not interrupt' + if (Test-ManagedBlockExcludes -BlockText $withLegacyWait -Patterns $forbiddenPolicyPatterns) { + Add-Failure 'Fault injection did not reject the unbounded wait clause.' + } +} + +$managedPolicyPatterns = @( + 'Managed source: config/AGENTS\.task-aware\.md', + 'Task-aware delegation policy v3\.1', + 'Fix the request-mode authority and mutation boundary', + 'Delegation never expands the authority granted to the parent', + 'D1 uses `gpt-6-luna`, D2 uses `gpt-6-sol`', + 'The target parent is GPT-6 Astra', + 'D3/D4 use `gpt-6-astra`', + 'Children never delegate', + 'Classify capability first, then choose reasoning effort', + 'luna_task_medium', + 'astra_architect', + 'astra_architect_max', + 'D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max', + 'D4 synthesis requires findings from at least two independent D3 work items', + 'same three-child limit', + 'NEEDS_INPUT', + '### Decision order', + 'D0 always remains with the parent and never spawns', + 'Only after an affirmative spawn decision, choose role and effort', + 'Classification alone never authorizes or requires delegation', + 'any delegation gate fails, the parent retains ownership and executes directly', + 'Reasoning effort alone never expands a role''s permissions', + 'Administrator privilege gate', + 'ADMIN_AUTHORIZED: yes', + 'Never auto-promote an escalation', + 'REQUIRED_ACCEPTANCE_CHECKS', + 'OPTIONAL_EVIDENCE', + 'STOP_CONDITION', + 'EXPANSION_TRIGGER', + 'NO_PROGRESS_LIMIT', + 'HARD_DEADLINE', + 'SAFE_CANCELLATION', + 'Upper roles', + '(?-i:\bSTALLED\b)', + 'Only a terminal child return may be validated and integrated' +) +$forbiddenPolicyPatterns = @( + 'tool-wait timeout is nonterminal: re-wait and do not interrupt', + 'After required checks pass, continue collecting any additional evidence available', + 'D1 default: call `spawn_agent`' +) Assert-FileContains -Path $configPath -Patterns @( '(?m)^[ \t]*\[agents\][ \t]*(?:#[^\r\n]*)?\r?$', @@ -48,56 +308,56 @@ Assert-FileContains -Path $configPath -Patterns @( '(?m)^max_concurrent_threads_per_session\s*=\s*3\s*$' ) -Assert-FileContains -Path $agentsMdPath -Patterns @( - '', - 'Task-aware delegation policy', - 'agent_type\s*=\s*"luna_task"', - 'agent_type\s*=\s*"luna_task_max"', - 'agent_type\s*=\s*"terra_worker"', - 'agent_type\s*=\s*"terra_worker_max"', - 'agent_type\s*=\s*"sol_specialist"', - 'agent_type\s*=\s*"sol_specialist_max"', - 'Classify capability first, then choose reasoning effort', - 'Higher effort never expands a role''s permissions', - 'Lower model prices reduce the\s+threshold for elevated effort', - 'Use Max as the single elevated effort for D1-D3', - 'Do not add an xhigh middle lane', - 'concrete reason the\s+base effort is likely to be materially more error-prone', - 'bounded read-only investigation or verification', - 'Inputs, the\s+output contract, and the success condition must be explicit', - 'State-changing implementation', - 'tool-heavy multi-step work', - 'requires ordinary judgment', - 'Do not split an atomic D0 item solely because Luna is inexpensive', - 'Do not route an obvious D2 or D3 item through a cheaper role', - 'fork_turns\s*=\s*"none"', - 'packet must explicitly tell the child not to delegate', - '' -) +$sourcePolicyArtifact = Get-ManagedPolicyArtifact -Path $policySource +$livePolicyArtifact = Get-ManagedPolicyArtifact -Path $agentsMdPath +Assert-ManagedPolicyParity -SourceArtifact $sourcePolicyArtifact -LiveArtifact $livePolicyArtifact +# Keep always-loaded guidance compact; do not cap the user's unmanaged text. +if ($null -ne $sourcePolicyArtifact -and $sourcePolicyArtifact.RawBytes.Length -gt 10240) { + Add-Failure 'Managed policy exceeds the 10 KiB maintenance budget; consolidate existing rules before adding more.' +} +Assert-ManagedPolicyContains -Artifact $sourcePolicyArtifact -Patterns $managedPolicyPatterns +Assert-ManagedPolicyContains -Artifact $livePolicyArtifact -Patterns $managedPolicyPatterns +Assert-ManagedPolicyExcludes -Artifact $sourcePolicyArtifact -Patterns $forbiddenPolicyPatterns +Assert-ManagedPolicyExcludes -Artifact $livePolicyArtifact -Patterns $forbiddenPolicyPatterns +Invoke-PolicyFaultInjection -SourceArtifact $sourcePolicyArtifact $expectedAgents = [ordered]@{ - 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-5.6-luna'; Effort = 'low'; Sandbox = 'read-only' } - 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-5.6-luna'; Effort = 'max'; Sandbox = 'read-only' } - 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-5.6-terra'; Effort = 'medium'; Sandbox = $null } - 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-5.6-terra'; Effort = 'max'; Sandbox = $null } - 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-5.6-sol'; Effort = 'high'; Sandbox = 'read-only' } - 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-5.6-sol'; Effort = 'max'; Sandbox = 'read-only' } + 'luna-task.toml' = [ordered]@{ Name = 'luna_task'; Model = 'gpt-6-luna'; Effort = 'low'; Sandbox = 'read-only'; Approval = 'never' } + 'luna-task-medium.toml' = [ordered]@{ Name = 'luna_task_medium'; Model = 'gpt-6-luna'; Effort = 'medium'; Sandbox = 'read-only'; Approval = 'never' } + 'luna-task-max.toml' = [ordered]@{ Name = 'luna_task_max'; Model = 'gpt-6-luna'; Effort = 'high'; Sandbox = 'read-only'; Approval = 'never' } + 'terra-worker.toml' = [ordered]@{ Name = 'terra_worker'; Model = 'gpt-6-sol'; Effort = 'medium'; Sandbox = $null; Approval = 'never' } + 'terra-worker-max.toml' = [ordered]@{ Name = 'terra_worker_max'; Model = 'gpt-6-sol'; Effort = 'high'; Sandbox = $null; Approval = 'never' } + 'sol-specialist.toml' = [ordered]@{ Name = 'sol_specialist'; Model = 'gpt-6-astra'; Effort = 'high'; Sandbox = 'read-only'; Approval = 'never' } + 'sol-specialist-max.toml' = [ordered]@{ Name = 'sol_specialist_max'; Model = 'gpt-6-astra'; Effort = 'xhigh'; Sandbox = 'read-only'; Approval = 'never' } + 'astra-architect.toml' = [ordered]@{ Name = 'astra_architect'; Model = 'gpt-6-astra'; Effort = 'xhigh'; Sandbox = 'read-only'; Approval = 'never' } + 'astra-architect-max.toml' = [ordered]@{ Name = 'astra_architect_max'; Model = 'gpt-6-astra'; Effort = 'max'; Sandbox = 'read-only'; Approval = 'never' } + 'sol-admin-max.toml' = [ordered]@{ Name = 'sol_admin_max'; Model = 'gpt-6-astra'; Effort = 'max'; Sandbox = 'danger-full-access'; Approval = 'on-request' } } foreach ($entry in $expectedAgents.GetEnumerator()) { - $escapedModel = [regex]::Escape([string]$entry.Value.Model) - $escapedEffort = [regex]::Escape([string]$entry.Value.Effort) - $escapedName = [regex]::Escape([string]$entry.Value.Name) + $agentFile = $entry.Key + $definition = $entry.Value + $escapedName = [regex]::Escape([string]$definition.Name) + $escapedModel = [regex]::Escape([string]$definition.Model) + $escapedEffort = [regex]::Escape([string]$definition.Effort) + $escapedSandbox = [regex]::Escape([string]$definition.Sandbox) + $escapedApproval = [regex]::Escape([string]$definition.Approval) + $installedPath = Join-Path $agentsPath $agentFile + $sourcePath = Join-Path $agentsSource $agentFile $patterns = @( - "(?m)^name\s*=\s*`"$escapedName`"\s*$", + "(?m)^name\s*=\s*""$escapedName""\s*$", '(?m)^description\s*=\s*"""', '(?m)^developer_instructions\s*=\s*"""', - "(?m)^model\s*=\s*`"$escapedModel`"\s*$", - "(?m)^model_reasoning_effort\s*=\s*`"$escapedEffort`"\s*$" + "(?m)^model\s*=\s*""$escapedModel""\s*$", + "(?m)^model_reasoning_effort\s*=\s*""$escapedEffort""\s*$", + "(?m)^approval_policy\s*=\s*""$escapedApproval""\s*$" ) - if ($entry.Value.Sandbox) { - $escapedSandbox = [regex]::Escape([string]$entry.Value.Sandbox) - $patterns += "(?m)^sandbox_mode\s*=\s*`"$escapedSandbox`"\s*$" + if ($null -ne $definition.Sandbox) { + $patterns += "(?m)^sandbox_mode\s*=\s*""$escapedSandbox""\s*$" + } + elseif ((Test-Path -LiteralPath $installedPath) -and + (Get-Content -Raw -LiteralPath $installedPath) -match '(?m)^\s*sandbox_mode\s*=') { + Add-Failure "D2 role must inherit the parent sandbox: $agentFile" } if ($entry.Key -eq 'luna-task.toml') { $patterns += 'Use as the default for compact, homogeneous D1' @@ -106,11 +366,18 @@ foreach ($entry in $expectedAgents.GetEnumerator()) { $patterns += 'success condition' $patterns += 'Do not use for material judgment, broad investigation, or state changes' } + elseif ($entry.Key -eq 'luna-task-medium.toml') { + $patterns += 'bounded D1 work with fixed inputs' + $patterns += 'modest reconciliation across files or' + $patterns += 'objective success condition' + $patterns += 'Do not use for material judgment, broad investigation, or state changes' + $patterns += 'Do not broaden scope or delegate' + } elseif ($entry.Key -eq 'luna-task-max.toml') { $patterns += 'D1 work that remains deterministic, read-only, and objectively' $patterns += 'dense cross-checking across heterogeneous inputs' $patterns += 'Do not use for material judgment, broad investigation, or state changes' - $patterns += 'Use Max reasoning for completeness and cross-checking' + $patterns += 'Use High reasoning for completeness and cross-checking' $patterns += 'not to broaden the task''s\s+capability boundary' } elseif ($entry.Key -eq 'terra-worker.toml') { @@ -122,7 +389,7 @@ foreach ($entry in $expectedAgents.GetEnumerator()) { $patterns += 'D2 work that stays within ordinary engineering judgment' $patterns += 'many\s+coupled constraints' $patterns += 'Do not use for unresolved architectural trade-offs' - $patterns += 'Use Max reasoning for coupled constraints, edge cases, and verification' + $patterns += 'Use High reasoning for coupled constraints, edge cases, and verification' $patterns += 'not to\s+broaden the task''s capability boundary' } elseif ($entry.Key -eq 'sol-specialist-max.toml') { @@ -134,8 +401,51 @@ foreach ($entry in $expectedAgents.GetEnumerator()) { $patterns += 'Use as the default for one bounded D3' $patterns += 'Prefer sol_specialist_max when uncertainty and consequence are both' } - Assert-FileContains -Path (Join-Path $agentsPath $entry.Key) -Patterns $patterns + elseif ($entry.Key -in @('astra-architect.toml', 'astra-architect-max.toml')) { + $patterns += 'bounded D4 synthesis' + $patterns += 'at least two independent D3' + $patterns += 'read-only synthesis role; the parent owns orchestration' + $patterns += 'NEEDS_INPUT' + $patterns += 'Do not repeat completed investigations' + $patterns += 'Do not delegate' + $patterns += 'Do not modify files or external state' + } + Assert-FileContains -Path $installedPath -Patterns $patterns + Assert-FileByteParity -SourcePath $sourcePath -InstalledPath $installedPath -Label "agent $agentFile" +} + +foreach ($agentFile in @( + 'luna-task.toml', + 'luna-task-medium.toml', + 'astra-architect.toml', + 'astra-architect-max.toml', + 'luna-task-max.toml', + 'terra-worker.toml', + 'terra-worker-max.toml', + 'sol-specialist.toml', + 'sol-specialist-max.toml' + )) { + Assert-FileContains -Path (Join-Path $agentsPath $agentFile) -Patterns @('Never invoke or request sudo') } +Assert-FileContains -Path (Join-Path $agentsPath 'sol-admin-max.toml') -Patterns @( + 'ADMIN_AUTHORIZED: yes', + 'Before every command that uses sudo', + 'never use sudo -S', + 'This role definition alone is not root or an administrator token' +) + +Assert-FileContains -Path $fullAdminRulePath -Patterns @( + 'decision\s*=\s*"prompt"', + '"sudo"', + '"doas"', + '"pkexec"', + '"su"', + '"runas"', + '"gsudo"', + '"Start-Process"', + 'explicit sol_admin_max full-admin gate' +) +Assert-FileByteParity -SourcePath $rulesSource -InstalledPath $fullAdminRulePath -Label 'full-admin rule' Assert-FileAbsent -Path (Join-Path $agentsPath 'luna-task-high.toml') Assert-FileAbsent -Path (Join-Path $agentsPath 'terra-worker-high.toml') @@ -150,6 +460,21 @@ if ($codex -and -not $SkipRuntime) { $previousCodexHome = $env:CODEX_HOME try { $env:CODEX_HOME = [IO.Path]::GetFullPath($CodexHome) + $execPolicyOutput = (& $codex.Source execpolicy check --pretty --rules $fullAdminRulePath -- sudo -n true | Out-String) + if ($LASTEXITCODE -ne 0) { + throw "codex execpolicy check failed with exit code $LASTEXITCODE for $fullAdminRulePath" + } + try { + $execPolicyReport = $execPolicyOutput | ConvertFrom-Json -Depth 20 + } + catch { + throw "codex execpolicy check did not return valid JSON: $($execPolicyOutput.Trim())" + } + if ($execPolicyReport.decision -ne 'prompt') { + throw "Full-admin execpolicy did not prompt for sudo: $($execPolicyOutput.Trim())" + } + Write-Host "Full-admin execpolicy prompt passed for CODEX_HOME=$($env:CODEX_HOME)." + if ($ConfigOnlyRuntime) { $doctorOutput = (& $codex.Source --strict-config doctor --json --no-color | Out-String) $doctorExitCode = $LASTEXITCODE @@ -159,7 +484,6 @@ if ($codex -and -not $SkipRuntime) { catch { throw "codex doctor did not return valid JSON for CODEX_HOME=$($env:CODEX_HOME): $($doctorOutput.Trim())" } - $configCheck = $doctorReport.checks.'config.load' if (-not $configCheck -or $configCheck.status -ne 'ok') { throw "Codex strict config load failed for CODEX_HOME=$($env:CODEX_HOME) (doctor exit $doctorExitCode)." diff --git a/scripts/install-task-aware-agent.sh b/scripts/install-task-aware-agent.sh index 41b2c9b..b824d60 100755 --- a/scripts/install-task-aware-agent.sh +++ b/scripts/install-task-aware-agent.sh @@ -10,7 +10,8 @@ Install the Codex Task-Aware Agent configuration globally. Options: --codex-home PATH Target Codex home (default: $CODEX_HOME or ~/.codex) - --set-sol-default Set gpt-5.6-sol with xhigh reasoning as the default + --set-astra-default Set gpt-6-astra with xhigh reasoning as the default + --set-sol-default Retired; use --set-astra-default instead --enable-full-access Set approval_policy=never and danger-full-access -h, --help Show this help EOF @@ -22,7 +23,7 @@ die() { } codex_home="${CODEX_HOME:-$HOME/.codex}" -set_sol_default=false +set_astra_default=false enable_full_access=false while (($# > 0)); do @@ -33,7 +34,10 @@ while (($# > 0)); do shift 2 ;; --set-sol-default) - set_sol_default=true + die '--set-sol-default was retired; use --set-astra-default instead' + ;; + --set-astra-default) + set_astra_default=true shift ;; --enable-full-access) @@ -53,9 +57,12 @@ done script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd -P) repository_root=$(cd -- "$script_dir/.." && pwd -P) agents_source="$repository_root/agents" +rules_source="$repository_root/rules/full-admin.rules" policy_source="$repository_root/config/AGENTS.task-aware.md" config_path="$codex_home/config.toml" agents_path="$codex_home/agents" +rules_path="$codex_home/rules" +full_admin_rule_path="$rules_path/task-aware-full-admin.rules" agents_md_path="$codex_home/AGENTS.md" timestamp=$(date '+%Y%m%d-%H%M%S') backup_path="$codex_home/task-aware-backups/$timestamp" @@ -67,16 +74,21 @@ while [[ -e "$backup_path" ]]; do done [[ -d "$agents_source" ]] || die "missing agents directory: $agents_source" +[[ -f "$rules_source" ]] || die "missing full-admin rule file: $rules_source" [[ -f "$policy_source" ]] || die "missing policy file: $policy_source" command -v awk >/dev/null || die 'awk is required' expected_agent_files=( luna-task.toml + luna-task-medium.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml sol-specialist.toml sol-specialist-max.toml + astra-architect.toml + astra-architect-max.toml + sol-admin-max.toml ) retired_agent_files=(luna-task-high.toml terra-worker-high.toml) for agent_file in "${expected_agent_files[@]}"; do @@ -92,14 +104,17 @@ validate_policy_markers() { end_marker = "" } - $0 == begin_marker { - begin_count++ - if (begin_count > 1 || end_count > 0) invalid = 1 - } - - $0 == end_marker { - end_count++ - if (begin_count != 1 || end_count > 1) invalid = 1 + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + begin_count++ + if (begin_count > 1 || end_count > 0) invalid = 1 + } + if (line == end_marker) { + end_count++ + if (begin_count != 1 || end_count > 1) invalid = 1 + } } END { @@ -260,18 +275,24 @@ merge_policy_block() { for (i = 1; i <= policy_count; i++) print policy[i] } - $0 == begin_marker { - if (!policy_written) { - emit_policy() - policy_written = 1 + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + if (!policy_written) { + emit_policy() + policy_written = 1 + } + block_found = 1 + skipping = 1 + next } - block_found = 1 - skipping = 1 - next } skipping { - if ($0 == end_marker) skipping = 0 + line = $0 + sub(/\r$/, "", line) + if (line == end_marker) skipping = 0 next } @@ -296,10 +317,11 @@ if [[ "$enable_full_access" == true && -f "$config_path" ]] && die 'cannot use --enable-full-access while config.toml defines default_permissions; remove one permission system before retrying' fi -mkdir -p -- "$codex_home" "$agents_path" "$backup_path" +mkdir -p -- "$codex_home" "$agents_path" "$rules_path" "$backup_path" backup_if_present "$config_path" backup_if_present "$agents_md_path" +backup_if_present "$full_admin_rule_path" for agent_file in "${expected_agent_files[@]}"; do backup_if_present "$agents_path/$agent_file" done @@ -315,8 +337,8 @@ remove_toml_section_key "$config_path" agents max_threads remove_toml_section_key "$config_path" agents max_depth remove_toml_section_key "$config_path" features multi_agent -if [[ "$set_sol_default" == true ]]; then - set_top_level_toml_value "$config_path" model '"gpt-5.6-sol"' +if [[ "$set_astra_default" == true ]]; then + set_top_level_toml_value "$config_path" model '"gpt-6-astra"' set_top_level_toml_value "$config_path" model_reasoning_effort '"xhigh"' fi @@ -329,6 +351,7 @@ merge_policy_block "$agents_md_path" for agent_file in "${expected_agent_files[@]}"; do cp -f -- "$agents_source/$agent_file" "$agents_path/$agent_file" done +cp -f -- "$rules_source" "$full_admin_rule_path" for agent_file in "${retired_agent_files[@]}"; do rm -f -- "$agents_path/$agent_file" done diff --git a/scripts/test-task-aware-agent.sh b/scripts/test-task-aware-agent.sh index 1061e09..29231d1 100755 --- a/scripts/test-task-aware-agent.sh +++ b/scripts/test-task-aware-agent.sh @@ -10,7 +10,7 @@ Validate an installed Codex Task-Aware Agent configuration. Options: --codex-home PATH Target Codex home (default: $CODEX_HOME or ~/.codex) - --skip-runtime Skip the codex doctor runtime check + --skip-runtime Skip the codex execpolicy and doctor checks --config-only-runtime Require strict config loading but ignore unrelated doctor failures -h, --help Show this help @@ -24,10 +24,7 @@ config_only_runtime=false while (($# > 0)); do case "$1" in --codex-home) - if (($# < 2)); then - printf 'Error: --codex-home requires a path\n' >&2 - exit 1 - fi + (($# >= 2)) || { printf '%s\n' 'Error: --codex-home requires a path' >&2; exit 1; } codex_home=$2 shift 2 ;; @@ -51,78 +48,370 @@ while (($# > 0)); do done failures=0 +script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd -P) +repository_root=$(cd -- "$script_dir/.." && pwd -P) +source_policy_path="$repository_root/config/AGENTS.task-aware.md" +agents_source="$repository_root/agents" +rules_source="$repository_root/rules/full-admin.rules" +config_path="$codex_home/config.toml" +agents_md_path="$codex_home/AGENTS.md" +agents_path="$codex_home/agents" +full_admin_rule_path="$codex_home/rules/task-aware-full-admin.rules" + +record_failure() { + printf '%s\n' "$*" >&2 + failures=$((failures + 1)) +} assert_file_contains() { local path=$1 shift if [[ ! -f "$path" ]]; then - printf 'Missing file: %s\n' "$path" >&2 - failures=$((failures + 1)) + record_failure "Missing file: $path" return fi local pattern for pattern in "$@"; do if ! grep -Eq -- "$pattern" "$path"; then - printf "Missing pattern '%s' in %s\n" "$pattern" "$path" >&2 - failures=$((failures + 1)) + record_failure "Missing pattern '$pattern' in $path" fi done } assert_file_absent() { local path=$1 - if [[ -e "$path" ]]; then - printf 'Unexpected retired file: %s\n' "$path" >&2 - failures=$((failures + 1)) + record_failure "Unexpected retired file: $path" fi } -config_path="$codex_home/config.toml" -agents_md_path="$codex_home/AGENTS.md" -agents_path="$codex_home/agents" +assert_file_byte_parity() { + local source_path=$1 + local installed_path=$2 + local label=$3 + + if [[ ! -f "$source_path" ]]; then + record_failure "Missing $label source: $source_path" + elif [[ ! -f "$installed_path" ]]; then + record_failure "Missing installed $label: $installed_path" + elif ! cmp -s -- "$source_path" "$installed_path"; then + record_failure "Installed $label does not match source bytes: $label" + fi +} + +count_exact_marker() { + local path=$1 + local marker=$2 + + LC_ALL=C awk -v marker="$marker" ' + { + line = $0 + sub(/\r$/, "", line) + if (line == marker) { + count += 1 + } + } + END { print count + 0 } + ' "$path" +} + +has_one_ordered_marker_pair() { + local path=$1 + local begin_marker='' + local end_marker='' + + LC_ALL=C awk -v begin_marker="$begin_marker" -v end_marker="$end_marker" ' + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + begin_count += 1 + if (end_count > 0 || begin_count != 1) invalid = 1 + } else if (line == end_marker) { + end_count += 1 + if (begin_count != 1 || end_count != 1) invalid = 1 + } + } + END { exit !(begin_count == 1 && end_count == 1 && !invalid) } + ' "$path" +} + +extract_managed_policy_block() { + local path=$1 + local begin_marker='' + local end_marker='' + + LC_ALL=C awk -v begin_marker="$begin_marker" -v end_marker="$end_marker" ' + { + line = $0 + sub(/\r$/, "", line) + if (line == begin_marker) { + begin_count += 1 + if (end_count > 0 || begin_count != 1) invalid = 1 + if (begin_count == 1 && end_count == 0) emitting = 1 + } + if (emitting) print $0 + if (line == end_marker) { + end_count += 1 + if (begin_count != 1 || !emitting || end_count != 1) invalid = 1 + if (emitting) { + emitting = 0 + completed = 1 + } + } + } + END { exit !(begin_count == 1 && end_count == 1 && completed && !invalid) } + ' "$path" +} + +assert_managed_policy_parity() { + local source_path=$1 + local installed_path=$2 + local begin_marker='' + local end_marker='' + local path begin_count end_count valid=true + + for path in "$source_path" "$installed_path"; do + if [[ ! -f "$path" ]]; then + record_failure "Missing managed policy file: $path" + valid=false + continue + fi + begin_count=$(count_exact_marker "$path" "$begin_marker") + end_count=$(count_exact_marker "$path" "$end_marker") + if [[ "$begin_count" != 1 || "$end_count" != 1 ]]; then + record_failure "Expected exactly one managed marker pair in $path; found begin=$begin_count end=$end_count." + valid=false + fi + if ! has_one_ordered_marker_pair "$path"; then + record_failure "Managed markers are not one ordered begin/end pair in $path." + valid=false + fi + done + + if [[ "$valid" == true ]] && ! cmp -s -- "$source_path" <(extract_managed_policy_block "$installed_path"); then + record_failure 'Managed Task-Aware policy source/live byte parity failed.' + fi +} + +managed_block_matches_patterns() { + local block=$1 + shift + + local pattern + for pattern in "$@"; do + if ! grep -Eq -- "$pattern" <<< "$block"; then + return 1 + fi + done + return 0 +} + +managed_block_has_no_patterns() { + local block=$1 + shift + + local pattern + for pattern in "$@"; do + if grep -Eq -- "$pattern" <<< "$block"; then + return 1 + fi + done + return 0 +} + +assert_managed_policy_patterns() { + local path=$1 + shift + local block + + if [[ ! -f "$path" ]]; then + record_failure "Missing managed policy file: $path" + return + fi + if ! block=$(extract_managed_policy_block "$path"); then + record_failure "Could not extract one ordered managed marker pair from $path." + return + fi + + local pattern + for pattern in "$@"; do + if ! grep -Eq -- "$pattern" <<< "$block"; then + record_failure "Missing managed pattern '$pattern' in $path" + fi + done +} + +assert_managed_policy_excludes() { + local path=$1 + shift + local block + + if [[ ! -f "$path" ]]; then + record_failure "Missing managed policy file: $path" + return + fi + if ! block=$(extract_managed_policy_block "$path"); then + record_failure "Could not extract one ordered managed marker pair from $path." + return + fi + + local pattern + for pattern in "$@"; do + if grep -Eq -- "$pattern" <<< "$block"; then + record_failure "Forbidden managed pattern '$pattern' found in $path" + fi + done +} + +run_policy_fault_injection() { + local source_block without_deadline without_stalled without_classification + local without_required without_optional without_stop with_legacy_wait + + if ! source_block=$(extract_managed_policy_block "$source_policy_path"); then + record_failure 'Could not extract source policy for fault injection.' + return + fi + if ! managed_block_matches_patterns "$source_block" "${managed_policy_patterns[@]}" || + ! managed_block_has_no_patterns "$source_block" "${forbidden_policy_patterns[@]}"; then + record_failure 'Source policy does not satisfy the managed policy contract before fault injection.' + return + fi + + without_deadline=${source_block//HARD_DEADLINE/DEADLINE_REMOVED} + if managed_block_matches_patterns "$without_deadline" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed HARD_DEADLINE clauses.' + fi + without_stalled=${source_block//STALLED/LIFECYCLE_STOPPED} + if managed_block_matches_patterns "$without_stalled" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed STALLED clauses.' + fi + without_classification=${source_block//Classification alone never authorizes or requires delegation/Classification authorization removed} + if managed_block_matches_patterns "$without_classification" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed classification-not-authorization clause.' + fi + without_required=${source_block//REQUIRED_ACCEPTANCE_CHECKS/REQUIRED_CHECKS_REMOVED} + if managed_block_matches_patterns "$without_required" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed REQUIRED_ACCEPTANCE_CHECKS clauses.' + fi + without_optional=${source_block//OPTIONAL_EVIDENCE/OPTIONAL_REMOVED} + if managed_block_matches_patterns "$without_optional" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed OPTIONAL_EVIDENCE clauses.' + fi + without_stop=${source_block//STOP_CONDITION/STOP_REMOVED} + if managed_block_matches_patterns "$without_stop" "${managed_policy_patterns[@]}"; then + record_failure 'Fault injection did not detect removed STOP_CONDITION clauses.' + fi + with_legacy_wait="$source_block"$'\n''a tool-wait timeout is nonterminal: re-wait and do not interrupt' + if managed_block_has_no_patterns "$with_legacy_wait" "${forbidden_policy_patterns[@]}"; then + record_failure 'Fault injection did not reject the unbounded wait clause.' + fi +} assert_file_contains "$config_path" \ '^[[:space:]]*\[agents\][[:space:]]*(#.*)?$' \ '^enabled[[:space:]]*=[[:space:]]*true[[:space:]]*$' \ '^max_concurrent_threads_per_session[[:space:]]*=[[:space:]]*3[[:space:]]*$' -assert_file_contains "$agents_md_path" \ - '' \ - 'Task-aware delegation policy' \ - 'agent_type[[:space:]]*=[[:space:]]*"luna_task"' \ - 'agent_type[[:space:]]*=[[:space:]]*"luna_task_max"' \ - 'agent_type[[:space:]]*=[[:space:]]*"terra_worker"' \ - 'agent_type[[:space:]]*=[[:space:]]*"terra_worker_max"' \ - 'agent_type[[:space:]]*=[[:space:]]*"sol_specialist"' \ - 'agent_type[[:space:]]*=[[:space:]]*"sol_specialist_max"' \ - 'Classify capability first, then choose reasoning effort' \ - "Higher effort never expands a role's permissions" \ - 'Lower model prices reduce the' \ - 'threshold for elevated effort' \ - 'Use Max as the single elevated effort for D1-D3' \ - 'Do not add an xhigh middle lane' \ - 'concrete reason the' \ - 'base effort is likely to be materially more error-prone' \ - 'bounded read-only investigation or verification' \ - 'Inputs, the' \ - 'output contract, and the success condition must be explicit' \ - 'State-changing implementation' \ - 'tool-heavy multi-step work' \ - 'requires ordinary judgment' \ - 'Do not split an atomic D0 item solely because Luna is inexpensive' \ - 'Do not route an obvious D2 or D3 item through a cheaper role' \ - 'fork_turns[[:space:]]*=[[:space:]]*"none"' \ - 'packet must explicitly tell the child not to delegate' \ - '' +assert_managed_policy_parity "$source_policy_path" "$agents_md_path" + +# Keep always-loaded guidance compact; do not cap the user's unmanaged text. +if [[ -f "$source_policy_path" ]] && (( $(wc -c < "$source_policy_path") > 10240 )); then + record_failure 'Managed policy exceeds the 10 KiB maintenance budget; consolidate existing rules before adding more.' +fi + +managed_policy_patterns=( + 'Managed source: config/AGENTS\.task-aware\.md' + 'Task-aware delegation policy v3\.1' + 'Fix the request-mode authority and mutation boundary' + 'Delegation never expands the authority granted to the parent' + 'D1 uses `gpt-6-luna`, D2 uses `gpt-6-sol`' + 'The target parent is GPT-6 Astra' + 'D3/D4 use `gpt-6-astra`' + 'Children never delegate' + 'Classify capability first, then choose reasoning effort' + 'luna_task_medium' + 'astra_architect' + 'astra_architect_max' + 'D1 Low/Medium/High, D2 Medium/High, D3 High/xhigh, and D4 xhigh/Max' + 'D4 synthesis requires findings from at least two independent D3 work items' + 'same three-child limit' + 'NEEDS_INPUT' + '### Decision order' + 'D0 always remains with the parent and never spawns' + 'Only after an affirmative spawn decision, choose role and effort' + 'Classification alone never authorizes or requires delegation' + 'any delegation gate fails, the parent retains ownership and executes directly' + 'Reasoning effort alone never expands a role'\''s permissions' + 'Administrator privilege gate' + 'ADMIN_AUTHORIZED: yes' + 'Never auto-promote an escalation' + 'REQUIRED_ACCEPTANCE_CHECKS' + 'OPTIONAL_EVIDENCE' + 'STOP_CONDITION' + 'EXPANSION_TRIGGER' + 'NO_PROGRESS_LIMIT' + 'HARD_DEADLINE' + 'SAFE_CANCELLATION' + 'Upper roles' + 'STALLED' + 'Only a terminal child return may be validated and integrated' +) +forbidden_policy_patterns=( + 'tool-wait timeout is nonterminal: re-wait and do not interrupt' + 'After required checks pass, continue collecting any additional evidence available' + 'D1 default: call `spawn_agent`' +) + +assert_managed_policy_patterns "$source_policy_path" "${managed_policy_patterns[@]}" +assert_managed_policy_patterns "$agents_md_path" "${managed_policy_patterns[@]}" +assert_managed_policy_excludes "$source_policy_path" "${forbidden_policy_patterns[@]}" +assert_managed_policy_excludes "$agents_md_path" "${forbidden_policy_patterns[@]}" +run_policy_fault_injection + +expected_agent_specs=( + 'luna-task.toml|luna_task|gpt-6-luna|low|read-only|never' + 'luna-task-medium.toml|luna_task_medium|gpt-6-luna|medium|read-only|never' + 'luna-task-max.toml|luna_task_max|gpt-6-luna|high|read-only|never' + 'terra-worker.toml|terra_worker|gpt-6-sol|medium||never' + 'terra-worker-max.toml|terra_worker_max|gpt-6-sol|high||never' + 'sol-specialist.toml|sol_specialist|gpt-6-astra|high|read-only|never' + 'sol-specialist-max.toml|sol_specialist_max|gpt-6-astra|xhigh|read-only|never' + 'astra-architect.toml|astra_architect|gpt-6-astra|xhigh|read-only|never' + 'astra-architect-max.toml|astra_architect_max|gpt-6-astra|max|read-only|never' + 'sol-admin-max.toml|sol_admin_max|gpt-6-astra|max|danger-full-access|on-request' +) + +for spec in "${expected_agent_specs[@]}"; do + IFS='|' read -r agent_file agent_name agent_model agent_effort agent_sandbox agent_approval <<< "$spec" + installed_agent_path="$agents_path/$agent_file" + source_agent_path="$agents_source/$agent_file" + assert_file_contains "$installed_agent_path" \ + "^name[[:space:]]*=[[:space:]]*\"$agent_name\"[[:space:]]*$" \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + "^model[[:space:]]*=[[:space:]]*\"$agent_model\"[[:space:]]*$" \ + "^model_reasoning_effort[[:space:]]*=[[:space:]]*\"$agent_effort\"[[:space:]]*$" \ + "^approval_policy[[:space:]]*=[[:space:]]*\"$agent_approval\"[[:space:]]*$" + if [[ -n "$agent_sandbox" ]]; then + assert_file_contains "$installed_agent_path" "^sandbox_mode[[:space:]]*=[[:space:]]*\"$agent_sandbox\"[[:space:]]*$" + elif [[ -f "$installed_agent_path" ]] && grep -Eq '^[[:space:]]*sandbox_mode[[:space:]]*=' "$installed_agent_path"; then + record_failure "D2 role must inherit the parent sandbox: $agent_file" + fi + assert_file_byte_parity "$source_agent_path" "$installed_agent_path" "agent $agent_file" +done +for agent_file in luna-task.toml luna-task-medium.toml luna-task-max.toml terra-worker.toml terra-worker-max.toml sol-specialist.toml sol-specialist-max.toml astra-architect.toml astra-architect-max.toml; do + assert_file_contains "$agents_path/$agent_file" 'Never invoke or request sudo' +done +# Retain the GPT-6 family role capability contracts. assert_file_contains "$agents_path/luna-task.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-luna"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-luna"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"low"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'Use as the default for compact, homogeneous D1' \ @@ -131,17 +420,30 @@ assert_file_contains "$agents_path/luna-task.toml" \ 'success condition' \ 'Do not use for material judgment, broad investigation, or state changes' +assert_file_contains "$agents_path/luna-task-medium.toml" \ + '^name[[:space:]]*=[[:space:]]*"luna_task_medium"[[:space:]]*$' \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-luna"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' \ + '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ + 'bounded D1 work with fixed inputs' \ + 'modest reconciliation across files or' \ + 'objective success condition' \ + 'Do not use for material judgment, broad investigation, or state changes' \ + 'Do not broaden scope or delegate' + assert_file_contains "$agents_path/luna-task-max.toml" \ '^name[[:space:]]*=[[:space:]]*"luna_task_max"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-luna"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-luna"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'D1 work that remains deterministic, read-only, and objectively' \ 'dense cross-checking across heterogeneous inputs' \ 'Do not use for material judgment, broad investigation, or state changes' \ - 'Use Max reasoning for completeness and cross-checking' \ + 'Use High reasoning for completeness and cross-checking' \ "not to broaden the task's" \ 'capability boundary' @@ -154,7 +456,7 @@ assert_file_contains "$agents_path/terra-worker.toml" \ 'multi-step work' \ 'requires ordinary' \ 'judgment while keeping clear success criteria' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-terra"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-sol"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"medium"[[:space:]]*$' assert_file_contains "$agents_path/terra-worker-max.toml" \ @@ -165,17 +467,17 @@ assert_file_contains "$agents_path/terra-worker-max.toml" \ 'many' \ 'coupled constraints' \ 'Do not use for unresolved architectural trade-offs' \ - 'Use Max reasoning for coupled constraints, edge cases, and verification' \ + 'Use High reasoning for coupled constraints, edge cases, and verification' \ 'not to' \ "broaden the task's capability boundary" \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-terra"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' + '^model[[:space:]]*=[[:space:]]*"gpt-6-sol"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' assert_file_contains "$agents_path/sol-specialist.toml" \ '^name[[:space:]]*=[[:space:]]*"sol_specialist"[[:space:]]*$' \ '^description[[:space:]]*=[[:space:]]*"""' \ '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-sol"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ '^model_reasoning_effort[[:space:]]*=[[:space:]]*"high"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ 'Use as the default for one bounded D3' \ @@ -188,10 +490,58 @@ assert_file_contains "$agents_path/sol-specialist-max.toml" \ 'D3 work when both uncertainty and consequence are high' \ 'security-sensitive trade-offs' \ 'reasoning variance' \ - '^model[[:space:]]*=[[:space:]]*"gpt-5\.6-sol"[[:space:]]*$' \ - '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"xhigh"[[:space:]]*$' \ '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' +assert_file_contains "$agents_path/astra-architect.toml" \ + '^name[[:space:]]*=[[:space:]]*"astra_architect"[[:space:]]*$' \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"xhigh"[[:space:]]*$' \ + '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ + 'bounded D4 synthesis' \ + 'at least two independent D3' \ + 'read-only synthesis role; the parent owns orchestration' \ + 'NEEDS_INPUT' \ + 'Do not repeat completed investigations' \ + 'Do not delegate' \ + 'Do not modify files or external state' + +assert_file_contains "$agents_path/astra-architect-max.toml" \ + '^name[[:space:]]*=[[:space:]]*"astra_architect_max"[[:space:]]*$' \ + '^description[[:space:]]*=[[:space:]]*"""' \ + '^developer_instructions[[:space:]]*=[[:space:]]*"""' \ + '^model[[:space:]]*=[[:space:]]*"gpt-6-astra"[[:space:]]*$' \ + '^model_reasoning_effort[[:space:]]*=[[:space:]]*"max"[[:space:]]*$' \ + '^sandbox_mode[[:space:]]*=[[:space:]]*"read-only"[[:space:]]*$' \ + 'bounded D4 synthesis' \ + 'at least two independent D3' \ + 'read-only synthesis role; the parent owns orchestration' \ + 'NEEDS_INPUT' \ + 'Do not repeat completed investigations' \ + 'Do not delegate' \ + 'Do not modify files or external state' + +assert_file_contains "$agents_path/sol-admin-max.toml" \ + 'ADMIN_AUTHORIZED: yes' \ + 'Before every command that uses sudo' \ + 'never use sudo -S' \ + 'This role definition alone is not root or an administrator token' + +assert_file_contains "$full_admin_rule_path" \ + 'decision[[:space:]]*=[[:space:]]*"prompt"' \ + '"sudo"' \ + '"doas"' \ + '"pkexec"' \ + '"su"' \ + '"runas"' \ + '"gsudo"' \ + '"Start-Process"' \ + 'explicit sol_admin_max full-admin gate' +assert_file_byte_parity "$rules_source" "$full_admin_rule_path" 'full-admin rule' + assert_file_absent "$agents_path/luna-task-high.toml" assert_file_absent "$agents_path/terra-worker-high.toml" @@ -202,12 +552,23 @@ fi if [[ "$skip_runtime" == false ]]; then if command -v codex >/dev/null 2>&1; then + set +e + execpolicy_output=$(CODEX_HOME="$codex_home" codex execpolicy check --pretty --rules "$full_admin_rule_path" -- sudo -n true 2>&1) + execpolicy_exit=$? + set -e + if ((execpolicy_exit != 0)) || ! grep -Eq '"decision"[[:space:]]*:[[:space:]]*"prompt"' <<< "$execpolicy_output"; then + printf '%s\n' "$execpolicy_output" >&2 + printf 'Full-admin execpolicy did not return prompt for harmless sudo text (exit %d).\n' "$execpolicy_exit" >&2 + exit 1 + fi + printf 'Full-admin execpolicy prompt passed for CODEX_HOME=%s.\n' "$codex_home" + if [[ "$config_only_runtime" == true ]]; then set +e doctor_output=$(CODEX_HOME="$codex_home" codex --strict-config doctor --json --no-color 2>&1) doctor_exit=$? set -e - config_status=$(printf '%s\n' "$doctor_output" | awk ' + config_status=$(awk ' /"config.load"[[:space:]]*:/ { in_config = 1 } in_config && /"status"[[:space:]]*:/ { status = $0 @@ -216,7 +577,7 @@ if [[ "$skip_runtime" == false ]]; then print status exit } - ') + ' <<< "$doctor_output") if [[ "$config_status" != ok ]]; then printf '%s\n' "$doctor_output" >&2 printf 'Codex strict config load failed with doctor exit %d for CODEX_HOME=%s.\n' "$doctor_exit" "$codex_home" >&2