From eea02206e3f3e00f3ba15be8bb48b928d89fdfe8 Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 13:46:38 +0800 Subject: [PATCH 1/7] feat(spawn): support codex ultra effort --- .agents/skills/harness-adapters/SKILL.md | 4 ++- bin/fm-bootstrap.sh | 2 +- bin/fm-control.sh | 8 ++--- bin/fm-remote-secondmate-control.sh | 2 +- bin/fm-spawn.sh | 27 +++++++------- docs/configuration.md | 4 ++- tests/fm-bootstrap.test.sh | 2 ++ tests/fm-control-relaunch.test.sh | 17 +++++++++ tests/fm-muse-harness.test.sh | 13 ++++++- ...fm-remote-secondmate-trace-context.test.sh | 8 ++++- tests/fm-spawn-dispatch-profile.test.sh | 36 +++++++++++++++++++ 11 files changed, 100 insertions(+), 23 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index d20a7dfaa62..9f03126fdf3 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -120,13 +120,15 @@ Use `low` for well-understood work with an explicit bounded path and `xhigh` for Choose intermediate levels proportionally as complexity, uncertainty, blast radius, or open-ended reasoning increases. When a verified adapter lacks `xhigh`, cap the choice at its highest supported non-`max` level rather than omitting the intended effort silently. Never select `max` from this fallback; use it only when the captain has explicitly expressed that per-task or standing preference. +Never select Codex-only `ultra` from this fallback. +Before selecting `ultra`, explain that it uses internal sub-agent decomposition with significantly higher token spend per turn and obtain the captain's current explicit approval. The supported launch-profile flags below are verified locally; each row records its evidence. | Harness | Model flag | Effort flag | Notes | |---|---|---|---| | claude | `--model ` | `--effort ` | Verified on Claude Code 2.1.196. | -| codex | `--model ` | `-c 'model_reasoning_effort=""'` | Verified on codex-cli 0.142.1. The installed binary schema contains `model_reasoning_effort`, the active config uses it, and the bundled model catalog advertises only low/medium/high/xhigh. `max` is omitted. | +| codex | `--model ` | `-c 'model_reasoning_effort=""'` | Verified from codex-cli 0.149.1's embedded schema for low/medium/high/xhigh/ultra, with ultra exclusive to gpt-5.6-sol and unsupported by the currently nix-pinned PATH codex 0.133.0 until its separate cutover; `max` is omitted. | | grok | `--model ` | `--reasoning-effort ` | Verified on grok 0.2.99 (2026-07-13). `--effort` is an alias, but firstmate's profile axis is reasoning effort. As of 0.2.99 the ceiling is `high`; both `xhigh` and `max` are rejected with `use one of: high, medium, low`, so firstmate omits them. | | pi / pi-signed | `--model ` | `--thinking ` | Verified 2026-07-27 on Pi and pi-signed 0.82.0. Both expose the same accepted thinking levels and completed the same model-qualified max-thinking smoke. | | opencode | `--model ` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 62568cf5771..37e402b82e6 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -1085,7 +1085,7 @@ crew_dispatch_validate() { if $e == null then true elif ($e | type) != "string" then false elif $h == "claude" then (["low","medium","high","xhigh","max"] | index($e)) - elif $h == "codex" then (["low","medium","high","xhigh"] | index($e)) + elif $h == "codex" then (["low","medium","high","xhigh","ultra"] | index($e)) elif $h == "grok" then (["low","medium","high"] | index($e)) elif $h == "pi" or $h == "pi-signed" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "muse" then (["low","medium","high","xhigh","max"] | index($e)) diff --git a/bin/fm-control.sh b/bin/fm-control.sh index 251cc679912..62f8f4b0d7a 100755 --- a/bin/fm-control.sh +++ b/bin/fm-control.sh @@ -243,8 +243,8 @@ fi [ "$MODEL_SET" = 0 ] || [ -n "$NEW_MODEL" ] || die "--model requires a non-empty value" [ "$EFFORT_SET" = 0 ] || [ -n "$NEW_EFFORT" ] || die "--effort requires a non-empty value" case "$NEW_EFFORT" in - ''|low|medium|high|xhigh|max) ;; - *) die "--effort must be one of low, medium, high, xhigh, max" ;; + ''|low|medium|high|xhigh|max|ultra) ;; + *) die "--effort must be one of low, medium, high, xhigh, max, ultra" ;; esac # --- exact task-id resolution ---------------------------------------------- @@ -632,9 +632,9 @@ resolve_relaunch_profile() { CONFIG_MODEL=$("$SCRIPT_DIR/fm-harness.sh" secondmate-model 2>/dev/null || true) CONFIG_EFFORT=$("$SCRIPT_DIR/fm-harness.sh" secondmate-effort 2>/dev/null || true) case "$CONFIG_EFFORT" in - ''|low|medium|high|xhigh|max) ;; + ''|low|medium|high|xhigh|max|ultra) ;; *) - echo "warning: config/secondmate-harness effort token '$CONFIG_EFFORT' is not one of low, medium, high, xhigh, max; ignoring" >&2 + echo "warning: config/secondmate-harness effort token '$CONFIG_EFFORT' is not one of low, medium, high, xhigh, max, ultra; ignoring" >&2 CONFIG_EFFORT= ;; esac diff --git a/bin/fm-remote-secondmate-control.sh b/bin/fm-remote-secondmate-control.sh index fc8cc5ec72e..677e9550c88 100755 --- a/bin/fm-remote-secondmate-control.sh +++ b/bin/fm-remote-secondmate-control.sh @@ -144,7 +144,7 @@ cmd_launch() { claude|codex|opencode|pi|pi-signed|grok|kimi|cursor) ;; *) die "unverified remote secondmate harness: $harness" ;; esac - case "$effort" in -|low|medium|high|xhigh|max) ;; *) die "invalid remote secondmate effort: $effort" ;; esac + case "$effort" in -|low|medium|high|xhigh|max|ultra) ;; *) die "invalid remote secondmate effort: $effort" ;; esac # Herdr is required on this host, not merely preferred: its server belongs to # the GUI login session, so the endpoint survives every SSH disconnection that # a remote route depends on. bin/fm-remote-doctor.sh is the readiness owner. diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 325eefd389c..45f904d4986 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -34,10 +34,11 @@ # the new incarnation. # --harness is the explicit per-spawn harness/profile adapter. The old # positional harness arg still works for back-compat. -# --model and --effort are concrete profile -# axes chosen by firstmate at intake. They are only threaded into harnesses whose -# installed CLIs were verified to support that axis; unsupported axes are omitted -# from that harness's launch rather than guessed. +# --model and --effort are concrete +# profile axes chosen by firstmate at intake. ultra is Codex-only. Values are +# threaded only into harnesses whose installed CLIs were verified to support +# them; unsupported axes are omitted from that harness's launch rather than +# guessed. # --backend is the explicit runtime session-provider backend for this # exact task only (docs/configuration.md "Runtime backend" owns when that flag # is authorized). Without it, the script resolves FM_BACKEND, then @@ -346,8 +347,8 @@ if [ "$TRACEPARENT_SET" -eq 1 ]; then } fi case "$EFFORT" in - ''|low|medium|high|xhigh|max) ;; - *) echo "error: --effort must be one of low, medium, high, xhigh, max" >&2; exit 1 ;; + ''|low|medium|high|xhigh|max|ultra) ;; + *) echo "error: --effort must be one of low, medium, high, xhigh, max, ultra" >&2; exit 1 ;; esac # --relaunch reuses an existing task's endpoint, worktree, project, and kind, @@ -473,7 +474,7 @@ spawn_remote_secondmate() { ;; esac case "$effort" in - -|low|medium|high|xhigh|max) ;; + -|low|medium|high|xhigh|max|ultra) ;; *) fm_lock_release "$registry_lock" || true fm_lock_release "$SPAWN_TASK_LOCK" || true @@ -1288,8 +1289,8 @@ if [ "$KIND" = secondmate ] && [ -z "$ARG3" ]; then SM_EFFORT=$("$SCRIPT_DIR/fm-harness.sh" secondmate-effort) if [ -n "$SM_EFFORT" ]; then case "$SM_EFFORT" in - low|medium|high|xhigh|max) EFFORT=$SM_EFFORT ;; - *) echo "warning: config/secondmate-harness effort token '$SM_EFFORT' is not one of low, medium, high, xhigh, max; ignoring" >&2 ;; + low|medium|high|xhigh|max|ultra) EFFORT=$SM_EFFORT ;; + *) echo "warning: config/secondmate-harness effort token '$SM_EFFORT' is not one of low, medium, high, xhigh, max, ultra; ignoring" >&2 ;; esac fi fi @@ -1392,11 +1393,11 @@ effort_flag_for_harness() { esac ;; codex) - # The installed codex config schema uses model_reasoning_effort, and the - # bundled model catalog advertises low|medium|high|xhigh. Omit max rather - # than passing an unsupported value. + # codex-cli 0.149.1's schema accepts model_reasoning_effort values + # low|medium|high|xhigh|ultra. ultra is exclusive to gpt-5.6-sol. Omit + # max rather than passing an unsupported value. case "$effort" in - low|medium|high|xhigh) printf -- '-c %s ' "$(shell_quote "model_reasoning_effort=\"$effort\"")" ;; + low|medium|high|xhigh|ultra) printf -- '-c %s ' "$(shell_quote "model_reasoning_effort=\"$effort\"")" ;; esac ;; grok) diff --git a/docs/configuration.md b/docs/configuration.md index 49d26f573a7..bf3e21468bb 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -321,7 +321,7 @@ This section is the single owner of the canonical schema and its per-field seman { "when": "", "use": [ - { "harness": "", "model": "", "effort": "" } + { "harness": "", "model": "", "effort": "" } ], "why": "" } @@ -337,6 +337,8 @@ Both `use` and the optional top-level `default` accept either one profile object The single-object form stays fully backward-compatible, and every profile needs `harness`. Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. +`ultra` is a Codex-only profile value, and bootstrap validates that harness pairing while every other harness keeps its existing effort set. +Codex supports `ultra` only on `gpt-5.6-sol`, so select that model when using it. Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. diff --git a/tests/fm-bootstrap.test.sh b/tests/fm-bootstrap.test.sh index 1810e6b5f0b..6710015123e 100755 --- a/tests/fm-bootstrap.test.sh +++ b/tests/fm-bootstrap.test.sh @@ -1119,6 +1119,8 @@ test_crew_dispatch_validation() { malformed dispatch config is flagged^{"rules":[^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - malformed JSON unverified dispatch harness is flagged^{"rules":[{"when":"anything","use":{"harness":"spaceship"}}],"default":{"harness":"codex"}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - unverified harness: spaceship unsupported codex max effort is flagged^{"rules":[{"when":"big feature","use":{"harness":"codex","model":"gpt-5","effort":"max"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: codex:max +codex ultra effort is accepted^{"rules":[{"when":"ultra coding","use":{"harness":"codex","model":"gpt-5.6-sol","effort":"ultra"}}]}^empty^ +unsupported claude ultra effort is flagged^{"rules":[{"when":"ultra coding","use":{"harness":"claude","model":"claude-opus-4-6","effort":"ultra"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: claude:ultra unsupported grok max effort is flagged^{"rules":[{"when":"deep current work","use":{"harness":"grok","model":"grok-4","effort":"max"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: grok:max unsupported grok xhigh effort is flagged^{"rules":[{"when":"deep current work","use":{"harness":"grok","model":"grok-4","effort":"xhigh"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: grok:xhigh pi max effort is accepted^{"rules":[{"when":"deep coding","use":{"harness":"pi","model":"openai-codex/gpt-5.6-sol","effort":"max"}}]}^empty^ diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index 9a7b4285bab..29647cc4859 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -530,6 +530,22 @@ test_explicit_model_wins_over_the_recorded_one() { pass "fm-control relaunch: explicit model and effort win over the recorded ones" } +test_codex_ultra_effort_passes_through_relaunch() { + local dir out rc launch + dir=$(new_case codexultra rl7b) + add_ship_task "$dir" rl7b claude + printf 'codex' > "$dir/fake/becomes" + out=$(run_control "$dir" rl7b relaunch --harness codex --model gpt-5.6-sol --effort ultra --note "use the ultra tier"); rc=$? + expect_code 0 "$rc" "relaunch with codex ultra effort should succeed"$'\n'"$out" + [ "$(meta_field "$dir" rl7b model)" = gpt-5.6-sol ] || fail "the codex ultra model should be recorded" + [ "$(meta_field "$dir" rl7b effort)" = ultra ] || fail "the ultra effort should be recorded" + [ "$(journal_field "$dir" rl7b to_effort)" = ultra ] || fail "the relaunch journal should retain ultra" + launch=$(cat "$dir/fake/literal") + assert_contains "$launch" "-c 'model_reasoning_effort=\"ultra\"'" \ + "fm-control did not pass ultra through to the codex launch" + pass "fm-control relaunch: codex ultra effort reaches the replacement launch" +} + test_relaunch_onto_an_unverified_harness_is_refused() { local dir out rc dir=$(new_case badharness rl8) @@ -1324,6 +1340,7 @@ test_harness_switch_resolves_a_prefixed_recorded_harness test_prefixed_recorded_harness_requires_explicit_replacement test_same_harness_relaunch_keeps_the_profile_axes test_explicit_model_wins_over_the_recorded_one +test_codex_ultra_effort_passes_through_relaunch test_relaunch_onto_an_unverified_harness_is_refused test_prior_harness_turnend_registry_entry_is_cleared test_wiring_removal_failure_refuses_before_replacement_arm diff --git a/tests/fm-muse-harness.test.sh b/tests/fm-muse-harness.test.sh index 83a0747458b..06a5b748532 100755 --- a/tests/fm-muse-harness.test.sh +++ b/tests/fm-muse-harness.test.sh @@ -290,7 +290,18 @@ EOF || fail "muse spawn without an effort axis failed" launch=$(cat "$home/launch.log") assert_not_contains "$launch" '--reasoning-effort' "muse spawn invented an effort when none was chosen" - pass "muse maps the shared effort vocabulary and reaches ultra only via explicit max" + + rec=$(make_spawn_case effort-codex-ultra) + IFS='|' read -r case_dir home proj wt fakebin id </dev/null \ + || fail "muse spawn with codex-only ultra failed" + launch=$(cat "$home/launch.log") + assert_grep 'effort=ultra' "$home/state/$id.meta" "muse meta did not retain codex-only ultra" + assert_not_contains "$launch" '--reasoning-effort' "muse launch exposed codex-only ultra directly" + pass "muse maps max to its ultra tier but omits the codex-only ultra profile value" } # An unauthenticated muse pane does not exit: it sits on an OAuth device-code diff --git a/tests/fm-remote-secondmate-trace-context.test.sh b/tests/fm-remote-secondmate-trace-context.test.sh index d2989364689..e5a67c25656 100755 --- a/tests/fm-remote-secondmate-trace-context.test.sh +++ b/tests/fm-remote-secondmate-trace-context.test.sh @@ -120,7 +120,7 @@ exec "$FM_FAKE_REMOTE_ENTRYPOINT" "$@" SH chmod +x "$FAKEBIN/fake-ssh" -printf 'codex\n' > "$PARENT/config/secondmate-harness" +printf 'codex gpt-5.6-sol ultra\n' > "$PARENT/config/secondmate-harness" printf 'tmux\n' > "$PARENT/config/backend" printf 'codex\n' > "$PARENT/config/crew-harness" printf '## In flight\n\n## Queued\n\n## Done\n' > "$PARENT/data/backlog.md" @@ -169,6 +169,12 @@ freeze_parent_session remote_env "$ROOT/bin/fm-spawn.sh" ios --secondmate >/dev/null 2>&1 \ || fail "default-off remote secondmate spawn failed" assert_present "$PARENT/state/ios.meta" "default-off remote spawn published no parent metadata" +assert_grep 'model=gpt-5.6-sol' "$PARENT/state/ios.meta" \ + "remote codex ultra spawn did not retain its model" +assert_grep 'effort=ultra' "$PARENT/state/ios.meta" \ + "remote codex ultra spawn did not retain its effort" +assert_contains "$(cat "$HERDR_LOG")" 'model_reasoning_effort="ultra"' \ + "remote codex ultra spawn did not emit model_reasoning_effort" ! grep -q '^traceparent=' "$PARENT/state/ios.meta" \ || fail "default-off remote spawn must not record a traceparent= line" ! grep -q 'export TRACEPARENT=' "$HERDR_LOG" \ diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index d1f1effb41a..f072f3a7fe8 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -435,6 +435,23 @@ test_claude_threads_model_and_effort() { pass "claude receives --model and --effort profile flags" } +test_claude_omits_codex_only_ultra_effort() { + local rec id out status launch + id=profile-claude-ultra-z2b + rec=$(make_spawn_case profile-claude-ultra claude "$id") + read_case_record "$rec" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --model sonnet --effort ultra) + status=$? + expect_code 0 "$status" "claude spawn with codex-only ultra effort should omit the effort flag" + assert_meta_profile "$HOME_DIR/state/$id.meta" claude sonnet ultra + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "claude --dangerously-skip-permissions --model 'sonnet'" \ + "claude launch did not preserve the model flag when ultra effort was omitted" + assert_not_contains "$launch" "--effort" "claude launch must omit codex-only ultra effort" + pass "non-codex launches retain ultra in metadata and omit its unsupported flag" +} + test_codex_threads_model_and_effort() { local rec id out status launch id=profile-codex-z3 @@ -451,6 +468,23 @@ test_codex_threads_model_and_effort() { pass "codex receives --model and model_reasoning_effort profile flags" } +test_codex_threads_ultra_effort() { + local rec id out status launch + id=profile-codex-ultra-z3b + rec=$(make_spawn_case profile-codex-ultra codex "$id") + read_case_record "$rec" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" \ + --model gpt-5.6-sol --effort ultra) + status=$? + expect_code 0 "$status" "codex spawn with ultra effort should pass validation" + assert_meta_profile "$HOME_DIR/state/$id.meta" codex gpt-5.6-sol ultra + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "codex --model 'gpt-5.6-sol' -c 'model_reasoning_effort=\"ultra\"' --dangerously-bypass-approvals-and-sandbox" \ + "codex launch did not thread ultra through model_reasoning_effort" + pass "codex receives the ultra model_reasoning_effort value" +} + test_codex_omits_invalid_max_effort() { local rec id out status launch id=profile-codex-max-z4 @@ -838,7 +872,9 @@ test_active_dispatch_profile_allows_explicit_harness test_active_dispatch_profile_allows_positional_harness test_active_dispatch_profile_allows_raw_launch_command test_claude_threads_model_and_effort +test_claude_omits_codex_only_ultra_effort test_codex_threads_model_and_effort +test_codex_threads_ultra_effort test_codex_omits_invalid_max_effort test_grok_threads_model_and_reasoning_effort test_grok_omits_invalid_max_reasoning_effort From b0a6448ec257a0eb9c97f9324ac1d0982b7d50b8 Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 14:01:13 +0800 Subject: [PATCH 2/7] chore(review): docs: record codex ultra CLI floor and disambiguate muse ultra --- .agents/skills/harness-adapters/SKILL.md | 4 ++-- docs/configuration.md | 1 + 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 9f03126fdf3..58c6f7ed4e3 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -128,13 +128,13 @@ The supported launch-profile flags below are verified locally; each row records | Harness | Model flag | Effort flag | Notes | |---|---|---|---| | claude | `--model ` | `--effort ` | Verified on Claude Code 2.1.196. | -| codex | `--model ` | `-c 'model_reasoning_effort=""'` | Verified from codex-cli 0.149.1's embedded schema for low/medium/high/xhigh/ultra, with ultra exclusive to gpt-5.6-sol and unsupported by the currently nix-pinned PATH codex 0.133.0 until its separate cutover; `max` is omitted. | +| codex | `--model ` | `-c 'model_reasoning_effort=""'` | Verified on codex-cli 0.142.1 for low/medium/high/xhigh: the installed binary schema contained `model_reasoning_effort`, the active config used it, and the bundled model catalog advertised those four. codex-cli 0.149.1's embedded schema adds `ultra`, exclusive to gpt-5.6-sol. The PATH codex is currently nix-pinned at 0.133.0 and does not support `ultra` until its separate cutover, so an `ultra` launch fails there. `max` is omitted. | | grok | `--model ` | `--reasoning-effort ` | Verified on grok 0.2.99 (2026-07-13). `--effort` is an alias, but firstmate's profile axis is reasoning effort. As of 0.2.99 the ceiling is `high`; both `xhigh` and `max` are rejected with `use one of: high, medium, low`, so firstmate omits them. | | pi / pi-signed | `--model ` | `--thinking ` | Verified 2026-07-27 on Pi and pi-signed 0.82.0. Both expose the same accepted thinking levels and completed the same model-qualified max-thinking smoke. | | opencode | `--model ` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | | kimi | `--model ` | none | Verified 2026-07-25 on Kimi Code CLI 0.29.1. | | cursor | `--model ` | none | Verified 2026-08-11 on Cursor Agent CLI 2026.08.11-e8db854. No effort flag exists, so firstmate records the requested effort in task metadata and omits it from the launch. Validate ids against `cursor-agent --list-models` rather than assuming a low/medium/high family: the live catalog carries only `-high` Grok ids. | -| muse | `--model ` | `--reasoning-effort `, and `ultra` only for an explicit `max` | Verified 2026-08-05 on Muse Code 0.1.0-R708.1. The flag accepts `none\|minimal\|low\|medium\|high\|xhigh\|ultra` and defaults to `high`. `ultra` is muse's max-class level, so it is reachable only through an explicit captain `max`, never from the generic fallback; `none` and `minimal` sit below the shared vocabulary and stay unreachable. | +| muse | `--model ` | `--reasoning-effort `, and muse-native `ultra` only for an explicit `max` | Verified 2026-08-05 on Muse Code 0.1.0-R708.1. The flag accepts `none\|minimal\|low\|medium\|high\|xhigh\|ultra` and defaults to `high`. Muse's native `ultra` is its own max-class level and is unrelated to firstmate's Codex-only `ultra` profile value: firstmate reaches muse's level only by mapping an explicit captain `max` onto it, never from the generic fallback, while a profile carrying the Codex-only `ultra` is recorded in task meta and omitted from the muse launch, leaving muse on its `high` default. `none` and `minimal` sit below the shared vocabulary and stay unreachable. | The concrete `harness` field owns adapter identity independently of the model provider: `harness=pi` with `model=xai/grok-*` is Pi using xAI, not `harness=grok`, and does not require Grok CLI login; `harness=grok` remains the standalone Grok Build CLI adapter. Likewise, `harness=cursor` with `model=cursor-grok-4.5-*` is Cursor Agent CLI routing a Grok model, not the xAI Grok Build `grok` harness. diff --git a/docs/configuration.md b/docs/configuration.md index bf3e21468bb..02344b63f57 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -339,6 +339,7 @@ Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. `ultra` is a Codex-only profile value, and bootstrap validates that harness pairing while every other harness keeps its existing effort set. Codex supports `ultra` only on `gpt-5.6-sol`, so select that model when using it. +`ultra` also needs codex-cli 0.149.1 or newer on `PATH`; the PATH codex is currently nix-pinned at 0.133.0, which does not support it, so an `ultra` profile fails at pane launch until that CLI is rebuilt and cut over. Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. From 62b1efaecdfb02d05f5d58580a46ed3eb818caae Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 14:33:55 +0800 Subject: [PATCH 3/7] chore(review): move remote ultra coverage to its own suite, note codex pin --- bin/fm-spawn.sh | 6 +- bin/fm-test-run.sh | 1 + docs/remote-secondmates.md | 3 + .../fm-remote-secondmate-profile-axes.test.sh | 226 ++++++++++++++++++ ...fm-remote-secondmate-trace-context.test.sh | 8 +- 5 files changed, 235 insertions(+), 9 deletions(-) create mode 100755 tests/fm-remote-secondmate-profile-axes.test.sh diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 45f904d4986..b135e14ef7e 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1394,8 +1394,10 @@ effort_flag_for_harness() { ;; codex) # codex-cli 0.149.1's schema accepts model_reasoning_effort values - # low|medium|high|xhigh|ultra. ultra is exclusive to gpt-5.6-sol. Omit - # max rather than passing an unsupported value. + # low|medium|high|xhigh|ultra. ultra is exclusive to gpt-5.6-sol, and the + # currently nix-pinned PATH codex 0.133.0 rejects it outright until its + # separately owned cutover. Omit max rather than passing an unsupported + # value. case "$effort" in low|medium|high|xhigh|ultra) printf -- '-c %s ' "$(shell_quote "model_reasoning_effort=\"$effort\"")" ;; esac diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 2d3a64bcadb..1dd47a8fd1f 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -173,6 +173,7 @@ family_for_basename() { fm-backlog-handoff.test.sh|fm-on.test.sh|fm-remote-backlog-handoff.test.sh|\ fm-remote-doctor.test.sh|fm-remote-job.test.sh|fm-remote-job-orphan-reap.test.sh|\ fm-remote-reply.test.sh|fm-remote-secondmate-lifecycle-e2e.test.sh|\ + fm-remote-secondmate-profile-axes.test.sh|\ fm-remote-secondmate-trace-context.test.sh|\ fm-secondmate-harness.test.sh|fm-secondmate-lifecycle-e2e.test.sh|\ fm-secondmate-liveness.test.sh|fm-secondmate-safety.test.sh|fm-secondmate-sync.test.sh|\ diff --git a/docs/remote-secondmates.md b/docs/remote-secondmates.md index 891f85cf07f..8e3da1bd8be 100644 --- a/docs/remote-secondmates.md +++ b/docs/remote-secondmates.md @@ -251,9 +251,12 @@ bin/fm-test-run.sh tests/fm-project-origin.test.sh bin/fm-test-run.sh tests/fm-remote-reply.test.sh bin/fm-test-run.sh tests/fm-remote-backlog-handoff.test.sh bin/fm-test-run.sh tests/fm-remote-secondmate-lifecycle-e2e.test.sh +bin/fm-test-run.sh tests/fm-remote-secondmate-profile-axes.test.sh bin/fm-test-run.sh tests/fm-remote-secondmate-trace-context.test.sh ``` +`tests/fm-remote-secondmate-profile-axes.test.sh` owns the launch-profile axes on this route: the harness, model, and effort a `config/secondmate-harness` line pins are validated by the parent, replayed across the SSH boundary by `bin/fm-remote-secondmate-control.sh`, and applied by the remote host's own `bin/fm-spawn.sh`, so the launch literal the remote pane receives is what the suite reads back. + The account-level checks the doctor performs - a real Aqua login session, a real `launchctl` domain, and a real herdr server - are only ever exercised against fixtures here, so the readiness gate's behavior on a genuine Mac remains an operator-run smoke test. For a real-host smoke test, provision a disposable remote account and project, run the doctor and its repair against that account, launch the second mate, send one marked request, verify its correlated reply and structured fleet projection, simulate an unreachable host to confirm unknown-without-failover behavior, then retire only after the remote queue is empty. diff --git a/tests/fm-remote-secondmate-profile-axes.test.sh b/tests/fm-remote-secondmate-profile-axes.test.sh new file mode 100755 index 00000000000..9e43e817e82 --- /dev/null +++ b/tests/fm-remote-secondmate-profile-axes.test.sh @@ -0,0 +1,226 @@ +#!/usr/bin/env bash +# tests/fm-remote-secondmate-profile-axes.test.sh - launch-profile axis regressions +# for the REMOTE second mate route, over the deterministic generic SSH boundary. +# +# The local spawn path's coverage lives in tests/fm-secondmate-harness.test.sh +# (capability C, the optional model/effort tokens config/secondmate-harness +# carries) and tests/fm-spawn-dispatch-profile.test.sh. A remote second mate +# never reaches that path: bin/fm-spawn.sh routes it through +# spawn_remote_secondmate, which validates the configured effort, hands it to +# bin/fm-remote-secondmate-control.sh, and that control script re-validates and +# replays it as --model/--effort into the remote host's OWN bin/fm-spawn.sh. +# Both ends must therefore accept the axis, and only the remote end decides +# whether it reaches the launch command. +# +# These assertions drive the real chain - parent fm-spawn -> fm-on -> the real +# remote entrypoint -> fm-remote-secondmate-control -> the remote host's own +# fm-spawn - against a fake herdr CLI, so the launch literal the remote pane +# received is observable. +# +# See docs/remote-secondmates.md for the remote route's maintained contract and +# .agents/skills/harness-adapters/SKILL.md for the per-harness effort flags. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +# shellcheck source=tests/remote-herdr-fixture.sh +. "$(dirname "${BASH_SOURCE[0]}")/remote-herdr-fixture.sh" + +ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P) +TMP_ROOT=$(fm_test_tmproot fm-remote-profile-axes) +mkdir -p "$TMP_ROOT" +TMP_ROOT=$(cd "$TMP_ROOT" && pwd -P) +PARENT="$TMP_ROOT/parent" +REMOTE_ROOT="$TMP_ROOT/remote-root" +REMOTE_HOME="$TMP_ROOT/remote-home" +FAKEBIN=$(fm_fakebin "$TMP_ROOT/fake") +HERDR_LOG="$TMP_ROOT/remote-herdr.log" +HERDR_STATE="$TMP_ROOT/remote-herdr.state" +TMUX_LOG="$TMP_ROOT/remote-tmux.log" +TMUX_STATE="$TMP_ROOT/remote-tmux.state" +CLAIMS="$TMP_ROOT/claims" +REMOTE_META="$REMOTE_HOME/state/parent-route/ios.meta" +mkdir -p "$PARENT/data" "$PARENT/state" "$PARENT/config" "$PARENT/projects" "$REMOTE_ROOT" "$CLAIMS" +trap 'FM_HOME="$PARENT" FM_PROCEVENT_CLAIM_ROOT="$CLAIMS" "$ROOT/bin/fm-procevent.sh" sweep-home >/dev/null 2>&1 || true; if [ -f "$TMP_ROOT/remote-jobs/worker.pid" ]; then kill "$(cat "$TMP_ROOT/remote-jobs/worker.pid")" 2>/dev/null || true; fi; rm -rf -- "$TMP_ROOT"' EXIT + +# The remote host's tracked code root is this branch, as a real git repository: +# fm-on and the remote entrypoint both require the dispatched command to be +# tracked there, and the remote side runs the real scripts under test. +( + cd "$ROOT" || exit + tar --exclude=.git --exclude=.no-mistakes --exclude=data --exclude=state --exclude=config -cf - . +) | (cd "$REMOTE_ROOT" && tar -xf -) + +# The remote host runs the Herdr fixture, whose every invocation is logged +# verbatim, so the launch literal the pane received is read back exactly. The +# tmux fixture below only keeps the remote home's own non-second-mate tooling +# resolvable. +cat > "$REMOTE_ROOT/bin/tmux" <> "\$log" +case "\${1:-}" in + has-session|new-session|set-window-option) exit 0 ;; + list-windows) + [ -f "\$state" ] || exit 0 + name=\$(cut -d'|' -f1 "\$state") + case "\$*" in *'#{session_name}:#{window_name}'*) printf 'firstmate:%s\n' "\$name" ;; *) printf '%s\n' "\$name" ;; esac + exit 0 + ;; + new-window) + name=; cwd= + while [ "\$#" -gt 0 ]; do + case "\$1" in -n) shift; name=\$1 ;; -c) shift; cwd=\$1 ;; esac + shift + done + printf '%s|%s\n' "\$name" "\$cwd" > "\$state" + printf '@1\n' + exit 0 + ;; + display-message) + case "\$*" in + *'#{pane_current_path}'*) cut -d'|' -f2- "\$state" ;; + *'#{pane_current_command}'*) printf 'codex\n' ;; + *'#{cursor_y}'*) printf '0\n' ;; + *'#S'*) printf 'firstmate\n' ;; + *) printf '%%1\n' ;; + esac + exit 0 + ;; + capture-pane) printf '❯\n'; exit 0 ;; + send-keys) exit 0 ;; + kill-window) rm -f -- "\$state"; exit 0 ;; + list-panes) printf 'codex\n'; exit 0 ;; +esac +exit 0 +SH +chmod +x "$REMOTE_ROOT/bin/tmux" +install_remote_herdr_fixture "$REMOTE_ROOT" "$HERDR_STATE" "$HERDR_LOG" \ + "$TMP_ROOT/herdr-send-fail" "$TMP_ROOT/herdr.sock" +git -C "$REMOTE_ROOT" init -q -b main +git -C "$REMOTE_ROOT" config user.email test@example.com +git -C "$REMOTE_ROOT" config user.name Test +git -C "$REMOTE_ROOT" add . +git -C "$REMOTE_ROOT" commit -qm 'remote fixture root' + +cat > "$FAKEBIN/fake-ssh" <<'SH' +#!/usr/bin/env bash +while [ "$#" -gt 0 ]; do + case "$1" in -o) shift 2 ;; --) shift; break ;; *) exit 90 ;; esac +done +host=$1 +entry=$2 +shift 2 +[ "$host" = remote-mac ] || exit 91 +[ "$entry" = fm-remote-entrypoint.sh ] || exit 92 +cd "$FM_FAKE_REMOTE_CWD" || exit 93 +# The readiness gate is answered here rather than by the real doctor, which +# would inspect the RUNNER's own account; tests/fm-remote-doctor.test.sh owns +# the doctor's behavior against controlled account fixtures. +if printf '%s' "$4" | base64 --decode 2>/dev/null | tr '\0' '\n' | head -1 | grep -q '^fm-remote-doctor.sh$'; then + printf 'ok: remote second-mate readiness confirmed on this host\n' + exit 0 +fi +exec "$FM_FAKE_REMOTE_ENTRYPOINT" "$@" +SH +chmod +x "$FAKEBIN/fake-ssh" + +printf 'tmux\n' > "$PARENT/config/backend" +printf 'codex\n' > "$PARENT/config/crew-harness" +printf '## In flight\n\n## Queued\n\n## Done\n' > "$PARENT/data/backlog.md" + +remote_env() { + FM_HOME="$PARENT" \ + FM_ROOT_OVERRIDE="$REMOTE_ROOT" \ + FM_PROCEVENT_CLAIM_ROOT="$CLAIMS" \ + FM_SSH_BIN="$FAKEBIN/fake-ssh" \ + FM_FAKE_REMOTE_ENTRYPOINT="$REMOTE_ROOT/bin/fm-remote-entrypoint.sh" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux \ + FM_REMOTE_JOB_STATE_ROOT="$TMP_ROOT/remote-jobs" \ + FM_FAKE_REMOTE_CWD="$TMP_ROOT" \ + FM_SEND_SETTLE=0 FM_SEND_SLEEP=0 \ + "$@" +} + +# Relaunch the one seeded route under a fresh profile pin. The previous endpoint +# is retired first, so this is an ordinary relaunch that re-resolves +# config/secondmate-harness rather than a duplicate-launch refusal. +relaunch_with_profile() { # + printf '%s\n' "$1" > "$PARENT/config/secondmate-harness" + reset_remote_herdr_fixture "$HERDR_STATE" + : > "$HERDR_LOG" + remote_env "$ROOT/bin/fm-spawn.sh" ios --secondmate >/dev/null 2>&1 \ + || fail "$2" +} + +meta_axis() { sed -n "s/^$2=//p" "$1"; } + +# Provision and register the remote route from the captain-facing primary. +FM_SECONDMATE_CHARTER='Own iOS delivery on the build Mac.' \ + FM_SECONDMATE_SCOPE='iOS implementation and Xcode validation' \ + remote_env "$ROOT/bin/fm-remote-home-seed.sh" ios remote-mac "$REMOTE_ROOT" "$REMOTE_HOME" --no-projects >/dev/null \ + || fail "remote seed did not provision the route" + +# --- codex: the Codex-only ultra tier survives both ends and reaches the pane -- +# Before ultra was accepted, the parent's own effort validation refused this +# spawn outright, so the launch below never happened on either host. +relaunch_with_profile 'codex gpt-5.6-sol ultra' \ + "the remote route refused a configured codex+ultra profile" +[ "$(meta_axis "$PARENT/state/ios.meta" harness)" = codex ] \ + || fail "parent metadata did not record the configured remote harness" +[ "$(meta_axis "$PARENT/state/ios.meta" model)" = gpt-5.6-sol ] \ + || fail "parent metadata did not record the configured remote model" +[ "$(meta_axis "$PARENT/state/ios.meta" effort)" = ultra ] \ + || fail "parent metadata did not record the configured remote ultra effort" +[ "$(meta_axis "$REMOTE_META" effort)" = ultra ] \ + || fail "the remote host's own route metadata did not record ultra" +assert_grep "--model 'gpt-5.6-sol'" "$HERDR_LOG" \ + "the remote codex launch did not carry the configured model" +assert_grep "-c 'model_reasoning_effort=\"ultra\"'" "$HERDR_LOG" \ + "the remote codex launch did not emit the ultra reasoning effort" +pass "codex: a configured ultra profile crosses the SSH boundary and reaches the remote pane's launch command" + +# --- pi: the shared vocabulary still emits its own top tier -------------------- +# The control below proves the pi adapter's effort flag is live on this route, +# so the omission asserted next is specific to ultra rather than a dead flag. +relaunch_with_profile 'pi anthropic/claude-opus-5 max' \ + "the remote route refused a configured pi+max profile" +[ "$(meta_axis "$PARENT/state/ios.meta" effort)" = max ] \ + || fail "parent metadata did not record the configured remote max effort" +assert_grep "--thinking 'max'" "$HERDR_LOG" \ + "the remote pi launch did not emit its own max thinking level" +pass "pi: the shared max level still reaches the remote pi launch command" + +# --- pi: ultra is recorded for traceability and omitted from the launch -------- +# Pi has no ultra concept and its ladder ends at max, so the remote end applies +# the documented record-and-omit contract rather than passing a rejected value. +relaunch_with_profile 'pi anthropic/claude-opus-5 ultra' \ + "the remote route refused a configured pi+ultra profile" +[ "$(meta_axis "$PARENT/state/ios.meta" effort)" = ultra ] \ + || fail "the parent dropped an unsupported remote effort instead of recording it" +[ "$(meta_axis "$REMOTE_META" effort)" = ultra ] \ + || fail "the remote host's own route metadata dropped the unsupported effort" +assert_grep "--model 'anthropic/claude-opus-5'" "$HERDR_LOG" \ + "the remote pi launch dropped the model alongside the unsupported effort" +assert_no_grep '--thinking' "$HERDR_LOG" \ + "the remote pi launch passed the Codex-only ultra to an adapter that rejects it" +assert_no_grep 'ultra' "$HERDR_LOG" \ + "the Codex-only ultra token leaked into a non-codex remote launch" +pass "pi: a Codex-only ultra profile is recorded on both hosts and omitted from the remote launch" + +# --- an unverified effort is refused before anything is published ------------- +printf 'codex gpt-5.6-sol supreme\n' > "$PARENT/config/secondmate-harness" +reset_remote_herdr_fixture "$HERDR_STATE" +: > "$HERDR_LOG" +if out=$(remote_env "$ROOT/bin/fm-spawn.sh" ios --secondmate 2>&1); then + fail "the remote route launched an unverified configured effort" +fi +assert_contains "$out" 'invalid configured remote secondmate effort' \ + "the remote route did not name the unverified effort it refused" +assert_no_grep 'tab create' "$HERDR_LOG" \ + "an unverified remote effort reached the remote backend" +pass "an unverified configured effort is refused by the parent before any remote launch" + +echo "ALL TESTS PASSED" diff --git a/tests/fm-remote-secondmate-trace-context.test.sh b/tests/fm-remote-secondmate-trace-context.test.sh index e5a67c25656..d2989364689 100755 --- a/tests/fm-remote-secondmate-trace-context.test.sh +++ b/tests/fm-remote-secondmate-trace-context.test.sh @@ -120,7 +120,7 @@ exec "$FM_FAKE_REMOTE_ENTRYPOINT" "$@" SH chmod +x "$FAKEBIN/fake-ssh" -printf 'codex gpt-5.6-sol ultra\n' > "$PARENT/config/secondmate-harness" +printf 'codex\n' > "$PARENT/config/secondmate-harness" printf 'tmux\n' > "$PARENT/config/backend" printf 'codex\n' > "$PARENT/config/crew-harness" printf '## In flight\n\n## Queued\n\n## Done\n' > "$PARENT/data/backlog.md" @@ -169,12 +169,6 @@ freeze_parent_session remote_env "$ROOT/bin/fm-spawn.sh" ios --secondmate >/dev/null 2>&1 \ || fail "default-off remote secondmate spawn failed" assert_present "$PARENT/state/ios.meta" "default-off remote spawn published no parent metadata" -assert_grep 'model=gpt-5.6-sol' "$PARENT/state/ios.meta" \ - "remote codex ultra spawn did not retain its model" -assert_grep 'effort=ultra' "$PARENT/state/ios.meta" \ - "remote codex ultra spawn did not retain its effort" -assert_contains "$(cat "$HERDR_LOG")" 'model_reasoning_effort="ultra"' \ - "remote codex ultra spawn did not emit model_reasoning_effort" ! grep -q '^traceparent=' "$PARENT/state/ios.meta" \ || fail "default-off remote spawn must not record a traceparent= line" ! grep -q 'export TRACEPARENT=' "$HERDR_LOG" \ From dd330b110e48535fd462a393b076630c548f7dd2 Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 14:53:27 +0800 Subject: [PATCH 4/7] chore(review): stub pi in remote suite, disambiguate muse ultra comment --- bin/fm-spawn.sh | 11 ++++++---- .../fm-remote-secondmate-profile-axes.test.sh | 22 +++++++++++++++++++ 2 files changed, 29 insertions(+), 4 deletions(-) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index b135e14ef7e..25c9ea8421f 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1421,10 +1421,13 @@ effort_flag_for_harness() { muse) # muse 0.1.0-R708.1 --reasoning-effort accepts none|minimal|low|medium| # high|xhigh|ultra and defaults to high, so low..xhigh map straight across. - # ultra is muse's max-CLASS level, so firstmate's max maps onto it - but - # only ever as an EXPLICIT captain choice, never as a fallback, because - # AGENTS.md section 4 forbids selecting max without captain preference and - # the omitted effort here leaves muse on its own high default. muse's extra + # muse's native ultra is muse's own max-CLASS level and is unrelated to the + # Codex-only ultra profile value emitted above. firstmate's max maps onto + # muse's level, but only ever as an EXPLICIT captain choice, never as a + # fallback, because AGENTS.md section 4 forbids selecting max without + # captain preference and the omitted effort here leaves muse on its own + # high default. The Codex-only ultra is therefore deliberately absent from + # the case below and falls through to that same high default. muse's extra # none/minimal levels sit below firstmate's shared vocabulary and are # deliberately unreachable rather than remapped onto low. case "$effort" in diff --git a/tests/fm-remote-secondmate-profile-axes.test.sh b/tests/fm-remote-secondmate-profile-axes.test.sh index 9e43e817e82..b4c75b8f2e4 100755 --- a/tests/fm-remote-secondmate-profile-axes.test.sh +++ b/tests/fm-remote-secondmate-profile-axes.test.sh @@ -36,6 +36,7 @@ REMOTE_HOME="$TMP_ROOT/remote-home" FAKEBIN=$(fm_fakebin "$TMP_ROOT/fake") HERDR_LOG="$TMP_ROOT/remote-herdr.log" HERDR_STATE="$TMP_ROOT/remote-herdr.state" +PI_LOG="$TMP_ROOT/remote-pi.log" TMUX_LOG="$TMP_ROOT/remote-tmux.log" TMUX_STATE="$TMP_ROOT/remote-tmux.state" CLAIMS="$TMP_ROOT/claims" @@ -97,6 +98,23 @@ esac exit 0 SH chmod +x "$REMOTE_ROOT/bin/tmux" + +# bin/fm-remote-job-lib.sh rebuilds the child PATH from scratch and leads it with +# the remote root's own bin, so this stub shadows any host pi and the suite never +# touches a real Pi install. bin/fm-spawn.sh resolves the harness name on that +# PATH and probes the resolved executable with `--help` before composing +# --tui-mode, so the stub answers exactly that probe and logs every call, which +# is how the pi cases below prove the launch resolved here. +cat > "$REMOTE_ROOT/bin/pi" <> '$PI_LOG' +if [ "\${1:-}" = --help ]; then + printf '%s\n' 'Pi 0.84.0' 'Options: --help --tui-mode ' +fi +exit 0 +SH +chmod +x "$REMOTE_ROOT/bin/pi" install_remote_herdr_fixture "$REMOTE_ROOT" "$HERDR_STATE" "$HERDR_LOG" \ "$TMP_ROOT/herdr-send-fail" "$TMP_ROOT/herdr.sock" git -C "$REMOTE_ROOT" init -q -b main @@ -187,6 +205,10 @@ pass "codex: a configured ultra profile crosses the SSH boundary and reaches the # so the omission asserted next is specific to ultra rather than a dead flag. relaunch_with_profile 'pi anthropic/claude-opus-5 max' \ "the remote route refused a configured pi+max profile" +assert_present "$PI_LOG" \ + "the remote pi launch never probed the stub, so it resolved a host pi instead" +assert_grep '--help' "$PI_LOG" \ + "the remote pi launch did not run the stub's help probe" [ "$(meta_axis "$PARENT/state/ios.meta" effort)" = max ] \ || fail "parent metadata did not record the configured remote max effort" assert_grep "--thinking 'max'" "$HERDR_LOG" \ From 55ef906fdb108fd43fc680e3a49923429c0ae3e5 Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 16:49:16 +0800 Subject: [PATCH 5/7] chore(document): point secondmate harness pin at codex-only ultra rules --- docs/configuration.md | 1 + tests/fm-control-relaunch.test.sh | 43 +++++++++++++++++++++++++++ tests/fm-secondmate-harness.test.sh | 45 +++++++++++++++++++++++++++++ 3 files changed, 89 insertions(+) diff --git a/docs/configuration.md b/docs/configuration.md index 02344b63f57..a0f3f1319ff 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -288,6 +288,7 @@ When it is absent or contains `default`, crewmates mirror the firstmate's own ha `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. The first non-empty, non-comment line is parsed as ` [] []`. A bare `` preserves the previous behavior: harness only, with no model or effort launch flag. +The effort token accepts the Codex-only `ultra` as well, and [Crew dispatch profiles](#crew-dispatch-profiles-configcrew-dispatchjson) below owns its Codex pairing and codex-cli floor. When the harness token is absent or `default`, secondmate launch falls back through `config/crew-harness` and then the primary's own harness, and no model or effort is read from that file. `fm-harness.sh secondmate-model` and `fm-harness.sh secondmate-effort` expose only the optional tokens from `config/secondmate-harness`; `config/crew-harness` remains a bare adapter-name file. Changing this pin affects the next secondmate spawn or control-plane relaunch; the relaunch profile rules are owned by [`docs/agent-control.md`](agent-control.md#transactional-relaunch). diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index 29647cc4859..8811f793817 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -658,6 +658,48 @@ test_secondmate_relaunch_picks_up_the_configured_harness_pin() { pass "fm-control relaunch: a secondmate relaunch re-resolves its durable configured harness pin" } +# relaunch resolves config/secondmate-harness through its OWN effort validation, +# not fm-spawn's, so the Codex-only ultra tier needs proving on this path too. +test_secondmate_relaunch_picks_up_a_configured_ultra_pin() { + local dir home out rc launch + dir=$(new_case smultra sm3b) + home="$dir/home" + mkdir -p "$home/config" "$home/data/sm3b" + printf 'codex gpt-5.6-sol ultra\n' > "$home/config/secondmate-harness" + printf '# secondmate brief\n' > "$home/data/sm3b/brief.md" + fm_git_worktree "$dir/proj" "$dir/smhome" sm-branch + mkdir -p "$dir/smhome/state" "$dir/smhome/data" "$dir/smhome/bin" + printf 'sm3b\n' > "$dir/smhome/.fm-secondmate-home" + printf '# agents\n' > "$dir/smhome/AGENTS.md" + { + echo "window=fmses:fm-sm3b" + echo "endpoint_task_id=sm3b" + echo "worktree=$dir/smhome" + echo "project=$dir/smhome" + echo "harness=claude" + echo "kind=secondmate" + echo "mode=secondmate" + echo "yolo=off" + echo "model=default" + echo "effort=default" + echo "home=$dir/smhome" + } > "$home/state/sm3b.meta" + printf '%s\n' "fm-sm3b" > "$dir/fake/windows" + printf '%s' "$dir/smhome" > "$dir/fake/cwd" + printf 'codex' > "$dir/fake/becomes" + out=$(run_control "$dir" sm3b relaunch); rc=$? + expect_code 0 "$rc" "a configured codex+ultra pin should relaunch"$'\n'"$out" + assert_not_contains "$out" "effort token 'ultra'" \ + "relaunch must not reject the configured ultra tier as an unknown token" + [ "$(journal_field "$dir" sm3b to_effort)" = ultra ] \ + || fail "the configured ultra tier should come with the pin, got '$(journal_field "$dir" sm3b to_effort)'" + [ "$(meta_field "$dir" sm3b effort)" = ultra ] || fail "the durable record should retain ultra" + launch=$(cat "$dir/fake/literal") + assert_contains "$launch" "-c 'model_reasoning_effort=\"ultra\"'" \ + "the replacement secondmate launch did not carry the configured ultra tier" + pass "fm-control relaunch: a configured codex ultra pin reaches the replacement secondmate launch" +} + test_secondmate_relaunch_ignores_invalid_configured_effort_before_stop() { local dir home out rc dir=$(new_case invalid-effort sm6) @@ -1346,6 +1388,7 @@ test_prior_harness_turnend_registry_entry_is_cleared test_wiring_removal_failure_refuses_before_replacement_arm test_turnend_auth_paths_are_owned_by_the_control_adapter test_secondmate_relaunch_picks_up_the_configured_harness_pin +test_secondmate_relaunch_picks_up_a_configured_ultra_pin test_secondmate_relaunch_ignores_invalid_configured_effort_before_stop test_secondmate_relaunch_onto_a_crewmate_only_adapter_refuses_before_stop test_explicit_secondmate_harness_ignores_configured_profile_axes diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index 4db6f78169f..65852afcfdc 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -794,6 +794,50 @@ test_spawn_secondmate_harness_model_and_effort_tokens() { pass "C4 spawn: config/secondmate-harness's model+effort tokens thread into the launch and meta" } +# The Codex-only `ultra` tier is reachable from this durable pin, and a pin on a +# harness that has no such level records the request without handing a CLI a +# value it rejects. This route resolves the token separately from the remote one +# (tests/fm-remote-secondmate-profile-axes.test.sh owns that), so both need it. +test_spawn_secondmate_harness_ultra_token() { + local w sm meta launchlog launch out status + w="$TMP_ROOT/spawn-ultra-token" + sm="$w/sm" + launchlog="$w/launch.log" + mkdir -p "$w/home/config" + printf 'codex gpt-5.6-sol ultra\n' > "$w/home/config/secondmate-harness" + make_seeded_home "$sm" sm + + out=$(spawn_secondmate_capture "$w" sm "$sm" "$launchlog" 2>&1); status=$? + expect_code 0 "$status" "a configured codex+ultra secondmate pin should spawn"$'\n'"$out" + assert_not_contains "$out" "effort token 'ultra'" \ + "ultra-token: ultra must not be rejected as an unknown effort token" + meta="$w/home/state/sm.meta" + [ "$(meta_field "$meta" effort)" = ultra ] \ + || fail "ultra-token: meta effort not ultra (got '$(meta_field "$meta" effort)')" + launch=$(cat "$launchlog") + assert_contains "$launch" "codex --model 'gpt-5.6-sol' -c 'model_reasoning_effort=\"ultra\"'" \ + "ultra-token: launch did not thread the configured ultra tier" + + w="$TMP_ROOT/spawn-ultra-token-non-codex" + sm="$w/sm" + launchlog="$w/launch.log" + mkdir -p "$w/home/config" + printf 'claude opus ultra\n' > "$w/home/config/secondmate-harness" + make_seeded_home "$sm" sm + + out=$(spawn_secondmate_capture "$w" sm "$sm" "$launchlog" 2>&1); status=$? + expect_code 0 "$status" "a configured claude+ultra secondmate pin should spawn"$'\n'"$out" + meta="$w/home/state/sm.meta" + [ "$(meta_field "$meta" effort)" = ultra ] \ + || fail "ultra-token: a non-codex pin dropped ultra instead of recording it" + launch=$(cat "$launchlog") + assert_contains "$launch" "--model 'opus'" \ + "ultra-token: a non-codex pin dropped the model alongside the omitted effort" + assert_not_contains "$launch" "--effort" \ + "ultra-token: the Codex-only ultra reached a launch that rejects it" + pass "C4b spawn: config/secondmate-harness pins the Codex-only ultra tier, and a non-codex pin records it without launching it" +} + # Precedence: an explicit per-spawn --model overrides the file's model token. test_spawn_explicit_model_overrides_secondmate_harness_token() { local w sm meta launchlog launch @@ -2564,6 +2608,7 @@ test_spawn_explicit_backend_precedence_over_env_and_inherited_config test_spawn_bare_harness_no_model_effort_flag test_spawn_secondmate_harness_model_token test_spawn_secondmate_harness_model_and_effort_tokens +test_spawn_secondmate_harness_ultra_token test_spawn_explicit_model_overrides_secondmate_harness_token test_spawn_explicit_effort_overrides_secondmate_harness_token test_spawn_explicit_harness_does_not_inherit_secondmate_harness_tokens From 6916caf584045adb0442bb87a3d73402072251eb Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 18:43:32 +0800 Subject: [PATCH 6/7] chore(document): correct Pi effort comment for Codex-only ultra --- bin/fm-spawn.sh | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 25c9ea8421f..5cc792732af 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1412,8 +1412,9 @@ effort_flag_for_harness() { esac ;; pi|pi-signed) - # Pi 0.80.6 accepts the full shared effort vocabulary, including max, through - # its --thinking flag. + # Pi 0.80.6 accepts low through max through its --thinking flag. Pi has no + # ultra concept and its ladder ends at max, so the Codex-only ultra is + # omitted rather than passed to a flag that rejects it. case "$effort" in low|medium|high|xhigh|max) printf -- '--thinking %s ' "$(shell_quote "$effort")" ;; esac From f20591f4d68a1cbebf4ea619d0fb74bae47d4fd3 Mon Sep 17 00:00:00 2001 From: Brandon Goh Date: Wed, 26 Aug 2026 22:54:52 +0800 Subject: [PATCH 7/7] chore(ci): apply CI fixes --- tests/fm-no-mistakes-required.test.sh | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/tests/fm-no-mistakes-required.test.sh b/tests/fm-no-mistakes-required.test.sh index 4807371289a..a481c9fd431 100755 --- a/tests/fm-no-mistakes-required.test.sh +++ b/tests/fm-no-mistakes-required.test.sh @@ -5,7 +5,13 @@ set -u # shellcheck source=tests/lib.sh disable=SC1091 . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" -ACTION_REF=32d396ac0f29135daf7fcb9964aba9d5f4e796d6 +# The gate workflow's pin is the only source of truth for which action commit CI +# runs; a second copy here would let the suite certify an implementation the gate +# does not use. +GATE_WORKFLOW="$ROOT/.github/workflows/no-mistakes-required.yml" +ACTION_REF=$(grep -o 'require-no-mistakes@[0-9a-f]\{40\}' "$GATE_WORKFLOW" | head -1) +ACTION_REF=${ACTION_REF##*@} +[ -n "$ACTION_REF" ] || fail "$GATE_WORKFLOW does not pin require-no-mistakes to a commit SHA" TMP_ROOT=$(fm_test_tmproot fm-no-mistakes-required) VERIFY="$TMP_ROOT/verify.py" OLD_SHA=1111111111111111111111111111111111111111 @@ -40,6 +46,20 @@ test_matching_head_and_completed_steps_pass() { pass "shared action accepts a matching head_sha with completed required steps" } +test_attestation_without_signature_fails() { + local body output rc + body=" +" + rc=0 + output=$(run_verifier "$body" "$NEW_SHA") || rc=$? + [ "$rc" -ne 0 ] || fail "shared action accepted a body carrying no no-mistakes signature line" + assert_contains "$output" "This PR was not raised through no-mistakes." \ + "missing-signature failure did not report the not-raised-via-no-mistakes guidance" + assert_contains "$output" "$SIGNATURE" \ + "missing-signature failure did not quote the exact line the body must carry" + pass "shared action rejects a head-bound complete attestation with no signature line" +} + test_mismatched_head_fails_with_both_shas() { local body output rc body="$SIGNATURE @@ -68,5 +88,6 @@ test_missing_head_fails() { fetch_shared_verifier test_matching_head_and_completed_steps_pass +test_attestation_without_signature_fails test_mismatched_head_fails_with_both_shas test_missing_head_fails