From 0207d54b2827cb80eee5bd8825b550dcd24d54b0 Mon Sep 17 00:00:00 2001 From: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:47:02 -0700 Subject: [PATCH 1/2] fix: refresh model catalogs and preserve delegation intent Generated with Amplifier Co-Authored-By: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> --- .github/workflows/ci.yml | 9 +- .github/workflows/model-catalogs.yml | 36 ++++ README.md | 42 +++++ docs/MATRIX_CURATOR_GUIDE.md | 6 +- .../catalog_audit.py | 90 ++++++++++ .../knob_consistency.py | 26 ++- .../hooks-routing/tests/test_catalog_audit.py | 87 ++++++++++ .../tests/test_knob_consistency.py | 10 +- .../tests/test_knob_consistent_routing.py | 10 +- routing/balanced.yaml | 26 +-- routing/copilot.yaml | 30 ++-- routing/economy.yaml | 26 +-- routing/openai.yaml | 10 +- routing/quality.yaml | 26 +-- scripts/check_model_catalogs.py | 88 ++++++++++ ...esolution_golden_pre_knob_consistency.json | 26 +-- tests/test_catalog_cli.py | 56 ++++++ tests/test_catalog_refresh.py | 161 ++++++++++++++++++ 18 files changed, 680 insertions(+), 85 deletions(-) create mode 100644 .github/workflows/model-catalogs.yml create mode 100644 modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py create mode 100644 modules/hooks-routing/tests/test_catalog_audit.py create mode 100644 scripts/check_model_catalogs.py create mode 100644 tests/test_catalog_cli.py create mode 100644 tests/test_catalog_refresh.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0249b04..abff621 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -78,7 +78,7 @@ jobs: - name: Run root tests run: > uv run --no-project --python ${{ matrix.python-version }} - --with pytest --with pytest-asyncio --with pyyaml + --with pytest --with pytest-asyncio --with pyyaml --with click python -m pytest tests -q --tb=short # --------------------------------------------------------------------------- @@ -137,6 +137,13 @@ jobs: --with git+https://github.com/microsoft/amplifier-foundation python -m pytest -q --tb=short + - name: Run catalog refresh runtime seam tests with real dependencies + working-directory: modules/${{ matrix.module }} + run: > + uv run --frozen --extra dev + --with git+https://github.com/microsoft/amplifier-foundation + python -c "import amplifier_core, amplifier_foundation.spawn_utils, pytest; raise SystemExit(pytest.main(['../../tests/test_catalog_refresh.py', '-q', '--tb=short']))" + # --------------------------------------------------------------------------- # Bundle structure — a cheap YAML parse of bundle.md's frontmatter, every # behaviors/*.yaml and every routing/*.yaml. No network, no keys, ~1 second. diff --git a/.github/workflows/model-catalogs.yml b/.github/workflows/model-catalogs.yml new file mode 100644 index 0000000..eb86a28 --- /dev/null +++ b/.github/workflows/model-catalogs.yml @@ -0,0 +1,36 @@ +name: Live model catalogs + +# No inference. Disabled until dedicated credentials and the enable variable +# are provisioned; missing credentials on a requested check fail, not skip. +on: + workflow_dispatch: + schedule: + - cron: "23 15 * * 1" + +permissions: + contents: read + +jobs: + catalogs: + if: github.event_name == 'workflow_dispatch' || vars.MODEL_CATALOG_CHECKS_ENABLED == 'true' + runs-on: ubuntu-latest + timeout-minutes: 10 + strategy: + fail-fast: false + matrix: + provider: [openai, anthropic, gemini, github-copilot] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: astral-sh/setup-uv@ae62891fec2bb8e7d6c99fc78c9fec3a63790f8d # v10.0.0 + - name: Check fresh provider catalog + env: + OPENAI_API_KEY: ${{ matrix.provider == 'openai' && secrets.OPENAI_API_KEY || '' }} + ANTHROPIC_API_KEY: ${{ matrix.provider == 'anthropic' && secrets.ANTHROPIC_API_KEY || '' }} + GOOGLE_API_KEY: ${{ matrix.provider == 'gemini' && secrets.GOOGLE_API_KEY || '' }} + COPILOT_GITHUB_TOKEN: ${{ matrix.provider == 'github-copilot' && secrets.COPILOT_GITHUB_TOKEN || '' }} + run: > + uv run --no-project --python 3.13 + --with ./modules/hooks-routing --with click + --with git+https://github.com/microsoft/amplifier-module-provider-${{ matrix.provider }}@main + --with git+https://github.com/microsoft/amplifier-core@main + python scripts/check_model_catalogs.py --provider '${{ matrix.provider }}' \ No newline at end of file diff --git a/README.md b/README.md index 68db10f..1c04aa0 100644 --- a/README.md +++ b/README.md @@ -24,6 +24,48 @@ Eight curated matrices ship with this bundle, plus one explicit-name alias Browse the matrix files directly in the [`routing/`](routing/) directory. +### October 2026 catalog refresh + +Copilot pins use Sonnet/Opus 5.5, Luna 6 and Sol 6.1; Copilot vision uses +advertised Sonnet 5.5 instead of the unadvertised Gemini 3.5 Flash pin. OpenAI +Luna globs now accept both dotted and whole-generation IDs without `-fast` +siblings. Terra remains the mid-tier selection; the OpenAI Sol pause remains. + +The `openai` preset canonicalizes the ChatGPT backend through the same provider +family aliases as model selection. When no curated candidate fits the caller's +ceiling, in-family substitution preserves the caller's exact model and effort, +instead of turning a broad classification glob into a `-fast` selection. Explicit +fast callers remain fast. The ladder recognizes GPT-6 Luna, Sol and Astra; +Astra shares the upper ordinal rung for ceiling classification only, not as a +cost-equivalence claim or a new role candidate. + +### Catalog freshness checks + +`scripts/check_model_catalogs.py` is a catalog-only Click wrapper around reusable +checks in the routing module. Install the routing hook, Click and the requested +provider modules, then run: + +```bash +python scripts/check_model_catalogs.py --provider openai --provider anthropic \ + --provider gemini --provider github-copilot +``` + +It fails on missing pins/glob matches, newer same-family standard IDs, or vision +without advertised support. It checks lower-priority candidates too, preserves +the Sol/Terra policy boundary, disables Copilot disk fallback, and never generates +content or edits model settings. Missing/empty/failed catalogs are failures, not +freshness success. Output contains coverage counts and repo-declared patterns, +not credentials, exception bodies or account-specific inventories. + +`.github/workflows/model-catalogs.yml` provides manual dispatch and a weekly +schedule. The schedule is **disabled** unless the repository variable +`MODEL_CATALOG_CHECKS_ENABLED=true` is set and dedicated provider secrets are +provisioned (`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GOOGLE_API_KEY`, +`COPILOT_GITHUB_TOKEN`). The Actions token alone does not grant Copilot access. +ChatGPT OAuth and local Ollama are outside this hosted workflow; their omission +is coverage not provided, not proof of freshness. Ordinary PR tests remain +secret-free. Fresh catalogs do not replace inference smoke tests or quality evals. + ## Including the Bundle **Foundation already includes this bundle** — no extra configuration needed if you use Foundation. diff --git a/docs/MATRIX_CURATOR_GUIDE.md b/docs/MATRIX_CURATOR_GUIDE.md index 99a1f31..b8c4be9 100644 --- a/docs/MATRIX_CURATOR_GUIDE.md +++ b/docs/MATRIX_CURATOR_GUIDE.md @@ -484,6 +484,10 @@ suffixed sibling, a class glob will find it. > glob **selects** one model (narrow is right — it is how `-fast` is excluded); > a ladder glob **classifies** a model already chosen (broad is right — a user > who hand-pins `gpt-5.6-terra-fast` must still land on the terra rung). +> In-family ceiling substitution uses the caller's exact model, not that broad +> classification glob. This prevents an unrequested fast-sibling substitution. +> ChatGPT caller contexts use the canonical OpenAI ladder unless a custom preset +> explicitly declares a backend-specific ladder. **`openai-chatgpt` is `openai` to the resolver.** It is a separate provider MODULE (the OAuth/ChatGPT-subscription backend) — same models, different bill — @@ -529,7 +533,7 @@ reject `thinking_level`. | `gpt-?.?-sol*` | any dotted-version sol / flagship tier (e.g. `gpt-5.6-sol`) | base, terra, mini, nano, luna, pro | | `gpt-?.?-terra*` | any dotted-version terra / mid tier (e.g. `gpt-5.6-terra`) | base, sol, mini, nano, luna, pro | | `gpt-?.?-terra` | the standard terra id ONLY — **the shipped form** | everything above, **plus `-fast` variants and dated snapshots** | -| `gpt-?.?-luna` | the standard luna id ONLY — **the shipped form** | everything above, **plus `-fast` variants and dated snapshots** | +| `gpt-[0-9]*-luna` | standard dotted or whole-generation Luna IDs — **the shipped form** | other named tiers, **plus `-fast` variants and dated snapshots** | | `gpt-?.?-luna*` | any dotted-version luna / cheap-fast tier (e.g. `gpt-5.6-luna`) | base, terra, mini, nano, sol, pro | | `gpt-?.?-mini*` | any dotted-version mini (e.g. `gpt-5.4-mini`) | base, pro, nano, sol, luna, `gpt-5-mini` (no dot) | | `gpt-?.?-nano*` | any dotted-version nano | base, mini, pro, sol, luna | diff --git a/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py b/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py new file mode 100644 index 0000000..9c04ce2 --- /dev/null +++ b/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py @@ -0,0 +1,90 @@ +"""Catalog freshness checks; no inference, configuration writes or model policy.""" + +from __future__ import annotations + +import asyncio +import fnmatch +import re +from typing import Any + +from .resolver import _is_glob, _version_sort_key + + +def _family(model: str) -> str | None: + """Only compare standard IDs within a named class, never cross tiers.""" + match = re.fullmatch(r"gpt-\d+(?:\.\d+)?-(sol|terra|luna)", model) + if match: + return "gpt-" + match[1] + match = re.fullmatch(r"claude-(sonnet|opus|haiku)-\d+(?:[.-]\d+)*", model) + if match: + return "claude-" + match[1] + match = re.fullmatch( + r"gemini-\d+(?:\.\d+)?-(flash|flash-lite|pro-preview|pro-image|flash-image)", model + ) + if match: + return "gemini-" + match[1] + return None + + +async def collect_catalogs(providers: dict[str, Any], timeout: float = 60) -> dict[str, dict]: + """Fetch actual provider menus; callers must disable static/cache fallback. + + Serialize only public model IDs/capabilities and exception class names. + Never serialize provider objects, config, exception text or request headers. + """ + catalogs = {} + for name, provider in providers.items(): + try: + models = await asyncio.wait_for(provider.list_models(), timeout) + catalogs[name] = { + "status": "ok" if models else "empty", + "models": [ + {"id": m.id, "capabilities": list(getattr(m, "capabilities", []))} + for m in models + ], + } + except Exception as error: + catalogs[name] = {"status": "error", "error_type": type(error).__name__, "models": []} + return catalogs + + +def audit_matrix_catalogs(matrices: dict[str, dict], catalogs: dict[str, dict]) -> dict: + """Validate all candidates on requested backends, including lower fallbacks. + + Unrequested providers are outside this check, not evidence of availability. + Exact-pin acceptance in the runtime resolver intentionally remains unchanged. + """ + issues, checked = [], 0 + for backend, catalog in catalogs.items(): + if catalog.get("status") != "ok" or not catalog.get("models"): + issues.append({"kind": "catalog_unavailable", "provider": backend}) + continue + models = {m["id"]: m for m in catalog["models"]} + names = list(models) + for matrix_name, matrix in matrices.items(): + for role, definition in matrix["roles"].items(): + for index, candidate in enumerate(definition["candidates"]): + wanted = candidate["provider"] + if backend != wanted and not (wanted == "openai" and backend == "openai-chatgpt"): + continue + checked += 1 + pattern = candidate["model"] + context = {"provider": backend, "matrix": matrix_name, "role": role, + "candidate": index + 1, "pattern": pattern} + matches = [m for m in names if fnmatch.fnmatch(m.lower(), pattern.lower())] + if not matches: + issues.append({"kind": "missing_glob" if _is_glob(pattern) else "missing_pin", + **context}) + continue + selected = max(matches, key=_version_sort_key) + family = _family(selected) + peers = [m for m in names if family and _family(m) == family] + if peers: + latest = max(peers, key=_version_sort_key) + if _version_sort_key(latest) > _version_sort_key(selected): + issues.append({"kind": "stale_model", **context, + "selected": selected, "latest": latest}) + if role == "vision" and "vision" not in models[selected].get("capabilities", []): + issues.append({"kind": "vision_not_advertised", **context, "selected": selected}) + return {"checked_candidates": checked, "issues": issues, + "status": "fail" if issues or not checked else "pass"} \ No newline at end of file diff --git a/modules/hooks-routing/amplifier_module_hooks_routing/knob_consistency.py b/modules/hooks-routing/amplifier_module_hooks_routing/knob_consistency.py index 6df5ab0..4a2380c 100644 --- a/modules/hooks-routing/amplifier_module_hooks_routing/knob_consistency.py +++ b/modules/hooks-routing/amplifier_module_hooks_routing/knob_consistency.py @@ -605,13 +605,23 @@ def derive_caller_context( module = str(best.get("module") or "") family = module.replace("provider-", "") or str(best.get("id") or "") + # Model selection and tier inheritance must agree about backend aliases. + # Preserve an explicitly declared backend ladder in a custom preset. + if preset is not None and family not in preset.tier_ladder: + from .resolver import PROVIDER_FAMILY_ALIASES + + family = next( + (canonical for canonical, aliases in PROVIDER_FAMILY_ALIASES.items() + if family in aliases and canonical in preset.tier_ladder), + family, + ) effort_key = (preset or Preset()).effort_key_for(family) effort = cfg.get(effort_key) return CallerContext( family=family, model=model, effort=str(effort) if effort is not None else None, - provider_key=str(best.get("id") or module), + provider_key=str(best.get("instance_id") or best.get("id") or module), ) @@ -840,10 +850,20 @@ def plan_candidates( return candidates, record if preset.report_unhonored else None rung_index = min(caller_rung, len(sub_ladder) - 1) + # A ladder classifies suffix variants; it is not a safe selection glob. + # In-family fallback can honor the caller's exact chosen model, including + # an explicitly chosen -fast variant, without inventing a different sibling. + substitute_model = sub_ladder[rung_index][0] + substitute_provider = sub_family + if sub_family == caller.family and rung_of(caller.model, sub_ladder) == rung_index: + substitute_model = caller.model + # Backend-specific models (for example ChatGPT -fast variants) must + # remain on the mount that selected them, not API-first family lookup. + substitute_provider = caller.provider_key or sub_family substitute = _with_effort( { - "provider": sub_family, - "model": sub_ladder[rung_index][0], + "provider": substitute_provider, + "model": substitute_model, "config": dict(original_top.get("config") or {}), }, preset, diff --git a/modules/hooks-routing/tests/test_catalog_audit.py b/modules/hooks-routing/tests/test_catalog_audit.py new file mode 100644 index 0000000..ff1fcf1 --- /dev/null +++ b/modules/hooks-routing/tests/test_catalog_audit.py @@ -0,0 +1,87 @@ +"""Freshness checks must fail closed without changing runtime pin semantics.""" + +from types import SimpleNamespace +from unittest.mock import AsyncMock + +import pytest + +from amplifier_module_hooks_routing.catalog_audit import audit_matrix_catalogs, collect_catalogs + + +def matrices(model, provider="github-copilot", role="general"): + return {"test": {"roles": {role: {"candidates": [{"provider": provider, "model": model}]}}}} + + +def catalog(*models, status="ok"): + return {"status": status, "models": [{"id": m, "capabilities": ["vision"]} for m in models]} + + +def test_absent_exact_pin_is_detected_even_though_runtime_accepts_it(): + result = audit_matrix_catalogs(matrices("missing"), {"github-copilot": catalog("other")}) + assert result["status"] == "fail" + assert result["issues"][0]["kind"] == "missing_pin" + + +def test_stale_pin_compares_only_same_family(): + result = audit_matrix_catalogs(matrices("claude-sonnet-5"), { + "github-copilot": catalog("claude-sonnet-5", "claude-sonnet-5.5", "claude-opus-9") + }) + assert result["issues"][0]["latest"] == "claude-sonnet-5.5" + + +def test_terra_is_not_demoted_for_a_newer_sol(): + result = audit_matrix_catalogs(matrices("gpt-5.6-terra", "openai"), { + "openai": catalog("gpt-5.6-terra", "gpt-6.1-sol") + }) + assert result["status"] == "pass" + + +def test_old_luna_glob_is_detected(): + result = audit_matrix_catalogs(matrices("gpt-?.?-luna", "openai"), { + "openai": catalog("gpt-5.6-luna", "gpt-6-luna") + }) + assert result["issues"][0]["kind"] == "stale_model" + + +def test_new_luna_glob_excludes_fast(): + result = audit_matrix_catalogs(matrices("gpt-[0-9]*-luna", "openai"), { + "openai-chatgpt": catalog("gpt-5.6-luna", "gpt-6-luna", "gpt-6-luna-fast") + }) + assert result["status"] == "pass" and result["checked_candidates"] == 1 + + +@pytest.mark.parametrize("status", ["error", "empty", "static-fallback"]) +def test_unavailable_or_fallback_catalog_never_passes(status): + result = audit_matrix_catalogs(matrices("any"), {"github-copilot": catalog("any", status=status)}) + assert result["status"] == "fail" + assert result["issues"][0]["kind"] == "catalog_unavailable" + + +def test_no_requested_candidate_is_not_success(): + assert audit_matrix_catalogs({}, {"openai": catalog("gpt-6-luna")})["status"] == "fail" + + +def test_vision_requires_advertised_capability(): + data = catalog("claude-sonnet-5.5") + data["models"][0]["capabilities"] = [] + result = audit_matrix_catalogs(matrices("claude-sonnet-5.5", role="vision"), {"github-copilot": data}) + assert result["issues"][0]["kind"] == "vision_not_advertised" + + +@pytest.mark.asyncio +async def test_collection_sanitizes_errors_and_records_real_model_ids(): + providers = { + "bad": SimpleNamespace(list_models=AsyncMock(side_effect=RuntimeError("private credential"))), + "good": SimpleNamespace(list_models=AsyncMock(return_value=[SimpleNamespace(id="model")])), + } + data = await collect_catalogs(providers) + assert data["bad"] == {"status": "error", "error_type": "RuntimeError", "models": []} + assert data["good"]["models"] == [{"id": "model", "capabilities": []}] + + +def test_gemini_standard_class_detects_new_generation_but_not_specialized_ids(): + result = audit_matrix_catalogs(matrices("gemini-[3-9]*-flash", "gemini"), { + "gemini": catalog("gemini-3.8-flash", "gemini-10-flash", + "gemini-omni-100-flash", "gemini-100-flash-image") + }) + assert result["issues"][0]["latest"] == "gemini-10-flash" \ No newline at end of file diff --git a/modules/hooks-routing/tests/test_knob_consistency.py b/modules/hooks-routing/tests/test_knob_consistency.py index 78a76eb..3063e62 100644 --- a/modules/hooks-routing/tests/test_knob_consistency.py +++ b/modules/hooks-routing/tests/test_knob_consistency.py @@ -471,12 +471,12 @@ def test_the_defect_fixed_sol_becomes_terra(self) -> None: EscalationState(), ) assert len(planned) == 1 - assert planned[0]["model"] == "gpt-?.?-terra*" + assert planned[0]["model"] == "gpt-5.6-terra" assert planned[0]["config"][CANONICAL_EFFORT_KEY] == "medium" assert record is not None assert record.honored is True assert record.requested_model == "gpt-?.?-sol*" - assert record.granted_model == "gpt-?.?-terra*" + assert record.granted_model == "gpt-5.6-terra" assert "substituted the ladder rung" in record.reason def test_candidate_already_below_ceiling_is_kept(self) -> None: @@ -536,7 +536,7 @@ def test_strict_denies_escalation_even_for_allowed_roles(self) -> None: planned, record = plan_candidates( "reasoning", REASONING_CANDIDATES, TERRA_CALLER, preset, state ) - assert planned[0]["model"] == "gpt-?.?-terra*" + assert planned[0]["model"] == "gpt-5.6-terra" assert state.used == 0 assert record is not None and record.escalated is False @@ -597,7 +597,7 @@ def test_budget_exhausts_then_clamps(self) -> None: "reasoning", REASONING_CANDIDATES, TERRA_CALLER, preset, state ) assert first[0]["model"] == "gpt-?.?-sol*" - assert second[0]["model"] == "gpt-?.?-terra*" + assert second[0]["model"] == "gpt-5.6-terra" assert state.remaining == 0 assert record is not None and record.escalated is False @@ -611,7 +611,7 @@ def test_role_not_on_allow_list_never_escalates(self) -> None: preset, state, ) - assert planned[0]["model"] == "gpt-?.?-terra*" + assert planned[0]["model"] == "gpt-5.6-terra" assert state.used == 0 def test_escalation_not_consumed_when_candidate_is_already_below(self) -> None: diff --git a/modules/hooks-routing/tests/test_knob_consistent_routing.py b/modules/hooks-routing/tests/test_knob_consistent_routing.py index 6597406..2c6030d 100644 --- a/modules/hooks-routing/tests/test_knob_consistent_routing.py +++ b/modules/hooks-routing/tests/test_knob_consistent_routing.py @@ -93,7 +93,7 @@ def _providers() -> dict[str, Any]: TERRA_CALLER = CallerContext( - family="openai", model="gpt-5.6-terra", effort="medium", provider_key="terra" + family="openai", model="gpt-5.6-terra", effort="medium", provider_key="openai" ) @@ -231,7 +231,7 @@ async def sink(record: Any) -> None: ) assert len(seen) == 1 assert seen[0].role == "reasoning" - assert seen[0].granted_model == "gpt-?.?-terra*" + assert seen[0].granted_model == "gpt-5.6-terra" @pytest.mark.asyncio async def test_no_record_when_nothing_resolves(self) -> None: @@ -296,7 +296,7 @@ def _coordinator_for_resolver( config[CANONICAL_EFFORT_KEY] = effort coordinator = MagicMock() coordinator.config = { - "providers": [{"module": "provider-openai", "id": "terra", "config": config}] + "providers": [{"module": "provider-openai", "id": "openai", "config": config}] } return coordinator @@ -487,7 +487,7 @@ def _get(key: str) -> Any: "providers": [ { "module": "provider-openai", - "id": "terra", + "id": "openai", "config": provider_config if provider_config is not None else { @@ -629,7 +629,7 @@ async def test_clamp_event_is_emitted_not_injected(self, tmp_path: Path) -> None payload = emitted[0][1] assert payload["role"] == "reasoning" assert payload["requested"]["model"] == "gpt-?.?-sol*" - assert payload["granted"]["model"] == "gpt-?.?-terra*" + assert payload["granted"]["model"] == "gpt-5.6-terra" # provider:request must NOT have been turned into an injection carrier # for this record -- it goes to the event log only. assert "context_injection" not in payload diff --git a/routing/balanced.yaml b/routing/balanced.yaml index 84f8f69..4eb3711 100644 --- a/routing/balanced.yaml +++ b/routing/balanced.yaml @@ -125,7 +125,7 @@ name: balanced description: "Quality/cost balance for mixed workloads. Curated by Amplifier Foundation team." -updated: "2026-09-07" +updated: "2026-10-02" roles: general: @@ -146,7 +146,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium - provider: github-copilot @@ -160,7 +160,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -196,7 +196,7 @@ roles: config: thinking_budget_tokens: 32000 - provider: github-copilot - model: gpt-5.6-luna + model: gpt-6-luna config: reasoning_effort: medium - provider: ollama @@ -220,7 +220,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium - provider: github-copilot @@ -234,7 +234,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: max - provider: anthropic @@ -248,7 +248,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium - provider: github-copilot @@ -296,7 +296,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: high - provider: anthropic @@ -344,7 +344,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium @@ -366,7 +366,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium @@ -388,7 +388,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium - provider: anthropic @@ -424,7 +424,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -452,6 +452,6 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium diff --git a/routing/copilot.yaml b/routing/copilot.yaml index 244e79c..1f866a2 100644 --- a/routing/copilot.yaml +++ b/routing/copilot.yaml @@ -21,14 +21,18 @@ name: copilot description: "GitHub Copilot-only routing. Strongest model per role across the Claude, GPT and Gemini families Copilot serves." -updated: "2026-07-28" +updated: "2026-10-02" + +# October catalog refresh: Claude 5.5, Sol 6.1 and advertised Sonnet vision. +# The Sol pause applies to OpenAI-family routing, not this existing Copilot +# security-audit specialist. This refresh does not add Sol to other matrices. roles: general: description: "Versatile catch-all, no specialization needed" candidates: - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: high @@ -44,7 +48,7 @@ roles: description: "Code generation, implementation, debugging" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: xhigh @@ -52,7 +56,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: xhigh @@ -60,7 +64,7 @@ roles: description: "Vulnerability assessment, attack surface analysis, code auditing" candidates: - provider: github-copilot - model: gpt-5.6-sol + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -68,7 +72,7 @@ roles: description: "Deep architectural reasoning, system design, complex multi-step analysis" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: max @@ -76,7 +80,7 @@ roles: description: "Analytical evaluation — finding flaws in existing work, not generating solutions" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: xhigh @@ -84,7 +88,7 @@ roles: description: "Design direction, aesthetic judgment, high-quality creative output" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: high @@ -92,7 +96,7 @@ roles: description: "Long-form content — documentation, marketing, case studies, storytelling" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: high @@ -100,7 +104,7 @@ roles: description: "Deep investigation, information synthesis across multiple sources" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: high @@ -108,7 +112,7 @@ roles: description: "Understanding visual input — screenshots, diagrams, UI mockups" candidates: - provider: github-copilot - model: gemini-3.5-flash + model: claude-sonnet-5.5 config: reasoning_effort: high @@ -118,12 +122,12 @@ roles: description: "Image generation (limited — no native Copilot image API)" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 critical-ops: description: "High-reliability operational tasks — infrastructure, orchestration, coordination" candidates: - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: xhigh diff --git a/routing/economy.yaml b/routing/economy.yaml index b218388..e898797 100644 --- a/routing/economy.yaml +++ b/routing/economy.yaml @@ -122,14 +122,14 @@ name: economy description: "Budget-first routing. Cheap tier for utility roles, mid-tier where quality matters." -updated: "2026-09-07" +updated: "2026-10-02" roles: general: description: "Versatile catch-all, no specialization needed" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -173,7 +173,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -211,7 +211,7 @@ roles: config: thinking_budget_tokens: 32000 - provider: github-copilot - model: gpt-5.6-luna + model: gpt-6-luna config: reasoning_effort: medium - provider: ollama @@ -221,7 +221,7 @@ roles: description: "Code generation, implementation, debugging" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -263,7 +263,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: max - provider: anthropic @@ -317,7 +317,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: high @@ -339,7 +339,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium - provider: github-copilot @@ -387,7 +387,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -409,7 +409,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -431,7 +431,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -458,7 +458,7 @@ roles: config: thinking_budget_tokens: 32000 - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -499,6 +499,6 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium diff --git a/routing/openai.yaml b/routing/openai.yaml index 6041338..a64bf5d 100644 --- a/routing/openai.yaml +++ b/routing/openai.yaml @@ -151,7 +151,7 @@ name: openai description: "OpenAI routing for BOTH backends -- the pay-per-use API (`openai`) and the ChatGPT subscription (`openai-chatgpt`). Flagship (sol) tier paused pending evals; knob-consistent delegation ON by default -- sub-agents inherit the caller's tier/effort ceiling." -updated: "2026-09-07" +updated: "2026-10-02" # --------------------------------------------------------------------------- # Knob-consistent delegation preset -- DEFAULT ON for this matrix. @@ -180,9 +180,9 @@ preset: # covers both. tier_ladder: openai: - - ["gpt-?.?-luna*", "gpt-?.?-mini*", "gpt-?.?-nano*"] + - ["gpt-[0-9]*-luna", "gpt-[0-9]*-luna*", "gpt-?.?-mini*", "gpt-?.?-nano*"] - ["gpt-?.?-terra*"] - - ["gpt-?.?-sol*", "gpt-[0-9].[0-9]"] + - ["gpt-[0-9]*-sol*", "gpt-[0-9]*-astra*", "gpt-[0-9].[0-9]"] delegation: # none | effort | tier-and-effort | strict @@ -206,7 +206,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: medium @@ -222,7 +222,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: max diff --git a/routing/quality.yaml b/routing/quality.yaml index 5ca87a2..03b27f0 100644 --- a/routing/quality.yaml +++ b/routing/quality.yaml @@ -123,7 +123,7 @@ name: quality description: "Best available models. Prioritizes capability over cost." -updated: "2026-09-07" +updated: "2026-10-02" roles: general: @@ -144,7 +144,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -166,7 +166,7 @@ roles: thinking_config: thinking_level: low - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -188,7 +188,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -196,7 +196,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: gpt-?.?-luna + model: "gpt-[0-9]*-luna" config: reasoning_effort: max - provider: anthropic @@ -210,7 +210,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -254,7 +254,7 @@ roles: config: reasoning_effort: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: high @@ -276,7 +276,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: xhigh @@ -298,7 +298,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium @@ -320,7 +320,7 @@ roles: thinking_config: thinking_level: medium - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium @@ -342,7 +342,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium @@ -364,7 +364,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: claude-sonnet-5 + model: claude-sonnet-5.5 config: reasoning_effort: medium @@ -392,6 +392,6 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: claude-opus-5 + model: claude-opus-5.5 config: reasoning_effort: medium diff --git a/scripts/check_model_catalogs.py b/scripts/check_model_catalogs.py new file mode 100644 index 0000000..a693438 --- /dev/null +++ b/scripts/check_model_catalogs.py @@ -0,0 +1,88 @@ +"""Thin catalog-only CLI. Install requested provider modules before running.""" + +import asyncio +import json +import logging +import os +from contextlib import ExitStack +from datetime import datetime, timezone +from pathlib import Path +from unittest.mock import patch + +import click +import yaml + +from amplifier_module_hooks_routing.catalog_audit import audit_matrix_catalogs, collect_catalogs + + +def make_provider(name): + if name == "openai": + from amplifier_module_provider_openai import OpenAIProvider + return OpenAIProvider(os.environ.get("OPENAI_API_KEY", "")) + if name == "anthropic": + from amplifier_module_provider_anthropic import AnthropicProvider + # Audit the full menu before the provider's lexical family prefilter. + return AnthropicProvider(os.environ.get("ANTHROPIC_API_KEY", ""), config={"filtered": False}) + if name == "gemini": + from amplifier_module_provider_gemini import GeminiProvider + return GeminiProvider(os.environ.get("GOOGLE_API_KEY") or os.environ.get("GEMINI_API_KEY", "")) + from amplifier_module_provider_github_copilot import GitHubCopilotProvider + return GitHubCopilotProvider() + + +async def check(names, routing_dir): + providers = {} + catalogs = {} + with ExitStack() as stack: + if "github-copilot" in names: + # A cached catalog cannot certify today's pin availability. + stack.enter_context(patch( + "amplifier_module_provider_github_copilot.provider.read_cache", return_value=None + )) + stack.enter_context(patch("amplifier_module_provider_github_copilot.provider.write_cache")) + for name in names: + try: + providers[name] = make_provider(name) + except Exception as error: + catalogs[name] = {"status": "error", "error_type": type(error).__name__, "models": []} + try: + catalogs.update(await collect_catalogs(providers)) + finally: + for provider in providers.values(): + try: + await asyncio.wait_for(provider.close(), 10) + except Exception: + pass + matrices = {p.stem: yaml.safe_load(p.read_text()) for p in routing_dir.glob("*.yaml")} + if not matrices: + raise click.ClickException("No routing matrices found") + report = audit_matrix_catalogs(matrices, catalogs) + # Public CI must not upload account-specific inventories or private new IDs. + # Detailed comparisons remain in-memory; emit only repo-owned patterns. + for issue in report["issues"]: + issue.pop("selected", None) + issue.pop("latest", None) + report.update( + observed_at=datetime.now(timezone.utc).isoformat(), + coverage={name: {"status": data["status"], "count": len(data["models"])} + for name, data in catalogs.items()}, + ) + return report + + +@click.command() +@click.option("--provider", "providers", multiple=True, required=True, + type=click.Choice(["openai", "anthropic", "gemini", "github-copilot"])) +@click.option("--routing-dir", type=click.Path(path_type=Path, file_okay=False, exists=True), + default=Path(__file__).resolve().parents[1] / "routing") +def main(providers, routing_dir): + """Fail on missing/stale candidates using fresh menus, without calling an LLM.""" + # Provider exception text can contain endpoints; expose sanitized JSON only. + logging.disable(logging.CRITICAL) + report = asyncio.run(check(tuple(dict.fromkeys(providers)), routing_dir)) + click.echo(json.dumps(report, indent=2)) + raise SystemExit(0 if report["status"] == "pass" else 1) + + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/tests/fixtures/resolution_golden_pre_knob_consistency.json b/tests/fixtures/resolution_golden_pre_knob_consistency.json index d102ff9..52348d8 100644 --- a/tests/fixtures/resolution_golden_pre_knob_consistency.json +++ b/tests/fixtures/resolution_golden_pre_knob_consistency.json @@ -263,7 +263,7 @@ "config": { "reasoning_effort": "max" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -272,7 +272,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -281,7 +281,7 @@ "config": { "reasoning_effort": "high" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -290,7 +290,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -299,7 +299,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -315,14 +315,14 @@ "config": { "reasoning_effort": "high" }, - "model": "claude-sonnet-5", + "model": "claude-sonnet-5.5", "provider": "github-copilot" } ], "image-gen": [ { "config": {}, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -331,7 +331,7 @@ "config": { "reasoning_effort": "max" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -340,7 +340,7 @@ "config": { "reasoning_effort": "high" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -349,7 +349,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-sol", + "model": "gpt-6.1-sol", "provider": "github-copilot" } ], @@ -358,7 +358,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ], @@ -367,7 +367,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gemini-3.5-flash", + "model": "claude-sonnet-5.5", "provider": "github-copilot" } ], @@ -376,7 +376,7 @@ "config": { "reasoning_effort": "high" }, - "model": "claude-opus-5", + "model": "claude-opus-5.5", "provider": "github-copilot" } ] diff --git a/tests/test_catalog_cli.py b/tests/test_catalog_cli.py new file mode 100644 index 0000000..de6afb7 --- /dev/null +++ b/tests/test_catalog_cli.py @@ -0,0 +1,56 @@ +"""Public CLI output must not expose private catalogs or SDK exception text.""" + +import importlib.util +import sys +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import AsyncMock + +import yaml +from click.testing import CliRunner + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "modules" / "hooks-routing")) + + +def cli_module(): + spec = importlib.util.spec_from_file_location("catalog_cli", ROOT / "scripts/check_model_catalogs.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_cli_only_emits_repo_patterns_not_private_ids(tmp_path, monkeypatch): + module = cli_module() + path = tmp_path / "test.yaml" + path.write_text(yaml.safe_dump({"roles": {"general": {"candidates": [ + {"provider": "openai", "model": "gpt-5.6-luna"}, + ]}}})) + provider = SimpleNamespace( + list_models=AsyncMock(return_value=[ + SimpleNamespace(id="gpt-5.6-luna", capabilities=[]), + SimpleNamespace(id="gpt-99-luna", capabilities=[]), + SimpleNamespace(id="private-project-model", capabilities=[]), + ]), + close=AsyncMock(), + ) + monkeypatch.setattr(module, "make_provider", lambda name: provider) + result = CliRunner().invoke(module.main, ["--provider", "openai", "--routing-dir", str(tmp_path)]) + assert result.exit_code == 1 + assert "stale_model" in result.output and "gpt-5.6-luna" in result.output + assert "gpt-99-luna" not in result.output and "private-project-model" not in result.output + + +def test_cli_does_not_print_sensitive_error_body(tmp_path, monkeypatch): + module = cli_module() + (tmp_path / "test.yaml").write_text(yaml.safe_dump({"roles": {"general": {"candidates": [ + {"provider": "openai", "model": "gpt-6-luna"}, + ]}}})) + provider = SimpleNamespace( + list_models=AsyncMock(side_effect=RuntimeError("sensitive SDK exception body")), + close=AsyncMock(), + ) + monkeypatch.setattr(module, "make_provider", lambda name: provider) + result = CliRunner().invoke(module.main, ["--provider", "openai", "--routing-dir", str(tmp_path)]) + assert result.exit_code == 1 and "catalog_unavailable" in result.output + assert "sensitive SDK exception body" not in result.output \ No newline at end of file diff --git a/tests/test_catalog_refresh.py b/tests/test_catalog_refresh.py new file mode 100644 index 0000000..5d9b7ba --- /dev/null +++ b/tests/test_catalog_refresh.py @@ -0,0 +1,161 @@ +"""October catalog refresh: test outcomes, not just model-pattern spelling.""" + +import sys +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import AsyncMock + +import pytest +import yaml + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "modules" / "hooks-routing")) + +from amplifier_module_hooks_routing.knob_consistency import ( # noqa: E402 + CallerContext, + parse_preset, + rung_of, +) +from amplifier_module_hooks_routing.resolver import resolve_model_role # noqa: E402 +from amplifier_module_hooks_routing.resolver_class import MatrixModelRoleResolver # noqa: E402 + + +def matrix(name): + return yaml.safe_load((ROOT / "routing" / f"{name}.yaml").read_text()) + + +def provider(ids): + return SimpleNamespace(list_models=AsyncMock(return_value=[SimpleNamespace(id=i) for i in ids])) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("backend", ["openai", "openai-chatgpt"]) +@pytest.mark.parametrize("name", ["balanced", "quality", "economy", "openai"]) +async def test_luna_selection_tracks_gpt6_without_selecting_fast(name, backend): + data = matrix(name) + mounted = {backend: provider(["gpt-5.6-luna", "gpt-6-luna", "gpt-6-luna-fast"])} + checked = 0 + for role, definition in data["roles"].items(): + for candidate in definition["candidates"]: + if candidate["provider"] == "openai" and candidate["model"].endswith("-luna"): + result = await resolve_model_role( + [role], {role: {"candidates": [candidate]}}, mounted + ) + assert result[0]["model"] == "gpt-6-luna" + checked += 1 + assert checked > 0 + + +@pytest.mark.parametrize("model,rung", [ + ("gpt-6-luna", 0), ("gpt-6-luna-fast", 0), + ("gpt-6-sol", 2), ("gpt-6-astra", 2), ("gpt-6-astra-fast", 2), + ("gpt-6.1-sol", 2), ("gpt-5.6-terra", 1), +]) +def test_current_gpt6_models_are_classified(model, rung): + preset = parse_preset(matrix("openai")) + assert rung_of(model, preset.tier_ladder["openai"]) == rung + + +@pytest.mark.asyncio +@pytest.mark.parametrize("backend", ["openai", "openai-chatgpt"]) +@pytest.mark.parametrize("model", ["gpt-5.6-luna", "gpt-6-luna", "gpt-6-luna-fast"]) +async def test_default_derived_strict_context_keeps_exact_caller_model(backend, model): + pytest.importorskip("amplifier_foundation.spawn_utils") + data = matrix("openai") + mounted = {backend: provider([ + "gpt-5.6-terra", "gpt-5.6-luna", "gpt-5.6-luna-fast", "gpt-6-luna", "gpt-6-luna-fast", + ])} + plan = {"providers": [{ + "module": f"provider-{backend}", "id": backend, + "config": {"priority": 0, "default_model": model, "reasoning_effort": "medium"}, + }]} + coordinator = SimpleNamespace(config=plan) + resolver = MatrixModelRoleResolver( + data["roles"], mounted, "openai", coordinator, preset=parse_preset(data) + ) + prefs = await resolver.resolve("reasoning") # No manufactured canonical caller. + assert prefs[0].provider == backend + assert prefs[0].model == model + assert prefs[0].config["reasoning_effort"] == "medium" + assert resolver.clamp_records[-1].honored + + +@pytest.mark.asyncio +async def test_explicit_context_does_not_substitute_fast_sibling(): + data = matrix("openai") + result = await resolve_model_role( + ["reasoning"], data["roles"], + {"openai-chatgpt": provider(["gpt-5.6-luna", "gpt-5.6-luna-fast"])}, + caller_context=CallerContext("openai", "gpt-5.6-luna", "medium"), + preset=parse_preset(data), + ) + assert result[0]["model"] == "gpt-5.6-luna" + + +def test_copilot_pins_refresh_all_matrices(): + forbidden = {"claude-sonnet-5", "claude-opus-5", "gpt-5.6-luna", "gpt-5.6-sol", "gemini-3.5-flash"} + for path in (ROOT / "routing").glob("*.yaml"): + data = yaml.safe_load(path.read_text()) + for role in data["roles"].values(): + for candidate in role["candidates"]: + if candidate["provider"] == "github-copilot": + assert candidate["model"] not in forbidden, path.name + vision = matrix("copilot")["roles"]["vision"]["candidates"][0] + assert vision["model"] == "claude-sonnet-5.5" + + +def test_sol_pause_stays_in_place(): + for name in ["balanced", "quality", "economy", "openai"]: + for role in matrix(name)["roles"].values(): + assert all("-sol" not in c["model"] and "-astra" not in c["model"] + for c in role["candidates"] if c["provider"] == "openai") + + +@pytest.mark.asyncio +async def test_exact_chatgpt_fallback_preserves_origin_with_both_backends(): + data = matrix("openai") + result = await resolve_model_role( + ["reasoning"], data["roles"], + {"openai": provider(["gpt-6-luna"]), "subscription": provider(["gpt-6-luna-fast"])}, + caller_context=CallerContext("openai", "gpt-6-luna-fast", "medium", "subscription"), + preset=parse_preset(data), + ) + assert result[0]["provider"] == "subscription" + assert result[0]["model"] == "gpt-6-luna-fast" + + +def test_explicit_backend_ladder_is_not_normalized(): + from amplifier_module_hooks_routing.knob_consistency import derive_caller_context + data = matrix("openai") + data["preset"]["tier_ladder"]["openai-chatgpt"] = [["gpt-6-luna*"]] + coordinator = SimpleNamespace(config={"providers": [{ + "module": "provider-openai-chatgpt", "id": "subscription", + "config": {"default_model": "gpt-6-luna"}, + }]}) + caller = derive_caller_context(coordinator, parse_preset(data)) + assert caller.family == "openai-chatgpt" and caller.provider_key == "subscription" + + +@pytest.mark.asyncio +async def test_derived_exact_fallback_prefers_instance_id_over_conflicting_id(): + pytest.importorskip("amplifier_foundation.spawn_utils") + data = matrix("openai") + mounted = {"openai": provider(["gpt-6-luna"]), + "subscription": provider(["gpt-6-luna-fast"])} + coordinator = SimpleNamespace(config={"providers": [ + {"module": "provider-openai", "id": "openai", + "config": {"priority": 1, "default_model": "gpt-5.6-terra"}}, + {"module": "provider-openai-chatgpt", "instance_id": "subscription", "id": "openai", + "config": {"priority": 0, "default_model": "gpt-6-luna-fast", "reasoning_effort": "medium"}}, + ]}) + resolver = MatrixModelRoleResolver(data["roles"], mounted, "openai", coordinator, + preset=parse_preset(data)) + preferences = await resolver.resolve("reasoning") + assert preferences[0].provider == "subscription" + assert preferences[0].model == "gpt-6-luna-fast" + from amplifier_foundation.spawn_utils import apply_provider_preferences_with_resolution + coordinator.get = lambda key: mounted if key == "providers" else None + child = await apply_provider_preferences_with_resolution(coordinator.config, preferences, coordinator) + assert child["providers"][1]["config"]["default_model"] == "gpt-6-luna-fast" + assert child["providers"][1]["config"]["priority"] == 0 + assert child["providers"][0]["config"]["priority"] > 0 \ No newline at end of file From 0d1fb32625c9582b16e89d90f8b46b42eacfcf7c Mon Sep 17 00:00:00 2001 From: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> Date: Fri, 2 Oct 2026 07:50:32 -0700 Subject: [PATCH 2/2] fix: ship interim Sol 6.1 routing and complete freshness guards Generated with Amplifier Co-Authored-By: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> --- README.md | 16 +- docs/MATRIX_CURATOR_GUIDE.md | 15 +- .../catalog_audit.py | 15 +- .../hooks-routing/tests/test_catalog_audit.py | 26 +++- .../tests/test_knob_consistent_routing.py | 25 ++- routing/balanced.yaml | 46 +++--- routing/copilot.yaml | 4 +- routing/economy.yaml | 35 +++-- routing/openai.yaml | 66 ++++---- routing/quality.yaml | 36 +++-- ...esolution_golden_pre_knob_consistency.json | 144 ++++++++---------- tests/test_catalog_refresh.py | 26 +++- tests/test_single_provider_coverage.py | 6 + 13 files changed, 244 insertions(+), 216 deletions(-) diff --git a/README.md b/README.md index 1c04aa0..d1d3a5f 100644 --- a/README.md +++ b/README.md @@ -15,7 +15,7 @@ Eight curated matrices ship with this bundle, plus one explicit-name alias | **quality** | Maximum capability. Uses the strongest models for every role, regardless of cost. | | **economy** | Cost-optimized. Prefers free tiers, smaller models, and local providers like Ollama. | | **anthropic** | Anthropic Claude models exclusively. No knob-consistent delegation -- no measured win for this family yet (the "Anthropic guardrail"). | -| **openai** | OpenAI models exclusively. **Ships knob-consistent delegation ON by default** (2026-09-02, a measured win -- see below). Covers **both OpenAI backends** — the pay-per-use API (`openai`) and the ChatGPT subscription (`openai-chatgpt`) — with a single candidate per role: the resolver treats them as one family, API key first. The flagship `sol` tier is PAUSED as of 2026-09-07 pending further evals; roles that used it now run `terra` one effort notch higher, and `ui-coding` runs `luna` at `max`. | +| **openai** | OpenAI models exclusively, API-first across both backends. Knob-consistent delegation is ON by default. The interim October 2 update selects Sol 6.1 instead of Terra and current Luna for fast/UI roles; `ui-coding` keeps `max` effort. Comparative selection evals follow separately. | | **gemini** | Google Gemini models exclusively. | | **copilot** | Strongest model per role across the Claude, GPT and Gemini families Copilot serves, through a single provider. | | **ollama** | A **template**, deliberately minimal: only the two required roles (`general`, `fast`), both `model: "*"`. Every Ollama user has pulled a different set of models, so there is no useful curation to ship -- copy it and pin what you actually have. | @@ -28,8 +28,12 @@ Browse the matrix files directly in the [`routing/`](routing/) directory. Copilot pins use Sonnet/Opus 5.5, Luna 6 and Sol 6.1; Copilot vision uses advertised Sonnet 5.5 instead of the unadvertised Gemini 3.5 Flash pin. OpenAI -Luna globs now accept both dotted and whole-generation IDs without `-fast` -siblings. Terra remains the mid-tier selection; the OpenAI Sol pause remains. +Luna globs now accept both dotted and whole-generation GPT-6+ IDs without `-fast` +siblings. The subsequent approved interim patch replaces every shipped Terra +candidate (including mixed-matrix Copilot fallbacks) with exact `gpt-6.1-sol`. +This supersedes the September Sol pause. Role ordering and efforts are retained; +this is an opportunistic model refresh, not evidence of a comparative eval win. +Legacy Terra remains only in caller classification and historical examples. The `openai` preset canonicalizes the ChatGPT backend through the same provider family aliases as model selection. When no curated candidate fits the caller's @@ -52,7 +56,7 @@ python scripts/check_model_catalogs.py --provider openai --provider anthropic \ It fails on missing pins/glob matches, newer same-family standard IDs, or vision without advertised support. It checks lower-priority candidates too, preserves -the Sol/Terra policy boundary, disables Copilot disk fallback, and never generates +same-family comparisons, disables Copilot disk fallback, and never generates content or edits model settings. Missing/empty/failed catalogs are failures, not freshness success. Output contains coverage counts and repo-declared patterns, not credentials, exception bodies or account-specific inventories. @@ -178,9 +182,9 @@ Levels 1 and 2 stay strictly above level 3, so an author who deliberately pinned preset: tier_ladder: # cheapest -> most expensive, declared openai: - - ["gpt-?.?-luna*", "gpt-?.?-mini*", "gpt-?.?-nano*"] + - ["gpt-[6-9]*-luna", "gpt-[0-9]*-luna*", "gpt-?.?-mini*", "gpt-?.?-nano*"] - ["gpt-?.?-terra*"] - - ["gpt-?.?-sol*", "gpt-[0-9].[0-9]"] + - ["gpt-[0-9]*-sol*", "gpt-[0-9]*-astra*", "gpt-[0-9].[0-9]"] delegation: inherit: strict # none | effort | tier-and-effort | strict report_unhonored: true diff --git a/docs/MATRIX_CURATOR_GUIDE.md b/docs/MATRIX_CURATOR_GUIDE.md index b8c4be9..073b026 100644 --- a/docs/MATRIX_CURATOR_GUIDE.md +++ b/docs/MATRIX_CURATOR_GUIDE.md @@ -459,7 +459,10 @@ After reviewing benchmark data and weather report alignment, follow this 3-step ### Pin Model Names -Always use exact, versioned model names in matrix files. Globs are for user overrides and local providers only. +Use exact versioned names for deliberate pins, or bounded class-scoped globs +when live discovery is supported. Never use a broad glob as a tier policy. +The October 2026 interim update pins Sol 6.1, replacing Terra selections while +retaining role efforts. This approved opportunistic choice is not a quality eval. **Good ✅ — class-scoped globs** for providers whose `list_models()` is backed by a live API. These auto-track new releases within a class without silently @@ -532,8 +535,8 @@ reject `thinking_level`. > the base alias explicitly. | `gpt-?.?-sol*` | any dotted-version sol / flagship tier (e.g. `gpt-5.6-sol`) | base, terra, mini, nano, luna, pro | | `gpt-?.?-terra*` | any dotted-version terra / mid tier (e.g. `gpt-5.6-terra`) | base, sol, mini, nano, luna, pro | -| `gpt-?.?-terra` | the standard terra id ONLY — **the shipped form** | everything above, **plus `-fast` variants and dated snapshots** | -| `gpt-[0-9]*-luna` | standard dotted or whole-generation Luna IDs — **the shipped form** | other named tiers, **plus `-fast` variants and dated snapshots** | +| `gpt-?.?-terra` | legacy standard Terra IDs for custom overrides; no longer a stock candidate | everything above, **plus `-fast` variants and dated snapshots** | +| `gpt-[6-9]*-luna` | standard dotted or whole-generation Luna IDs, generation 6–9 — **the shipped form** | pre-6 and other named tiers, **plus `-fast` variants and dated snapshots** | | `gpt-?.?-luna*` | any dotted-version luna / cheap-fast tier (e.g. `gpt-5.6-luna`) | base, terra, mini, nano, sol, pro | | `gpt-?.?-mini*` | any dotted-version mini (e.g. `gpt-5.4-mini`) | base, pro, nano, sol, luna, `gpt-5-mini` (no dot) | | `gpt-?.?-nano*` | any dotted-version nano | base, mini, pro, sol, luna | @@ -631,10 +634,10 @@ Different providers use different naming conventions for the **same underlying m | Claude Sonnet 4.x | `claude-sonnet-*` (glob) | — | — | `claude-sonnet-4.6` (pin) | | Claude Opus 4.x | `claude-opus-*` (glob) | — | — | `claude-opus-4.8` (pin) | | Claude Haiku 4.x | `claude-haiku-*` (glob) | — | — | `claude-haiku-4.5` (pin) | -| GPT mid-tier (terra) | — | `gpt-?.?-terra` (glob, **no `*`**) | — | pinned, e.g. `gpt-5.6-terra` | +| GPT mid-tier (terra) | — | legacy caller classification only | — | no stock Terra pin | | GPT base / pre-5.6 migration fallback | — | `gpt-[0-9].[0-9]` (glob) | — | — | -| GPT flagship (sol) | — | `gpt-?.?-sol*` (glob) | — | pinned, e.g. `gpt-5.6-sol` | -| GPT cheap-fast (luna) | — | `gpt-?.?-luna` (glob, **no `*`**) | — | pinned, e.g. `gpt-5.6-luna` | +| GPT flagship (sol) | — | exact `gpt-6.1-sol` | — | exact `gpt-6.1-sol` | +| GPT cheap-fast (luna) | — | `gpt-[6-9]*-luna` (glob, **no trailing `*`**) | — | exact `gpt-6-luna` | | GPT-5.x mini | — | `gpt-?.?-mini*` (glob) | — | pinned, e.g. `gpt-5.4-mini` | | Gemini Pro | — | — | `gemini-[3-9]*-pro-preview` (glob) | — | | Gemini Flash | — | — | `gemini-[3-9]*-flash` (glob) | — | diff --git a/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py b/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py index 9c04ce2..f7f15ef 100644 --- a/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py +++ b/modules/hooks-routing/amplifier_module_hooks_routing/catalog_audit.py @@ -19,13 +19,20 @@ def _family(model: str) -> str | None: if match: return "claude-" + match[1] match = re.fullmatch( - r"gemini-\d+(?:\.\d+)?-(flash|flash-lite|pro-preview|pro-image|flash-image)", model + r"gemini-\d+(?:\.\d+)?-(flash|flash-lite|pro|pro-preview|pro-image|flash-image)", model ) if match: - return "gemini-" + match[1] + return "gemini-" + ("pro" if match[1] == "pro-preview" else match[1]) return None +def _freshness_key(model: str) -> tuple: + """Compare Pro GA/preview as peers; prefer GA on an equal release version.""" + if _family(model) == "gemini-pro": + return (_version_sort_key(model.removesuffix("-preview")), not model.endswith("-preview")) + return (_version_sort_key(model), True) + + async def collect_catalogs(providers: dict[str, Any], timeout: float = 60) -> dict[str, dict]: """Fetch actual provider menus; callers must disable static/cache fallback. @@ -80,8 +87,8 @@ def audit_matrix_catalogs(matrices: dict[str, dict], catalogs: dict[str, dict]) family = _family(selected) peers = [m for m in names if family and _family(m) == family] if peers: - latest = max(peers, key=_version_sort_key) - if _version_sort_key(latest) > _version_sort_key(selected): + latest = max(peers, key=_freshness_key) + if _freshness_key(latest) > _freshness_key(selected): issues.append({"kind": "stale_model", **context, "selected": selected, "latest": latest}) if role == "vision" and "vision" not in models[selected].get("capabilities", []): diff --git a/modules/hooks-routing/tests/test_catalog_audit.py b/modules/hooks-routing/tests/test_catalog_audit.py index ff1fcf1..e08d8c5 100644 --- a/modules/hooks-routing/tests/test_catalog_audit.py +++ b/modules/hooks-routing/tests/test_catalog_audit.py @@ -84,4 +84,28 @@ def test_gemini_standard_class_detects_new_generation_but_not_specialized_ids(): "gemini": catalog("gemini-3.8-flash", "gemini-10-flash", "gemini-omni-100-flash", "gemini-100-flash-image") }) - assert result["issues"][0]["latest"] == "gemini-10-flash" \ No newline at end of file + assert result["issues"][0]["latest"] == "gemini-10-flash" + + +def test_gemini_ga_pro_successor_is_visible_to_preview_freshness_check(): + result = audit_matrix_catalogs(matrices("gemini-[3-9]*-pro-preview", "gemini"), { + "gemini": catalog("gemini-3.1-pro-preview", "gemini-3.2-pro", + "gemini-9-pro-image", "gemini-9-pro-preview-customtools", + "gemini-omni-9-pro") + }) + assert result["issues"][0]["kind"] == "stale_model" + assert result["issues"][0]["latest"] == "gemini-3.2-pro" + + +def test_gemini_ga_only_catalog_does_not_false_pass_preview_selector(): + result = audit_matrix_catalogs(matrices("gemini-[3-9]*-pro-preview", "gemini"), { + "gemini": catalog("gemini-3.2-pro") + }) + assert result["issues"][0]["kind"] == "missing_glob" + + +def test_gemini_ga_is_newer_than_same_version_preview(): + result = audit_matrix_catalogs(matrices("gemini-[3-9]*-pro-preview", "gemini"), { + "gemini": catalog("gemini-3.1-pro-preview", "gemini-3.1-pro") + }) + assert result["issues"][0]["latest"] == "gemini-3.1-pro" \ No newline at end of file diff --git a/modules/hooks-routing/tests/test_knob_consistent_routing.py b/modules/hooks-routing/tests/test_knob_consistent_routing.py index 2c6030d..53f5529 100644 --- a/modules/hooks-routing/tests/test_knob_consistent_routing.py +++ b/modules/hooks-routing/tests/test_knob_consistent_routing.py @@ -36,6 +36,7 @@ ROUTING_DIR = REPO_ROOT / "routing" OPENAI_MODELS = [ + "gpt-6-luna", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", @@ -786,9 +787,8 @@ async def test_shipped_openai_matrix_opt_out_restores_legacy_at_mount( `disable_delegation_preset: true` on a terra root restores the matrix's own uninherited answer with no matrix edit required. - That answer is `gpt-5.6-terra`, not `gpt-5.6-sol`: the flagship tier - was paused matrix-wide on 2026-09-07 pending further evals, so no - role selects sol any more. What this test guards is the OPT-OUT + That answer is now `gpt-6.1-sol` after the approved interim migration. + What this test guards is the OPT-OUT mechanism -- that the flag stops the caller's knobs being inherited -- not which model the matrix happens to name.""" agents = {"explorer": {"model_role": "reasoning"}} @@ -803,7 +803,7 @@ async def test_shipped_openai_matrix_opt_out_restores_legacy_at_mount( ) await _run_session_start(coordinator) assert agents["explorer"]["provider_preferences"][0]["model"] == ( - "gpt-5.6-terra" + "gpt-6.1-sol" ) @@ -830,8 +830,8 @@ async def test_openai_matrix_cold_resolution_is_unchanged(self) -> None: resolves), the preset is INERT -- cold resolution returns whatever the roles block says, untouched. - The expected model changed from `gpt-5.6-sol` to `gpt-5.6-terra` on - 2026-09-07 when the flagship tier was paused pending further evals. + The expected model changed to `gpt-6.1-sol` on 2026-10-02 during + the approved opportunistic refresh. That is a matrix change, not a preset change; the invariant under test -- "a preset does nothing without a caller" -- is unaffected.""" import yaml @@ -839,7 +839,7 @@ async def test_openai_matrix_cold_resolution_is_unchanged(self) -> None: data = yaml.safe_load((ROUTING_DIR / "openai.yaml").read_text(encoding="utf-8")) assert parse_preset(data) is not None result = await resolve_model_role(["reasoning"], data["roles"], _providers()) - assert result[0]["model"] == "gpt-5.6-terra" + assert result[0]["model"] == "gpt-6.1-sol" assert result[0]["config"][CANONICAL_EFFORT_KEY] == "xhigh" @pytest.mark.asyncio @@ -869,7 +869,7 @@ async def test_openai_root_delegate_resolution_stays_in_tier_by_default( escalations=escalations, ) assert result, f"role {role} resolved to nothing" - assert result[0]["model"] != "gpt-5.6-sol", ( + assert result[0]["model"] not in {"gpt-5.6-sol", "gpt-6.1-sol"}, ( f"role {role} still resolves to sol under a terra caller " "-- the default-on preset did not clamp it" ) @@ -935,11 +935,8 @@ async def test_a_top_tier_root_is_not_downgraded(self) -> None: `gpt-5.6-sol` / xhigh caller still gets the matrix's own top candidate at full effort, rather than being pulled down a rung. - Previously this read `openai-knob-consistent.yaml` (deleted - 2026-09-07) and asserted the result was sol itself, because sol was in - the roles block. With the flagship tier paused, the matrix's ceiling - for `reasoning` is `gpt-5.6-terra` @ xhigh -- so that is what a sol - caller must still receive, undiminished.""" + The approved October migration selects Sol 6.1 at the top rung; + the caller's effort ceiling still applies.""" import yaml data = yaml.safe_load((ROUTING_DIR / "openai.yaml").read_text(encoding="utf-8")) @@ -955,7 +952,7 @@ async def test_a_top_tier_root_is_not_downgraded(self) -> None: preset=preset, escalations=EscalationState(), ) - assert result[0]["model"] == "gpt-5.6-terra" + assert result[0]["model"] == "gpt-6.1-sol" assert result[0]["config"][CANONICAL_EFFORT_KEY] == "xhigh" # `test_roles_block_is_identical_to_the_stock_openai_matrix` lived here. It diff --git a/routing/balanced.yaml b/routing/balanced.yaml index 4eb3711..059398d 100644 --- a/routing/balanced.yaml +++ b/routing/balanced.yaml @@ -24,12 +24,10 @@ # particular `vision` stays Gemini-led. If this changes, revisit balanced.yaml, # quality.yaml, economy.yaml together. # -# OpenAI tier status (2026-09-07): the flagship `gpt-?.?-sol*` tier is PAUSED -# pending further evals; sol candidates became `gpt-?.?-terra*` one effort -# notch higher, clamped at xhigh, and the `gpt-[0-9].[0-9]` base-alias twin -# is gone (gpt-5.6 is live; a gpt-5.5 fallback bought nothing). `ui-coding` -# deliberately runs `gpt-?.?-luna*` at `max` effort -- cheap tier, maximum -# thinking. +# OpenAI tier status (2026-10-02): approved interim Terra -> exact Sol 6.1 +# migration, including Copilot GPT fallbacks. Existing role efforts remain. +# Luna is current GPT-6; a comparative eval-driven selection review follows. +# Historical September pause and rationale are retained in the review log. # # Gemini notes: # * Globs are DIGIT-ANCHORED (`gemini-[3-9]*-...`). The leading `[3-9]` @@ -132,7 +130,7 @@ roles: description: "Versatile catch-all, no specialization needed" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -150,7 +148,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: ollama @@ -160,7 +158,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -206,7 +204,7 @@ roles: description: "Code generation, implementation, debugging" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high - provider: anthropic @@ -224,7 +222,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: ollama @@ -234,7 +232,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: max - provider: anthropic @@ -252,7 +250,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: medium @@ -260,7 +258,7 @@ roles: description: "Vulnerability assessment, attack surface analysis, code auditing" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -274,7 +272,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -282,7 +280,7 @@ roles: description: "Deep architectural reasoning, system design, complex multi-step analysis" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -308,7 +306,7 @@ roles: description: "Analytical evaluation — finding flaws in existing work, not generating solutions" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -322,7 +320,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -330,7 +328,7 @@ roles: description: "Design direction, aesthetic judgment, high-quality creative output" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high - provider: anthropic @@ -352,7 +350,7 @@ roles: description: "Long-form content — documentation, marketing, case studies, storytelling" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high - provider: anthropic @@ -374,7 +372,7 @@ roles: description: "Deep investigation, information synthesis across multiple sources" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -412,7 +410,7 @@ roles: thinking_config: thinking_level: low - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -420,7 +418,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: github-copilot @@ -438,7 +436,7 @@ roles: description: "High-reliability operational tasks — infrastructure, orchestration, coordination" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic diff --git a/routing/copilot.yaml b/routing/copilot.yaml index 1f866a2..3887c65 100644 --- a/routing/copilot.yaml +++ b/routing/copilot.yaml @@ -24,8 +24,8 @@ description: "GitHub Copilot-only routing. Strongest model per role across the C updated: "2026-10-02" # October catalog refresh: Claude 5.5, Sol 6.1 and advertised Sonnet vision. -# The Sol pause applies to OpenAI-family routing, not this existing Copilot -# security-audit specialist. This refresh does not add Sol to other matrices. +# This existing specialist is retained. The subsequent approved interim patch +# also migrates Terra fallbacks in mixed matrices to Sol 6.1. roles: general: diff --git a/routing/economy.yaml b/routing/economy.yaml index e898797..86c7421 100644 --- a/routing/economy.yaml +++ b/routing/economy.yaml @@ -5,10 +5,13 @@ # that matches an installed provider. # # Philosophy: budget-first. Prefer cheap tier (Haiku, *-luna, Flash, -# Flash-Lite) for utility roles. Reserve mid-tier (Sonnet, -# gpt-?.?-terra*) for quality-sensitive roles like security-audit, critique, +# Flash-Lite) for utility roles. Reserve Sonnet / Sol 6.1 +# for quality-sensitive roles like security-audit, critique, # creative, writing, research. # +# 2026-10-02: approved opportunistic Terra -> Sol 6.1 migration keeps existing +# role efforts but raises the model tier. Proper quality/cost evals follow. +# # Schema: # provider: Module mount name (e.g., "anthropic", "gemini", "github-copilot") # model: Exact model name or class-scoped glob. Copilot PINNED. @@ -129,7 +132,7 @@ roles: description: "Versatile catch-all, no specialization needed" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -173,7 +176,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -221,7 +224,7 @@ roles: description: "Code generation, implementation, debugging" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -263,7 +266,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: max - provider: anthropic @@ -303,7 +306,7 @@ roles: description: "Vulnerability assessment, attack surface analysis, code auditing" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high - provider: anthropic @@ -325,7 +328,7 @@ roles: description: "Deep architectural reasoning, system design, complex multi-step analysis" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -343,7 +346,7 @@ roles: config: reasoning_effort: medium - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: medium @@ -351,7 +354,7 @@ roles: description: "Analytical evaluation — finding flaws in existing work, not generating solutions" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -365,7 +368,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -373,7 +376,7 @@ roles: description: "Design direction, aesthetic judgment, high-quality creative output" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -395,7 +398,7 @@ roles: description: "Long-form content — documentation, marketing, case studies, storytelling" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -417,7 +420,7 @@ roles: description: "Deep investigation, information synthesis across multiple sources" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -458,7 +461,7 @@ roles: config: thinking_budget_tokens: 32000 - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: medium - provider: anthropic @@ -485,7 +488,7 @@ roles: description: "High-reliability operational tasks — infrastructure, orchestration, coordination" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic diff --git a/routing/openai.yaml b/routing/openai.yaml index a64bf5d..e24b8de 100644 --- a/routing/openai.yaml +++ b/routing/openai.yaml @@ -5,7 +5,7 @@ # # BOTH OPENAI BACKENDS, ONE CANDIDATE. `provider: openai` matches EITHER the # pay-per-use API-key backend (module `openai`) or the ChatGPT-subscription/ -# OAuth backend (module `openai-chatgpt`). Same gpt-5.6 family, same model +# OAuth backend (module `openai-chatgpt`). Same GPT-6 family, same model # ids, same `reasoning_effort` key -- the only difference is which account # pays. The resolver's `PROVIDER_FAMILY_ALIASES` (resolver.py) tries the API # key across all mounts first, then the subscription, so a user with both @@ -20,18 +20,16 @@ # bridge in the provider (mirroring provider-openai __init__.py:1046) -- and # the twins were removed. One candidate, one glob vocabulary, one knob. # -# TIER STATUS (2026-09-07): the flagship `gpt-?.?-sol*` tier is PAUSED pending -# further evals. No role selects sol. Roles that previously did now use -# `gpt-?.?-terra*` one effort notch higher, clamped at xhigh. +# TIER STATUS (2026-10-02): interim, opportunistic GPT-6 update. The prior +# Sol pause is superseded by the approved Terra -> exact gpt-6.1-sol migration. +# Existing role effort settings remain; comparative quality/cost evals follow. # # `luna` is used for exactly two roles: `fast` (cheap utility work) and # `ui-coding`, the latter at `max` effort -- cheapest tier, maximum thinking. # -# TWO GLOBS, NO FALLBACKS, NO TRAILING `*`. Every candidate in this file is -# either `gpt-?.?-terra` or `gpt-?.?-luna`. The `gpt-[0-9].[0-9]` base-alias -# twin (a pre-5.6 migration fallback) and the `gpt-?.?-mini*` rung are both -# gone -- gpt-5.6 terra and luna are live on both backends, so neither -# fallback bought anything. +# Sol is an exact gpt-6.1-sol pin; Luna uses suffix-free gpt-[6-9]*-luna. +# No legacy-generation fallback is added. Catalog CI validates exact Sol +# membership separately because runtime pins do not call list_models(). # # THE MISSING `*` IS LOAD-BEARING. The ChatGPT backend serves a `-fast` # sibling for every model (`gpt-5.6-terra` AND `gpt-5.6-terra-fast`). A @@ -42,21 +40,11 @@ # `gpt-5.6-sol-fast` "Internal only" trap on GitHub Copilot.) It also stops # matching dated snapshots, which is wanted: the clean alias is the target. # -# CONSEQUENCE, STATED PLAINLY: each role offers one candidate per backend and -# nothing else, and neither openai provider has a static model-list fallback -# (provider-openai's `list_models()` docstring: "no fallback; caller handles -# empty lists"). If that call fails, the glob resolves to nothing and the role -# yields no preference -- the agent falls back to the session's default -# provider rather than erroring. If that degradation is not acceptable, add a -# pinned exact name (e.g. `gpt-5.6-terra`) as a further candidate: an exact -# name never calls list_models(), the same glob+pin shape copilot.yaml uses. -# -# NOTE: `gpt-?.?-sol*` is deliberately RETAINED on the preset tier ladder -# below even though nothing routes to it. The ladder describes the provider -# family for knob-consistent tier comparison; removing the rung would -# misclassify a sol model that a user pins by hand. With the base alias gone -# from the roles, nothing resolves onto the sol rung any more, so a terra-tier -# caller can no longer be clamped by a fallback sitting a rung above it. +# Luna still requires a usable catalog; Sol exact pins retain offline name +# resolution. Availability failures keep the declared fallback semantics. +# The preset retains legacy Terra classification for explicitly pinned callers: +# strict inheritance must not escalate an existing caller merely because the +# stock matrix moved to Sol. # # Knob-consistent delegation is ON by default for this matrix (measured win on # OpenAI roots: `gpt-5.6-sol` call share 27.8% -> 0.0%, cost -29.7% (S3 median) @@ -150,7 +138,7 @@ # selecting one, so it must still match a hand-pinned suffixed id. name: openai -description: "OpenAI routing for BOTH backends -- the pay-per-use API (`openai`) and the ChatGPT subscription (`openai-chatgpt`). Flagship (sol) tier paused pending evals; knob-consistent delegation ON by default -- sub-agents inherit the caller's tier/effort ceiling." +description: "Interim GPT-6+ routing for both OpenAI backends. Sol 6.1 replaces Terra; Luna tracks current clean IDs. Knob-consistent delegation preserves the caller's tier/effort ceiling." updated: "2026-10-02" # --------------------------------------------------------------------------- @@ -180,7 +168,7 @@ preset: # covers both. tier_ladder: openai: - - ["gpt-[0-9]*-luna", "gpt-[0-9]*-luna*", "gpt-?.?-mini*", "gpt-?.?-nano*"] + - ["gpt-[6-9]*-luna", "gpt-[0-9]*-luna*", "gpt-?.?-mini*", "gpt-?.?-nano*"] - ["gpt-?.?-terra*"] - ["gpt-[0-9]*-sol*", "gpt-[0-9]*-astra*", "gpt-[0-9].[0-9]"] @@ -198,7 +186,7 @@ roles: description: "Versatile catch-all, no specialization needed" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -206,7 +194,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: medium @@ -214,7 +202,7 @@ roles: description: "Code generation, implementation, debugging" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -222,7 +210,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: max @@ -230,7 +218,7 @@ roles: description: "Vulnerability assessment, attack surface analysis, code auditing" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -238,7 +226,7 @@ roles: description: "Deep architectural reasoning, system design, complex multi-step analysis" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -246,7 +234,7 @@ roles: description: "Analytical evaluation — finding flaws in existing work, not generating solutions" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -254,7 +242,7 @@ roles: description: "Design direction, aesthetic judgment, high-quality creative output" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -262,7 +250,7 @@ roles: description: "Long-form content — documentation, marketing, case studies, storytelling" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -270,7 +258,7 @@ roles: description: "Deep investigation, information synthesis across multiple sources" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh @@ -278,7 +266,7 @@ roles: description: "Understanding visual input — screenshots, diagrams, UI mockups" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium @@ -286,7 +274,7 @@ roles: description: "Image generation (limited — gpt-image not available via chat API)" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -294,6 +282,6 @@ roles: description: "High-reliability operational tasks — infrastructure, orchestration, coordination" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh diff --git a/routing/quality.yaml b/routing/quality.yaml index 03b27f0..885c4d8 100644 --- a/routing/quality.yaml +++ b/routing/quality.yaml @@ -22,12 +22,10 @@ # particular `vision` stays Gemini-led. If this changes, revisit balanced.yaml, # quality.yaml, economy.yaml together. # -# OpenAI tier status (2026-09-07): the flagship `gpt-?.?-sol*` tier is PAUSED -# pending further evals; sol candidates became `gpt-?.?-terra*` one effort -# notch higher, clamped at xhigh, and the `gpt-[0-9].[0-9]` base-alias twin -# is gone (gpt-5.6 is live; a gpt-5.5 fallback bought nothing). `ui-coding` -# deliberately runs `gpt-?.?-luna*` at `max` effort -- cheap tier, maximum -# thinking. +# OpenAI tier status (2026-10-02): approved interim Terra -> exact Sol 6.1 +# migration, including Copilot GPT fallbacks. Existing role efforts remain. +# Luna is current GPT-6; a comparative eval-driven selection review follows. +# Historical September pause and rationale are retained in the review log. # # KNOWN ODDITY, retained deliberately: `fast` selects Sonnet, not Haiku. In a # quality-first matrix "fast" means "the quickest acceptable model", and this @@ -130,7 +128,7 @@ roles: description: "Versatile catch-all, no specialization needed" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: low - provider: anthropic @@ -152,7 +150,7 @@ roles: description: "Quick utility tasks — parsing, classification, file ops, bulk work" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -174,7 +172,7 @@ roles: description: "Code generation, implementation, debugging" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high - provider: anthropic @@ -196,7 +194,7 @@ roles: description: "Frontend/UI code — components, layouts, styling, spatial reasoning" candidates: - provider: openai - model: "gpt-[0-9]*-luna" + model: "gpt-[6-9]*-luna" config: reasoning_effort: max - provider: anthropic @@ -218,7 +216,7 @@ roles: description: "Vulnerability assessment, attack surface analysis, code auditing" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -232,7 +230,7 @@ roles: thinking_config: thinking_level: high - provider: github-copilot - model: gpt-5.6-terra + model: gpt-6.1-sol config: reasoning_effort: high @@ -240,7 +238,7 @@ roles: description: "Deep architectural reasoning, system design, complex multi-step analysis" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: gemini @@ -262,7 +260,7 @@ roles: description: "Analytical evaluation — finding flaws in existing work, not generating solutions" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -284,7 +282,7 @@ roles: description: "Design direction, aesthetic judgment, high-quality creative output" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -306,7 +304,7 @@ roles: description: "Long-form content — documentation, marketing, case studies, storytelling" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -328,7 +326,7 @@ roles: description: "Deep investigation, information synthesis across multiple sources" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: xhigh - provider: anthropic @@ -356,7 +354,7 @@ roles: thinking_config: thinking_level: low - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: medium - provider: anthropic @@ -378,7 +376,7 @@ roles: description: "High-reliability operational tasks — infrastructure, orchestration, coordination" candidates: - provider: openai - model: gpt-?.?-terra + model: gpt-6.1-sol config: reasoning_effort: high - provider: anthropic diff --git a/tests/fixtures/resolution_golden_pre_knob_consistency.json b/tests/fixtures/resolution_golden_pre_knob_consistency.json index 52348d8..f11f612 100644 --- a/tests/fixtures/resolution_golden_pre_knob_consistency.json +++ b/tests/fixtures/resolution_golden_pre_knob_consistency.json @@ -133,7 +133,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -142,7 +142,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -151,7 +151,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -160,7 +160,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -169,17 +169,17 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], "fast": [ { "config": { - "reasoning_effort": "medium" + "thinking_budget_tokens": 32000 }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-haiku-4-5", + "provider": "anthropic" } ], "general": [ @@ -187,7 +187,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -203,7 +203,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -212,7 +212,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -221,17 +221,17 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], "ui-coding": [ { "config": { - "reasoning_effort": "max" + "reasoning_effort": "medium" }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-sonnet-5", + "provider": "anthropic" } ], "vision": [ @@ -252,7 +252,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ] @@ -387,17 +387,17 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], "coding": [ { "config": { - "reasoning_effort": "medium" + "thinking_budget_tokens": 32000 }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-haiku-4-5", + "provider": "anthropic" } ], "creative": [ @@ -405,7 +405,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -414,7 +414,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -423,26 +423,26 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], "fast": [ { "config": { - "reasoning_effort": "medium" + "thinking_budget_tokens": 32000 }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-haiku-4-5", + "provider": "anthropic" } ], "general": [ { "config": { - "reasoning_effort": "medium" + "thinking_budget_tokens": 32000 }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-haiku-4-5", + "provider": "anthropic" } ], "image-gen": [ @@ -457,7 +457,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -466,7 +466,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -475,17 +475,17 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], "ui-coding": [ { "config": { - "reasoning_effort": "max" + "thinking_budget_tokens": 32000 }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-haiku-4-5", + "provider": "anthropic" } ], "vision": [ @@ -502,7 +502,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ] @@ -690,7 +690,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -699,7 +699,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -708,7 +708,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -717,7 +717,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -726,25 +726,17 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", - "provider": "openai" - } - ], - "fast": [ - { - "config": { - "reasoning_effort": "medium" - }, - "model": "gpt-5.6-luna", + "model": "gpt-6.1-sol", "provider": "openai" } ], + "fast": [], "general": [ { "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -753,7 +745,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -762,7 +754,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -771,7 +763,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -780,25 +772,17 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", - "provider": "openai" - } - ], - "ui-coding": [ - { - "config": { - "reasoning_effort": "max" - }, - "model": "gpt-5.6-luna", + "model": "gpt-6.1-sol", "provider": "openai" } ], + "ui-coding": [], "vision": [ { "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -807,7 +791,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ] @@ -818,7 +802,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -827,7 +811,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -836,7 +820,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -845,7 +829,7 @@ "config": { "reasoning_effort": "high" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -854,7 +838,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -863,7 +847,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -872,7 +856,7 @@ "config": { "reasoning_effort": "low" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -888,7 +872,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -897,7 +881,7 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], @@ -906,17 +890,17 @@ "config": { "reasoning_effort": "xhigh" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ], "ui-coding": [ { "config": { - "reasoning_effort": "max" + "reasoning_effort": "medium" }, - "model": "gpt-5.6-luna", - "provider": "openai" + "model": "claude-sonnet-5", + "provider": "anthropic" } ], "vision": [ @@ -937,7 +921,7 @@ "config": { "reasoning_effort": "medium" }, - "model": "gpt-5.6-terra", + "model": "gpt-6.1-sol", "provider": "openai" } ] diff --git a/tests/test_catalog_refresh.py b/tests/test_catalog_refresh.py index 5d9b7ba..a3c6cf4 100644 --- a/tests/test_catalog_refresh.py +++ b/tests/test_catalog_refresh.py @@ -104,11 +104,27 @@ def test_copilot_pins_refresh_all_matrices(): assert vision["model"] == "claude-sonnet-5.5" -def test_sol_pause_stays_in_place(): - for name in ["balanced", "quality", "economy", "openai"]: - for role in matrix(name)["roles"].values(): - assert all("-sol" not in c["model"] and "-astra" not in c["model"] - for c in role["candidates"] if c["provider"] == "openai") +def test_interim_openai_candidates_are_v6_and_terra_is_no_longer_selected(): + checked = 0 + for path in (ROOT / "routing").glob("*.yaml"): + for role in yaml.safe_load(path.read_text())["roles"].values(): + for candidate in role["candidates"]: + assert "terra" not in candidate["model"] + if candidate["provider"] == "openai": + assert candidate["model"] in {"gpt-6.1-sol", "gpt-[6-9]*-luna"} + checked += 1 + if candidate["provider"] == "github-copilot" and candidate["model"].startswith("gpt-"): + assert candidate["model"] in {"gpt-6.1-sol", "gpt-6-luna"} + assert checked == 49 + + +@pytest.mark.asyncio +async def test_current_luna_selector_does_not_demote_to_pre6_catalog(): + data = matrix("openai") + result = await resolve_model_role( + ["fast"], data["roles"], {"openai": provider(["gpt-5.6-luna", "gpt-5.6-luna-fast"])} + ) + assert result == [] @pytest.mark.asyncio diff --git a/tests/test_single_provider_coverage.py b/tests/test_single_provider_coverage.py index dea8420..0dc14ae 100644 --- a/tests/test_single_provider_coverage.py +++ b/tests/test_single_provider_coverage.py @@ -72,6 +72,8 @@ "claude-fable-5-1", ], "openai": [ + "gpt-6.1-sol", + "gpt-6-luna", "gpt-6-astra", "gpt-5.6-sol", "gpt-5.6-terra", @@ -82,6 +84,10 @@ "gpt-5.4-nano", ], "openai-chatgpt": [ + "gpt-6.1-sol", + "gpt-6.1-sol-fast", + "gpt-6-luna", + "gpt-6-luna-fast", "gpt-6-astra", "gpt-6-astra-fast", "gpt-5.6-sol",