From c00c54f4afa770bc959ce01005db839fe9f2c69f Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Sun, 12 Jul 2026 23:27:19 +0000 Subject: [PATCH 01/11] Support Anthropic BYOK endpoints --- ARCHITECTURE.md | 11 +- README.md | 34 ++++-- config.toml | 5 +- src/config.rs | 62 +++++++++-- src/doctor.rs | 52 +++------- src/llm.rs | 267 +++++++++++++++++++++++++++++++++++++++++------- src/main.rs | 1 + tests/e2e.rs | 237 ++++++++++++++++++++++++++++++++++++++++-- 8 files changed, 570 insertions(+), 99 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 101d6ee..2bb9c53 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -22,8 +22,12 @@ acquire diff ──> parse + index ──> prompt ──> model (cascade/consens - `filter.rs` — grounding (uncited findings dropped; all-uncited = untrusted run), policy suppression (ignore globs, severityThreshold, minConfidence, maxFindings), and baseline reconciliation (resolved / carried) for incremental reviews. -- `llm.rs` — OpenAI-compatible client; model cascade on failure, one JSON-repair retry, - optional N-model consensus (agreement by path + line proximity). +- `llm.rs` — shared model transport for OpenAI-compatible chat completions and the + native Anthropic Messages API; model cascade on failure, one JSON-repair retry, + optional N-model consensus (agreement by path + line proximity). Request construction + and response decoding vary by API format while retry, timeout, deadline, cascade, and + secret-redaction semantics remain shared. Optional private-endpoint authentication is + a separate header whose name cannot collide with provider-managed headers. - `review.rs` — orchestration; owns fail-closed semantics (`fail_closed_finding`) and check-run lifecycle ordering (checks are created before the model runs so a crash can still be reported against them). @@ -41,7 +45,8 @@ acquire diff ──> parse + index ──> prompt ──> model (cascade/consens Exception to precedence: `model.apiBase` from a config file is ignored by default (a repo could redirect the base URL that receives the inference credential); honored only with `POSTIL_ALLOW_CONFIG_API_BASE=1`. The - `POSTIL_API_BASE` environment variable is always applied. + `POSTIL_API_BASE` environment variable is always applied. `model.apiFormat` and + `POSTIL_API_FORMAT` select `openai-compatible` (default) or `anthropic`. ## Prompt-injected policy sources diff --git a/README.md b/README.md index 8b59613..fe6b1c8 100644 --- a/README.md +++ b/README.md @@ -142,13 +142,13 @@ gate: # contentPolicy: # enabled: false model: - name: deepseek/deepseek-v4-pro + name: z-ai/glm-5.2 cascade: - - google/gemini-3.1-flash-lite - moonshotai/kimi-k2.7-code - - mistralai/mistral-large-2512 + - deepseek/deepseek-v4-flash scorer: anthropic/claude-haiku-4.5 apiBase: https://openrouter.ai/api/v1 # ignored from config by default; see note below + apiFormat: openai-compatible # or anthropic for the native Messages API consensus: 1 # >1: only findings multiple models agree on survive ``` @@ -160,7 +160,10 @@ single-user local setup where the checked-out repo is trusted, set `POSTIL_ALLOW_CONFIG_API_BASE=1` to honor the config value. Environment: `POSTIL_API_KEY`, `OPENROUTER_API_KEY`, `MODEL_API_KEY`, or -`LLM_API_KEY`, `POSTIL_API_BASE`, `POSTIL_DETAILS_URL` (optional HTTP(S) target +`LLM_API_KEY`, `POSTIL_API_BASE`, `POSTIL_API_FORMAT` (`openai-compatible` by +default, or `anthropic`), `POSTIL_ENDPOINT_AUTH_HEADER` and +`POSTIL_ENDPOINT_AUTH_VALUE` (optional additional authentication for a private +endpoint), `POSTIL_DETAILS_URL` (optional HTTP(S) target for GitHub check-run details links), `REVIEW_MODEL`, `REVIEW_MODEL_CASCADE`, `REVIEW_SCORER_MODEL`, `GITHUB_TOKEN`/`GITHUB_API_URL`, @@ -171,7 +174,24 @@ for GitHub check-run details links), See the measured benchmark results at [postil.dev/docs/models](https://postil.dev/docs/models), which are sourced from the published bench aggregate. Any model served through an -OpenAI-compatible endpoint works. +OpenAI-compatible endpoint works. Native Anthropic Messages API endpoints also work: + +```sh +POSTIL_API_BASE=https://api.anthropic.com/v1 \ +POSTIL_API_FORMAT=anthropic \ +MODEL_API_KEY=... \ +REVIEW_MODEL=claude-sonnet-4-6 \ +postil doctor +``` + +OpenAI-compatible requests use `Authorization: Bearer`; Anthropic requests use +`x-api-key` and `anthropic-version`. A private gateway can require an additional +header through `POSTIL_ENDPOINT_AUTH_HEADER` and `POSTIL_ENDPOINT_AUTH_VALUE`. +Postil rejects additional-header names that collide with `x-api-key`, +`anthropic-version`, or `content-type`. OpenAI-compatible endpoints also reserve +`Authorization` for the provider key; Anthropic endpoints may use an additional +`Authorization` value alongside their provider-owned `x-api-key`. Postil never +prints credential values. Local endpoints use the same OpenAI-compatible contract: @@ -190,14 +210,14 @@ REVIEW_MODEL= \ postil review --staged --output json ``` -Hosted remote reviews use a 240-second request timeout with a single timeout retry capped at 90 seconds, reducing unnecessary fallback to weaker models when the primary model is slow but working. The entire review model phase is capped at 420 seconds, with the remaining 120 seconds of the 540-second total LLM budget reserved for scoring inside the worker watchdog. A timeout triggers one automatic retry at the same model level before cascading to the next model. Local reviews default to a 480-second request timeout and do not use a total deadline unless `POSTIL_LLM_TOTAL_TIMEOUT_SECS` is set. Exhausting a review or total deadline is terminal. +Hosted remote reviews use a 240-second initial request timeout with a single timeout retry capped at 90 seconds, reducing unnecessary fallback to weaker models when the primary model is slow but working. The entire review model phase is capped at 420 seconds, with the remaining 120 seconds of the 540-second total LLM budget reserved for scoring inside the worker watchdog. A timeout triggers one automatic retry at the same model level before cascading to the next model. Local reviews use a 480-second initial request timeout and the same timeout-retry rule, so a timed-out model can receive one additional attempt of up to 90 seconds. Local reviews do not have a total deadline unless `POSTIL_LLM_TOTAL_TIMEOUT_SECS` is set. Exhausting a review or total deadline is terminal. Use the live benchmark harness before standardizing on a model: ```sh cargo build --quiet --release cd bench -MODEL_API_KEY=... REVIEW_MODEL=deepseek/deepseek-v4-pro bun run bench:live -- --json +MODEL_API_KEY=... REVIEW_MODEL=z-ai/glm-5.2 bun run bench:live -- --json ``` ## Preview a config change before deploying it diff --git a/config.toml b/config.toml index 2fbb0f6..04e2537 100644 --- a/config.toml +++ b/config.toml @@ -1,9 +1,8 @@ version = 1 -default_model = "deepseek/deepseek-v4-pro" +default_model = "z-ai/glm-5.2" cascade = [ - "google/gemini-3.1-flash-lite", "moonshotai/kimi-k2.7-code", - "mistralai/mistral-large-2512", + "deepseek/deepseek-v4-flash", ] [scorer] diff --git a/src/config.rs b/src/config.rs index 9afadb3..b962ebf 100644 --- a/src/config.rs +++ b/src/config.rs @@ -20,6 +20,31 @@ use crate::envelope::{Kind, Severity}; const MODEL_DEFAULTS_TOML: &str = include_str!("../config.toml"); pub const DEFAULT_API_BASE: &str = "https://openrouter.ai/api/v1"; +#[derive(Debug, Clone, Copy, Default, Deserialize, Serialize, PartialEq, Eq)] +#[serde(rename_all = "kebab-case")] +pub enum ApiFormat { + #[default] + OpenaiCompatible, + Anthropic, +} + +impl ApiFormat { + pub fn parse(value: &str) -> Result { + match value.trim().to_ascii_lowercase().as_str() { + "openai-compatible" => Ok(Self::OpenaiCompatible), + "anthropic" => Ok(Self::Anthropic), + _ => anyhow::bail!("invalid API format {value:?} (openai-compatible|anthropic)"), + } + } + + pub fn as_str(self) -> &'static str { + match self { + Self::OpenaiCompatible => "openai-compatible", + Self::Anthropic => "anthropic", + } + } +} + #[derive(Debug, Clone, PartialEq, Eq)] pub struct ModelDefaults { pub version: u64, @@ -193,6 +218,7 @@ pub struct Config { pub cascade: Vec, pub scorer: String, pub api_base: String, + pub api_format: ApiFormat, /// Run the first N models of [model + cascade] and keep agreeing findings. pub consensus: usize, /// Contents of `.postil/guardrails.md`, injected into the prompt as repo @@ -229,6 +255,7 @@ impl Default for Config { cascade: defaults.cascade.clone(), scorer: defaults.scorer_model.clone(), api_base: DEFAULT_API_BASE.to_string(), + api_format: ApiFormat::default(), consensus: 1, guardrails: None, content_policy: Some(BUILTIN_CONTENT_POLICY.to_string()), @@ -283,6 +310,7 @@ pub struct ModelSection { pub cascade: Option>, pub scorer: Option, pub api_base: Option, + pub api_format: Option, pub consensus: Option, } @@ -322,7 +350,7 @@ impl Config { } else { Config::default() }; - cfg.apply_env(); + cfg.apply_env()?; // Repo guardrails are a separate file so they can be long-form prose. let guardrails_path = root.join(".postil").join("guardrails.md"); if let Ok(text) = std::fs::read_to_string(&guardrails_path) { @@ -449,6 +477,9 @@ impl Config { ); } } + if let Some(format) = m.api_format { + self.api_format = format; + } if let Some(n) = m.consensus { anyhow::ensure!(n >= 1, "model.consensus must be >= 1"); self.consensus = n; @@ -509,7 +540,7 @@ impl Config { Ok(cfg) } - fn apply_env(&mut self) { + fn apply_env(&mut self) -> Result<()> { if let Ok(m) = std::env::var("REVIEW_MODEL") && !m.is_empty() { @@ -530,6 +561,12 @@ impl Config { { self.api_base = b; } + if let Ok(format) = std::env::var("POSTIL_API_FORMAT") + && !format.trim().is_empty() + { + self.api_format = ApiFormat::parse(&format)?; + } + Ok(()) } /// All models to try, in order, deduplicated. @@ -606,7 +643,8 @@ model: name: __DEFAULT_MODEL__ cascade: __DEFAULT_CASCADE__ scorer: __DEFAULT_SCORER_MODEL__ - # apiBase: https://openrouter.ai/api/v1 # any OpenAI-compatible endpoint (Ollama, vLLM, Azure). + # apiBase: https://openrouter.ai/api/v1 # OpenAI-compatible or Anthropic endpoint base URL. + # apiFormat: openai-compatible # openai-compatible (default) or anthropic. # # Ignored from config by default (a repo could redirect # # the inference credential). Prefer POSTIL_API_BASE; to # # honor this key set POSTIL_ALLOW_CONFIG_API_BASE=1 for a @@ -868,10 +906,9 @@ fallback = "example/scorer-fallback" assert_eq!( c.model_chain(), vec![ - "deepseek/deepseek-v4-pro", - "google/gemini-3.1-flash-lite", + "z-ai/glm-5.2", "moonshotai/kimi-k2.7-code", - "mistralai/mistral-large-2512", + "deepseek/deepseek-v4-flash", ] ); } @@ -935,6 +972,19 @@ fallback = "example/scorer-fallback" ); } + #[test] + fn anthropic_api_format_parses_from_postil_config() { + let f: FileConfig = serde_yaml::from_str("model:\n apiFormat: anthropic\n").unwrap(); + let mut config = Config::default(); + config.apply_file(f).unwrap(); + assert_eq!(config.api_format, ApiFormat::Anthropic); + } + + #[test] + fn openai_compatible_is_the_default_api_format() { + assert_eq!(Config::default().api_format, ApiFormat::OpenaiCompatible); + } + #[test] fn guardrails_file_is_loaded() { let dir = tempfile::tempdir().unwrap(); diff --git a/src/doctor.rs b/src/doctor.rs index 4bffdde..875b9fe 100644 --- a/src/doctor.rs +++ b/src/doctor.rs @@ -5,11 +5,10 @@ //! silently does nothing. Doctor checks each link in the chain and says //! exactly what to fix. Secret values are never printed. -use anyhow::Result; -use serde_json::json; - use crate::api_key; use crate::config::Config; +use crate::llm::LlmClient; +use anyhow::Result; pub struct Check { pub name: &'static str, @@ -59,43 +58,26 @@ pub async fn run(cfg: &Config) -> Result> { }, }); - // Live probe: a 1-token completion proves base URL + key + model in one shot. + // Live probe: a 1-token response proves base URL + key + model + selected + // API format in one shot. LlmClient also applies optional endpoint auth and + // redacts both secrets from any provider error body. if let Some(key) = key { - let http = reqwest::Client::builder() - .timeout(std::time::Duration::from_secs(30)) - .build()?; - let url = format!("{}/chat/completions", cfg.api_base.trim_end_matches('/')); - let resp = http - .post(&url) - .bearer_auth(&key) - .json(&json!({ - "model": cfg.model, - "max_tokens": 1, - "messages": [{"role": "user", "content": "ping"}], - })) - .send() - .await; - let (ok, detail) = match resp { - Ok(r) if r.status().is_success() => ( + let (ok, detail) = match LlmClient::doctor_probe(cfg, key).await { + Ok(()) => ( true, - format!("{} answered for model {}", cfg.api_base, cfg.model), + format!( + "{} answered for model {} using {}", + cfg.api_base, + cfg.model, + cfg.api_format.as_str() + ), ), - Ok(r) => { - let status = r.status(); - let body = r.text().await.unwrap_or_default(); - let snippet: String = body.chars().take(200).collect(); - let hint = match status.as_u16() { - 401 | 403 => " (key rejected: wrong key for this endpoint?)", - 404 => " (404: wrong apiBase path or unknown model name?)", - _ => "", - }; - (false, format!("{status}{hint}: {snippet}")) - } - Err(e) => ( + Err(error) => ( false, format!( - "cannot reach {} ({e}); check model.apiBase — for Ollama use http://localhost:11434/v1", - cfg.api_base + "cannot use {} as {} ({error:#}); check model.apiBase, model.apiFormat, credentials, and model name", + cfg.api_base, + cfg.api_format.as_str() ), ), }; diff --git a/src/llm.rs b/src/llm.rs index 6b22071..4ebabc2 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -1,17 +1,18 @@ -//! OpenAI-compatible chat client with model cascade, one JSON-repair retry, +//! OpenAI-compatible and native Anthropic chat client with model cascade, one JSON-repair retry, //! optional multi-model consensus, and fail-closed semantics. //! -//! Works against OpenRouter (default), Ollama, vLLM, LiteLLM, Azure OpenAI, or -//! any other endpoint that speaks `POST {base}/chat/completions`. +//! OpenAI-compatible endpoints use `POST {base}/chat/completions` by default. +//! Native Anthropic endpoints use `POST {base}/messages` when explicitly selected. use std::time::{Duration, Instant}; use anyhow::{Context, Result, anyhow}; +use reqwest::header::{HeaderName, HeaderValue}; use serde::Deserialize; use serde_json::json; use crate::api_key; -use crate::config::Config; +use crate::config::{ApiFormat, Config}; use crate::envelope::{Finding, Kind, Usage}; #[derive(Debug, Clone)] @@ -157,12 +158,21 @@ pub struct LlmClient { http: reqwest::Client, api_base: String, api_key: String, + api_format: ApiFormat, + endpoint_auth: Option, request_timeout: Duration, timeout_retry_timeout: Duration, review_deadline: Option, total_deadline: Option, } +#[derive(Clone)] +struct EndpointAuth { + name: HeaderName, + value: HeaderValue, + secret: String, +} + #[derive(Debug, Clone, Copy)] struct LlmTimeouts { request: Duration, @@ -184,9 +194,14 @@ const TIMEOUT_RETRY_CAP_SECS: u64 = 90; /// use their provider default. const REVIEW_MAX_TOKENS: u32 = 16384; const SCORER_MAX_TOKENS: u32 = 4096; +const ANTHROPIC_DEFAULT_MAX_TOKENS: u32 = 4096; +const ANTHROPIC_VERSION: &str = "2023-06-01"; pub(crate) const DEFAULT_REQUEST_TIMEOUT_SECS: u64 = 480; const REQUEST_TIMEOUT_ENV: &str = "POSTIL_LLM_REQUEST_TIMEOUT_SECS"; const TOTAL_TIMEOUT_ENV: &str = "POSTIL_LLM_TOTAL_TIMEOUT_SECS"; +const ENDPOINT_AUTH_HEADER_ENV: &str = "POSTIL_ENDPOINT_AUTH_HEADER"; +const ENDPOINT_AUTH_VALUE_ENV: &str = "POSTIL_ENDPOINT_AUTH_VALUE"; +const ALWAYS_MANAGED_HEADERS: &[&str] = &["x-api-key", "anthropic-version", "content-type"]; /// Marker context attached to transport/provider-level failures (endpoint /// unreachable, HTTP error status, timeout, malformed HTTP envelope) — the @@ -242,6 +257,23 @@ impl std::fmt::Display for RequestTimedOut { impl std::error::Error for RequestTimedOut {} impl LlmClient { + pub(crate) async fn doctor_probe(cfg: &Config, api_key: String) -> Result<()> { + let client = Self::build(cfg, api_key, Duration::from_secs(30), None, None)?; + let body = client.request_body(&cfg.model, "", "ping", Some(1), 0.0); + let (status, text) = + tokio::time::timeout(Duration::from_secs(30), client.request_once(&body)) + .await + .map_err(|_| RequestTimedOut)??; + if !status.is_success() { + return Err(anyhow!( + "model endpoint returned {status}: {}", + client.safe_error_snippet(&text) + )); + } + client.parse_response(&text, &mut Usage::default())?; + Ok(()) + } + /// Local and interactive clients have no built-in total deadline. They only /// get one when POSTIL_LLM_TOTAL_TIMEOUT_SECS is explicitly set. pub fn from_env(cfg: &Config) -> Result { @@ -290,6 +322,7 @@ impl LlmClient { review_deadline: Option, total_deadline: Option, ) -> Result { + let endpoint_auth = endpoint_auth_from_env(cfg.api_format)?; Ok(LlmClient { // The attempt timeout wraps both sending the request and consuming // the complete response body, so header and body stalls take the @@ -297,6 +330,8 @@ impl LlmClient { http: reqwest::Client::builder().build()?, api_base: cfg.api_base.trim_end_matches('/').to_string(), api_key, + api_format: cfg.api_format, + endpoint_auth, request_timeout, timeout_retry_timeout: request_timeout.min(Duration::from_secs(TIMEOUT_RETRY_CAP_SECS)), review_deadline, @@ -689,17 +724,7 @@ impl LlmClient { temperature: f64, phase: LlmPhase, ) -> Result { - let mut body = json!({ - "model": model, - "temperature": temperature, - "messages": [ - {"role": "system", "content": system}, - {"role": "user", "content": user}, - ], - }); - if let Some(max_tokens) = max_tokens { - body["max_tokens"] = json!(max_tokens); - } + let body = self.request_body(model, system, user, max_tokens, temperature); let mut retries = 0u32; let mut timeout_retries = 0u32; let mut attempt_timeout = self.request_timeout; @@ -735,7 +760,7 @@ impl LlmClient { if status.is_success() { break text; } - let snippet: String = text.chars().take(300).collect(); + let snippet = self.safe_error_snippet(&text); if timeout_status(status.as_u16()) && timeout_retries < TIMEOUT_RETRIES && retries < TRANSIENT_RETRIES @@ -812,33 +837,114 @@ impl LlmClient { } } }; - let parsed: ChatResponse = - serde_json::from_str(&text).context("model endpoint returned non-JSON body")?; - if let Some(u) = parsed.usage { - usage.prompt_tokens += u.prompt_tokens.unwrap_or(0); - usage.completion_tokens += u.completion_tokens.unwrap_or(0); + self.parse_response(&text, usage) + } + + fn request_body( + &self, + model: &str, + system: &str, + user: &str, + max_tokens: Option, + temperature: f64, + ) -> serde_json::Value { + match self.api_format { + ApiFormat::OpenaiCompatible => { + let mut body = json!({ + "model": model, + "temperature": temperature, + "messages": [ + {"role": "system", "content": system}, + {"role": "user", "content": user}, + ], + }); + if let Some(max_tokens) = max_tokens { + body["max_tokens"] = json!(max_tokens); + } + body + } + ApiFormat::Anthropic => json!({ + "model": model, + "system": system, + "messages": [{"role": "user", "content": user}], + "max_tokens": max_tokens.unwrap_or(ANTHROPIC_DEFAULT_MAX_TOKENS), + "temperature": temperature, + }), + } + } + + fn parse_response(&self, text: &str, usage: &mut Usage) -> Result { + match self.api_format { + ApiFormat::OpenaiCompatible => { + let parsed: ChatResponse = serde_json::from_str(text) + .context("model endpoint returned non-JSON OpenAI-compatible body")?; + if let Some(u) = parsed.usage { + usage.prompt_tokens += u.prompt_tokens.unwrap_or(0); + usage.completion_tokens += u.completion_tokens.unwrap_or(0); + } + parsed + .choices + .into_iter() + .next() + .and_then(|choice| choice.message.content) + .ok_or_else(|| anyhow!("model response had no choices/content")) + } + ApiFormat::Anthropic => { + let parsed: AnthropicResponse = serde_json::from_str(text) + .context("model endpoint returned non-JSON Anthropic body")?; + if let Some(u) = parsed.usage { + usage.prompt_tokens += u.input_tokens.unwrap_or(0); + usage.completion_tokens += u.output_tokens.unwrap_or(0); + } + let content = parsed + .content + .into_iter() + .filter(|block| block.kind == "text") + .filter_map(|block| block.text) + .collect::>() + .join("\n"); + if content.is_empty() { + Err(anyhow!("model response had no text content blocks")) + } else { + Ok(content) + } + } + } + } + + fn safe_error_snippet(&self, text: &str) -> String { + let mut redacted = text.replace(&self.api_key, "[REDACTED]"); + if let Some(auth) = &self.endpoint_auth { + redacted = redacted.replace(&auth.secret, "[REDACTED]"); } - parsed - .choices - .into_iter() - .next() - .and_then(|c| c.message.content) - .ok_or_else(|| anyhow!("model response had no choices/content")) + redacted.chars().take(300).collect() } async fn request_once( &self, body: &serde_json::Value, ) -> std::result::Result<(reqwest::StatusCode, String), reqwest::Error> { - let response = self - .http - .post(format!("{}/chat/completions", self.api_base)) - .bearer_auth(&self.api_key) - .header("HTTP-Referer", "https://postil.dev") - .header("X-Title", "Postil") - .json(body) - .send() - .await?; + let mut request = match self.api_format { + ApiFormat::OpenaiCompatible => { + let url = format!("{}/chat/completions", self.api_base); + self.http + .post(&url) + .bearer_auth(&self.api_key) + .header("HTTP-Referer", "https://postil.dev") + .header("X-Title", "Postil") + } + ApiFormat::Anthropic => { + let url = format!("{}/messages", self.api_base); + self.http + .post(&url) + .header("x-api-key", &self.api_key) + .header("anthropic-version", ANTHROPIC_VERSION) + } + }; + if let Some(auth) = &self.endpoint_auth { + request = request.header(auth.name.clone(), auth.value.clone()); + } + let response = request.json(body).send().await?; let status = response.status(); let text = response.text().await?; Ok((status, text)) @@ -929,6 +1035,65 @@ struct ChatUsage { completion_tokens: Option, } +#[derive(Debug, Deserialize)] +struct AnthropicResponse { + #[serde(default)] + content: Vec, + usage: Option, +} + +#[derive(Debug, Deserialize)] +struct AnthropicContentBlock { + #[serde(rename = "type")] + kind: String, + text: Option, +} + +#[derive(Debug, Deserialize)] +struct AnthropicUsage { + input_tokens: Option, + output_tokens: Option, +} + +fn endpoint_auth_from_env(api_format: ApiFormat) -> Result> { + let header = std::env::var(ENDPOINT_AUTH_HEADER_ENV) + .ok() + .filter(|value| !value.trim().is_empty()); + let value = std::env::var(ENDPOINT_AUTH_VALUE_ENV) + .ok() + .filter(|value| !value.is_empty()); + match (header, value) { + (None, None) => Ok(None), + (Some(_), None) => Err(anyhow!( + "{ENDPOINT_AUTH_VALUE_ENV} must be set when {ENDPOINT_AUTH_HEADER_ENV} is set" + )), + (None, Some(_)) => Err(anyhow!( + "{ENDPOINT_AUTH_HEADER_ENV} must be set when {ENDPOINT_AUTH_VALUE_ENV} is set" + )), + (Some(header), Some(secret)) => { + let normalized = header.trim().to_ascii_lowercase(); + anyhow::ensure!( + !(ALWAYS_MANAGED_HEADERS.contains(&normalized.as_str()) + || (api_format == ApiFormat::OpenaiCompatible + && normalized == "authorization")), + "{ENDPOINT_AUTH_HEADER_ENV} cannot override provider-managed header {header:?}" + ); + let name = HeaderName::from_bytes(header.trim().as_bytes()).with_context(|| { + format!("{ENDPOINT_AUTH_HEADER_ENV} is not a valid HTTP header name") + })?; + let mut value = HeaderValue::from_bytes(secret.as_bytes()).with_context(|| { + format!("{ENDPOINT_AUTH_VALUE_ENV} is not a valid HTTP header value") + })?; + value.set_sensitive(true); + Ok(Some(EndpointAuth { + name, + value, + secret, + })) + } + } +} + /// Extract and validate the review JSON from model text. Tolerates code fences /// and leading/trailing prose, nothing else. fn parse_review(content: &str) -> Result { @@ -1186,6 +1351,36 @@ mod tests { ); } + #[test] + fn endpoint_auth_rejects_headers_managed_by_each_api_format() { + let _lock = env_lock().lock().unwrap(); + let _env = EnvRestore::capture(&[ENDPOINT_AUTH_HEADER_ENV, ENDPOINT_AUTH_VALUE_ENV]); + EnvRestore::set(ENDPOINT_AUTH_VALUE_ENV, "secret-value"); + for name in ["X-API-Key", "Anthropic-Version", "Content-Type"] { + EnvRestore::set(ENDPOINT_AUTH_HEADER_ENV, name); + for format in [ApiFormat::OpenaiCompatible, ApiFormat::Anthropic] { + let error = endpoint_auth_from_env(format) + .err() + .expect("collision rejected"); + assert!(error.to_string().contains("provider-managed header")); + assert!(!error.to_string().contains("secret-value")); + } + } + + EnvRestore::set(ENDPOINT_AUTH_HEADER_ENV, "Authorization"); + let openai_error = endpoint_auth_from_env(ApiFormat::OpenaiCompatible) + .err() + .expect("OpenAI-compatible Authorization collision rejected"); + assert!(openai_error.to_string().contains("provider-managed header")); + assert!(!openai_error.to_string().contains("secret-value")); + + let anthropic = endpoint_auth_from_env(ApiFormat::Anthropic) + .unwrap() + .expect("Anthropic additional Authorization accepted"); + assert_eq!(anthropic.name, reqwest::header::AUTHORIZATION); + assert_eq!(anthropic.value, "secret-value"); + } + impl Drop for EnvRestore { fn drop(&mut self) { for (name, value) in &self.saved { diff --git a/src/main.rs b/src/main.rs index 58b06f0..9b78083 100644 --- a/src/main.rs +++ b/src/main.rs @@ -155,6 +155,7 @@ async fn dispatch(cli: Cli) -> anyhow::Result { println!("model.cascade: {:?}", cfg.cascade); println!("model.scorer: {}", cfg.scorer); println!("model.apiBase: {}", cfg.api_base); + println!("model.apiFormat: {}", cfg.api_format.as_str()); println!("model.consensus: {}", cfg.consensus); Ok(0) } diff --git a/tests/e2e.rs b/tests/e2e.rs index a3d5d89..f25ba75 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -59,6 +59,24 @@ fn llm_text(text: &str) -> Value { }) } +fn anthropic_content(findings: Value, input_tokens: u64, output_tokens: u64) -> Value { + let summary = if findings.as_array().is_none_or(|items| items.is_empty()) { + "" + } else { + "SQL injection risk in auth path." + }; + json!({ + "content": [ + {"type": "thinking", "thinking": "omitted"}, + {"type": "text", "text": json!({ + "summary": summary, + "findings": findings + }).to_string()} + ], + "usage": {"input_tokens": input_tokens, "output_tokens": output_tokens} + }) +} + fn finding_at(line: u32, severity: &str, confidence: f64) -> Value { json!({ "path": "src/auth.rs", @@ -108,6 +126,9 @@ fn postil() -> Command { .env_remove("OPENROUTER_API_KEY") .env_remove("POSTIL_API_KEY") .env_remove("POSTIL_API_BASE") + .env_remove("POSTIL_API_FORMAT") + .env_remove("POSTIL_ENDPOINT_AUTH_HEADER") + .env_remove("POSTIL_ENDPOINT_AUTH_VALUE") .env_remove("POSTIL_DETAILS_URL") .env_remove("GITHUB_SERVER_URL") .env_remove("POSTIL_ENABLE_BITBUCKET_INCREMENTAL") @@ -115,6 +136,206 @@ fn postil() -> Command { cmd } +#[tokio::test] +async fn native_anthropic_review_uses_messages_shape_auth_and_usage() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/messages")) + .and(header("x-api-key", "anthropic-provider-key")) + .and(header("anthropic-version", "2023-06-01")) + .and(header("authorization", "Bearer private-endpoint-secret")) + .respond_with(ResponseTemplate::new(200).set_body_json(anthropic_content(json!([]), 11, 7))) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("MODEL_API_KEY", "anthropic-provider-key") + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_API_FORMAT", "anthropic") + .env("POSTIL_ENDPOINT_AUTH_HEADER", "Authorization") + .env( + "POSTIL_ENDPOINT_AUTH_VALUE", + "Bearer private-endpoint-secret", + ) + .args(["review", "--diff-file"]) + .arg(&diff) + .arg("--output-json") + .assert() + .success(); + + let envelope: Value = serde_json::from_slice(&out.get_output().stdout).unwrap(); + assert_eq!(envelope["usage"]["promptTokens"], 11); + assert_eq!(envelope["usage"]["completionTokens"], 7); + let requests = server.received_requests().await.unwrap(); + assert_eq!(requests.len(), 1); + let body: Value = requests[0].body_json().unwrap(); + assert!(body["system"].as_str().is_some()); + assert_eq!(body["messages"].as_array().unwrap().len(), 1); + assert_eq!(body["messages"][0]["role"], "user"); + assert_eq!(body["max_tokens"], 16384); + assert!(body.get("choices").is_none()); +} + +#[tokio::test] +async fn openai_compatible_rejects_additional_authorization_without_leaking() { + let server = MockServer::start().await; + let endpoint_secret = "Bearer endpoint-secret-never-print"; + let dir = tempfile::tempdir().unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_ENDPOINT_AUTH_HEADER", "Authorization") + .env("POSTIL_ENDPOINT_AUTH_VALUE", endpoint_secret) + .args(["review", "--diff-file"]) + .arg(&diff) + .arg("--output-json") + .assert() + .code(2); + let stdout = String::from_utf8_lossy(&out.get_output().stdout); + let stderr = String::from_utf8_lossy(&out.get_output().stderr); + assert!(!stdout.contains(endpoint_secret)); + assert!(!stderr.contains(endpoint_secret)); + assert!(stderr.contains("cannot override provider-managed header")); + assert!(server.received_requests().await.unwrap().is_empty()); +} + +#[tokio::test] +async fn native_anthropic_retries_transient_status_with_the_same_shape() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/messages")) + .respond_with(ResponseTemplate::new(529).set_body_string("overloaded")) + .up_to_n_times(1) + .mount(&server) + .await; + Mock::given(method("POST")) + .and(path("/messages")) + .respond_with(ResponseTemplate::new(200).set_body_json(anthropic_content(json!([]), 3, 2))) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_API_FORMAT", "anthropic") + .args(["review", "--diff-file"]) + .arg(&diff) + .arg("--output-json") + .assert() + .success(); + let stderr = String::from_utf8_lossy(&out.get_output().stderr); + assert!(stderr.contains("retryable HTTP 529")); + let requests = server.received_requests().await.unwrap(); + assert_eq!(requests.len(), 2); + assert!( + requests + .iter() + .all(|request| request.url.path() == "/messages") + ); +} + +#[tokio::test] +async fn provider_errors_redact_provider_and_endpoint_auth_secrets() { + let server = MockServer::start().await; + let provider_secret = "provider-secret-never-print"; + let endpoint_secret = "endpoint-secret-never-print"; + Mock::given(method("POST")) + .and(path("/messages")) + .respond_with(ResponseTemplate::new(401).set_body_string(format!( + "bad credentials: {provider_secret} and {endpoint_secret}" + ))) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("MODEL_API_KEY", provider_secret) + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_API_FORMAT", "anthropic") + .env("POSTIL_ENDPOINT_AUTH_HEADER", "X-Private-Endpoint-Token") + .env("POSTIL_ENDPOINT_AUTH_VALUE", endpoint_secret) + .args(["review", "--diff-file"]) + .arg(&diff) + .arg("--output-json") + .assert() + .code(1); + let stderr = String::from_utf8_lossy(&out.get_output().stderr); + let stdout = String::from_utf8_lossy(&out.get_output().stdout); + assert!(!stderr.contains(provider_secret)); + assert!(!stderr.contains(endpoint_secret)); + assert!(!stdout.contains(provider_secret)); + assert!(!stdout.contains(endpoint_secret)); + assert!(stderr.contains("[REDACTED]")); + assert!(stdout.contains("[REDACTED]")); +} + +#[tokio::test] +async fn doctor_probes_native_anthropic_format() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/messages")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "content": [{"type": "text", "text": "p"}], + "usage": {"input_tokens": 1, "output_tokens": 1} + }))) + .mount(&server) + .await; + let dir = tempfile::tempdir().unwrap(); + assert!( + std::process::Command::new("git") + .args(["init", "--quiet"]) + .current_dir(dir.path()) + .status() + .unwrap() + .success() + ); + let out = postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_API_FORMAT", "anthropic") + .arg("doctor") + .assert() + .success(); + let stderr = String::from_utf8_lossy(&out.get_output().stderr); + assert!(stderr.contains("using anthropic")); +} + +#[tokio::test] +async fn doctor_probes_openai_compatible_format_by_default() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(header("authorization", "Bearer test-key")) + .respond_with(ResponseTemplate::new(200).set_body_json(llm_text("p"))) + .mount(&server) + .await; + let dir = tempfile::tempdir().unwrap(); + assert!( + std::process::Command::new("git") + .args(["init", "--quiet"]) + .current_dir(dir.path()) + .status() + .unwrap() + .success() + ); + let out = postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .arg("doctor") + .assert() + .success(); + let stderr = String::from_utf8_lossy(&out.get_output().stderr); + assert!(stderr.contains("using openai-compatible")); +} + fn write_diff(dir: &std::path::Path) -> std::path::PathBuf { let p = dir.join("change.diff"); std::fs::write(&p, DIFF).unwrap(); @@ -1373,11 +1594,11 @@ async fn garbage_output_fails_closed_after_repair_attempt() { let env: Value = serde_json::from_str(&String::from_utf8(out.get_output().stdout.clone()).unwrap()).unwrap(); assert_eq!(env["findings"][0]["path"], ".postil/model-output"); - assert_eq!(env["usage"]["promptTokens"], 640); - assert_eq!(env["usage"]["completionTokens"], 240); + assert_eq!(env["usage"]["promptTokens"], 480); + assert_eq!(env["usage"]["completionTokens"], 180); // Initial call + repair call for each default model in the retry roster. let requests = server.received_requests().await.unwrap(); - assert_eq!(requests.len(), 8); + assert_eq!(requests.len(), 6); let models: Vec = requests .iter() .map(|request| { @@ -1390,14 +1611,12 @@ async fn garbage_output_fails_closed_after_repair_attempt() { assert_eq!( models, vec![ - "deepseek/deepseek-v4-pro", - "deepseek/deepseek-v4-pro", - "google/gemini-3.1-flash-lite", - "google/gemini-3.1-flash-lite", + "z-ai/glm-5.2", + "z-ai/glm-5.2", "moonshotai/kimi-k2.7-code", "moonshotai/kimi-k2.7-code", - "mistralai/mistral-large-2512", - "mistralai/mistral-large-2512", + "deepseek/deepseek-v4-flash", + "deepseek/deepseek-v4-flash", ] ); } From 3dae818d02c340d7f9eb0a1000c6be5fa2d7e525 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Sun, 12 Jul 2026 23:58:08 +0000 Subject: [PATCH 02/11] Harden provider transport and Anthropic scoring --- Cargo.lock | 2 +- Cargo.toml | 2 +- README.md | 13 ++- src/config.rs | 46 +++++++++++ src/llm.rs | 213 +++++++++++++++++++++++++++++++++++++++++++++++--- src/review.rs | 2 +- tests/e2e.rs | 153 +++++++++++++++++++++++++++++++++++- 7 files changed, 413 insertions(+), 18 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 52d4c76..0fe75c4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1041,7 +1041,7 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "postil-cli" -version = "0.4.6" +version = "0.5.0" dependencies = [ "anyhow", "assert_cmd", diff --git a/Cargo.toml b/Cargo.toml index f0dd6e6..df62e9f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "postil-cli" -version = "0.4.6" +version = "0.5.0" edition = "2024" description = "Postil: a low-noise AI review gate. Silent on clean PRs, hard gate on real risk." license = "Apache-2.0" diff --git a/README.md b/README.md index fe6b1c8..bcad7d5 100644 --- a/README.md +++ b/README.md @@ -163,7 +163,8 @@ Environment: `POSTIL_API_KEY`, `OPENROUTER_API_KEY`, `MODEL_API_KEY`, or `LLM_API_KEY`, `POSTIL_API_BASE`, `POSTIL_API_FORMAT` (`openai-compatible` by default, or `anthropic`), `POSTIL_ENDPOINT_AUTH_HEADER` and `POSTIL_ENDPOINT_AUTH_VALUE` (optional additional authentication for a private -endpoint), `POSTIL_DETAILS_URL` (optional HTTP(S) target +endpoint), `POSTIL_ALLOW_PRIVATE_API_BASE=1` (explicit opt-in for a local or +private-network endpoint), `POSTIL_DETAILS_URL` (optional HTTP(S) target for GitHub check-run details links), `REVIEW_MODEL`, `REVIEW_MODEL_CASCADE`, `REVIEW_SCORER_MODEL`, `GITHUB_TOKEN`/`GITHUB_API_URL`, @@ -191,7 +192,13 @@ Postil rejects additional-header names that collide with `x-api-key`, `anthropic-version`, or `content-type`. OpenAI-compatible endpoints also reserve `Authorization` for the provider key; Anthropic endpoints may use an additional `Authorization` value alongside their provider-owned `x-api-key`. Postil never -prints credential values. +prints credential values. Provider requests do not follow redirects. Postil +resolves the API hostname once, rejects non-public addresses, and pins the HTTP +client to the accepted addresses while retaining hostname-based TLS checks. + +The built-in scorer roster uses OpenRouter model identifiers, so native Anthropic +skips implicit scoring. Set `model.scorer` or `REVIEW_SCORER_MODEL` to an +Anthropic model identifier to enable scoring through a native Anthropic endpoint. Local endpoints use the same OpenAI-compatible contract: @@ -199,12 +206,14 @@ Local endpoints use the same OpenAI-compatible contract: # Ollama ollama pull qwen3-coder:30b POSTIL_API_BASE=http://localhost:11434/v1 \ +POSTIL_ALLOW_PRIVATE_API_BASE=1 \ MODEL_API_KEY=ollama \ REVIEW_MODEL=qwen3-coder:30b \ postil doctor # vLLM, SGLang, LiteLLM, or another local gateway POSTIL_API_BASE=http://localhost:8000/v1 \ +POSTIL_ALLOW_PRIVATE_API_BASE=1 \ MODEL_API_KEY=local \ REVIEW_MODEL= \ postil review --staged --output json diff --git a/src/config.rs b/src/config.rs index b962ebf..cad5a96 100644 --- a/src/config.rs +++ b/src/config.rs @@ -217,6 +217,9 @@ pub struct Config { pub model: String, pub cascade: Vec, pub scorer: String, + /// True when the scorer was selected by config or environment rather than + /// inherited from the OpenRouter-oriented built-in defaults. + pub scorer_explicit: bool, pub api_base: String, pub api_format: ApiFormat, /// Run the first N models of [model + cascade] and keep agreeing findings. @@ -254,6 +257,7 @@ impl Default for Config { model: defaults.default_model.clone(), cascade: defaults.cascade.clone(), scorer: defaults.scorer_model.clone(), + scorer_explicit: false, api_base: DEFAULT_API_BASE.to_string(), api_format: ApiFormat::default(), consensus: 1, @@ -458,6 +462,7 @@ impl Config { } if let Some(s) = m.scorer { self.scorer = s; + self.scorer_explicit = true; } if let Some(b) = m.api_base { // `model.apiBase` from `.postil.yaml` is repo-controlled, and the @@ -555,6 +560,7 @@ impl Config { && !s.is_empty() { self.scorer = s; + self.scorer_explicit = true; } if let Ok(b) = std::env::var("POSTIL_API_BASE") && !b.is_empty() @@ -582,6 +588,18 @@ impl Config { /// Scorer models to try, in order, deduplicated. pub fn scorer_chain(&self) -> Vec { + if self.api_format == ApiFormat::Anthropic { + let defaults = model_defaults(); + return if self.scorer.starts_with("claude-") + || (self.scorer_explicit + && self.scorer != defaults.scorer_model + && self.scorer != defaults.scorer_fallback) + { + vec![self.scorer.clone()] + } else { + Vec::new() + }; + } let mut chain = vec![self.scorer.clone()]; let fallback = model_defaults().scorer_fallback.clone(); if !chain.contains(&fallback) { @@ -589,6 +607,10 @@ impl Config { } chain } + + pub fn scorer_enabled(&self) -> bool { + !self.scorer_chain().is_empty() + } } /// Whether a repo-controlled `model.apiBase` may be applied. Opt-in only: @@ -972,6 +994,30 @@ fallback = "example/scorer-fallback" ); } + #[test] + fn native_anthropic_skips_implicit_openrouter_scorers() { + let mut config = Config { + api_format: ApiFormat::Anthropic, + ..Config::default() + }; + assert!(!config.scorer_enabled()); + assert!(config.scorer_chain().is_empty()); + + let generated_default: FileConfig = serde_yaml::from_str(&format!( + "model:\n scorer: {}\n", + model_defaults().scorer_model + )) + .unwrap(); + config.apply_file(generated_default).unwrap(); + assert!(config.scorer_chain().is_empty()); + + let file: FileConfig = + serde_yaml::from_str("model:\n scorer: claude-haiku-4-5\n").unwrap(); + config.apply_file(file).unwrap(); + assert!(config.scorer_enabled()); + assert_eq!(config.scorer_chain(), vec!["claude-haiku-4-5"]); + } + #[test] fn anthropic_api_format_parses_from_postil_config() { let f: FileConfig = serde_yaml::from_str("model:\n apiFormat: anthropic\n").unwrap(); diff --git a/src/llm.rs b/src/llm.rs index 4ebabc2..4794c8a 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -4,6 +4,8 @@ //! OpenAI-compatible endpoints use `POST {base}/chat/completions` by default. //! Native Anthropic endpoints use `POST {base}/messages` when explicitly selected. +use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr, ToSocketAddrs}; +use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant}; use anyhow::{Context, Result, anyhow}; @@ -155,7 +157,7 @@ fn default_confidence() -> f64 { #[derive(Clone)] pub struct LlmClient { - http: reqwest::Client, + http: Arc>>, api_base: String, api_key: String, api_format: ApiFormat, @@ -201,6 +203,7 @@ const REQUEST_TIMEOUT_ENV: &str = "POSTIL_LLM_REQUEST_TIMEOUT_SECS"; const TOTAL_TIMEOUT_ENV: &str = "POSTIL_LLM_TOTAL_TIMEOUT_SECS"; const ENDPOINT_AUTH_HEADER_ENV: &str = "POSTIL_ENDPOINT_AUTH_HEADER"; const ENDPOINT_AUTH_VALUE_ENV: &str = "POSTIL_ENDPOINT_AUTH_VALUE"; +const ALLOW_PRIVATE_API_BASE_ENV: &str = "POSTIL_ALLOW_PRIVATE_API_BASE"; const ALWAYS_MANAGED_HEADERS: &[&str] = &["x-api-key", "anthropic-version", "content-type"]; /// Marker context attached to transport/provider-level failures (endpoint @@ -225,6 +228,12 @@ fn timeout_status(status: u16) -> bool { matches!(status, 408 | 504) } +fn reqwest_error(error: &anyhow::Error) -> Option<&reqwest::Error> { + error + .chain() + .find_map(|cause| cause.downcast_ref::()) +} + #[derive(Debug, Clone, Copy)] enum LlmPhase { Review, @@ -327,7 +336,7 @@ impl LlmClient { // The attempt timeout wraps both sending the request and consuming // the complete response body, so header and body stalls take the // same retry path. - http: reqwest::Client::builder().build()?, + http: Arc::new(Mutex::new(None)), api_base: cfg.api_base.trim_end_matches('/').to_string(), api_key, api_format: cfg.api_format, @@ -800,7 +809,7 @@ impl LlmClient { return Err(anyhow!("model endpoint returned {status}: {snippet}")); } Err(error) - if error.is_timeout() + if reqwest_error(&error).is_some_and(reqwest::Error::is_timeout) && timeout_retries < TIMEOUT_RETRIES && retries < TRANSIENT_RETRIES => { @@ -817,7 +826,10 @@ impl LlmClient { self.sleep_with_budget(phase, wait).await?; attempt_timeout = self.timeout_retry_timeout; } - Err(error) if error.is_connect() && retries < TRANSIENT_RETRIES => { + Err(error) + if reqwest_error(&error).is_some_and(reqwest::Error::is_connect) + && retries < TRANSIENT_RETRIES => + { retries += 1; let wait = Duration::from_secs(2 * retries as u64); eprintln!( @@ -831,9 +843,7 @@ impl LlmClient { attempt_timeout = self.request_timeout; } Err(error) => { - return Err( - anyhow::Error::from(error).context("request to model endpoint failed") - ); + return Err(error.context("request to model endpoint failed")); } } }; @@ -923,20 +933,19 @@ impl LlmClient { async fn request_once( &self, body: &serde_json::Value, - ) -> std::result::Result<(reqwest::StatusCode, String), reqwest::Error> { + ) -> Result<(reqwest::StatusCode, String)> { + let http = self.http_client()?; let mut request = match self.api_format { ApiFormat::OpenaiCompatible => { let url = format!("{}/chat/completions", self.api_base); - self.http - .post(&url) + http.post(&url) .bearer_auth(&self.api_key) .header("HTTP-Referer", "https://postil.dev") .header("X-Title", "Postil") } ApiFormat::Anthropic => { let url = format!("{}/messages", self.api_base); - self.http - .post(&url) + http.post(&url) .header("x-api-key", &self.api_key) .header("anthropic-version", ANTHROPIC_VERSION) } @@ -950,6 +959,19 @@ impl LlmClient { Ok((status, text)) } + fn http_client(&self) -> Result { + let mut client = self + .http + .lock() + .map_err(|_| anyhow!("model provider HTTP client lock is poisoned"))?; + if let Some(client) = client.as_ref() { + return Ok(client.clone()); + } + let built = secure_http_client(&self.api_base)?; + *client = Some(built.clone()); + Ok(built) + } + fn remaining_budget(&self, phase: LlmPhase) -> Result> { let deadline = match phase { LlmPhase::Review => self.review_deadline, @@ -1055,6 +1077,122 @@ struct AnthropicUsage { output_tokens: Option, } +fn secure_http_client(api_base: &str) -> Result { + let (hostname, addresses) = resolve_api_endpoint(api_base)?; + reqwest::Client::builder() + // Provider credentials and prompts must never follow an endpoint's + // redirect to another origin or into an internal network. + .redirect(reqwest::redirect::Policy::none()) + // Connect only to the addresses approved by the single resolution + // above. Reqwest retains the URL hostname for TLS SNI/certificate + // verification while replacing DNS lookup results with this set. + .resolve_to_addrs(&hostname, &addresses) + .build() + .context("build model provider HTTP client") +} + +fn resolve_api_endpoint(api_base: &str) -> Result<(String, Vec)> { + resolve_api_endpoint_with(api_base, |hostname, port| { + (hostname, port) + .to_socket_addrs() + .map(|items| items.collect()) + }) +} + +fn resolve_api_endpoint_with(api_base: &str, resolver: F) -> Result<(String, Vec)> +where + F: FnOnce(&str, u16) -> std::io::Result>, +{ + let url = reqwest::Url::parse(api_base).context("model API base must be an absolute URL")?; + anyhow::ensure!( + matches!(url.scheme(), "http" | "https"), + "model API base must use HTTP or HTTPS" + ); + anyhow::ensure!( + url.username().is_empty() && url.password().is_none(), + "model API base must not contain credentials" + ); + let hostname = url + .host_str() + .filter(|hostname| !hostname.is_empty()) + .context("model API base must include a hostname")? + .to_string(); + let port = url + .port_or_known_default() + .context("model API base must include a port for its URL scheme")?; + let addresses = resolver(&hostname, port) + .with_context(|| format!("model API hostname {hostname:?} could not be resolved"))?; + anyhow::ensure!( + !addresses.is_empty(), + "model API hostname {hostname:?} did not resolve to any addresses" + ); + let allow_private = std::env::var(ALLOW_PRIVATE_API_BASE_ENV) + .map(|value| value == "1" || value.eq_ignore_ascii_case("true")) + .unwrap_or(false); + if !allow_private { + for address in &addresses { + anyhow::ensure!( + is_public_ip(address.ip()), + "model API hostname {hostname:?} resolved to a private, loopback, link-local, or non-public address" + ); + } + } + Ok((hostname, addresses)) +} + +fn is_public_ip(address: IpAddr) -> bool { + match address { + IpAddr::V4(address) => is_public_ipv4(address), + IpAddr::V6(address) => is_public_ipv6(address), + } +} + +fn is_public_ipv4(address: Ipv4Addr) -> bool { + let [a, b, c, _] = address.octets(); + !(a == 0 + || a == 10 + || a == 127 + || (a == 100 && (64..=127).contains(&b)) + || (a == 169 && b == 254) + || (a == 172 && (16..=31).contains(&b)) + || (a == 192 && b == 0 && c == 0) + || (a == 192 && b == 0 && c == 2) + || (a == 192 && b == 168) + || (a == 198 && (b == 18 || b == 19)) + || (a == 198 && b == 51 && c == 100) + || (a == 203 && b == 0 && c == 113) + || a >= 224) +} + +fn is_public_ipv6(address: Ipv6Addr) -> bool { + if let Some(mapped) = address.to_ipv4_mapped() { + return is_public_ipv4(mapped); + } + let segments = address.segments(); + // Deprecated IPv4-compatible addresses retain the embedded IPv4's + // reachability semantics even though they are not `::ffff:` mapped. + if segments[..6].iter().all(|segment| *segment == 0) { + return false; + } + // The well-known NAT64 prefix embeds the destination IPv4 in the final + // 32 bits. Reject it when it would translate to a non-public destination. + if segments[..6] == [0x0064, 0xff9b, 0, 0, 0, 0] { + return is_public_ipv4(Ipv4Addr::new( + (segments[6] >> 8) as u8, + segments[6] as u8, + (segments[7] >> 8) as u8, + segments[7] as u8, + )); + } + !(address.is_unspecified() + || address.is_loopback() + || address.is_multicast() + || (segments[0] & 0xfe00) == 0xfc00 + || (segments[0] & 0xffc0) == 0xfe80 + || (segments[0] & 0xffc0) == 0xfec0 + || (segments[0] == 0x2001 && segments[1] == 0x0db8)) +} + fn endpoint_auth_from_env(api_format: ApiFormat) -> Result> { let header = std::env::var(ENDPOINT_AUTH_HEADER_ENV) .ok() @@ -1381,6 +1519,57 @@ mod tests { assert_eq!(anthropic.value, "secret-value"); } + #[test] + fn api_endpoint_resolution_rejects_any_non_public_result() { + let _lock = env_lock().lock().unwrap(); + let _env = EnvRestore::capture(&[ALLOW_PRIVATE_API_BASE_ENV]); + EnvRestore::remove(ALLOW_PRIVATE_API_BASE_ENV); + let error = resolve_api_endpoint_with("https://models.example/v1", |hostname, port| { + assert_eq!(hostname, "models.example"); + assert_eq!(port, 443); + Ok(vec![ + "8.8.8.8:443".parse().unwrap(), + "169.254.169.254:443".parse().unwrap(), + ]) + }) + .expect_err("a mixed public/private DNS answer must fail closed"); + assert!(error.to_string().contains("non-public address")); + } + + #[test] + fn api_endpoint_resolution_preserves_public_addresses_for_pinning() { + let _lock = env_lock().lock().unwrap(); + let _env = EnvRestore::capture(&[ALLOW_PRIVATE_API_BASE_ENV]); + EnvRestore::remove(ALLOW_PRIVATE_API_BASE_ENV); + let expected = vec![ + "8.8.8.8:8443".parse().unwrap(), + "[2606:4700:4700::1111]:8443".parse().unwrap(), + ]; + let (hostname, addresses) = resolve_api_endpoint_with( + "https://models.example:8443/v1", + |_, _| Ok(expected.clone()), + ) + .unwrap(); + assert_eq!(hostname, "models.example"); + assert_eq!(addresses, expected); + } + + #[test] + fn api_endpoint_rejects_ipv4_mapped_compatible_and_nat64_private_targets() { + for address in [ + "::ffff:127.0.0.1", + "::a00:1", + "64:ff9b::a9fe:a9fe", + "fec0::1", + ] { + assert!( + !is_public_ip(address.parse().unwrap()), + "accepted {address}" + ); + } + assert!(is_public_ip("64:ff9b::808:808".parse().unwrap())); + } + impl Drop for EnvRestore { fn drop(&mut self) { for (name, value) in &self.saved { diff --git a/src/review.rs b/src/review.rs index 2105ab1..9cc45fd 100644 --- a/src/review.rs +++ b/src/review.rs @@ -684,7 +684,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> full_review_trustworthy = true; summary = model_review.summary; let mut kept = outcome.kept; - if !kept.is_empty() { + if !kept.is_empty() && cfg.scorer_enabled() { let inputs = scorer_inputs(&parsed, &kept); let scorer_system = prompt::scorer_system_prompt(cfg); let scorer_user = prompt::scorer_user_prompt(&inputs); diff --git a/tests/e2e.rs b/tests/e2e.rs index f25ba75..7491ca5 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -77,6 +77,13 @@ fn anthropic_content(findings: Value, input_tokens: u64, output_tokens: u64) -> }) } +fn anthropic_text(text: &str, input_tokens: u64, output_tokens: u64) -> Value { + json!({ + "content": [{"type": "text", "text": text}], + "usage": {"input_tokens": input_tokens, "output_tokens": output_tokens} + }) +} + fn finding_at(line: u32, severity: &str, confidence: f64) -> Value { json!({ "path": "src/auth.rs", @@ -129,10 +136,15 @@ fn postil() -> Command { .env_remove("POSTIL_API_FORMAT") .env_remove("POSTIL_ENDPOINT_AUTH_HEADER") .env_remove("POSTIL_ENDPOINT_AUTH_VALUE") + .env_remove("POSTIL_ALLOW_PRIVATE_API_BASE") .env_remove("POSTIL_DETAILS_URL") .env_remove("GITHUB_SERVER_URL") .env_remove("POSTIL_ENABLE_BITBUCKET_INCREMENTAL") - .env("MODEL_API_KEY", "test-key"); + .env("MODEL_API_KEY", "test-key") + // Mock providers bind loopback. Production and normal CLI invocations + // reject private API endpoints unless this explicit local-only escape + // hatch is set by the caller. + .env("POSTIL_ALLOW_PRIVATE_API_BASE", "1"); cmd } @@ -179,6 +191,98 @@ async fn native_anthropic_review_uses_messages_shape_auth_and_usage() { assert!(body.get("choices").is_none()); } +#[tokio::test] +async fn native_anthropic_findings_skip_incompatible_default_scorer() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/messages")) + .and(header("x-api-key", "anthropic-provider-key")) + .respond_with(ResponseTemplate::new(200).set_body_json(anthropic_content( + json!([finding_at(42, "warn", 0.9)]), + 17, + 9, + ))) + .expect(1) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("MODEL_API_KEY", "anthropic-provider-key") + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_API_FORMAT", "anthropic") + .args(["review", "--diff-file"]) + .arg(&diff) + .arg("--output-json") + .assert() + .success(); + + let envelope: Value = serde_json::from_slice(&out.get_output().stdout).unwrap(); + assert_eq!(envelope["findings"].as_array().unwrap().len(), 1); + assert!(envelope["scorerModel"].is_null()); + assert!(envelope["scorerError"].is_null()); + assert_eq!(envelope["usage"]["promptTokens"], 17); + assert_eq!(envelope["usage"]["completionTokens"], 9); +} + +#[tokio::test] +async fn native_anthropic_findings_use_explicit_native_scorer() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/messages")) + .and(body_string_contains("\"model\":\"claude-sonnet-4-6\"")) + .respond_with(ResponseTemplate::new(200).set_body_json(anthropic_content( + json!([finding_at(42, "warn", 0.9)]), + 17, + 9, + ))) + .expect(1) + .mount(&server) + .await; + Mock::given(method("POST")) + .and(path("/messages")) + .and(body_string_contains("\"model\":\"claude-haiku-4-5\"")) + .respond_with( + ResponseTemplate::new(200).set_body_json(anthropic_text( + &json!([{ + "index": 0, + "confidence": 0.82, + "kind": "risk", + "reason": "The changed line contains the reported flow." + }]) + .to_string(), + 5, + 3, + )), + ) + .expect(1) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("MODEL_API_KEY", "anthropic-provider-key") + .env("POSTIL_API_BASE", server.uri()) + .env("POSTIL_API_FORMAT", "anthropic") + .env("REVIEW_MODEL", "claude-sonnet-4-6") + .env("REVIEW_SCORER_MODEL", "claude-haiku-4-5") + .args(["review", "--diff-file"]) + .arg(&diff) + .arg("--output-json") + .assert() + .success(); + + let envelope: Value = serde_json::from_slice(&out.get_output().stdout).unwrap(); + assert_eq!(envelope["scorerModel"], "claude-haiku-4-5"); + assert_eq!(envelope["findings"][0]["scorerConfidence"], 0.82); + assert_eq!(envelope["usage"]["promptTokens"], 22); + assert_eq!(envelope["usage"]["completionTokens"], 12); +} + #[tokio::test] async fn openai_compatible_rejects_additional_authorization_without_leaking() { let server = MockServer::start().await; @@ -336,6 +440,53 @@ async fn doctor_probes_openai_compatible_format_by_default() { assert!(stderr.contains("using openai-compatible")); } +#[tokio::test] +async fn doctor_does_not_follow_provider_redirect_or_forward_auth() { + let redirect_target = MockServer::start().await; + let provider = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(header("authorization", "Bearer provider-secret")) + .and(header("x-private-endpoint-token", "endpoint-secret")) + .respond_with( + ResponseTemplate::new(307) + .insert_header("Location", format!("{}/captured", redirect_target.uri())), + ) + .expect(1) + .mount(&provider) + .await; + + let dir = tempfile::tempdir().unwrap(); + assert!( + std::process::Command::new("git") + .args(["init", "--quiet"]) + .current_dir(dir.path()) + .status() + .unwrap() + .success() + ); + let out = postil() + .current_dir(dir.path()) + .env("MODEL_API_KEY", "provider-secret") + .env("POSTIL_API_BASE", provider.uri()) + .env("POSTIL_ENDPOINT_AUTH_HEADER", "X-Private-Endpoint-Token") + .env("POSTIL_ENDPOINT_AUTH_VALUE", "endpoint-secret") + .arg("doctor") + .assert() + .failure(); + let stderr = String::from_utf8_lossy(&out.get_output().stderr); + assert!(stderr.contains("307")); + assert!(!stderr.contains("provider-secret")); + assert!(!stderr.contains("endpoint-secret")); + assert!( + redirect_target + .received_requests() + .await + .unwrap() + .is_empty() + ); +} + fn write_diff(dir: &std::path::Path) -> std::path::PathBuf { let p = dir.join("change.diff"); std::fs::write(&p, DIFF).unwrap(); From 7ed6235f498d9b4f0fe3409f7da0abb976a10a70 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 00:02:19 +0000 Subject: [PATCH 03/11] Allow private endpoints in hermetic benches --- bench/src/harness.ts | 1 + bench/src/scorer-eval.ts | 1 + 2 files changed, 2 insertions(+) diff --git a/bench/src/harness.ts b/bench/src/harness.ts index feba051..f0c1e24 100644 --- a/bench/src/harness.ts +++ b/bench/src/harness.ts @@ -372,6 +372,7 @@ function isolatedEnv( GIT_CONFIG_NOSYSTEM: "1", GIT_TERMINAL_PROMPT: "0", POSTIL_API_BASE: modelBaseUrl, + POSTIL_ALLOW_PRIVATE_API_BASE: "1", POSTIL_API_KEY: "benchmark-api-key", GITHUB_API_URL: githubBaseUrl, GITHUB_TOKEN: "benchmark-github-token", diff --git a/bench/src/scorer-eval.ts b/bench/src/scorer-eval.ts index bf5372e..85a8f23 100644 --- a/bench/src/scorer-eval.ts +++ b/bench/src/scorer-eval.ts @@ -379,6 +379,7 @@ export function isolatedEnv( GIT_CONFIG_NOSYSTEM: "1", GIT_TERMINAL_PROMPT: "0", POSTIL_API_BASE: modelBaseUrl, + POSTIL_ALLOW_PRIVATE_API_BASE: "1", POSTIL_API_KEY: "scorer-eval-proxy-key", GITHUB_API_URL: githubBaseUrl, GITHUB_TOKEN: "benchmark-github-token", From b129ccf34710f942d3beebee737ac348b60e754d Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 00:26:48 +0000 Subject: [PATCH 04/11] Emit private respond usage receipts --- README.md | 12 ++++++-- src/llm.rs | 49 +++++++++++++++++++++++++++-- src/respond.rs | 84 ++++++++++++++++++++++++++++++++++++++++++++++++-- tests/e2e.rs | 76 +++++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 213 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index bcad7d5..e1cee94 100644 --- a/README.md +++ b/README.md @@ -164,13 +164,21 @@ Environment: `POSTIL_API_KEY`, `OPENROUTER_API_KEY`, `MODEL_API_KEY`, or default, or `anthropic`), `POSTIL_ENDPOINT_AUTH_HEADER` and `POSTIL_ENDPOINT_AUTH_VALUE` (optional additional authentication for a private endpoint), `POSTIL_ALLOW_PRIVATE_API_BASE=1` (explicit opt-in for a local or -private-network endpoint), `POSTIL_DETAILS_URL` (optional HTTP(S) target -for GitHub check-run details links), +private-network endpoint), `POSTIL_USAGE_RECEIPT_PATH` (optional worker-owned +path for a successful `respond` usage receipt), `POSTIL_DETAILS_URL` (optional +HTTP(S) target for GitHub check-run details links), `REVIEW_MODEL`, `REVIEW_MODEL_CASCADE`, `REVIEW_SCORER_MODEL`, `GITHUB_TOKEN`/`GITHUB_API_URL`, `GITLAB_TOKEN`/`GITLAB_API_URL`, `BITBUCKET_TOKEN`/`BITBUCKET_USER`/`BITBUCKET_API_URL`, `AZURE_DEVOPS_TOKEN`/`AZURE_DEVOPS_API_URL`. +When `POSTIL_USAGE_RECEIPT_PATH` is set, `postil respond` creates that path with +mode `0600` before provider access and writes JSON only after the reply succeeds. +The version 1 receipt contains `operation: "respond"`, aggregate +`promptTokens`/`completionTokens`, and a `models` array with token usage for each +model that returned cost-relevant usage during cascade attempts. The receipt is +never written to stdout, stderr, or command arguments. The caller owns deletion. + ## Models and local inference See the measured benchmark results at [postil.dev/docs/models](https://postil.dev/docs/models), diff --git a/src/llm.rs b/src/llm.rs index 4794c8a..d69a970 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -40,6 +40,20 @@ pub struct ScorerReview { pub usage: Usage, } +#[derive(Debug, Clone)] +pub struct ModelUsage { + pub model: String, + pub usage: Usage, +} + +#[derive(Debug, Clone)] +pub struct Answer { + pub content: String, + pub model_used: String, + pub usage: Usage, + pub models: Vec, +} + #[derive(Debug)] pub struct ModelError { error: anyhow::Error, @@ -197,6 +211,7 @@ const TIMEOUT_RETRY_CAP_SECS: u64 = 90; const REVIEW_MAX_TOKENS: u32 = 16384; const SCORER_MAX_TOKENS: u32 = 4096; const ANTHROPIC_DEFAULT_MAX_TOKENS: u32 = 4096; +const RESPOND_MAX_TOKENS: u32 = 4096; const ANTHROPIC_VERSION: &str = "2023-06-01"; pub(crate) const DEFAULT_REQUEST_TIMEOUT_SECS: u64 = 480; const REQUEST_TIMEOUT_ENV: &str = "POSTIL_LLM_REQUEST_TIMEOUT_SECS"; @@ -489,17 +504,45 @@ impl LlmClient { /// Free-form answer (no JSON contract). Used by the interactive bot to reply /// to a maintainer's question or mention. Tries the model chain in order. - pub async fn answer(&self, cfg: &Config, system: &str, user: &str) -> Result<(String, String)> { + pub async fn answer(&self, cfg: &Config, system: &str, user: &str) -> Result { let mut usage = Usage::default(); + let mut models = Vec::new(); let mut last_err = None; for model in cfg.model_chain() { + let mut model_usage = Usage::default(); match self - .chat(&model, system, user, &mut usage, None, LlmPhase::Total) + .chat( + &model, + system, + user, + &mut model_usage, + Some(RESPOND_MAX_TOKENS), + LlmPhase::Total, + ) .await { - Ok(content) => return Ok((content.trim().to_string(), model)), + Ok(content) => { + add_usage(&mut usage, model_usage); + models.push(ModelUsage { + model: model.clone(), + usage: model_usage, + }); + return Ok(Answer { + content: content.trim().to_string(), + model_used: model, + usage, + models, + }); + } Err(e) => { eprintln!("postil: model {model} failed: {e:#}"); + if model_usage.prompt_tokens > 0 || model_usage.completion_tokens > 0 { + add_usage(&mut usage, model_usage); + models.push(ModelUsage { + model: model.clone(), + usage: model_usage, + }); + } if e.downcast_ref::().is_some() { return Err(e); } diff --git a/src/respond.rs b/src/respond.rs index 754c8ce..d82d801 100644 --- a/src/respond.rs +++ b/src/respond.rs @@ -6,16 +6,20 @@ //! issues and pulls; Bitbucket and Azure DevOps are scoped to PRs (their issue //! trackers / work items use endpoints we cannot verify against a live host). +use std::fs::{File, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; use std::path::PathBuf; use anyhow::{Context, Result, anyhow}; +use serde::Serialize; use crate::config::Config; use crate::diff; use crate::forge::{ Forge, ThreadKind, azure::Azure, bitbucket::Bitbucket, github::GitHub, gitlab::GitLab, }; -use crate::llm::LlmClient; +use crate::llm::{Answer, LlmClient}; use crate::prompt; use crate::review::ForgeKind; @@ -36,6 +40,68 @@ pub struct RespondArgs { } const MAX_DIFF_BYTES: usize = 200_000; +const USAGE_RECEIPT_PATH_ENV: &str = "POSTIL_USAGE_RECEIPT_PATH"; + +#[derive(Serialize)] +#[serde(rename_all = "camelCase")] +struct RespondUsageReceipt<'a> { + version: u32, + operation: &'static str, + prompt_tokens: u64, + completion_tokens: u64, + models: Vec>, +} + +#[derive(Serialize)] +#[serde(rename_all = "camelCase")] +struct RespondModelUsage<'a> { + model: &'a str, + prompt_tokens: u64, + completion_tokens: u64, +} + +struct UsageReceiptWriter(File); + +impl UsageReceiptWriter { + fn from_env() -> Result> { + let Some(path) = std::env::var_os(USAGE_RECEIPT_PATH_ENV) else { + return Ok(None); + }; + anyhow::ensure!( + !path.is_empty(), + "{USAGE_RECEIPT_PATH_ENV} must not be empty" + ); + let file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(PathBuf::from(path)) + .context("creating usage receipt")?; + Ok(Some(Self(file))) + } + + fn commit(mut self, answer: &Answer) -> Result<()> { + let receipt = RespondUsageReceipt { + version: 1, + operation: "respond", + prompt_tokens: answer.usage.prompt_tokens, + completion_tokens: answer.usage.completion_tokens, + models: answer + .models + .iter() + .map(|model| RespondModelUsage { + model: &model.model, + prompt_tokens: model.usage.prompt_tokens, + completion_tokens: model.usage.completion_tokens, + }) + .collect(), + }; + serde_json::to_writer(&mut self.0, &receipt).context("serializing usage receipt")?; + self.0.write_all(b"\n").context("writing usage receipt")?; + self.0.sync_all().context("syncing usage receipt")?; + Ok(()) + } +} pub async fn run(args: RespondArgs) -> Result { let cwd = std::env::current_dir()?; @@ -54,6 +120,7 @@ pub async fn run(args: RespondArgs) -> Result { .or_else(|| std::env::var("POSTIL_COMMENT").ok()) .filter(|c| !c.trim().is_empty()) .ok_or_else(|| anyhow!("the mention text is required: --comment or POSTIL_COMMENT"))?; + let usage_receipt = UsageReceiptWriter::from_env()?; // The number the mention is on, and whether it is a PR/MR or an issue. let (number, kind) = match (args.pr, args.issue) { @@ -75,6 +142,7 @@ pub async fn run(args: RespondArgs) -> Result { kind, &comment, args.no_post, + usage_receipt, ) .await } @@ -87,6 +155,7 @@ pub async fn run(args: RespondArgs) -> Result { kind, &comment, args.no_post, + usage_receipt, ) .await } @@ -99,6 +168,7 @@ pub async fn run(args: RespondArgs) -> Result { kind, &comment, args.no_post, + usage_receipt, ) .await } @@ -111,6 +181,7 @@ pub async fn run(args: RespondArgs) -> Result { kind, &comment, args.no_post, + usage_receipt, ) .await } @@ -127,6 +198,7 @@ async fn respond_with( kind: ThreadKind, comment: &str, no_post: bool, + usage_receipt: Option, ) -> Result { let context = build_context(&forge, repo, number, kind).await?; @@ -136,9 +208,12 @@ async fn respond_with( comment.trim() ); let client = LlmClient::from_env(cfg)?; - let (answer, model_used) = client.answer(cfg, &system, &user).await?; + let answer = client.answer(cfg, &system, &user).await?; - let reply = format!("{answer}\n\nPostil · {model_used}"); + let reply = format!( + "{}\n\nPostil · {}", + answer.content, answer.model_used + ); if no_post { println!("{reply}"); @@ -149,6 +224,9 @@ async fn respond_with( .context("posting reply")?; eprintln!("postil: replied on {repo}#{number}"); } + if let Some(writer) = usage_receipt { + writer.commit(&answer)?; + } Ok(0) } diff --git a/tests/e2e.rs b/tests/e2e.rs index 7491ca5..e61d297 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -3532,6 +3532,82 @@ async fn respond_to_pr_mention_posts_grounded_reply() { assert!(text.contains("Postil ·")); // footer with model attribution } +#[tokio::test] +async fn respond_writes_private_usage_receipt_across_model_fallback() { + use std::os::unix::fs::PermissionsExt; + + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("primary-model")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "choices": [], + "usage": {"prompt_tokens": 10, "completion_tokens": 2} + }))) + .expect(1) + .mount(&server) + .await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("backup-model")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "choices": [{"message": {"content": "Use a bounded worker pool."}}], + "usage": {"prompt_tokens": 20, "completion_tokens": 3} + }))) + .expect(1) + .mount(&server) + .await; + Mock::given(method("GET")) + .and(path("/repos/acme/api/issues/9")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "title": "Timeouts", "body": "Requests hang under load." + }))) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let receipt_path = dir.path().join("respond-usage.json"); + let out = postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("GITHUB_API_URL", server.uri()) + .env("GITHUB_TOKEN", "gh-test-token") + .env("REVIEW_MODEL", "primary-model") + .env("REVIEW_MODEL_CASCADE", "backup-model") + .env("POSTIL_USAGE_RECEIPT_PATH", &receipt_path) + .args([ + "respond", + "--repo", + "acme/api", + "--issue", + "9", + "--comment", + "@postil how should this be bounded?", + "--no-post", + ]) + .assert() + .success(); + + let stdout = String::from_utf8_lossy(&out.get_output().stdout); + assert!(stdout.contains("Use a bounded worker pool.")); + assert!(!stdout.contains("promptTokens")); + let receipt: Value = serde_json::from_slice(&std::fs::read(&receipt_path).unwrap()).unwrap(); + assert_eq!(receipt["version"], 1); + assert_eq!(receipt["operation"], "respond"); + assert_eq!(receipt["promptTokens"], 30); + assert_eq!(receipt["completionTokens"], 5); + assert_eq!(receipt["models"][0]["model"], "primary-model"); + assert_eq!(receipt["models"][1]["model"], "backup-model"); + assert_eq!( + std::fs::metadata(&receipt_path) + .unwrap() + .permissions() + .mode() + & 0o777, + 0o600, + ); +} + #[tokio::test] async fn respond_to_issue_mention_uses_issue_body() { let server = MockServer::start().await; From 9ea8f02ef2da7f3e6dd8009352cf20259a34524b Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 01:00:12 +0000 Subject: [PATCH 05/11] Attribute review usage by model --- README.md | 8 +++- src/envelope.rs | 16 +++++++ src/forge/github.rs | 2 + src/forge/mod.rs | 4 ++ src/llm.rs | 103 ++++++++++++++++++++++++++++++++++++++------ src/output.rs | 1 + src/plan.rs | 1 + src/respond.rs | 67 ++++++++++++++++++++++------ src/review.rs | 9 +++- src/sarif.rs | 1 + tests/e2e.rs | 35 +++++++++++++++ 11 files changed, 218 insertions(+), 29 deletions(-) diff --git a/README.md b/README.md index e1cee94..e8806bf 100644 --- a/README.md +++ b/README.md @@ -177,6 +177,8 @@ mode `0600` before provider access and writes JSON only after the reply succeeds The version 1 receipt contains `operation: "respond"`, aggregate `promptTokens`/`completionTokens`, and a `models` array with token usage for each model that returned cost-relevant usage during cascade attempts. The receipt is +synced before stdout or forge delivery, so a hosted worker can persist accounting +before it posts an answer. The receipt is never written to stdout, stderr, or command arguments. The caller owns deletion. ## Models and local inference @@ -263,7 +265,11 @@ Bitbucket incremental reviews are disabled unless `--output json` prints a stable versioned envelope (`summary`, `silent`, `findings`, `resolved`, `counts`, `confidenceBuckets`, `gate`, `modelUsed`, scorer metadata, -`usage`, SHAs) consumed by the hosted platform and `postil plan`. `--output yaml` and +aggregate `usage`, per-model `modelUsage`, SHAs) consumed by the hosted platform +and `postil plan`. `modelUsage` includes the successful generator, scorers, and +token-bearing failed fallbacks; its totals equal aggregate `usage`. Older v1 +envelopes omit this additive field. Failed attempts that report zero tokens are +omitted because they carry no billable usage. `--output yaml` and `--output csv` print the same review result in YAML or CSV. `--output-file ` writes the selected format to a file instead of stdout. `--output-json` is deprecated in v0.2.1 as an alias for `--output json` and emits a stderr warning. Schema: diff --git a/src/envelope.rs b/src/envelope.rs index 30ef527..78fce76 100644 --- a/src/envelope.rs +++ b/src/envelope.rs @@ -144,6 +144,17 @@ pub struct Usage { pub completion_tokens: u64, } +/// Token usage attributed to one provider model attempt. Entries include +/// successful generation/scoring calls and failed attempts that returned +/// provider usage, so hosted accounting can price the complete review. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ModelUsage { + pub model: String, + pub prompt_tokens: u64, + pub completion_tokens: u64, +} + #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] pub struct Envelope { @@ -167,6 +178,10 @@ pub struct Envelope { #[serde(skip_serializing_if = "Option::is_none", default)] pub scorer_disagreements: Option, pub usage: Usage, + /// Per-model usage for exact provider pricing. Older v1 envelopes omit + /// this additive field and are handled conservatively by the control plane. + #[serde(skip_serializing_if = "Vec::is_empty", default)] + pub model_usage: Vec, /// Wall-clock duration of the review engine run in milliseconds. #[serde(default)] pub duration_ms: u64, @@ -426,6 +441,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Usage::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/forge/github.rs b/src/forge/github.rs index 6825b0f..0ebbb79 100644 --- a/src/forge/github.rs +++ b/src/forge/github.rs @@ -491,6 +491,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Usage::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, @@ -522,6 +523,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Usage::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/forge/mod.rs b/src/forge/mod.rs index e57c410..21ec086 100644 --- a/src/forge/mod.rs +++ b/src/forge/mod.rs @@ -442,6 +442,7 @@ mod tests { prompt_tokens: 10, completion_tokens: 5, }, + model_usage: vec![], duration_ms: 1_250, base_sha: None, head_sha: Some("abcdef123456".into()), @@ -489,6 +490,7 @@ mod tests { scorer_error: Some("[click me](https://attacker.invalid)".into()), scorer_disagreements: None, usage: Default::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, @@ -549,6 +551,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Default::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, @@ -578,6 +581,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Default::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/llm.rs b/src/llm.rs index d69a970..df2de29 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -15,7 +15,7 @@ use serde_json::json; use crate::api_key; use crate::config::{ApiFormat, Config}; -use crate::envelope::{Finding, Kind, Usage}; +use crate::envelope::{Finding, Kind, ModelUsage, Usage}; #[derive(Debug, Clone)] pub struct ModelReview { @@ -23,6 +23,7 @@ pub struct ModelReview { pub findings: Vec, pub model_used: String, pub usage: Usage, + pub model_usage: Vec, } #[derive(Debug, Clone)] @@ -38,12 +39,7 @@ pub struct ScorerReview { pub scores: Vec, pub model_used: String, pub usage: Usage, -} - -#[derive(Debug, Clone)] -pub struct ModelUsage { - pub model: String, - pub usage: Usage, + pub model_usage: Vec, } #[derive(Debug, Clone)] @@ -58,17 +54,26 @@ pub struct Answer { pub struct ModelError { error: anyhow::Error, usage: Usage, + model_usage: Vec, } impl ModelError { fn new(error: anyhow::Error, usage: Usage) -> Self { - Self { error, usage } + Self { + error, + usage, + model_usage: Vec::new(), + } } pub fn usage(&self) -> Usage { self.usage } + pub fn model_usage(&self) -> &[ModelUsage] { + &self.model_usage + } + pub fn is_provider(&self) -> bool { self.error.downcast_ref::().is_some() } @@ -405,12 +410,23 @@ impl LlmClient { .collect(); let mut ok: Vec = Vec::new(); let mut failed_usage = Usage::default(); + let mut failed_model_usage = Vec::new(); let mut last_err: Option = None; for (model, handle) in handles { let model_log = log_text(&model); match handle.await { Ok(Ok(r)) => ok.push(r), Ok(Err(mut e)) => { + if e.model_usage.is_empty() + && (e.usage.prompt_tokens > 0 || e.usage.completion_tokens > 0) + { + e.model_usage.push(ModelUsage { + model: model.clone(), + prompt_tokens: e.usage.prompt_tokens, + completion_tokens: e.usage.completion_tokens, + }); + } + failed_model_usage.extend(e.model_usage.clone()); add_usage(&mut failed_usage, e.usage); e.usage = failed_usage; last_err = Some(e); @@ -424,10 +440,14 @@ impl LlmClient { // Wrap the last failure so its error class (provider vs // content) survives for gate.onError classification. 0 => Err(match last_err { - Some(e) => ModelError::new( - e.error.context(format!("all {n} consensus models failed")), - failed_usage, - ), + Some(e) => { + let mut error = ModelError::new( + e.error.context(format!("all {n} consensus models failed")), + failed_usage, + ); + error.model_usage = failed_model_usage; + error + } None => { ModelError::new(anyhow!("all {n} consensus models failed"), failed_usage) } @@ -435,16 +455,19 @@ impl LlmClient { 1 => { let mut review = ok.into_iter().next().unwrap(); add_usage(&mut review.usage, failed_usage); + review.model_usage.extend(failed_model_usage); Ok(review) } _ => { let mut review = consensus_merge(ok); add_usage(&mut review.usage, failed_usage); + review.model_usage.extend(failed_model_usage); Ok(review) } } } else { let mut failed_usage = Usage::default(); + let mut failed_model_usage = Vec::new(); let mut last_err = None; for (index, model) in chain.iter().enumerate() { let model_log = log_text(model); @@ -457,9 +480,20 @@ impl LlmClient { elapsed_text(started_at.elapsed()) ); add_usage(&mut r.usage, failed_usage); + r.model_usage.splice(0..0, failed_model_usage); return Ok(r); } Err(mut e) => { + if e.model_usage.is_empty() + && (e.usage.prompt_tokens > 0 || e.usage.completion_tokens > 0) + { + e.model_usage.push(ModelUsage { + model: model.clone(), + prompt_tokens: e.usage.prompt_tokens, + completion_tokens: e.usage.completion_tokens, + }); + } + failed_model_usage.extend(e.model_usage.clone()); let elapsed = elapsed_text(started_at.elapsed()); if e.is_deadline_exceeded() { add_usage(&mut failed_usage, e.usage); @@ -467,6 +501,7 @@ impl LlmClient { eprintln!( "postil: model {model_log} stopped after {elapsed}: {e}; cascade fallback is disabled after deadline exhaustion" ); + e.model_usage = failed_model_usage; return Err(e); } let has_fallback = index + 1 < chain.len(); @@ -498,6 +533,10 @@ impl LlmClient { } } Err(last_err + .map(|mut error| { + error.model_usage = failed_model_usage; + error + }) .unwrap_or_else(|| ModelError::new(anyhow!("empty model chain"), failed_usage))) } } @@ -525,7 +564,8 @@ impl LlmClient { add_usage(&mut usage, model_usage); models.push(ModelUsage { model: model.clone(), - usage: model_usage, + prompt_tokens: model_usage.prompt_tokens, + completion_tokens: model_usage.completion_tokens, }); return Ok(Answer { content: content.trim().to_string(), @@ -536,11 +576,16 @@ impl LlmClient { } Err(e) => { eprintln!("postil: model {model} failed: {e:#}"); + // Provider failures that report no tokens have no billable + // usage to attribute. Omit them rather than emitting a + // misleading accounting entry; token-bearing failures are + // retained and priced by the hosted control plane. if model_usage.prompt_tokens > 0 || model_usage.completion_tokens > 0 { add_usage(&mut usage, model_usage); models.push(ModelUsage { model: model.clone(), - usage: model_usage, + prompt_tokens: model_usage.prompt_tokens, + completion_tokens: model_usage.completion_tokens, }); } if e.downcast_ref::().is_some() { @@ -561,6 +606,7 @@ impl LlmClient { expected_len: usize, ) -> std::result::Result { let mut failed_usage = Usage::default(); + let mut failed_model_usage = Vec::new(); let mut last_err = None; let chain = cfg.scorer_chain(); for (index, model) in chain.iter().enumerate() { @@ -577,9 +623,20 @@ impl LlmClient { elapsed_text(started_at.elapsed()) ); add_usage(&mut r.usage, failed_usage); + r.model_usage.splice(0..0, failed_model_usage); return Ok(r); } Err(mut e) => { + if e.model_usage.is_empty() + && (e.usage.prompt_tokens > 0 || e.usage.completion_tokens > 0) + { + e.model_usage.push(ModelUsage { + model: model.clone(), + prompt_tokens: e.usage.prompt_tokens, + completion_tokens: e.usage.completion_tokens, + }); + } + failed_model_usage.extend(e.model_usage.clone()); let elapsed = elapsed_text(started_at.elapsed()); if e.is_deadline_exceeded() { add_usage(&mut failed_usage, e.usage); @@ -587,6 +644,7 @@ impl LlmClient { eprintln!( "postil: scorer {model_log} stopped after {elapsed}: {e}; scorer fallback is disabled after deadline exhaustion" ); + e.model_usage = failed_model_usage; return Err(e); } let has_fallback = index + 1 < chain.len(); @@ -616,6 +674,10 @@ impl LlmClient { } } Err(last_err + .map(|mut error| { + error.model_usage = failed_model_usage; + error + }) .unwrap_or_else(|| ModelError::new(anyhow!("empty scorer model chain"), failed_usage))) } @@ -731,6 +793,11 @@ impl LlmClient { scores, model_used: model.to_string(), usage, + model_usage: vec![ModelUsage { + model: model.to_string(), + prompt_tokens: usage.prompt_tokens, + completion_tokens: usage.completion_tokens, + }], }) } @@ -1432,6 +1499,11 @@ fn into_review(raw: RawReview, model: &str, usage: Usage) -> ModelReview { findings, model_used: model.to_string(), usage, + model_usage: vec![ModelUsage { + model: model.to_string(), + prompt_tokens: usage.prompt_tokens, + completion_tokens: usage.completion_tokens, + }], } } @@ -1446,6 +1518,7 @@ fn consensus_merge(runs: Vec) -> ModelReview { completion_tokens: runs.iter().map(|r| r.usage.completion_tokens).sum(), }; let models: Vec = runs.iter().map(|r| r.model_used.clone()).collect(); + let model_usage = runs.iter().flat_map(|r| r.model_usage.clone()).collect(); let summary = runs[0].summary.clone(); // Flatten in run order; greedy clustering then anchors each cluster on its // earliest report. @@ -1487,6 +1560,7 @@ fn consensus_merge(runs: Vec) -> ModelReview { findings: kept, model_used: format!("consensus({})", models.join(", ")), usage: total_usage, + model_usage, } } @@ -1930,6 +2004,7 @@ mod tests { prompt_tokens: 10, completion_tokens: 5, }, + model_usage: vec![], } } diff --git a/src/output.rs b/src/output.rs index a3ec76d..87a9232 100644 --- a/src/output.rs +++ b/src/output.rs @@ -274,6 +274,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Default::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/plan.rs b/src/plan.rs index 9c5566e..9fc9bf3 100644 --- a/src/plan.rs +++ b/src/plan.rs @@ -173,6 +173,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Usage::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/respond.rs b/src/respond.rs index d82d801..0ac197a 100644 --- a/src/respond.rs +++ b/src/respond.rs @@ -9,7 +9,8 @@ use std::fs::{File, OpenOptions}; use std::io::Write; use std::os::unix::fs::OpenOptionsExt; -use std::path::PathBuf; +use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicU64, Ordering}; use anyhow::{Context, Result, anyhow}; use serde::Serialize; @@ -60,7 +61,13 @@ struct RespondModelUsage<'a> { completion_tokens: u64, } -struct UsageReceiptWriter(File); +static RECEIPT_TEMP_SEQUENCE: AtomicU64 = AtomicU64::new(0); + +struct UsageReceiptWriter { + file: Option, + temp_path: PathBuf, + final_path: PathBuf, +} impl UsageReceiptWriter { fn from_env() -> Result> { @@ -71,13 +78,29 @@ impl UsageReceiptWriter { !path.is_empty(), "{USAGE_RECEIPT_PATH_ENV} must not be empty" ); + let final_path = PathBuf::from(path); + let parent = final_path.parent().unwrap_or_else(|| Path::new(".")); + let file_name = final_path + .file_name() + .ok_or_else(|| anyhow!("{USAGE_RECEIPT_PATH_ENV} must name a file"))? + .to_string_lossy(); + let sequence = RECEIPT_TEMP_SEQUENCE.fetch_add(1, Ordering::Relaxed); + let temp_path = parent.join(format!( + ".{file_name}.{}.{}.tmp", + std::process::id(), + sequence + )); let file = OpenOptions::new() .write(true) .create_new(true) .mode(0o600) - .open(PathBuf::from(path)) - .context("creating usage receipt")?; - Ok(Some(Self(file))) + .open(&temp_path) + .context("creating private usage receipt temporary file")?; + Ok(Some(Self { + file: Some(file), + temp_path, + final_path, + })) } fn commit(mut self, answer: &Answer) -> Result<()> { @@ -91,18 +114,33 @@ impl UsageReceiptWriter { .iter() .map(|model| RespondModelUsage { model: &model.model, - prompt_tokens: model.usage.prompt_tokens, - completion_tokens: model.usage.completion_tokens, + prompt_tokens: model.prompt_tokens, + completion_tokens: model.completion_tokens, }) .collect(), }; - serde_json::to_writer(&mut self.0, &receipt).context("serializing usage receipt")?; - self.0.write_all(b"\n").context("writing usage receipt")?; - self.0.sync_all().context("syncing usage receipt")?; + let file = self.file.as_mut().expect("usage receipt file is present"); + serde_json::to_writer(&mut *file, &receipt).context("serializing usage receipt")?; + file.write_all(b"\n").context("writing usage receipt")?; + file.sync_all().context("syncing usage receipt")?; + drop(self.file.take()); + std::fs::rename(&self.temp_path, &self.final_path) + .context("atomically publishing usage receipt")?; + if let Some(parent) = self.final_path.parent() { + File::open(parent) + .and_then(|directory| directory.sync_all()) + .context("syncing usage receipt directory")?; + } Ok(()) } } +impl Drop for UsageReceiptWriter { + fn drop(&mut self) { + let _ = std::fs::remove_file(&self.temp_path); + } +} + pub async fn run(args: RespondArgs) -> Result { let cwd = std::env::current_dir()?; let mut cfg = Config::load(&cwd, args.config.as_deref())?; @@ -215,6 +253,12 @@ async fn respond_with( answer.content, answer.model_used ); + // Hosted execution requires the durable usage receipt before any external + // delivery. Commit it before stdout or forge posting so the control plane + // can reconcile spend and own idempotent delivery. + if let Some(writer) = usage_receipt { + writer.commit(&answer)?; + } if no_post { println!("{reply}"); } else { @@ -224,9 +268,6 @@ async fn respond_with( .context("posting reply")?; eprintln!("postil: replied on {repo}#{number}"); } - if let Some(writer) = usage_receipt { - writer.commit(&answer)?; - } Ok(0) } diff --git a/src/review.rs b/src/review.rs index 9cc45fd..480a603 100644 --- a/src/review.rs +++ b/src/review.rs @@ -7,7 +7,7 @@ use anyhow::{Context, Result, anyhow}; use crate::config::{Config, GateLevel, OnError}; use crate::diff::{self, DiffIndex}; -use crate::envelope::{Envelope, Finding, Gate, Kind, Usage, fail_closed_finding}; +use crate::envelope::{Envelope, Finding, Gate, Kind, ModelUsage, Usage, fail_closed_finding}; use crate::filter; use crate::forge::{ CheckState, Forge, PrMeta, azure::Azure, bitbucket::Bitbucket, github::GitHub, gitlab::GitLab, @@ -608,6 +608,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> let mut summary = String::new(); let mut model_used = "none (empty diff)".to_string(); let mut usage = Usage::default(); + let mut model_usage: Vec = Vec::new(); let mut suppressed = 0u32; let mut ungrounded = 0u32; let mut findings: Vec = Vec::new(); @@ -656,6 +657,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> let outcome = filter::apply(cfg, &index, model_review.findings)?; model_used = model_review.model_used; usage = model_review.usage; + model_usage = model_review.model_usage; suppressed = outcome.suppressed; ungrounded = outcome.ungrounded; if outcome.all_ungrounded { @@ -700,6 +702,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> scorer_model = Some(scored.model_used); usage.prompt_tokens += scored.usage.prompt_tokens; usage.completion_tokens += scored.usage.completion_tokens; + model_usage.extend(scored.model_usage); scorer_disagreements = Some(disagreements); sort_findings_for_display(&mut kept); } @@ -711,6 +714,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> let scorer_usage = e.usage(); usage.prompt_tokens += scorer_usage.prompt_tokens; usage.completion_tokens += scorer_usage.completion_tokens; + model_usage.extend_from_slice(e.model_usage()); scorer_error = Some(detail); } Err(_) => { @@ -727,6 +731,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> Err(e) => { model_used = cfg.model_chain().join(" -> "); usage = e.usage(); + model_usage = e.model_usage().to_vec(); let detail = format!("{e:#}"); // Provider-class failures (outage, timeout) are the only ones // `gate.onError: advisory` may stand aside for; unusable model @@ -835,6 +840,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> scorer_error, scorer_disagreements, usage, + model_usage, duration_ms: review_started.elapsed().as_millis() as u64, base_sha: meta.map(|m| m.base_sha.clone()), head_sha, @@ -1014,6 +1020,7 @@ fn error_envelope( scorer_error: None, scorer_disagreements: None, usage: Usage::default(), + model_usage: vec![], duration_ms, base_sha: Some(meta.base_sha.clone()), head_sha: Some(head_sha.to_string()), diff --git a/src/sarif.rs b/src/sarif.rs index 4437b42..596748b 100644 --- a/src/sarif.rs +++ b/src/sarif.rs @@ -135,6 +135,7 @@ mod tests { scorer_error: None, scorer_disagreements: None, usage: Usage::default(), + model_usage: vec![], duration_ms: 0, base_sha: None, head_sha: None, diff --git a/tests/e2e.rs b/tests/e2e.rs index e61d297..fea9504 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -561,6 +561,15 @@ async fn local_review_reports_grounded_finding_and_gates() { assert_eq!(env["gate"]["failing"], true); assert_eq!(env["counts"]["error"], 1); assert_eq!(env["usage"]["promptTokens"], 300); + let model_usage = env["modelUsage"].as_array().unwrap(); + assert!(!model_usage.is_empty()); + assert_eq!( + model_usage + .iter() + .map(|entry| entry["promptTokens"].as_u64().unwrap()) + .sum::(), + env["usage"]["promptTokens"].as_u64().unwrap() + ); let requests = server.received_requests().await.unwrap(); let request: Value = requests[0].body_json().unwrap(); @@ -738,6 +747,15 @@ async fn slow_scorer_request_times_out_and_falls_back() { let stderr = String::from_utf8(out.get_output().stderr.clone()).unwrap(); let env: Value = serde_json::from_str(&stdout).unwrap(); assert_eq!(env["scorerModel"], "openai/gpt-5-mini"); + let models: Vec<_> = env["modelUsage"] + .as_array() + .unwrap() + .iter() + .map(|entry| entry["model"].as_str().unwrap()) + .collect(); + assert!(models.contains(&"generator-model")); + assert!(models.contains(&"openai/gpt-5-mini")); + assert!(!models.contains(&"anthropic/claude-haiku-4.5")); assert!(stderr.contains("postil: scorer anthropic/claude-haiku-4.5 timed out after")); assert!(stderr.contains("falling back to next scorer")); assert!(stderr.contains("postil: running scorer with openai/gpt-5-mini")); @@ -924,6 +942,18 @@ async fn scorer_error_fails_open_and_preserves_generator_values() { .contains("scorer output invalid") ); assert!(env.get("scorerDisagreements").is_none()); + assert_eq!(env["modelUsage"][0]["model"], "generator-model"); + assert_eq!(env["modelUsage"][1]["model"], "anthropic/claude-haiku-4.5"); + assert_eq!(env["modelUsage"][2]["model"], "openai/gpt-5-mini"); + assert_eq!( + env["modelUsage"] + .as_array() + .unwrap() + .iter() + .map(|entry| entry["promptTokens"].as_u64().unwrap()) + .sum::(), + env["usage"]["promptTokens"].as_u64().unwrap() + ); assert_eq!(finding["confidence"], 0.92); assert_eq!(finding["kind"], "risk"); assert!(finding.get("generatorConfidence").is_none()); @@ -3567,6 +3597,11 @@ async fn respond_writes_private_usage_receipt_across_model_fallback() { let dir = tempfile::tempdir().unwrap(); let receipt_path = dir.path().join("respond-usage.json"); + std::fs::write( + &receipt_path, + b"stale receipt from an interrupted attempt\n", + ) + .unwrap(); let out = postil() .current_dir(dir.path()) .env("POSTIL_API_BASE", server.uri()) From b95717ed0b5422e32e8e65f41cb2a5ac278d7212 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 01:22:00 +0000 Subject: [PATCH 06/11] Mark incomplete provider usage accounting --- README.md | 7 ++- src/envelope.rs | 5 +++ src/forge/github.rs | 2 + src/forge/mod.rs | 4 ++ src/llm.rs | 106 +++++++++++++++++++++++++++++++++++--------- src/output.rs | 1 + src/plan.rs | 1 + src/respond.rs | 2 + src/review.rs | 8 ++++ src/sarif.rs | 1 + tests/e2e.rs | 61 +++++++++++++++++++++++++ 11 files changed, 175 insertions(+), 23 deletions(-) diff --git a/README.md b/README.md index e8806bf..2e88e13 100644 --- a/README.md +++ b/README.md @@ -178,7 +178,9 @@ The version 1 receipt contains `operation: "respond"`, aggregate `promptTokens`/`completionTokens`, and a `models` array with token usage for each model that returned cost-relevant usage during cascade attempts. The receipt is synced before stdout or forge delivery, so a hosted worker can persist accounting -before it posts an answer. The receipt is +before it posts an answer. `usageAccountingComplete` is false when a sent request +can have unknown provider-billed usage, such as a timeout or ambiguous transport +failure. The receipt is never written to stdout, stderr, or command arguments. The caller owns deletion. ## Models and local inference @@ -269,7 +271,8 @@ aggregate `usage`, per-model `modelUsage`, SHAs) consumed by the hosted platform and `postil plan`. `modelUsage` includes the successful generator, scorers, and token-bearing failed fallbacks; its totals equal aggregate `usage`. Older v1 envelopes omit this additive field. Failed attempts that report zero tokens are -omitted because they carry no billable usage. `--output yaml` and +omitted because they carry no billable usage. The envelope's +`usageAccountingComplete` has the same conservative semantics. `--output yaml` and `--output csv` print the same review result in YAML or CSV. `--output-file ` writes the selected format to a file instead of stdout. `--output-json` is deprecated in v0.2.1 as an alias for `--output json` and emits a stderr warning. Schema: diff --git a/src/envelope.rs b/src/envelope.rs index 78fce76..4628288 100644 --- a/src/envelope.rs +++ b/src/envelope.rs @@ -182,6 +182,10 @@ pub struct Envelope { /// this additive field and are handled conservatively by the control plane. #[serde(skip_serializing_if = "Vec::is_empty", default)] pub model_usage: Vec, + /// False when any sent provider request can have unknown billed usage, + /// including timeouts and ambiguous transport failures. + #[serde(default)] + pub usage_accounting_complete: bool, /// Wall-clock duration of the review engine run in milliseconds. #[serde(default)] pub duration_ms: u64, @@ -442,6 +446,7 @@ mod tests { scorer_disagreements: None, usage: Usage::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/forge/github.rs b/src/forge/github.rs index 0ebbb79..56afc00 100644 --- a/src/forge/github.rs +++ b/src/forge/github.rs @@ -492,6 +492,7 @@ mod tests { scorer_disagreements: None, usage: Usage::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, @@ -524,6 +525,7 @@ mod tests { scorer_disagreements: None, usage: Usage::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/forge/mod.rs b/src/forge/mod.rs index 21ec086..3280f5e 100644 --- a/src/forge/mod.rs +++ b/src/forge/mod.rs @@ -443,6 +443,7 @@ mod tests { completion_tokens: 5, }, model_usage: vec![], + usage_accounting_complete: true, duration_ms: 1_250, base_sha: None, head_sha: Some("abcdef123456".into()), @@ -491,6 +492,7 @@ mod tests { scorer_disagreements: None, usage: Default::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, @@ -552,6 +554,7 @@ mod tests { scorer_disagreements: None, usage: Default::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, @@ -582,6 +585,7 @@ mod tests { scorer_disagreements: None, usage: Default::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/llm.rs b/src/llm.rs index df2de29..81d8b35 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -24,6 +24,7 @@ pub struct ModelReview { pub model_used: String, pub usage: Usage, pub model_usage: Vec, + pub usage_accounting_complete: bool, } #[derive(Debug, Clone)] @@ -40,6 +41,7 @@ pub struct ScorerReview { pub model_used: String, pub usage: Usage, pub model_usage: Vec, + pub usage_accounting_complete: bool, } #[derive(Debug, Clone)] @@ -48,6 +50,7 @@ pub struct Answer { pub model_used: String, pub usage: Usage, pub models: Vec, + pub usage_accounting_complete: bool, } #[derive(Debug)] @@ -55,14 +58,16 @@ pub struct ModelError { error: anyhow::Error, usage: Usage, model_usage: Vec, + usage_accounting_complete: bool, } impl ModelError { - fn new(error: anyhow::Error, usage: Usage) -> Self { + fn new(error: anyhow::Error, usage: Usage, usage_accounting_complete: bool) -> Self { Self { error, usage, model_usage: Vec::new(), + usage_accounting_complete, } } @@ -74,6 +79,10 @@ impl ModelError { &self.model_usage } + pub fn usage_accounting_complete(&self) -> bool { + self.usage_accounting_complete + } + pub fn is_provider(&self) -> bool { self.error.downcast_ref::().is_some() } @@ -411,12 +420,14 @@ impl LlmClient { let mut ok: Vec = Vec::new(); let mut failed_usage = Usage::default(); let mut failed_model_usage = Vec::new(); + let mut usage_accounting_complete = true; let mut last_err: Option = None; for (model, handle) in handles { let model_log = log_text(&model); match handle.await { Ok(Ok(r)) => ok.push(r), Ok(Err(mut e)) => { + usage_accounting_complete &= e.usage_accounting_complete; if e.model_usage.is_empty() && (e.usage.prompt_tokens > 0 || e.usage.completion_tokens > 0) { @@ -432,6 +443,7 @@ impl LlmClient { last_err = Some(e); } Err(e) => { + usage_accounting_complete = false; eprintln!("postil: consensus model {model_log} task panicked: {e}") } } @@ -444,30 +456,36 @@ impl LlmClient { let mut error = ModelError::new( e.error.context(format!("all {n} consensus models failed")), failed_usage, + usage_accounting_complete, ); error.model_usage = failed_model_usage; error } - None => { - ModelError::new(anyhow!("all {n} consensus models failed"), failed_usage) - } + None => ModelError::new( + anyhow!("all {n} consensus models failed"), + failed_usage, + usage_accounting_complete, + ), }), 1 => { let mut review = ok.into_iter().next().unwrap(); add_usage(&mut review.usage, failed_usage); review.model_usage.extend(failed_model_usage); + review.usage_accounting_complete &= usage_accounting_complete; Ok(review) } _ => { let mut review = consensus_merge(ok); add_usage(&mut review.usage, failed_usage); review.model_usage.extend(failed_model_usage); + review.usage_accounting_complete &= usage_accounting_complete; Ok(review) } } } else { let mut failed_usage = Usage::default(); let mut failed_model_usage = Vec::new(); + let mut usage_accounting_complete = true; let mut last_err = None; for (index, model) in chain.iter().enumerate() { let model_log = log_text(model); @@ -481,9 +499,11 @@ impl LlmClient { ); add_usage(&mut r.usage, failed_usage); r.model_usage.splice(0..0, failed_model_usage); + r.usage_accounting_complete &= usage_accounting_complete; return Ok(r); } Err(mut e) => { + usage_accounting_complete &= e.usage_accounting_complete; if e.model_usage.is_empty() && (e.usage.prompt_tokens > 0 || e.usage.completion_tokens > 0) { @@ -535,9 +555,12 @@ impl LlmClient { Err(last_err .map(|mut error| { error.model_usage = failed_model_usage; + error.usage_accounting_complete = usage_accounting_complete; error }) - .unwrap_or_else(|| ModelError::new(anyhow!("empty model chain"), failed_usage))) + .unwrap_or_else(|| { + ModelError::new(anyhow!("empty model chain"), failed_usage, true) + })) } } @@ -547,6 +570,7 @@ impl LlmClient { let mut usage = Usage::default(); let mut models = Vec::new(); let mut last_err = None; + let mut usage_accounting_complete = true; for model in cfg.model_chain() { let mut model_usage = Usage::default(); match self @@ -572,9 +596,16 @@ impl LlmClient { model_used: model, usage, models, + usage_accounting_complete, }); } Err(e) => { + // Usage parsed from a provider response is complete even + // when the response has no usable answer. A transport + // failure with no response usage is ambiguous. + if model_usage.prompt_tokens == 0 && model_usage.completion_tokens == 0 { + usage_accounting_complete = false; + } eprintln!("postil: model {model} failed: {e:#}"); // Provider failures that report no tokens have no billable // usage to attribute. Omit them rather than emitting a @@ -607,6 +638,7 @@ impl LlmClient { ) -> std::result::Result { let mut failed_usage = Usage::default(); let mut failed_model_usage = Vec::new(); + let mut usage_accounting_complete = true; let mut last_err = None; let chain = cfg.scorer_chain(); for (index, model) in chain.iter().enumerate() { @@ -624,9 +656,11 @@ impl LlmClient { ); add_usage(&mut r.usage, failed_usage); r.model_usage.splice(0..0, failed_model_usage); + r.usage_accounting_complete &= usage_accounting_complete; return Ok(r); } Err(mut e) => { + usage_accounting_complete &= e.usage_accounting_complete; if e.model_usage.is_empty() && (e.usage.prompt_tokens > 0 || e.usage.completion_tokens > 0) { @@ -676,9 +710,12 @@ impl LlmClient { Err(last_err .map(|mut error| { error.model_usage = failed_model_usage; + error.usage_accounting_complete = usage_accounting_complete; error }) - .unwrap_or_else(|| ModelError::new(anyhow!("empty scorer model chain"), failed_usage))) + .unwrap_or_else(|| { + ModelError::new(anyhow!("empty scorer model chain"), failed_usage, true) + })) } async fn review_with_model( @@ -698,7 +735,10 @@ impl LlmClient { LlmPhase::Review, ) .await - .map_err(|e| ModelError::new(e, usage))?; + .map_err(|e| { + let complete = usage.prompt_tokens > 0 || usage.completion_tokens > 0; + ModelError::new(e, usage, complete) + })?; let raw = match parse_review(&content) { Ok(raw) => raw, Err(parse_err) => { @@ -718,9 +758,15 @@ impl LlmClient { LlmPhase::Review, ) .await - .map_err(|e| ModelError::new(e.context("JSON repair call failed"), usage))?; + .map_err(|e| { + ModelError::new(e.context("JSON repair call failed"), usage, false) + })?; parse_review(&repaired).map_err(|e| { - ModelError::new(anyhow!("model output invalid after repair: {e}"), usage) + ModelError::new( + anyhow!("model output invalid after repair: {e}"), + usage, + true, + ) })? } }; @@ -740,7 +786,7 @@ impl LlmClient { return exactly {{\"summary\": \"\", \"findings\": []}}." ); let mut retry_usage = usage; - if let Ok(retried) = self + match self .chat( model, system, @@ -751,17 +797,27 @@ impl LlmClient { ) .await { - review.usage = retry_usage; - if let Ok(retried_raw) = parse_review(&retried) { - let candidate = into_review(retried_raw, model, retry_usage); - let still_contradictory = - candidate.findings.is_empty() && !candidate.summary.is_empty(); - if !still_contradictory { - review = candidate; + Ok(retried) => { + review.usage = retry_usage; + if let Ok(retried_raw) = parse_review(&retried) { + let candidate = into_review(retried_raw, model, retry_usage); + let still_contradictory = + candidate.findings.is_empty() && !candidate.summary.is_empty(); + if !still_contradictory { + review = candidate; + } + } + } + Err(_) => { + review.usage_accounting_complete = false; + if let Err(error) = self.remaining_budget(LlmPhase::Review) { + return Err(ModelError::new( + error.context(ProviderError), + retry_usage, + false, + )); } } - } else if let Err(error) = self.remaining_budget(LlmPhase::Review) { - return Err(ModelError::new(error.context(ProviderError), retry_usage)); } } Ok(review) @@ -786,9 +842,12 @@ impl LlmClient { LlmPhase::Total, ) .await - .map_err(|e| ModelError::new(e, usage))?; + .map_err(|e| { + let complete = usage.prompt_tokens > 0 || usage.completion_tokens > 0; + ModelError::new(e, usage, complete) + })?; let scores = parse_scores(&content, expected_len) - .map_err(|e| ModelError::new(anyhow!("scorer output invalid: {e}"), usage))?; + .map_err(|e| ModelError::new(anyhow!("scorer output invalid: {e}"), usage, true))?; Ok(ScorerReview { scores, model_used: model.to_string(), @@ -798,6 +857,7 @@ impl LlmClient { prompt_tokens: usage.prompt_tokens, completion_tokens: usage.completion_tokens, }], + usage_accounting_complete: true, }) } @@ -1504,6 +1564,7 @@ fn into_review(raw: RawReview, model: &str, usage: Usage) -> ModelReview { prompt_tokens: usage.prompt_tokens, completion_tokens: usage.completion_tokens, }], + usage_accounting_complete: true, } } @@ -1519,6 +1580,7 @@ fn consensus_merge(runs: Vec) -> ModelReview { }; let models: Vec = runs.iter().map(|r| r.model_used.clone()).collect(); let model_usage = runs.iter().flat_map(|r| r.model_usage.clone()).collect(); + let usage_accounting_complete = runs.iter().all(|r| r.usage_accounting_complete); let summary = runs[0].summary.clone(); // Flatten in run order; greedy clustering then anchors each cluster on its // earliest report. @@ -1561,6 +1623,7 @@ fn consensus_merge(runs: Vec) -> ModelReview { model_used: format!("consensus({})", models.join(", ")), usage: total_usage, model_usage, + usage_accounting_complete, } } @@ -2005,6 +2068,7 @@ mod tests { completion_tokens: 5, }, model_usage: vec![], + usage_accounting_complete: true, } } diff --git a/src/output.rs b/src/output.rs index 87a9232..f87262c 100644 --- a/src/output.rs +++ b/src/output.rs @@ -275,6 +275,7 @@ mod tests { scorer_disagreements: None, usage: Default::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/plan.rs b/src/plan.rs index 9fc9bf3..b2ab306 100644 --- a/src/plan.rs +++ b/src/plan.rs @@ -174,6 +174,7 @@ mod tests { scorer_disagreements: None, usage: Usage::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, diff --git a/src/respond.rs b/src/respond.rs index 0ac197a..875c8dc 100644 --- a/src/respond.rs +++ b/src/respond.rs @@ -51,6 +51,7 @@ struct RespondUsageReceipt<'a> { prompt_tokens: u64, completion_tokens: u64, models: Vec>, + usage_accounting_complete: bool, } #[derive(Serialize)] @@ -118,6 +119,7 @@ impl UsageReceiptWriter { completion_tokens: model.completion_tokens, }) .collect(), + usage_accounting_complete: answer.usage_accounting_complete, }; let file = self.file.as_mut().expect("usage receipt file is present"); serde_json::to_writer(&mut *file, &receipt).context("serializing usage receipt")?; diff --git a/src/review.rs b/src/review.rs index 480a603..14e759f 100644 --- a/src/review.rs +++ b/src/review.rs @@ -609,6 +609,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> let mut model_used = "none (empty diff)".to_string(); let mut usage = Usage::default(); let mut model_usage: Vec = Vec::new(); + let mut usage_accounting_complete = true; let mut suppressed = 0u32; let mut ungrounded = 0u32; let mut findings: Vec = Vec::new(); @@ -658,6 +659,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> model_used = model_review.model_used; usage = model_review.usage; model_usage = model_review.model_usage; + usage_accounting_complete = model_review.usage_accounting_complete; suppressed = outcome.suppressed; ungrounded = outcome.ungrounded; if outcome.all_ungrounded { @@ -703,6 +705,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> usage.prompt_tokens += scored.usage.prompt_tokens; usage.completion_tokens += scored.usage.completion_tokens; model_usage.extend(scored.model_usage); + usage_accounting_complete &= scored.usage_accounting_complete; scorer_disagreements = Some(disagreements); sort_findings_for_display(&mut kept); } @@ -715,9 +718,11 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> usage.prompt_tokens += scorer_usage.prompt_tokens; usage.completion_tokens += scorer_usage.completion_tokens; model_usage.extend_from_slice(e.model_usage()); + usage_accounting_complete &= e.usage_accounting_complete(); scorer_error = Some(detail); } Err(_) => { + usage_accounting_complete = false; let detail = format!("scorer timed out after {SCORER_TIMEOUT_SECS}s"); eprintln!("postil: scorer failed open: {detail}"); @@ -732,6 +737,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> model_used = cfg.model_chain().join(" -> "); usage = e.usage(); model_usage = e.model_usage().to_vec(); + usage_accounting_complete = e.usage_accounting_complete(); let detail = format!("{e:#}"); // Provider-class failures (outage, timeout) are the only ones // `gate.onError: advisory` may stand aside for; unusable model @@ -841,6 +847,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> scorer_disagreements, usage, model_usage, + usage_accounting_complete, duration_ms: review_started.elapsed().as_millis() as u64, base_sha: meta.map(|m| m.base_sha.clone()), head_sha, @@ -1021,6 +1028,7 @@ fn error_envelope( scorer_disagreements: None, usage: Usage::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms, base_sha: Some(meta.base_sha.clone()), head_sha: Some(head_sha.to_string()), diff --git a/src/sarif.rs b/src/sarif.rs index 596748b..5ec34c4 100644 --- a/src/sarif.rs +++ b/src/sarif.rs @@ -136,6 +136,7 @@ mod tests { scorer_disagreements: None, usage: Usage::default(), model_usage: vec![], + usage_accounting_complete: true, duration_ms: 0, base_sha: None, head_sha: None, diff --git a/tests/e2e.rs b/tests/e2e.rs index fea9504..ed4d121 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -561,6 +561,7 @@ async fn local_review_reports_grounded_finding_and_gates() { assert_eq!(env["gate"]["failing"], true); assert_eq!(env["counts"]["error"], 1); assert_eq!(env["usage"]["promptTokens"], 300); + assert_eq!(env["usageAccountingComplete"], true); let model_usage = env["modelUsage"].as_array().unwrap(); assert!(!model_usage.is_empty()); assert_eq!( @@ -747,6 +748,7 @@ async fn slow_scorer_request_times_out_and_falls_back() { let stderr = String::from_utf8(out.get_output().stderr.clone()).unwrap(); let env: Value = serde_json::from_str(&stdout).unwrap(); assert_eq!(env["scorerModel"], "openai/gpt-5-mini"); + assert_eq!(env["usageAccountingComplete"], false); let models: Vec<_> = env["modelUsage"] .as_array() .unwrap() @@ -755,6 +757,9 @@ async fn slow_scorer_request_times_out_and_falls_back() { .collect(); assert!(models.contains(&"generator-model")); assert!(models.contains(&"openai/gpt-5-mini")); + // The timed-out request has no validated token count. It is not invented + // as a per-model entry; the explicit incomplete flag makes hosted billing + // consume the conservative reservation instead. assert!(!models.contains(&"anthropic/claude-haiku-4.5")); assert!(stderr.contains("postil: scorer anthropic/claude-haiku-4.5 timed out after")); assert!(stderr.contains("falling back to next scorer")); @@ -3629,6 +3634,7 @@ async fn respond_writes_private_usage_receipt_across_model_fallback() { let receipt: Value = serde_json::from_slice(&std::fs::read(&receipt_path).unwrap()).unwrap(); assert_eq!(receipt["version"], 1); assert_eq!(receipt["operation"], "respond"); + assert_eq!(receipt["usageAccountingComplete"], true); assert_eq!(receipt["promptTokens"], 30); assert_eq!(receipt["completionTokens"], 5); assert_eq!(receipt["models"][0]["model"], "primary-model"); @@ -3643,6 +3649,61 @@ async fn respond_writes_private_usage_receipt_across_model_fallback() { ); } +#[tokio::test] +async fn respond_marks_receipt_incomplete_after_ambiguous_fallback() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("primary-model")) + .respond_with(ResponseTemplate::new(503)) + .mount(&server) + .await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("backup-model")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "choices": [{"message": {"content": "Retry with a bounded backoff."}}], + "usage": {"prompt_tokens": 20, "completion_tokens": 3} + }))) + .mount(&server) + .await; + Mock::given(method("GET")) + .and(path("/repos/acme/api/issues/10")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "title": "Retries", "body": "Requests fail intermittently." + }))) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let receipt_path = dir.path().join("respond-usage.json"); + postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("GITHUB_API_URL", server.uri()) + .env("GITHUB_TOKEN", "gh-test-token") + .env("REVIEW_MODEL", "primary-model") + .env("REVIEW_MODEL_CASCADE", "backup-model") + .env("POSTIL_USAGE_RECEIPT_PATH", &receipt_path) + .args([ + "respond", + "--repo", + "acme/api", + "--issue", + "10", + "--comment", + "@postil how should this retry?", + "--no-post", + ]) + .assert() + .success(); + + let receipt: Value = serde_json::from_slice(&std::fs::read(receipt_path).unwrap()).unwrap(); + assert_eq!(receipt["usageAccountingComplete"], false); + assert_eq!(receipt["models"].as_array().unwrap().len(), 1); + assert_eq!(receipt["models"][0]["model"], "backup-model"); +} + #[tokio::test] async fn respond_to_issue_mention_uses_issue_body() { let server = MockServer::start().await; From 20095ebf2b9d48484344543ffecde6e1511d4edd Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 01:41:53 +0000 Subject: [PATCH 07/11] Propagate retry accounting ambiguity --- src/llm.rs | 79 ++++++++++++++++++++++++++++++++++++++++++---------- tests/e2e.rs | 2 ++ 2 files changed, 66 insertions(+), 15 deletions(-) diff --git a/src/llm.rs b/src/llm.rs index 81d8b35..1dad67d 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -573,12 +573,14 @@ impl LlmClient { let mut usage_accounting_complete = true; for model in cfg.model_chain() { let mut model_usage = Usage::default(); + let mut model_accounting_complete = true; match self .chat( &model, system, user, &mut model_usage, + &mut model_accounting_complete, Some(RESPOND_MAX_TOKENS), LlmPhase::Total, ) @@ -603,7 +605,9 @@ impl LlmClient { // Usage parsed from a provider response is complete even // when the response has no usable answer. A transport // failure with no response usage is ambiguous. - if model_usage.prompt_tokens == 0 && model_usage.completion_tokens == 0 { + if !model_accounting_complete + || (model_usage.prompt_tokens == 0 && model_usage.completion_tokens == 0) + { usage_accounting_complete = false; } eprintln!("postil: model {model} failed: {e:#}"); @@ -725,18 +729,21 @@ impl LlmClient { user: &str, ) -> std::result::Result { let mut usage = Usage::default(); + let mut usage_accounting_complete = true; let content = self .chat( model, system, user, &mut usage, + &mut usage_accounting_complete, Some(REVIEW_MAX_TOKENS), LlmPhase::Review, ) .await .map_err(|e| { - let complete = usage.prompt_tokens > 0 || usage.completion_tokens > 0; + let complete = usage_accounting_complete + && (usage.prompt_tokens > 0 || usage.completion_tokens > 0); ModelError::new(e, usage, complete) })?; let raw = match parse_review(&content) { @@ -754,6 +761,7 @@ impl LlmClient { "You repair malformed JSON. Output only valid JSON.", &repair_user, &mut usage, + &mut usage_accounting_complete, Some(REVIEW_MAX_TOKENS), LlmPhase::Review, ) @@ -765,12 +773,13 @@ impl LlmClient { ModelError::new( anyhow!("model output invalid after repair: {e}"), usage, - true, + usage_accounting_complete, ) })? } }; let mut review = into_review(raw, model, usage); + review.usage_accounting_complete = usage_accounting_complete; // Semantic consistency retry: a summary that narrates risk next to an // empty findings array is the contract violation behind "clean status, @@ -792,6 +801,7 @@ impl LlmClient { system, &retry_user, &mut retry_usage, + &mut usage_accounting_complete, Some(REVIEW_MAX_TOKENS), LlmPhase::Review, ) @@ -799,8 +809,10 @@ impl LlmClient { { Ok(retried) => { review.usage = retry_usage; + review.usage_accounting_complete = usage_accounting_complete; if let Ok(retried_raw) = parse_review(&retried) { - let candidate = into_review(retried_raw, model, retry_usage); + let mut candidate = into_review(retried_raw, model, retry_usage); + candidate.usage_accounting_complete = usage_accounting_complete; let still_contradictory = candidate.findings.is_empty() && !candidate.summary.is_empty(); if !still_contradictory { @@ -831,23 +843,31 @@ impl LlmClient { expected_len: usize, ) -> std::result::Result { let mut usage = Usage::default(); + let mut usage_accounting_complete = true; let content = self .chat_with_temperature( model, system, user, &mut usage, + &mut usage_accounting_complete, Some(SCORER_MAX_TOKENS), 0.0, LlmPhase::Total, ) .await .map_err(|e| { - let complete = usage.prompt_tokens > 0 || usage.completion_tokens > 0; + let complete = usage_accounting_complete + && (usage.prompt_tokens > 0 || usage.completion_tokens > 0); ModelError::new(e, usage, complete) })?; - let scores = parse_scores(&content, expected_len) - .map_err(|e| ModelError::new(anyhow!("scorer output invalid: {e}"), usage, true))?; + let scores = parse_scores(&content, expected_len).map_err(|e| { + ModelError::new( + anyhow!("scorer output invalid: {e}"), + usage, + usage_accounting_complete, + ) + })?; Ok(ScorerReview { scores, model_used: model.to_string(), @@ -857,22 +877,33 @@ impl LlmClient { prompt_tokens: usage.prompt_tokens, completion_tokens: usage.completion_tokens, }], - usage_accounting_complete: true, + usage_accounting_complete, }) } + #[allow(clippy::too_many_arguments)] async fn chat( &self, model: &str, system: &str, user: &str, usage: &mut Usage, + usage_accounting_complete: &mut bool, max_tokens: Option, phase: LlmPhase, ) -> Result { - self.chat_with_temperature(model, system, user, usage, max_tokens, 0.1, phase) - .await - .map_err(|e| e.context(ProviderError)) + self.chat_with_temperature( + model, + system, + user, + usage, + usage_accounting_complete, + max_tokens, + 0.1, + phase, + ) + .await + .map_err(|e| e.context(ProviderError)) } #[allow(clippy::too_many_arguments)] @@ -882,13 +913,23 @@ impl LlmClient { system: &str, user: &str, usage: &mut Usage, + usage_accounting_complete: &mut bool, max_tokens: Option, temperature: f64, phase: LlmPhase, ) -> Result { - self.chat_inner(model, system, user, usage, max_tokens, temperature, phase) - .await - .map_err(|e| e.context(ProviderError)) + self.chat_inner( + model, + system, + user, + usage, + usage_accounting_complete, + max_tokens, + temperature, + phase, + ) + .await + .map_err(|e| e.context(ProviderError)) } /// Transport + HTTP envelope handling; every error here is provider-class. @@ -899,6 +940,7 @@ impl LlmClient { system: &str, user: &str, usage: &mut Usage, + usage_accounting_complete: &mut bool, max_tokens: Option, temperature: f64, phase: LlmPhase, @@ -914,8 +956,12 @@ impl LlmClient { let timeout = remaining.map_or(attempt_timeout, |value| value.min(attempt_timeout)); let response = match tokio::time::timeout(timeout, self.request_once(&body)).await { Ok(result) => result, - Err(_) if deadline_limited => return Err(DeadlineExceeded(phase).into()), + Err(_) if deadline_limited => { + *usage_accounting_complete = false; + return Err(DeadlineExceeded(phase).into()); + } Err(_) => { + *usage_accounting_complete = false; if timeout_retries < TIMEOUT_RETRIES && retries < TRANSIENT_RETRIES { retries += 1; timeout_retries += 1; @@ -944,6 +990,7 @@ impl LlmClient { && timeout_retries < TIMEOUT_RETRIES && retries < TRANSIENT_RETRIES { + *usage_accounting_complete = false; retries += 1; timeout_retries += 1; let wait = Duration::from_secs(2 * retries as u64); @@ -983,6 +1030,7 @@ impl LlmClient { && timeout_retries < TIMEOUT_RETRIES && retries < TRANSIENT_RETRIES => { + *usage_accounting_complete = false; retries += 1; timeout_retries += 1; let wait = Duration::from_secs(2 * retries as u64); @@ -1000,6 +1048,7 @@ impl LlmClient { if reqwest_error(&error).is_some_and(reqwest::Error::is_connect) && retries < TRANSIENT_RETRIES => { + *usage_accounting_complete = false; retries += 1; let wait = Duration::from_secs(2 * retries as u64); eprintln!( diff --git a/tests/e2e.rs b/tests/e2e.rs index ed4d121..b697f8b 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -1920,6 +1920,7 @@ async fn slow_model_request_retries_same_model_then_succeeds() { let env: Value = serde_json::from_str(&String::from_utf8(out.get_output().stdout.clone()).unwrap()).unwrap(); assert_eq!(env["modelUsed"], "primary-model"); + assert_eq!(env["usageAccountingComplete"], false); let stderr = String::from_utf8(out.get_output().stderr.clone()).unwrap(); assert!(stderr.contains("postil: model primary-model hit a request timeout after")); assert!(stderr.contains("timeout retry 1/1")); @@ -1973,6 +1974,7 @@ async fn timeout_http_status_retries_same_model_then_succeeds() { let envelope: Value = serde_json::from_slice(&out.get_output().stdout).expect("review output should be JSON"); assert_eq!(envelope["modelUsed"], "primary-model"); + assert_eq!(envelope["usageAccountingComplete"], false); let stderr = String::from_utf8(out.get_output().stderr.clone()).unwrap(); assert!(stderr.contains("returned timeout HTTP 408 Request Timeout")); assert!(stderr.contains("timeout retry 1/1")); From ef3ae6ba834d013530bdc3806bfa5d4161f42520 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 01:46:19 +0000 Subject: [PATCH 08/11] Preserve retry ambiguity in answers --- src/llm.rs | 1 + tests/e2e.rs | 66 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 67 insertions(+) diff --git a/src/llm.rs b/src/llm.rs index 1dad67d..4f50e99 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -587,6 +587,7 @@ impl LlmClient { .await { Ok(content) => { + usage_accounting_complete &= model_accounting_complete; add_usage(&mut usage, model_usage); models.push(ModelUsage { model: model.clone(), diff --git a/tests/e2e.rs b/tests/e2e.rs index b697f8b..0010078 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -3706,6 +3706,72 @@ async fn respond_marks_receipt_incomplete_after_ambiguous_fallback() { assert_eq!(receipt["models"][0]["model"], "backup-model"); } +#[tokio::test] +async fn respond_marks_receipt_incomplete_after_internal_retry_succeeds() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("primary-model")) + .respond_with(ResponseTemplate::new(408).set_body_string("request timed out")) + .up_to_n_times(1) + .mount(&server) + .await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("primary-model")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "choices": [{"message": {"content": "Use a bounded retry."}}], + "usage": {"prompt_tokens": 20, "completion_tokens": 3} + }))) + .mount(&server) + .await; + Mock::given(method("GET")) + .and(path("/repos/acme/api/issues/11")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "title": "Retries", "body": "Requests fail intermittently." + }))) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + let receipt_path = dir.path().join("respond-usage.json"); + postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("GITHUB_API_URL", server.uri()) + .env("GITHUB_TOKEN", "gh-test-token") + .env("REVIEW_MODEL", "primary-model") + .env("REVIEW_MODEL_CASCADE", "") + .env("POSTIL_USAGE_RECEIPT_PATH", &receipt_path) + .args([ + "respond", + "--repo", + "acme/api", + "--issue", + "11", + "--comment", + "@postil how should this retry?", + "--no-post", + ]) + .assert() + .success(); + + let receipt: Value = serde_json::from_slice(&std::fs::read(receipt_path).unwrap()).unwrap(); + assert_eq!(receipt["usageAccountingComplete"], false); + assert_eq!(receipt["models"].as_array().unwrap().len(), 1); + assert_eq!(receipt["models"][0]["model"], "primary-model"); + assert_eq!( + server + .received_requests() + .await + .unwrap() + .iter() + .filter(|request| { request.url.path() == "/chat/completions" }) + .count(), + 2 + ); +} + #[tokio::test] async fn respond_to_issue_mention_uses_issue_body() { let server = MockServer::start().await; From 70e8a95243d31fd3189715f92e749e1767a4a844 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 01:54:23 +0000 Subject: [PATCH 09/11] Document accounting durability invariants --- src/llm.rs | 3 +++ src/respond.rs | 8 ++++++++ 2 files changed, 11 insertions(+) diff --git a/src/llm.rs b/src/llm.rs index 4f50e99..98f7943 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -946,6 +946,9 @@ impl LlmClient { temperature: f64, phase: LlmPhase, ) -> Result { + // This mutable flag is stack-local state held through one exclusively + // borrowed async call. Request retries run sequentially in this loop, + // so updating it before continuing or returning needs no atomic type. let body = self.request_body(model, system, user, max_tokens, temperature); let mut retries = 0u32; let mut timeout_retries = 0u32; diff --git a/src/respond.rs b/src/respond.rs index 875c8dc..25fc2f4 100644 --- a/src/respond.rs +++ b/src/respond.rs @@ -62,6 +62,8 @@ struct RespondModelUsage<'a> { completion_tokens: u64, } +// PID separates concurrent processes; this sequence separates writers within +// one process. create_new below also fails closed if a path already exists. static RECEIPT_TEMP_SEQUENCE: AtomicU64 = AtomicU64::new(0); struct UsageReceiptWriter { @@ -94,6 +96,7 @@ impl UsageReceiptWriter { let file = OpenOptions::new() .write(true) .create_new(true) + // Usage receipts contain private provider-accounting metadata. .mode(0o600) .open(&temp_path) .context("creating private usage receipt temporary file")?; @@ -124,6 +127,8 @@ impl UsageReceiptWriter { let file = self.file.as_mut().expect("usage receipt file is present"); serde_json::to_writer(&mut *file, &receipt).context("serializing usage receipt")?; file.write_all(b"\n").context("writing usage receipt")?; + // Publish only after file contents are durable. The directory sync + // below makes the rename durable before stdout or forge delivery. file.sync_all().context("syncing usage receipt")?; drop(self.file.take()); std::fs::rename(&self.temp_path, &self.final_path) @@ -139,6 +144,9 @@ impl UsageReceiptWriter { impl Drop for UsageReceiptWriter { fn drop(&mut self) { + // Drop cannot return cleanup errors. Best-effort removal is safe: an + // unpublished temp file is never treated as a committed receipt, and + // create_new prevents a later writer from clobbering it. let _ = std::fs::remove_file(&self.temp_path); } } From bf4087657109ae60c051f8cb506ecfc4dab9e237 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 02:02:34 +0000 Subject: [PATCH 10/11] Reapply confidence policy after scoring --- src/review.rs | 29 +++++++++++++++++++++++++++++ tests/e2e.rs | 48 ++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 77 insertions(+) diff --git a/src/review.rs b/src/review.rs index 14e759f..0ea139b 100644 --- a/src/review.rs +++ b/src/review.rs @@ -560,6 +560,12 @@ fn apply_scorer_scores(cfg: &Config, findings: &mut [Finding], scores: Vec) -> u32 { + let before = findings.len(); + findings.retain(|finding| finding.confidence >= cfg.min_confidence); + (before - findings.len()) as u32 +} + fn sort_findings_for_display(findings: &mut [Finding]) { findings.sort_by(|a, b| { b.severity @@ -701,6 +707,7 @@ async fn review_diff(cfg: &Config, args: &ReviewArgs, input: ReviewInput<'_>) -> Ok(Ok(scored)) => { let disagreements = apply_scorer_scores(cfg, &mut kept, scored.scores); + suppressed += suppress_below_min_confidence(cfg, &mut kept); scorer_model = Some(scored.model_used); usage.prompt_tokens += scored.usage.prompt_tokens; usage.completion_tokens += scored.usage.completion_tokens; @@ -1218,6 +1225,28 @@ mod tests { assert_eq!(findings[0].scorer_kind, Some(Kind::Risk)); } + #[test] + fn scorer_confidence_below_policy_is_suppressed_after_calibration() { + let cfg = Config { + min_confidence: 0.6, + ..Config::default() + }; + let mut findings = vec![finding("low.rs", 10, "low"), finding("kept.rs", 20, "kept")]; + + apply_scorer_scores( + &cfg, + &mut findings, + vec![score(0, 0.1, Kind::Risk), score(1, 0.8, Kind::Risk)], + ); + let suppressed = suppress_below_min_confidence(&cfg, &mut findings); + + assert_eq!(suppressed, 1); + assert_eq!(findings.len(), 1); + assert_eq!(findings[0].path, "kept.rs"); + assert_eq!(findings[0].generator_confidence, Some(0.9)); + assert_eq!(findings[0].scorer_confidence, Some(0.8)); + } + #[test] fn scorer_kind_can_escalate_into_a_blocking_kind() { let cfg = Config::default(); diff --git a/tests/e2e.rs b/tests/e2e.rs index 0010078..ac2f93a 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -700,6 +700,54 @@ async fn scorer_lowers_confidence_and_stores_both_values() { assert!(scorer_user.contains("diffHunk")); } +#[tokio::test] +async fn scorer_confidence_below_minimum_is_suppressed_and_nonblocking() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("generator-model")) + .respond_with( + ResponseTemplate::new(200) + .set_body_json(llm_content(json!([finding_at(41, "error", 0.92)]))), + ) + .mount(&server) + .await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .and(body_string_contains("anthropic/claude-haiku-4.5")) + .respond_with( + ResponseTemplate::new(200).set_body_json(scorer_content(json!([{ + "index": 0, + "confidence": 0.1, + "kind": "risk", + "reason": "independent evidence does not support the generator claim" + }]))), + ) + .mount(&server) + .await; + + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join(".postil.yaml"), "minConfidence: 0.6\n").unwrap(); + let diff = write_diff(dir.path()); + let out = postil() + .current_dir(dir.path()) + .env("POSTIL_API_BASE", server.uri()) + .env("REVIEW_MODEL", "generator-model") + .args(["review", "--diff-file"]) + .arg(&diff) + .args(["--output", "json"]) + .assert() + .code(0); + + let envelope: Value = serde_json::from_slice(&out.get_output().stdout).unwrap(); + assert_eq!(envelope["findings"].as_array().unwrap().len(), 0); + assert_eq!(envelope["silent"], true); + assert_eq!(envelope["gate"]["failing"], false); + assert_eq!(envelope["counts"]["error"], 0); + assert_eq!(envelope["counts"]["suppressed"], 1); + assert_eq!(envelope["scorerModel"], "anthropic/claude-haiku-4.5"); +} + #[tokio::test] async fn slow_scorer_request_times_out_and_falls_back() { let server = MockServer::start().await; From 515be6e14f9c81f45c5165fb75acf97c9bbc5538 Mon Sep 17 00:00:00 2001 From: Postil Maintainer Date: Mon, 13 Jul 2026 02:05:37 +0000 Subject: [PATCH 11/11] Keep scorer disagreement coverage above threshold --- tests/e2e.rs | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/e2e.rs b/tests/e2e.rs index ac2f93a..e9bb439 100644 --- a/tests/e2e.rs +++ b/tests/e2e.rs @@ -913,7 +913,7 @@ async fn large_confidence_disagreement_escalates_to_uncertainty_with_default_gat mock_review_model( &server, "generator-model", - json!([finding_at(41, "warn", 0.95)]), + json!([finding_at(41, "warn", 1.0)]), ) .await; mock_scorer_model( @@ -921,7 +921,7 @@ async fn large_confidence_disagreement_escalates_to_uncertainty_with_default_gat "anthropic/claude-haiku-4.5", json!([{ "index": 0, - "confidence": 0.5, + "confidence": 0.6, "kind": "risk", "reason": "weak evidence" }]), @@ -942,7 +942,7 @@ async fn large_confidence_disagreement_escalates_to_uncertainty_with_default_gat let stdout = String::from_utf8(out.get_output().stdout.clone()).unwrap(); let env: Value = serde_json::from_str(&stdout).unwrap(); - assert_eq!(env["findings"][0]["confidence"], 0.5); + assert_eq!(env["findings"][0]["confidence"], 0.6); assert_eq!(env["findings"][0]["kind"], "uncertainty"); assert_eq!(env["findings"][0]["generatorKind"], "risk"); assert_eq!(env["findings"][0]["scorerKind"], "risk");