diff --git a/.env.example b/.env.example index 095b67b..9b5b127 100644 --- a/.env.example +++ b/.env.example @@ -1,4 +1,4 @@ # Copy to .env (never commit .env). Production uses Secret Manager. # GROQ_API_KEY=gsk_xxxx -# GROQ_MODEL=openai/gpt-oss-20b +# GROQ_MODEL=qwen/qwen3.8-27b # GROQ_BASE_URL=https://api.groq.com/openai/v1 diff --git a/README.md b/README.md index ad06515..55b8e66 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,7 @@ Deterministic systems first; LLM only where semantic reasoning is required. provider-neutral ModelClient). Python is spec artifacts + bounded future service per ADR-001. - Agent count: 1 Investigation Agent (bounded, provider-neutral). No APPROVE/DENY/PAY by AI, ever. - - LLM provider: Groq (`llama-3.1-8b-instant`, code default) contracted per ADR-002, wiring + - LLM provider: Groq (`qwen/qwen3.8-27b`, code default) contracted per ADR-002, wiring deferred behind ModelClient (stub/FakeModelClient until key); no Ollama. ## Prerequisites diff --git a/apps/api/cmd/agent/groq_provider_selection_test.go b/apps/api/cmd/agent/groq_provider_selection_test.go index 05f0a51..43b8c26 100644 --- a/apps/api/cmd/agent/groq_provider_selection_test.go +++ b/apps/api/cmd/agent/groq_provider_selection_test.go @@ -51,10 +51,11 @@ import ( const ( // canonicalGroqModel is the ADR-002 adjudicated canonical model string - // (Option A, 2026-09-26). It must equal the code default in + // (Option A, 2026-09-26; re-adjudicated to a servable production model + // in APA-47, 2026-09-28). It must equal the code default in // groq_model.go and the README line; groq_env_contract_test.go pins // that three-way agreement so the three cannot drift apart silently. - canonicalGroqModel = "llama-3.1-8b-instant" + canonicalGroqModel = "qwen/qwen3.8-27b" // sentinelGroqKey is a fake key material. It exists so the no-leak // assertions have something specific to hunt for. It is not a real diff --git a/apps/api/internal/investigate/orchestrate/groq_env_contract_test.go b/apps/api/internal/investigate/orchestrate/groq_env_contract_test.go index 0782c49..4636035 100644 --- a/apps/api/internal/investigate/orchestrate/groq_env_contract_test.go +++ b/apps/api/internal/investigate/orchestrate/groq_env_contract_test.go @@ -35,10 +35,11 @@ import ( const ( // canonicalModelString is the ADR-002 adjudicated canonical model - // string (Option A, 2026-09-26). It is duplicated here as a literal on + // string (Option A, 2026-09-26; re-adjudicated to a servable production + // model in APA-47, 2026-09-28). It is duplicated here as a literal on // purpose: if the code default ever changes, this test fails and forces // an explicit decision rather than letting code and docs drift. - canonicalModelString = "llama-3.1-8b-instant" + canonicalModelString = "qwen/qwen3.8-27b" // sentinelKey is fake key material used as a leak canary. It is not a // real credential and must never be committed as one. diff --git a/apps/api/internal/investigate/orchestrate/groq_model.go b/apps/api/internal/investigate/orchestrate/groq_model.go index f6d1502..9249700 100644 --- a/apps/api/internal/investigate/orchestrate/groq_model.go +++ b/apps/api/internal/investigate/orchestrate/groq_model.go @@ -15,8 +15,16 @@ import ( ) // Groq defaults. +// +// defaultGroqModel is the ADR-002 adjudicated canonical model, re-adjudicated +// to a currently servable production model in APA-47 (2026-09-28). The prior +// value, llama-3.1-8b-instant, was retired from Groq and answers HTTP 404 +// model_not_found, so it cannot serve production traffic. Re-verified live the +// same day: HTTP 200 with non-empty choices[0].message.content under this +// client's existing wire format. Provider-neutral seam and client contract are +// unchanged; only the model identity moved. const ( - defaultGroqModel = "llama-3.1-8b-instant" + defaultGroqModel = "qwen/qwen3.8-27b" defaultGroqBaseURL = "https://api.groq.com/openai/v1" groqTimeout = 30 * time.Second ) diff --git a/docs/adr/002-groq-llm-provider.md b/docs/adr/002-groq-llm-provider.md index 6f7addd..5c86a60 100644 --- a/docs/adr/002-groq-llm-provider.md +++ b/docs/adr/002-groq-llm-provider.md @@ -19,12 +19,23 @@ outcome is recorded here. ## Decision -- Provider: Groq. Model: `llama-3.1-8b-instant` (code default in +- Provider: Groq. Model: `qwen/qwen3.8-27b` (code default in `groq_model.go`; canonical string for the Groq qualification run). - Adjudication 2026-09-26 (Option A): the ADR originally contracted - `openai/gpt-oss-20b`, but the shipped code default is + `openai/gpt-oss-20b`, but the shipped code default was `llama-3.1-8b-instant` and working code is not changed to satisfy a stale ADR. ADR and README amended to the code default instead. +- Re-adjudication 2026-09-28 (APA-47): `llama-3.1-8b-instant` has since + been retired from Groq and answers HTTP 404 `model_not_found`, so the + config contracted above would have failed every inference call. Model + re-adjudicated to `qwen/qwen3.8-27b`, confirmed servable live against + the configured provider (HTTP 200, non-empty + `choices[0].message.content`) under the existing client wire format + (`max_tokens`, `temperature: 0`, `stream: false`). This is a config and + documentation change only: no `response_format`/JSON mode, no reasoning + parameter, no prompt change, and no model-specific response parsing were + added, and the model remains untrusted input behind the unchanged + deterministic validation/grounding boundary. - Client pattern: OpenAI-compatible (`base_url` + env key). Works with `openai.OpenAI`, LangChain `ChatOpenAI`, LangGraph nodes unchanged. - Ollama is out of scope. localgcp Vertex proxy is reserved for