baikai 0.7.1.0 → 0.7.2.0
raw patch · 12 files changed
+317/−18 lines, 12 filesPVP: major bump suggested
API removals or changes: PVP suggests a major version bump
API changes (from Hackage documentation)
+ Baikai.Models.Generated: anthropic_claude_sonnet_5_5 :: Model
+ Baikai.Models.Generated: openai_gpt_6_1_sol :: Model
+ Baikai.Provider.Cli.Internal: [structuredOutput] :: ClaudeCliReport -> !Maybe Value
+ Baikai.Provider.Cli.Internal: unsupportedFlagError :: Text -> String -> Int -> Text -> Maybe BaikaiError
+ Baikai.ResponseFormat: NativeJsonSchema :: StructuredOutputSupport
+ Baikai.ResponseFormat: NoStructuredOutput :: StructuredOutputSupport
+ Baikai.ResponseFormat: data StructuredOutputSupport
+ Baikai.ResponseFormat: declaredStructuredOutput :: Api -> StructuredOutputSupport
+ Baikai.ResponseFormat: instance GHC.Classes.Eq Baikai.ResponseFormat.StructuredOutputSupport
+ Baikai.ResponseFormat: instance GHC.Classes.Ord Baikai.ResponseFormat.StructuredOutputSupport
+ Baikai.ResponseFormat: instance GHC.Internal.Enum.Bounded Baikai.ResponseFormat.StructuredOutputSupport
+ Baikai.ResponseFormat: instance GHC.Internal.Enum.Enum Baikai.ResponseFormat.StructuredOutputSupport
+ Baikai.ResponseFormat: instance GHC.Internal.Generics.Generic Baikai.ResponseFormat.StructuredOutputSupport
+ Baikai.ResponseFormat: instance GHC.Internal.Show.Show Baikai.ResponseFormat.StructuredOutputSupport
- Baikai.Provider.Cli.Internal: ClaudeCliReport :: !Text -> !Bool -> !Maybe Text -> !Maybe Text -> !Maybe Usage -> ClaudeCliReport
+ Baikai.Provider.Cli.Internal: ClaudeCliReport :: !Text -> !Bool -> !Maybe Text -> !Maybe Text -> !Maybe Usage -> !Maybe Value -> ClaudeCliReport
Files
- CHANGELOG.md +61/−0
- baikai.cabal +1/−1
- fetch/FetchModelsCore.hs +18/−1
- src/Baikai/Models/Generated.hs +69/−0
- src/Baikai/Provider/Cli/Internal.hs +35/−3
- src/Baikai/Provider/Registry.hs +17/−5
- src/Baikai/ResponseFormat.hs +41/−0
- test/CatalogSpec.hs +4/−3
- test/CliInternalSpec.hs +48/−1
- test/PricingPolicySpec.hs +9/−2
- test/StrictEvidenceSpec.hs +1/−1
- test/SurfaceSpec.hs +13/−1
CHANGELOG.md view
@@ -7,6 +7,67 @@ ## [Unreleased] +## [baikai 0.7.2.0] - 2026-09-30++### Added++- Curated GPT-6.1 Sol on OpenAI Responses and Claude Sonnet 5.5 on Anthropic+ Messages (`openai_gpt_6_1_sol`, `anthropic_claude_sonnet_5_5`), with endpoint+ compatibility facts, standard prices, GPT-6.1 Sol's 272K-token context tier,+ and Sonnet 5.5's one-hour cache-write rate. Sonnet 5.5 rejects forced tool+ choice locally and has no fast mode. Both passed live text and function-tool+ acceptance on 2026-09-30 through `/v1/responses` and `/v1/messages`+ ([record](docs/validation/plan-85/2026-09-30-complete.json),+ [plan 85](docs/plans/85-prove-gpt-6-1-sol-and-claude-sonnet-5-5-live-compatibility.md)).+ The focused smoke runner gains the `sol61-*` and `sonnet55-*` cases.+- `Baikai.ResponseFormat.StructuredOutputSupport`+ (`NativeJsonSchema | NoStructuredOutput`), `declaredStructuredOutput :: Api ->+ StructuredOutputSupport`, and a `structuredOutput` field on `ApiProvider`+ (default `NoStructuredOutput` in `apiProviderWith`), so a caller can ask+ whether a transport enforces a schema without calling it. Every built-in+ provider declares `NativeJsonSchema`.++## [baikai-claude 0.7.1.0] - 2026-09-30++Requires `baikai >=0.7.2`.++### Added++- `claude -p` honours a `JsonSchema` response format (IR-11): it receives+ `--json-schema '<schema>'` and the response text is the tool's validated+ `structured_output`. A missing `structured_output` is a `DecodeFailure`; a+ `claude` too old for the flag yields an `InvalidRequest` error with the exit+ code rather than unconstrained text. `JsonObject`, `name` and `strict` are not+ forwarded; requests without a schema render the same argument vector as+ before. Both Claude providers declare `structuredOutput = NativeJsonSchema`.++### Fixed++- Anthropic adaptive `ThinkingHigh` now sends+ `output_config.effort: "high"` instead of omitting the field. Claude Opus 5.5+ defaults to `medium`, so it previously ran `ThinkingHigh` at medium effort;+ other adaptive models default to `high` and behave as before. Evidence for+ these calls records `effortText = "high"` and no longer carries+ `effort_omitted`, so strict evidence mode no longer refuses them.+ `Baikai.Evidence.EffortOmitted` stays exported, and older records still decode.++## [baikai-openai 0.7.1.0] - 2026-09-30++Requires `baikai >=0.7.2`.++### Added++- `codex exec` honours a `JsonSchema` response format (IR-11): the schema is+ written to a temporary file passed as `--output-schema <file>` and deleted+ however the call ends. A `codex` too old for the flag yields an+ `InvalidRequest` error with the exit code rather than unconstrained text.+ `JsonObject`, `name` and `strict` are not forwarded; requests without a schema+ render the same argument vector as before. All three OpenAI providers declare+ `structuredOutput = NativeJsonSchema`.+- `codexCliCommandWith`, which renders the `codex exec`+ vector with a given `--output-schema` file. `codexCliCommand` is unchanged+ and never renders the flag.+ ## [baikai 0.7.1.0] - 2026-09-23 ### Added
baikai.cabal view
@@ -1,6 +1,6 @@ cabal-version: 3.4 name: baikai-version: 0.7.1.0+version: 0.7.2.0 synopsis: Unified Haskell interface for multiple AI providers description: baikai provides a unified, provider-agnostic Haskell interface for working
fetch/FetchModelsCore.hs view
@@ -267,6 +267,7 @@ Map.union ( Map.fromList [ ("gpt-6-astra", Just (CatalogResponsesCompat astraResponsesFacts)),+ ("gpt-6.1-sol", Just (CatalogResponsesCompat gpt61ResponsesFacts)), ("gpt-6-sol", Just (CatalogResponsesCompat gpt6ResponsesFacts)), ("gpt-6-luna", Just (CatalogResponsesCompat gpt6ResponsesFacts)) ]@@ -304,6 +305,11 @@ -- https://developers.openai.com/api/docs/models/gpt-6-sol -- https://developers.openai.com/api/docs/models/gpt-6-luna gpt6ResponsesFacts = astraResponsesFacts+ -- 2026-09-29: as Astra, tool calling requires Responses; efforts low+ -- through max (no none or minimal); sampling removed; 30m cache TTL.+ -- https://developers.openai.com/api/docs/models/gpt-6.1-sol+ -- https://developers.openai.com/api/docs/guides/latest-model+ gpt61ResponsesFacts = astraResponsesFacts -- 2026-09-07: native tools require Responses; only modern 30m cache TTL. -- https://developers.openai.com/api/docs/guides/latest-model astraResponsesFacts =@@ -355,6 +361,12 @@ -- This is the finding: the retired prefix table did not know this id -- and sent it budget_tokens — same source. ("claude-sonnet-5", adaptiveNoSampling),+ -- 2026-09-29: adaptive thinking on by default; forced tool choice and+ -- nondefault sampling return 400; no fast mode. Baikai never sends+ -- the rejected "disabled" thinking type.+ -- https://platform.claude.com/docs/en/models/sonnet-5-5/whats-new-sonnet-5-5+ -- https://platform.claude.com/docs/en/build-with-claude/fast-mode+ ("claude-sonnet-5-5", adaptiveNoSampling & #supportsForcedToolChoice .~ False), -- 2026-08-27: as claude-opus-4-6 — budget deprecated but functional, -- sampling accepted; baikai prefers the non-deprecated shape — same -- source. Plan 40 left this membership to a live check that never@@ -639,11 +651,14 @@ where c = m ^. #cost --- | Provider documentation verified 2026-09-23. These rules supplement base+-- | Provider documentation verified 2026-09-23; GPT-6.1 Sol and Sonnet 5.5+-- verified 2026-09-29. These rules supplement base -- models.dev rates, which do not describe the full request billing policy. -- https://developers.openai.com/api/docs/models/gpt-6-astra -- https://developers.openai.com/api/docs/models/gpt-6-sol -- https://developers.openai.com/api/docs/models/gpt-6-luna+-- https://developers.openai.com/api/docs/models/gpt-6.1-sol+-- https://platform.claude.com/docs/en/models/sonnet-5-5/overview -- https://platform.claude.com/docs/en/models/fable-5-1/overview -- https://platform.claude.com/docs/en/models/opus-5-5/overview -- https://platform.claude.com/docs/en/build-with-claude/fast-mode@@ -654,6 +669,8 @@ (("anthropic", "claude-opus-5"), Model.PricingPolicy [] (Just 10)), (("anthropic", "claude-opus-4-8"), Model.PricingPolicy [] (Just 10)), (("openai", "gpt-6-astra"), Model.PricingPolicy [Model.InputPriceTier 272000 (Model.ModelCost 20 75 2 25)] Nothing),+ (("anthropic", "claude-sonnet-5-5"), Model.PricingPolicy [] (Just 4)),+ (("openai", "gpt-6.1-sol"), Model.PricingPolicy [Model.InputPriceTier 272000 (Model.ModelCost 4 15 0.2 5)] Nothing), (("openai", "gpt-6-sol"), Model.PricingPolicy [Model.InputPriceTier 272000 (Model.ModelCost 4 15 0.4 5)] Nothing), (("openai", "gpt-6-luna"), Model.PricingPolicy [Model.InputPriceTier 272000 (Model.ModelCost 0.2 0.75 0.02 0.25)] Nothing), (("anthropic", "claude-fable-5-1"), Model.PricingPolicy [] (Just 20))
src/Baikai/Models/Generated.hs view
@@ -509,6 +509,41 @@ } } +anthropic_claude_sonnet_5_5 :: Model+anthropic_claude_sonnet_5_5 =+ emptyModel+ { modelId = "claude-sonnet-5-5",+ name = "Claude Sonnet 5.5",+ api = AnthropicMessages,+ provider = "anthropic",+ baseUrl = "https://api.anthropic.com",+ reasoning = True,+ input = [InputText, InputImage],+ cost =+ ModelCost+ { inputCost = 2 % 1,+ outputCost = 10 % 1,+ cacheReadCost = 1 % 5,+ cacheWriteCost = 5 % 2+ },+ fastModeCost = Nothing,+ pricingPolicy = Just (PricingPolicy [] (Just (4 % 1))),+ contextWindow = 1000000,+ maxOutputTokens = 128000,+ headers = Map.empty,+ compat =+ CompatAnthropicMessages+ defaultAnthropicMessagesCompat+ { supportsLongCacheRetention = True,+ supportsCacheControlOnTools = True,+ sendSessionAffinityHeaders = False,+ thinkingStyle = AnthropicThinkingAdaptive,+ supportsSamplingParameters = False,+ supportsFastMode = False,+ supportsForcedToolChoice = False+ }+ }+ deepseek_deepseek_chat :: Model deepseek_deepseek_chat = emptyModel@@ -1009,6 +1044,38 @@ compat = CompatNone } +openai_gpt_6_1_sol :: Model+openai_gpt_6_1_sol =+ emptyModel+ { modelId = "gpt-6.1-sol",+ name = "GPT-6.1 Sol",+ api = OpenAIResponses,+ provider = "openai",+ baseUrl = "https://api.openai.com",+ reasoning = True,+ input = [InputText, InputImage],+ cost =+ ModelCost+ { inputCost = 2 % 1,+ outputCost = 10 % 1,+ cacheReadCost = 1 % 10,+ cacheWriteCost = 5 % 2+ },+ fastModeCost = Nothing,+ pricingPolicy = Just (PricingPolicy [InputPriceTier 272000 (ModelCost (4 % 1) (15 % 1) (1 % 5) (5 % 1))] Nothing),+ contextWindow = 1050000,+ maxOutputTokens = 128000,+ headers = Map.empty,+ compat =+ CompatOpenAIResponses+ defaultOpenAIResponsesCompat+ { supportedReasoningEfforts = Just [ThinkingLow, ThinkingMedium, ThinkingHigh, ThinkingXHigh, ThinkingMax],+ supportsSamplingParameters = False,+ supportsLongCacheRetention = False,+ supportsPromptCacheOptions = True+ }+ }+ openai_gpt_6_astra :: Model openai_gpt_6_astra = emptyModel@@ -1270,6 +1337,7 @@ anthropic_claude_sonnet_4_5, anthropic_claude_sonnet_4_6, anthropic_claude_sonnet_5,+ anthropic_claude_sonnet_5_5, deepseek_deepseek_chat, deepseek_deepseek_reasoner, openai_gpt_4_1,@@ -1290,6 +1358,7 @@ openai_gpt_5_6_terra, openai_gpt_5_mini, openai_gpt_5_nano,+ openai_gpt_6_1_sol, openai_gpt_6_astra, openai_gpt_6_luna, openai_gpt_6_sol,
src/Baikai/Provider/Cli/Internal.hs view
@@ -22,6 +22,9 @@ ClaudeCliReport (..), decodeClaudeCliResult, + -- * Failures a CLI reports about its own arguments+ unsupportedFlagError,+ -- * What baikai knows about the process it launched ExecutableIdentity (..), executableIdentity,@@ -41,7 +44,7 @@ ) import Baikai.Context (Context) import Baikai.Cost (Cost (..), providerReportedBasis, zeroCost, zeroCostBreakdown)-import Baikai.Error (BaikaiError, decodeError)+import Baikai.Error (BaikaiError (..), ErrorCategory (..), decodeError, processError) import Baikai.Evidence (EvidenceStrength (..), Observed (..), deriveStrength, usageEnvelope) import Baikai.Message ( AssistantPayload (..),@@ -447,7 +450,10 @@ reportedModel :: !(Maybe Text), -- | The token counts and reported cost, when the tool included a -- usage block.- usage :: !(Maybe Usage)+ usage :: !(Maybe Usage),+ -- | The validated value of the tool's internal @StructuredOutput@+ -- tool, present when the run was given @--json-schema@.+ structuredOutput :: !(Maybe Value) } deriving stock (Eq, Show, Generic) @@ -483,14 +489,40 @@ body <- o .: "result" failed <- o .: "is_error" session <- o .:? "session_id"+ structured <- o .:? "structured_output" pure ClaudeCliReport { result = body, isError = failed, sessionId = session, reportedModel = KeyMap.lookup "modelUsage" o >>= soleModelUsageKey,- usage = claudeUsage o+ usage = claudeUsage o,+ structuredOutput = structured }++-- | Recognise a CLI's refusal of a flag it does not know.+--+-- Given the tool name, the flag baikai sent, the exit code and the+-- decoded stderr, this returns an 'InvalidRequest' error (keeping the+-- exit code) exactly when stderr is the tool's argument parser rejecting+-- that flag: commander's @unknown option '<flag>'@ (@claude@) or clap's+-- @unexpected argument '<flag>'@ (@codex@). Any other failure, including+-- one about the flag's /value/, is 'Nothing' and stays a+-- 'ProcessFailure'.+unsupportedFlagError :: Text -> String -> Int -> Text -> Maybe BaikaiError+unsupportedFlagError tool flag code stderr+ | any (`Text.isInfixOf` stderr) spellings =+ Just ((processError code msg) {category = InvalidRequest})+ | otherwise = Nothing+ where+ quoted = "'" <> Text.pack flag <> "'"+ spellings = ["unknown option " <> quoted, "unexpected argument " <> quoted]+ msg =+ tool+ <> " does not accept "+ <> Text.pack flag+ <> "; upgrade it or unset Options.responseFormat: "+ <> stderr -- | The model @claude@ reported as having consumed tokens. --
src/Baikai/Provider/Registry.hs view
@@ -15,7 +15,7 @@ -- error-shaped 'Response' in the 'Baikai.Error.ProviderUnavailable' -- category. module Baikai.Provider.Registry- ( ApiProvider (apiTag, stream, complete, describeThinking, strengthCeiling),+ ( ApiProvider (apiTag, stream, complete, describeThinking, strengthCeiling, structuredOutput), apiProviderWith, describeApi, ProviderRegistry,@@ -51,6 +51,7 @@ import Baikai.Options (Options, emptyOptions) import Baikai.Options qualified as Options import Baikai.Response (Response (..), errorResponse, flattenAssistantBlocks, flattenAssistantText, responseError)+import Baikai.ResponseFormat (StructuredOutputSupport (..)) import Baikai.StopReason (StopReason (..)) import Baikai.Stream.Event (AssistantMessageEvent) import Control.Exception qualified as Exception@@ -111,7 +112,16 @@ -- declare 'Evidence.EvidenceRequestedOnly', and will still fail a -- strict caller at the terminal — see -- @docs\/adr\/0014-strict-evidence-means-a-record-exists.md@.- strengthCeiling :: !Evidence.EvidenceStrength+ strengthCeiling :: !Evidence.EvidenceStrength,+ -- | Whether this provider forwards a 'Baikai.ResponseFormat.JsonSchema'+ -- request to a mechanism the host enforces: a static declaration a+ -- caller can read before dispatch.+ --+ -- Like 'strengthCeiling', declare only what the provider delivers.+ -- The built-in providers take their value from+ -- 'Baikai.ResponseFormat.declaredStructuredOutput'; a caller-supplied+ -- transport that honours schemas sets 'NativeJsonSchema' itself.+ structuredOutput :: !StructuredOutputSupport } deriving stock (Generic) @@ -127,8 +137,9 @@ -- 'describeThinking' defaults to reporting that nothing was requested -- and nothing translated, which is honest for a transport with no -- reasoning controls; 'strengthCeiling' defaults to--- 'Evidence.EvidenceRequestedOnly', matching @declaredStrength (Custom _)@.--- Override either by record update.+-- 'Evidence.EvidenceRequestedOnly', matching @declaredStrength (Custom _)@;+-- 'structuredOutput' defaults to 'NoStructuredOutput'. Override any of+-- them by record update. apiProviderWith :: Api -> (Model -> Context -> Options -> Stream IO AssistantMessageEvent) ->@@ -140,7 +151,8 @@ stream = producer, complete = completer, describeThinking = \_ _ -> Evidence.noThinkingRequested,- strengthCeiling = Evidence.EvidenceRequestedOnly+ strengthCeiling = Evidence.EvidenceRequestedOnly,+ structuredOutput = NoStructuredOutput } -- | How an 'Api' tag reads in a dispatch failure.
src/Baikai/ResponseFormat.hs view
@@ -1,3 +1,4 @@+{-# LANGUAGE LambdaCase #-} {-# LANGUAGE OverloadedRecordDot #-} -- | Provider-agnostic structured-output preference.@@ -10,9 +11,14 @@ ( ResponseFormat (..), JsonSchemaFormat (name, schema, strict), jsonSchemaFormat,++ -- * Which transports enforce a schema+ StructuredOutputSupport (..),+ declaredStructuredOutput, ) where +import Baikai.Api (Api (..)) import Data.Aeson ( FromJSON (parseJSON), ToJSON (toJSON),@@ -96,3 +102,38 @@ (jsonSchemaFormat schemaName schemaDoc) {strict = fromMaybe False isStrict} ) other -> fail ("unknown ResponseFormat tag: " <> show other)++-- | Whether a transport forwards a 'JsonSchema' request to a mechanism+-- the host enforces.+--+-- Answerable without spawning or calling anything: read it from a+-- registered provider (@Baikai.Provider.Registry.structuredOutput@) or+-- from a model's transport tag ('declaredStructuredOutput').+data StructuredOutputSupport+ = -- | 'JsonSchema' reaches the provider and the provider enforces it.+ NativeJsonSchema+ | -- | 'Baikai.Options.responseFormat' is ignored by this transport.+ NoStructuredOutput+ deriving stock (Eq, Ord, Show, Enum, Bounded, Generic)++-- | The structured-output support of each built-in transport.+--+-- 'NativeJsonSchema' covers 'JsonSchema' only. The HTTP providers send+-- the schema in the request body; the subscription CLIs receive it as a+-- flag (@claude -p --json-schema@, @codex exec --output-schema@) and+-- 'JsonObject', 'strict' and 'name' are not forwarded to them. An+-- installed CLI too old to know the flag still rejects it at run time;+-- that call returns an error-shaped response in the+-- 'Baikai.Error.InvalidRequest' category rather than unconstrained text.+--+-- A 'Custom' transport is 'NoStructuredOutput' here because this table+-- cannot know what a caller-registered provider does; that provider's+-- own @structuredOutput@ field is the authoritative answer.+declaredStructuredOutput :: Api -> StructuredOutputSupport+declaredStructuredOutput = \case+ AnthropicMessages -> NativeJsonSchema+ OpenAIChatCompletions -> NativeJsonSchema+ OpenAIResponses -> NativeJsonSchema+ AnthropicMessagesCli -> NativeJsonSchema+ OpenAICompletionsCli -> NativeJsonSchema+ Custom _ -> NoStructuredOutput
test/CatalogSpec.hs view
@@ -60,7 +60,7 @@ c.supportsSamplingParameters @?= False c.supportedReasoningEfforts @?= Just [ThinkingLow, ThinkingMedium, ThinkingHigh, ThinkingXHigh, ThinkingMax] _ -> assertFailure "Astra needs explicit OpenAI endpoint facts",- testCase "Sol and Luna select Responses with explicit endpoint facts" $+ testCase "Sol, Luna, and GPT-6.1 Sol select Responses with explicit endpoint facts" $ mapM_ ( \mid -> do [api m | m <- allModels, modelId m == mid] @?= [OpenAIResponses]@@ -72,7 +72,7 @@ c.supportedReasoningEfforts @?= Just [ThinkingLow, ThinkingMedium, ThinkingHigh, ThinkingXHigh, ThinkingMax] _ -> assertFailure "GPT-6 Sol/Luna need explicit OpenAI Responses facts" )- ["gpt-6-sol", "gpt-6-luna"],+ ["gpt-6-sol", "gpt-6-luna", "gpt-6.1-sol"], testCase "regenerating from data/models produces no diff" $ withSystemTempDirectory "baikai-catalog-spec" $ \tmpDir -> do let regenPath = tmpDir <> "/Generated.hs"@@ -119,7 +119,8 @@ ("claude-opus-5-5", (AnthropicThinkingAdaptive, False, False, True)), ("claude-sonnet-4-5", (AnthropicThinkingBudget, True, True, False)), ("claude-sonnet-4-6", (AnthropicThinkingAdaptive, True, True, False)),- ("claude-sonnet-5", (AnthropicThinkingAdaptive, False, True, False))+ ("claude-sonnet-5", (AnthropicThinkingAdaptive, False, True, False)),+ ("claude-sonnet-5-5", (AnthropicThinkingAdaptive, False, False, False)) ] assertFacts :: Model -> IO ()
test/CliInternalSpec.hs view
@@ -12,6 +12,7 @@ import Baikai import Baikai.Provider.Cli.Internal import Control.Lens ((&), (.~), (^.))+import Data.Aeson qualified as Aeson import Data.ByteString (ByteString) import Data.ByteString qualified as BS import Data.ByteString.Char8 qualified as BS8@@ -35,6 +36,7 @@ [ promptTests, codexParserTests, claudeParserTests,+ unsupportedFlagTests, executableIdentityTests, evidenceHelperTests ]@@ -267,7 +269,52 @@ testCase "malformed stdout is a decode error rather than an exception" $ case decodeClaudeCliResult "not json" of Left _ -> pure ()- Right r -> assertFailure ("expected a decode error, got: " <> show r)+ Right r -> assertFailure ("expected a decode error, got: " <> show r),+ -- Shape recorded from Claude Code 2.1.285 run with --json-schema:+ -- the validated value arrives beside a compact copy in "result".+ testCase "a --json-schema run's structured_output decodes" $+ case decodeClaudeCliResult+ "{\"type\":\"result\",\"is_error\":false,\"stop_reason\":\"tool_use\",\+ \\"result\":\"{\\\"items\\\":[]}\",\"structured_output\":{\"items\":[]}}" of+ Left err -> assertFailure ("expected the document to decode: " <> show err)+ Right r -> do+ r ^. #result @?= "{\"items\":[]}"+ r ^. #structuredOutput @?= Just (Aeson.object ["items" Aeson..= ([] :: [Aeson.Value])]),+ testCase "a run without --json-schema has no structured output" $+ case decodeClaudeCliResult "{\"result\":\"pong\",\"is_error\":false}" of+ Left err -> assertFailure ("expected a bare object to decode: " <> show err)+ Right r -> r ^. #structuredOutput @?= Nothing+ ]++-- ============================================================+-- Unsupported-flag classification+-- ============================================================++unsupportedFlagTests :: TestTree+unsupportedFlagTests =+ testGroup+ "unsupportedFlagError"+ [ testCase "commander's unknown option (claude) is InvalidRequest with the exit code" $+ case unsupportedFlagError "claude" "--json-schema" 1 "error: unknown option '--json-schema'\n" of+ Nothing -> assertFailure "expected the rejection to be recognised"+ Just e -> do+ e ^. #category @?= InvalidRequest+ e ^. #exitCode @?= Just 1+ assertBool+ ("message names the tool and flag: " <> show (e ^. #message))+ ("claude does not accept --json-schema" `Text.isPrefixOf` (e ^. #message)),+ testCase "clap's unexpected argument (codex) is InvalidRequest with the exit code" $+ case unsupportedFlagError "codex" "--output-schema" 2 "error: unexpected argument '--output-schema' found\n" of+ Nothing -> assertFailure "expected the rejection to be recognised"+ Just e -> do+ e ^. #category @?= InvalidRequest+ e ^. #exitCode @?= Just 2,+ testCase "an unreadable schema file is not an unsupported flag" $+ unsupportedFlagError "codex" "--output-schema" 1 "Failed to read output schema file /tmp/x.json: No such file"+ @?= Nothing,+ testCase "a rejection naming a different flag is not this flag" $+ unsupportedFlagError "claude" "--json-schema" 1 "error: unknown option '--bogus-flag'"+ @?= Nothing ] -- ============================================================
test/PricingPolicySpec.hs view
@@ -23,8 +23,8 @@ tests = testGroup "Pricing policy"- [ testCase "Sol and Luna prices switch the complete request above 272K" $- forM_ [(Models.openai_gpt_6_sol, M.ModelCost 2 10 (1 / 5) (5 / 2), M.ModelCost 4 15 (2 / 5) 5), (Models.openai_gpt_6_luna, M.ModelCost (1 / 10) (1 / 2) (1 / 100) (1 / 8), M.ModelCost (1 / 5) (3 / 4) (1 / 50) (1 / 4))] $ \(m, standard, highRates) -> do+ [ testCase "Sol, Luna, and GPT-6.1 Sol prices switch the complete request above 272K" $+ forM_ [(Models.openai_gpt_6_sol, M.ModelCost 2 10 (1 / 5) (5 / 2), M.ModelCost 4 15 (2 / 5) 5), (Models.openai_gpt_6_luna, M.ModelCost (1 / 10) (1 / 2) (1 / 100) (1 / 8), M.ModelCost (1 / 5) (3 / 4) (1 / 50) (1 / 4)), (Models.openai_gpt_6_1_sol, M.ModelCost 2 10 (1 / 10) (5 / 2), M.ModelCost 4 15 (1 / 5) 5)] $ \(m, standard, highRates) -> do forM_ [(272000, standard), (272001, highRates)] $ \(n, expected) -> do let u = U.zeroUsage & #inputTokens .~ (n - 2000) & #cacheReadTokens .~ 1000 & #cacheWriteTokens .~ 1000 & #outputTokens .~ 100 resolveRates Nothing m u @?= Right expected@@ -38,6 +38,13 @@ (computeCostAtSpeed m SpeedFast u).usd @?= 10 (computeCostForService (Just CacheRetentionLong) Nothing m observedFast).usd @?= 16 Set.member (C.UnsupportedSpeed "fast") (computeCostForService (Just CacheRetentionLong) Nothing m observedFast).basis.estimateReasons @?= False,+ testCase "Sonnet 5.5 prices short and long cache writes and has no fast rates" $ do+ let m = Models.anthropic_claude_sonnet_5_5+ u = U.zeroUsage & #cacheWriteTokens .~ 1000000+ (computeCostWith (Just CacheRetentionShort) m u).usd @?= 5 / 2+ (computeCostWith (Just CacheRetentionLong) m u).usd @?= 4+ (computeCost m (U.zeroUsage & #cacheReadTokens .~ 1000000)).usd @?= 1 / 5+ Set.member (C.UnsupportedSpeed "fast") (computeCostAtSpeed m SpeedFast u).basis.estimateReasons @?= True, testCase "requested tiers never substitute for observed service" $ do let unknown = N.normalizeUsage N.InclusiveInput (N.ReportedUsage (Just 1000) (Just 0) (Just 0) (Just 0) Nothing) standard = U.observeBilling [U.BillingServiceTier "default"] unknown
test/StrictEvidenceSpec.hs view
@@ -180,7 +180,7 @@ (EffortCollapsedToToggle ThinkingMax) "bare on/off toggle", refusesDowngrade- "an adaptive high sends no effort field at all"+ "an omitted effort field is indistinguishable from the provider default" (EffortOmitted ThinkingHigh) "indistinguishable on the wire", refusesDowngrade
test/SurfaceSpec.hs view
@@ -64,11 +64,23 @@ logCfg = callLogConfig "/dev/null" provider ^. #apiTag @?= Custom "probe" provider ^. #strengthCeiling @?= EvidenceRequestedOnly+ provider ^. #structuredOutput @?= NoStructuredOutput req ^. #attempt @?= 2 req ^. #runId @?= "r" tool ^. #name @?= "t" tool ^. #parameters @?= Aeson.Null embedding ^. #modelId @?= "e" logCfg ^. #path @?= "/dev/null"- logCfg ^. #enabled @?= True+ logCfg ^. #enabled @?= True,+ testCase "declaredStructuredOutput: every built-in transport enforces a schema, Custom does not" $ do+ let builtIns =+ [ AnthropicMessages,+ OpenAIChatCompletions,+ OpenAIResponses,+ AnthropicMessagesCli,+ OpenAICompletionsCli+ ]+ map declaredStructuredOutput builtIns @?= map (const NativeJsonSchema) builtIns+ declaredStructuredOutput (Custom "x") @?= NoStructuredOutput+ [minBound .. maxBound] @?= [NativeJsonSchema, NoStructuredOutput] ]