shikumi-eval 0.3.0.0 → 0.3.0.1
raw patch · 12 files changed
+102/−85 lines, 12 filesdep ~baikaidep ~effectfuldep ~shikumi-evalPVP ok
version bump matches the API change (PVP)
Dependency ranges changed: baikai, effectful, shikumi-eval
API changes (from Hackage documentation)
Files
- CHANGELOG.md +4/−0
- shikumi-eval.cabal +54/−45
- src/Shikumi/Eval.hs +10/−6
- src/Shikumi/Eval/Embedding.hs +2/−2
- src/Shikumi/Eval/Evaluate.hs +4/−4
- src/Shikumi/Eval/Golden.hs +6/−6
- src/Shikumi/Eval/Metric.hs +2/−2
- src/Shikumi/Eval/Report.hs +7/−7
- src/Shikumi/Eval/Types.hs +9/−9
- src/Shikumi/Eval/Usage.hs +1/−1
- test/GoldenSpec.hs +1/−1
- test/MetricSpec.hs +2/−2
CHANGELOG.md view
@@ -2,6 +2,10 @@ ## Unreleased +## 0.3.0.1 — 2026-10-05++- Move the dependency on `mori://shinzui/baikai/packages/baikai` to `>=0.7.1.0 && <0.8` and widen the `effectful` bound to `>=2.6 && <2.8`, so both effectful 2.6 and 2.7 are supported (effectful 2.7 needs `baikai-effectful` 0.4.0.2, effectful 2.6 needs 0.4.0.1). Bounds only; no source changed.+ ## 0.3.0.0 — 2026-09-08 - Raise the internal `shikumi` bound to `^>=0.4.0.0` for the breaking core release.
shikumi-eval.cabal view
@@ -1,8 +1,8 @@-cabal-version: 3.4-name: shikumi-eval-version: 0.3.0.0-synopsis: Typed evaluation framework for shikumi LM programs (EP-8)-category: AI+cabal-version: 3.4+name: shikumi-eval+version: 0.3.0.1+synopsis: Typed evaluation framework for shikumi LM programs (EP-8)+category: AI description: The evaluation framework for shikumi: the owned data model (@Example@, @Prediction@, @Dataset@, @Metric@, @Score@, @Report@ — MasterPlan integration@@ -11,20 +11,26 @@ parallelism and per-example error boundaries, and golden testing that pins a program's behaviour deterministically under a mock or replayed LM. -license: BSD-3-Clause-author: Nadeem Bitar-maintainer: nadeem@gmail.com-build-type: Simple+license: BSD-3-Clause+author: Nadeem Bitar+maintainer: nadeem@gmail.com+build-type: Simple extra-doc-files: CHANGELOG.md common common-options ghc-options:- -Wall -Wcompat -Widentities -Wincomplete-uni-patterns- -Wincomplete-record-updates -Wredundant-constraints- -fhide-source-paths -Wmissing-export-lists -Wpartial-fields+ -Wall+ -Wcompat+ -Widentities+ -Wincomplete-uni-patterns+ -Wincomplete-record-updates+ -Wredundant-constraints+ -fhide-source-paths+ -Wmissing-export-lists+ -Wpartial-fields -Wmissing-deriving-strategies - default-language: GHC2024+ default-language: GHC2024 default-extensions: DeriveAnyClass DuplicateRecordFields@@ -32,8 +38,8 @@ OverloadedStrings library- import: common-options- hs-source-dirs: src+ import: common-options+ hs-source-dirs: src exposed-modules: Shikumi.Eval Shikumi.Eval.Embedding@@ -45,26 +51,29 @@ Shikumi.Eval.Usage build-depends:- , aeson >=2.2 && <2.3- , baikai >=0.7.0.0 && <0.8- , base >=4.20 && <5- , bytestring >=0.11 && <0.13- , containers >=0.6 && <0.9- , effectful >=2.5 && <2.7- , generic-lens >=2.2 && <2.4- , lens ^>=5.3- , shikumi ^>=0.4.0.0- , tasty >=1.4 && <1.6- , tasty-golden >=2.3 && <2.4- , text ^>=2.1- , vector >=0.13 && <0.14+ aeson >=2.2 && <2.3,+ baikai >=0.7.1.0 && <0.8,+ base >=4.20 && <5,+ bytestring >=0.11 && <0.13,+ containers >=0.6 && <0.9,+ effectful >=2.6 && <2.8,+ generic-lens >=2.2 && <2.4,+ lens ^>=5.3,+ shikumi ^>=0.4.0.0,+ tasty >=1.4 && <1.6,+ tasty-golden >=2.3 && <2.4,+ text ^>=2.1,+ vector >=0.13 && <0.14, test-suite shikumi-eval-test- import: common-options- type: exitcode-stdio-1.0+ import: common-options+ type: exitcode-stdio-1.0 hs-source-dirs: test- main-is: Main.hs- ghc-options: -threaded -with-rtsopts=-N+ main-is: Main.hs+ ghc-options:+ -threaded+ -with-rtsopts=-N+ other-modules: DocSpec EmbeddingSpec@@ -78,16 +87,16 @@ UsageSpec build-depends:- , aeson- , baikai >=0.7.0.0 && <0.8- , base- , effectful- , generic-lens- , lens- , shikumi ^>=0.4.0.0- , shikumi-eval ^>=0.3.0.0- , tasty- , tasty-golden- , tasty-hunit- , text- , vector+ aeson,+ baikai >=0.7.1.0 && <0.8,+ base,+ effectful,+ generic-lens,+ lens,+ shikumi ^>=0.4.0.0,+ shikumi-eval ^>=0.3.0.1,+ tasty,+ tasty-golden,+ tasty-hunit,+ text,+ vector,
src/Shikumi/Eval.hs view
@@ -1,8 +1,8 @@ -- | The shikumi evaluation framework — one import for the whole public surface -- (MasterPlan integration point #5). Re-exports the owned data model--- ('Shikumi.Eval.Types'), the metrics ('Shikumi.Eval.Metric'), the report--- ('Shikumi.Eval.Report'), the 'evaluate' runner ('Shikumi.Eval.Evaluate'), and--- golden testing ('Shikumi.Eval.Golden').+-- ("Shikumi.Eval.Types"), the metrics ("Shikumi.Eval.Metric"), the report+-- ("Shikumi.Eval.Report"), the 'evaluate' runner ("Shikumi.Eval.Evaluate"), and+-- golden testing ("Shikumi.Eval.Golden"). -- -- Worked example: measure a @Question -> Answer@ program over a small typed -- dataset with the exact-match metric, then read the aggregate score. (The@@ -12,15 +12,19 @@ -- @ -- import Shikumi.Eval ----- qaData :: 'Dataset' Question Answer+-- qaData :: t'Dataset' Question Answer -- qaData = 'dataset' -- [ 'example' (Question \"What is 2+2?\") (Answer \"4\") -- , 'example' (Question \"Capital of France?\") (Answer \"Paris\") -- ] -- -- -- inside an Eff stack with (LLM, Concurrent, Error ShikumiError, IOE):--- run :: ('LLM' :> es, 'Concurrent' :> es, 'Error' 'ShikumiError' :> es, 'IOE' :> es)--- => 'Program' Question Answer -> Eff es Double+-- run ::+-- ( t'Shikumi.LLM.LLM' :> es+-- , 'Effectful.Concurrent.Concurrent' :> es+-- , 'Effectful.Error.Static.Error' t'Shikumi.Error.ShikumiError' :> es+-- , 'Effectful.IOE' :> es+-- ) => t'Shikumi.Program.Program' Question Answer -> Eff es Double -- run prog = do -- report <- 'evaluatePure' qaData 'exactMatch' prog -- pure ('aggregateScore' report)
src/Shikumi/Eval/Embedding.hs view
@@ -2,10 +2,10 @@ -- | A real embeddings interpreter for the existing 'Embedding' effect (EP-15). ----- 'Shikumi.Eval.Metric' ships only the pure 'Shikumi.Eval.Metric.runEmbedding'+-- "Shikumi.Eval.Metric" ships only the pure 'Shikumi.Eval.Metric.runEmbedding' -- (a @Text -> Vector Double@ table), so @semanticSimilarity@ has no way to call a -- real backend. This module adds interpreters that drive the upstream--- 'Baikai.Embedding' client (an OpenAI-compatible @\/v1\/embeddings@ endpoint),+-- "Baikai.Embedding" client (an OpenAI-compatible @\/v1\/embeddings@ endpoint), -- so @semanticSimilarity@ runs end to end. The @Embedding@ effect itself is -- unchanged (integration point #5); only new interpreters are added. --
src/Shikumi/Eval/Evaluate.hs view
@@ -1,10 +1,10 @@--- | The core deliverable: @evaluate@ runs a 'Program' over a typed 'Dataset',+-- | The core deliverable: @evaluate@ runs a t'Program' over a typed 'Dataset', -- scores each example with a metric, and returns a 'Report' summarising the -- aggregate score, the per-example breakdown, token usage, cost, and latency. -- -- Examples run with __bounded__ concurrency ('pooledForConcurrentlyN', worker -- count from 'concurrency'), which preserves dataset order in the result. Each--- example sits inside a per-example error boundary: a 'ShikumiError' from+-- example sits inside a per-example error boundary: a t'ShikumiError' from -- 'runProgram' becomes a 'ProgramError', one from an effectful metric becomes a -- 'MetricError', and the 'FailurePolicy' decides whether that example is scored -- and the run continues ('FailScore') or the whole run aborts ('FailAbort').@@ -154,7 +154,7 @@ FailAbort -> throwError e FailScore s -> pure (s, Just reason) --- | Build a 'Prediction' by running the program @numSamples@ times (at least+-- | Build a t'Prediction' by running the program @numSamples@ times (at least -- once); a single sample yields a one-element prediction. buildPrediction :: (LLM :> es, Error ShikumiError :> es) =>@@ -173,7 +173,7 @@ where n = max 1 (numSamples cfg) --- | Run an action, catching a 'ShikumiError' into a 'Left'.+-- | Run an action, catching a t'ShikumiError' into a 'Left'. tryShikumi :: (Error ShikumiError :> es) => Eff es a -> Eff es (Either ShikumiError a) tryShikumi act = (Right <$> act) `catchError` \_ e -> pure (Left e)
src/Shikumi/Eval/Golden.hs view
@@ -6,15 +6,15 @@ -- stack under a /mock or replayed/ LM (never a live one), so the only thing that -- can change the output is a genuine change in program behaviour. ----- 'goldenProgram' compares a per-example transcript (one @index\\t<output>@ line--- per example, in dataset order); 'goldenReport' compares the rendered 'Report'+-- 'goldenProgram' compares a per-example transcript (one @index\\t\<output\>@ line+-- per example, in dataset order); 'goldenReport' compares the rendered t'Shikumi.Eval.Report.Report' -- from 'evaluate'. Both are built on @tasty-golden@'s 'goldenVsString' — -- regenerate the golden file with @cabal test --test-options=--accept@. -- -- The runner is a single rank-2 function @forall a. Eff es a -> IO a@, so the -- helper stays agnostic to the exact effect stack: the caller picks a concrete -- @es@ (e.g. @'[LLM, Error ShikumiError, IOE]@) and a function collapsing it to--- 'IO' (throwing on a 'ShikumiError').+-- 'IO' (throwing on a t'ShikumiError'). module Shikumi.Eval.Golden ( goldenProgram, goldenReport,@@ -49,7 +49,7 @@ TestName -> -- | golden file path (relative to the package directory) FilePath ->- -- | how to run the effect stack (mock/replay), throwing on a 'ShikumiError'+ -- | how to run the effect stack (mock/replay), throwing on a t'ShikumiError' (forall a. Eff es a -> IO a) -> Dataset i o -> Program i o ->@@ -62,7 +62,7 @@ outs <- runner (traverse (runProgram prog) inputs) pure (encodeText (renderTranscript (zip [0 ..] (map render outs)))) --- | Like 'goldenProgram' but compares the rendered 'Report' produced by+-- | Like 'goldenProgram' but compares the rendered t'Shikumi.Eval.Report.Report' produced by -- 'evaluate' with the given metric, rather than a raw transcript. goldenReport :: (LLM :> es, Concurrent :> es, Error ShikumiError :> es, Time :> es, Prim :> es) =>@@ -78,7 +78,7 @@ report <- runner (evaluate ds metric prog) pure (encodeText (renderReportText report)) --- | One @index\\t<rendered output>@ line per example, in dataset order.+-- | One @index\\t\<rendered output\>@ line per example, in dataset order. renderTranscript :: [(Int, Text)] -> Text renderTranscript = T.unlines . map (\(i, t) -> T.pack (show i) <> "\t" <> t)
src/Shikumi/Eval/Metric.hs view
@@ -232,7 +232,7 @@ instance Validatable Grade -- | The judge program: a single structured-output predictor that returns a--- 'Grade'. The @rubricText@ becomes the signature instruction.+-- t'Grade'. The @rubricText@ becomes the signature instruction. judgeProgram :: Text -> Program JudgeInput Grade judgeProgram rubricText = predict (judgeSig rubricText) @@ -247,7 +247,7 @@ -- | LLM-as-judge: ask a model to grade the prediction against the expected -- output. The rubric is supplied as text; the model returns a numeric grade, -- decoded through the structured-output path. A decode failure surfaces as a--- 'ShikumiError' (the per-example boundary in @evaluate@ turns it into a+-- t'ShikumiError' (the per-example boundary in @evaluate@ turns it into a -- 'Shikumi.Eval.Report.MetricError'). Threads @Error ShikumiError@ because -- 'runProgram' does (EP-4): decode failures throw typed errors. modelJudge ::
src/Shikumi/Eval/Report.hs view
@@ -1,11 +1,11 @@--- | The evaluation report and its aggregation. A 'Report' summarises a run over+-- | The evaluation report and its aggregation. A t'Report' summarises a run over -- a dataset: the mean per-example score, pass/fail counts, summed token usage and--- cost, summed per-example latency, and a per-example breakdown ('ExampleResult',+-- cost, summed per-example latency, and a per-example breakdown ([ExampleResult]("Shikumi.Eval.Report#t:ExampleResult"), -- retained in dataset order for failure analysis). 'mkReport' is the pure aggregator; -- 'renderReportText' is a deterministic human-readable rendering reused by the--- CLI (@docs/plans/12-cli-and-developer-experience.md@) and by golden tests.+-- CLI (@docs\/plans\/12-cli-and-developer-experience.md@) and by golden tests. ----- 'EvalConfig' carries the run knobs (bounded 'concurrency', the 'FailurePolicy'+-- t'EvalConfig' carries the run knobs (bounded 'concurrency', the 'FailurePolicy' -- for per-example errors, optional per-example timeout, and 'numSamples' per example -- for multi-sample metrics); 'defaultEvalConfig' scores failures @0@ and keeps going. module Shikumi.Eval.Report@@ -47,7 +47,7 @@ -- | The reason an example did not complete normally. data FailureReason- = -- | @runProgram@ threw a 'Shikumi.Error.ShikumiError' (rendered to text)+ = -- | @runProgram@ threw a t'Shikumi.Error.ShikumiError' (rendered to text) ProgramError !Text | -- | the metric itself failed (an effectful metric raised an error) MetricError !Text@@ -84,7 +84,7 @@ } deriving stock (Eq, Show) --- | The zero of 'UsageTotals'.+-- | The zero of t'UsageTotals'. emptyUsageTotals :: UsageTotals emptyUsageTotals = UsageTotals 0 0 0 0 Nothing 0 @@ -118,7 +118,7 @@ -- | Knobs for an evaluation run. data EvalConfig = EvalConfig- { -- | max examples evaluated at once (forced to @>= 1@ by 'evaluate')+ { -- | max examples evaluated at once (forced to @>= 1@ by 'Shikumi.Eval.Evaluate.evaluate') concurrency :: !Int, failurePolicy :: !FailurePolicy, -- | wall-clock budget per example in milliseconds; 'Nothing' means no timeout
src/Shikumi/Eval/Types.hs view
@@ -1,15 +1,15 @@ -- | The evaluation data model — MasterPlan integration point #5. The optimizer--- (@docs/plans/10-optimizer-framework.md@) and the CLI--- (@docs/plans/12-cli-and-developer-experience.md@) /consume/ these types and+-- (@docs\/plans\/10-optimizer-framework.md@) and the CLI+-- (@docs\/plans\/12-cli-and-developer-experience.md@) /consume/ these types and -- must not redefine them. ----- A 'Score' is a 'Double' clamped to the closed interval @[0, 1]@ at+-- A t'Score' is a 'Double' clamped to the closed interval @[0, 1]@ at -- construction, so downstream ranking code can assume @0 <= s <= 1@ without--- re-checking (a @Bool@ metric maps @True -> 1@, @False -> 0@). An 'Example' is--- one labelled datum (input + expected output); a 'Dataset' is a list of them. A--- 'Prediction' is a program's actual output for one example, carrying a primary+-- re-checking (a @Bool@ metric maps @True -> 1@, @False -> 0@). An t'Example' is+-- one labelled datum (input + expected output); a t'Dataset' is a list of them. A+-- t'Prediction' is a program's actual output for one example, carrying a primary -- output plus the non-empty set of sampled outputs (a single-output program--- yields a one-element 'Prediction'), so agreement-rewarding metrics+-- yields a one-element t'Prediction'), so agreement-rewarding metrics -- (majority-vote, ensemble) can inspect every sample. module Shikumi.Eval.Types ( -- * Scores@@ -68,7 +68,7 @@ } deriving stock (Eq, Show) --- | Build an 'Example' from an input and its expected output.+-- | Build an t'Example' from an input and its expected output. example :: i -> o -> Example i o example = Example @@ -76,7 +76,7 @@ newtype Dataset i o = Dataset [Example i o] deriving stock (Eq, Show) --- | Build a 'Dataset' from a list of examples.+-- | Build a t'Dataset' from a list of examples. dataset :: [Example i o] -> Dataset i o dataset = Dataset
src/Shikumi/Eval/Usage.hs view
@@ -6,7 +6,7 @@ -- EP-6's @cachedLLM@ and EP-7's @tracedLLM@ use. Because @LLM.complete@ returns -- the full 'Response' (which already carries 'Baikai.Usage.Usage' and -- 'Baikai.Cost.Cost'), no substrate hook or writer effect is needed; the--- interposed handler reads usage straight off each response. A shared 'IORef'+-- interposed handler reads usage straight off each response. A shared 'Data.IORef.IORef' -- mutated with 'atomicModifyIORef'' makes the accumulation safe under the bounded -- concurrency @evaluate@ uses (each pooled worker gets a cloned env that still -- references the one ref).
test/GoldenSpec.hs view
@@ -1,7 +1,7 @@ {-# LANGUAGE RankNTypes #-} -- | A golden test for a stub program, run deterministically and offline under a--- constant mock LM. The committed @test/golden/qa-program.golden@ pins the+-- constant mock LM. The committed @test\/golden\/qa-program.golden@ pins the -- transcript; the plan records the fail-before/pass-after demonstration. module GoldenSpec (tests) where
test/MetricSpec.hs view
@@ -1,5 +1,5 @@ -- | Unit tests for the pure metrics and combinators: exact match, normalized--- string similarity (identical / near / disjoint), and the @threshold@,+-- string similarity (identical \/ near \/ disjoint), and the @threshold@, -- @weightedMean@, and @invert@ combinators. All offline and deterministic. module MetricSpec (tests) where @@ -50,7 +50,7 @@ testCase "threshold fails a low score" $ threshold 0.5 (constMetric 0.3) () (prediction ()) @?= scoreZero, testCase "weightedMean averages by weight" $- -- (1*0.2 + 3*0.6) / (1+3) = 2.0/4 = 0.5 (within float tolerance)+ -- (1*0.2 + 3*0.6) \/ (1+3) = 2.0\/4 = 0.5 (within float tolerance) let s = unScore (weightedMean [(1, constMetric 0.2), (3, constMetric 0.6)] () (prediction ())) in assertBool ("expected ~0.5, got " <> show s) (abs (s - 0.5) < 1e-9), testCase "weightedMean of empty is zero" $