packages feed

shikumi-eval 0.3.0.0 → 0.3.0.1

raw patch · 12 files changed

+102/−85 lines, 12 filesdep ~baikaidep ~effectfuldep ~shikumi-evalPVP ok

version bump matches the API change (PVP)

Dependency ranges changed: baikai, effectful, shikumi-eval

API changes (from Hackage documentation)

Files

CHANGELOG.md view
@@ -2,6 +2,10 @@  ## Unreleased +## 0.3.0.1 — 2026-10-05++- Move the dependency on `mori://shinzui/baikai/packages/baikai` to `>=0.7.1.0 && <0.8` and widen the `effectful` bound to `>=2.6 && <2.8`, so both effectful 2.6 and 2.7 are supported (effectful 2.7 needs `baikai-effectful` 0.4.0.2, effectful 2.6 needs 0.4.0.1). Bounds only; no source changed.+ ## 0.3.0.0 — 2026-09-08  - Raise the internal `shikumi` bound to `^>=0.4.0.0` for the breaking core release.
shikumi-eval.cabal view
@@ -1,8 +1,8 @@-cabal-version:   3.4-name:            shikumi-eval-version:         0.3.0.0-synopsis:        Typed evaluation framework for shikumi LM programs (EP-8)-category:        AI+cabal-version: 3.4+name: shikumi-eval+version: 0.3.0.1+synopsis: Typed evaluation framework for shikumi LM programs (EP-8)+category: AI description:   The evaluation framework for shikumi: the owned data model (@Example@,   @Prediction@, @Dataset@, @Metric@, @Score@, @Report@ — MasterPlan integration@@ -11,20 +11,26 @@   parallelism and per-example error boundaries, and golden testing that pins a   program's behaviour deterministically under a mock or replayed LM. -license:         BSD-3-Clause-author:          Nadeem Bitar-maintainer:      nadeem@gmail.com-build-type:      Simple+license: BSD-3-Clause+author: Nadeem Bitar+maintainer: nadeem@gmail.com+build-type: Simple extra-doc-files: CHANGELOG.md  common common-options   ghc-options:-    -Wall -Wcompat -Widentities -Wincomplete-uni-patterns-    -Wincomplete-record-updates -Wredundant-constraints-    -fhide-source-paths -Wmissing-export-lists -Wpartial-fields+    -Wall+    -Wcompat+    -Widentities+    -Wincomplete-uni-patterns+    -Wincomplete-record-updates+    -Wredundant-constraints+    -fhide-source-paths+    -Wmissing-export-lists+    -Wpartial-fields     -Wmissing-deriving-strategies -  default-language:   GHC2024+  default-language: GHC2024   default-extensions:     DeriveAnyClass     DuplicateRecordFields@@ -32,8 +38,8 @@     OverloadedStrings  library-  import:          common-options-  hs-source-dirs:  src+  import: common-options+  hs-source-dirs: src   exposed-modules:     Shikumi.Eval     Shikumi.Eval.Embedding@@ -45,26 +51,29 @@     Shikumi.Eval.Usage    build-depends:-    , aeson         >=2.2      && <2.3-    , baikai        >=0.7.0.0  && <0.8-    , base          >=4.20     && <5-    , bytestring    >=0.11     && <0.13-    , containers    >=0.6      && <0.9-    , effectful     >=2.5      && <2.7-    , generic-lens  >=2.2      && <2.4-    , lens          ^>=5.3-    , shikumi       ^>=0.4.0.0-    , tasty         >=1.4      && <1.6-    , tasty-golden  >=2.3      && <2.4-    , text          ^>=2.1-    , vector        >=0.13     && <0.14+    aeson >=2.2 && <2.3,+    baikai >=0.7.1.0 && <0.8,+    base >=4.20 && <5,+    bytestring >=0.11 && <0.13,+    containers >=0.6 && <0.9,+    effectful >=2.6 && <2.8,+    generic-lens >=2.2 && <2.4,+    lens ^>=5.3,+    shikumi ^>=0.4.0.0,+    tasty >=1.4 && <1.6,+    tasty-golden >=2.3 && <2.4,+    text ^>=2.1,+    vector >=0.13 && <0.14,  test-suite shikumi-eval-test-  import:         common-options-  type:           exitcode-stdio-1.0+  import: common-options+  type: exitcode-stdio-1.0   hs-source-dirs: test-  main-is:        Main.hs-  ghc-options:    -threaded -with-rtsopts=-N+  main-is: Main.hs+  ghc-options:+    -threaded+    -with-rtsopts=-N+   other-modules:     DocSpec     EmbeddingSpec@@ -78,16 +87,16 @@     UsageSpec    build-depends:-    , aeson-    , baikai        >=0.7.0.0  && <0.8-    , base-    , effectful-    , generic-lens-    , lens-    , shikumi       ^>=0.4.0.0-    , shikumi-eval  ^>=0.3.0.0-    , tasty-    , tasty-golden-    , tasty-hunit-    , text-    , vector+    aeson,+    baikai >=0.7.1.0 && <0.8,+    base,+    effectful,+    generic-lens,+    lens,+    shikumi ^>=0.4.0.0,+    shikumi-eval ^>=0.3.0.1,+    tasty,+    tasty-golden,+    tasty-hunit,+    text,+    vector,
src/Shikumi/Eval.hs view
@@ -1,8 +1,8 @@ -- | The shikumi evaluation framework — one import for the whole public surface -- (MasterPlan integration point #5). Re-exports the owned data model--- ('Shikumi.Eval.Types'), the metrics ('Shikumi.Eval.Metric'), the report--- ('Shikumi.Eval.Report'), the 'evaluate' runner ('Shikumi.Eval.Evaluate'), and--- golden testing ('Shikumi.Eval.Golden').+-- ("Shikumi.Eval.Types"), the metrics ("Shikumi.Eval.Metric"), the report+-- ("Shikumi.Eval.Report"), the 'evaluate' runner ("Shikumi.Eval.Evaluate"), and+-- golden testing ("Shikumi.Eval.Golden"). -- -- Worked example: measure a @Question -> Answer@ program over a small typed -- dataset with the exact-match metric, then read the aggregate score. (The@@ -12,15 +12,19 @@ -- @ -- import Shikumi.Eval ----- qaData :: 'Dataset' Question Answer+-- qaData :: t'Dataset' Question Answer -- qaData = 'dataset' --   [ 'example' (Question \"What is 2+2?\")     (Answer \"4\") --   , 'example' (Question \"Capital of France?\") (Answer \"Paris\") --   ] -- -- -- inside an Eff stack with (LLM, Concurrent, Error ShikumiError, IOE):--- run :: ('LLM' :> es, 'Concurrent' :> es, 'Error' 'ShikumiError' :> es, 'IOE' :> es)---     => 'Program' Question Answer -> Eff es Double+-- run ::+--   ( t'Shikumi.LLM.LLM' :> es+--   , 'Effectful.Concurrent.Concurrent' :> es+--   , 'Effectful.Error.Static.Error' t'Shikumi.Error.ShikumiError' :> es+--   , 'Effectful.IOE' :> es+--   ) => t'Shikumi.Program.Program' Question Answer -> Eff es Double -- run prog = do --   report <- 'evaluatePure' qaData 'exactMatch' prog --   pure ('aggregateScore' report)
src/Shikumi/Eval/Embedding.hs view
@@ -2,10 +2,10 @@  -- | A real embeddings interpreter for the existing 'Embedding' effect (EP-15). ----- 'Shikumi.Eval.Metric' ships only the pure 'Shikumi.Eval.Metric.runEmbedding'+-- "Shikumi.Eval.Metric" ships only the pure 'Shikumi.Eval.Metric.runEmbedding' -- (a @Text -> Vector Double@ table), so @semanticSimilarity@ has no way to call a -- real backend. This module adds interpreters that drive the upstream--- 'Baikai.Embedding' client (an OpenAI-compatible @\/v1\/embeddings@ endpoint),+-- "Baikai.Embedding" client (an OpenAI-compatible @\/v1\/embeddings@ endpoint), -- so @semanticSimilarity@ runs end to end. The @Embedding@ effect itself is -- unchanged (integration point #5); only new interpreters are added. --
src/Shikumi/Eval/Evaluate.hs view
@@ -1,10 +1,10 @@--- | The core deliverable: @evaluate@ runs a 'Program' over a typed 'Dataset',+-- | The core deliverable: @evaluate@ runs a t'Program' over a typed 'Dataset', -- scores each example with a metric, and returns a 'Report' summarising the -- aggregate score, the per-example breakdown, token usage, cost, and latency. -- -- Examples run with __bounded__ concurrency ('pooledForConcurrentlyN', worker -- count from 'concurrency'), which preserves dataset order in the result. Each--- example sits inside a per-example error boundary: a 'ShikumiError' from+-- example sits inside a per-example error boundary: a t'ShikumiError' from -- 'runProgram' becomes a 'ProgramError', one from an effectful metric becomes a -- 'MetricError', and the 'FailurePolicy' decides whether that example is scored -- and the run continues ('FailScore') or the whole run aborts ('FailAbort').@@ -154,7 +154,7 @@       FailAbort -> throwError e       FailScore s -> pure (s, Just reason) --- | Build a 'Prediction' by running the program @numSamples@ times (at least+-- | Build a t'Prediction' by running the program @numSamples@ times (at least -- once); a single sample yields a one-element prediction. buildPrediction ::   (LLM :> es, Error ShikumiError :> es) =>@@ -173,7 +173,7 @@   where     n = max 1 (numSamples cfg) --- | Run an action, catching a 'ShikumiError' into a 'Left'.+-- | Run an action, catching a t'ShikumiError' into a 'Left'. tryShikumi :: (Error ShikumiError :> es) => Eff es a -> Eff es (Either ShikumiError a) tryShikumi act = (Right <$> act) `catchError` \_ e -> pure (Left e) 
src/Shikumi/Eval/Golden.hs view
@@ -6,15 +6,15 @@ -- stack under a /mock or replayed/ LM (never a live one), so the only thing that -- can change the output is a genuine change in program behaviour. ----- 'goldenProgram' compares a per-example transcript (one @index\\t<output>@ line--- per example, in dataset order); 'goldenReport' compares the rendered 'Report'+-- 'goldenProgram' compares a per-example transcript (one @index\\t\<output\>@ line+-- per example, in dataset order); 'goldenReport' compares the rendered t'Shikumi.Eval.Report.Report' -- from 'evaluate'. Both are built on @tasty-golden@'s 'goldenVsString' — -- regenerate the golden file with @cabal test --test-options=--accept@. -- -- The runner is a single rank-2 function @forall a. Eff es a -> IO a@, so the -- helper stays agnostic to the exact effect stack: the caller picks a concrete -- @es@ (e.g. @'[LLM, Error ShikumiError, IOE]@) and a function collapsing it to--- 'IO' (throwing on a 'ShikumiError').+-- 'IO' (throwing on a t'ShikumiError'). module Shikumi.Eval.Golden   ( goldenProgram,     goldenReport,@@ -49,7 +49,7 @@   TestName ->   -- | golden file path (relative to the package directory)   FilePath ->-  -- | how to run the effect stack (mock/replay), throwing on a 'ShikumiError'+  -- | how to run the effect stack (mock/replay), throwing on a t'ShikumiError'   (forall a. Eff es a -> IO a) ->   Dataset i o ->   Program i o ->@@ -62,7 +62,7 @@     outs <- runner (traverse (runProgram prog) inputs)     pure (encodeText (renderTranscript (zip [0 ..] (map render outs)))) --- | Like 'goldenProgram' but compares the rendered 'Report' produced by+-- | Like 'goldenProgram' but compares the rendered t'Shikumi.Eval.Report.Report' produced by -- 'evaluate' with the given metric, rather than a raw transcript. goldenReport ::   (LLM :> es, Concurrent :> es, Error ShikumiError :> es, Time :> es, Prim :> es) =>@@ -78,7 +78,7 @@     report <- runner (evaluate ds metric prog)     pure (encodeText (renderReportText report)) --- | One @index\\t<rendered output>@ line per example, in dataset order.+-- | One @index\\t\<rendered output\>@ line per example, in dataset order. renderTranscript :: [(Int, Text)] -> Text renderTranscript = T.unlines . map (\(i, t) -> T.pack (show i) <> "\t" <> t) 
src/Shikumi/Eval/Metric.hs view
@@ -232,7 +232,7 @@ instance Validatable Grade  -- | The judge program: a single structured-output predictor that returns a--- 'Grade'. The @rubricText@ becomes the signature instruction.+-- t'Grade'. The @rubricText@ becomes the signature instruction. judgeProgram :: Text -> Program JudgeInput Grade judgeProgram rubricText = predict (judgeSig rubricText) @@ -247,7 +247,7 @@ -- | LLM-as-judge: ask a model to grade the prediction against the expected -- output. The rubric is supplied as text; the model returns a numeric grade, -- decoded through the structured-output path. A decode failure surfaces as a--- 'ShikumiError' (the per-example boundary in @evaluate@ turns it into a+-- t'ShikumiError' (the per-example boundary in @evaluate@ turns it into a -- 'Shikumi.Eval.Report.MetricError'). Threads @Error ShikumiError@ because -- 'runProgram' does (EP-4): decode failures throw typed errors. modelJudge ::
src/Shikumi/Eval/Report.hs view
@@ -1,11 +1,11 @@--- | The evaluation report and its aggregation. A 'Report' summarises a run over+-- | The evaluation report and its aggregation. A t'Report' summarises a run over -- a dataset: the mean per-example score, pass/fail counts, summed token usage and--- cost, summed per-example latency, and a per-example breakdown ('ExampleResult',+-- cost, summed per-example latency, and a per-example breakdown ([ExampleResult]("Shikumi.Eval.Report#t:ExampleResult"), -- retained in dataset order for failure analysis). 'mkReport' is the pure aggregator; -- 'renderReportText' is a deterministic human-readable rendering reused by the--- CLI (@docs/plans/12-cli-and-developer-experience.md@) and by golden tests.+-- CLI (@docs\/plans\/12-cli-and-developer-experience.md@) and by golden tests. ----- 'EvalConfig' carries the run knobs (bounded 'concurrency', the 'FailurePolicy'+-- t'EvalConfig' carries the run knobs (bounded 'concurrency', the 'FailurePolicy' -- for per-example errors, optional per-example timeout, and 'numSamples' per example -- for multi-sample metrics); 'defaultEvalConfig' scores failures @0@ and keeps going. module Shikumi.Eval.Report@@ -47,7 +47,7 @@  -- | The reason an example did not complete normally. data FailureReason-  = -- | @runProgram@ threw a 'Shikumi.Error.ShikumiError' (rendered to text)+  = -- | @runProgram@ threw a t'Shikumi.Error.ShikumiError' (rendered to text)     ProgramError !Text   | -- | the metric itself failed (an effectful metric raised an error)     MetricError !Text@@ -84,7 +84,7 @@   }   deriving stock (Eq, Show) --- | The zero of 'UsageTotals'.+-- | The zero of t'UsageTotals'. emptyUsageTotals :: UsageTotals emptyUsageTotals = UsageTotals 0 0 0 0 Nothing 0 @@ -118,7 +118,7 @@  -- | Knobs for an evaluation run. data EvalConfig = EvalConfig-  { -- | max examples evaluated at once (forced to @>= 1@ by 'evaluate')+  { -- | max examples evaluated at once (forced to @>= 1@ by 'Shikumi.Eval.Evaluate.evaluate')     concurrency :: !Int,     failurePolicy :: !FailurePolicy,     -- | wall-clock budget per example in milliseconds; 'Nothing' means no timeout
src/Shikumi/Eval/Types.hs view
@@ -1,15 +1,15 @@ -- | The evaluation data model — MasterPlan integration point #5. The optimizer--- (@docs/plans/10-optimizer-framework.md@) and the CLI--- (@docs/plans/12-cli-and-developer-experience.md@) /consume/ these types and+-- (@docs\/plans\/10-optimizer-framework.md@) and the CLI+-- (@docs\/plans\/12-cli-and-developer-experience.md@) /consume/ these types and -- must not redefine them. ----- A 'Score' is a 'Double' clamped to the closed interval @[0, 1]@ at+-- A t'Score' is a 'Double' clamped to the closed interval @[0, 1]@ at -- construction, so downstream ranking code can assume @0 <= s <= 1@ without--- re-checking (a @Bool@ metric maps @True -> 1@, @False -> 0@). An 'Example' is--- one labelled datum (input + expected output); a 'Dataset' is a list of them. A--- 'Prediction' is a program's actual output for one example, carrying a primary+-- re-checking (a @Bool@ metric maps @True -> 1@, @False -> 0@). An t'Example' is+-- one labelled datum (input + expected output); a t'Dataset' is a list of them. A+-- t'Prediction' is a program's actual output for one example, carrying a primary -- output plus the non-empty set of sampled outputs (a single-output program--- yields a one-element 'Prediction'), so agreement-rewarding metrics+-- yields a one-element t'Prediction'), so agreement-rewarding metrics -- (majority-vote, ensemble) can inspect every sample. module Shikumi.Eval.Types   ( -- * Scores@@ -68,7 +68,7 @@   }   deriving stock (Eq, Show) --- | Build an 'Example' from an input and its expected output.+-- | Build an t'Example' from an input and its expected output. example :: i -> o -> Example i o example = Example @@ -76,7 +76,7 @@ newtype Dataset i o = Dataset [Example i o]   deriving stock (Eq, Show) --- | Build a 'Dataset' from a list of examples.+-- | Build a t'Dataset' from a list of examples. dataset :: [Example i o] -> Dataset i o dataset = Dataset 
src/Shikumi/Eval/Usage.hs view
@@ -6,7 +6,7 @@ -- EP-6's @cachedLLM@ and EP-7's @tracedLLM@ use. Because @LLM.complete@ returns -- the full 'Response' (which already carries 'Baikai.Usage.Usage' and -- 'Baikai.Cost.Cost'), no substrate hook or writer effect is needed; the--- interposed handler reads usage straight off each response. A shared 'IORef'+-- interposed handler reads usage straight off each response. A shared 'Data.IORef.IORef' -- mutated with 'atomicModifyIORef'' makes the accumulation safe under the bounded -- concurrency @evaluate@ uses (each pooled worker gets a cloned env that still -- references the one ref).
test/GoldenSpec.hs view
@@ -1,7 +1,7 @@ {-# LANGUAGE RankNTypes #-}  -- | A golden test for a stub program, run deterministically and offline under a--- constant mock LM. The committed @test/golden/qa-program.golden@ pins the+-- constant mock LM. The committed @test\/golden\/qa-program.golden@ pins the -- transcript; the plan records the fail-before/pass-after demonstration. module GoldenSpec (tests) where 
test/MetricSpec.hs view
@@ -1,5 +1,5 @@ -- | Unit tests for the pure metrics and combinators: exact match, normalized--- string similarity (identical / near / disjoint), and the @threshold@,+-- string similarity (identical \/ near \/ disjoint), and the @threshold@, -- @weightedMean@, and @invert@ combinators. All offline and deterministic. module MetricSpec (tests) where @@ -50,7 +50,7 @@       testCase "threshold fails a low score" $         threshold 0.5 (constMetric 0.3) () (prediction ()) @?= scoreZero,       testCase "weightedMean averages by weight" $-        -- (1*0.2 + 3*0.6) / (1+3) = 2.0/4 = 0.5 (within float tolerance)+        -- (1*0.2 + 3*0.6) \/ (1+3) = 2.0\/4 = 0.5 (within float tolerance)         let s = unScore (weightedMean [(1, constMetric 0.2), (3, constMetric 0.6)] () (prediction ()))          in assertBool ("expected ~0.5, got " <> show s) (abs (s - 0.5) < 1e-9),       testCase "weightedMean of empty is zero" $