diff --git a/CHANGELOG.md b/CHANGELOG.md
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -2,6 +2,10 @@
 
 ## Unreleased
 
+## 0.3.0.1 — 2026-10-05
+
+- Move the dependency on `mori://shinzui/baikai/packages/baikai` to `>=0.7.1.0 && <0.8` and widen the `effectful` bound to `>=2.6 && <2.8`, so both effectful 2.6 and 2.7 are supported (effectful 2.7 needs `baikai-effectful` 0.4.0.2, effectful 2.6 needs 0.4.0.1). Bounds only; no source changed.
+
 ## 0.3.0.0 — 2026-09-08
 
 - Raise the internal `shikumi` bound to `^>=0.4.0.0` for the breaking core release.
diff --git a/shikumi-eval.cabal b/shikumi-eval.cabal
--- a/shikumi-eval.cabal
+++ b/shikumi-eval.cabal
@@ -1,8 +1,8 @@
-cabal-version:   3.4
-name:            shikumi-eval
-version:         0.3.0.0
-synopsis:        Typed evaluation framework for shikumi LM programs (EP-8)
-category:        AI
+cabal-version: 3.4
+name: shikumi-eval
+version: 0.3.0.1
+synopsis: Typed evaluation framework for shikumi LM programs (EP-8)
+category: AI
 description:
   The evaluation framework for shikumi: the owned data model (@Example@,
   @Prediction@, @Dataset@, @Metric@, @Score@, @Report@ — MasterPlan integration
@@ -11,20 +11,26 @@
   parallelism and per-example error boundaries, and golden testing that pins a
   program's behaviour deterministically under a mock or replayed LM.
 
-license:         BSD-3-Clause
-author:          Nadeem Bitar
-maintainer:      nadeem@gmail.com
-build-type:      Simple
+license: BSD-3-Clause
+author: Nadeem Bitar
+maintainer: nadeem@gmail.com
+build-type: Simple
 extra-doc-files: CHANGELOG.md
 
 common common-options
   ghc-options:
-    -Wall -Wcompat -Widentities -Wincomplete-uni-patterns
-    -Wincomplete-record-updates -Wredundant-constraints
-    -fhide-source-paths -Wmissing-export-lists -Wpartial-fields
+    -Wall
+    -Wcompat
+    -Widentities
+    -Wincomplete-uni-patterns
+    -Wincomplete-record-updates
+    -Wredundant-constraints
+    -fhide-source-paths
+    -Wmissing-export-lists
+    -Wpartial-fields
     -Wmissing-deriving-strategies
 
-  default-language:   GHC2024
+  default-language: GHC2024
   default-extensions:
     DeriveAnyClass
     DuplicateRecordFields
@@ -32,8 +38,8 @@
     OverloadedStrings
 
 library
-  import:          common-options
-  hs-source-dirs:  src
+  import: common-options
+  hs-source-dirs: src
   exposed-modules:
     Shikumi.Eval
     Shikumi.Eval.Embedding
@@ -45,26 +51,29 @@
     Shikumi.Eval.Usage
 
   build-depends:
-    , aeson         >=2.2      && <2.3
-    , baikai        >=0.7.0.0  && <0.8
-    , base          >=4.20     && <5
-    , bytestring    >=0.11     && <0.13
-    , containers    >=0.6      && <0.9
-    , effectful     >=2.5      && <2.7
-    , generic-lens  >=2.2      && <2.4
-    , lens          ^>=5.3
-    , shikumi       ^>=0.4.0.0
-    , tasty         >=1.4      && <1.6
-    , tasty-golden  >=2.3      && <2.4
-    , text          ^>=2.1
-    , vector        >=0.13     && <0.14
+    aeson >=2.2 && <2.3,
+    baikai >=0.7.1.0 && <0.8,
+    base >=4.20 && <5,
+    bytestring >=0.11 && <0.13,
+    containers >=0.6 && <0.9,
+    effectful >=2.6 && <2.8,
+    generic-lens >=2.2 && <2.4,
+    lens ^>=5.3,
+    shikumi ^>=0.4.0.0,
+    tasty >=1.4 && <1.6,
+    tasty-golden >=2.3 && <2.4,
+    text ^>=2.1,
+    vector >=0.13 && <0.14,
 
 test-suite shikumi-eval-test
-  import:         common-options
-  type:           exitcode-stdio-1.0
+  import: common-options
+  type: exitcode-stdio-1.0
   hs-source-dirs: test
-  main-is:        Main.hs
-  ghc-options:    -threaded -with-rtsopts=-N
+  main-is: Main.hs
+  ghc-options:
+    -threaded
+    -with-rtsopts=-N
+
   other-modules:
     DocSpec
     EmbeddingSpec
@@ -78,16 +87,16 @@
     UsageSpec
 
   build-depends:
-    , aeson
-    , baikai        >=0.7.0.0  && <0.8
-    , base
-    , effectful
-    , generic-lens
-    , lens
-    , shikumi       ^>=0.4.0.0
-    , shikumi-eval  ^>=0.3.0.0
-    , tasty
-    , tasty-golden
-    , tasty-hunit
-    , text
-    , vector
+    aeson,
+    baikai >=0.7.1.0 && <0.8,
+    base,
+    effectful,
+    generic-lens,
+    lens,
+    shikumi ^>=0.4.0.0,
+    shikumi-eval ^>=0.3.0.1,
+    tasty,
+    tasty-golden,
+    tasty-hunit,
+    text,
+    vector,
diff --git a/src/Shikumi/Eval.hs b/src/Shikumi/Eval.hs
--- a/src/Shikumi/Eval.hs
+++ b/src/Shikumi/Eval.hs
@@ -1,8 +1,8 @@
 -- | The shikumi evaluation framework — one import for the whole public surface
 -- (MasterPlan integration point #5). Re-exports the owned data model
--- ('Shikumi.Eval.Types'), the metrics ('Shikumi.Eval.Metric'), the report
--- ('Shikumi.Eval.Report'), the 'evaluate' runner ('Shikumi.Eval.Evaluate'), and
--- golden testing ('Shikumi.Eval.Golden').
+-- ("Shikumi.Eval.Types"), the metrics ("Shikumi.Eval.Metric"), the report
+-- ("Shikumi.Eval.Report"), the 'evaluate' runner ("Shikumi.Eval.Evaluate"), and
+-- golden testing ("Shikumi.Eval.Golden").
 --
 -- Worked example: measure a @Question -> Answer@ program over a small typed
 -- dataset with the exact-match metric, then read the aggregate score. (The
@@ -12,15 +12,19 @@
 -- @
 -- import Shikumi.Eval
 --
--- qaData :: 'Dataset' Question Answer
+-- qaData :: t'Dataset' Question Answer
 -- qaData = 'dataset'
 --   [ 'example' (Question \"What is 2+2?\")     (Answer \"4\")
 --   , 'example' (Question \"Capital of France?\") (Answer \"Paris\")
 --   ]
 --
 -- -- inside an Eff stack with (LLM, Concurrent, Error ShikumiError, IOE):
--- run :: ('LLM' :> es, 'Concurrent' :> es, 'Error' 'ShikumiError' :> es, 'IOE' :> es)
---     => 'Program' Question Answer -> Eff es Double
+-- run ::
+--   ( t'Shikumi.LLM.LLM' :> es
+--   , 'Effectful.Concurrent.Concurrent' :> es
+--   , 'Effectful.Error.Static.Error' t'Shikumi.Error.ShikumiError' :> es
+--   , 'Effectful.IOE' :> es
+--   ) => t'Shikumi.Program.Program' Question Answer -> Eff es Double
 -- run prog = do
 --   report <- 'evaluatePure' qaData 'exactMatch' prog
 --   pure ('aggregateScore' report)
diff --git a/src/Shikumi/Eval/Embedding.hs b/src/Shikumi/Eval/Embedding.hs
--- a/src/Shikumi/Eval/Embedding.hs
+++ b/src/Shikumi/Eval/Embedding.hs
@@ -2,10 +2,10 @@
 
 -- | A real embeddings interpreter for the existing 'Embedding' effect (EP-15).
 --
--- 'Shikumi.Eval.Metric' ships only the pure 'Shikumi.Eval.Metric.runEmbedding'
+-- "Shikumi.Eval.Metric" ships only the pure 'Shikumi.Eval.Metric.runEmbedding'
 -- (a @Text -> Vector Double@ table), so @semanticSimilarity@ has no way to call a
 -- real backend. This module adds interpreters that drive the upstream
--- 'Baikai.Embedding' client (an OpenAI-compatible @\/v1\/embeddings@ endpoint),
+-- "Baikai.Embedding" client (an OpenAI-compatible @\/v1\/embeddings@ endpoint),
 -- so @semanticSimilarity@ runs end to end. The @Embedding@ effect itself is
 -- unchanged (integration point #5); only new interpreters are added.
 --
diff --git a/src/Shikumi/Eval/Evaluate.hs b/src/Shikumi/Eval/Evaluate.hs
--- a/src/Shikumi/Eval/Evaluate.hs
+++ b/src/Shikumi/Eval/Evaluate.hs
@@ -1,10 +1,10 @@
--- | The core deliverable: @evaluate@ runs a 'Program' over a typed 'Dataset',
+-- | The core deliverable: @evaluate@ runs a t'Program' over a typed 'Dataset',
 -- scores each example with a metric, and returns a 'Report' summarising the
 -- aggregate score, the per-example breakdown, token usage, cost, and latency.
 --
 -- Examples run with __bounded__ concurrency ('pooledForConcurrentlyN', worker
 -- count from 'concurrency'), which preserves dataset order in the result. Each
--- example sits inside a per-example error boundary: a 'ShikumiError' from
+-- example sits inside a per-example error boundary: a t'ShikumiError' from
 -- 'runProgram' becomes a 'ProgramError', one from an effectful metric becomes a
 -- 'MetricError', and the 'FailurePolicy' decides whether that example is scored
 -- and the run continues ('FailScore') or the whole run aborts ('FailAbort').
@@ -154,7 +154,7 @@
       FailAbort -> throwError e
       FailScore s -> pure (s, Just reason)
 
--- | Build a 'Prediction' by running the program @numSamples@ times (at least
+-- | Build a t'Prediction' by running the program @numSamples@ times (at least
 -- once); a single sample yields a one-element prediction.
 buildPrediction ::
   (LLM :> es, Error ShikumiError :> es) =>
@@ -173,7 +173,7 @@
   where
     n = max 1 (numSamples cfg)
 
--- | Run an action, catching a 'ShikumiError' into a 'Left'.
+-- | Run an action, catching a t'ShikumiError' into a 'Left'.
 tryShikumi :: (Error ShikumiError :> es) => Eff es a -> Eff es (Either ShikumiError a)
 tryShikumi act = (Right <$> act) `catchError` \_ e -> pure (Left e)
 
diff --git a/src/Shikumi/Eval/Golden.hs b/src/Shikumi/Eval/Golden.hs
--- a/src/Shikumi/Eval/Golden.hs
+++ b/src/Shikumi/Eval/Golden.hs
@@ -6,15 +6,15 @@
 -- stack under a /mock or replayed/ LM (never a live one), so the only thing that
 -- can change the output is a genuine change in program behaviour.
 --
--- 'goldenProgram' compares a per-example transcript (one @index\\t<output>@ line
--- per example, in dataset order); 'goldenReport' compares the rendered 'Report'
+-- 'goldenProgram' compares a per-example transcript (one @index\\t\<output\>@ line
+-- per example, in dataset order); 'goldenReport' compares the rendered t'Shikumi.Eval.Report.Report'
 -- from 'evaluate'. Both are built on @tasty-golden@'s 'goldenVsString' —
 -- regenerate the golden file with @cabal test --test-options=--accept@.
 --
 -- The runner is a single rank-2 function @forall a. Eff es a -> IO a@, so the
 -- helper stays agnostic to the exact effect stack: the caller picks a concrete
 -- @es@ (e.g. @'[LLM, Error ShikumiError, IOE]@) and a function collapsing it to
--- 'IO' (throwing on a 'ShikumiError').
+-- 'IO' (throwing on a t'ShikumiError').
 module Shikumi.Eval.Golden
   ( goldenProgram,
     goldenReport,
@@ -49,7 +49,7 @@
   TestName ->
   -- | golden file path (relative to the package directory)
   FilePath ->
-  -- | how to run the effect stack (mock/replay), throwing on a 'ShikumiError'
+  -- | how to run the effect stack (mock/replay), throwing on a t'ShikumiError'
   (forall a. Eff es a -> IO a) ->
   Dataset i o ->
   Program i o ->
@@ -62,7 +62,7 @@
     outs <- runner (traverse (runProgram prog) inputs)
     pure (encodeText (renderTranscript (zip [0 ..] (map render outs))))
 
--- | Like 'goldenProgram' but compares the rendered 'Report' produced by
+-- | Like 'goldenProgram' but compares the rendered t'Shikumi.Eval.Report.Report' produced by
 -- 'evaluate' with the given metric, rather than a raw transcript.
 goldenReport ::
   (LLM :> es, Concurrent :> es, Error ShikumiError :> es, Time :> es, Prim :> es) =>
@@ -78,7 +78,7 @@
     report <- runner (evaluate ds metric prog)
     pure (encodeText (renderReportText report))
 
--- | One @index\\t<rendered output>@ line per example, in dataset order.
+-- | One @index\\t\<rendered output\>@ line per example, in dataset order.
 renderTranscript :: [(Int, Text)] -> Text
 renderTranscript = T.unlines . map (\(i, t) -> T.pack (show i) <> "\t" <> t)
 
diff --git a/src/Shikumi/Eval/Metric.hs b/src/Shikumi/Eval/Metric.hs
--- a/src/Shikumi/Eval/Metric.hs
+++ b/src/Shikumi/Eval/Metric.hs
@@ -232,7 +232,7 @@
 instance Validatable Grade
 
 -- | The judge program: a single structured-output predictor that returns a
--- 'Grade'. The @rubricText@ becomes the signature instruction.
+-- t'Grade'. The @rubricText@ becomes the signature instruction.
 judgeProgram :: Text -> Program JudgeInput Grade
 judgeProgram rubricText = predict (judgeSig rubricText)
 
@@ -247,7 +247,7 @@
 -- | LLM-as-judge: ask a model to grade the prediction against the expected
 -- output. The rubric is supplied as text; the model returns a numeric grade,
 -- decoded through the structured-output path. A decode failure surfaces as a
--- 'ShikumiError' (the per-example boundary in @evaluate@ turns it into a
+-- t'ShikumiError' (the per-example boundary in @evaluate@ turns it into a
 -- 'Shikumi.Eval.Report.MetricError'). Threads @Error ShikumiError@ because
 -- 'runProgram' does (EP-4): decode failures throw typed errors.
 modelJudge ::
diff --git a/src/Shikumi/Eval/Report.hs b/src/Shikumi/Eval/Report.hs
--- a/src/Shikumi/Eval/Report.hs
+++ b/src/Shikumi/Eval/Report.hs
@@ -1,11 +1,11 @@
--- | The evaluation report and its aggregation. A 'Report' summarises a run over
+-- | The evaluation report and its aggregation. A t'Report' summarises a run over
 -- a dataset: the mean per-example score, pass/fail counts, summed token usage and
--- cost, summed per-example latency, and a per-example breakdown ('ExampleResult',
+-- cost, summed per-example latency, and a per-example breakdown ([ExampleResult]("Shikumi.Eval.Report#t:ExampleResult"),
 -- retained in dataset order for failure analysis). 'mkReport' is the pure aggregator;
 -- 'renderReportText' is a deterministic human-readable rendering reused by the
--- CLI (@docs/plans/12-cli-and-developer-experience.md@) and by golden tests.
+-- CLI (@docs\/plans\/12-cli-and-developer-experience.md@) and by golden tests.
 --
--- 'EvalConfig' carries the run knobs (bounded 'concurrency', the 'FailurePolicy'
+-- t'EvalConfig' carries the run knobs (bounded 'concurrency', the 'FailurePolicy'
 -- for per-example errors, optional per-example timeout, and 'numSamples' per example
 -- for multi-sample metrics); 'defaultEvalConfig' scores failures @0@ and keeps going.
 module Shikumi.Eval.Report
@@ -47,7 +47,7 @@
 
 -- | The reason an example did not complete normally.
 data FailureReason
-  = -- | @runProgram@ threw a 'Shikumi.Error.ShikumiError' (rendered to text)
+  = -- | @runProgram@ threw a t'Shikumi.Error.ShikumiError' (rendered to text)
     ProgramError !Text
   | -- | the metric itself failed (an effectful metric raised an error)
     MetricError !Text
@@ -84,7 +84,7 @@
   }
   deriving stock (Eq, Show)
 
--- | The zero of 'UsageTotals'.
+-- | The zero of t'UsageTotals'.
 emptyUsageTotals :: UsageTotals
 emptyUsageTotals = UsageTotals 0 0 0 0 Nothing 0
 
@@ -118,7 +118,7 @@
 
 -- | Knobs for an evaluation run.
 data EvalConfig = EvalConfig
-  { -- | max examples evaluated at once (forced to @>= 1@ by 'evaluate')
+  { -- | max examples evaluated at once (forced to @>= 1@ by 'Shikumi.Eval.Evaluate.evaluate')
     concurrency :: !Int,
     failurePolicy :: !FailurePolicy,
     -- | wall-clock budget per example in milliseconds; 'Nothing' means no timeout
diff --git a/src/Shikumi/Eval/Types.hs b/src/Shikumi/Eval/Types.hs
--- a/src/Shikumi/Eval/Types.hs
+++ b/src/Shikumi/Eval/Types.hs
@@ -1,15 +1,15 @@
 -- | The evaluation data model — MasterPlan integration point #5. The optimizer
--- (@docs/plans/10-optimizer-framework.md@) and the CLI
--- (@docs/plans/12-cli-and-developer-experience.md@) /consume/ these types and
+-- (@docs\/plans\/10-optimizer-framework.md@) and the CLI
+-- (@docs\/plans\/12-cli-and-developer-experience.md@) /consume/ these types and
 -- must not redefine them.
 --
--- A 'Score' is a 'Double' clamped to the closed interval @[0, 1]@ at
+-- A t'Score' is a 'Double' clamped to the closed interval @[0, 1]@ at
 -- construction, so downstream ranking code can assume @0 <= s <= 1@ without
--- re-checking (a @Bool@ metric maps @True -> 1@, @False -> 0@). An 'Example' is
--- one labelled datum (input + expected output); a 'Dataset' is a list of them. A
--- 'Prediction' is a program's actual output for one example, carrying a primary
+-- re-checking (a @Bool@ metric maps @True -> 1@, @False -> 0@). An t'Example' is
+-- one labelled datum (input + expected output); a t'Dataset' is a list of them. A
+-- t'Prediction' is a program's actual output for one example, carrying a primary
 -- output plus the non-empty set of sampled outputs (a single-output program
--- yields a one-element 'Prediction'), so agreement-rewarding metrics
+-- yields a one-element t'Prediction'), so agreement-rewarding metrics
 -- (majority-vote, ensemble) can inspect every sample.
 module Shikumi.Eval.Types
   ( -- * Scores
@@ -68,7 +68,7 @@
   }
   deriving stock (Eq, Show)
 
--- | Build an 'Example' from an input and its expected output.
+-- | Build an t'Example' from an input and its expected output.
 example :: i -> o -> Example i o
 example = Example
 
@@ -76,7 +76,7 @@
 newtype Dataset i o = Dataset [Example i o]
   deriving stock (Eq, Show)
 
--- | Build a 'Dataset' from a list of examples.
+-- | Build a t'Dataset' from a list of examples.
 dataset :: [Example i o] -> Dataset i o
 dataset = Dataset
 
diff --git a/src/Shikumi/Eval/Usage.hs b/src/Shikumi/Eval/Usage.hs
--- a/src/Shikumi/Eval/Usage.hs
+++ b/src/Shikumi/Eval/Usage.hs
@@ -6,7 +6,7 @@
 -- EP-6's @cachedLLM@ and EP-7's @tracedLLM@ use. Because @LLM.complete@ returns
 -- the full 'Response' (which already carries 'Baikai.Usage.Usage' and
 -- 'Baikai.Cost.Cost'), no substrate hook or writer effect is needed; the
--- interposed handler reads usage straight off each response. A shared 'IORef'
+-- interposed handler reads usage straight off each response. A shared 'Data.IORef.IORef'
 -- mutated with 'atomicModifyIORef'' makes the accumulation safe under the bounded
 -- concurrency @evaluate@ uses (each pooled worker gets a cloned env that still
 -- references the one ref).
diff --git a/test/GoldenSpec.hs b/test/GoldenSpec.hs
--- a/test/GoldenSpec.hs
+++ b/test/GoldenSpec.hs
@@ -1,7 +1,7 @@
 {-# LANGUAGE RankNTypes #-}
 
 -- | A golden test for a stub program, run deterministically and offline under a
--- constant mock LM. The committed @test/golden/qa-program.golden@ pins the
+-- constant mock LM. The committed @test\/golden\/qa-program.golden@ pins the
 -- transcript; the plan records the fail-before/pass-after demonstration.
 module GoldenSpec (tests) where
 
diff --git a/test/MetricSpec.hs b/test/MetricSpec.hs
--- a/test/MetricSpec.hs
+++ b/test/MetricSpec.hs
@@ -1,5 +1,5 @@
 -- | Unit tests for the pure metrics and combinators: exact match, normalized
--- string similarity (identical / near / disjoint), and the @threshold@,
+-- string similarity (identical \/ near \/ disjoint), and the @threshold@,
 -- @weightedMean@, and @invert@ combinators. All offline and deterministic.
 module MetricSpec (tests) where
 
@@ -50,7 +50,7 @@
       testCase "threshold fails a low score" $
         threshold 0.5 (constMetric 0.3) () (prediction ()) @?= scoreZero,
       testCase "weightedMean averages by weight" $
-        -- (1*0.2 + 3*0.6) / (1+3) = 2.0/4 = 0.5 (within float tolerance)
+        -- (1*0.2 + 3*0.6) \/ (1+3) = 2.0\/4 = 0.5 (within float tolerance)
         let s = unScore (weightedMean [(1, constMetric 0.2), (3, constMetric 0.6)] () (prediction ()))
          in assertBool ("expected ~0.5, got " <> show s) (abs (s - 0.5) < 1e-9),
       testCase "weightedMean of empty is zero" $
