diff --git a/CHANGELOG.md b/CHANGELOG.md
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -2,6 +2,18 @@
 
 ## Unreleased
 
+## 0.3.0.0 — 2026-09-08
+
+- Raise the internal `shikumi` bound to `^>=0.4.0.0` for the breaking core release.
+
+- Preserve logical usage quality and unknown-call counts, and explicitly attach/render whole-run transport billing separately. Public UsageTotals and Report fields require PVP major review; empty/default constructors retain zero semantics.
+
+- Render structured provider failures through the shared readable renderer while retaining existing failure policy and legacy diagnostics.
+
+- Upgrade the dependency on `mori://shinzui/baikai/packages/baikai` to `>=0.7.0.0 && <0.8`.
+
+- Expose `scoreExecution` and `tryShikumi` for alternate typed runners retaining execution evidence. Existing evaluation failure, timing, concurrency, and usage behavior is unchanged.
+
 ## 0.2.0.3 — 2026-08-29
 
 ### Changed
diff --git a/shikumi-eval.cabal b/shikumi-eval.cabal
--- a/shikumi-eval.cabal
+++ b/shikumi-eval.cabal
@@ -1,6 +1,6 @@
 cabal-version:   3.4
 name:            shikumi-eval
-version:         0.2.0.3
+version:         0.3.0.0
 synopsis:        Typed evaluation framework for shikumi LM programs (EP-8)
 category:        AI
 description:
@@ -46,14 +46,14 @@
 
   build-depends:
     , aeson         >=2.2      && <2.3
-    , baikai        >=0.6      && <0.7
+    , baikai        >=0.7.0.0  && <0.8
     , base          >=4.20     && <5
     , bytestring    >=0.11     && <0.13
     , containers    >=0.6      && <0.9
     , effectful     >=2.5      && <2.7
     , generic-lens  >=2.2      && <2.4
     , lens          ^>=5.3
-    , shikumi       ^>=0.3.0.0
+    , shikumi       ^>=0.4.0.0
     , tasty         >=1.4      && <1.6
     , tasty-golden  >=2.3      && <2.4
     , text          ^>=2.1
@@ -79,13 +79,13 @@
 
   build-depends:
     , aeson
-    , baikai        >=0.6      && <0.7
+    , baikai        >=0.7.0.0  && <0.8
     , base
     , effectful
     , generic-lens
     , lens
-    , shikumi       ^>=0.3.0.0
-    , shikumi-eval  ^>=0.2.0.0
+    , shikumi       ^>=0.4.0.0
+    , shikumi-eval  ^>=0.3.0.0
     , tasty
     , tasty-golden
     , tasty-hunit
diff --git a/src/Shikumi/Eval/Evaluate.hs b/src/Shikumi/Eval/Evaluate.hs
--- a/src/Shikumi/Eval/Evaluate.hs
+++ b/src/Shikumi/Eval/Evaluate.hs
@@ -15,6 +15,8 @@
   ( evaluate,
     evaluatePure,
     evaluateWith,
+    scoreExecution,
+    tryShikumi,
   )
 where
 
@@ -27,7 +29,7 @@
 import Effectful.Error.Static (Error, catchError, throwError)
 import Effectful.Prim (Prim)
 import Shikumi.Effect.Time (Time, getMonotonicTimeNSec)
-import Shikumi.Error (ShikumiError (..))
+import Shikumi.Error (ShikumiError (..), renderShikumiError)
 import Shikumi.Eval.Metric (Metric, MetricM, liftMetric)
 import Shikumi.Eval.Report
   ( EvalConfig (..),
@@ -119,16 +121,36 @@
   o ->
   Eff es (Score, Maybe FailureReason)
 scoreExample cfg metric prog inp expd = do
-  predOrErr <- tryShikumi (buildPrediction cfg prog inp)
-  case predOrErr of
+  (_, result) <-
+    scoreExecution
+      (const (failurePolicy cfg))
+      id
+      (tryShikumi (buildPrediction cfg prog inp))
+      (metric expd)
+  pure result
+
+-- | Shared typed execution boundary. The runner retains arbitrary evidence even
+-- on failure; the projection identifies its root result. Only checked errors are
+-- caught. Callers choose their error classification without losing evidence.
+scoreExecution ::
+  (Error ShikumiError :> es) =>
+  (ShikumiError -> FailurePolicy) ->
+  (a -> Either ShikumiError b) ->
+  Eff es a ->
+  (b -> Eff es Score) ->
+  Eff es (a, (Score, Maybe FailureReason))
+scoreExecution policy project runner metric = do
+  evidence <- runner
+  result <- case project evidence of
     Left e -> boundary (ProgramError (renderErr e)) e
-    Right pr -> do
-      scoreOrErr <- tryShikumi (metric expd pr)
-      case scoreOrErr of
+    Right value -> do
+      scored <- tryShikumi (metric value)
+      case scored of
         Left e -> boundary (MetricError (renderErr e)) e
         Right s -> pure (s, Nothing)
+  pure (evidence, result)
   where
-    boundary reason e = case failurePolicy cfg of
+    boundary reason e = case policy e of
       FailAbort -> throwError e
       FailScore s -> pure (s, Just reason)
 
@@ -157,4 +179,5 @@
 
 -- | Render a shikumi error for a 'FailureReason'.
 renderErr :: ShikumiError -> T.Text
-renderErr = T.pack . show
+renderErr e@ProviderError {} = renderShikumiError e
+renderErr e = T.pack (show e)
diff --git a/src/Shikumi/Eval/Report.hs b/src/Shikumi/Eval/Report.hs
--- a/src/Shikumi/Eval/Report.hs
+++ b/src/Shikumi/Eval/Report.hs
@@ -26,16 +26,24 @@
     Report (..),
     mkReport,
     renderReportText,
+    attachBillingSummary,
+    usageTotalsFromUsage,
   )
 where
 
+import Baikai.Cost qualified as C
+import Baikai.Usage qualified as U
+import Data.Aeson (encode)
+import Data.ByteString.Lazy qualified as BL
 import Data.Maybe (isJust, mapMaybe)
 import Data.Text (Text)
 import Data.Text qualified as T
+import Data.Text.Encoding qualified as TE
 import GHC.Generics (Generic)
 import Numeric (showFFloat)
 import Numeric.Natural (Natural)
 import Shikumi.Eval.Types (Score, scoreZero, unScore)
+import Shikumi.LLM.Observation (BillingSummary, UsageRecord (..), renderBillingSummary, usageUnknown)
 
 -- | The reason an example did not complete normally.
 data FailureReason
@@ -70,13 +78,15 @@
   { totalInputTokens :: !Natural,
     totalOutputTokens :: !Natural,
     totalTokens :: !Natural,
-    totalCostUsd :: !Rational
+    totalCostUsd :: !Rational,
+    usageQuality :: !(Maybe UsageRecord),
+    unknownUsageCalls :: !Int
   }
   deriving stock (Eq, Show)
 
 -- | The zero of 'UsageTotals'.
 emptyUsageTotals :: UsageTotals
-emptyUsageTotals = UsageTotals 0 0 0 0
+emptyUsageTotals = UsageTotals 0 0 0 0 Nothing 0
 
 instance Semigroup UsageTotals where
   a <> b =
@@ -85,7 +95,24 @@
       (totalOutputTokens a + totalOutputTokens b)
       (totalTokens a + totalTokens b)
       (totalCostUsd a + totalCostUsd b)
+      (combineQuality (usageQuality a) (usageQuality b))
+      (unknownUsageCalls a + unknownUsageCalls b)
 
+combineQuality :: Maybe UsageRecord -> Maybe UsageRecord -> Maybe UsageRecord
+combineQuality Nothing b = b
+combineQuality a Nothing = a
+combineQuality (Just (UsageRecord a)) (Just (UsageRecord b)) = Just (UsageRecord (a <> b))
+
+usageTotalsFromUsage :: U.Usage -> UsageTotals
+usageTotalsFromUsage u =
+  UsageTotals
+    (U.inputTokens u)
+    (U.outputTokens u)
+    (U.totalTokens u)
+    (C.usd (U.cost u))
+    (Just (UsageRecord u))
+    (if usageUnknown (Just (UsageRecord u)) then 1 else 0)
+
 instance Monoid UsageTotals where
   mempty = emptyUsageTotals
 
@@ -127,7 +154,8 @@
     usage :: !UsageTotals,
     -- | sum of per-example latencies; under concurrent evaluation this can exceed
     -- wall-clock time because it is total compute latency, not elapsed time
-    totalLatencyMs :: !Integer
+    totalLatencyMs :: !Integer,
+    transportBilling :: !(Maybe BillingSummary)
   }
   deriving stock (Eq, Show)
 
@@ -149,9 +177,14 @@
       total = length rs,
       results = rs,
       usage = u,
-      totalLatencyMs = sum (map latencyMs rs)
+      totalLatencyMs = sum (map latencyMs rs),
+      transportBilling = Nothing
     }
 
+-- | Attach whole-run transport totals, without adding them to logical usage.
+attachBillingSummary :: BillingSummary -> Report -> Report
+attachBillingSummary b r = r {transportBilling = Just b}
+
 -- | A deterministic, human-readable multi-line summary of a report. The format
 -- is stable (fixed 4-decimal score and cost, examples in index order) so the CLI
 -- and golden tests can rely on it. The exact shape, for a 3-example run with one
@@ -167,7 +200,7 @@
 -- When there are no failures the trailing @failures:@ block is omitted.
 renderReportText :: Report -> Text
 renderReportText r =
-  T.intercalate "\n" (header ++ failureLines)
+  T.intercalate "\n" (header ++ qualityLines ++ billingLines ++ failureLines)
   where
     header =
       [ "score="
@@ -187,6 +220,10 @@
         "cost: $" <> fixed4 (fromRational (totalCostUsd (usage r))),
         "latency-sum: " <> tshow (totalLatencyMs r) <> " ms"
       ]
+    qualityLines = case usageQuality (usage r) of
+      Nothing -> []
+      Just u -> ["logical usage quality: unknown-calls=" <> tshow (unknownUsageCalls (usage r)) <> " " <> TE.decodeUtf8 (BL.toStrict (encode u))]
+    billingLines = maybe [] (pure . renderBillingSummary) (transportBilling r)
     failingResults = mapMaybe asFailure (results r)
     asFailure er = (\fr -> (index er, fr)) <$> failure er
     failureLines
diff --git a/src/Shikumi/Eval/Usage.hs b/src/Shikumi/Eval/Usage.hs
--- a/src/Shikumi/Eval/Usage.hs
+++ b/src/Shikumi/Eval/Usage.hs
@@ -21,7 +21,7 @@
 import Effectful (Eff, (:>))
 import Effectful.Dispatch.Dynamic (interpose)
 import Effectful.Prim.IORef (Prim, atomicModifyIORef', newIORef, readIORef)
-import Shikumi.Eval.Report (UsageTotals (..), emptyUsageTotals)
+import Shikumi.Eval.Report (UsageTotals, emptyUsageTotals, usageTotalsFromUsage)
 import Shikumi.LLM (LLM (..), Response, complete, stream)
 
 -- | Run @act@, accumulating every @LLM@ call's usage/cost into a 'UsageTotals'.
@@ -55,13 +55,7 @@
 
 -- | Project an assistant payload's token usage and cost into a 'UsageTotals'.
 usageOfAssistant :: AssistantPayload -> UsageTotals
-usageOfAssistant msg =
-  UsageTotals
-    { totalInputTokens = msg ^. #usage . #inputTokens,
-      totalOutputTokens = msg ^. #usage . #outputTokens,
-      totalTokens = msg ^. #usage . #totalTokens,
-      totalCostUsd = msg ^. #usage . #cost . #usd
-    }
+usageOfAssistant msg = usageTotalsFromUsage (msg ^. #usage)
 
 -- | The usage of one streamed call: read off terminal events. Baikai streams
 -- emit exactly one 'EventDone' or 'EventError' carrying the assembled message
diff --git a/test/EvalFixtures.hs b/test/EvalFixtures.hs
--- a/test/EvalFixtures.hs
+++ b/test/EvalFixtures.hs
@@ -53,7 +53,7 @@
 import GHC.Generics (Generic)
 import Shikumi.Adapter (ToPrompt)
 import Shikumi.Error (ShikumiError (..))
-import Shikumi.Eval.Report (UsageTotals (..))
+import Shikumi.Eval.Report (UsageTotals, usageTotalsFromUsage)
 import Shikumi.LLM (LLM (..))
 import Shikumi.Module (predict)
 import Shikumi.Program (Program)
@@ -117,13 +117,7 @@
     & #message . #usage . #cost . #usd .~ (1 % 1000)
 
 usageTotalsPerCall :: UsageTotals
-usageTotalsPerCall =
-  UsageTotals
-    { totalInputTokens = 100,
-      totalOutputTokens = 20,
-      totalTokens = 120,
-      totalCostUsd = 1 % 1000
-    }
+usageTotalsPerCall = usageTotalsFromUsage (usageResponse "yes" ^. #message . #usage)
 
 -- | A successful terminal stream whose assembled message carries the same known
 -- non-zero usage as 'usageResponse'.
diff --git a/test/EvaluateSpec.hs b/test/EvaluateSpec.hs
--- a/test/EvaluateSpec.hs
+++ b/test/EvaluateSpec.hs
@@ -9,7 +9,7 @@
 import Data.List.NonEmpty qualified as NE
 import Effectful (runEff)
 import Effectful.Concurrent (runConcurrent)
-import Effectful.Error.Static (runErrorNoCallStack)
+import Effectful.Error.Static (runErrorNoCallStack, throwError)
 import Effectful.Prim (runPrim)
 import EvalFixtures
   ( Answer (..),
@@ -23,7 +23,7 @@
   )
 import Shikumi.Effect.Time (runTime)
 import Shikumi.Error (ShikumiError (..))
-import Shikumi.Eval.Evaluate (evaluatePure, evaluateWith)
+import Shikumi.Eval.Evaluate (evaluatePure, evaluateWith, scoreExecution)
 import Shikumi.Eval.Metric (exactMatch, liftMetric)
 import Shikumi.Eval.Report
   ( ExampleResult (..),
@@ -53,7 +53,27 @@
 tests =
   testGroup
     "Evaluate"
-    [ testCase "four-of-five exact match -> aggregateScore 0.8" $ do
+    [ testCase "alternate runner retains typed failure evidence" $ do
+        let err = InvalidJSON "original"
+            envelope = (Left err :: Either ShikumiError Answer, ["attempt"] :: [String])
+        result <-
+          runEff . runErrorNoCallStack @ShikumiError $
+            scoreExecution (const (FailScore scoreZero)) fst (pure envelope) (const (pure (boolScore True)))
+        result @?= Right (envelope, (scoreZero, Just (ProgramError "InvalidJSON \"original\"")))
+        aborted <-
+          runEff . runErrorNoCallStack @ShikumiError $
+            scoreExecution (const FailAbort) fst (pure envelope) (const (pure scoreZero))
+        aborted @?= Left err,
+      testCase "alternate runner labels metric failures separately" $ do
+        result <-
+          runEff . runErrorNoCallStack @ShikumiError $
+            scoreExecution
+              (const (FailScore scoreZero))
+              id
+              (pure (Right ()))
+              (const (throwError (ValidationFailure "metric")))
+        result @?= Right (Right (), (scoreZero, Just (MetricError "ValidationFailure \"metric\""))),
+      testCase "four-of-five exact match -> aggregateScore 0.8" $ do
         let ds = dataset [example q a | (q, a) <- aggregateData]
         report <-
           runEff . runPrim . runTime . runConcurrent . runErrorNoCallStack @ShikumiError $
diff --git a/test/ReportSpec.hs b/test/ReportSpec.hs
--- a/test/ReportSpec.hs
+++ b/test/ReportSpec.hs
@@ -32,7 +32,7 @@
   ]
 
 fixtureUsage :: UsageTotals
-fixtureUsage = UsageTotals {totalInputTokens = 120, totalOutputTokens = 45, totalTokens = 165, totalCostUsd = 23 % 10000}
+fixtureUsage = UsageTotals {totalInputTokens = 120, totalOutputTokens = 45, totalTokens = 165, totalCostUsd = 23 % 10000, usageQuality = Nothing, unknownUsageCalls = 0}
 
 fixtureReport :: Report
 fixtureReport = mkReport fixtureResults fixtureUsage
@@ -61,8 +61,8 @@
       testCase "results retained in order" $ map index (results fixtureReport) @?= [0, 1, 2],
       testCase "empty report is zero" $ aggregateScore (mkReport [] emptyUsageTotals) @?= 0,
       testCase "UsageTotals Monoid sums" $
-        UsageTotals 10 20 30 (1 % 100) <> UsageTotals 1 2 3 (2 % 100)
-          @?= UsageTotals 11 22 33 (3 % 100),
+        UsageTotals 10 20 30 (1 % 100) Nothing 0 <> UsageTotals 1 2 3 (2 % 100) Nothing 0
+          @?= UsageTotals 11 22 33 (3 % 100) Nothing 0,
       testCase "UsageTotals mempty is empty" $ (mempty :: UsageTotals) @?= emptyUsageTotals,
       testCase "renderReportText matches the documented format" $
         renderReportText fixtureReport @?= expectedRender
diff --git a/test/UsageSpec.hs b/test/UsageSpec.hs
--- a/test/UsageSpec.hs
+++ b/test/UsageSpec.hs
@@ -20,7 +20,7 @@
 import Shikumi.Error (ShikumiError)
 import Shikumi.Eval.Evaluate (evaluatePure)
 import Shikumi.Eval.Metric (exactMatch)
-import Shikumi.Eval.Report (Report (..), UsageTotals (..))
+import Shikumi.Eval.Report (Report (..))
 import Shikumi.Eval.Types (dataset, example)
 import Shikumi.Eval.Usage (withUsageTotals)
 import Shikumi.LLM (stream)
@@ -65,10 +65,5 @@
           Left e -> assertFailure ("unexpected error: " <> show e)
           Right r ->
             usage r
-              @?= UsageTotals
-                { totalInputTokens = 300,
-                  totalOutputTokens = 60,
-                  totalTokens = 360,
-                  totalCostUsd = 3 / 1000
-                }
+              @?= mconcat (replicate 3 usageTotalsPerCall)
     ]
