diff --git a/CHANGELOG.md b/CHANGELOG.md
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -7,6 +7,32 @@
 
 ## [Unreleased]
 
+## [0.1.1.0] - 2026-08-23
+
+### Changed
+
+- Widen dependency bounds for GHC 9.8–9.12 (`base`, `bytestring`, `containers`).
+  `tested-with`: GHC 9.6.7, 9.8.4, 9.10.2, 9.12.2. GitHub Actions CI runs
+  `cabal test` and `cabal haddock` on that matrix.
+- Example `model-catalog.json`: default Gemini entry is `gemini_lite` (`gemini-3.1-flash-lite`);
+  removed deprecated `gemini-2.5-flash` config; raised example `maxTokens` to 4096; added
+  optional `gpt_5_6_terra` (`gpt-5.6-terra`); corrected `gemini_lite` and `deepseek4flash`
+  pricing (DeepSeek rates are peak cache-miss; off-peak is half).
+- `get_history` tool description now documents the `"(no earlier history)"` /
+  `"(no more history)"` sentinels instead of claiming an empty result.
+
+### Fixed
+
+- Claude and Gemini gateways honor catalog `baseUrl` / `baseUrlEnv` (previously
+  ignored; always hit the public Anthropic/Google hosts). New
+  `claudeGatewayWith` / `geminiGatewayWith` constructors; optional
+  `CLAUDE_BASE_URL` and `GEMINI_BASE_URL` env overrides.
+- `get_history` no longer hangs when the visible context window has zero user
+  turns (page size 0); `chunkBackward` treats `n <= 0` as a single unpaged chunk.
+- `generateObject` / `generateObjectUntyped` never advertise tools (including
+  auto-injected `get_history`), even when `agContextWindow` is set. New
+  `createGenRequestNoTools` helper; windowing still truncates messages.
+
 ## [0.1.0.2] - 2026-07-15
 
 ### Added
@@ -59,6 +85,7 @@
   DeepSeek), single-shot generation with fallbacks, agent tool loops, structured
   output, JSON model catalog loading, and workspace-scoped filesystem tools.
 
+[0.1.1.0]: https://github.com/aische/llm-simple/compare/v0.1.0.2...v0.1.1.0
 [0.1.0.2]: https://github.com/aische/llm-simple/compare/v0.1.0.1...v0.1.0.2
 [0.1.0.1]: https://github.com/aische/llm-simple/compare/v0.1.0.0...v0.1.0.1
 [0.1.0.0]: https://github.com/aische/llm-simple/releases/tag/v0.1.0.0
diff --git a/Readme.md b/Readme.md
--- a/Readme.md
+++ b/Readme.md
@@ -21,7 +21,7 @@
 cabal build
 ```
 
-Requires GHC 9.6+ with `GHC2021` (see `llm-simple.cabal`).
+Requires GHC 9.6–9.12 with `GHC2021` (see `tested-with` in `llm-simple.cabal`).
 
 ## Quick start
 
@@ -84,7 +84,7 @@
 | ------------- | --------------------------------------------------------------------------- |
 | `providerName`| Key referenced by `providerName` in model catalog entries                   |
 | `protocol`    | `"openai"`, `"claude"`, `"gemini"`, `"ollama"`, or `"deepseek"`             |
-| `baseUrl`     | Provider host/base URL (OpenAI-compatible protocols append `/v1/chat/completions`) |
+| `baseUrl`     | Provider origin (scheme+host[+port[+optional path prefix]]). OpenAI/DeepSeek/Ollama append `/v1/chat/completions`; Claude appends `/v1/messages`; Gemini appends `/v1beta/models/{model}:generateContent` |
 | `apiKeyEnv`   | Environment variable holding the API key (omit for keyless providers)       |
 | `baseUrlEnv`  | Optional env var that overrides `baseUrl` at runtime                      |
 
@@ -93,8 +93,9 @@
 `providerName` from your model catalog.
 
 Optional base URL overrides for built-in providers: `OPENAI_BASE_URL`,
-`DEEPSEEK_BASE_URL`, `OLLAMA_BASE_URL`.
-
+`CLAUDE_BASE_URL`, `GEMINI_BASE_URL`, `DEEPSEEK_BASE_URL`, `OLLAMA_BASE_URL`.
+Do not include the protocol path (`/v1/...` or `/v1beta/...`) in `baseUrl`; the
+library appends it.
 `loadGateways` and `loadGatewaysWithDotenv` always use built-in provider defaults.
 Custom providers from a co-located `providers.json` apply when loading models via
 `loadModelOrThrow` or `loadModelsOrThrow`.
@@ -109,7 +110,7 @@
 main = do
   loadFile defaultConfig `catch` \(_ :: SomeException) -> pure ()
   (gemini, deepseek) <-
-    loadModelsOrThrow "./model-catalog.json" ("gemini_2_5_flash", "deepseek4flash")
+    loadModelsOrThrow "./model-catalog.json" ("gemini_lite", "deepseek4flash")
   ...
 ```
 
@@ -135,7 +136,7 @@
   loadFile defaultConfig `catch` \(_ :: SomeException) -> pure ()
 
   (gemini, deepseek) <-
-    loadModelsOrThrow "./model-catalog.json" ("gemini_2_5_flash", "deepseek4flash")
+    loadModelsOrThrow "./model-catalog.json" ("gemini_lite", "deepseek4flash")
 
   let models = ModelWithFallbacks { mwfModel = gemini, mwfFallbacks = [deepseek] }
 
diff --git a/app/Main.hs b/app/Main.hs
--- a/app/Main.hs
+++ b/app/Main.hs
@@ -30,7 +30,7 @@
   (_gpt, llama, _haiku, gemini, mistral, deepseek) <-
     loadModelsOrThrow
       "./model-catalog.json"
-      ("gpt_4_1", "llama_3_2", "haiku_4_5", "gemini_2_5_flash", "mistral", "deepseek4flash")
+      ("gpt_4_1", "llama_3_2", "haiku_4_5", "gemini_lite", "mistral", "deepseek4flash")
 
   let _models1 = ModelWithFallbacks {mwfModel = llama, mwfFallbacks = []}
       _models2 = ModelWithFallbacks {mwfModel = mistral, mwfFallbacks = []}
diff --git a/llm-simple.cabal b/llm-simple.cabal
--- a/llm-simple.cabal
+++ b/llm-simple.cabal
@@ -2,7 +2,7 @@
 
 name:               llm-simple
 
-version:            0.1.0.2
+version:            0.1.1.0
 
 synopsis:
     Multi-provider LLM library with agent tool loops and filesystem tools
@@ -44,6 +44,9 @@
 
 tested-with:
     GHC ==9.6.7
+  , GHC ==9.8.4
+  , GHC ==9.10.2
+  , GHC ==9.12.2
 
 source-repository head
     type:     git
@@ -142,9 +145,9 @@
         aeson-pretty   >=0.8   && <0.9,
         autodocodec    >=0.5 && < 0.6,
         autodocodec-schema >=0.2 && <0.3,
-        base           ^>=4.18,
-        bytestring     >=0.11  && <0.12,
-        containers     ^>=0.6.7,
+        base           >=4.18  && <4.22,
+        bytestring     >=0.11  && <0.13,
+        containers     >=0.6.7 && <0.8,
         directory      >=1.3   && <1.4,
         dotenv         >=0.9   && <0.13,
         filepath       >=1.4   && <1.6,
@@ -169,9 +172,9 @@
         aeson          >=2.0   && <2.3,
         autodocodec    >=0.5 && < 0.6,
         autodocodec-schema >=0.2 && <0.3,
-        base           ^>=4.18,
-        bytestring     >=0.11 && <0.12,
-        containers     ^>=0.6.7,
+        base           >=4.18  && <4.22,
+        bytestring     >=0.11  && <0.13,
+        containers     >=0.6.7 && <0.8,
         dotenv         >=0.9   && <0.13,
         directory      >=1.3   && <1.4,
         filepath       >=1.4   && <1.6,
@@ -193,9 +196,9 @@
         aeson          >=2.0   && <2.3,
         aeson-pretty   >=0.8   && <0.9,
         autodocodec    >=0.5 && < 0.6,
-        base           ^>=4.18,
-        bytestring     >=0.11 && <0.12,
-        containers     ^>=0.6.7,
+        base           >=4.18  && <4.22,
+        bytestring     >=0.11  && <0.13,
+        containers     >=0.6.7 && <0.8,
         dotenv         >=0.9   && <0.13,
         directory      >=1.3   && <1.4,
         filepath       >=1.4   && <1.6,
@@ -228,8 +231,8 @@
                         LLM.WeatherTool
     hs-source-dirs:     test
     build-depends:
-        base           ^>=4.18,
-        containers ^>=0.6.7,
+        base           >=4.18  && <4.22,
+        containers     >=0.6.7 && <0.8,
         directory      >=1.3   && <1.4,
         filepath       >=1.4   && <1.6,
         temporary      >=1.3   && <1.4,
@@ -237,7 +240,7 @@
         aeson          >=2.0   && <2.3,
         aeson-pretty   >=0.8   && <0.9,
         autodocodec    >=0.5 && < 0.6,
-        bytestring     >=0.11  && <0.12,
+        bytestring     >=0.11  && <0.13,
         hspec          >=2.10  && <2.12,
         heptapod       >=1.1   && <1.2,
         llm-simple,
diff --git a/model-catalog.json b/model-catalog.json
--- a/model-catalog.json
+++ b/model-catalog.json
@@ -7,7 +7,7 @@
             "pricePerMillionInput": 2.0,
             "pricePerMillionOutput": 8.0
         },
-        "maxTokens": 1024,
+        "maxTokens": 4096,
         "temperature": 0.5,
         "requestTimeout": 10000,
         "throttleDelay": 1000,
@@ -15,6 +15,21 @@
         "jitterBackoff": 1000
     },
     {
+        "modelConfigName": "gpt_5_6_terra",
+        "providerName": "openai",
+        "modelName": "gpt-5.6-terra",
+        "pricing": {
+            "pricePerMillionInput": 2.0,
+            "pricePerMillionOutput": 12.0
+        },
+        "maxTokens": 4096,
+        "temperature": 0.5,
+        "requestTimeout": 10000,
+        "throttleDelay": 1000,
+        "retryCount": 3,
+        "jitterBackoff": 1000
+    },
+    {
         "modelConfigName": "llama_3_2",
         "providerName": "ollama",
         "modelName": "llama3.2:latest",
@@ -22,7 +37,7 @@
             "pricePerMillionInput": 0.0,
             "pricePerMillionOutput": 0.0
         },
-        "maxTokens": 1024,
+        "maxTokens": 4096,
         "temperature": 0.5,
         "requestTimeout": 3000,
         "throttleDelay": 1000,
@@ -37,7 +52,7 @@
             "pricePerMillionInput": 0.0,
             "pricePerMillionOutput": 0.0
         },
-        "maxTokens": 1024,
+        "maxTokens": 4096,
         "temperature": 0.5,
         "requestTimeout": 30000,
         "throttleDelay": 1000,
@@ -45,29 +60,14 @@
         "jitterBackoff": 1000
     },
     {
-        "modelConfigName": "gemini_2_5_flash",
-        "providerName": "gemini",
-        "modelName": "gemini-2.5-flash",
-        "pricing": {
-            "pricePerMillionInput": 0.1,
-            "pricePerMillionOutput": 0.4
-        },
-        "maxTokens": 1024,
-        "temperature": 0.5,
-        "requestTimeout": 10000,
-        "throttleDelay": 1000,
-        "retryCount": 3,
-        "jitterBackoff": 1000
-    },
-    {
         "modelConfigName": "gemini_lite",
         "providerName": "gemini",
         "modelName": "gemini-3.1-flash-lite",
         "pricing": {
-            "pricePerMillionInput": 0.1,
-            "pricePerMillionOutput": 0.4
+            "pricePerMillionInput": 0.25,
+            "pricePerMillionOutput": 1.5
         },
-        "maxTokens": 1024,
+        "maxTokens": 4096,
         "temperature": 0.5,
         "requestTimeout": 10000,
         "throttleDelay": 1000,
@@ -82,7 +82,7 @@
             "pricePerMillionInput": 1,
             "pricePerMillionOutput": 5
         },
-        "maxTokens": 1024,
+        "maxTokens": 4096,
         "temperature": 0.5,
         "requestTimeout": 30000,
         "throttleDelay": 5000,
@@ -94,10 +94,10 @@
         "providerName": "deepseek",
         "modelName": "deepseek-v4-flash",
         "pricing": {
-            "pricePerMillionInput": 0.14,
-            "pricePerMillionOutput": 0.28
+            "pricePerMillionInput": 0.44,
+            "pricePerMillionOutput": 1.32
         },
-        "maxTokens": 1024,
+        "maxTokens": 4096,
         "temperature": 0.5,
         "requestTimeout": 20000,
         "throttleDelay": 3000,
diff --git a/providers.json b/providers.json
--- a/providers.json
+++ b/providers.json
@@ -10,13 +10,15 @@
         "providerName": "claude",
         "protocol": "claude",
         "baseUrl": "https://api.anthropic.com",
-        "apiKeyEnv": "CLAUDE_API_KEY"
+        "apiKeyEnv": "CLAUDE_API_KEY",
+        "baseUrlEnv": "CLAUDE_BASE_URL"
     },
     {
         "providerName": "gemini",
         "protocol": "gemini",
         "baseUrl": "https://generativelanguage.googleapis.com",
-        "apiKeyEnv": "GEMINI_API_KEY"
+        "apiKeyEnv": "GEMINI_API_KEY",
+        "baseUrlEnv": "GEMINI_BASE_URL"
     },
     {
         "providerName": "deepseek",
diff --git a/src/LLM.hs b/src/LLM.hs
--- a/src/LLM.hs
+++ b/src/LLM.hs
@@ -98,6 +98,7 @@
     Tool (..),
     ToolContext (..),
     createGenRequest,
+    createGenRequestNoTools,
     toTool,
     GenerateEvent (..),
     GenerateEventDetail (..),
@@ -125,6 +126,7 @@
     Tool (..),
     ToolContext (..),
     createGenRequest,
+    createGenRequestNoTools,
     generateText,
     noEventObserver,
     streamText,
diff --git a/src/LLM/Agent.hs b/src/LLM/Agent.hs
--- a/src/LLM/Agent.hs
+++ b/src/LLM/Agent.hs
@@ -26,6 +26,7 @@
     noEventObserver,
     toTool,
     createGenRequest,
+    createGenRequestNoTools,
   )
 where
 
diff --git a/src/LLM/Agent/GenerateObject.hs b/src/LLM/Agent/GenerateObject.hs
--- a/src/LLM/Agent/GenerateObject.hs
+++ b/src/LLM/Agent/GenerateObject.hs
@@ -1,8 +1,7 @@
 module LLM.Agent.GenerateObject where
 
 import Data.Aeson (Value)
-import Data.Map qualified as Map
-import LLM.Agent.ToolUtils (createGenRequest)
+import LLM.Agent.ToolUtils (createGenRequestNoTools)
 import LLM.Agent.Types
   ( Agent (..),
     RuntimeArgs (..),
@@ -19,7 +18,8 @@
 
 -- | Generate a typed Haskell value from the model via Autodocodec.
 --
--- Tools are not used; conversation context is taken from @turns@ only.
+-- Tools are never sent: @grTools@ is always empty. 'agContextWindow' still
+-- truncates messages, but does not inject @get_history@.
 generateObject ::
   (GeneratableObject t) =>
   Agent ->
@@ -27,9 +27,12 @@
   RuntimeArgs ->
   [Turn] ->
   IO (Either GenerateErrorResult (t, Usage))
-generateObject a m r t = genObject (createGenRequest id a Map.empty r t) m
+generateObject a m r t = genObject (createGenRequestNoTools a r t) m
 
 -- | Generate a JSON 'Value' from the model using a caller-supplied schema.
+--
+-- Same tool policy as 'generateObject': windowing may truncate, tools are not
+-- advertised or executed.
 generateObjectUntyped ::
   Agent ->
   ModelWithFallbacks ->
@@ -37,4 +40,4 @@
   [Turn] ->
   Value ->
   IO (Either GenerateErrorResult (Value, Usage))
-generateObjectUntyped a m r t = genObjectUntyped (createGenRequest id a Map.empty r t) m
+generateObjectUntyped a m r t = genObjectUntyped (createGenRequestNoTools a r t) m
diff --git a/src/LLM/Agent/ToolUtils.hs b/src/LLM/Agent/ToolUtils.hs
--- a/src/LLM/Agent/ToolUtils.hs
+++ b/src/LLM/Agent/ToolUtils.hs
@@ -9,6 +9,7 @@
     getResolvedTools,
     windowOffset,
     createGenRequest,
+    createGenRequestNoTools,
     embedTextTool,
   )
 where
@@ -165,6 +166,23 @@
    in GenRequest
         { grSystemPrompt = agent.agSystemPrompt,
           grTools = map (\x -> x.toolDef) tools,
+          grMessages = drop offset messages,
+          grAbortSignal = rt.rtAbortSignal,
+          grLLMHooks = rt.rtLLMHooks,
+          grHooks = rt.rtHooks
+        }
+
+-- | Like 'createGenRequest', but never attaches tool definitions.
+--
+-- Still applies 'agContextWindow' truncation to 'grMessages'. Use this for
+-- structured-output paths ('generateObject' / 'generateObjectUntyped'), which
+-- cannot execute tools and must not advertise @get_history@.
+createGenRequestNoTools :: Agent -> RuntimeArgs -> [Turn] -> GenRequest
+createGenRequestNoTools agent rt messages =
+  let offset = windowOffset agent.agContextWindow messages
+   in GenRequest
+        { grSystemPrompt = agent.agSystemPrompt,
+          grTools = [],
           grMessages = drop offset messages,
           grAbortSignal = rt.rtAbortSignal,
           grLLMHooks = rt.rtLLMHooks,
diff --git a/src/LLM/Agent/Tools/HistoryTool.hs b/src/LLM/Agent/Tools/HistoryTool.hs
--- a/src/LLM/Agent/Tools/HistoryTool.hs
+++ b/src/LLM/Agent/Tools/HistoryTool.hs
@@ -42,7 +42,8 @@
       ttoolDescription =
         "Retrieve earlier conversation history that is not in your current context window. "
           <> "Pass chunk=0 for the most recent hidden history, chunk=1 for the one before that, etc. "
-          <> "Returns an empty result when there is no more history.",
+          <> "Returns \"(no earlier history)\" when nothing is hidden, and \"(no more history)\" "
+          <> "when the chunk index is out of range.",
       ttoolReadonly = True,
       ttoolExecute = getHistoryExecTyped
     }
@@ -59,15 +60,23 @@
 -- backward from the end. Each page starts at a 'UserTurn'.
 -- Chunk 0 is the most recent page, chunk 1 the one before, etc.
 -- The oldest chunk (highest index) may contain fewer than @n@ user messages.
+--
+-- When @n <= 0@ (e.g. the visible window has no user turns), paging is
+-- disabled and the whole conversation is returned as a single chunk.
+-- This avoids an infinite loop in the page walker when @start == end@.
 chunkBackward :: Int -> [Turn] -> [[Turn]]
 chunkBackward _ [] = []
-chunkBackward n conv = reverse (go (length conv) [])
+chunkBackward n conv
+  | n <= 0 = [conv]
+  | otherwise = reverse (go (length conv) [])
   where
     go 0 acc = acc
     go end acc =
       let start = findNthUserBack n (take end conv)
           page = slice start end conv
-       in go start (page : acc)
+       in if start >= end
+            then page : acc -- should not happen for n > 0; refuse to loop
+            else go start (page : acc)
 
 -- | Find the start index for a page containing @n@ user messages,
 -- scanning backward from the end of the given prefix.
diff --git a/src/LLM/Generate/GenerateUtils.hs b/src/LLM/Generate/GenerateUtils.hs
--- a/src/LLM/Generate/GenerateUtils.hs
+++ b/src/LLM/Generate/GenerateUtils.hs
@@ -19,7 +19,7 @@
   )
 import LLM.Core.Usage (Usage (..), estimateCost)
 import LLM.Core.Utils (withRetry, withTimeout)
-import LLM.Generate.Logger (Hooks (..), LogLevel (..), onLog)
+import LLM.Generate.Logger (Hooks (..), LogLevel (..))
 import LLM.Generate.ModelConfig
   ( ModelConfig (..),
     ModelWithFallbacks (..),
diff --git a/src/LLM/Load/LoadGateways.hs b/src/LLM/Load/LoadGateways.hs
--- a/src/LLM/Load/LoadGateways.hs
+++ b/src/LLM/Load/LoadGateways.hs
@@ -29,9 +29,9 @@
     defaultProviderCatalogMap,
   )
 import LLM.Load.Types (LoadConfigError (LoadModelConfigError))
-import LLM.Providers.Claude (claudeGateway)
+import LLM.Providers.Claude (claudeGatewayWith)
 import LLM.Providers.DeepSeek (deepSeekGatewayWith)
-import LLM.Providers.Gemini (geminiGateway)
+import LLM.Providers.Gemini (geminiGatewayWith)
 import LLM.Providers.Ollama (ollamaGatewayWith)
 import LLM.Providers.OpenAI (openAIGatewayWithName)
 import Network.HTTP.Client qualified as HC
@@ -91,8 +91,12 @@
 buildGateway :: ProviderCatalogItem -> ProviderBaseUrl -> Text -> LLMGateway
 buildGateway item baseUrl apiKey =
   case item.protocol of
-    ClaudeProtocol -> claudeGateway apiKey
-    GeminiProtocol -> geminiGateway apiKey
+    ClaudeProtocol -> case baseUrl of
+      HttpsProviderBase url opts -> claudeGatewayWith url opts apiKey
+      HttpProviderBase url opts -> claudeGatewayWith url opts apiKey
+    GeminiProtocol -> case baseUrl of
+      HttpsProviderBase url opts -> geminiGatewayWith url opts apiKey
+      HttpProviderBase url opts -> geminiGatewayWith url opts apiKey
     OpenAIProtocol -> case baseUrl of
       HttpsProviderBase url opts -> openAIGatewayWithName item.providerName url opts apiKey
       HttpProviderBase url opts -> openAIGatewayWithName item.providerName url opts apiKey
diff --git a/src/LLM/Load/ProviderCatalog.hs b/src/LLM/Load/ProviderCatalog.hs
--- a/src/LLM/Load/ProviderCatalog.hs
+++ b/src/LLM/Load/ProviderCatalog.hs
@@ -85,7 +85,7 @@
           protocol = ClaudeProtocol,
           baseUrl = "https://api.anthropic.com",
           apiKeyEnv = Just "CLAUDE_API_KEY",
-          baseUrlEnv = Nothing
+          baseUrlEnv = Just "CLAUDE_BASE_URL"
         }
     geminiItem =
       ProviderCatalogItem
@@ -93,7 +93,7 @@
           protocol = GeminiProtocol,
           baseUrl = "https://generativelanguage.googleapis.com",
           apiKeyEnv = Just "GEMINI_API_KEY",
-          baseUrlEnv = Nothing
+          baseUrlEnv = Just "GEMINI_BASE_URL"
         }
     deepseekItem =
       ProviderCatalogItem
diff --git a/src/LLM/Providers.hs b/src/LLM/Providers.hs
--- a/src/LLM/Providers.hs
+++ b/src/LLM/Providers.hs
@@ -6,11 +6,15 @@
 
     -- * Gemini
     geminiProvider,
+    geminiProviderWith,
     geminiGateway,
+    geminiGatewayWith,
 
     -- * Claude
     claudeProvider,
+    claudeProviderWith,
     claudeGateway,
+    claudeGatewayWith,
 
     -- * Ollama
     ollamaProvider,
@@ -22,8 +26,8 @@
   )
 where
 
-import LLM.Providers.Claude (claudeGateway, claudeProvider)
+import LLM.Providers.Claude (claudeGateway, claudeGatewayWith, claudeProvider, claudeProviderWith)
 import LLM.Providers.DeepSeek (deepSeekGateway, deepSeekProvider)
-import LLM.Providers.Gemini (geminiGateway, geminiProvider)
+import LLM.Providers.Gemini (geminiGateway, geminiGatewayWith, geminiProvider, geminiProviderWith)
 import LLM.Providers.Ollama (ollamaGateway, ollamaProvider)
 import LLM.Providers.OpenAI (openAIGateway, openAIProvider)
diff --git a/src/LLM/Providers/Claude.hs b/src/LLM/Providers/Claude.hs
--- a/src/LLM/Providers/Claude.hs
+++ b/src/LLM/Providers/Claude.hs
@@ -1,6 +1,8 @@
 module LLM.Providers.Claude
   ( claudeGateway,
+    claudeGatewayWith,
     claudeProvider,
+    claudeProviderWith,
     parseClaudeResponse,
     parseClaudeUsage,
   )
@@ -53,7 +55,6 @@
   ( Option,
     POST (POST),
     ReqBodyJson (ReqBodyJson),
-    Scheme (Https),
     Url,
     header,
     https,
@@ -66,51 +67,57 @@
     (/:),
   )
 
--- | Create a LLMGateway for the Claude provider. Takes the API key as a parameter.
+-- | Create a LLMGateway for the Claude provider at api.anthropic.com.
 claudeGateway :: Text -> LLMGateway
 claudeGateway apiKey = toGateway $ claudeProvider apiKey
 
--- | Create a LLMProvider for the Claude provider. Takes the API key as a parameter.
+-- | Create a Claude-compatible client with a custom base URL (origin).
+-- The library appends @/v1/messages@.
+claudeGatewayWith :: Url scheme -> Option scheme -> Text -> LLMGateway
+claudeGatewayWith baseUrl baseOpts apiKey = toGateway (claudeProviderWith baseUrl baseOpts apiKey)
+
+-- | Create a LLMProvider for the Claude provider at api.anthropic.com.
 claudeProvider :: Text -> LLMProvider
-claudeProvider apiKey =
+claudeProvider = claudeProviderWith (https "api.anthropic.com") mempty
+
+-- | Claude-compatible provider with a custom base URL (origin).
+-- The library appends @/v1/messages@.
+claudeProviderWith :: Url scheme -> Option scheme -> Text -> LLMProvider
+claudeProviderWith baseUrl baseOpts apiKey =
   LLMProvider
     { providerName = "claude",
       buildBody = claudeBuildBody,
       sendRequest = sendRequest,
       sendStreamRequest = \body callback ->
-        runReq lenientConfig $
-          reqBr POST claudeUrl (ReqBodyJson body) (claudeOpts apiKey) $ \resp ->
+        runReq lenientConfig $ do
+          let url = baseUrl /: "v1" /: "messages"
+              opts = baseOpts <> claudeAuthOpts apiKey
+          reqBr POST url (ReqBodyJson body) opts $ \resp ->
             handleStreamResponse resp (`parseClaudeStream` callback),
       parseResponse = pure . parseClaudeResponse,
-      -- buildObjectBody _ r schema = claudeBuildBody False (r {reqConversation = reqConversation r <> Conversation [UserTurn ("Generate a JSON object matching this schema: " <> T.pack (show schema))]})
       buildObjectBody = \r schema ->
         let schemaText = TL.toStrict . decodeUtf8 $ encode schema
             instruction = "Respond with a raw JSON object matching this schema. No markdown, no explanation, no code fences:\n" <> schemaText
             conv' = r.reqConversation <> [UserTurn instruction]
          in claudeBuildBody False (r {reqConversation = conv'}),
       sendObjectRequest = sendRequest,
-      -- parseObjectResponse _ = parseClaudeObjectResponse
       parseObjectResponse = parseClaudeObjectResponse
     }
   where
     sendRequest body =
       runReq lenientConfig $ do
-        resp <- req POST claudeUrl (ReqBodyJson body) jsonResponse (claudeOpts apiKey)
+        let url = baseUrl /: "v1" /: "messages"
+            opts = baseOpts <> claudeAuthOpts apiKey
+        resp <- req POST url (ReqBodyJson body) jsonResponse opts
         pure (responseStatusCode resp, responseBody resp)
 
 -- Internal helpers
 
-claudeUrl :: Url 'Https
-claudeUrl = https "api.anthropic.com" /: "v1" /: "messages"
-
-claudeOpts :: Text -> Option 'Https
-claudeOpts apiKey =
+claudeAuthOpts :: Text -> Option scheme
+claudeAuthOpts apiKey =
   header "x-api-key" (encodeUtf8 apiKey)
     <> header "anthropic-version" "2023-06-01"
 
--- | Create an LLMClient from Claude credentials
--- claudeGateway :: Text -> LLMGateway
--- claudeGateway apiKey = toGateway (Claude apiKey)
 parseClaudeStream :: HC.BodyReader -> (StreamEvent -> IO ()) -> IO LLMTextResult
 parseClaudeStream reader callback = do
   blocksRef <- newIORef ([] :: [ContentBlock])
diff --git a/src/LLM/Providers/Gemini.hs b/src/LLM/Providers/Gemini.hs
--- a/src/LLM/Providers/Gemini.hs
+++ b/src/LLM/Providers/Gemini.hs
@@ -1,6 +1,8 @@
 module LLM.Providers.Gemini
   ( geminiGateway,
+    geminiGatewayWith,
     geminiProvider,
+    geminiProviderWith,
     parseGeminiResponse,
     parseGeminiUsage,
   )
@@ -53,8 +55,10 @@
 import LLM.Core.Usage (Usage (..))
 import Network.HTTP.Client qualified as HC
 import Network.HTTP.Req
-  ( POST (POST),
+  ( Option,
+    POST (POST),
     ReqBodyJson (ReqBodyJson),
+    Url,
     header,
     https,
     jsonResponse,
@@ -67,13 +71,23 @@
     (=:),
   )
 
--- | Create a LLMGateway for the Gemini provider. Takes the API key as a parameter.
+-- | Create a LLMGateway for the Gemini provider at generativelanguage.googleapis.com.
 geminiGateway :: Text -> LLMGateway
 geminiGateway apiKey = toGateway (geminiProvider apiKey)
 
--- | Create a LLMProvider for the Gemini provider. Takes the API key as a parameter.
+-- | Create a Gemini-compatible client with a custom base URL (origin).
+-- The library appends @/v1beta/models/{model}:generateContent@ (and the stream variant).
+geminiGatewayWith :: Url scheme -> Option scheme -> Text -> LLMGateway
+geminiGatewayWith baseUrl baseOpts apiKey = toGateway (geminiProviderWith baseUrl baseOpts apiKey)
+
+-- | Create a LLMProvider for the Gemini provider at generativelanguage.googleapis.com.
 geminiProvider :: Text -> LLMProvider
-geminiProvider apiKey =
+geminiProvider = geminiProviderWith (https "generativelanguage.googleapis.com") mempty
+
+-- | Gemini-compatible provider with a custom base URL (origin).
+-- The library appends @/v1beta/models/{model}:generateContent@ (and the stream variant).
+geminiProviderWith :: Url scheme -> Option scheme -> Text -> LLMProvider
+geminiProviderWith baseUrl baseOpts apiKey =
   LLMProvider
     { providerName = "gemini",
       buildBody = const geminiBuildBody,
@@ -82,11 +96,12 @@
         runReq lenientConfig $ do
           let model = extractModel body
               url =
-                https "generativelanguage.googleapis.com"
+                baseUrl
                   /: "v1beta"
                   /: "models"
                   /: (model <> ":streamGenerateContent")
-          reqBr POST url (ReqBodyJson (stripBoundsAndComments $ stripModel body)) (header "x-goog-api-key" (encodeUtf8 apiKey) <> "alt" =: ("sse" :: Text)) $ \resp ->
+              opts = baseOpts <> geminiAuthOpts apiKey <> "alt" =: ("sse" :: Text)
+          reqBr POST url (ReqBodyJson (stripBoundsAndComments $ stripModel body)) opts $ \resp ->
             handleStreamResponse resp (`parseGeminiStream` callback),
       parseResponse = parseGeminiResponse,
       buildObjectBody = \r schema ->
@@ -118,12 +133,16 @@
         -- We extract it from the request body JSON since the LLMProvider only passes Value.
         let model = extractModel body
             url =
-              https "generativelanguage.googleapis.com"
+              baseUrl
                 /: "v1beta"
                 /: "models"
                 /: (model <> ":generateContent")
-        resp <- req POST url (ReqBodyJson (stripBoundsAndComments $ stripModel body)) jsonResponse (header "x-goog-api-key" (encodeUtf8 apiKey))
+            opts = baseOpts <> geminiAuthOpts apiKey
+        resp <- req POST url (ReqBodyJson (stripBoundsAndComments $ stripModel body)) jsonResponse opts
         pure (responseStatusCode resp, responseBody resp)
+
+geminiAuthOpts :: Text -> Option scheme
+geminiAuthOpts apiKey = header "x-goog-api-key" (encodeUtf8 apiKey)
 
 -- | Extract model name stashed in the request body by geminiBuildBody.
 extractModel :: Value -> Text
diff --git a/test/LLM/GenerateObjectSpec.hs b/test/LLM/GenerateObjectSpec.hs
--- a/test/LLM/GenerateObjectSpec.hs
+++ b/test/LLM/GenerateObjectSpec.hs
@@ -3,13 +3,14 @@
 module LLM.GenerateObjectSpec (spec) where
 
 import Data.Aeson (Value, object, (.=))
+import Data.IORef (IORef, newIORef, readIORef, writeIORef)
 import Data.Text (Text)
 import Data.Text qualified as T
 import Heptapod (generate)
 import LLM.Agent.GenerateObject (generateObject, generateObjectUntyped)
 import LLM.Agent.Types (Agent (..), RuntimeArgs (..))
 import LLM.Core.Abort (AbortSignal, abort, newAbortSignal)
-import LLM.Core.Types (ChatResponse (..), LLMError (..), LLMGateway (..), LLMHooks (..), Turn (..))
+import LLM.Core.Types (ChatRequest (..), ChatResponse (..), LLMError (..), LLMGateway (..), LLMHooks (..), Turn (..))
 import LLM.Core.Usage (PricingInfo (..), Usage (..))
 import LLM.Generate.Logger (noHooks)
 import LLM.Generate.ModelConfig
@@ -88,6 +89,26 @@
           T.unpack msg `shouldContain` "Can't decode object"
         _ -> expectationFailure "expected GErrParseObjectError"
 
+    it "does not advertise tools even when agContextWindow would inject get_history" $ do
+      captured <- newIORef Nothing
+      let gw = capturingObjectGateway captured (object ["location" .= ("Paris" :: Text)]) (Usage 1 0 0)
+          models = ModelWithFallbacks (mockModel gw) []
+          agent = defaultAgent {agContextWindow = Just 1}
+          conv =
+            [ UserTurn "first",
+              AssistantTurn "a1" Nothing [],
+              UserTurn "second"
+            ]
+      rt <- mkRuntime Nothing
+      result <- generateObjectUntyped agent models rt conv (object ["type" .= ("object" :: Text)])
+      case result of
+        Right _ -> do
+          mReq <- readIORef captured
+          case mReq of
+            Nothing -> expectationFailure "expected gwGenerateObject to be called"
+            Just req -> req.reqTools `shouldBe` []
+        Left err -> expectationFailure $ show err
+
 objectGateway :: Value -> Usage -> LLMGateway
 objectGateway value usage =
   LLMGateway
@@ -95,6 +116,15 @@
       gwGenerateText = \_ _ -> pure (Right (ChatResponse "" [] Nothing Nothing)),
       gwStreamText = \_ _ _ -> pure (Right (ChatResponse "" [] Nothing Nothing)),
       gwGenerateObject = \_ _ _ -> pure (Right (value, Just usage))
+    }
+
+-- | Mock that records the 'ChatRequest' seen by 'gwGenerateObject'.
+capturingObjectGateway :: IORef (Maybe ChatRequest) -> Value -> Usage -> LLMGateway
+capturingObjectGateway ref value usage =
+  (objectGateway value usage)
+    { gwGenerateObject = \_ _ req -> do
+        writeIORef ref (Just req)
+        pure (Right (value, Just usage))
     }
 
 errorObjectGateway :: LLMError -> LLMGateway
diff --git a/test/LLM/HistoryToolSpec.hs b/test/LLM/HistoryToolSpec.hs
--- a/test/LLM/HistoryToolSpec.hs
+++ b/test/LLM/HistoryToolSpec.hs
@@ -52,6 +52,22 @@
       result <- runHistory windowedAgent sampleConversation 9
       result `shouldBe` "(no more history)"
 
+    -- When the visible window has zero UserTurns, page size is 0. Previously
+    -- chunkBackward looped forever; it now returns the whole hidden prefix as
+    -- a single chunk.
+    it "returns the full hidden prefix when the visible window has no user turns" $ do
+      let conv =
+            [ UserTurn "hidden question",
+              AssistantTurn "hidden answer" Nothing [],
+              AssistantTurn "visible assistant only" Nothing []
+            ]
+          -- Offset past both user+assistant hidden turns; visible slice is
+          -- assistant-only so countUserTurns == 0.
+          offset = 2
+      result <- runHistoryAtOffset offset conv 0
+      T.unpack result `shouldContain` "[User] hidden question"
+      T.unpack result `shouldContain` "[Assistant] hidden answer"
+
 sampleConversation :: [Turn]
 sampleConversation =
   [ UserTurn "first question",
@@ -76,10 +92,12 @@
 windowedAgent = noWindowAgent {agContextWindow = Just 1}
 
 runHistory :: Agent -> [Turn] -> Int -> IO Text
-runHistory agent conv chunk = do
+runHistory agent conv = runHistoryAtOffset (windowOffset agent.agContextWindow conv) conv
+
+runHistoryAtOffset :: Int -> [Turn] -> Int -> IO Text
+runHistoryAtOffset offset conv chunk = do
   genId <- generate
-  let offset = windowOffset agent.agContextWindow conv
-      ctx =
+  let ctx =
         ToolContext
           { tcConversation = conv,
             tcUsage = mempty,
diff --git a/test/LLM/LoadSpec.hs b/test/LLM/LoadSpec.hs
--- a/test/LLM/LoadSpec.hs
+++ b/test/LLM/LoadSpec.hs
@@ -95,6 +95,34 @@
           LLMGateway {gwName = name} = gateway
       name `shouldBe` "openrouter"
 
+    it "builds a claude gateway from a custom baseUrl origin" $ do
+      let item =
+            ProviderCatalogItem
+              { providerName = "claude",
+                protocol = ClaudeProtocol,
+                baseUrl = "https://claude-proxy.example.com",
+                apiKeyEnv = Just "CLAUDE_API_KEY",
+                baseUrlEnv = Nothing
+              }
+      Right baseUrl <- pure $ parseProviderBaseUrl item.baseUrl
+      let gateway = buildGateway item baseUrl "test-key"
+          LLMGateway {gwName = name} = gateway
+      name `shouldBe` "claude"
+
+    it "builds a gemini gateway from a custom baseUrl origin" $ do
+      let item =
+            ProviderCatalogItem
+              { providerName = "gemini",
+                protocol = GeminiProtocol,
+                baseUrl = "https://gemini-proxy.example.com",
+                apiKeyEnv = Just "GEMINI_API_KEY",
+                baseUrlEnv = Nothing
+              }
+      Right baseUrl <- pure $ parseProviderBaseUrl item.baseUrl
+      let gateway = buildGateway item baseUrl "test-key"
+          LLMGateway {gwName = name} = gateway
+      name `shouldBe` "gemini"
+
     it "loads gateways from a provider catalog file" $ do
       result <- runExceptT $ loadProviderCatalog providersOllamaOnly
       case result of
diff --git a/test/LLM/TypesSpec.hs b/test/LLM/TypesSpec.hs
--- a/test/LLM/TypesSpec.hs
+++ b/test/LLM/TypesSpec.hs
@@ -13,10 +13,6 @@
     addUsage,
     emptyUsage,
     estimateCost,
-    pricePerMillionInput,
-    pricePerMillionOutput,
-    usageInputTokens,
-    usageOutputTokens,
   )
 import LLM.Core.Utils
   ( getToolCalls,
