Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 38 additions & 4 deletions crates/openhuman-core/src/inference/context_window.rs
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,11 @@
//! through `tinyinference_llm::model::discover` (bounded, cached per
//! endpoint and model), lowered by any window the provider stated in a
//! context-overflow error on that endpoint (learned by the chat adapters).
//! Ollama's OpenAI-compatible listing carries no window, so for the local
//! Ollama provider and any endpoint that looks like an Ollama server
//! (including a custom OpenAI-compatible provider at `:11434/v1`) the same
//! discovery asks the native `POST /api/show`; its result, or its failure,
//! is cached and the whole lookup is time-bounded.
//! 3. **Local runtime profile** for local providers (Ollama, LM Studio, ...).
//! 4. **Static guess**: the tier aliases, the cost catalog, and the generic
//! id-pattern table ([`super::model_context::static_context_window_for_model`]),
Expand All @@ -23,7 +28,8 @@ use std::collections::{HashMap, HashSet};
use std::sync::{Mutex, OnceLock};

use tinyinference_llm::model::discover::{
discover_model_limits_with, model_limits_cache, ModelLimitsCache, ModelListingFetcher,
discover_model_limits_with, model_limits_cache, DiscoveryRequest, ModelLimitsCache,
ModelListingFetcher,
};

use crate::config::Config;
Expand Down Expand Up @@ -132,6 +138,27 @@ fn warn_static_guess(model: &str, window: Option<u64>, provider: &str) {
}
}

/// Discovery request for the built-in local Ollama provider: its OpenAI
/// listing carries no window, so the native `/api/show` probe is forced on.
/// A custom OpenAI-compatible provider pointed at an Ollama server gets the
/// same probe from the endpoint auto-detection in `tinyinference-llm`.
fn ollama_limits_request(model: &str, config: &Config) -> Option<DiscoveryRequest> {
let root = tinyinference_local::ollama::ollama_base_url_from_override(
config.local_ai.base_url.as_deref(),
);
let model = model
.split_once('@')
.map_or(model, |(model, _)| model)
.trim();
if model.is_empty() {
return None;
}
Some(
DiscoveryRequest::new(format!("{}/v1", root.trim_end_matches('/')), model)
.with_ollama_native(true),
)
}

/// The fetcher discovery uses. Unit tests never reach the network.
fn default_fetcher() -> Box<dyn ModelListingFetcher> {
#[cfg(test)]
Expand Down Expand Up @@ -234,10 +261,17 @@ async fn resolve_inner(
}

let local_kind = tinyinference_local::profile::kind_from_provider_string(provider);
if local_kind.is_none() {
if let Some(request) =
let request = match local_kind {
None => {
crate::inference::provider::factory::model_limits_request(role, provider, model, config)
{
}
Some(tinyinference_local::profile::LocalProviderKind::Ollama) => {
ollama_limits_request(model, config)
}
Comment on lines +268 to +270

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Badge Respect the effective Ollama num_ctx limit

When local_ai.num_ctx is smaller than the model's architecture limit (or is unset and Ollama uses a smaller runtime default), this new branch treats /api/show's *.context_length as the effective turn limit. The chat builder in provider/factory/local_runtime.rs still sends only config.local_ai.num_ctx as options.num_ctx; for example, an 8,192 override with the test's 40,960 metadata makes trimming and compaction wait for 40,960 while every request has an 8,192 window, so Ollama can truncate or reject long histories. Use the allocated/requested num_ctx as the effective limit, bounded by the reported model maximum, rather than always accepting the architecture metadata.

Useful? React with 👍 / 👎.

Some(_) => None,
};
{
if let Some(request) = request {
let limits = discover_model_limits_with(fetcher, cache, &request).await;
if let Some((window, limits)) = limits
.as_ref()
Expand Down
115 changes: 113 additions & 2 deletions crates/openhuman-core/src/inference/context_window_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,8 @@ const V41_FLASH: &str = "deepseek/deepseek-v4.1-flash";
#[derive(Default)]
struct FakeFetcher {
routes: Vec<(String, Value)>,
posts: Vec<(String, Value)>,
post_calls: StdMutex<Vec<String>>,
calls: StdMutex<Vec<String>>,
}

Expand All @@ -35,6 +37,15 @@ impl FakeFetcher {
self
}

fn with_post(mut self, url: &str, body: Value) -> Self {
self.posts.push((url.to_string(), body));
self
}

fn post_count(&self) -> usize {
self.post_calls.lock().unwrap().len()
}

fn call_count(&self) -> usize {
self.calls.lock().unwrap().len()
}
Expand All @@ -54,6 +65,20 @@ impl ModelListingFetcher for FakeFetcher {
.map(|(_, body)| body.clone())
.ok_or_else(|| tinyinference_llm::Error::Catalog(format!("GET {url} returned 404")))
}

async fn post_json(
&self,
url: &str,
_headers: &[(String, String)],
_body: &Value,
) -> tinyinference_llm::Result<Value> {
self.post_calls.lock().unwrap().push(url.to_string());
self.posts
.iter()
.find(|(route, _)| route == url)
.map(|(_, body)| body.clone())
.ok_or_else(|| tinyinference_llm::Error::Catalog(format!("POST {url} returned 404")))
}
}

fn openrouter_listing() -> Value {
Expand Down Expand Up @@ -230,7 +255,7 @@ async fn unknown_model_without_provider_data_is_unknown() {
}

#[tokio::test]
async fn local_routes_use_the_local_profile_without_discovery() {
async fn local_ollama_without_a_reachable_server_falls_back_to_the_local_profile() {
let tmp = TempDir::new().unwrap();
let config = openrouter_config(&tmp);
let fetcher = FakeFetcher::default();
Expand All @@ -245,7 +270,93 @@ async fn local_routes_use_the_local_profile_without_discovery() {
.await;
assert!(resolved.window.is_some());
assert_eq!(resolved.source, WindowSource::LocalProfile);
assert_eq!(fetcher.call_count(), 0);
}

fn ollama_show() -> Value {
json!({
"model_info": { "general.architecture": "qwen3", "qwen3.context_length": 40960 },
"capabilities": ["completion", "tools"]
})
}

#[tokio::test]
async fn builtin_local_ollama_reads_the_window_from_api_show() {
let tmp = TempDir::new().unwrap();
let mut config = openrouter_config(&tmp);
config.local_ai.base_url = Some("http://127.0.0.1:11434".to_string());
let fetcher =
FakeFetcher::default().with_post("http://127.0.0.1:11434/api/show", ollama_show());
let resolved = resolve_context_window_with(
&fetcher,
&ModelLimitsCache::default(),
"chat",
"ollama:qwen3:14b",
"qwen3:14b",
&config,
)
.await;
assert_eq!(resolved.window, Some(40_960));
assert_eq!(resolved.source, WindowSource::ProviderReported);
}

#[tokio::test]
async fn custom_openai_provider_at_ollama_v1_reads_the_window_from_api_show() {
let tmp = TempDir::new().unwrap();
let mut config = openrouter_config(&tmp);
config.cloud_providers.push(CloudProviderCreds {
id: "p_ollama".to_string(),
slug: "myollama".to_string(),
label: "Ollama".to_string(),
endpoint: "http://127.0.0.1:11434/v1".to_string(),
auth_style: AuthStyle::None,
..Default::default()
});
// `/v1/models` lists the model without any window, as Ollama does.
let fetcher = FakeFetcher::default()
.with(
"http://127.0.0.1:11434/v1/models",
json!({"data": [{"id": "qwen3:14b", "object": "model", "owned_by": "library"}]}),
)
.with_post("http://127.0.0.1:11434/api/show", ollama_show());
let cache = ModelLimitsCache::default();
let resolve = || {
resolve_context_window_with(
&fetcher,
&cache,
"chat",
"myollama:qwen3:14b",
"qwen3:14b",
&config,
)
};
let resolved = resolve().await;
assert_eq!(resolved.window, Some(40_960));
assert_eq!(resolved.source, WindowSource::ProviderReported);
// Cached: the second turn makes no further requests.
resolve().await;
assert_eq!(fetcher.post_count(), 1);
}

#[tokio::test]
async fn failed_ollama_probe_is_remembered_and_not_retried() {
let tmp = TempDir::new().unwrap();
let mut config = openrouter_config(&tmp);
config.local_ai.base_url = Some("http://127.0.0.1:11434".to_string());
let fetcher = FakeFetcher::default();
let cache = ModelLimitsCache::default();
for _ in 0..3 {
let resolved = resolve_context_window_with(
&fetcher,
&cache,
"chat",
"ollama:qwen3:14b",
"qwen3:14b",
&config,
)
.await;
assert_eq!(resolved.source, WindowSource::LocalProfile);
}
assert_eq!(fetcher.post_count(), 1);
}

#[test]
Expand Down
2 changes: 1 addition & 1 deletion vendor/tinyagents
Submodule tinyagents updated 40 files
+91 −15 Cargo.lock
+1 −1 Cargo.toml
+2 −2 crates/tinyagents-graph/Cargo.toml
+4 −2 crates/tinyagents-harness/Cargo.toml
+7 −2 crates/tinyagents-harness/src/multimodal/README.md
+1 −1 crates/tinyagents-harness/src/multimodal/mod.rs
+39 −10 crates/tinyagents-harness/src/multimodal/resolve.rs
+142 −51 crates/tinyagents-harness/src/multimodal/resolve_tests.rs
+12 −4 crates/tinyagents-harness/src/providers/claude_code/README.md
+178 −14 crates/tinyagents-harness/src/providers/claude_code/driver.rs
+398 −0 crates/tinyagents-harness/src/providers/claude_code/driver_tests.rs
+29 −3 crates/tinyagents-harness/src/providers/claude_code/event_mapper.rs
+39 −0 crates/tinyagents-harness/src/providers/claude_code/event_mapper_tests.rs
+99 −13 crates/tinyagents-harness/src/providers/claude_code/input_builder.rs
+111 −0 crates/tinyagents-harness/src/providers/claude_code/input_builder_tests.rs
+33 −4 crates/tinyagents-harness/src/providers/claude_code/mod.rs
+23 −0 crates/tinyagents-harness/src/providers/claude_code/mod_tests.rs
+217 −9 crates/tinyagents-harness/src/providers/claude_code/session_store.rs
+188 −0 crates/tinyagents-harness/src/providers/claude_code/session_store_tests.rs
+5 −1 crates/tinyagents-harness/src/providers/claude_code/stream_parser.rs
+10 −0 crates/tinyagents-harness/src/providers/claude_code/stream_parser_tests.rs
+136 −12 crates/tinyagents-integration-tests/tests/dependency_boundary.rs
+1 −0 crates/tinyagents-integration-tests/tests/feature_session_transcript.rs
+2 −2 crates/tinyagents-live/Cargo.toml
+8 −8 crates/tinyagents-orchestration/Cargo.toml
+2 −2 crates/tinyagents-registry/Cargo.toml
+2 −2 crates/tinyagents-runtime/Cargo.toml
+1 −0 crates/tinyagents-runtime/src/lib_tests.rs
+2 −2 crates/tinyagents-session/Cargo.toml
+5 −1 crates/tinyagents-session/src/transcript/README.md
+1 −0 crates/tinyagents-session/src/transcript/adoption_tests.rs
+28 −0 crates/tinyagents-session/src/transcript/spend.rs
+88 −0 crates/tinyagents-session/src/transcript/spend_tests.rs
+2 −0 crates/tinyagents-session/src/transcript/transcript_tests.rs
+15 −1 crates/tinyagents-session/src/transcript/types.rs
+1 −0 crates/tinyagents-session/src/transcript/view/transcript_ordering_tests.rs
+1 −0 crates/tinyagents-session/src/transcript/view/transcript_view_tests.rs
+1 −1 crates/tinyagents-tasks/Cargo.toml
+1 −1 vendor/tinyinference
+1 −1 vendor/tinytools
Loading