Advertise a provider's configured context window in /v1/models

Two bugs made codex_think advertise a context window it was not
configured with, so clients that size their context from the model
listing refused to run against it.

1. The endpoint-level model cache was keyed on type+endpoint alone.
   codex_think, openai_think and bigscreen are all
   codex:https://api.openai.com/v1 but authenticate as different
   accounts, so whichever prefetched first populated the shared entry
   and the others served its model list -- an OAuth ChatGPT provider
   inherited an API-key provider's generic OpenAI models, contexts and
   all. Same collision across the three kilo-* providers. Key the entry
   on a digest of the provider's credentials too. Also fix
   invalidate_provider_cache(), which tried to find a provider's
   endpoint entry with a substring match on a key that never contains
   the provider id.

2. Provider models were published exactly as the upstream API returned
   them, so default_context_size / default_max_request_tokens never
   reached the listing. Stamp them on, letting explicit configuration
   override the fetched value.

_configured_context_size() deliberately does not call
get_context_config_for_model(): that helper ends in
_infer_context_size_from_model(), whose generic 8192 fallback is right
for sizing a request but would overwrite a real fetched window (272000)
with a guess when published.

Bump version to 0.99.89.
Co-Authored-By: 's avatarClaude Opus 4.8 (1M context) <noreply@anthropic.com>
parent 9aab99d5
...@@ -55,7 +55,7 @@ from .auth.qwen import QwenOAuth2 ...@@ -55,7 +55,7 @@ from .auth.qwen import QwenOAuth2
from .handlers import RequestHandler, RotationHandler, AutoselectHandler from .handlers import RequestHandler, RotationHandler, AutoselectHandler
from .utils import count_messages_tokens, split_messages_into_chunks, get_max_request_tokens_for_model, get_max_completion_tokens_for_model from .utils import count_messages_tokens, split_messages_into_chunks, get_max_request_tokens_for_model, get_max_completion_tokens_for_model
__version__ = "0.99.88" __version__ = "0.99.89"
__all__ = [ __all__ = [
# Config # Config
"config", "config",
......
...@@ -3,6 +3,7 @@ Provider model fetching, caching, and background refresh. ...@@ -3,6 +3,7 @@ Provider model fetching, caching, and background refresh.
Extracted from main.py. Extracted from main.py.
""" """
import time import time
import hashlib
import logging import logging
import asyncio import asyncio
from typing import Optional from typing import Optional
...@@ -42,6 +43,37 @@ def _cache_key_for_provider(provider_id: str, user_id: Optional[int] = None) -> ...@@ -42,6 +43,37 @@ def _cache_key_for_provider(provider_id: str, user_id: Optional[int] = None) ->
return f"{provider_id}:{user_id}" if user_id is not None else provider_id return f"{provider_id}:{user_id}" if user_id is not None else provider_id
def _auth_fingerprint(provider_config) -> str:
"""Short digest of a provider's credentials, or '' when it has none.
Two providers that share a type and endpoint only return the same model list
when they authenticate as the same account. Codex/Kiro/Kilo providers each
point at one shared upstream URL but carry their own OAuth credentials file,
so keying the endpoint cache on type+endpoint alone made them serve whichever
provider happened to populate the entry first — an OAuth ChatGPT provider
would inherit an API-key provider's model list, contexts and all.
"""
if provider_config is None:
return ''
material = []
api_key = getattr(provider_config, 'api_key', None)
if api_key:
material.append(str(api_key))
for attr in dir(provider_config):
if not attr.endswith('_config'):
continue
sub = getattr(provider_config, attr, None)
if not isinstance(sub, dict):
continue
for field in ('credentials_file', 'api_key', 'token', 'account_id', 'issuer'):
value = sub.get(field)
if value:
material.append(f"{attr}.{field}={value}")
if not material:
return ''
return hashlib.sha256('|'.join(sorted(material)).encode()).hexdigest()[:16]
def _endpoint_cache_key(provider_config) -> Optional[str]: def _endpoint_cache_key(provider_config) -> Optional[str]:
if provider_config is None: if provider_config is None:
return None return None
...@@ -49,7 +81,7 @@ def _endpoint_cache_key(provider_config) -> Optional[str]: ...@@ -49,7 +81,7 @@ def _endpoint_cache_key(provider_config) -> Optional[str]:
endpoint = getattr(provider_config, 'endpoint', '') or '' endpoint = getattr(provider_config, 'endpoint', '') or ''
if not prov_type and not endpoint: if not prov_type and not endpoint:
return None return None
return f"{prov_type}:{endpoint}" return f"{prov_type}:{endpoint}:{_auth_fingerprint(provider_config)}"
def _get_cached_provider_models(cache_key: str) -> Optional[list]: def _get_cached_provider_models(cache_key: str) -> Optional[list]:
...@@ -66,9 +98,17 @@ def invalidate_provider_cache(provider_id: str, user_id: Optional[int] = None) - ...@@ -66,9 +98,17 @@ def invalidate_provider_cache(provider_id: str, user_id: Optional[int] = None) -
cache_key = _cache_key_for_provider(provider_id, user_id) cache_key = _cache_key_for_provider(provider_id, user_id)
_model_cache.pop(cache_key, None) _model_cache.pop(cache_key, None)
_model_cache_timestamps.pop(cache_key, None) _model_cache_timestamps.pop(cache_key, None)
# Also drop the endpoint-level cache so a different user_id doesn't serve stale data # Also drop the endpoint-level cache so a different user_id doesn't serve stale data.
# The endpoint key never contains the provider id, so resolve this provider's own
# key from its config rather than relying on a substring match that only ever hit
# by coincidence (e.g. an endpoint URL that happened to embed the provider name).
try:
from aisbf.config import config as _global_config
own_key = _endpoint_cache_key(_global_config.get_provider(provider_id))
except Exception:
own_key = None
for key in list(_endpoint_model_cache.keys()): for key in list(_endpoint_model_cache.keys()):
if key.startswith(f"coderai:") or provider_id in key: if key.startswith('coderai:') or key == own_key or provider_id in key:
_endpoint_model_cache.pop(key, None) _endpoint_model_cache.pop(key, None)
......
...@@ -8,6 +8,7 @@ from aisbf.database import DatabaseRegistry ...@@ -8,6 +8,7 @@ from aisbf.database import DatabaseRegistry
from aisbf.app.model_cache import get_provider_models, _refresh_provider_usage_if_stale, _background_tasks from aisbf.app.model_cache import get_provider_models, _refresh_provider_usage_if_stale, _background_tasks
from aisbf.studio_services import studio_service from aisbf.studio_services import studio_service
from aisbf.context import get_context_config_for_model from aisbf.context import get_context_config_for_model
from aisbf.utils import get_max_request_tokens_for_model
router = APIRouter() router = APIRouter()
_config = None _config = None
...@@ -270,6 +271,72 @@ def _resolve_autoselect_context(autoselect_config) -> Optional[int]: ...@@ -270,6 +271,72 @@ def _resolve_autoselect_context(autoselect_config) -> Optional[int]:
return best return best
def _configured_context_size(model_name: str, provider_config) -> Optional[int]:
"""Context window explicitly configured for a model, or None.
Deliberately not get_context_config_for_model(): that helper ends in
_infer_context_size_from_model(), whose generic fallback is 8192. That is the
right answer when sizing a request we are about to send, but publishing it in
/v1/models would overwrite a real fetched context window (272000) with a
guess. Only configuration the operator actually wrote may override the
upstream listing, so this stops at the two explicit steps.
"""
if provider_config is None:
return None
if isinstance(provider_config, dict):
models = provider_config.get('models') or []
provider_default = provider_config.get('default_context_size')
else:
models = getattr(provider_config, 'models', None) or []
provider_default = getattr(provider_config, 'default_context_size', None)
for model in models:
name = model.get('name') if isinstance(model, dict) else getattr(model, 'name', None)
if name != model_name:
continue
size = model.get('context_size') if isinstance(model, dict) else getattr(model, 'context_size', None)
if size:
return size
break
return provider_default or None
def _apply_provider_context_config(models: list, provider_config) -> None:
"""Stamp the provider's configured context/token limits onto its model entries.
Rotations and autoselects already advertise a resolved context window, but
provider models were published exactly as the upstream API (or its cache)
returned them. A provider configured with default_context_size never showed
it in /v1/models, so clients that size their context from the model listing
fell back to their own default and refused to run.
Explicit configuration wins over the fetched value — it is the only way to
correct a model whose upstream listing understates its window.
"""
for entry in models:
model_name = entry.get('model_name') or entry.get('name')
if not model_name:
continue
try:
ctx = _configured_context_size(model_name, provider_config)
except Exception as e:
logger.debug(f"Could not resolve context for model {entry.get('id')}: {e}")
continue
if not ctx:
# Fetched listings set only context_window; mirror it so clients
# reading either field see the same number.
ctx = entry.get('context_window') or entry.get('context_length') or entry.get('context_size')
if ctx:
entry['context_size'] = ctx
entry['context_length'] = ctx
entry['context_window'] = ctx
if entry.get('max_request_tokens') is None:
try:
entry['max_request_tokens'] = get_max_request_tokens_for_model(model_name, provider_config)
except Exception:
pass
async def _build_model_list(request: Request) -> dict: async def _build_model_list(request: Request) -> dict:
"""Shared model listing logic used by all /models endpoints.""" """Shared model listing logic used by all /models endpoints."""
all_models = [] all_models = []
...@@ -287,6 +354,7 @@ async def _build_model_list(request: Request) -> dict: ...@@ -287,6 +354,7 @@ async def _build_model_list(request: Request) -> dict:
for provider_id, provider_config in _config.providers.items(): for provider_id, provider_config in _config.providers.items():
try: try:
provider_models = await get_provider_models(provider_id, provider_config, _config, user_id=user_id) provider_models = await get_provider_models(provider_id, provider_config, _config, user_id=user_id)
_apply_provider_context_config(provider_models, provider_config)
all_models.extend(provider_models) all_models.extend(provider_models)
except Exception as e: except Exception as e:
logger.warning(f"Error listing models for provider {provider_id}: {e}") logger.warning(f"Error listing models for provider {provider_id}: {e}")
......
...@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" ...@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project] [project]
name = "aisbf" name = "aisbf"
version = "0.99.88" version = "0.99.89"
description = "AISBF - AI Service Broker Framework || AI Should Be Free - A modular proxy server for managing multiple AI provider integrations" description = "AISBF - AI Service Broker Framework || AI Should Be Free - A modular proxy server for managing multiple AI provider integrations"
readme = "README.md" readme = "README.md"
license = "GPL-3.0-or-later" license = "GPL-3.0-or-later"
......
...@@ -106,7 +106,7 @@ class InstallCommand(_install): ...@@ -106,7 +106,7 @@ class InstallCommand(_install):
setup( setup(
name="aisbf", name="aisbf",
version="0.99.88", version="0.99.89",
author="AISBF Contributors", author="AISBF Contributors",
author_email="stefy@nexlab.net", author_email="stefy@nexlab.net",
description="AISBF - AI Service Broker Framework || AI Should Be Free - A modular proxy server for managing multiple AI provider integrations", description="AISBF - AI Service Broker Framework || AI Should Be Free - A modular proxy server for managing multiple AI provider integrations",
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment