Commit 8bb0a549 authored by Stefy Lanza (nextime / spora )'s avatar Stefy Lanza (nextime / spora )

Merge branch 'fix/manual-disable-direct-and-image-cache-key'

parents 9f9ec1a0 e6616376
...@@ -1382,8 +1382,30 @@ class ResponseCache: ...@@ -1382,8 +1382,30 @@ class ResponseCache:
if isinstance(temperature, (int, float)): if isinstance(temperature, (int, float)):
temperature = round(temperature * 4) / 4 # Round to nearest 0.25 temperature = round(temperature * 4) / 4 # Round to nearest 0.25
# Create message content hash for semantic deduplication # Create message content hash for semantic deduplication.
# Include only the text content of messages, ignore metadata # Include the text AND any attached media (images/video/audio) so two
# requests that share a prompt but carry different attachments never
# collapse to the same key. Vision/multimodal callers commonly send a
# fixed instruction with a varying image; hashing text alone would make
# every image reuse the first image's cached answer.
def _media_ref(part: dict) -> str:
"""Stable string identity for a non-text multimodal part."""
for media_key in ('image_url', 'video_url', 'audio_url'):
media = part.get(media_key)
if isinstance(media, dict):
url = media.get('url')
if url:
return f"{media_key}:{url}"
elif isinstance(media, str) and media:
return f"{media_key}:{media}"
# input_audio / file style parts: fold in the whole part deterministically.
if part.get('type') in ('input_audio', 'file', 'input_image'):
try:
return f"{part.get('type')}:{json.dumps(part, sort_keys=True)}"
except Exception:
return f"{part.get('type')}:{part}"
return ''
message_texts = [] message_texts = []
for msg in messages: for msg in messages:
if isinstance(msg, dict): if isinstance(msg, dict):
...@@ -1391,11 +1413,16 @@ class ResponseCache: ...@@ -1391,11 +1413,16 @@ class ResponseCache:
content = msg.get('content', '') content = msg.get('content', '')
# Handle both string and list content (for multimodal) # Handle both string and list content (for multimodal)
if isinstance(content, list): if isinstance(content, list):
# For multimodal content, extract text parts # For multimodal content, extract text parts and media identities
text_parts = [] text_parts = []
for part in content: for part in content:
if isinstance(part, dict) and part.get('type') == 'text': if isinstance(part, dict):
text_parts.append(part.get('text', '')) if part.get('type') == 'text':
text_parts.append(part.get('text', ''))
else:
ref = _media_ref(part)
if ref:
text_parts.append(ref)
elif isinstance(part, str): elif isinstance(part, str):
text_parts.append(part) text_parts.append(part)
content = ' '.join(text_parts) content = ' '.join(text_parts)
......
...@@ -103,30 +103,40 @@ def _should_record_failure(error) -> bool: ...@@ -103,30 +103,40 @@ def _should_record_failure(error) -> bool:
def _raise_if_provider_unavailable(handler, provider_id: str) -> None: def _raise_if_provider_unavailable(handler, provider_id: str) -> None:
"""Reject the request with 503 when the provider is disabled, saying why. """Reject a direct request with 503 only when the provider is genuinely
unavailable, saying why.
is_rate_limited() collapses three very different states — the dashboard is_rate_limited() collapses three very different states — the dashboard
disable toggle, the failure cooldown and the usage-limit cooldown — into one disable toggle, the failure cooldown and the usage-limit cooldown — into one
boolean. Returning a bare "Provider temporarily unavailable" with no log line boolean. Those are NOT equivalent for a direct call:
made a provider that had simply been toggled off in the dashboard look
identical to an upstream outage, with nothing in debug.log to tell them * The dashboard "disable" toggle only removes a provider from *rotation* and
apart. Log the reason and put it in the response detail. *autoselect* (those scans call is_provider_disabled_cheap, which honours the
manual flag). A direct request that names the provider/model explicitly must
still go through — manually disabling a provider must never take it offline
across the whole API.
* A failure cooldown or a usage-limit cooldown is a real availability problem:
calling upstream would just fail again, so a direct call is still rejected,
with the reason logged and echoed in the response detail.
""" """
if not handler.is_rate_limited(): if not handler.is_rate_limited():
return return
if handler.is_manually_disabled(): now = time.time()
reason = "manually disabled from the dashboard (re-enable it there)" cooldown_until = handler.error_tracking.get('disabled_until') if getattr(handler, 'error_tracking', None) else None
usage_until = getattr(handler, '_usage_disabled_until', None)
in_failure_cooldown = bool(cooldown_until and cooldown_until > now)
in_usage_cooldown = bool(usage_until and usage_until > now)
# The only remaining reason is_rate_limited() can be true is the manual
# dashboard toggle, which is rotation-only — let the direct call proceed.
if not in_failure_cooldown and not in_usage_cooldown:
return
if in_failure_cooldown:
reason = f"in failure cooldown for another {int(cooldown_until - now)}s"
else: else:
now = time.time() reason = f"disabled by a usage limit for another {int(usage_until - now)}s"
cooldown_until = handler.error_tracking.get('disabled_until') if getattr(handler, 'error_tracking', None) else None
usage_until = getattr(handler, '_usage_disabled_until', None)
if cooldown_until and cooldown_until > now:
reason = f"in failure cooldown for another {int(cooldown_until - now)}s"
elif usage_until and usage_until > now:
reason = f"disabled by a usage limit for another {int(usage_until - now)}s"
else:
reason = "temporarily unavailable"
logging.getLogger(__name__).warning( logging.getLogger(__name__).warning(
f"[{provider_id}] Rejecting request with 503: provider is {reason}" f"[{provider_id}] Rejecting request with 503: provider is {reason}"
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment