mirror of
https://github.com/VectifyAI/PageIndex.git
synced 2026-10-02 07:44:37 +08:00
fix: flip litellm's banner switch at the provider lookups, not after them
_openai_agent asks litellm which provider serves a model twice (_openai_protocol, _cache_extra_args) before _openai_model flips suppress_debug_info, and openai_agent_config's litellm/ branch never flipped it at all; a managed cloud client runs no preload, so each failed lookup still print()ed the "Provider List:" banner. Quiet at the two get_llm_provider call sites instead, which covers every lane. Also: LITELLM_LOG at ERROR or above now clamps to that level (CRITICAL used to skip the clamp and end up noisier than unset); the retry notice is logged only when a retry follows; test_preload_stamps_litellm_log_level no longer leaks LITELLM_LOG=ERROR into the session; the config-time repair test covers _quiet_litellm with the preload thread pinned off. Claude-Session: https://claude.ai/code/session_01Gq5Kwi7wEgQNVT4k1XUhq9
This commit is contained in:
@@ -246,6 +246,8 @@ def _litellm_claude_marks(wire: str) -> Optional[dict]:
|
||||
hands bare names to LiteLLM's own resolution)."""
|
||||
try:
|
||||
from litellm import get_llm_provider
|
||||
from .utils import _quiet_litellm
|
||||
_quiet_litellm()
|
||||
model, provider, _, _ = get_llm_provider(model=wire)
|
||||
except Exception:
|
||||
return None
|
||||
@@ -281,6 +283,8 @@ def _openai_protocol(model_name: str) -> bool:
|
||||
return True
|
||||
try:
|
||||
import litellm
|
||||
from .utils import _quiet_litellm
|
||||
_quiet_litellm()
|
||||
_, provider, _, _ = litellm.get_llm_provider(model=wire)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
+6
-4
@@ -74,9 +74,11 @@ def _quiet_litellm() -> None:
|
||||
and log with their full text."""
|
||||
import litellm
|
||||
litellm.suppress_debug_info = True
|
||||
if os.environ.get("LITELLM_LOG", "ERROR").upper() == "ERROR":
|
||||
level = getattr(logging, os.environ.get("LITELLM_LOG", "ERROR").upper(),
|
||||
logging.ERROR)
|
||||
if level >= logging.ERROR:
|
||||
for name in ("LiteLLM", "LiteLLM Router", "LiteLLM Proxy", "litellm"):
|
||||
logging.getLogger(name).setLevel(logging.ERROR)
|
||||
logging.getLogger(name).setLevel(level)
|
||||
|
||||
# Backward compatibility: support CHATGPT_API_KEY as alias for OPENAI_API_KEY
|
||||
if not os.getenv("OPENAI_API_KEY") and os.getenv("CHATGPT_API_KEY"):
|
||||
@@ -189,9 +191,9 @@ def llm_completion(model, prompt, chat_history=None, return_finish_reason=False)
|
||||
except Exception as e:
|
||||
if getattr(e, "status_code", None) in _NO_RETRY_STATUS:
|
||||
raise
|
||||
logging.warning("Retrying LLM completion")
|
||||
logging.error(f"Error: {e}")
|
||||
if i < max_retries - 1:
|
||||
logging.warning("Retrying LLM completion")
|
||||
time.sleep(1)
|
||||
else:
|
||||
raise LLMRetriesExhausted(
|
||||
@@ -221,9 +223,9 @@ async def llm_acompletion(model, prompt):
|
||||
except Exception as e:
|
||||
if getattr(e, "status_code", None) in _NO_RETRY_STATUS:
|
||||
raise
|
||||
logging.warning("Retrying LLM completion")
|
||||
logging.error(f"Error: {e}")
|
||||
if i < max_retries - 1:
|
||||
logging.warning("Retrying LLM completion")
|
||||
await asyncio.sleep(1)
|
||||
else:
|
||||
raise LLMRetriesExhausted(
|
||||
|
||||
@@ -873,14 +873,18 @@ def test_non_string_doc_name_stays_not_found(client, store_path):
|
||||
|
||||
|
||||
@pytest.mark.parametrize("repair", ["_repair_litellm_types",
|
||||
"_mute_litellm_bridge_usage_warning"])
|
||||
"_mute_litellm_bridge_usage_warning",
|
||||
"_quiet_litellm"])
|
||||
def test_openai_agent_config_repairs_litellm_types(tmp_path, monkeypatch,
|
||||
repair):
|
||||
"""The BYO path resolves its model through LiteLLM in the caller's
|
||||
process, outside our completion helpers — the LiteLLM repairs must
|
||||
run at config time, and only for LiteLLM-routed models."""
|
||||
pytest.importorskip("agents")
|
||||
import pageindex.client
|
||||
import pageindex.utils
|
||||
# The preload thread runs _quiet_litellm too; keep it out of the count.
|
||||
monkeypatch.setattr(pageindex.client, "_litellm_preload_started", True)
|
||||
calls = []
|
||||
monkeypatch.setattr(pageindex.utils, repair,
|
||||
lambda: calls.append(True))
|
||||
|
||||
+18
-1
@@ -1975,7 +1975,9 @@ def test_preload_stamps_litellm_log_level(monkeypatch):
|
||||
stderr; setdefault, so a caller's explicit choice wins."""
|
||||
import pageindex.client as client_mod
|
||||
monkeypatch.setattr(client_mod, "_litellm_preload_started", True)
|
||||
monkeypatch.delenv("LITELLM_LOG", raising=False)
|
||||
# delenv on an absent name records no undo; setenv first so it does.
|
||||
monkeypatch.setenv("LITELLM_LOG", "x")
|
||||
monkeypatch.delenv("LITELLM_LOG")
|
||||
client_mod._preload_litellm()
|
||||
assert os.environ["LITELLM_LOG"] == "ERROR"
|
||||
monkeypatch.setenv("LITELLM_LOG", "DEBUG")
|
||||
@@ -2004,3 +2006,18 @@ def test_retry_notice_logs_instead_of_stdout(monkeypatch, capsys, caplog):
|
||||
assert utils.llm_completion("openai/gpt-x", "hi") == "ok"
|
||||
assert capsys.readouterr().out == ""
|
||||
assert any("Retrying" in r.getMessage() for r in caplog.records)
|
||||
|
||||
|
||||
def test_retry_notice_only_when_a_retry_follows(monkeypatch, caplog):
|
||||
"""The notice announces a retry; the terminal attempt raises instead."""
|
||||
litellm = pytest.importorskip("litellm")
|
||||
from pageindex import utils
|
||||
|
||||
def broken(**kwargs):
|
||||
raise RuntimeError("boom")
|
||||
|
||||
monkeypatch.setattr(litellm, "completion", broken)
|
||||
monkeypatch.setattr(utils.time, "sleep", lambda s: None)
|
||||
with pytest.raises(utils.LLMRetriesExhausted):
|
||||
utils.llm_completion("openai/gpt-x", "hi")
|
||||
assert sum("Retrying" in r.getMessage() for r in caplog.records) == 9
|
||||
|
||||
@@ -2122,6 +2122,36 @@ def test_litellm_lane_gates_litellm_logging(monkeypatch):
|
||||
assert gated.getEffectiveLevel() == logging.WARNING
|
||||
|
||||
|
||||
def test_provider_lookups_flip_the_switch_before_asking(monkeypatch, capsys):
|
||||
"""The lane asks litellm which provider serves a model before the
|
||||
model is built, and openai_agent_config asks with no model built at
|
||||
all; a failed ask print()s the banner, so the switch flips at the ask."""
|
||||
litellm = pytest.importorskip("litellm")
|
||||
for lookup in (local_chat._openai_protocol,
|
||||
local_chat._litellm_claude_marks):
|
||||
monkeypatch.setattr(litellm, "suppress_debug_info", False)
|
||||
assert not lookup("z-ai/not-in-the-map")
|
||||
assert litellm.suppress_debug_info is True
|
||||
assert "Provider List" not in capsys.readouterr().out
|
||||
|
||||
|
||||
def test_quiet_litellm_honors_a_quieter_level(monkeypatch):
|
||||
"""A caller asking for less than ERROR via LITELLM_LOG gets that level,
|
||||
not the WARNING chatter the gate exists to remove."""
|
||||
pytest.importorskip("litellm")
|
||||
import logging
|
||||
from pageindex.utils import _quiet_litellm
|
||||
monkeypatch.setenv("LITELLM_LOG", "CRITICAL")
|
||||
gated = logging.getLogger("LiteLLM")
|
||||
prior = gated.level
|
||||
gated.setLevel(logging.WARNING)
|
||||
try:
|
||||
_quiet_litellm()
|
||||
assert gated.getEffectiveLevel() == logging.CRITICAL
|
||||
finally:
|
||||
gated.setLevel(prior)
|
||||
|
||||
|
||||
def test_openai_protocol_predicate_follows_litellm_routing():
|
||||
pytest.importorskip("litellm")
|
||||
for name in ("gpt-5", "openai/gpt-4o", "litellm/gpt-4o",
|
||||
|
||||
Reference in New Issue
Block a user