mirror of
https://github.com/VectifyAI/PageIndex.git
synced 2026-10-02 07:44:37 +08:00
Keep litellm's terminal noise out of answers: stdout banners, logger chatter, retry print (#455)
* fix: mute litellm's stdout Provider List banner in the litellm lanes litellm's OpenRouter adapter probes supports_reasoning() with the provider-stripped model name on every completion, so any model missing from its static map (e.g. openrouter/z-ai/glm-5.3-flash) makes get_llm_provider print a red "Provider List:" banner straight into stdout — interleaved with the streamed answer, once per agent turn. suppress_debug_info is litellm's own embedder switch (its Router sets it too) and gates only this banner and the "Give Feedback / Get Help" one; errors still raise with their full text. Applied at the same lazy hook points as the existing litellm repairs, plus the background preload, so merely importing pageindex still leaves the host's litellm untouched. Claude-Session: https://claude.ai/code/session_018psbiPrxdiCqFsS7Tk3eFL * fix: retry notice rides logging, not the caller's stdout llm_completion/llm_acompletion printed '* Retrying *' straight into stdout on every retried request — the channel that belongs to answers and CLI output. The notice moves to logging.warning beside the error line that already accompanies it. Claude-Session: https://claude.ai/code/session_018psbiPrxdiCqFsS7Tk3eFL * fix: gate litellm's stderr WARNING chatter alongside the stdout banners The banner mute grows into _quiet_litellm: litellm's own logger sprays WARNING records (remote-map fetch fallbacks, cost hiccups) onto stderr from inside requests — not actionable for SDK callers, whose real failures raise as exceptions. The preload stamps LITELLM_LOG=ERROR before litellm's import initializes its logger (setdefault, so an explicit caller choice wins, and litellm honors a chosen level itself); the hook's setLevel covers litellm imported before us, plus the dotted litellm namespace its adapters log under. Claude-Session: https://claude.ai/code/session_018psbiPrxdiCqFsS7Tk3eFL
This commit is contained in:
@@ -26,6 +26,9 @@ def _preload_litellm() -> None:
|
||||
# import, so merely importing pageindex leaves the host process's own
|
||||
# litellm untouched; setdefault, so an explicit user choice wins.
|
||||
os.environ.setdefault("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
# Its logger initializes from LITELLM_LOG at import; ERROR keeps
|
||||
# WARNING chatter off the caller's stderr from the first record.
|
||||
os.environ.setdefault("LITELLM_LOG", "ERROR")
|
||||
global _litellm_preload_started
|
||||
if _litellm_preload_started:
|
||||
return
|
||||
@@ -34,6 +37,8 @@ def _preload_litellm() -> None:
|
||||
def _import() -> None:
|
||||
try:
|
||||
import litellm # noqa: F401
|
||||
from .utils import _quiet_litellm
|
||||
_quiet_litellm()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
@@ -219,9 +219,10 @@ def _openai_model(protocol: str, model_name: str, backend=None):
|
||||
"installed. Run: pip install 'litellm>=1.97'"
|
||||
)
|
||||
from .utils import (_litellm_model, _mute_litellm_bridge_usage_warning,
|
||||
_repair_litellm_types)
|
||||
_quiet_litellm, _repair_litellm_types)
|
||||
_repair_litellm_types()
|
||||
_mute_litellm_bridge_usage_warning()
|
||||
_quiet_litellm()
|
||||
try:
|
||||
wire = _litellm_model(model_name)
|
||||
except litellm.NotFoundError as exc:
|
||||
|
||||
+24
-2
@@ -58,6 +58,26 @@ def _mute_litellm_bridge_usage_warning() -> None:
|
||||
message=r"Pydantic serializer warnings:\s+"
|
||||
r"(PydanticSerializationUnexpectedValue\()?Expected `ResponseAPIUsage`")
|
||||
|
||||
|
||||
def _quiet_litellm() -> None:
|
||||
"""litellm print()s a red "Provider List:" banner into stdout when a
|
||||
provider lookup fails — its OpenRouter adapter probes supports_reasoning()
|
||||
with the provider-stripped model name on every completion, so any model
|
||||
missing from litellm's static map stamps the banner into the middle of a
|
||||
streamed answer (get_llm_provider_logic, seen on 1.97). Flip litellm's own
|
||||
embedder switch, as its Router does; it gates only this banner and the
|
||||
"Give Feedback / Get Help" one. Its logger quiets the same way — WARNING
|
||||
chatter (remote-map fetch fallbacks, cost hiccups) is not actionable for
|
||||
SDK callers — unless the caller picked a level via LITELLM_LOG, which
|
||||
litellm honors at import and this respects. The dotted litellm name
|
||||
covers the module-level loggers its adapters create; errors still raise
|
||||
and log with their full text."""
|
||||
import litellm
|
||||
litellm.suppress_debug_info = True
|
||||
if os.environ.get("LITELLM_LOG", "ERROR").upper() == "ERROR":
|
||||
for name in ("LiteLLM", "LiteLLM Router", "LiteLLM Proxy", "litellm"):
|
||||
logging.getLogger(name).setLevel(logging.ERROR)
|
||||
|
||||
# Backward compatibility: support CHATGPT_API_KEY as alias for OPENAI_API_KEY
|
||||
if not os.getenv("OPENAI_API_KEY") and os.getenv("CHATGPT_API_KEY"):
|
||||
import warnings
|
||||
@@ -150,6 +170,7 @@ def llm_completion(model, prompt, chat_history=None, return_finish_reason=False)
|
||||
backend = _llm_backend.get()
|
||||
model = _litellm_model(model)
|
||||
_repair_litellm_types()
|
||||
_quiet_litellm()
|
||||
for i in range(max_retries):
|
||||
try:
|
||||
response = litellm.completion(**{
|
||||
@@ -168,7 +189,7 @@ def llm_completion(model, prompt, chat_history=None, return_finish_reason=False)
|
||||
except Exception as e:
|
||||
if getattr(e, "status_code", None) in _NO_RETRY_STATUS:
|
||||
raise
|
||||
print('************* Retrying *************')
|
||||
logging.warning("Retrying LLM completion")
|
||||
logging.error(f"Error: {e}")
|
||||
if i < max_retries - 1:
|
||||
time.sleep(1)
|
||||
@@ -186,6 +207,7 @@ async def llm_acompletion(model, prompt):
|
||||
backend = _llm_backend.get()
|
||||
model = _litellm_model(model)
|
||||
_repair_litellm_types()
|
||||
_quiet_litellm()
|
||||
for i in range(max_retries):
|
||||
try:
|
||||
response = await litellm.acompletion(**{
|
||||
@@ -199,7 +221,7 @@ async def llm_acompletion(model, prompt):
|
||||
except Exception as e:
|
||||
if getattr(e, "status_code", None) in _NO_RETRY_STATUS:
|
||||
raise
|
||||
print('************* Retrying *************')
|
||||
logging.warning("Retrying LLM completion")
|
||||
logging.error(f"Error: {e}")
|
||||
if i < max_retries - 1:
|
||||
await asyncio.sleep(1)
|
||||
|
||||
@@ -1967,3 +1967,40 @@ def test_blank_chat_model_carries_no_model_into_agent_config():
|
||||
client = PageIndexClient()
|
||||
client.chat_model = " "
|
||||
assert "model" not in client.openai_agent_config()
|
||||
|
||||
|
||||
def test_preload_stamps_litellm_log_level(monkeypatch):
|
||||
"""The background preload stamps LITELLM_LOG before litellm's import
|
||||
initializes its logger, so import-time WARNING chatter never reaches
|
||||
stderr; setdefault, so a caller's explicit choice wins."""
|
||||
import pageindex.client as client_mod
|
||||
monkeypatch.setattr(client_mod, "_litellm_preload_started", True)
|
||||
monkeypatch.delenv("LITELLM_LOG", raising=False)
|
||||
client_mod._preload_litellm()
|
||||
assert os.environ["LITELLM_LOG"] == "ERROR"
|
||||
monkeypatch.setenv("LITELLM_LOG", "DEBUG")
|
||||
client_mod._preload_litellm()
|
||||
assert os.environ["LITELLM_LOG"] == "DEBUG"
|
||||
|
||||
|
||||
def test_retry_notice_logs_instead_of_stdout(monkeypatch, capsys, caplog):
|
||||
"""A retried completion must not write into the caller's stdout — that
|
||||
channel belongs to answers and CLI output; the notice rides logging
|
||||
beside the error it accompanies."""
|
||||
litellm = pytest.importorskip("litellm")
|
||||
from pageindex import utils
|
||||
attempts = []
|
||||
|
||||
def flaky(**kwargs):
|
||||
if not attempts:
|
||||
attempts.append(1)
|
||||
raise RuntimeError("boom")
|
||||
message = types.SimpleNamespace(content="ok")
|
||||
choice = types.SimpleNamespace(message=message, finish_reason="stop")
|
||||
return types.SimpleNamespace(choices=[choice])
|
||||
|
||||
monkeypatch.setattr(litellm, "completion", flaky)
|
||||
monkeypatch.setattr(utils.time, "sleep", lambda s: None)
|
||||
assert utils.llm_completion("openai/gpt-x", "hi") == "ok"
|
||||
assert capsys.readouterr().out == ""
|
||||
assert any("Retrying" in r.getMessage() for r in caplog.records)
|
||||
|
||||
@@ -2557,6 +2557,41 @@ def test_litellm_lane_hides_the_bridge_usage_warning():
|
||||
assert any("Expected `int`" in m for m in seen)
|
||||
|
||||
|
||||
@needs_agents
|
||||
def test_litellm_lane_mutes_the_provider_list_banner(monkeypatch, capsys):
|
||||
"""litellm's OpenRouter adapter probes supports_reasoning() with the
|
||||
provider-stripped model name, so any model missing from its static map
|
||||
print()s a red "Provider List:" banner into the middle of the streamed
|
||||
answer; building the lane's model flips litellm's embedder switch."""
|
||||
litellm = pytest.importorskip("litellm")
|
||||
monkeypatch.setattr(litellm, "suppress_debug_info", False)
|
||||
local_chat._openai_model("chat", "openrouter/z-ai/not-in-the-map")
|
||||
assert litellm.suppress_debug_info is True
|
||||
with pytest.raises(litellm.BadRequestError):
|
||||
litellm.get_llm_provider("z-ai/not-in-the-map")
|
||||
assert "Provider List" not in capsys.readouterr().out
|
||||
|
||||
|
||||
@needs_agents
|
||||
def test_litellm_lane_gates_litellm_logging(monkeypatch):
|
||||
"""litellm's own logger sprays WARNING chatter (remote-map fetch
|
||||
fallbacks, cost hiccups) onto stderr from inside requests; building
|
||||
the lane's model gates it to ERROR — unless the caller picked a
|
||||
level via LITELLM_LOG, which stays theirs."""
|
||||
pytest.importorskip("litellm")
|
||||
import logging
|
||||
monkeypatch.delenv("LITELLM_LOG", raising=False)
|
||||
gated = logging.getLogger("LiteLLM")
|
||||
gated.setLevel(logging.WARNING)
|
||||
local_chat._openai_model("chat", "openrouter/z-ai/not-in-the-map")
|
||||
assert gated.getEffectiveLevel() == logging.ERROR
|
||||
assert not gated.isEnabledFor(logging.WARNING)
|
||||
monkeypatch.setenv("LITELLM_LOG", "DEBUG")
|
||||
gated.setLevel(logging.WARNING)
|
||||
local_chat._openai_model("chat", "openrouter/z-ai/not-in-the-map")
|
||||
assert gated.getEffectiveLevel() == logging.WARNING
|
||||
|
||||
|
||||
def test_openai_protocol_predicate_follows_litellm_routing():
|
||||
pytest.importorskip("litellm")
|
||||
for name in ("gpt-5", "openai/gpt-4o", "litellm/gpt-4o",
|
||||
|
||||
Reference in New Issue
Block a user