Keep litellm's terminal noise out of answers: stdout banners, logger chatter, retry print (#455)

* fix: mute litellm's stdout Provider List banner in the litellm lanes

litellm's OpenRouter adapter probes supports_reasoning() with the
provider-stripped model name on every completion, so any model missing
from its static map (e.g. openrouter/z-ai/glm-5.3-flash) makes
get_llm_provider print a red "Provider List:" banner straight into
stdout — interleaved with the streamed answer, once per agent turn.
suppress_debug_info is litellm's own embedder switch (its Router sets
it too) and gates only this banner and the "Give Feedback / Get Help"
one; errors still raise with their full text. Applied at the same lazy
hook points as the existing litellm repairs, plus the background
preload, so merely importing pageindex still leaves the host's litellm
untouched.

Claude-Session: https://claude.ai/code/session_018psbiPrxdiCqFsS7Tk3eFL

* fix: retry notice rides logging, not the caller's stdout

llm_completion/llm_acompletion printed '* Retrying *' straight into
stdout on every retried request — the channel that belongs to answers
and CLI output. The notice moves to logging.warning beside the error
line that already accompanies it.

Claude-Session: https://claude.ai/code/session_018psbiPrxdiCqFsS7Tk3eFL

* fix: gate litellm's stderr WARNING chatter alongside the stdout banners

The banner mute grows into _quiet_litellm: litellm's own logger sprays
WARNING records (remote-map fetch fallbacks, cost hiccups) onto stderr
from inside requests — not actionable for SDK callers, whose real
failures raise as exceptions. The preload stamps LITELLM_LOG=ERROR
before litellm's import initializes its logger (setdefault, so an
explicit caller choice wins, and litellm honors a chosen level itself);
the hook's setLevel covers litellm imported before us, plus the dotted
litellm namespace its adapters log under.

Claude-Session: https://claude.ai/code/session_018psbiPrxdiCqFsS7Tk3eFL
This commit is contained in:
Ray
2026-09-01 22:16:23 +08:00
committed by GitHub
parent 555c370f66
commit ad7956d3c3
5 changed files with 103 additions and 3 deletions
+5
View File
@@ -26,6 +26,9 @@ def _preload_litellm() -> None:
# import, so merely importing pageindex leaves the host process's own
# litellm untouched; setdefault, so an explicit user choice wins.
os.environ.setdefault("LITELLM_LOCAL_MODEL_COST_MAP", "True")
# Its logger initializes from LITELLM_LOG at import; ERROR keeps
# WARNING chatter off the caller's stderr from the first record.
os.environ.setdefault("LITELLM_LOG", "ERROR")
global _litellm_preload_started
if _litellm_preload_started:
return
@@ -34,6 +37,8 @@ def _preload_litellm() -> None:
def _import() -> None:
try:
import litellm # noqa: F401
from .utils import _quiet_litellm
_quiet_litellm()
except Exception:
pass
+2 -1
View File
@@ -219,9 +219,10 @@ def _openai_model(protocol: str, model_name: str, backend=None):
"installed. Run: pip install 'litellm>=1.97'"
)
from .utils import (_litellm_model, _mute_litellm_bridge_usage_warning,
_repair_litellm_types)
_quiet_litellm, _repair_litellm_types)
_repair_litellm_types()
_mute_litellm_bridge_usage_warning()
_quiet_litellm()
try:
wire = _litellm_model(model_name)
except litellm.NotFoundError as exc:
+24 -2
View File
@@ -58,6 +58,26 @@ def _mute_litellm_bridge_usage_warning() -> None:
message=r"Pydantic serializer warnings:\s+"
r"(PydanticSerializationUnexpectedValue\()?Expected `ResponseAPIUsage`")
def _quiet_litellm() -> None:
"""litellm print()s a red "Provider List:" banner into stdout when a
provider lookup fails — its OpenRouter adapter probes supports_reasoning()
with the provider-stripped model name on every completion, so any model
missing from litellm's static map stamps the banner into the middle of a
streamed answer (get_llm_provider_logic, seen on 1.97). Flip litellm's own
embedder switch, as its Router does; it gates only this banner and the
"Give Feedback / Get Help" one. Its logger quiets the same way — WARNING
chatter (remote-map fetch fallbacks, cost hiccups) is not actionable for
SDK callers — unless the caller picked a level via LITELLM_LOG, which
litellm honors at import and this respects. The dotted litellm name
covers the module-level loggers its adapters create; errors still raise
and log with their full text."""
import litellm
litellm.suppress_debug_info = True
if os.environ.get("LITELLM_LOG", "ERROR").upper() == "ERROR":
for name in ("LiteLLM", "LiteLLM Router", "LiteLLM Proxy", "litellm"):
logging.getLogger(name).setLevel(logging.ERROR)
# Backward compatibility: support CHATGPT_API_KEY as alias for OPENAI_API_KEY
if not os.getenv("OPENAI_API_KEY") and os.getenv("CHATGPT_API_KEY"):
import warnings
@@ -150,6 +170,7 @@ def llm_completion(model, prompt, chat_history=None, return_finish_reason=False)
backend = _llm_backend.get()
model = _litellm_model(model)
_repair_litellm_types()
_quiet_litellm()
for i in range(max_retries):
try:
response = litellm.completion(**{
@@ -168,7 +189,7 @@ def llm_completion(model, prompt, chat_history=None, return_finish_reason=False)
except Exception as e:
if getattr(e, "status_code", None) in _NO_RETRY_STATUS:
raise
print('************* Retrying *************')
logging.warning("Retrying LLM completion")
logging.error(f"Error: {e}")
if i < max_retries - 1:
time.sleep(1)
@@ -186,6 +207,7 @@ async def llm_acompletion(model, prompt):
backend = _llm_backend.get()
model = _litellm_model(model)
_repair_litellm_types()
_quiet_litellm()
for i in range(max_retries):
try:
response = await litellm.acompletion(**{
@@ -199,7 +221,7 @@ async def llm_acompletion(model, prompt):
except Exception as e:
if getattr(e, "status_code", None) in _NO_RETRY_STATUS:
raise
print('************* Retrying *************')
logging.warning("Retrying LLM completion")
logging.error(f"Error: {e}")
if i < max_retries - 1:
await asyncio.sleep(1)
+37
View File
@@ -1967,3 +1967,40 @@ def test_blank_chat_model_carries_no_model_into_agent_config():
client = PageIndexClient()
client.chat_model = " "
assert "model" not in client.openai_agent_config()
def test_preload_stamps_litellm_log_level(monkeypatch):
"""The background preload stamps LITELLM_LOG before litellm's import
initializes its logger, so import-time WARNING chatter never reaches
stderr; setdefault, so a caller's explicit choice wins."""
import pageindex.client as client_mod
monkeypatch.setattr(client_mod, "_litellm_preload_started", True)
monkeypatch.delenv("LITELLM_LOG", raising=False)
client_mod._preload_litellm()
assert os.environ["LITELLM_LOG"] == "ERROR"
monkeypatch.setenv("LITELLM_LOG", "DEBUG")
client_mod._preload_litellm()
assert os.environ["LITELLM_LOG"] == "DEBUG"
def test_retry_notice_logs_instead_of_stdout(monkeypatch, capsys, caplog):
"""A retried completion must not write into the caller's stdout — that
channel belongs to answers and CLI output; the notice rides logging
beside the error it accompanies."""
litellm = pytest.importorskip("litellm")
from pageindex import utils
attempts = []
def flaky(**kwargs):
if not attempts:
attempts.append(1)
raise RuntimeError("boom")
message = types.SimpleNamespace(content="ok")
choice = types.SimpleNamespace(message=message, finish_reason="stop")
return types.SimpleNamespace(choices=[choice])
monkeypatch.setattr(litellm, "completion", flaky)
monkeypatch.setattr(utils.time, "sleep", lambda s: None)
assert utils.llm_completion("openai/gpt-x", "hi") == "ok"
assert capsys.readouterr().out == ""
assert any("Retrying" in r.getMessage() for r in caplog.records)
+35
View File
@@ -2557,6 +2557,41 @@ def test_litellm_lane_hides_the_bridge_usage_warning():
assert any("Expected `int`" in m for m in seen)
@needs_agents
def test_litellm_lane_mutes_the_provider_list_banner(monkeypatch, capsys):
"""litellm's OpenRouter adapter probes supports_reasoning() with the
provider-stripped model name, so any model missing from its static map
print()s a red "Provider List:" banner into the middle of the streamed
answer; building the lane's model flips litellm's embedder switch."""
litellm = pytest.importorskip("litellm")
monkeypatch.setattr(litellm, "suppress_debug_info", False)
local_chat._openai_model("chat", "openrouter/z-ai/not-in-the-map")
assert litellm.suppress_debug_info is True
with pytest.raises(litellm.BadRequestError):
litellm.get_llm_provider("z-ai/not-in-the-map")
assert "Provider List" not in capsys.readouterr().out
@needs_agents
def test_litellm_lane_gates_litellm_logging(monkeypatch):
"""litellm's own logger sprays WARNING chatter (remote-map fetch
fallbacks, cost hiccups) onto stderr from inside requests; building
the lane's model gates it to ERROR — unless the caller picked a
level via LITELLM_LOG, which stays theirs."""
pytest.importorskip("litellm")
import logging
monkeypatch.delenv("LITELLM_LOG", raising=False)
gated = logging.getLogger("LiteLLM")
gated.setLevel(logging.WARNING)
local_chat._openai_model("chat", "openrouter/z-ai/not-in-the-map")
assert gated.getEffectiveLevel() == logging.ERROR
assert not gated.isEnabledFor(logging.WARNING)
monkeypatch.setenv("LITELLM_LOG", "DEBUG")
gated.setLevel(logging.WARNING)
local_chat._openai_model("chat", "openrouter/z-ai/not-in-the-map")
assert gated.getEffectiveLevel() == logging.WARNING
def test_openai_protocol_predicate_follows_litellm_routing():
pytest.importorskip("litellm")
for name in ("gpt-5", "openai/gpt-4o", "litellm/gpt-4o",