mirror of
https://github.com/TencentCloud/Octop.git
synced 2026-10-04 08:58:20 +08:00
* fix: schedule proactive care for agents that never saved config Proactive care defaults to on, but the scheduler only listed persisted enabled rows, so most experts never entered the push loop. * fix: attach gateway connector tools without rebuilding the agent prepare_chat_mcp rebuilt the agent whenever a builtin MCP server was missing from the live runtime. For gateway-mode connectors that rebuild is pure loss: they carry no HTTP transport, their tools are built in-process from stored credentials, and _post_start_agent runs the very same injection at the end anyway. Meanwhile the rebuild drops the harness instance and with it the checkpointer pool an in-flight turn is still writing to. Split the missing names on connector mode: gateway ones refresh their credentials and inject into the running agent, the rest keep the existing reload path. Co-authored-by: Cursor <cursoragent@cursor.com> * fix: await cancellation on scheduler shutdown, and keep its sleeps out of tests ProactiveCareScheduler.shutdown() cancelled each task and returned without awaiting it, so a loop parked in a multi-hour sleep could outlive the shutdown that was supposed to end it. It now gathers the cancelled tasks before returning. Proactive care defaults to on, so creating an agent starts one of those sleeps. pytest-asyncio waits for leftover tasks before fixture teardown, which hangs the suite. octop_client shuts the scheduler down and calls the new suspend() so nothing reschedules; an autouse fixture covers tests that boot OctopServer another way. The scheduler's own unit tests opt out. suspend() exists for the tests. The alternative — every test remembering to tear the scheduler down — is the arrangement that produced the hang. * refactor: drive websockets on one event loop and drop the cross-loop workaround starlette's sync TestClient runs the app on its own anyio portal loop. That left the process with two loops over one OctopServer, so primitives bound to a loop and shared by both — AgentManager._lock — deadlocked or raised "bound to a different event loop". The tests worked around it by running the ws session in a worker thread. tests.support.http now speaks ASGI directly, so the handler, the gateway workers and the test all share one loop, the way uvicorn runs it in production. That removes the reason the production handlers marshalled every outbound frame across loops with run_coroutine_threadsafe + wrap_future. The comment on that code named the cause outright — "this handler may run on a different loop (e.g. starlette's TestClient portal)" — so it was test-shaped machinery sitting on the path of every frame the dashboard receives. Both handlers now await the send directly. * update harness-agent & harness-memory ---------
76 lines
2.9 KiB
Python
76 lines
2.9 KiB
Python
"""OctopServer + ASGI client lifecycle for integration tests."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from collections.abc import AsyncIterator
|
|
from contextlib import asynccontextmanager, nullcontext
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import httpx
|
|
|
|
from octop.api.app import build_app
|
|
from octop.config import DatabaseConfig, load_config
|
|
from octop.infra.db.rebind import persist_database_config
|
|
from octop.infra.server import OctopServer
|
|
from tests.support.harness import patch_harness
|
|
|
|
|
|
def write_octop_config(home: Path, **overrides: object) -> None:
|
|
"""Ensure ``home/config.json`` exists and apply overrides (for tests)."""
|
|
cfg_path = home / "config.json"
|
|
load_config(cfg_path)
|
|
data = json.loads(cfg_path.read_text(encoding="utf-8"))
|
|
data.update(overrides)
|
|
cfg_path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
|
|
|
|
|
async def ensure_control_plane_bound(srv: OctopServer) -> None:
|
|
"""Bind default SQLite when greenfield start deferred the control-plane DB."""
|
|
if srv.database_bound:
|
|
return
|
|
persist_database_config(srv.paths.config, DatabaseConfig())
|
|
await srv.bind_control_plane()
|
|
|
|
|
|
@asynccontextmanager
|
|
async def octop_client(
|
|
home: Path,
|
|
*,
|
|
fake_agent: Any | None = None,
|
|
patch_llm: bool = True,
|
|
bind_database: bool = True,
|
|
) -> AsyncIterator[tuple[httpx.AsyncClient, OctopServer]]:
|
|
"""Start OctopServer, yield ``(httpx client, server)``, then stop.
|
|
|
|
Greenfield starts defer the control-plane DB until ``/setup/database``.
|
|
Most tests set ``bind_database=True`` (default) to bind SQLite immediately.
|
|
Pass ``bind_database=False`` to exercise deferred password / status paths.
|
|
"""
|
|
ctx = patch_harness(fake_agent) if patch_llm else nullcontext(fake_agent)
|
|
with ctx:
|
|
srv = OctopServer(home=home)
|
|
await srv.start()
|
|
if bind_database and not srv.database_bound:
|
|
await ensure_control_plane_bound(srv)
|
|
# Creating an agent now starts a random-interval sleep (default ON).
|
|
# pytest-asyncio waits for leftover tasks before fixture teardown, so
|
|
# those sleeps hang the suite. Production shutdown still cancels them.
|
|
if srv.app_runtime is not None:
|
|
await srv.app_runtime.proactive_scheduler.shutdown()
|
|
srv.app_runtime.proactive_scheduler.suspend()
|
|
app = build_app(srv)
|
|
try:
|
|
async with httpx.AsyncClient(
|
|
transport=httpx.ASGITransport(app=app),
|
|
base_url="http://testserver",
|
|
) as client:
|
|
# Expose the ASGI app so tests can open WebSocket sessions on
|
|
# this same event loop (tests.support.http.ws_connect); httpx
|
|
# removed AsyncClient.websocket_connect.
|
|
client._octop_app = app # type: ignore[attr-defined]
|
|
yield client, srv
|
|
finally:
|
|
await srv.stop()
|