refactor(inference): remove managed inference routes (#3195)

* refactor(inference): remove managed inference routes

Closes #3172

Remove the inference route control plane, inference.local data path, built-in router crate, and SDK surface. Move inference workloads to explicitly imported provider profiles and native endpoints, with migration cleanup and updated tests and documentation.

Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com>

* fix(policy): preserve alternate upstream isolation

Restore the provider policy activation guard so legacy OpenAI and Anthropic providers configured for alternate base URLs do not grant egress to the built-in public vendor endpoints.

Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com>

---------

Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com>
This commit is contained in:
John T. Myers
2026-09-09 18:47:22 +00:00
committed by GitHub
parent 7f4bd49a47
commit f4dc6be4b2
142 changed files with 1524 additions and 19041 deletions
-4
View File
@@ -9,8 +9,6 @@ from .sandbox import (
ClientCredentialsAuth,
ExecChunk,
ExecResult,
InferenceRouteClient,
InferenceRouteConfig,
Sandbox,
SandboxClient,
SandboxError,
@@ -35,8 +33,6 @@ __all__ = [
"ClientCredentialsAuth",
"ExecChunk",
"ExecResult",
"InferenceRouteClient",
"InferenceRouteConfig",
"Sandbox",
"SandboxClient",
"SandboxError",
-70
View File
@@ -25,8 +25,6 @@ import httpx
from ._proto import (
datamodel_pb2,
inference_pb2,
inference_pb2_grpc,
openshell_pb2,
openshell_pb2_grpc,
)
@@ -1204,74 +1202,6 @@ class SandboxTemplateClient:
return bool(response.deleted)
@dataclass(frozen=True)
class InferenceRouteConfig:
provider_name: str
model_id: str
version: int
class InferenceRouteClient:
"""gRPC client for workspace-scoped inference route configuration."""
def __init__(self, channel: grpc.Channel, *, timeout: float = 30.0) -> None:
self._stub = inference_pb2_grpc.InferenceStub(channel)
self._timeout = timeout
@classmethod
def from_sandbox_client(cls, client: SandboxClient) -> InferenceRouteClient:
return cls(client._channel, timeout=client._timeout)
def set_route(
self,
*,
workspace: str,
provider_name: str,
model_id: str,
no_verify: bool = False,
) -> InferenceRouteConfig:
response = self._stub.SetInferenceRoute(
inference_pb2.SetInferenceRouteRequest(
workspace=workspace,
provider_name=provider_name,
model_id=model_id,
no_verify=no_verify,
),
timeout=self._timeout,
)
return InferenceRouteConfig(
provider_name=response.provider_name,
model_id=response.model_id,
version=response.version,
)
def get_route(self, *, workspace: str) -> InferenceRouteConfig:
response = self._stub.GetInferenceRoute(
inference_pb2.GetInferenceRouteRequest(workspace=workspace),
timeout=self._timeout,
)
return InferenceRouteConfig(
provider_name=response.provider_name,
model_id=response.model_id,
version=response.version,
)
def delete_route(
self,
*,
workspace: str,
route_name: str = "",
) -> bool:
response = self._stub.DeleteInferenceRoute(
inference_pb2.DeleteInferenceRouteRequest(
workspace=workspace,
route_name=route_name,
),
timeout=self._timeout,
)
return response.deleted
@dataclass(frozen=True)
class WorkspaceRef:
name: str
-62
View File
@@ -23,7 +23,6 @@ from openshell.sandbox import (
_PYTHON_CLOUDPICKLE_BOOTSTRAP,
_SANDBOX_PYTHON_BIN,
ClientCredentialsAuth,
InferenceRouteClient,
Sandbox,
SandboxClient,
SandboxError,
@@ -403,34 +402,6 @@ class _FakeStub:
)
class _FakeInferenceStub:
def __init__(self) -> None:
self.set_request = None
self.get_request = None
def SetInferenceRoute(self, request: Any, timeout: float | None = None) -> Any:
self.set_request = request
_ = timeout
class _Response:
provider_name = request.provider_name
model_id = request.model_id
version = 1
return _Response()
def GetInferenceRoute(self, request: Any, timeout: float | None = None) -> Any:
self.get_request = request
_ = timeout
class _Response:
provider_name = "openai-dev"
model_id = "gpt-4.1"
version = 2
return _Response()
def _client_with_fake_stub(stub: object) -> SandboxClient:
client = cast("SandboxClient", object.__new__(SandboxClient))
client._timeout = 30.0
@@ -1877,39 +1848,6 @@ def test_sandbox_wrapper_defaults_match_from_active_cluster(
assert captured["insecure"] is False
def test_inference_set_route_forwards_workspace_and_no_verify() -> None:
stub = _FakeInferenceStub()
client = cast("InferenceRouteClient", object.__new__(InferenceRouteClient))
client._timeout = 30.0
client._stub = cast("Any", stub)
client.set_route(
workspace="production",
provider_name="openai-dev",
model_id="gpt-4.1",
no_verify=True,
)
assert stub.set_request is not None
assert stub.set_request.no_verify is True
assert stub.set_request.workspace == "production"
def test_inference_get_route_forwards_workspace() -> None:
stub = _FakeInferenceStub()
client = cast("InferenceRouteClient", object.__new__(InferenceRouteClient))
client._timeout = 30.0
client._stub = cast("Any", stub)
config = client.get_route(workspace="staging")
assert stub.get_request is not None
assert stub.get_request.workspace == "staging"
assert config.provider_name == "openai-dev"
assert config.model_id == "gpt-4.1"
assert config.version == 2
# ---------------------------------------------------------------------------
# Encoding regression tests (utf-8 explicit on all config file reads/writes)
# ---------------------------------------------------------------------------
+1 -1
View File
@@ -32,7 +32,7 @@ def _wheel_files() -> set[str]:
"openshell/py.typed",
"openshell/_proto/__init__.py",
}
for stem in ("datamodel", "inference", "openshell", "options", "sandbox"):
for stem in ("datamodel", "openshell", "options", "sandbox"):
files.add(f"openshell/_proto/{stem}_pb2.py")
files.add(f"openshell/_proto/{stem}_pb2.pyi")
files.add(f"openshell/_proto/{stem}_pb2_grpc.py")