mirror of
https://github.com/NVIDIA/OpenShell.git
synced 2026-10-02 07:34:45 +08:00
refactor(inference): remove managed inference routes (#3195)
* refactor(inference): remove managed inference routes Closes #3172 Remove the inference route control plane, inference.local data path, built-in router crate, and SDK surface. Move inference workloads to explicitly imported provider profiles and native endpoints, with migration cleanup and updated tests and documentation. Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com> * fix(policy): preserve alternate upstream isolation Restore the provider policy activation guard so legacy OpenAI and Anthropic providers configured for alternate base URLs do not grant egress to the built-in public vendor endpoints. Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com> --------- Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com>
This commit is contained in:
@@ -9,8 +9,6 @@ from .sandbox import (
|
||||
ClientCredentialsAuth,
|
||||
ExecChunk,
|
||||
ExecResult,
|
||||
InferenceRouteClient,
|
||||
InferenceRouteConfig,
|
||||
Sandbox,
|
||||
SandboxClient,
|
||||
SandboxError,
|
||||
@@ -35,8 +33,6 @@ __all__ = [
|
||||
"ClientCredentialsAuth",
|
||||
"ExecChunk",
|
||||
"ExecResult",
|
||||
"InferenceRouteClient",
|
||||
"InferenceRouteConfig",
|
||||
"Sandbox",
|
||||
"SandboxClient",
|
||||
"SandboxError",
|
||||
|
||||
@@ -25,8 +25,6 @@ import httpx
|
||||
|
||||
from ._proto import (
|
||||
datamodel_pb2,
|
||||
inference_pb2,
|
||||
inference_pb2_grpc,
|
||||
openshell_pb2,
|
||||
openshell_pb2_grpc,
|
||||
)
|
||||
@@ -1204,74 +1202,6 @@ class SandboxTemplateClient:
|
||||
return bool(response.deleted)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InferenceRouteConfig:
|
||||
provider_name: str
|
||||
model_id: str
|
||||
version: int
|
||||
|
||||
|
||||
class InferenceRouteClient:
|
||||
"""gRPC client for workspace-scoped inference route configuration."""
|
||||
|
||||
def __init__(self, channel: grpc.Channel, *, timeout: float = 30.0) -> None:
|
||||
self._stub = inference_pb2_grpc.InferenceStub(channel)
|
||||
self._timeout = timeout
|
||||
|
||||
@classmethod
|
||||
def from_sandbox_client(cls, client: SandboxClient) -> InferenceRouteClient:
|
||||
return cls(client._channel, timeout=client._timeout)
|
||||
|
||||
def set_route(
|
||||
self,
|
||||
*,
|
||||
workspace: str,
|
||||
provider_name: str,
|
||||
model_id: str,
|
||||
no_verify: bool = False,
|
||||
) -> InferenceRouteConfig:
|
||||
response = self._stub.SetInferenceRoute(
|
||||
inference_pb2.SetInferenceRouteRequest(
|
||||
workspace=workspace,
|
||||
provider_name=provider_name,
|
||||
model_id=model_id,
|
||||
no_verify=no_verify,
|
||||
),
|
||||
timeout=self._timeout,
|
||||
)
|
||||
return InferenceRouteConfig(
|
||||
provider_name=response.provider_name,
|
||||
model_id=response.model_id,
|
||||
version=response.version,
|
||||
)
|
||||
|
||||
def get_route(self, *, workspace: str) -> InferenceRouteConfig:
|
||||
response = self._stub.GetInferenceRoute(
|
||||
inference_pb2.GetInferenceRouteRequest(workspace=workspace),
|
||||
timeout=self._timeout,
|
||||
)
|
||||
return InferenceRouteConfig(
|
||||
provider_name=response.provider_name,
|
||||
model_id=response.model_id,
|
||||
version=response.version,
|
||||
)
|
||||
|
||||
def delete_route(
|
||||
self,
|
||||
*,
|
||||
workspace: str,
|
||||
route_name: str = "",
|
||||
) -> bool:
|
||||
response = self._stub.DeleteInferenceRoute(
|
||||
inference_pb2.DeleteInferenceRouteRequest(
|
||||
workspace=workspace,
|
||||
route_name=route_name,
|
||||
),
|
||||
timeout=self._timeout,
|
||||
)
|
||||
return response.deleted
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class WorkspaceRef:
|
||||
name: str
|
||||
|
||||
@@ -23,7 +23,6 @@ from openshell.sandbox import (
|
||||
_PYTHON_CLOUDPICKLE_BOOTSTRAP,
|
||||
_SANDBOX_PYTHON_BIN,
|
||||
ClientCredentialsAuth,
|
||||
InferenceRouteClient,
|
||||
Sandbox,
|
||||
SandboxClient,
|
||||
SandboxError,
|
||||
@@ -403,34 +402,6 @@ class _FakeStub:
|
||||
)
|
||||
|
||||
|
||||
class _FakeInferenceStub:
|
||||
def __init__(self) -> None:
|
||||
self.set_request = None
|
||||
self.get_request = None
|
||||
|
||||
def SetInferenceRoute(self, request: Any, timeout: float | None = None) -> Any:
|
||||
self.set_request = request
|
||||
_ = timeout
|
||||
|
||||
class _Response:
|
||||
provider_name = request.provider_name
|
||||
model_id = request.model_id
|
||||
version = 1
|
||||
|
||||
return _Response()
|
||||
|
||||
def GetInferenceRoute(self, request: Any, timeout: float | None = None) -> Any:
|
||||
self.get_request = request
|
||||
_ = timeout
|
||||
|
||||
class _Response:
|
||||
provider_name = "openai-dev"
|
||||
model_id = "gpt-4.1"
|
||||
version = 2
|
||||
|
||||
return _Response()
|
||||
|
||||
|
||||
def _client_with_fake_stub(stub: object) -> SandboxClient:
|
||||
client = cast("SandboxClient", object.__new__(SandboxClient))
|
||||
client._timeout = 30.0
|
||||
@@ -1877,39 +1848,6 @@ def test_sandbox_wrapper_defaults_match_from_active_cluster(
|
||||
assert captured["insecure"] is False
|
||||
|
||||
|
||||
def test_inference_set_route_forwards_workspace_and_no_verify() -> None:
|
||||
stub = _FakeInferenceStub()
|
||||
client = cast("InferenceRouteClient", object.__new__(InferenceRouteClient))
|
||||
client._timeout = 30.0
|
||||
client._stub = cast("Any", stub)
|
||||
|
||||
client.set_route(
|
||||
workspace="production",
|
||||
provider_name="openai-dev",
|
||||
model_id="gpt-4.1",
|
||||
no_verify=True,
|
||||
)
|
||||
|
||||
assert stub.set_request is not None
|
||||
assert stub.set_request.no_verify is True
|
||||
assert stub.set_request.workspace == "production"
|
||||
|
||||
|
||||
def test_inference_get_route_forwards_workspace() -> None:
|
||||
stub = _FakeInferenceStub()
|
||||
client = cast("InferenceRouteClient", object.__new__(InferenceRouteClient))
|
||||
client._timeout = 30.0
|
||||
client._stub = cast("Any", stub)
|
||||
|
||||
config = client.get_route(workspace="staging")
|
||||
|
||||
assert stub.get_request is not None
|
||||
assert stub.get_request.workspace == "staging"
|
||||
assert config.provider_name == "openai-dev"
|
||||
assert config.model_id == "gpt-4.1"
|
||||
assert config.version == 2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Encoding regression tests (utf-8 explicit on all config file reads/writes)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -32,7 +32,7 @@ def _wheel_files() -> set[str]:
|
||||
"openshell/py.typed",
|
||||
"openshell/_proto/__init__.py",
|
||||
}
|
||||
for stem in ("datamodel", "inference", "openshell", "options", "sandbox"):
|
||||
for stem in ("datamodel", "openshell", "options", "sandbox"):
|
||||
files.add(f"openshell/_proto/{stem}_pb2.py")
|
||||
files.add(f"openshell/_proto/{stem}_pb2.pyi")
|
||||
files.add(f"openshell/_proto/{stem}_pb2_grpc.py")
|
||||
|
||||
Reference in New Issue
Block a user