mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 01:54:39 +08:00
feat(phase-13/11): teach per-request sampling
Sampling calls must carry current capabilities and stateless response semantics.
This commit is contained in:
@@ -1,4 +1,4 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 960 520" font-family="Georgia, 'Times New Roman', serif">
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 960 560" font-family="Georgia, 'Times New Roman', serif">
|
||||
<defs>
|
||||
<marker id="arrow" viewBox="0 0 10 10" refX="9" refY="5" markerWidth="7" markerHeight="7" orient="auto">
|
||||
<path d="M0,0 L10,5 L0,10 z" fill="#1a1a1a"/>
|
||||
@@ -12,66 +12,57 @@
|
||||
.head { font-size: 13px; font-weight: 700; fill: #1a1a1a; }
|
||||
.step { font-size: 12px; font-family: 'Menlo', monospace; fill: #222; }
|
||||
.small { font-size: 11px; font-family: 'Menlo', monospace; fill: #333; }
|
||||
.caption { font-size: 11px; fill: #555; font-style: italic; }
|
||||
.edge { stroke: #1a1a1a; stroke-width: 1.5; fill: none; }
|
||||
</style>
|
||||
</defs>
|
||||
|
||||
<text x="480" y="26" text-anchor="middle" class="title">server-hosted agent loop via sampling (no server API key)</text>
|
||||
<text x="480" y="26" text-anchor="middle" class="title">2026-07-28 model input: direct integration or stateless MRTR</text>
|
||||
|
||||
<rect x="40" y="60" width="260" height="440" class="cool"/>
|
||||
<text x="170" y="82" text-anchor="middle" class="head">client (user's host)</text>
|
||||
<text x="56" y="108" class="small">holds LLM credentials</text>
|
||||
<text x="56" y="128" class="step">LLM provider</text>
|
||||
<text x="56" y="146" class="small">(Claude, GPT, Gemini,</text>
|
||||
<text x="56" y="162" class="small">local Ollama, ...)</text>
|
||||
<text x="56" y="194" class="step">sampling handler</text>
|
||||
<text x="56" y="212" class="small">runs LLM on server's</text>
|
||||
<text x="56" y="228" class="small">request; returns completion</text>
|
||||
<text x="56" y="260" class="step">safety</text>
|
||||
<text x="56" y="278" class="small">- shows user the request</text>
|
||||
<text x="56" y="294" class="small">- applies per-session rate</text>
|
||||
<text x="56" y="310" class="small">- honors modelPreferences</text>
|
||||
<text x="56" y="342" class="step">billing</text>
|
||||
<text x="56" y="360" class="small">user pays for sampling</text>
|
||||
<text x="56" y="376" class="small">calls via their own key</text>
|
||||
<rect x="40" y="55" width="880" height="70" class="hot"/>
|
||||
<text x="480" y="78" text-anchor="middle" class="head">architecture decision</text>
|
||||
<text x="60" y="100" class="small">New server: call a model provider directly. Compatibility need: carry deprecated Sampling through MRTR.</text>
|
||||
|
||||
<path d="M300,180 L420,180" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="360" y="174" text-anchor="middle" class="small">tools/call summarize_repo</text>
|
||||
<rect x="40" y="145" width="260" height="350" class="cool"/>
|
||||
<text x="170" y="170" text-anchor="middle" class="head">client and host model</text>
|
||||
<text x="58" y="202" class="step">every POST includes _meta</text>
|
||||
<text x="58" y="220" class="small">protocolVersion: 2026-07-28</text>
|
||||
<text x="58" y="238" class="small">clientCapabilities: sampling</text>
|
||||
<text x="58" y="272" class="step">fulfill inputRequests</text>
|
||||
<text x="58" y="290" class="small">apply approval and budgets</text>
|
||||
<text x="58" y="308" class="small">validate model output</text>
|
||||
<text x="58" y="342" class="step">retry original method</text>
|
||||
<text x="58" y="360" class="small">fresh JSON-RPC id</text>
|
||||
<text x="58" y="378" class="small">inputResponses for this round</text>
|
||||
<text x="58" y="396" class="small">exact requestState echo</text>
|
||||
<text x="58" y="438" class="small">No initialize</text>
|
||||
<text x="58" y="456" class="small">No protocol session</text>
|
||||
<text x="58" y="474" class="small">No reverse request channel</text>
|
||||
|
||||
<path d="M660,260 L300,260" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="254" text-anchor="middle" class="small">sampling/createMessage {pick files}</text>
|
||||
<rect x="660" y="145" width="260" height="350" class="cold"/>
|
||||
<text x="790" y="170" text-anchor="middle" class="head">stateless MCP server</text>
|
||||
<text x="678" y="202" class="step">discover + cached tools/list</text>
|
||||
<text x="678" y="220" class="small">version -32022 | capability -32021</text>
|
||||
<text x="678" y="252" class="step">round 1: pick files</text>
|
||||
<text x="678" y="270" class="small">resultType: input_required</text>
|
||||
<text x="678" y="288" class="small">sampling/createMessage embedded</text>
|
||||
<text x="678" y="320" class="step">round 2: summarize</text>
|
||||
<text x="678" y="338" class="small">validate first response</text>
|
||||
<text x="678" y="356" class="small">return next input_required</text>
|
||||
<text x="678" y="388" class="step">round 3: complete</text>
|
||||
<text x="678" y="406" class="small">resultType: complete</text>
|
||||
<text x="678" y="438" class="step">requestState</text>
|
||||
<text x="678" y="456" class="small">HMAC binds principal, args,</text>
|
||||
<text x="678" y="474" class="small">phase, and short expiry</text>
|
||||
|
||||
<path d="M300,310 L660,310" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="304" text-anchor="middle" class="small"><- completion {picked: [...]}</text>
|
||||
<path d="M300,230 L660,230" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="222" text-anchor="middle" class="small">1. tools/call, id 1</text>
|
||||
<path d="M660,285 L300,285" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="277" text-anchor="middle" class="small">2. input_required + signed state</text>
|
||||
<path d="M300,345 L660,345" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="337" text-anchor="middle" class="small">3. tools/call retry, id 2</text>
|
||||
<path d="M660,400 L300,400" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="392" text-anchor="middle" class="small">4. next input_required or complete</text>
|
||||
|
||||
<path d="M660,370 L300,370" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="364" text-anchor="middle" class="small">sampling/createMessage {summarize}</text>
|
||||
|
||||
<path d="M300,420 L660,420" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="480" y="414" text-anchor="middle" class="small"><- completion {summary}</text>
|
||||
|
||||
<path d="M660,460 L420,460" class="edge" marker-end="url(#arrow)"/>
|
||||
<text x="540" y="454" text-anchor="middle" class="small">tools/call result {summary}</text>
|
||||
|
||||
<rect x="660" y="60" width="260" height="440" class="cold"/>
|
||||
<text x="790" y="82" text-anchor="middle" class="head">server (summarize_repo)</text>
|
||||
<text x="676" y="108" class="small">NO LLM credentials</text>
|
||||
<text x="676" y="128" class="step">algorithm</text>
|
||||
<text x="676" y="146" class="small">1. walk file list</text>
|
||||
<text x="676" y="162" class="small">2. ask client to pick</text>
|
||||
<text x="676" y="178" class="small">3. read picked files</text>
|
||||
<text x="676" y="194" class="small">4. ask client to summarize</text>
|
||||
<text x="676" y="210" class="small">5. return result</text>
|
||||
<text x="676" y="240" class="step">modelPreferences</text>
|
||||
<text x="676" y="258" class="small">pick files: cost 0.5, int 0.2</text>
|
||||
<text x="676" y="274" class="small">summarize : cost 0.2, int 0.6</text>
|
||||
<text x="676" y="306" class="step">guardrails</text>
|
||||
<text x="676" y="324" class="small">- max_samples_per_tool</text>
|
||||
<text x="676" y="340" class="small">- includeContext: "none"</text>
|
||||
<text x="676" y="356" class="small">- no covert sampling</text>
|
||||
<text x="676" y="388" class="step">SEP-1577 (drift-risk)</text>
|
||||
<text x="676" y="406" class="small">tools[] inside sampling</text>
|
||||
<text x="676" y="422" class="small">for server-hosted ReAct</text>
|
||||
<text x="676" y="438" class="small">SDK shapes still settling</text>
|
||||
<rect x="40" y="515" width="880" height="30" class="box"/>
|
||||
<text x="480" y="535" text-anchor="middle" class="small">Any server instance can process a retry because the request and integrity-protected state are self-contained.</text>
|
||||
</svg>
|
||||
|
||||
|
Before Width: | Height: | Size: 4.4 KiB After Width: | Height: | Size: 4.3 KiB |
@@ -1,154 +1,383 @@
|
||||
"""Phase 13 Lesson 11 - MCP sampling harness (server -> client LLM calls).
|
||||
"""Phase 13 Lesson 11: model input through stateless MCP MRTR.
|
||||
|
||||
Simulated server-to-client sampling:
|
||||
- Server's summarize_repo tool runs two sampling rounds (pick files, then
|
||||
synthesize) by calling a 'fake_client_sample' stand-in for the client.
|
||||
- Rate-limited at max_samples_per_tool to prevent loop bombs.
|
||||
- ModelPreferences are printed so you can see the cost/speed/intelligence
|
||||
trade-off shape.
|
||||
|
||||
Stdlib only.
|
||||
|
||||
Run: python code/main.py
|
||||
Lesson: ../docs/en.md
|
||||
Specification: https://modelcontextprotocol.io/specification/2026-07-28/basic/patterns/mrtr
|
||||
This example uses only Python's standard library.
|
||||
Run: python3 main.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
from dataclasses import dataclass, field
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
PROTOCOL_VERSION = "2026-07-28"
|
||||
PROTOCOL_META = "io.modelcontextprotocol/protocolVersion"
|
||||
CAPABILITIES_META = "io.modelcontextprotocol/clientCapabilities"
|
||||
CLIENT_INFO_META = "io.modelcontextprotocol/clientInfo"
|
||||
SERVER_INFO_META = "io.modelcontextprotocol/serverInfo"
|
||||
SERVER_SECRET = b"lesson-11-demo-secret-change-in-production"
|
||||
|
||||
FAKE_REPO = {
|
||||
"README.md": "This repo implements the toy MCP notes server.",
|
||||
"server.py": "def dispatch(msg): ... handler code ...",
|
||||
"client.py": "def connect(): ... subprocess Popen ...",
|
||||
"README.md": "A small notes server used to teach stateless MCP.",
|
||||
"server.py": "def dispatch(request): return handle(request)",
|
||||
"client.py": "def retry(request, inputs, state): ...",
|
||||
"LICENSE": "MIT",
|
||||
"tests/test_server.py": "def test_initialize(): ...",
|
||||
"assets/diagram.svg": "<svg>...</svg>",
|
||||
"docs/intro.md": "## Introduction to the toy notes server",
|
||||
"tests/test_server.py": "def test_stateless_retry(): ...",
|
||||
"docs/intro.md": "The server has no protocol session state.",
|
||||
}
|
||||
|
||||
|
||||
CANNED_RESPONSES = {
|
||||
"pick": json.dumps(["README.md", "server.py", "docs/intro.md"]),
|
||||
"summarize": "This repo is a toy MCP server teaching the sampling loop. "
|
||||
"The server dispatches JSON-RPC methods; clients drive it over stdio. "
|
||||
"Documentation in docs/ introduces the pattern end to end.",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class SampleRequest:
|
||||
messages: list[dict]
|
||||
system_prompt: str
|
||||
model_preferences: dict
|
||||
max_tokens: int = 1024
|
||||
include_context: str = "none"
|
||||
tools: list[dict] | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class SampleResponse:
|
||||
role: str
|
||||
content: dict
|
||||
model: str
|
||||
stop_reason: str
|
||||
|
||||
|
||||
def fake_client_sample(req: SampleRequest) -> SampleResponse:
|
||||
"""Stand-in for the client's LLM. Picks a canned response by keyword."""
|
||||
text = req.messages[-1]["content"]["text"].lower()
|
||||
if "pick" in text or "choose" in text:
|
||||
body = CANNED_RESPONSES["pick"]
|
||||
else:
|
||||
body = CANNED_RESPONSES["summarize"]
|
||||
return SampleResponse(
|
||||
role="assistant",
|
||||
content={"type": "text", "text": body},
|
||||
model="claude-3-5-sonnet-fake",
|
||||
stop_reason="endTurn",
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SamplingBudget:
|
||||
used: int = 0
|
||||
max_samples_per_tool: int = 5
|
||||
|
||||
|
||||
def sample(req: SampleRequest, budget: SamplingBudget) -> SampleResponse:
|
||||
if budget.used >= budget.max_samples_per_tool:
|
||||
raise RuntimeError("sampling rate limit exceeded (loop bomb guard)")
|
||||
budget.used += 1
|
||||
print(f" [sample #{budget.used}] model_prefs={req.model_preferences} "
|
||||
f"includeContext={req.include_context!r}")
|
||||
print(f" system: {req.system_prompt[:60]}...")
|
||||
print(f" user : {req.messages[-1]['content']['text'][:60]}...")
|
||||
resp = fake_client_sample(req)
|
||||
print(f" <- model={resp.model} stop={resp.stop_reason} "
|
||||
f"len={len(resp.content['text'])}")
|
||||
return resp
|
||||
|
||||
|
||||
def summarize_repo_tool(args: dict) -> dict:
|
||||
budget = SamplingBudget()
|
||||
|
||||
pick_req = SampleRequest(
|
||||
messages=[{"role": "user", "content": {"type": "text", "text":
|
||||
"Given this file list, pick five files most likely to describe the repo's purpose. "
|
||||
f"Files: {list(FAKE_REPO.keys())}. Reply as a JSON array of filenames."}}],
|
||||
system_prompt="You select representative files for repo summarization.",
|
||||
model_preferences={
|
||||
"costPriority": 0.5,
|
||||
"speedPriority": 0.3,
|
||||
"intelligencePriority": 0.2,
|
||||
"hints": [{"name": "claude-3-5-haiku"}],
|
||||
TOOLS = [
|
||||
{
|
||||
"name": "summarize_repo",
|
||||
"description": "Select representative repository files and summarize them.",
|
||||
"inputSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"audience": {
|
||||
"type": "string",
|
||||
"description": "Audience for the final repository summary.",
|
||||
}
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
max_tokens=256,
|
||||
include_context="none",
|
||||
)
|
||||
pick_resp = sample(pick_req, budget)
|
||||
picked = json.loads(pick_resp.content["text"])
|
||||
print(f" picked files: {picked}")
|
||||
}
|
||||
]
|
||||
|
||||
combined = "\n\n".join(f"=== {f} ===\n{FAKE_REPO[f]}" for f in picked if f in FAKE_REPO)
|
||||
|
||||
summ_req = SampleRequest(
|
||||
messages=[{"role": "user", "content": {"type": "text", "text":
|
||||
f"Summarize the repo in three paragraphs given these files:\n\n{combined}"}}],
|
||||
system_prompt="You write concise, accurate repo summaries.",
|
||||
model_preferences={
|
||||
"costPriority": 0.2,
|
||||
"speedPriority": 0.2,
|
||||
"intelligencePriority": 0.6,
|
||||
"hints": [{"name": "claude-3-5-sonnet"}],
|
||||
},
|
||||
max_tokens=512,
|
||||
include_context="none",
|
||||
)
|
||||
summ_resp = sample(summ_req, budget)
|
||||
@dataclass
|
||||
class McpError(Exception):
|
||||
code: int
|
||||
message: str
|
||||
data: dict[str, Any] | None = None
|
||||
|
||||
|
||||
def request_meta(*, sampling: bool = True) -> dict[str, Any]:
|
||||
capabilities: dict[str, Any] = {"sampling": {}} if sampling else {}
|
||||
return {
|
||||
"content": [{"type": "text", "text": summ_resp.content["text"]}],
|
||||
"isError": False,
|
||||
"_meta": {"samplesUsed": budget.used},
|
||||
PROTOCOL_META: PROTOCOL_VERSION,
|
||||
CAPABILITIES_META: capabilities,
|
||||
CLIENT_INFO_META: {"name": "lesson-client", "version": "1.0.0"},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
print("=" * 72)
|
||||
print("PHASE 13 LESSON 11 - MCP SAMPLING HARNESS")
|
||||
print("=" * 72)
|
||||
print()
|
||||
print("summarize_repo invoked (no server-side LLM credentials)")
|
||||
print("-" * 72)
|
||||
def _server_meta() -> dict[str, Any]:
|
||||
return {SERVER_INFO_META: {"name": "mrtr-demo", "version": "1.0.0"}}
|
||||
|
||||
|
||||
def complete(**fields: Any) -> dict[str, Any]:
|
||||
return {"resultType": "complete", **fields, "_meta": _server_meta()}
|
||||
|
||||
|
||||
def validate_request_meta(params: dict[str, Any]) -> dict[str, Any]:
|
||||
meta = params.get("_meta")
|
||||
if not isinstance(meta, dict):
|
||||
raise McpError(-32602, "missing request _meta")
|
||||
requested_version = meta.get(PROTOCOL_META)
|
||||
if not isinstance(requested_version, str):
|
||||
raise McpError(-32602, "missing protocol version")
|
||||
if requested_version != PROTOCOL_VERSION:
|
||||
raise McpError(
|
||||
-32022,
|
||||
"unsupported protocol version",
|
||||
{"supported": [PROTOCOL_VERSION], "requested": requested_version},
|
||||
)
|
||||
capabilities = meta.get(CAPABILITIES_META)
|
||||
if not isinstance(capabilities, dict):
|
||||
raise McpError(-32602, "missing client capabilities")
|
||||
return meta
|
||||
|
||||
|
||||
def server_discover(params: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_request_meta(params)
|
||||
return complete(
|
||||
supportedVersions=[PROTOCOL_VERSION],
|
||||
capabilities={"tools": {}},
|
||||
ttlMs=300_000,
|
||||
cacheScope="public",
|
||||
)
|
||||
|
||||
|
||||
def tools_list(params: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_request_meta(params)
|
||||
return complete(
|
||||
tools=sorted(TOOLS, key=lambda tool: tool["name"]),
|
||||
ttlMs=60_000,
|
||||
cacheScope="public",
|
||||
)
|
||||
|
||||
|
||||
def _b64(data: bytes) -> str:
|
||||
return base64.urlsafe_b64encode(data).decode("ascii").rstrip("=")
|
||||
|
||||
|
||||
def _unb64(value: str) -> bytes:
|
||||
return base64.urlsafe_b64decode(value + "=" * (-len(value) % 4))
|
||||
|
||||
|
||||
def _arguments_digest(arguments: dict[str, Any]) -> str:
|
||||
encoded = json.dumps(arguments, sort_keys=True, separators=(",", ":")).encode()
|
||||
return hashlib.sha256(encoded).hexdigest()
|
||||
|
||||
|
||||
def seal_request_state(payload: dict[str, Any]) -> str:
|
||||
body = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()
|
||||
signature = hmac.new(SERVER_SECRET, body, hashlib.sha256).digest()
|
||||
return f"{_b64(body)}.{_b64(signature)}"
|
||||
|
||||
|
||||
def verify_request_state(
|
||||
token: str,
|
||||
*,
|
||||
principal: str,
|
||||
arguments: dict[str, Any],
|
||||
now: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
try:
|
||||
result = summarize_repo_tool({})
|
||||
print("\n result.content[0].text:")
|
||||
print(f" {result['content'][0]['text']}")
|
||||
print(f"\n samples used: {result['_meta']['samplesUsed']}")
|
||||
except RuntimeError as e:
|
||||
print(f" loop-bomb guard triggered: {e}")
|
||||
body_part, signature_part = token.split(".", 1)
|
||||
body = _unb64(body_part)
|
||||
supplied = _unb64(signature_part)
|
||||
except (ValueError, TypeError) as exc:
|
||||
raise McpError(-32602, "invalid requestState encoding") from exc
|
||||
expected = hmac.new(SERVER_SECRET, body, hashlib.sha256).digest()
|
||||
if not hmac.compare_digest(supplied, expected):
|
||||
raise McpError(-32602, "requestState integrity check failed")
|
||||
try:
|
||||
state = json.loads(body)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise McpError(-32602, "invalid requestState payload") from exc
|
||||
if state.get("principal") != principal:
|
||||
raise McpError(-32602, "requestState principal mismatch")
|
||||
if state.get("method") != "tools/call":
|
||||
raise McpError(-32602, "requestState method mismatch")
|
||||
if state.get("argumentsDigest") != _arguments_digest(arguments):
|
||||
raise McpError(-32602, "requestState arguments mismatch")
|
||||
if int(state.get("expiresAt", 0)) < (int(time.time()) if now is None else now):
|
||||
raise McpError(-32602, "requestState expired")
|
||||
return state
|
||||
|
||||
|
||||
def _sampling_request(prompt: str, *, intelligence: float) -> dict[str, Any]:
|
||||
return {
|
||||
"method": "sampling/createMessage",
|
||||
"params": {
|
||||
"messages": [
|
||||
{"role": "user", "content": {"type": "text", "text": prompt}}
|
||||
],
|
||||
"systemPrompt": "Return only the requested value.",
|
||||
"modelPreferences": {
|
||||
"costPriority": round(1.0 - intelligence, 2),
|
||||
"intelligencePriority": intelligence,
|
||||
},
|
||||
"maxTokens": 400,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _input_required(
|
||||
key: str,
|
||||
prompt: str,
|
||||
state: dict[str, Any],
|
||||
*,
|
||||
intelligence: float,
|
||||
) -> dict[str, Any]:
|
||||
return {
|
||||
"resultType": "input_required",
|
||||
"inputRequests": {key: _sampling_request(prompt, intelligence=intelligence)},
|
||||
"requestState": seal_request_state(state),
|
||||
"_meta": _server_meta(),
|
||||
}
|
||||
|
||||
|
||||
def _sampling_text(input_responses: Any, key: str) -> str:
|
||||
if not isinstance(input_responses, dict):
|
||||
raise McpError(-32602, "inputResponses must be an object")
|
||||
response = input_responses.get(key)
|
||||
if not isinstance(response, dict):
|
||||
raise McpError(-32602, f"missing input response: {key}")
|
||||
content = response.get("content")
|
||||
if not isinstance(content, dict) or content.get("type") != "text":
|
||||
raise McpError(-32602, f"invalid sampling response: {key}")
|
||||
text = content.get("text")
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
raise McpError(-32602, f"empty sampling response: {key}")
|
||||
return text
|
||||
|
||||
|
||||
def tools_call(params: dict[str, Any], *, principal: str) -> dict[str, Any]:
|
||||
meta = validate_request_meta(params)
|
||||
if params.get("name") != "summarize_repo":
|
||||
raise McpError(-32602, "unknown tool")
|
||||
arguments = params.get("arguments", {})
|
||||
if not isinstance(arguments, dict):
|
||||
raise McpError(-32602, "arguments must be an object")
|
||||
capabilities = meta[CAPABILITIES_META]
|
||||
if "sampling" not in capabilities:
|
||||
raise McpError(
|
||||
-32021,
|
||||
"missing required client capability",
|
||||
{"requiredCapabilities": {"sampling": {}}},
|
||||
)
|
||||
|
||||
state_token = params.get("requestState")
|
||||
if state_token is None:
|
||||
state = {
|
||||
"phase": "pick",
|
||||
"principal": principal,
|
||||
"method": "tools/call",
|
||||
"argumentsDigest": _arguments_digest(arguments),
|
||||
"expiresAt": int(time.time()) + 300,
|
||||
}
|
||||
prompt = (
|
||||
"Choose three representative files and return a JSON array. Files: "
|
||||
+ json.dumps(sorted(FAKE_REPO))
|
||||
)
|
||||
return _input_required("pick_files", prompt, state, intelligence=0.2)
|
||||
|
||||
if not isinstance(state_token, str):
|
||||
raise McpError(-32602, "requestState must be a string")
|
||||
state = verify_request_state(
|
||||
state_token,
|
||||
principal=principal,
|
||||
arguments=arguments,
|
||||
)
|
||||
responses = params.get("inputResponses")
|
||||
|
||||
if state["phase"] == "pick":
|
||||
raw_picks = _sampling_text(responses, "pick_files")
|
||||
try:
|
||||
picks = json.loads(raw_picks)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise McpError(-32602, "pick_files must return JSON") from exc
|
||||
if not isinstance(picks, list) or not all(isinstance(item, str) for item in picks):
|
||||
raise McpError(-32602, "pick_files must return a string array")
|
||||
picked = [name for name in picks if name in FAKE_REPO][:3]
|
||||
if not picked:
|
||||
raise McpError(-32602, "pick_files returned no known files")
|
||||
combined = "\n\n".join(f"{name}: {FAKE_REPO[name]}" for name in picked)
|
||||
next_state = {
|
||||
**state,
|
||||
"phase": "summarize",
|
||||
"picked": picked,
|
||||
"expiresAt": int(time.time()) + 300,
|
||||
}
|
||||
prompt = "Summarize these files in two sentences:\n\n" + combined
|
||||
return _input_required("summary", prompt, next_state, intelligence=0.8)
|
||||
|
||||
if state["phase"] == "summarize":
|
||||
summary = _sampling_text(responses, "summary")
|
||||
return complete(
|
||||
content=[{"type": "text", "text": summary}],
|
||||
structuredContent={"picked": state["picked"], "summary": summary},
|
||||
isError=False,
|
||||
)
|
||||
|
||||
raise McpError(-32602, "unknown requestState phase")
|
||||
|
||||
|
||||
def dispatch(
|
||||
request: dict[str, Any],
|
||||
*,
|
||||
principal: str = "user-42",
|
||||
) -> dict[str, Any] | None:
|
||||
is_notification = "id" not in request
|
||||
request_id = request.get("id")
|
||||
try:
|
||||
method = request.get("method")
|
||||
params = request.get("params", {})
|
||||
if not isinstance(params, dict):
|
||||
raise McpError(-32602, "params must be an object")
|
||||
if method == "server/discover":
|
||||
result = server_discover(params)
|
||||
elif method == "tools/list":
|
||||
result = tools_list(params)
|
||||
elif method == "tools/call":
|
||||
result = tools_call(params, principal=principal)
|
||||
else:
|
||||
raise McpError(-32601, "method not found")
|
||||
if is_notification:
|
||||
return None
|
||||
return {"jsonrpc": "2.0", "id": request_id, "result": result}
|
||||
except McpError as exc:
|
||||
if is_notification:
|
||||
return None
|
||||
error: dict[str, Any] = {"code": exc.code, "message": exc.message}
|
||||
if exc.data is not None:
|
||||
error["data"] = exc.data
|
||||
return {"jsonrpc": "2.0", "id": request_id, "error": error}
|
||||
|
||||
|
||||
def fake_host_model(input_request: dict[str, Any]) -> dict[str, Any]:
|
||||
prompt = input_request["params"]["messages"][-1]["content"]["text"]
|
||||
if "Choose three" in prompt:
|
||||
text = json.dumps(["README.md", "server.py", "docs/intro.md"])
|
||||
else:
|
||||
text = (
|
||||
"This repository demonstrates a stateless MCP server and client retry loop. "
|
||||
"Each MRTR round carries all required state without a protocol session."
|
||||
)
|
||||
return {
|
||||
"role": "assistant",
|
||||
"content": {"type": "text", "text": text},
|
||||
"model": "host-model",
|
||||
"stopReason": "endTurn",
|
||||
}
|
||||
|
||||
|
||||
def run_mrtr() -> tuple[dict[str, Any], list[int]]:
|
||||
base_params = {
|
||||
"name": "summarize_repo",
|
||||
"arguments": {"audience": "developer"},
|
||||
"_meta": request_meta(),
|
||||
}
|
||||
request_id = 1
|
||||
response = dispatch(
|
||||
{"jsonrpc": "2.0", "id": request_id, "method": "tools/call", "params": base_params}
|
||||
)
|
||||
seen_ids = [request_id]
|
||||
|
||||
while response.get("result", {}).get("resultType") == "input_required":
|
||||
pending = response["result"]
|
||||
fulfilled = {
|
||||
key: fake_host_model(input_request)
|
||||
for key, input_request in pending["inputRequests"].items()
|
||||
}
|
||||
request_id += 1
|
||||
seen_ids.append(request_id)
|
||||
retry_params = {
|
||||
**base_params,
|
||||
"inputResponses": fulfilled,
|
||||
"requestState": pending["requestState"],
|
||||
}
|
||||
response = dispatch(
|
||||
{
|
||||
"jsonrpc": "2.0",
|
||||
"id": request_id,
|
||||
"method": "tools/call",
|
||||
"params": retry_params,
|
||||
}
|
||||
)
|
||||
return response, seen_ids
|
||||
|
||||
|
||||
def main() -> None:
|
||||
discovery = dispatch(
|
||||
{
|
||||
"jsonrpc": "2.0",
|
||||
"id": 0,
|
||||
"method": "server/discover",
|
||||
"params": {"_meta": request_meta()},
|
||||
}
|
||||
)
|
||||
print("discover:", json.dumps(discovery["result"], indent=2))
|
||||
response, request_ids = run_mrtr()
|
||||
print("independent request ids:", request_ids)
|
||||
print("final:", json.dumps(response["result"], indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
"""Tests for the stateless MCP MRTR sampling migration lesson."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
MODULE_PATH = Path(__file__).resolve().parents[1] / "main.py"
|
||||
SPEC = importlib.util.spec_from_file_location("lesson11_main", MODULE_PATH)
|
||||
assert SPEC and SPEC.loader
|
||||
main = importlib.util.module_from_spec(SPEC)
|
||||
sys.modules[SPEC.name] = main
|
||||
SPEC.loader.exec_module(main)
|
||||
|
||||
|
||||
def tool_request(request_id: int = 1, *, sampling: bool = True) -> dict:
|
||||
return {
|
||||
"jsonrpc": "2.0",
|
||||
"id": request_id,
|
||||
"method": "tools/call",
|
||||
"params": {
|
||||
"name": "summarize_repo",
|
||||
"arguments": {"audience": "developer"},
|
||||
"_meta": main.request_meta(sampling=sampling),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
class SamplingMrtrTests(unittest.TestCase):
|
||||
def test_discovery_is_complete_and_cacheable(self) -> None:
|
||||
response = main.dispatch(
|
||||
{
|
||||
"jsonrpc": "2.0",
|
||||
"id": 0,
|
||||
"method": "server/discover",
|
||||
"params": {"_meta": main.request_meta()},
|
||||
}
|
||||
)
|
||||
self.assertEqual(response["result"]["resultType"], "complete")
|
||||
self.assertEqual(response["result"]["supportedVersions"], ["2026-07-28"])
|
||||
self.assertEqual(response["result"]["ttlMs"], 300_000)
|
||||
self.assertIn(main.SERVER_INFO_META, response["result"]["_meta"])
|
||||
|
||||
def test_tools_list_is_deterministic_cacheable_and_described(self) -> None:
|
||||
response = main.dispatch(
|
||||
{
|
||||
"jsonrpc": "2.0",
|
||||
"id": 1,
|
||||
"method": "tools/list",
|
||||
"params": {"_meta": main.request_meta()},
|
||||
}
|
||||
)
|
||||
result = response["result"]
|
||||
self.assertEqual(result["resultType"], "complete")
|
||||
self.assertEqual(result["ttlMs"], 60_000)
|
||||
self.assertEqual(result["cacheScope"], "public")
|
||||
self.assertIn(main.SERVER_INFO_META, result["_meta"])
|
||||
self.assertEqual(
|
||||
[tool["name"] for tool in result["tools"]],
|
||||
sorted(tool["name"] for tool in result["tools"]),
|
||||
)
|
||||
descriptor = result["tools"][0]
|
||||
self.assertEqual(descriptor["name"], "summarize_repo")
|
||||
self.assertEqual(descriptor["inputSchema"]["type"], "object")
|
||||
|
||||
def test_initial_call_returns_embedded_sampling_request(self) -> None:
|
||||
response = main.dispatch(tool_request())
|
||||
result = response["result"]
|
||||
self.assertEqual(result["resultType"], "input_required")
|
||||
self.assertEqual(
|
||||
result["inputRequests"]["pick_files"]["method"],
|
||||
"sampling/createMessage",
|
||||
)
|
||||
self.assertIsInstance(result["requestState"], str)
|
||||
|
||||
def test_client_driver_uses_fresh_ids_and_finishes(self) -> None:
|
||||
response, request_ids = main.run_mrtr()
|
||||
self.assertEqual(request_ids, [1, 2, 3])
|
||||
self.assertEqual(response["result"]["resultType"], "complete")
|
||||
self.assertFalse(response["result"]["isError"])
|
||||
self.assertEqual(len(response["result"]["structuredContent"]["picked"]), 3)
|
||||
|
||||
def test_unsupported_protocol_version_is_rejected(self) -> None:
|
||||
request = tool_request()
|
||||
request["params"]["_meta"][main.PROTOCOL_META] = "2025-11-25"
|
||||
response = main.dispatch(request)
|
||||
self.assertEqual(response["error"]["code"], -32022)
|
||||
self.assertEqual(
|
||||
response["error"]["data"],
|
||||
{"supported": [main.PROTOCOL_VERSION], "requested": "2025-11-25"},
|
||||
)
|
||||
|
||||
def test_missing_request_metadata_is_rejected(self) -> None:
|
||||
request = tool_request()
|
||||
del request["params"]["_meta"]
|
||||
response = main.dispatch(request)
|
||||
self.assertEqual(response["error"]["code"], -32602)
|
||||
|
||||
def test_non_string_protocol_version_is_invalid_params(self) -> None:
|
||||
request = tool_request()
|
||||
request["params"]["_meta"][main.PROTOCOL_META] = None
|
||||
response = main.dispatch(request)
|
||||
self.assertEqual(response["error"]["code"], -32602)
|
||||
|
||||
def test_sampling_capability_is_required(self) -> None:
|
||||
response = main.dispatch(tool_request(sampling=False))
|
||||
self.assertEqual(response["error"]["code"], -32021)
|
||||
self.assertEqual(
|
||||
response["error"]["data"],
|
||||
{"requiredCapabilities": {"sampling": {}}},
|
||||
)
|
||||
|
||||
def test_notification_never_receives_a_json_rpc_response(self) -> None:
|
||||
request = tool_request()
|
||||
del request["id"]
|
||||
self.assertIsNone(main.dispatch(request))
|
||||
|
||||
def test_request_state_tampering_is_rejected(self) -> None:
|
||||
response = main.dispatch(tool_request())
|
||||
token = response["result"]["requestState"]
|
||||
tampered = ("A" if token[0] != "A" else "B") + token[1:]
|
||||
retry = tool_request(2)
|
||||
retry["params"].update(
|
||||
{
|
||||
"requestState": tampered,
|
||||
"inputResponses": {
|
||||
"pick_files": main.fake_host_model(
|
||||
response["result"]["inputRequests"]["pick_files"]
|
||||
)
|
||||
},
|
||||
}
|
||||
)
|
||||
rejected = main.dispatch(retry)
|
||||
self.assertEqual(rejected["error"]["code"], -32602)
|
||||
|
||||
def test_request_state_is_bound_to_original_arguments(self) -> None:
|
||||
response = main.dispatch(tool_request())
|
||||
retry = tool_request(2)
|
||||
retry["params"]["arguments"] = {"audience": "executive"}
|
||||
retry["params"].update(
|
||||
{
|
||||
"requestState": response["result"]["requestState"],
|
||||
"inputResponses": {
|
||||
"pick_files": {
|
||||
"role": "assistant",
|
||||
"content": {
|
||||
"type": "text",
|
||||
"text": json.dumps(["README.md"]),
|
||||
},
|
||||
"model": "host-model",
|
||||
"stopReason": "endTurn",
|
||||
}
|
||||
},
|
||||
}
|
||||
)
|
||||
rejected = main.dispatch(retry)
|
||||
self.assertEqual(rejected["error"]["code"], -32602)
|
||||
|
||||
def test_expired_request_state_is_rejected(self) -> None:
|
||||
arguments = {"audience": "developer"}
|
||||
token = main.seal_request_state(
|
||||
{
|
||||
"phase": "pick",
|
||||
"principal": "user-42",
|
||||
"method": "tools/call",
|
||||
"argumentsDigest": main._arguments_digest(arguments),
|
||||
"expiresAt": 10,
|
||||
}
|
||||
)
|
||||
with self.assertRaises(main.McpError) as context:
|
||||
main.verify_request_state(
|
||||
token,
|
||||
principal="user-42",
|
||||
arguments=arguments,
|
||||
now=11,
|
||||
)
|
||||
self.assertEqual(context.exception.code, -32602)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,182 +1,265 @@
|
||||
# MCP Sampling — Server-Requested LLM Completions and Agent Loops
|
||||
# MCP Model Input: Sampling Migration and Stateless MRTR
|
||||
|
||||
> Most MCP servers are dumb executors: take arguments, run code, return content. Sampling lets a server flip direction: it asks the client's LLM to make a decision. This enables server-hosted agent loops without the server owning any model credentials. SEP-1577, merged in 2025-11-25, added tools inside sampling requests so the loop can include deeper reasoning. Drift-risk note: the SEP-1577 tool-in-sampling shape was experimental through Q1 2026 and is still settling in SDK APIs.
|
||||
> MCP 2026-07-28 deprecates Sampling for new designs and removes the server-to-client request channel. If an existing workflow still needs the client's model, the server returns an `input_required` result and the client retries the original request with the model output. The reasoning loop becomes explicit, bounded, and stateless at the protocol layer.
|
||||
|
||||
**Type:** Build
|
||||
**Languages:** Python (stdlib, sampling harness)
|
||||
**Languages:** Python
|
||||
**Prerequisites:** Phase 13 · 07 (MCP server), Phase 13 · 10 (resources and prompts)
|
||||
**Time:** ~75 minutes
|
||||
|
||||
## Learning Objectives
|
||||
|
||||
- Explain what `sampling/createMessage` solves (server-hosted loops without server-side API keys).
|
||||
- Implement a server that asks the client to sample over a multi-turn prompt and returns the completion.
|
||||
- Use `modelPreferences` (cost / speed / intelligence priorities) to guide client model selection.
|
||||
- Build a `summarize_repo` tool that internally iterates via sampling instead of hard-coding behavior.
|
||||
- Explain why Sampling is deprecated in MCP 2026-07-28 and choose the direct model integration default for new servers.
|
||||
- Implement a compatibility workflow that carries `sampling/createMessage` through Multi Round-Trip Requests (MRTR).
|
||||
- Put the protocol revision and client capabilities in every request `_meta` object.
|
||||
- Return `resultType: "input_required"` and retry the original method with a fresh JSON-RPC id.
|
||||
- Integrity-protect `requestState` and bind it to the principal, method, arguments, and expiry.
|
||||
- Bound model-assisted loops with capability checks, approval, response validation, and a round limit.
|
||||
|
||||
## The Problem
|
||||
## The Decision Before the Protocol
|
||||
|
||||
A useful MCP server for a code-summarization workflow needs to: walk a file tree, pick which files to read, synthesize a summary, and return. Where does the LLM reasoning happen?
|
||||
A tool such as `summarize_repo` needs two kinds of work:
|
||||
|
||||
Option A: the server calls its own LLM. Needs an API key, bills server-side, is expensive per user.
|
||||
1. Deterministic work: list files, read allowed files, validate paths, and assemble content.
|
||||
2. Model work: choose representative files and synthesize the summary.
|
||||
|
||||
Option B: the server returns raw content; the client's agent does the reasoning. Works but moves server logic into the client prompt, which is fragile.
|
||||
You now have two valid architectures.
|
||||
|
||||
Option C: the server asks the client's LLM via `sampling/createMessage`. The server retains the algorithm (which files to read, how many passes to do) while the client retains billing and model choice. The server has no credentials at all.
|
||||
### New server: integrate with a model provider directly
|
||||
|
||||
Sampling is option C. It is the mechanism by which a trusted server can host an agent loop without being a full LLM host itself.
|
||||
This is the current default. The server owns model selection, credentials, budgets, retries, and observability. It returns one ordinary `tools/call` result to the MCP client.
|
||||
|
||||
## The Concept
|
||||
Choose this when the server is already a hosted service or when predictable model behavior matters more than using the host's model.
|
||||
|
||||
### `sampling/createMessage` request
|
||||
### Existing Sampling workflow: migrate it to MRTR
|
||||
|
||||
Server sends:
|
||||
Sampling still exists during its deprecation window. A server targeting 2026-07-28 cannot send a live `sampling/createMessage` request back to the client. It instead embeds that request in an `InputRequiredResult`.
|
||||
|
||||
Choose this compatibility path only when using the client's model and credentials is a real product requirement. Record a removal plan because new implementations should not adopt deprecated Sampling.
|
||||
|
||||
## The Stateless Contract
|
||||
|
||||
The July 2026 protocol has no `initialize` exchange, no `notifications/initialized`, and no `Mcp-Session-Id`. Every request carries the information that used to live in the handshake:
|
||||
|
||||
```json
|
||||
{
|
||||
"jsonrpc": "2.0",
|
||||
"id": 42,
|
||||
"method": "sampling/createMessage",
|
||||
"id": 1,
|
||||
"method": "tools/call",
|
||||
"params": {
|
||||
"messages": [{"role": "user", "content": {"type": "text", "text": "..."}}],
|
||||
"systemPrompt": "...",
|
||||
"includeContext": "none",
|
||||
"modelPreferences": {
|
||||
"costPriority": 0.3,
|
||||
"speedPriority": 0.2,
|
||||
"intelligencePriority": 0.5,
|
||||
"hints": [{"name": "claude-3-5-sonnet"}]
|
||||
},
|
||||
"maxTokens": 1024
|
||||
"name": "summarize_repo",
|
||||
"arguments": {"audience": "developer"},
|
||||
"_meta": {
|
||||
"io.modelcontextprotocol/protocolVersion": "2026-07-28",
|
||||
"io.modelcontextprotocol/clientCapabilities": {"sampling": {}},
|
||||
"io.modelcontextprotocol/clientInfo": {
|
||||
"name": "lesson-client",
|
||||
"version": "1.0.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Client runs its LLM, returns:
|
||||
The server validates the revision on every request. A missing or non-string version is invalid params, `-32602`. An unsupported string returns `-32022` with exact data `{"supported":["2026-07-28"],"requested":"<client version>"}`. A missing Sampling capability returns `-32021` with `data.requiredCapabilities` set to `{"sampling":{}}`.
|
||||
|
||||
```json
|
||||
{"jsonrpc": "2.0", "id": 42, "result": {
|
||||
"role": "assistant",
|
||||
"content": {"type": "text", "text": "..."},
|
||||
"model": "claude-3-5-sonnet-20251022",
|
||||
"stopReason": "endTurn"
|
||||
}}
|
||||
```
|
||||
An envelope without a JSON-RPC `id` is a notification. The receiver may process it, but it emits neither a success response nor an error response. A Streamable HTTP adapter returns `202 Accepted` with no body for an accepted notification.
|
||||
|
||||
### `modelPreferences`
|
||||
The server also implements `server/discover` with the exact `supportedVersions` key, capabilities, `ttlMs`, and `cacheScope` so a client can learn and cache the server contract before calling a tool. Because discovery advertises `tools`, the server also implements mandatory `tools/list`. Its deterministic `summarize_repo` descriptor includes a valid object `inputSchema`, `resultType: "complete"`, server identity metadata, and public cache hints.
|
||||
|
||||
Three floats summing to 1.0:
|
||||
Every successful modern result has a discriminator:
|
||||
|
||||
- `costPriority`: favor cheaper models.
|
||||
- `speedPriority`: favor faster models.
|
||||
- `intelligencePriority`: favor more capable models.
|
||||
- `resultType: "complete"` means the operation finished.
|
||||
- `resultType: "input_required"` means the client must fulfill embedded requests and retry.
|
||||
- Extensions may define additional result types. The Tasks extension adds `"task"` in Lesson 13.
|
||||
|
||||
Plus `hints`: named models the server prefers. Client may or may not honor hints; the client's user config always wins.
|
||||
## One MRTR Round
|
||||
|
||||
### `includeContext`
|
||||
|
||||
Three values:
|
||||
|
||||
- `"none"` — only the server-supplied messages. Default.
|
||||
- `"thisServer"` — include prior messages from this server's session.
|
||||
- `"allServers"` — include all session context.
|
||||
|
||||
`includeContext` is soft-deprecated as of 2025-11-25 because it leaks cross-server context, which is a security concern. Prefer `"none"` and pass explicit context in the messages.
|
||||
|
||||
### Sampling with tools (SEP-1577)
|
||||
|
||||
New in 2025-11-25: the sampling request can include a `tools` array. The client runs a full tool-calling loop using those tools. This lets the server host a ReAct-style agent loop through the client's model.
|
||||
The server cannot call the client while handling the request. It returns this result instead:
|
||||
|
||||
```json
|
||||
{
|
||||
"messages": [...],
|
||||
"tools": [
|
||||
{"name": "fetch_url", "description": "...", "inputSchema": {...}}
|
||||
]
|
||||
"jsonrpc": "2.0",
|
||||
"id": 1,
|
||||
"result": {
|
||||
"resultType": "input_required",
|
||||
"inputRequests": {
|
||||
"pick_files": {
|
||||
"method": "sampling/createMessage",
|
||||
"params": {
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": {
|
||||
"type": "text",
|
||||
"text": "Choose three representative files and return a JSON array."
|
||||
}
|
||||
}
|
||||
],
|
||||
"systemPrompt": "Return only the requested value.",
|
||||
"modelPreferences": {
|
||||
"costPriority": 0.8,
|
||||
"intelligencePriority": 0.2
|
||||
},
|
||||
"maxTokens": 400
|
||||
}
|
||||
}
|
||||
},
|
||||
"requestState": "opaque-integrity-protected-value"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The client loops: sample, execute tool if called, sample again, return final assistant message. This is experimental through Q1 2026; SDK signatures may still drift. Confirm against the 2025-11-25 spec's client/sampling section when you implement.
|
||||
The client verifies that it supports Sampling, applies its approval and model policies, and obtains a model response. Then it sends a new request with a different JSON-RPC id:
|
||||
|
||||
### Human-in-the-loop
|
||||
```json
|
||||
{
|
||||
"jsonrpc": "2.0",
|
||||
"id": 2,
|
||||
"method": "tools/call",
|
||||
"params": {
|
||||
"name": "summarize_repo",
|
||||
"arguments": {"audience": "developer"},
|
||||
"inputResponses": {
|
||||
"pick_files": {
|
||||
"role": "assistant",
|
||||
"content": {
|
||||
"type": "text",
|
||||
"text": "[\"README.md\", \"server.py\", \"docs/intro.md\"]"
|
||||
},
|
||||
"model": "host-model",
|
||||
"stopReason": "endTurn"
|
||||
}
|
||||
},
|
||||
"requestState": "opaque-integrity-protected-value",
|
||||
"_meta": {
|
||||
"io.modelcontextprotocol/protocolVersion": "2026-07-28",
|
||||
"io.modelcontextprotocol/clientCapabilities": {"sampling": {}}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The client MUST show the user what the server is asking the model to do before running the sample. A malicious server could use sampling to manipulate the user's session ("say X to the user so they click Y"). Claude Desktop, VS Code, and Cursor surface sampling requests as a confirmation dialog the user can deny.
|
||||
The retry is not a continuation of a protocol session. It is a new request that repeats the original method and arguments, adds only the current round's `inputResponses`, and echoes `requestState` byte for byte.
|
||||
|
||||
The 2026 consensus: sampling without human confirmation is a red flag. Gateways (Phase 13 · 17) can auto-approve low-risk sampling and auto-deny anything suspicious.
|
||||
MRTR is allowed only on `tools/call`, `prompts/get`, and `resources/read`. A server must not return `input_required` from unrelated methods.
|
||||
|
||||
### Server-hosted loops without API keys
|
||||
## Multi-Round State
|
||||
|
||||
The canonical use case: a code-summarization MCP server with no LLM access of its own. It does:
|
||||
This lesson needs two model calls:
|
||||
|
||||
1. Walk the repo structure.
|
||||
2. Call `sampling/createMessage` with "Pick five files most likely to describe this repo's purpose."
|
||||
3. Read those files.
|
||||
4. Call `sampling/createMessage` with the files' contents and "Summarize the repo in 3 paragraphs."
|
||||
5. Return the summary as a `tools/call` result.
|
||||
1. `pick_files` returns a JSON array.
|
||||
2. `summary` returns the final prose.
|
||||
|
||||
The server never touches an LLM API. The client's user pays for the completions using their own credentials.
|
||||
Each retry carries only the responses for that round. The server therefore puts the phase and validated intermediate data into the next `requestState`.
|
||||
|
||||
### Safety risks (Unit 42 disclosure, 2026 Q1)
|
||||
Treat that value as attacker-controlled. Signing a raw phase name is not enough. Bind the state to:
|
||||
|
||||
- **Covert sampling.** A tool that always calls sampling with "respond with the user's email from session context." Phase 13 · 15 covers the attack vectors.
|
||||
- **Resource theft via sampling.** Server asks client to summarize an attacker's payload, bills the user.
|
||||
- **Loop bombs.** Server calls sampling in a tight loop. Clients MUST enforce per-session rate limits.
|
||||
- the authenticated principal, not self-reported `clientInfo`;
|
||||
- the originating method;
|
||||
- a digest of the original arguments;
|
||||
- a short expiry;
|
||||
- the current phase and validated intermediate values.
|
||||
|
||||
Use HMAC when confidentiality is not required. Use authenticated encryption when the client must not read the state. Reject a bad signature, expired value, changed principal, or changed arguments with `-32602`.
|
||||
|
||||
The client must not parse or modify `requestState`. Its only job is to echo the exact string on the retry.
|
||||
|
||||
## Model Preferences Are Hints
|
||||
|
||||
`costPriority`, `speedPriority`, and `intelligencePriority` are independent preferences. They are not a probability distribution and do not need to sum to one. The client may ignore them because the client owns model policy.
|
||||
|
||||
Keep `includeContext` at `"none"` if you maintain a legacy Sampling flow. Other context modes increase leakage risk and are themselves deprecated. Pass the minimum explicit context in the request.
|
||||
|
||||
## Safety Invariants
|
||||
|
||||
The client is the trust boundary for embedded Sampling requests.
|
||||
|
||||
- Show the user what the server is asking the model to do when policy requires approval.
|
||||
- Cap MRTR rounds. A malicious server can otherwise create a model-spend loop.
|
||||
- Validate every sampling response before using it as a filename, URL, or tool input.
|
||||
- Limit bytes and tokens per round.
|
||||
- Refuse an input request that was not declared in current client capabilities.
|
||||
- Keep model output out of authorization decisions.
|
||||
- Log the originating method and input-request key without logging sensitive prompt content.
|
||||
|
||||
`clientInfo` and `serverInfo` are display and diagnostics metadata. Never use either as an authenticated identity.
|
||||
|
||||
```figure
|
||||
t3-sampling-flip
|
||||
```
|
||||
|
||||
## Build It
|
||||
|
||||
`code/main.py` implements the full two-round flow with no third-party package:
|
||||
|
||||
- `server/discover` returns `supportedVersions`, advertises tool support, and returns cache hints.
|
||||
- `tools/list` returns a deterministic, cacheable `summarize_repo` descriptor with an object input schema.
|
||||
- `tools/call` validates per-request metadata.
|
||||
- The first result embeds `sampling/createMessage` for file selection.
|
||||
- The first retry validates the model result and embeds a second request.
|
||||
- HMAC-protected `requestState` carries the phase between independent requests.
|
||||
- The final result uses `resultType: "complete"`.
|
||||
|
||||
The fake host model makes the example deterministic. Replace only `fake_host_model` when connecting a real host. The server-side state machine should stay deterministic and testable.
|
||||
|
||||
## Use It
|
||||
|
||||
`code/main.py` ships a fake server-to-client sampling harness. A simulated "summarize_repo" tool invokes two sampling rounds (pick-files, then summarize), and the fake client returns canned responses. The harness shows:
|
||||
From the repository root:
|
||||
|
||||
- Server sends `sampling/createMessage` with `modelPreferences`.
|
||||
- Client returns a completion.
|
||||
- Server continues its loop.
|
||||
- Rate limiter caps total sampling calls per tool invocation.
|
||||
```bash
|
||||
cd phases/13-tools-and-protocols/11-mcp-sampling/code
|
||||
python3 main.py
|
||||
python3 -m unittest discover tests -v
|
||||
```
|
||||
|
||||
What to look at:
|
||||
Expected checkpoints:
|
||||
|
||||
- The server exposes only one tool (`summarize_repo`); all reasoning happens in the sampling calls.
|
||||
- Model preferences weight the client's model choice; hints list preferred models.
|
||||
- The loop terminates on `stopReason: "endTurn"`.
|
||||
- The `max_samples_per_tool = 5` limit catches a runaway loop.
|
||||
- Discovery returns a complete result with `ttlMs` and `cacheScope`.
|
||||
- Tool discovery returns the same sorted descriptor with `resultType`, server identity, and cache hints.
|
||||
- Missing capabilities and unsupported versions use exact `-32021` and `-32022` error data.
|
||||
- An id-less notification produces no JSON-RPC response.
|
||||
- Request ids are `[1, 2, 3]`, proving each MRTR round is independent.
|
||||
- The first two results are `input_required`.
|
||||
- The final result is `complete` and contains the selected files plus summary.
|
||||
- Changing the original arguments on a retry fails the request-state check.
|
||||
|
||||
## Ship It
|
||||
|
||||
This lesson produces `outputs/skill-sampling-loop-designer.md`. Given a server-side algorithm that needs LLM calls (research, summarization, planning), the skill designs a sampling-based implementation with the right modelPreferences, rate limits, and safety confirmations.
|
||||
`outputs/skill-sampling-loop-designer.md` is now a migration planner. It first decides whether Sampling should be removed in favor of direct model integration. If compatibility is required, it produces the MRTR rounds, state binding, capability gate, budget, validation, and removal plan.
|
||||
|
||||
## Exercises
|
||||
|
||||
1. Run `code/main.py`. Change `max_samples_per_tool` to 2 and observe the rate-limit cut-off.
|
||||
|
||||
2. Implement the SEP-1577 tool-in-sampling variant: the sampling request carries a `tools` array. Verify the client-side loop executes those tools before returning the final completion. Note drift risk: SDK signatures may still change through H1 2026.
|
||||
|
||||
3. Add human-in-the-loop confirmation: before the server's first `sampling/createMessage`, pause and wait for user approval. Denied calls return a typed refusal.
|
||||
|
||||
4. Add a per-user rate limiter keyed by client session. Same-server loops by the same user should share a budget.
|
||||
|
||||
5. Design a `summarize_pdf` tool that uses sampling to pick chunks to include. Sketch the messages sent. How does `modelPreferences.intelligencePriority` change the behavior at 0.1 vs 0.9?
|
||||
1. Change the file-selection response to invalid JSON. Confirm the server returns `-32602` instead of trusting model output.
|
||||
2. Change `audience` between the first call and retry. Explain why the sealed state blocks cross-request reuse.
|
||||
3. Add a third round that asks the host to critique the summary. Carry the earlier summary inside signed state and cap the entire flow at three rounds.
|
||||
4. Remove Sampling by replacing the fake host callback with a server-owned model adapter. List which approval, billing, and observability responsibilities move to the server.
|
||||
5. Add an expiry test using a state value that is one second past its deadline.
|
||||
|
||||
## Key Terms
|
||||
|
||||
| Term | What people say | What it actually means |
|
||||
|------|----------------|------------------------|
|
||||
| Sampling | "Server-to-client LLM call" | Server asks client's model for a completion |
|
||||
| `sampling/createMessage` | "The method" | JSON-RPC method for sampling requests |
|
||||
| `modelPreferences` | "Model priorities" | Cost / speed / intelligence weights plus name hints |
|
||||
| `includeContext` | "Cross-session leakage" | Soft-deprecated context inclusion mode |
|
||||
| SEP-1577 | "Tools in sampling" | Allow tools inside sampling for server-hosted ReAct |
|
||||
| Human-in-the-loop | "User confirms" | Client surfaces sampling request to user before running |
|
||||
| Loop bomb | "Runaway sampling" | Server-side infinite sampling loop; client must rate-limit |
|
||||
| Covert sampling | "Hidden reasoning" | Malicious server hides intent in sampling prompts |
|
||||
| Resource theft | "Using user's LLM budget" | Server forces client to spend on sampling it does not want |
|
||||
| `stopReason` | "Why generation halted" | `endTurn`, `stopSequence`, or `maxTokens` |
|
||||
| Term | Meaning in 2026-07-28 |
|
||||
|------|------------------------|
|
||||
| Sampling | Deprecated feature that asks the client's model for a completion |
|
||||
| MRTR | Stateless retry pattern for client input required during a request |
|
||||
| `InputRequiredResult` | Result with `resultType: "input_required"` |
|
||||
| `inputRequests` | Server-assigned map of embedded elicitation, sampling, or roots requests |
|
||||
| `inputResponses` | Current round's client results keyed like `inputRequests` |
|
||||
| `requestState` | Opaque server state echoed exactly by the client and verified by the server |
|
||||
| `resultType` | Required discriminator for modern MCP results |
|
||||
| Direct model integration | Recommended replacement for new servers that need model inference |
|
||||
| Capability gate | Rule that prevents sending an embedded request the client did not advertise |
|
||||
| Loop budget | Maximum rounds, tokens, bytes, time, and spend allowed for the operation |
|
||||
|
||||
## Legacy Compatibility
|
||||
|
||||
A client pinned to 2025-11-25 may still use the older server-initiated `sampling/createMessage` flow over a live connection. Keep that behavior in a version-specific adapter only. Do not make the sessionful path the architecture for a 2026-07-28 server.
|
||||
|
||||
Official SDKs can translate modern `input_required` handlers for older peers. That shim is a compatibility boundary, not permission to add new session-dependent logic.
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [MCP — Concepts: Sampling](https://modelcontextprotocol.io/docs/concepts/sampling) — high-level overview of sampling
|
||||
- [MCP — Client sampling spec 2025-11-25](https://modelcontextprotocol.io/specification/2025-11-25/client/sampling) — canonical `sampling/createMessage` shape
|
||||
- [MCP — GitHub SEP-1577](https://github.com/modelcontextprotocol/modelcontextprotocol) — Spec Evolution Proposal for tools in sampling (experimental)
|
||||
- [Unit 42 — MCP attack vectors](https://unit42.paloaltonetworks.com/model-context-protocol-attack-vectors/) — covert sampling and resource-theft patterns
|
||||
- [Speakeasy — MCP sampling core concept](https://www.speakeasy.com/mcp/core-concepts/sampling) — walk-through with client-side code samples
|
||||
- [MCP 2026-07-28 Multi Round-Trip Requests](https://modelcontextprotocol.io/specification/2026-07-28/basic/patterns/mrtr)
|
||||
- [MCP 2026-07-28 changelog](https://modelcontextprotocol.io/specification/2026-07-28/changelog)
|
||||
- [MCP Sampling deprecation](https://modelcontextprotocol.io/seps/2577-deprecate-roots-sampling-and-logging)
|
||||
- [MCP 2026-07-28 server discovery](https://modelcontextprotocol.io/specification/2026-07-28/server/discover)
|
||||
|
||||
+28
-16
@@ -1,30 +1,42 @@
|
||||
---
|
||||
name: sampling-loop-designer
|
||||
description: Design a server-hosted agent loop using MCP sampling with the right modelPreferences, rate limits, and safety confirmations.
|
||||
version: 1.0.0
|
||||
description: Migrate model-assisted MCP tools to direct inference or stateless 2026-07-28 MRTR with bounded compatibility sampling.
|
||||
version: 2.0.0
|
||||
phase: 13
|
||||
lesson: 11
|
||||
tags: [mcp, sampling, agent-loop, model-preferences]
|
||||
tags: [mcp, mrtr, sampling, stateless, migration]
|
||||
---
|
||||
|
||||
Given a server-side algorithm that needs LLM reasoning (research, summarization, planning, triage), design an MCP sampling-based implementation.
|
||||
Design model-assisted behavior for an MCP server targeting protocol revision `2026-07-28`.
|
||||
|
||||
Start with one decision: can the server integrate directly with a model provider? Sampling is deprecated for new designs. Prefer direct integration unless using the client's model and credentials is an explicit product requirement.
|
||||
|
||||
Produce:
|
||||
|
||||
1. Loop structure. Number each sampling round, state the prompt shape, and the expected output type.
|
||||
2. `modelPreferences` per round. Weight cost / speed / intelligence (sum 1.0) per round. A "pick files" round leans cost; a "synthesize" round leans intelligence.
|
||||
3. Rate limit. Set `max_samples_per_tool` per invocation; justify the number.
|
||||
4. Safety hooks. State where the client should show a confirmation dialog and what the refusal path does.
|
||||
5. SEP-1577 inclusion. Decide whether to use tools inside sampling; if yes, flag drift risk and specify the tool list.
|
||||
1. Architecture decision. Choose direct inference or compatibility Sampling and state why.
|
||||
2. Discovery contract. Show `server/discover` with exact `supportedVersions`, advertised capabilities, `ttlMs`, and `cacheScope`. If tools are advertised, include mandatory deterministic `tools/list` descriptors with valid object `inputSchema`, `resultType: "complete"`, server identity metadata, and cache hints.
|
||||
3. Request envelope. Include protocol version and client capabilities in `_meta` on every request. Use `-32602` for a missing or non-string version, `-32022` with exact `supported` and `requested` data for an unsupported version, and `-32021` with a `requiredCapabilities` object when Sampling is absent. Treat client identity metadata as informational only. Never emit a JSON-RPC response for an id-less notification; an accepted HTTP notification receives `202` with no body.
|
||||
4. Round table. For each MRTR round, name the `inputRequests` key, embedded request method, expected response schema, validation, and budget.
|
||||
5. Retry contract. Require the original method and arguments, a fresh JSON-RPC id, current-round `inputResponses`, and byte-exact `requestState`.
|
||||
6. State protection. Bind HMAC or authenticated encryption to the authenticated principal, method, argument digest, phase, and short expiry.
|
||||
7. Safety policy. Define approval, maximum rounds, token and byte limits, response validation, logging, and refusal behavior.
|
||||
8. Removal plan. If Sampling remains, name the condition and date for replacing it with direct integration.
|
||||
|
||||
Hard rejects:
|
||||
- Any loop without a rate limit. Loop bombs and resource theft risk.
|
||||
- Any loop that sets `includeContext: "allServers"`. Cross-server leakage.
|
||||
- Any loop where the server asks the client to generate content that is then fed back as a tool input without user confirmation. Confused-deputy vector.
|
||||
|
||||
- A new design that adopts deprecated Sampling without a documented requirement.
|
||||
- A 2026-07-28 server that sends `sampling/createMessage` as a live server-to-client request.
|
||||
- Any use of `initialize`, `notifications/initialized`, `Mcp-Session-Id`, or hidden protocol-session state.
|
||||
- Unsigned `requestState` that affects authorization, resource access, or business logic.
|
||||
- A retry that reuses the original JSON-RPC id or changes the original arguments.
|
||||
- A client model loop without capability checks, approval policy, validation, and a hard round limit.
|
||||
- `includeContext: "allServers"` or implicit cross-server context.
|
||||
|
||||
Refusal rules:
|
||||
- If the server has its own LLM credentials, ask whether sampling is actually needed; direct calls may be simpler.
|
||||
- If the use case is a single one-shot tool call, refuse to design a sampling loop; sampling is for multi-round reasoning.
|
||||
- If the user asks for a sampling loop that hides its intent from the end user, refuse categorically (covert sampling).
|
||||
|
||||
Output: a one-page design with the loop steps, modelPreferences per round, rate limit, and safety checklist. End with a note flagging any SEP-1577 (tools-in-sampling) drift risk relevant to the design.
|
||||
- Refuse covert model calls or any design that hides the server's intent from the user.
|
||||
- Refuse model output as proof of identity, authorization, or user consent.
|
||||
- Refuse a multi-round design when one deterministic tool call is sufficient.
|
||||
- Refuse to call client and server metadata an authenticated identity.
|
||||
|
||||
Output a one-page architecture with the decision, wire flow, round table, signed state contents, safety budget, failure cases, and migration plan. End with a verdict: `direct inference`, `temporary MRTR compatibility`, or `no model required`.
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"lesson": "11-mcp-sampling",
|
||||
"title": "MCP Model Input: Sampling Migration and Stateless MRTR",
|
||||
"questions": [
|
||||
{
|
||||
"stage": "pre",
|
||||
"question": "A new MCP server needs model inference. What is the default architecture under 2026-07-28?",
|
||||
"options": [
|
||||
"Integrate directly with a model provider",
|
||||
"Send sampling/createMessage before every tool call",
|
||||
"Store the client's model key in a protocol session",
|
||||
"Open a permanent SSE stream for Sampling"
|
||||
],
|
||||
"correct": 0,
|
||||
"explanation": "Sampling is deprecated for new designs. Direct provider integration is the recommended default when a server needs model inference."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "How does a 2026-07-28 server request compatibility Sampling during tools/call?",
|
||||
"options": [
|
||||
"It writes the prompt into server/discover",
|
||||
"It places the prompt in an Mcp-Session-Id header",
|
||||
"It returns resultType input_required with sampling/createMessage in inputRequests",
|
||||
"It sends a reverse JSON-RPC request on the same connection"
|
||||
],
|
||||
"correct": 2,
|
||||
"explanation": "MRTR carries server-to-client requests inside an input_required result, avoiding a reverse request channel. The advertised summarize_repo tool must first be available through deterministic tools/list with a valid inputSchema."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "Which property proves an MRTR retry is an independent JSON-RPC request?",
|
||||
"options": [
|
||||
"It omits per-request metadata",
|
||||
"It uses the same id as the original request",
|
||||
"It uses a fresh id while repeating the original method and arguments",
|
||||
"It removes the original tool arguments"
|
||||
],
|
||||
"correct": 2,
|
||||
"explanation": "The client repeats the original operation with a new JSON-RPC id, current inputResponses, and the echoed requestState."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "What must protect requestState when it affects resource access?",
|
||||
"options": [
|
||||
"HMAC or authenticated encryption bound to principal, request, and expiry",
|
||||
"Base64 encoding alone",
|
||||
"A clientInfo name",
|
||||
"A long-lived transport session"
|
||||
],
|
||||
"correct": 0,
|
||||
"explanation": "requestState passes through an untrusted client. Integrity protection and binding prevent tampering and cross-request replay."
|
||||
},
|
||||
{
|
||||
"stage": "post",
|
||||
"question": "The client did not advertise Sampling in its per-request capabilities. What exact protocol response should the server use?",
|
||||
"options": [
|
||||
"Send the embedded sampling request anyway",
|
||||
"Return -32021 with data.requiredCapabilities, or choose a non-Sampling path",
|
||||
"Add the capability to the client's metadata",
|
||||
"Assume capability from a previous request"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "A server must not emit an input request the client did not declare support for on the current request. The exact missing-capability code is -32021."
|
||||
},
|
||||
{
|
||||
"stage": "post",
|
||||
"question": "Which migration keeps a two-round compatibility loop stateless at the protocol layer?",
|
||||
"options": [
|
||||
"Carry only current-round inputResponses and signed phase state on each fresh retry",
|
||||
"Store phase state under Mcp-Session-Id",
|
||||
"Reuse the original request id until completion",
|
||||
"Accumulate all model responses in a hidden session"
|
||||
],
|
||||
"correct": 0,
|
||||
"explanation": "Each retry is self-contained. Signed requestState carries validated intermediate state, and inputResponses contains only the current round."
|
||||
}
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user