mirror of
https://github.com/agent-substrate/substrate.git
synced 2026-10-02 03:24:42 +08:00
## What this is We keep saying Substrate's sweet spot is "agents that are idle most of the time" — but none of our benchmarks actually behave like one. Glutton hammers one resource at a time, and sweperf needs an external image with a replay trace. This adds a workload that acts like the thing we're building for: a coding agent working through a task. Each locust user is one session. The actor gets told to do the kind of things a coding agent does — clone a repo, install deps, build, hit a failing test, fix it, write tests, refactor, package — twenty steps, each one costing the sandbox the CPU, memory, disk, and network the real action would. Between steps the "LLM is thinking," so the driver suspends the actor, and the next step's first request wakes it back up through the router. That parked wake (`WakeFirstTouch` in the stats) is the number this whole benchmark exists to measure. The part I care most about: the entire workload is one table in `internal/benchmarking/boomer/agentsession/script.go`. Every step says in plain English what the agent is doing and what it costs. If you want to know what step 6 does to the sandbox, you read step 6. If you want a different workload, you edit the table — tests will catch you if you write a step that reads a file nothing wrote, or blow the actor's memory budget. To act the steps out, glutton grew two RPCs: `BurnCPU` (compute-bound work) and `Ingest` (bytes that actually cross the network before hitting disk, so a "git clone" is a real download, not a local write). ## How it went when we ran it Validated on a fresh 2-node GKE cluster, micro-VM first, then gVisor on the same hardware. Smoke runs were clean on both classes (gVisor: 1340 requests over 7 full laps, zero failures, wake p50 1.4s; micro-VM: wake p50 2.1s). At 20 concurrent sessions with realistic think times, gVisor held 2.6% failures with wake p50 1.3s. Pushing past the knee (~7 concurrently-active sessions on 8 vCPUs) was also useful: it reproduced the ateom-socket-vanishing failure from #1133 and left six actors permanently wedged in DELETING — a live repro of #1665. Two things the first live run taught us are already folded in: RAM refills are in-place so repeat laps don't transiently double the guest heap (512Mi micro-VM actors OOM'd without this — use 1Gi), and the driver replaces an actor after three failed steps in a row, because a CRASHED actor never comes back on its own. ## Future changes Right now the script is compiled in — changing what steps do means editing the table and rebuilding the image. That's deliberate for this PR (one reviewable, test-guarded source of truth), but the follow-up we've agreed on is to make the script runtime-configurable: named script variants selectable per run first, then accepting a full script as a file so operators can define workloads without touching Go. That lands as its own PR once this one is in. --------- Co-authored-by: Aditya Shantanu <aditya-shantanu@users.noreply.github.com>
165 lines
5.7 KiB
Go
165 lines
5.7 KiB
Go
// Copyright 2026 Google LLC
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
package glutton
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc"
|
|
"go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp"
|
|
"go.opentelemetry.io/otel"
|
|
"google.golang.org/grpc"
|
|
"google.golang.org/grpc/codes"
|
|
"google.golang.org/grpc/reflection"
|
|
"google.golang.org/grpc/status"
|
|
"google.golang.org/protobuf/proto"
|
|
|
|
"github.com/agent-substrate/substrate/internal/ateinterceptors"
|
|
gluttonpb "github.com/agent-substrate/substrate/internal/proto/glutton"
|
|
)
|
|
|
|
// Handler builds the request handler for the given wire mode. ModeGRPC serves
|
|
// gRPC alongside the wakeup probe on a single listener; ModeHTTP serves the
|
|
// protobuf-over-HTTP route table. An unknown mode comes back as an error so
|
|
// the caller decides how to fail.
|
|
func Handler(mode string, svc *Service) (http.Handler, error) {
|
|
var handler http.Handler
|
|
switch mode {
|
|
case ModeGRPC:
|
|
srv := grpc.NewServer(
|
|
grpc.StatsHandler(otelgrpc.NewServerHandler()),
|
|
)
|
|
gluttonpb.RegisterGluttonServer(srv, svc)
|
|
reflection.Register(srv)
|
|
// The wakeup probe is an HTTP GET, so gRPC mode serves it next to
|
|
// the gRPC handler on the same listener.
|
|
handler = splitGRPC(srv, readyzMux())
|
|
case ModeHTTP:
|
|
// otelhttp at the mux level + per-handler span follows
|
|
// docs/dev/best-practices/tracing.md: extract incoming context,
|
|
// then name the span after the operation in each handler.
|
|
handler = otelhttp.NewHandler(newMux(svc), "/")
|
|
default:
|
|
return nil, fmt.Errorf("must be %s or %s: %q", ModeGRPC, ModeHTTP, mode)
|
|
}
|
|
return handler, nil
|
|
}
|
|
|
|
// NewServer enables unencrypted HTTP/2 so gRPC works on the plaintext
|
|
// listener, alongside HTTP/1.1 for the wakeup probe.
|
|
func NewServer(handler http.Handler) *http.Server {
|
|
protocols := new(http.Protocols)
|
|
protocols.SetHTTP1(true)
|
|
protocols.SetUnencryptedHTTP2(true)
|
|
return &http.Server{Handler: handler, Protocols: protocols}
|
|
}
|
|
|
|
// splitGRPC serves gRPC and plain HTTP on one listener: requests with a
|
|
// gRPC content-type go to grpcSrv, everything else to rest. All glutton
|
|
// RPCs are unary, which is what grpc.Server.ServeHTTP supports.
|
|
func splitGRPC(grpcSrv, rest http.Handler) http.Handler {
|
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
if r.ProtoMajor == 2 && strings.HasPrefix(r.Header.Get("Content-Type"), "application/grpc") {
|
|
grpcSrv.ServeHTTP(w, r)
|
|
return
|
|
}
|
|
rest.ServeHTTP(w, r)
|
|
})
|
|
}
|
|
|
|
// readyzMux serves the wakeup probe both modes need.
|
|
func readyzMux() *http.ServeMux {
|
|
mux := http.NewServeMux()
|
|
mux.HandleFunc(ReadyzRoute, func(w http.ResponseWriter, r *http.Request) {
|
|
w.WriteHeader(http.StatusOK)
|
|
})
|
|
return mux
|
|
}
|
|
|
|
// newMux builds the HTTP-mode route table on top of the wakeup probe.
|
|
func newMux(svc *Service) *http.ServeMux {
|
|
mux := readyzMux()
|
|
mux.HandleFunc(PingRoute, protoRoute("Ping", svc.Ping))
|
|
mux.HandleFunc(WriteDiskRoute, protoRoute("WriteDisk", svc.WriteDisk))
|
|
mux.HandleFunc(ReadDiskRoute, protoRoute("ReadDisk", svc.ReadDisk))
|
|
mux.HandleFunc(WriteRAMRoute, protoRoute("WriteRAM", svc.WriteRAM))
|
|
mux.HandleFunc(ReadRAMRoute, protoRoute("ReadRAM", svc.ReadRAM))
|
|
mux.HandleFunc(BurnCPURoute, protoRoute("BurnCPU", svc.BurnCPU))
|
|
mux.HandleFunc(IngestRoute, protoRoute("Ingest", svc.Ingest))
|
|
return mux
|
|
}
|
|
|
|
// protoRoute wraps a protobuf handler with POST-only routing, protobuf
|
|
// unmarshaling, status code mapping, and server-timing headers.
|
|
func protoRoute[Req any, Resp proto.Message, PtrReq interface {
|
|
*Req
|
|
proto.Message
|
|
}](spanName string, handler func(context.Context, PtrReq) (Resp, error)) http.HandlerFunc {
|
|
return func(w http.ResponseWriter, r *http.Request) {
|
|
start := time.Now()
|
|
if r.Method != http.MethodPost {
|
|
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
|
|
return
|
|
}
|
|
body, err := io.ReadAll(r.Body)
|
|
if err != nil {
|
|
http.Error(w, err.Error(), http.StatusBadRequest)
|
|
return
|
|
}
|
|
var req Req
|
|
ptrReq := PtrReq(&req)
|
|
if err := proto.Unmarshal(body, ptrReq); err != nil {
|
|
http.Error(w, "unmarshal: "+err.Error(), http.StatusBadRequest)
|
|
return
|
|
}
|
|
ctx, span := otel.Tracer(Name).Start(r.Context(), spanName)
|
|
defer span.End()
|
|
resp, err := handler(ctx, ptrReq)
|
|
if err != nil {
|
|
if st, ok := status.FromError(err); ok {
|
|
switch st.Code() {
|
|
case codes.InvalidArgument:
|
|
http.Error(w, st.Message(), http.StatusBadRequest)
|
|
case codes.NotFound:
|
|
http.Error(w, st.Message(), http.StatusNotFound)
|
|
default:
|
|
http.Error(w, st.Message(), http.StatusInternalServerError)
|
|
}
|
|
} else {
|
|
http.Error(w, err.Error(), http.StatusInternalServerError)
|
|
}
|
|
return
|
|
}
|
|
out, err := proto.Marshal(resp)
|
|
if err != nil {
|
|
http.Error(w, err.Error(), http.StatusInternalServerError)
|
|
return
|
|
}
|
|
// Glutton does not run ateinterceptors, so without this the serve path has no
|
|
// server-side timing at all. Mirrors the control-plane gRPC trailer so boomer's
|
|
// elapsedFromMD logic (source=server) works identically over HTTP.
|
|
w.Header().Set(ateinterceptors.ServerElapsedTrailer,
|
|
strconv.FormatInt(time.Since(start).Microseconds(), 10))
|
|
w.Header().Set("Content-Type", "application/x-protobuf")
|
|
_, _ = w.Write(out)
|
|
}
|
|
}
|