chore: merge main into adtrack landing fix

This commit is contained in:
Taus
2026-09-07 09:45:29 +06:00
43 changed files with 2004 additions and 164 deletions
+9 -2
View File
@@ -32,7 +32,9 @@ Regenerate via `scripts/build-map.py`.
| `scripts/build_plugin.py` | interface/skill.md |
| `scripts/catalog_drift.py` | architecture/catalog.md |
| `scripts/catalog_ingest.py` | architecture/catalog.md, architecture/instagram-oauth.md |
| `scripts/catalog_replicate_prices.py` | architecture/catalog.md |
| `scripts/catalog_validate.py` | architecture/catalog.md |
| `scripts/catalog_verify_extended.py` | architecture/catalog.md |
| `scripts/dev-local.sh` | ops/deploy.md |
| `scripts/dump_surface.py` | architecture/composition.md |
| `scripts/indexnow_submit.py` | interface/seo.md |
@@ -58,6 +60,11 @@ Regenerate via `scripts/build-map.py`.
| `src/treg/alembic/versions/0017_async_task_record.py` | architecture/data-model.md, architecture/money.md, architecture/multi-tenancy.md |
| `src/treg/alembic/versions/0018_async_resource_ownership.py` | architecture/data-model.md, architecture/money.md, architecture/multi-tenancy.md |
| `src/treg/alembic/versions/0019_async_poll_failures.py` | architecture/data-model.md, architecture/money.md |
| `src/treg/alembic/versions/0020_callrecord_created_at_indexes.py` | architecture/data-model.md |
| `src/treg/alembic/versions/0021_ledgerentry_org_created_at_index.py` | architecture/data-model.md |
| `src/treg/alembic/versions/0022_org_spent_today_counter.py` | architecture/data-model.md |
| `src/treg/alembic/versions/0023_callrecord_org_user_created_at_index.py` | architecture/data-model.md |
| `src/treg/alembic/versions/0024_membership_calls_today_counter.py` | architecture/data-model.md |
| `src/treg/analytics.py` | architecture/data-model.md |
| `src/treg/api.py` | architecture/archive.md, architecture/money.md, architecture/multi-tenancy.md, architecture/proxy-model.md, architecture/super-admin.md, interface/api.md, interface/dashboard.md, interface/landing-sandbox.md, interface/seo.md |
| `src/treg/application/__init__.py` | architecture/import-boundaries.md |
@@ -318,9 +325,9 @@ Regenerate via `scripts/build-map.py`.
| `architecture/ads-conversions.md` | `adsconv.py`, `signup.py`, `adtrack.js`, `gtag.js` |
| `architecture/archive.md` | `archive.py`, `0002_archive_tables.py`, `0003_callrecord_cached.py`, `0004_archivekey_request_shape.py`, `0011_callrecord_archive_link.py`, `service.py`, `backfill_call_archive_links.py`, `api.py`, `bootstrap.py`, `admin.py`, `asynctasks.py` |
| `architecture/auth-secrets.md` | `injectors.py`, `ssrf.py`, `crypto.py`, `oauth.py`, `__init__.py`, `authorization.py`, `oauth_flow.py`, `refresh.py`, `oauth_exchange.py`, `oauth_refresh.py`, `oauth_providers.py`, `health.py`, `connect.py`, `connections.py`, `resources.py`, `__init__.py`, `bindings.py`, `bundles.py`, `test_oauth_refresh.py`, `config.py` |
| `architecture/catalog.md` | `contracts.yaml`, `adapters.yaml`, `findymail.search.business-profile.json`, `__init__.py`, `contracts.py`, `paths.py`, `plan.py`, `synthetic.py`, `route.py`, `test_routing.py`, `catalog-drift.yml`, `catalog_drift.py`, `catalog_ingest.py`, `catalog_validate.py`, `aliases.yaml`, `fx.yaml`, `aviato.yaml`, `crustdata.yaml`, `aviato.companies.acquisitions.json`, `aviato.companies.employees.json`, `aviato.companies.enrich.bulk.json`, `aviato.companies.enrich.json`, `aviato.companies.founders.json`, `aviato.companies.funding_rounds.json`, `aviato.companies.investments.json`, `aviato.companies.outbound_investments.json`, `aviato.companies.search.json`, `aviato.linkedin.company.posts.json`, `aviato.linkedin.post.comments.json`, `aviato.linkedin.post.reactions.json`, `aviato.linkedin.post.reposts.json`, `aviato.linkedin.user.posts.json`, `aviato.people.contact.get.json`, `aviato.people.email.find.json`, `aviato.people.enrich.bulk.json`, `aviato.people.enrich.json`, `aviato.people.phone.find.json`, `aviato.people.search.json`, `aviato.people.search.simple.json`, `crustdata.companies.autocomplete.json`, `crustdata.companies.enrich.json`, `crustdata.companies.identify.json`, `crustdata.companies.jobs.search.json`, `crustdata.companies.search.json`, `crustdata.people.autocomplete.json`, `crustdata.people.enrich.json`, `crustdata.people.search.json`, `google-search-console.yaml`, `google-search-console.extended.yaml`, `google-tag-manager.yaml`, `google-tag-manager.extended.yaml`, `instagram.yaml`, `instagram.extended.yaml`, `justoneapi.extended.yaml`, `minimax.yaml`, `apify.yaml`, `brightdata.yaml`, `companyenrich.yaml`, `oceanio.yaml`, `akta.extended.yaml`, `dataforseo.extended.yaml`, `tikhub.extended.yaml`, `minimax.video-gen.result.retrieve.json`, `minimax.video-gen.from_image.json`, `minimax.video-gen.task.status.json`, `openrouter.yaml`, `openrouter.extended.yaml`, `openrouter.x.alibaba-wan-3-0.json`, `replicate.yaml`, `replicate.extended.yaml`, `replicate.image-gen.flux-schnell.json`, `__init__.py`, `store.py`, `settlement.py`, `stats.py`, `catalog_observations.py`, `catalog.py`, `test_aigc_pr_b.py`, `test_catalog_api.py`, `test_catalog_validate.py` |
| `architecture/catalog.md` | `contracts.yaml`, `adapters.yaml`, `findymail.search.business-profile.json`, `__init__.py`, `contracts.py`, `paths.py`, `plan.py`, `synthetic.py`, `route.py`, `test_routing.py`, `catalog-drift.yml`, `catalog_drift.py`, `catalog_ingest.py`, `catalog_replicate_prices.py`, `catalog_validate.py`, `catalog_verify_extended.py`, `aliases.yaml`, `fx.yaml`, `aviato.yaml`, `crustdata.yaml`, `aviato.companies.acquisitions.json`, `aviato.companies.employees.json`, `aviato.companies.enrich.bulk.json`, `aviato.companies.enrich.json`, `aviato.companies.founders.json`, `aviato.companies.funding_rounds.json`, `aviato.companies.investments.json`, `aviato.companies.outbound_investments.json`, `aviato.companies.search.json`, `aviato.linkedin.company.posts.json`, `aviato.linkedin.post.comments.json`, `aviato.linkedin.post.reactions.json`, `aviato.linkedin.post.reposts.json`, `aviato.linkedin.user.posts.json`, `aviato.people.contact.get.json`, `aviato.people.email.find.json`, `aviato.people.enrich.bulk.json`, `aviato.people.enrich.json`, `aviato.people.phone.find.json`, `aviato.people.search.json`, `aviato.people.search.simple.json`, `crustdata.companies.autocomplete.json`, `crustdata.companies.enrich.json`, `crustdata.companies.identify.json`, `crustdata.companies.jobs.search.json`, `crustdata.companies.search.json`, `crustdata.people.autocomplete.json`, `crustdata.people.enrich.json`, `crustdata.people.search.json`, `google-search-console.yaml`, `google-search-console.extended.yaml`, `google-tag-manager.yaml`, `google-tag-manager.extended.yaml`, `instagram.yaml`, `instagram.extended.yaml`, `justoneapi.extended.yaml`, `minimax.yaml`, `apify.yaml`, `brightdata.yaml`, `companyenrich.yaml`, `oceanio.yaml`, `akta.extended.yaml`, `dataforseo.extended.yaml`, `tikhub.extended.yaml`, `minimax.video-gen.result.retrieve.json`, `minimax.video-gen.from_image.json`, `minimax.video-gen.task.status.json`, `openrouter.yaml`, `openrouter.extended.yaml`, `openrouter.x.alibaba-wan-3-0.json`, `replicate.yaml`, `replicate.extended.yaml`, `replicate.image-gen.flux-schnell.json`, `__init__.py`, `store.py`, `settlement.py`, `stats.py`, `catalog_observations.py`, `catalog.py`, `test_aigc_pr_b.py`, `test_catalog_api.py`, `test_catalog_validate.py` |
| `architecture/composition.md` | `bootstrap.py`, `bootstrap_handlers.py`, `bootstrap_http.py`, `call_surface.py`, `connect.py`, `mcp_oauth.py`, `session.py`, `admin.py`, `auth.py`, `billing.py`, `call.py`, `connections.py`, `onboard.py`, `orgs.py`, `resources.py`, `referrals.py`, `web.py`, `dump_surface.py`, `test_app_roles.py` |
| `architecture/data-model.md` | `alembic.ini`, `env.py`, `0001_baseline_current_schema.py`, `0002_archive_tables.py`, `0003_callrecord_cached.py`, `0004_archivekey_request_shape.py`, `0005_capacity_policy_snapshot.py`, `0006_overflow_route.py`, `0007_overflow_spend.py`, `0008_org_platform_overflow_disabled.py`, `0009_callrecord_hit.py`, `0017_async_task_record.py`, `0018_async_resource_ownership.py`, `0019_async_poll_failures.py`, `0011_callrecord_archive_link.py`, `0015_idempotentcall_membership_cascade.py`, `maintenance.py`, `sitetrack.js`, `models.py`, `timeutil.py`, `db.py`, `referrals.py`, `audit.py`, `analytics.py`, `bootstrap_handlers.py`, `ratestore.py`, `auth.py`, `test_postgres_reset.py`, `test_alembic_expand_safety.py` |
| `architecture/data-model.md` | `alembic.ini`, `env.py`, `0001_baseline_current_schema.py`, `0002_archive_tables.py`, `0003_callrecord_cached.py`, `0004_archivekey_request_shape.py`, `0005_capacity_policy_snapshot.py`, `0006_overflow_route.py`, `0007_overflow_spend.py`, `0008_org_platform_overflow_disabled.py`, `0009_callrecord_hit.py`, `0017_async_task_record.py`, `0018_async_resource_ownership.py`, `0019_async_poll_failures.py`, `0020_callrecord_created_at_indexes.py`, `0021_ledgerentry_org_created_at_index.py`, `0022_org_spent_today_counter.py`, `0023_callrecord_org_user_created_at_index.py`, `0024_membership_calls_today_counter.py`, `0011_callrecord_archive_link.py`, `0015_idempotentcall_membership_cascade.py`, `maintenance.py`, `sitetrack.js`, `models.py`, `timeutil.py`, `db.py`, `referrals.py`, `audit.py`, `analytics.py`, `bootstrap_handlers.py`, `ratestore.py`, `auth.py`, `test_postgres_reset.py`, `test_alembic_expand_safety.py` |
| `architecture/import-boundaries.md` | `pyproject.toml`, `ci.yml`, `__init__.py`, `__init__.py`, `access.py`, `authorize.py`, `idempotency.py`, `overflow.py`, `route.py`, `__init__.py`, `intake.py`, `resolve.py`, `reserve.py`, `settle.py`, `evidence.py`, `service.py`, `types.py`, `client_identity.py`, `__init__.py`, `__init__.py`, `access.py`, `budgets.py`, `publicdemo.py`, `teams.py`, `usage.py`, `__init__.py`, `__init__.py`, `authorization.py`, `oauth_flow.py`, `refresh.py`, `__init__.py`, `__init__.py`, `__init__.py`, `__init__.py`, `injectors.py`, `relay.py`, `__init__.py`, `limiter.py`, `test_call_architecture.py`, `test_import_lightness.py` |
| `architecture/instagram-oauth.md` | `catalog_ingest.py`, `access.py`, `resolve.py`, `service.py`, `instagram.yaml`, `instagram.extended.yaml`, `cli.py`, `store.py`, `authorization.py`, `oauth_flow.py`, `oauth_exchange.py`, `mcp.py`, `call.py`, `index.html`, `0010_oauth_authorization_method.py`, `test_instagram_oauth_architecture.py` |
| `architecture/local-proxy.md` | `localproxy.py`, `server.js` |
+5 -4
View File
@@ -72,14 +72,15 @@ agents then built against a constitution that was wrong.
commit by design; a few other domain commits remain. Do not add another; move one out when you
touch it.
- **Table ownership.** One writer module per table; cross-domain reads are fine. Three recorded
exceptions: only money writes `org.balance_micro` and the auto-top-up fields; the call runtime
may persist an OAuth token refresh into `secret`; audit writes `callrecord`, domains only read it.
exceptions: only money writes `org.balance_micro`, the daily-spend counter (`spent_today_*`) and
the auto-top-up fields; the call runtime may persist an OAuth token refresh into `secret`; audit
writes `callrecord`, domains only read it.
- **The call runtime is self-contained.** `src/treg/application/call/` depends on no management
code (routes, login, OAuth consent, Stripe top-up), reads only membership, deny rules,
credentials, catalog prices and balances, and writes only what `tests/test_call_architecture.py`
allowlists (the ledger entries, idempotency claims, OAuth refresh, audit and telemetry, first-call
markers, tag budgets, capacity marks, overflow spend). Extend the test's allowlist in the same PR
as any new write, and expect the reviewer to ask why.
markers, tag budgets, capacity marks, overflow spend, the member's daily-cap slot). Extend the
test's allowlist in the same PR as any new write, and expect the reviewer to ask why.
- **Money.** Everything is **integer micro-USD** - never floats, never cents. The Stripe SDK lives
only in `infra/stripe.py`, orchestration in `application/billing.py`, and `reconcile.py` is
read-only. See `docs/context/architecture/money.md`.
+1
View File
@@ -298,6 +298,7 @@ Environment variables (prefix `TREG_`, read from `.env`):
| `TREG_META_CLIENT_ID` / `_SECRET` | *(empty)* | Meta app credentials for Facebook Pages, Meta Ads, and optional Instagram `page-tools` |
| `TREG_OAUTH_REVIEW_PENDING` | `instagram-login,page-messages` | Registry review keys awaiting production access. Remove `page-messages` after Page messaging approval; set empty after direct Instagram approval. |
| `TREG_RESEND_API_KEY` / `TREG_EMAIL_FROM` | *(empty)* | transactional email via Resend (OTP codes + invites); From must be a Resend-verified sender |
| `TREG_BLOCKED_EMAIL_DOMAINS` | *(empty)* | comma-separated email domains added to the built-in throwaway/farm blocklist, refused at every sign-up/sign-in door and at team creation (subdomains included, case-insensitive) |
| `TREG_ADMIN_TOKEN` | *(empty)* | cross-tenant **super-admin** bearer; authorizes every `/admin/*` endpoint. Empty disables the env path (only `is_superadmin` users reach `/admin`). Keep it long + secret. |
| `TREG_EMAIL_DEV_MODE` | `false` | when true, `/auth/email/start` returns the OTP in its response (no mail sender needed) — **dev/local only**, never in prod. |
+7
View File
@@ -30,6 +30,13 @@ environment) and enforcement happens server-side and in the operating system.
- **Server runs are resource-limited.** `treg run --server` executes each CLI with a scrubbed environment
(treg's own secrets removed), a per-run throwaway home, an allow-list of runnable commands, output
redaction, and POSIX resource limits (CPU, file size, no core dumps).
- **Signup abuse has a brake.** Every new team gets a small promotional balance, which makes
throwaway-email farming worth an attacker's time. An email-domain blocklist (throwaway-mail rules
and confirmed farm roots in code, plus `TREG_BLOCKED_EMAIL_DOMAINS` for a new root without a
redeploy; subdomains included, case-insensitive, domain only) refuses the address at every sign-up
and sign-in door and at both team-creating endpoints. The refusal names neither the list nor the
domain, every block is logged, and a classifier failure lets the sign-in through rather than
breaking real signups.
## Known limitations (by design, documented on purpose)
+18 -11
View File
@@ -287,20 +287,27 @@ worker whenever the archive records, `archive_prune_batch` (500) bodies per pass
`archive_prune_interval_s` (3600); batch 0 disables. Rollup counters (bodies_kept, kept_bytes)
move atomically with each strip.
## Recorder throttle (2026-09-03)
## Recorder throttle (2026-09-03, memory-bounded 2026-09-07)
At most `_MAX_CONCURRENT_WRITES` (4) recordings touch the database at once — audit's exact
loop-bound-semaphore pattern. Before it, a burst could put up to 512 concurrent short sessions in
front of the API's 15-slot pool (SToneX's pool-pressure report); those writes now land on the
BACKGROUND pool instead (`ops/deploy.md` § Three pools), so the semaphore is the inner bound rather
than the only one — a third module reaching for the wrong maker no longer needs its author to have
read this section. Queued recordings wait inside
their fire-and-forget task, so the caller is unaffected; the 30s bound covers wait+write, so a
stuck queue still sheds rather than wedges. Throttled, not shed: the burst test proves all 12
concurrent recordings land while peak DB concurrency stays ≤4.
At most `_MAX_CONCURRENT_WRITES` (2) recordings touch the database at once — audit's exact
loop-bound-semaphore pattern (four until 2026-09-07; every slot is paid per uvicorn worker and
again per rolling-deploy instance, and a recording is one INSERT of a body already in memory).
Before it, a burst could put up to 512 concurrent short sessions in front of the API's 15-slot pool
(SToneX's pool-pressure report); those writes now land on the BACKGROUND pool instead
(`ops/deploy.md` § Three pools), so the semaphore is the inner bound rather than the only one.
Queued recordings wait inside their fire-and-forget task, so the caller is unaffected; the 30s
bound covers wait+write, so a stuck queue still sheds rather than wedges. Throttled, not shed: the
burst test proves all 12 concurrent recordings land while peak DB concurrency stays ≤2.
**Memory bound (2026-09-07 OOM fix).** Each pending task holds its `body` bytes in a closure — up to
`_MAX_PENDING` (512) tasks × `archive_max_body_bytes` (2 MB) = 1 GB worst case. After #363 reduced
concurrent writes from 4 to 2, backlog built faster under heavy traffic and the 2026-09-07T00:43:06Z
OOM killed production at 4 GB. `_MAX_PENDING_BYTES` (256 MB) now caps total body bytes held by
pending work: `record()` sheds when EITHER the task count OR the bytes threshold is exceeded. The
done callback releases bytes when a task completes, keeping the budget accurate.
The semaphore is process-local, while production runs multiple processes. An exact in-process key
lock is acquired before the semaphore, so duplicate recordings queue without consuming all four
lock is acquired before the semaphore, so duplicate recordings queue without consuming both
database-write slots and unrelated keys keep moving; weak references discard inactive locks. Once
admitted, the writer locks and refreshes the matching `ArchiveKey` row before reading the newest
snapshot and allocating version N+1. The refresh matters because the earlier unlocked lookup
+30 -3
View File
@@ -17,6 +17,10 @@ sources:
- src/treg/alembic/versions/0018_async_resource_ownership.py
- src/treg/alembic/versions/0019_async_poll_failures.py
- src/treg/alembic/versions/0020_callrecord_created_at_indexes.py
- src/treg/alembic/versions/0021_ledgerentry_org_created_at_index.py
- src/treg/alembic/versions/0022_org_spent_today_counter.py
- src/treg/alembic/versions/0023_callrecord_org_user_created_at_index.py
- src/treg/alembic/versions/0024_membership_calls_today_counter.py
- src/treg/alembic/versions/0011_callrecord_archive_link.py
- src/treg/alembic/versions/0015_idempotentcall_membership_cascade.py
- src/treg/maintenance.py
@@ -92,8 +96,10 @@ uses this metadata, never the encrypted token's shape.
- **`Membership`** - links a user to an org: `user_id`, `org_id`, `role` (owner|admin|member),
`token_hash` (SHA-256 of the bearer token, shown once), `webhook_url` (health alerts POST here),
`daily_call_cap` (per-user, per-day usage cap; **-1 = unlimited**, the default - see
`api._enforce_daily_cap`); unique `(user_id, org_id)`. **A token = a `(user, org)` pair.** `ROLE_RANK`
orders the roles.
`governance/usage.enforce_daily_cap`) with `calls_today` / `calls_today_day`, the counter that cap
is checked against (one conditional UPDATE per capped event, revision 0024; only capped members are
counted, the roster reads the journal); unique `(user_id, org_id)`. **A token = a `(user, org)`
pair.** `ROLE_RANK` orders the roles.
- **`Invite`** - a one-time join code: `org_id, email, role, code_hash (idx), status`
(pending|accepted|revoked), `invited_by`. Carries a SECOND split secret, `email_token_hash (idx,
nullable)` - the inbox-only sign-in token embedded ONLY in the invite email's link (the
@@ -152,6 +158,27 @@ uses this metadata, never the encrypted token's shape.
here queues every other query and the API pool empties into `503 treg_saturated` - see
[deploy](../ops/deploy.md) § Three pools. The table has no retention sweep yet, so it only grows.
**`LedgerEntry` is the other one, and it was the larger.** It is append-only and never pruned
(4.38M rows / 2.3 GB on prod 2026-09-06, ~400k rows a day), and `ledger.spent_today` - the
fail-closed daily cap - reads it on EVERY metered call, inside the reserve transaction, on an
api-pool connection. With only single-column indexes the planner walked the whole platform's day
through `ix_ledgerentry_created_at` and filtered the org in memory: 322k rows discarded and 381k
buffer touches per call, 56-106 s once the day's pages had been evicted from a 512 MB cache, and
the heap had read 6.5 BILLION blocks - four times `callrecord`. Revision 0021 adds
`(org_id, created_at)`, which also serves `entries_of` (the `/billing` page, previously a
backward walk of the whole `created_at` index). That fixed light orgs and `/billing` but not the
two orgs writing half the day - their rows are on every page of the day, and the planner kept
walking it (395k buffer touches per call after 0021). So the cap no longer reads this table at
all: revision 0022 adds `Org.spent_today_micro` / `spent_today_day`, kept by `domain/money`
inside the balance UPDATE and read with one primary-key lookup; the journal aggregate survives
as `spent_today_from_ledger` for reconciliation. The same shape on `callrecord` - the per-user
daily call cap, `count_today`, which BitmapAnd-ed a member's whole history through
`ix_callrecord_user_email` (2.6 s of 3.0 s for a 287k-row member) - gets
`(org_id, user_email, created_at)` in revision 0023, and then the same answer as the ledger: the
index-only scan still fetched the heap for today's not-yet-vacuumed pages (110k heap fetches,
2.8 s), so revision 0024 moves the gate to `Membership.calls_today` and the journal count is
left to the roster and `/usage/me`.
`refused_by` distinguishes a treg refusal (`auth`, `policy`, `balance`, `cap`, `resolution`,
`request`, and other mechanism-specific values) from an upstream answer, where it is null.
In-handler audits mark their own outcome; the shared exception handler records earlier
@@ -374,7 +401,7 @@ key, including first sightings that never recurred to carry their own count out.
the shared queue is genuinely backing up (`_FAULT_QUEUE_SHARE` of `_MAX_PENDING`) - the congestion the
throttle was ever meant to prevent, rather than a wall-clock rate that fired against an empty queue.
**Losing data is ERROR, not WARNING.** `audit._write`, `audit._schedule`'s back-pressure shed, and
**Losing data is ERROR, not WARNING.** `audit._write_batch`, `audit._enqueue`'s back-pressure shed, and
`archive`'s `_store`/`_touch_write` drops all log at ERROR, because `FaultCaptureHandler` starts at ERROR:
below it the loss reaches container stdout and nothing else, so it can neither be alerted on nor found
without already suspecting it. Degradations that cost nothing (an archive lookup falling back to a live
+24 -3
View File
@@ -623,9 +623,12 @@ a compliant figure and together exceed the cap. Overshoot is bounded by `concurr
estimate`, and that is acceptable **only** because the hard gates sit behind it - the org balance and
the per-org daily cap.
Making it exact would need a second materialized authority on spend: reset daily, decremented on
release, corrected on settle divergence. Four new ways to disagree with `domain/money`, which is the one
module allowed to move money. Not worth it. Never document these caps to builders as hard limits.
Making it exact would need a second materialized authority on spend per tag: reset daily, decremented
on release, corrected on settle divergence. Four new ways to disagree with `domain/money`, which is the
one module allowed to move money. Not worth it for tag caps. (The per-org daily cap DOES have exactly
such a counter, `Org.spent_today_micro`, since 2026-09-06 - kept by `domain/money` itself, inside the
balance UPDATE, so there is no second writer to disagree with; see below.) Never document these caps
to builders as hard limits.
### Refusal bodies are not the org's
@@ -661,6 +664,24 @@ and the deployment's `platform_daily_cap_usd` ceiling (default $500/day). The te
limit and inspect it through `GET /orgs/{id}/settings`. A request above the platform ceiling is
refused, not silently clamped.
The check itself, `ledger.spent_today`, is the most-run query on the platform: every metered call,
inside the reserve transaction, on an api-pool connection, fail-closed. Its cost is therefore the
platform's throughput, and it is ONE primary-key read of `Org.spent_today_micro` /
`spent_today_day` (revision 0022). `domain/money` keeps that counter inside the same UPDATE that
moves the balance: reserve adds the charged estimate, settle adds what was consumed and removes
the estimate it replaces, release removes the estimate - in each case only when the hold was
opened today, because a hold opened yesterday was yesterday's commitment. The first movement of a
new UTC day resets the counter to that movement (`_spent_today_values`, one CASE expression).
`spent_today_from_ledger` computes the same number from the journal over `(org_id, created_at)`
(revision 0021) for reconciliation; `tests/test_daily_spend_counter.py` asserts the two agree
through reserve, settle (under and over the estimate), release and the day boundary.
Why a counter and not an index: until 2026-09-06 the check was that journal aggregate, and for an
org that writes a large share of the platform's day its rows sit on nearly every heap page of the
day, so no index makes the aggregate cheaper than reading the day - measured 395k buffer touches
per call, 56-171 s once those pages were cold, holding an api-pool slot throughout. That was the
API-pool saturation (see [deploy](../ops/deploy.md) § Three pools).
## Referrals
`domain/referrals.py` owns policy; credit moves only through `ledger.grant`. Rewards are flat,
@@ -114,6 +114,41 @@ pair, so every list/create/mutation and the proxy are scoped to the caller's org
Every identity door is blocked at the shared choke point `_find_or_create_user`, plus `register_user`
(which predates it and creates a `User` directly) and `auth_email_start` (refuse early, mint no code).
`list_members` carries `is_agent` so one roster can show people and machines apart.
- **Email-domain blocklist.** The same choke points, for throwaway mail and domains used for bulk
registration. A new team is created with a promotional balance, which is what makes registering in
bulk on throwaway addresses worth someone's while. **Two tiers, one classifier**
(`_is_blocked_email` in `domain/identity/access.py`, pure: it only answers). The CODE tier is
`BLOCKED_EMAIL_DOMAINS` (domains confirmed abusive in our own data, and `my.id` so every free
`.my.id` subdomain falls to the walk) plus `BLOCKED_EMAIL_KEYWORDS`, substring rules on the domain
(`tempmail`, `mailinator`, `guerrilla`, `10minute`, ...) that catch domains no static list has
seen. The OPS tier is `TREG_BLOCKED_EMAIL_DOMAINS`, comma-separated, **added to** the code tier
and parsed once per distinct value in `config.py` (trim, drop a leading `@`/`.`, lowercase, and
drop any dotless entry so a typed `com` cannot refuse the world): the next domain is a **dashboard
edit, no redeploy**. The rules, each of which exists because the obvious implementation is wrong:
match the **domain only**, never the whole address (matching the address false-flags real users
whose username happens to contain a keyword); **walk parent domains**, whole labels off the front
and never the bare last label, because registering `<random>.<blocked-root>` is otherwise a
one-line bypass; **sign-in as well as sign-up** (an account that predates the listing gets no
new session; existing accounts are suspended out of band). The DECISION lives in the application
layer, `signup.blocked_email(email, door)`: it refuses, writes one structured line per block
(`event=signup_blocked_domain door=<door> domain=<domain>` — the refusal reveals nothing, so the
log is the only detection a burst has), and **fails open**, logging `event=blocklist_error`
and letting the sign-in through if the classifier ever raises, because a misconfiguration must
never break a real sign-in. The doors: `start_email_login` (before the rate window, so no code and
no mail), `find_or_create_user` (so OTP verify, the GitHub and Google callbacks and the emailed
invite link `POST /auth/invite-signin` refuse before the row lookup, raising
`signup.BlockedEmailError` which each door translates to a `blocked_domain` kind), `register_user`
(`POST /users` mints user + team + promo in one call), `create_org` (`POST /orgs`, the other promo
door, reachable with a token minted before the listing) and the code-based `POST /invites/accept`
(which constructs a `User` directly, so it guards itself). Every refusal is the `machine_identity`
sibling's exact 403 `this address cannot be used to sign in` (a brand page on the browser doors,
like `suspended`): the caller learns neither that a list exists nor what is on it. Deliberately a
blocklist and nothing more: no allowlist, no table, no admin UI. Not covered: a session or identity
token already live when the domain was listed keeps working until suspension or expiry (the
out-of-band suspension); the promo grant and referral bonus are not separately gated, since with
the doors closed no promo-funded team on a blocked domain can come into existence; and vendoring a
full public disposable-domain list is a follow-up (megabytes of package data in the base wheel,
which also ships the light CLI, and not yet checked against real users).
**A rotate replaces the TOKEN, never the limits.** Because rotate is the same endpoint as create, an
absent optional field used to fall back to its permissive default — and the dashboard's Rotate button
sends only `{name, role, daily_call_cap}`, so a scoped agent silently became unrestricted
+17 -5
View File
@@ -246,10 +246,17 @@ validated before resolving the shared HTTP client. `/auth/logout` remains an HTT
(`PATCH /orgs/{id}/members/{user}/cap`, admin+) sets `Membership.daily_call_cap` (`-1` = unlimited,
rejects `< -1`). `my_usage` (`GET /usage/me`, any member) returns the caller's own `used_today` + `cap`.
`list_members` also returns each member's `daily_call_cap` + `used_today`. **Enforcement:**
`_enforce_daily_cap` runs at the top of `call_tool`, `run_tool_server`, and `grant_local_run` (so no
path dodges the cap); `count_today` = today's `CallRecord` + `RunRecord` for the user. `-1` (default)
skips the count entirely (zero overhead); the sandbox is exempt. **Soft by design** - it counts the
best-effort `CallRecord`, so under load it fails *open*, never closed.
`governance/usage.enforce_daily_cap` runs in `/call/`'s authorization gate and, through
`routers/call._enforce_daily_cap`, at the top of `run_tool_server` and `grant_local_run` (so no path
dodges the cap). It takes the slot with ONE conditional UPDATE of `Membership.calls_today` /
`calls_today_day` (`take_daily_slot`, revision 0024): the WHERE is the check and the SET is the
count, so the cap is exact under concurrency, a refused event is not counted, and the first event of
a new UTC day starts from 1. `-1` (default) skips it entirely (zero overhead); the sandbox is
exempt; a database error fails *open* (a courtesy limit, not a money gate). `used_today` in the
roster and `/usage/me` still comes from the journal (`count_today` = today's `CallRecord` +
`RunRecord`), and `set_member_cap` copies that journal count onto the counter when a member goes
from unlimited to capped, so a cap set mid-day does not start from zero. Until 2026-09-06 the gate
itself ran that journal count - 2.8 s per call for a member with 110k rows that day.
- **Super-admin (cross-tenant, `require_superadmin`):** `/admin/stats|orgs|orgs/{id}|users|tools|calls|
errors|health` (reads - `errors` is failed calls across every credential tier with captured,
admin-only request/response evidence, supports a `tier` filter, and runs the 14-day retention pass;
@@ -356,7 +363,12 @@ validated before resolving the shared HTTP client. `/auth/logout` remains an HTT
- **Identity doors:** GitHub, Google and email OTP share first-proof user provisioning. They create
a user without an automatic org; new users name their first team through onboarding or the CLI
login picker. Suspended users are refused at every door.
login picker. Suspended users are refused at every door, and so is any address on a blocked email
domain (throwaway-mail rules and confirmed farm roots in code, plus `TREG_BLOCKED_EMAIL_DOMAINS`;
subdomains included): the OTP start and verify, both social callbacks, the emailed invite link,
plus `POST /users`, `POST /orgs` and `POST /invites/accept`, all with the same 403 `this address
cannot be used to sign in` the machine-identity guard uses. See
[multi-tenancy](../architecture/multi-tenancy.md).
- GitHub/Google: `GET /auth/{provider}[/callback]`, optional `?cli=<id>`; callbacks validate
state before resolving the shared HTTP client, require a proven email, and set the session
+101 -12
View File
@@ -72,7 +72,7 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
|---|---|---|---|
| `api` | `session_maker` | 5 + 10 | every request handler, via `get_session` or directly |
| `admin` | `admin_session_maker` | 3 + **0** | `/admin/*` only, via `get_admin_session` |
| `background` | `background_session_maker` | 13 + **0** | audit, archive writes, ads worker, the observation reader, the error-evidence sweep |
| `background` | `background_session_maker` | 8 + **0** | audit (one batching writer), archive writes (two), ads worker, the observation reader, the error-evidence sweep |
Each class of work can exhaust only its own slots. Before this there was ONE pool of 15, and on
2026-09-03 a single admin browser tab polling `/admin/archive/panel` (every 5 s, no in-flight
@@ -82,6 +82,31 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
(`audit.py` did, `archive.py` did not), while a pool bounds every module routed to it. Overflow
is **0** on both minor pools for the same reason: it is the escape hatch a bulkhead must not have.
**Every number above is PER PROCESS, and the reference deployment runs two.** Render sets
`WEB_CONCURRENCY=2` on the web service's 2c-4g plan (not a dashboard variable - injected at
runtime, and absent on the crons) and uvicorn honors it: the boot log shows `Started parent
process` then two `Started server process` lines. Each worker opens its own three pools, runs its
own copy of every in-process background task (ads, archive refresh, prune, the gauge), and a
rolling deploy runs two instances for about a minute. So the budget is
`per_process × 2 workers × 2 instances` against `max_connections` (103 on the 1c-2g plan), and
`infra/db.connection_budget` logs it at boot:
| specs | per process | per instance | deploy peak | 103? |
|---|---|---|---|---|
| code defaults 15 + 3 + 13 (until 2026-09-07) | 31 | 62 | **124** | over |
| code defaults 15 + 3 + 8 (since 2026-09-07) | 26 | 52 | 104 | over by one |
| dashboard override 15 + 2 + 4 | 21 | 42 | 84 | fits |
That is the post-mortem of the 2026-09-04 defaults: `background = 13` did not overload the
database, it opened 124 connections at every deploy and restart until the override cut it to 84.
Every earlier passage in this file that multiplied by two instances only was counting half the
connections. Any resize must clear the deploy-peak column first; within it there are 2 spare
per process today (23 → 92). Batching the audit writer (4 → 1) and halving the archive semaphore
(4 → 2) on 2026-09-07 cut the derived `background` from 13 to 8, so a pool of 6 now serves every
consumer but two archive writers at once and still fits (15 + 2 + 6 = 23 → 92); the way to more
is `WEB_CONCURRENCY=1`, a larger database plan, or a pooler -
not a bigger number in the override.
Two sizing rules, both learned by getting them wrong first:
- **`background` is derived, not chosen.** `BACKGROUND_CONSUMERS` lists everything that can hold
@@ -116,11 +141,36 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
deployment ran `admin.pool_size=2,background.pool_size=4` from before the 2026-09-04 bulkhead
work until 2026-09-05 — pinning both minor pools BELOW the defaults that work had just raised
(`admin` to 3, `background` to the derived 13), including the exact `admin=2` whose post-mortem
is two bullets up. A `background` of 4 against 7 consumers needing 13 does not 503; it silently
drops audit rows. The knob being a dashboard edit rather than a deploy is what makes it useful
is two bullets up. A `background` of 4 against 7 consumers needing 13 (8 since 2026-09-07) does
not 503; it silently drops audit rows. The knob being a dashboard edit rather than a deploy is what makes it useful
mid-incident and what lets it survive the fix. Today it reads
`api.pool_size=10,admin.pool_size=2,background.pool_size=4`; the two stale entries are still
there deliberately, so that the `api` change could be observed on its own.
`admin.pool_size=2,background.pool_size=4` - and those two entries are no longer "stale": with
two uvicorn workers (§ above) they are what keeps a rolling deploy at 84 connections instead of
124, so removing them is not a cleanup, it is the 2026-09-04 outage again. `api.pool_size=10` was
added on 2026-09-05 and removed on 2026-09-06: against a database that is waiting on DISK (below),
five more slots meant five more readers of the same cold pages, and the worst hour on record
(2,136 pool faults at 11:00, on a third of the previous day's traffic) followed.
- **The pools are measured, not argued about: `db_pool_gauge`.** `bootstrap.pool_gauge` samples
`infra/db.pool_snapshot()` once a second and emits one PostHog event a minute per instance:
`<pool>_peak` (most connections that pool had checked out in the minute), `<pool>_capacity`
(`pool_size + max_overflow`) and `<pool>_headroom`. Telemetry, not a database consumer, so it is
not in `ROLE_BACKGROUND_TASKS` and runs in every role. Read it like this: a pool whose peak sits
at capacity is one whose waiters are timing out (`api`: `503 treg_saturated`; `background`: an
audit or archive row dropped after `pool_timeout`); a pool whose peak never nears capacity is
holding connections nothing uses. **Resize from the gauge, never from the arithmetic** - the
arithmetic got both minor pools wrong once each (above), and the 2026-09-05 `api` raise made the
saturation it meant to fix worse. The protocol: one pool at a time, one override at a time, each
setting across at least one full daily peak (the 01:00-04:00 UTC batch window), judged by the same
hour on consecutive days on three numbers - db_pool faults, `/call/` 503 rate, and the gap between
`tool_called` events and `callrecord` rows (dropped audit). A change that raises the 503 rate at
equal traffic is reverted, not tuned around.
```
SELECT toStartOfHour(timestamp) h, max(toFloat(properties.background_peak)) bg_peak,
any(properties.background_capacity) bg_cap, max(toFloat(properties.api_peak)) api_peak
FROM events WHERE event = 'db_pool_gauge' AND timestamp > now() - INTERVAL 2 DAY
GROUP BY h ORDER BY h DESC
```
- **No statement timeout yet.** The pools bound how many connections a class of work can hold, not
how long a query may run; `alembic/env.py` still has the only timeouts in the app. Adding per-pool
`statement_timeout` is deliberately a SEPARATE change: it is a behavior change on every query,
@@ -147,6 +197,23 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
through `ix_callrecord_endpoint_id_id` because no index carried `created_at` (revision 0020
adds the pairs). **Reach for a pool size only after ruling out a scan;** raising it buys headroom
and hides the cause.
That diagnosis was half right. Re-measured 2026-09-06 with wait events instead of response
times: 88 % of active backends sat in `IO DataFileRead` / `IPC BufferIO` (waiting for a page, or
for ANOTHER backend reading the same page), 18 of 879 samples were on CPU. The database is not
CPU-bound; it is a 512 MB buffer cache in front of 35 GB, and the query holding the pool was
not on `callrecord` at all: 674 of 879 active samples were `ledger.spent_today` on
`ledgerentry`, the fail-closed daily cap that runs inside EVERY metered call's reserve
transaction on an api-pool connection, scanning the whole platform's day because no index paired
`org_id` with `created_at` (revision 0021 adds it; `ledgerentry` had read 6.5 BILLION heap
blocks, four times `callrecord`). Whenever a large scan evicts the day's ledger pages - the
30-day observation refresh, the `/billing` page's backward index walk, the per-call
`idempotentcall` sweep, a concurrent index build - every in-flight `spent_today` stalls together
for tens of seconds, and 20 slots are gone. 0021 fixed light orgs and `/billing` only: the two
orgs writing half the day sit on every page of the day and the planner kept walking it, so
revision 0022 moved the cap to a counter on the org row (one primary-key read) and 0023 gave the
per-user cap its triple on `callrecord`. Two lessons: **sample `wait_event_type`, not
latency**, and on a disk-bound database a bigger pool is more contention, not more throughput.
- **SQLite aliases all three to one engine.** It has no pool to protect and file-level write locks
it cannot share, so three engines against one file would only manufacture "database is locked".
Tests therefore pin the ROUTING (which maker each module reaches for), not the isolation.
@@ -241,6 +308,18 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
code is exposed only through `Settings.expose_dev_code`, which requires `email_dev_mode` **and** a
**local sqlite** `database_url` — so even a stray `TREG_EMAIL_DEV_MODE=true` on Postgres (a real deploy)
can never leak a login code.
- `blocked_email_domains` (`TREG_BLOCKED_EMAIL_DOMAINS`, default empty) - the OPS tier of the
email-domain blocklist: comma-separated domains ADDED to the code tier (treg's confirmed farm
roots and the throwaway-mail keyword rules in `domain/identity/access.py`), refused at every
identity door and at both team-creating doors (`POST /users` and `POST /orgs`). Example:
`newfarm.io,other-farm.net`. Case-insensitive; a listed domain also blocks its subdomains; a
leading `@` or `.` and surrounding whitespace are tolerated; a dotless entry (`com`) is ignored.
The signup-grant-farm brake: edit it in the Render dashboard the moment a new root appears, no
redeploy; promote a root into the code tier in the next PR. Empty adds nothing (the code tier
stays in force). Existing accounts on a listed domain must be suspended separately (`/admin`);
the list only stops new sessions and new teams. Each block writes one
`event=signup_blocked_domain door=... domain=...` log line, so a wave is countable. See
[multi-tenancy](../architecture/multi-tenancy.md).
- `run_proof` (`TREG_RUN_PROOF`) — the **isolated-runner proof** for `treg run --local`. A local run whose
grant would return a secret the caller does **not** own (a shared-key tool a member may run but not read)
must present this value in the `X-Treg-Run-Proof` header — a value held **only** by the root-installed
@@ -317,13 +396,15 @@ UTC (SQLite is lax and hid this; it only bites on Postgres — the deploy target
and in the serial Postgres CI migration set. `env.py` bounds Postgres lock and statement wait time so
a contended migration fails before it queues the serving database behind DDL.
**Audit back-pressure (`audit.py`).** Audit rows are written off the request path (fire-and-forget), and
each write opens a DB connection — from the **background** pool since 2026-09-03, so a burst here can no
longer starve real requests, only other background work. Two limits still apply inside it: a loop-bound
semaphore caps concurrent audit writes at `_MAX_CONCURRENT_WRITES` (queueing in-process rather than
holding a pooled connection, and keeping `drain()` deterministic on SQLite, where all three makers share
one engine), and under an extreme burst the writer **sheds** load — it drops any audit row past
`_MAX_PENDING` rather than let the pending set grow without bound. Audit must never OOM or wedge the
**Audit back-pressure (`audit.py`).** Audit rows are written off the request path (fire-and-forget):
`record_call` appends to an in-process queue and ONE writer task per process drains it `_BATCH` rows
per INSERT on a **background**-pool connection (since 2026-09-07; before that four writers each took
one row per session, which cost four slots per process for millisecond inserts). A burst can therefore
never starve real requests, only other background work. Two limits still apply: a loop-bound semaphore
holds the writer to `_MAX_CONCURRENT_WRITES` (1), which keeps `drain()` deterministic on SQLite, where
all three makers share one engine, and under an extreme burst `_enqueue` **sheds** load — it drops any
audit row past `_MAX_PENDING` queued rows rather than let the queue grow without bound. A batch the
database refuses is retried row by row, so one bad row costs one row. Audit must never OOM or wedge the
server. Shedding is the *only* loss that should ever happen: `record_call` splats its telemetry dict
into `CallRecord(**fields)`, so a key with no matching column used to raise inside `_write`, where the
except swallowed it, and the whole row disappeared — a telemetry field deployed one commit ahead of its
@@ -333,6 +414,14 @@ is the whole point: `FaultCaptureHandler` starts at ERROR, so at WARNING a lost
stdout and nothing else, and the only way to learn audit was dropping was to already suspect it and go
grep. **A quiet audit table is now a bug you can alert on**, not one you find out about weeks later.
**Archive memory bound (`archive.py`).** Each pending archive recording holds its `body` bytes in a
task closure — up to `_MAX_PENDING` (512) tasks × `archive_max_body_bytes` (2 MB) = 1 GB worst case.
After #363 reduced `_MAX_CONCURRENT_WRITES` from 4 to 2, backlog built faster than it drained under
heavy `/call` + MCP traffic, and the 2026-09-07T00:43:06Z OOM killed the web service at 4 GB.
`_MAX_PENDING_BYTES` (256 MB) now caps total body bytes in pending work: `record()` sheds when
EITHER the task count OR the bytes threshold is exceeded. The done callback releases bytes when a
task completes; a regression test pins the bound.
The proxy is thin and IO-bound (a relay, low CPU/memory), so cheap machines scale it.
## The per-org daily spend cap
@@ -0,0 +1,96 @@
"""composite (org_id, created_at) on ledgerentry — the per-call daily-cap check stops scanning the
whole platform's day
Revision ID: 0021
Revises: 0020
Create Date: 2026-09-06
`ledger.spent_today` runs inside EVERY metered call's reserve transaction, on an api-pool
connection: `sum(amount_micro) WHERE org_id = ? AND kind = 'settle' AND created_at >= today`.
`ledgerentry` carried only single-column indexes, so the planner had two bad choices and took one
per org: walk `ix_ledgerentry_created_at` over the WHOLE platform's day and filter `org_id` in
memory (heavy orgs), or BitmapAnd the org's ENTIRE history against the day (light orgs). Measured
on prod 2026-09-06 at 4.38M rows / 2.3 GB, with ~400k rows written per day:
heaviest org, warm cache: Rows Removed by Filter: 322,638
Buffers: shared hit=367,715 read=13,777 (539 ms)
same query, cold cache: 56–106 s, holding an api-pool slot the whole time
ix_ledgerentry_org_id 1,858,056 scans 90,870,787,481 tuples read
ix_ledgerentry_created_at 1,295,504 scans 157,269,530,757 tuples read
ledgerentry heap 6.5 BILLION blocks read — 4× `callrecord`, the largest consumer
That is the API-pool saturation. The database is a 1 vCPU / 2 GB instance with a 512 MB buffer
cache in front of 35 GB; a 30-second activity sample showed 88 % of active backends in
`IO DataFileRead` / `IPC BufferIO` (waiting for a page, or for ANOTHER backend reading the same
page) and 674 of 879 active samples were this one query. Whenever a large scan evicts the day's
ledger pages, every in-flight `spent_today` stalls together on disk for tens of seconds, each one
holding an api-pool connection, and the pool empties into `503 treg_saturated`. Raising the pool
(15 → 20 on 2026-09-05) made it worse: more concurrent readers of the same cold pages.
`(org_id, created_at)` turns both plans into one tight range: the org's rows since midnight,
nothing else. It also serves `ledger.entries_of` (the `/billing` page), which walked the whole
`created_at` index BACKWARD filtering `org_id` — measured 57 s for a quiet org. `kind` is
deliberately not a third column: the pair serves both queries, a triple would serve only one, and
every extra index on a 400k-rows/day table is paid on every write.
Built with the 0020 discipline — see that revision's docstring for why raising `lock_timeout` for
a CONCURRENT build does not break the 2026-08-15 rule (its lock conflicts with neither reads nor
writes and a statement waiting for it blocks nobody), and why each index is inspected for INVALID
debris before building. The expand-safety linter counts the autocommit escape as non-additive, so
this revision declares a rollback floor pro forma: the operation is one additive index.
"""
from collections.abc import Sequence
import sqlalchemy as sa
from alembic import op
revision: str = "0021"
down_revision: str | Sequence[str] | None = "0020"
branch_labels: str | Sequence[str] | None = None
depends_on: str | Sequence[str] | None = None
contract = True # pro forma — see the rollback floor note; the operation is one additive index
_TABLE = "ledgerentry"
_INDEXES = (
("ix_ledgerentry_org_id_created_at", ["org_id", "created_at"]),
)
# Long enough to outlast an autovacuum pass on a 2.3 GB table; see 0020 for why waiting on THIS
# lock is safe. `env.py`'s values are restored before the autocommit block ends.
_LOCK_TIMEOUT = "180s"
_STATEMENT_TIMEOUT = "600s"
_ENV_LOCK_TIMEOUT = "5s"
_ENV_STATEMENT_TIMEOUT = "120s"
_VALIDITY = sa.text(
"SELECT i.indisvalid FROM pg_class c JOIN pg_index i ON i.indexrelid = c.oid "
"WHERE c.relname = :name")
def upgrade() -> None:
if op.get_bind().dialect.name != "postgresql":
for name, columns in _INDEXES: # SQLite: no concurrent mode, no traffic to block
op.create_index(name, _TABLE, columns)
return
# CONCURRENTLY cannot run inside a transaction; alembic opens one by default.
with op.get_context().autocommit_block():
bind = op.get_bind()
bind.execute(sa.text(f"SET lock_timeout = '{_LOCK_TIMEOUT}'"))
bind.execute(sa.text(f"SET statement_timeout = '{_STATEMENT_TIMEOUT}'"))
try:
for name, columns in _INDEXES:
valid = bind.execute(_VALIDITY, {"name": name}).scalar()
if valid is True:
continue
if valid is False: # debris from a killed build — unusable, and never repaired
op.drop_index(name, table_name=_TABLE, postgresql_concurrently=True)
op.create_index(name, _TABLE, columns, postgresql_concurrently=True)
finally:
bind.execute(sa.text(f"SET lock_timeout = '{_ENV_LOCK_TIMEOUT}'"))
bind.execute(sa.text(f"SET statement_timeout = '{_ENV_STATEMENT_TIMEOUT}'"))
def downgrade() -> None:
for name, _ in _INDEXES:
op.drop_index(name, table_name=_TABLE)
@@ -0,0 +1,81 @@
"""org.spent_today_micro / spent_today_day — the daily cap reads a counter, not the journal
Revision ID: 0022
Revises: 0021
Create Date: 2026-09-06
`ledger.spent_today` is the fail-closed per-org daily cap and runs inside EVERY metered call's
reserve transaction on an api-pool connection. Until now it was two aggregates over `ledgerentry`
and `hold` since midnight. Revision 0021 gave the ledger half an `(org_id, created_at)` index and
that fixed light orgs and `/billing`, but not the two orgs that write half the platform's day:
their rows sit on nearly every heap page of the day, so the planner (correctly) keeps walking the
whole day's `created_at` range — measured after 0021 on prod 2026-09-06:
org 5430 Rows Removed by Filter: 166,760 Buffers: shared hit=394,248 439 ms warm
org 4645 Rows Removed by Filter: 385,118 Buffers: shared hit=395,506 1,797 ms warm
cold (day's pages evicted by another scan): 56–171 s, holding an api-pool slot throughout
No index can make "sum of this org's rows today" cheaper than "this org's pages today" — for a
heavy org that IS the day. The counter makes it one primary-key read: `domain/money` folds every
reserve, settle and release into `org.spent_today_micro` inside the UPDATE that already moves the
balance, and `spent_today_day` says which UTC day it belongs to (the first movement of a new day
resets it). `spent_today_from_ledger` keeps the journal view for reconciliation.
**Backfill, in its own autocommit step.** The ALTER takes ACCESS EXCLUSIVE on `org`, the hottest
row-updated table (every reserve); adding a column with a constant default is metadata-only on this
Postgres and holds that lock for milliseconds. The backfill — one range aggregate over today's
ledger and holds — would hold it for seconds if it ran in the same transaction, which is exactly
the 2026-08-15 shape. So the ALTER commits first, then the two UPDATEs run autocommit, taking
only row locks on the orgs that moved money today. Calls that reserve between the backfill and the
new code starting are missed by at most that window; the cap is a blast-radius guard, not a bill.
Rollback floor: `spent_today_micro` is NOT NULL with a server default kept (the model carries the
same default), so older code still inserts orgs; downgrading drops both columns. The expand-safety
linter counts the autocommit escape as non-additive, hence the contract marker.
"""
from collections.abc import Sequence
import sqlalchemy as sa
from alembic import op
revision: str = "0022"
down_revision: str | Sequence[str] | None = "0021"
branch_labels: str | Sequence[str] | None = None
depends_on: str | Sequence[str] | None = None
contract = True
# `timezone('UTC', now())` because the app writes NAIVE UTC timestamps and dates (models._now);
# a session in another zone would otherwise draw the day boundary in the wrong place.
_BACKFILL_SETTLED = sa.text("""
UPDATE org SET spent_today_micro = x.v, spent_today_day = (timezone('UTC', now()))::date
FROM (SELECT org_id, -sum(amount_micro) AS v FROM ledgerentry
WHERE kind = 'settle' AND created_at >= date_trunc('day', timezone('UTC', now()))
GROUP BY org_id) x
WHERE org.id = x.org_id
""")
_BACKFILL_HELD = sa.text("""
UPDATE org SET spent_today_micro = spent_today_micro + x.v,
spent_today_day = (timezone('UTC', now()))::date
FROM (SELECT org_id, sum(amount_micro) AS v FROM hold
WHERE created_at >= date_trunc('day', timezone('UTC', now()))
GROUP BY org_id) x
WHERE org.id = x.org_id
""")
def upgrade() -> None:
op.add_column("org", sa.Column("spent_today_micro", sa.BigInteger(), nullable=False,
server_default="0"))
op.add_column("org", sa.Column("spent_today_day", sa.Date(), nullable=True))
if op.get_bind().dialect.name != "postgresql":
return # SQLite deployments are dev databases with no day of history worth carrying over
with op.get_context().autocommit_block():
bind = op.get_bind()
bind.execute(_BACKFILL_SETTLED)
bind.execute(_BACKFILL_HELD)
def downgrade() -> None:
with op.batch_alter_table("org") as batch:
batch.drop_column("spent_today_day")
batch.drop_column("spent_today_micro")
@@ -0,0 +1,74 @@
"""composite (org_id, user_email, created_at) on callrecord — the per-user daily cap stops reading
a member's whole history
Revision ID: 0023
Revises: 0022
Create Date: 2026-09-06
`governance/usage.count_today` backs the per-user daily call cap and runs on every capped call:
`count(*) WHERE org_id = ? AND user_email = ? AND created_at >= today`. With 0020's
`(org_id, created_at)` and the single-column `ix_callrecord_user_email` the planner BitmapAnd-ed
the two, and the `user_email` half read the member's WHOLE history. Measured on prod 2026-09-06,
after 0021, for the busiest member (287k rows, 104k of them today):
Bitmap Index Scan on ix_callrecord_user_email rows=287,527 2,621 ms of 3,016 ms
seen in the activity sample 96 times, longest 45.8 s cold
The triple is one tight range: this org, this member, since midnight. Built with the 0020
discipline (raised `lock_timeout` for the CONCURRENT build only, INVALID-debris check, `env.py`
timeouts restored) — see that revision for why waiting on this lock blocks nobody. The expand-safety
linter counts the autocommit escape as non-additive, so this revision declares a rollback floor
pro forma: the operation is one additive index.
"""
from collections.abc import Sequence
import sqlalchemy as sa
from alembic import op
revision: str = "0023"
down_revision: str | Sequence[str] | None = "0022"
branch_labels: str | Sequence[str] | None = None
depends_on: str | Sequence[str] | None = None
contract = True # pro forma — see the rollback floor note; the operation is one additive index
_TABLE = "callrecord"
_INDEXES = (
("ix_callrecord_org_id_user_email_created_at", ["org_id", "user_email", "created_at"]),
)
_LOCK_TIMEOUT = "180s"
_STATEMENT_TIMEOUT = "600s"
_ENV_LOCK_TIMEOUT = "5s"
_ENV_STATEMENT_TIMEOUT = "120s"
_VALIDITY = sa.text(
"SELECT i.indisvalid FROM pg_class c JOIN pg_index i ON i.indexrelid = c.oid "
"WHERE c.relname = :name")
def upgrade() -> None:
if op.get_bind().dialect.name != "postgresql":
for name, columns in _INDEXES: # SQLite: no concurrent mode, no traffic to block
op.create_index(name, _TABLE, columns)
return
# CONCURRENTLY cannot run inside a transaction; alembic opens one by default.
with op.get_context().autocommit_block():
bind = op.get_bind()
bind.execute(sa.text(f"SET lock_timeout = '{_LOCK_TIMEOUT}'"))
bind.execute(sa.text(f"SET statement_timeout = '{_STATEMENT_TIMEOUT}'"))
try:
for name, columns in _INDEXES:
valid = bind.execute(_VALIDITY, {"name": name}).scalar()
if valid is True:
continue
if valid is False: # debris from a killed build — unusable, and never repaired
op.drop_index(name, table_name=_TABLE, postgresql_concurrently=True)
op.create_index(name, _TABLE, columns, postgresql_concurrently=True)
finally:
bind.execute(sa.text(f"SET lock_timeout = '{_ENV_LOCK_TIMEOUT}'"))
bind.execute(sa.text(f"SET statement_timeout = '{_ENV_STATEMENT_TIMEOUT}'"))
def downgrade() -> None:
for name, _ in _INDEXES:
op.drop_index(name, table_name=_TABLE)
@@ -0,0 +1,72 @@
"""membership.calls_today / calls_today_day — the per-user daily cap takes a slot, not a count
Revision ID: 0024
Revises: 0023
Create Date: 2026-09-06
`governance/usage.enforce_daily_cap` runs on every call and run a capped member makes. Until now it
counted the member's `callrecord` and `runrecord` rows since midnight. Revision 0023's
`(org_id, user_email, created_at)` index turned that into an index-only scan, but today's pages are
not yet all-visible until autovacuum reaches them, so every row still fetched the heap — measured
on prod 2026-09-06 for the busiest member (110k rows today):
Index Only Scan ... Heap Fetches: 110,166 Buffers: shared hit=81,738 read=14,436 2,777 ms
O(this member's rows today), per call, on an api-pool connection. Same disease as `spent_today`
(revision 0022), same cure: a counter on the row the gate already has. `take_daily_slot` is one
conditional UPDATE — the WHERE is the check and the SET is the count, so the cap is exact under
concurrency and a refused call is not counted. Only capped members are counted; the roster and
`/usage/me` keep reading the journal (`count_today`), and `seed_counter` copies today's journal
onto the row when a cap is first set, so a member capped mid-day does not start from zero.
**Backfill, in its own autocommit step, capped members only.** 123 of 9,239 memberships carry a
cap; each backfill row is one journal count over 0023's index — the heaviest ~3 s, the rest
milliseconds. It runs after the ALTER commits, so the ACCESS EXCLUSIVE lock on `membership`
(read on every request) lasts milliseconds, not the length of those counts.
Rollback floor: `calls_today` is NOT NULL with a server default kept (the model carries the same
default), so older code still inserts memberships; downgrading drops both columns. The
expand-safety linter counts the autocommit escape as non-additive, hence the contract marker.
"""
from collections.abc import Sequence
import sqlalchemy as sa
from alembic import op
revision: str = "0024"
down_revision: str | Sequence[str] | None = "0023"
branch_labels: str | Sequence[str] | None = None
depends_on: str | Sequence[str] | None = None
contract = True
# `timezone('UTC', now())` because the app writes NAIVE UTC timestamps and dates; a session in
# another zone would draw the day boundary in the wrong place.
_BACKFILL = sa.text("""
UPDATE membership SET calls_today = x.n, calls_today_day = (timezone('UTC', now()))::date
FROM (SELECT m.id,
(SELECT count(*) FROM callrecord r
WHERE r.org_id = m.org_id AND r.user_email = u.email
AND r.created_at >= date_trunc('day', timezone('UTC', now())))
+ (SELECT count(*) FROM runrecord r
WHERE r.org_id = m.org_id AND r.user_email = u.email
AND r.created_at >= date_trunc('day', timezone('UTC', now()))) AS n
FROM membership m JOIN "user" u ON u.id = m.user_id
WHERE m.daily_call_cap >= 0) x
WHERE membership.id = x.id
""")
def upgrade() -> None:
op.add_column("membership", sa.Column("calls_today", sa.Integer(), nullable=False,
server_default="0"))
op.add_column("membership", sa.Column("calls_today_day", sa.Date(), nullable=True))
if op.get_bind().dialect.name != "postgresql":
return # SQLite deployments are dev databases with no day of history worth carrying over
with op.get_context().autocommit_block():
op.get_bind().execute(_BACKFILL)
def downgrade() -> None:
with op.batch_alter_table("membership") as batch:
batch.drop_column("calls_today_day")
batch.drop_column("calls_today")
+2 -1
View File
@@ -54,7 +54,8 @@ _FLUSH_INTERVAL_S = 2.0 # max staleness before a flush
_flusher: asyncio.Task | None = None
_SERVER_DISTINCT_ID = "treg-server"
SERVER_DISTINCT_ID = "treg-server"
_SERVER_DISTINCT_ID = SERVER_DISTINCT_ID # older name, kept for callers
_FAULT_VALUE_MAX = 500
_FAULT_WINDOW_S = 10.0 # one event per (fault type, site) per window; the rest are counted
_FAULT_MAX_KEYS = 500 # bound the ledger — it is what caps the cost, so it must be finite
+21 -7
View File
@@ -2,6 +2,8 @@
from __future__ import annotations
import logging
from collections.abc import Callable
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
@@ -207,6 +209,8 @@ async def start_email_login(email: str, client_ip: str) -> dict:
raise EmailAuthError("demo_address")
if _is_machine_email(email):
raise EmailAuthError("machine_identity")
if signup.blocked_email(email, "otp_start"): # refuse early: no code, no mail, no rate window
raise EmailAuthError("blocked_domain")
async with database.session_maker() as db:
await ratestore.sweep(db, OTP_START_NS)
@@ -251,9 +255,11 @@ async def verify_email_login(email: str, code: str) -> VerifiedEmail:
raise EmailAuthError("invalid_code")
await ratestore.kv_pop(db, OTP_NS, email)
try:
user = await signup.find_or_create_user(db, email)
user = await signup.find_or_create_user(db, email, door="otp_verify")
except signup.MachineIdentityError as exc:
raise EmailAuthError("machine_identity") from exc
except signup.BlockedEmailError as exc: # a code minted before the domain was listed
raise EmailAuthError("blocked_domain") from exc
if user.suspended:
raise EmailAuthError("suspended")
await db.commit()
@@ -425,12 +431,14 @@ def start_google_login(cli: str, callback_base: Callable[[], str]) -> SocialLogi
return SocialLoginStart(state=state, url=url)
async def _provision_social_user(email: str, state: str) -> SocialLoginProof:
async def _provision_social_user(email: str, state: str, door: str) -> SocialLoginProof:
async with database.session_maker() as db:
try:
user = await signup.find_or_create_user(db, email) # first login = registration (user only; no auto org)
user = await signup.find_or_create_user(db, email, door=door) # first login = registration (user only; no auto org)
except signup.MachineIdentityError as exc:
raise SocialLoginError("machine_identity") from exc
except signup.BlockedEmailError as exc: # a Google/GitHub account on a listed domain
raise SocialLoginError("blocked_domain") from exc
if user.suspended: # a banned account may prove its email but must not receive a live session
raise SocialLoginError("suspended")
await db.commit()
@@ -470,7 +478,7 @@ async def complete_github_login(
except Exception as exc: # noqa: BLE001
print(f"[auth] github callback error: {exc}") # keep internals server-side, not in the response
raise SocialLoginError("callback_failed") from exc
return await _provision_social_user(email, state)
return await _provision_social_user(email, state, "github")
async def complete_google_login(
@@ -507,7 +515,7 @@ async def complete_google_login(
except Exception as exc: # noqa: BLE001
print(f"[auth] google callback error: {exc}") # keep internals server-side, not in the response
raise SocialLoginError("callback_failed") from exc
return await _provision_social_user(email, state)
return await _provision_social_user(email, state, "google")
async def current_identity(x_treg_token: str, session_cookie: str) -> CurrentIdentity:
@@ -600,9 +608,11 @@ async def confirm_invite_signin(email_token: str) -> InviteSigninProof:
if invite is None: # consumed / expired / revoked / suspended org → the SPA's expired banner
raise InviteSigninError("expired")
try:
user = await signup.find_or_create_user(db, invite.email) # first click = registration (user only, no auto org)
user = await signup.find_or_create_user(db, invite.email, door="invite_link") # first click = registration (user only, no auto org)
except signup.MachineIdentityError as exc:
raise InviteSigninError("machine_identity") from exc
except signup.BlockedEmailError as exc: # an invite to a listed domain must not become a session
raise InviteSigninError("blocked_domain") from exc
if user is None or user.suspended: # a banned account may hold the link but must not get a session
raise InviteSigninError("suspended")
invite.email_token_hash = None # consume: one sign-in per emailed link
@@ -871,9 +881,13 @@ async def _refresh_grant(*, refresh_token: str, client_id: str, resource: str) -
# cost of being wrong is one sign-in; the cost of the other mistake is somebody's balance.
killed = await _revoke_refresh_family(row.family_id, "reuse detected", db)
await db.commit()
# `CallRecord` has no column for the family or the kill count; audit drops unknown
# telemetry keys with a warning on every occurrence, so they go to the log instead.
logging.getLogger("treg.auth").warning(
"refresh token reuse: family %s revoked (%s grants)", row.family_id, killed)
audit.record_call(org_id=row.org_id, user_email="", tool_name="oauth.refresh_reuse",
method="POST", path="/oauth/token", status_code=400, client="",
telemetry={"family": row.family_id, "revoked": killed})
refused_by="auth")
raise OAuthServerError(
"invalid_grant",
"this refresh token was already used — the grant has been revoked, sign in again",
+33 -2
View File
@@ -12,7 +12,7 @@ from .. import adsconv, health, sandbox as demo_sandbox
from ..domain import money as ledger
from ..domain import referrals
from ..domain.governance.teams import _make_org_membership, _slugify
from ..domain.identity.access import _is_machine_email, _norm_email
from ..domain.identity.access import _email_domain, _is_blocked_email, _is_machine_email, _norm_email
from ..infra.db import session_maker
from ..models import Org, User
from ..timeutil import utcnow_naive as _utcnow_naive
@@ -30,7 +30,28 @@ class MachineIdentityError(Exception):
"""A machine identity reached a human identity-provisioning command."""
async def find_or_create_user(db: AsyncSession, email: str) -> User:
class BlockedEmailError(Exception):
"""An address on a blocked email domain reached an identity door."""
def blocked_email(email: str, door: str) -> bool:
"""The blocklist DECISION for the identity doors, over the pure classifier `_is_blocked_email`:
True means refuse. One structured line per block, so a burst of refusals is countable from the
logs — the refusal itself tells the caller nothing, so the log is the only detection. Fails OPEN:
a classifier error is logged and the door stays open, because a misconfiguration must never break
a real sign-in. `door` names the endpoint for the log only; it never reaches the caller."""
log = logging.getLogger("treg.auth")
try:
if not _is_blocked_email(email):
return False
except Exception as exc: # noqa: BLE001 - fail open, by design
log.error("event=blocklist_error door=%s error=%s", door, exc)
return False
log.warning("event=signup_blocked_domain door=%s domain=%s", door, _email_domain(email))
return True
async def find_or_create_user(db: AsyncSession, email: str, *, door: str = "login") -> User:
"""Find a user by email, else register them — the user ONLY, **no auto personal org**. The shared
core of every identity door (GitHub / Google / email OTP). A brand-new user therefore lands with
zero teams and is asked to NAME + CREATE their first team (the dashboard's mandatory welcome, or
@@ -44,6 +65,10 @@ async def find_or_create_user(db: AsyncSession, email: str) -> User:
# (The domains are unroutable, so a code could never be delivered anyway; this makes it explicit.)
if _is_machine_email(email):
raise MachineIdentityError
# Same choke point for the domain blocklist, and BEFORE the lookup on purpose: a blocked domain
# gets no session whether or not it already has a row (sign-in, not just sign-up).
if blocked_email(email, door):
raise BlockedEmailError
user = (await db.execute(select(User).where(User.email == email))).scalar_one_or_none()
if user is None:
user = User(email=email)
@@ -176,6 +201,8 @@ async def register_user(
# otherwise a caller could squat an agent address before an admin mints that agent.
if _is_machine_email(email):
raise SignupError("machine_identity")
if blocked_email(email, "register"): # mints a promo-funded team in one call; refuse first
raise SignupError("blocked_domain")
if webhook_url and not health.safe_webhook_url(webhook_url): # SSRF guard on the alert URL
raise SignupError("unsafe_webhook")
if (await db.execute(select(User).where(User.email == email))).scalar_one_or_none():
@@ -222,6 +249,10 @@ async def create_org(
async with session_maker() as db:
if demo_sandbox.is_sandbox_user(user): # anonymous sandbox visitors cannot mint real teams
raise SignupError("sandbox_user")
# An identity registered BEFORE its domain was listed still holds a live token; every team it
# creates is another promo grant, so the blocklist covers this door too, not only sign-in.
if blocked_email(user.email, "create_org"):
raise SignupError("blocked_domain")
click_field, gclid, landing = _ad_attribution_from(ad_cookie)
# A browser sign-in reaches this door instead of /users, so both doors must read attribution.
for _ in range(3): # a concurrent create can claim the slug before commit; retry a fresh lookup
+28 -7
View File
@@ -203,11 +203,19 @@ from datetime import datetime, timedelta, timezone
_log = logging.getLogger("treg.archive")
_pending: set[asyncio.Task] = set()
_MAX_PENDING = 512
# Memory bound: each pending task holds its body bytes in a closure. 512 tasks × 8 MB = 4 GB in the
# worst case — the 2026-09-07 OOM. This cap sheds recordings when total pending body bytes exceeds
# the threshold, BEFORE the task count would shed them. 256 MB is generous for a 4 GB container and
# still allows ~128 concurrent recordings of typical 2 MB bodies.
_MAX_PENDING_BYTES = 256 * 1024 * 1024
_pending_bytes = 0
# At most this many recordings TOUCH THE DATABASE at once (audit's discipline, and its exact
# loop-bound pattern). Without it a traffic burst put up to 512 concurrent short sessions in
# front of the API's 15-slot pool — SToneX's pool-pressure report, 2026-09-03. Queued recordings
# wait INSIDE their task; the caller's response left long ago either way.
_MAX_CONCURRENT_WRITES = 4
# wait INSIDE their task; the caller's response left long ago either way. Two, not four: every
# slot here is paid twice (two uvicorn workers) and again at every deploy against the database's
# 103-connection ceiling, and a recording is one INSERT of a body that is already in memory.
_MAX_CONCURRENT_WRITES = 2
_sem: asyncio.Semaphore | None = None
_sem_loop = None
@@ -285,23 +293,36 @@ def record(
bytes (`/calls/{id}/result`). Computed here rather than in `_store` so they are computed
ONCE (the store reuses them) and are true whether or not the write lands: a shed recording
still names the answer the caller received."""
global _pending_bytes
kh = cache_key(method, endpoint_id, url, caller_body, headers)
ch = content_hash(body)
if len(_pending) >= _MAX_PENDING: # shed load; the stream self-heals on the next call
body_len = len(body)
# Shed on EITHER count OR bytes — whichever bound bites first. The bytes bound prevents OOM
# when a few large bodies queue while the semaphore is full; the count bound is the legacy
# backstop for many small bodies (archive_max_body_bytes is 2 MB, so 512 × 2 MB = 1 GB).
if len(_pending) >= _MAX_PENDING or _pending_bytes + body_len > _MAX_PENDING_BYTES:
return kh, ch
_pending_bytes += body_len
task = asyncio.create_task(asyncio.wait_for(_store(
method=method, endpoint_id=endpoint_id, provider=provider, url=url,
caller_body=caller_body, headers=headers, status_code=status_code,
media_type=media_type, body=body, origin=origin, key_hash=kh, body_hash=ch),
timeout=_STORE_TIMEOUT_S))
_pending.add(task)
# NOT redundant with drain()'s own removal: on a running server drain() never fires, and this
# callback is the only exit from `_pending` — without it the set fills to _MAX_PENDING and
# record() sheds every recording from then on.
task.add_done_callback(_pending.discard)
# Release bytes AND task when done. NOT redundant with drain()'s own removal: on a running
# server drain() never fires, and this callback is the only exit from `_pending` — without it
# the set fills to _MAX_PENDING and record() sheds every recording from then on.
task.add_done_callback(lambda t: _task_done(t, body_len))
return kh, ch
def _task_done(task: asyncio.Task, body_len: int) -> None:
"""Release the task and its body bytes from the pending budget."""
global _pending_bytes
_pending.discard(task)
_pending_bytes -= body_len
async def store_terminal_response(
call_id: str, provider: str, endpoint_id: str, status_code: int, body: bytes,
) -> None:
+98 -45
View File
@@ -1,29 +1,38 @@
"""Audit writes — deferred and fire-and-forget (rule #2: never block the proxied response).
`record_call` schedules an insert on its own session and returns immediately; the response
streams without waiting. A strong reference to each task is held until it finishes (otherwise
the event loop may GC a bare create_task). Failures are swallowed: an audit hiccup must never
break a real call. `drain()` flushes pending writes on shutdown / in tests.
`record_call` queues a row and returns immediately; the response streams without waiting. One
writer task per process drains the queue in batches on one connection (a strong reference to it
is held until it finishes, otherwise the event loop may GC a bare create_task). Failures are
swallowed: an audit hiccup must never break a real call. `drain()` flushes pending writes on
shutdown / in tests.
Back-pressure (why this matters): each write opens a DB connection, from the BACKGROUND pool
(db.py), so a burst here can starve other background work but never real calls. The semaphore stays
as the inner, cheaper bound — it queues in-process instead of holding a pooled connection, and it is
what keeps `drain()` deterministic on sqlite, where all three makers share one engine. Under an
extreme burst we DROP audit rows past `_MAX_PENDING` rather than grow without bound — audit is
best-effort; never OOM or wedge the server for it.
Back-pressure (why this matters): the writer's connection comes from the BACKGROUND pool (db.py),
so a burst here can starve other background work but never real calls. Rows queue in-process, not
as pooled connections: one writer per process takes them off the queue `_BATCH` at a time and lands
each batch in one INSERT round trip, which is what keeps `drain()` deterministic on sqlite, where
all three makers share one engine. Under an extreme burst we DROP audit rows past `_MAX_PENDING`
rather than grow without bound — audit is best-effort; never OOM or wedge the server for it.
"""
from __future__ import annotations
import asyncio
import logging
from collections import deque
from .infra.db import background_session_maker
from .models import CallRecord, RunRecord, SearchMiss
_pending: set[asyncio.Task] = set()
_MAX_CONCURRENT_WRITES = 4 # cap on audit writes holding a DB connection at once (protect the request pool)
# ONE writer per process, and it writes in batches. Audit rows are single-row inserts that cost
# milliseconds each, so four concurrent writers bought nothing but four `background` slots - and
# every slot is paid twice (two uvicorn workers) and again at every deploy against the database's
# 103-connection ceiling (ops/deploy.md). One writer draining a queue in batches of `_BATCH` rows
# lands the same rows in fewer round trips and holds one connection.
_MAX_CONCURRENT_WRITES = 1
_MAX_PENDING = 5000 # shed load past this: drop the audit row rather than grow unbounded
_BATCH = 200 # rows per INSERT round trip; a failed batch retries row by row
_queue: deque[tuple[type, dict]] = deque()
_sem: asyncio.Semaphore | None = None
_sem_loop = None
@@ -48,7 +57,7 @@ def record_call(
stay NULL. It is still fire-and-forget: the money landed in the ledger synchronously, so losing a
row here costs analytics, not accounting. `refused_by` marks a call TREG refused before anything
went upstream (see models.CallRecord) — NULL whenever the provider actually answered."""
_schedule(_write(CallRecord,
_enqueue(CallRecord, dict(
org_id=org_id, user_email=user_email, tool_name=tool_name,
method=method, path=path, status_code=status_code, client=client, refused_by=refused_by,
**_known_fields(CallRecord, telemetry),
@@ -78,14 +87,14 @@ def record_search_miss(*, query: str, source: str) -> None:
"""A catalog search that matched nothing — logged so the misses can steer ingest (see
models.SearchMiss). Same contract as every write here: fire-and-forget, and a dropped row
under load costs a data point, never a search response."""
_schedule(_write(SearchMiss, query=query[:300], source=source))
_enqueue(SearchMiss, dict(query=query[:300], source=source))
def record_run(
*, org_id: int | None = None, user_email: str, bundle_name: str, argv: list, exit_code: int,
duration_ms: int, client: str = ""
) -> None:
_schedule(_write(RunRecord,
_enqueue(RunRecord, dict(
org_id=org_id, user_email=user_email, bundle_name=bundle_name,
argv=argv, exit_code=exit_code, duration_ms=duration_ms, client=client,
))
@@ -95,39 +104,74 @@ _shed = 0 # audit rows dropped by back-pressure this process; only ever grows
def _schedule(coro) -> None:
if len(_pending) >= _MAX_PENDING: # shed load — audit is best-effort, never OOM the server
coro.close()
# Say so. Shedding bypasses `_write` entirely, so the warning there never fires for it, and
# a shed row is invisible: the audit table simply has less in it. For a table whose job is
# to record what happened, "quiet" and "quietly broken" must not look identical — and the
# failure-evidence columns ride this same path, so a burst silently loses exactly the errors
# someone would go looking for. Logged on the first drop and then every 1,000th, so a long
# incident cannot itself flood the log.
global _shed
_shed += 1
if _shed == 1 or _shed % 1000 == 0:
logging.getLogger("treg.audit").error(
"audit back-pressure: %d row(s) dropped this process (pending at %d)",
_shed, _MAX_PENDING)
return
"""Run the writer as a tracked task. Shedding happens in `_enqueue`, on the queue: a shed row is
invisible - the audit table simply has less in it - so for a table whose job is to record what
happened, "quiet" and "quietly broken" must not look identical. The failure-evidence columns
ride this same path, so a burst would otherwise silently lose exactly the errors someone would
go looking for. Logged on the first drop and then every 1,000th."""
task = asyncio.create_task(coro)
_pending.add(task)
task.add_done_callback(_pending.discard)
task.add_done_callback(_writer_done)
async def _write(model, **fields) -> None:
async with _get_sem(): # cap concurrent DB connections held by audit — never starve the request pool
try:
async with background_session_maker() as session:
session.add(model(**fields))
await session.commit()
except Exception: # noqa: BLE001 — audit must never surface into a call's result
# Swallowed on purpose, but neither silent nor local: the row is lost — that is the
# contract — and ERROR is what puts that loss in front of someone. At WARNING it stayed
# in the container's stdout, below the fault handler's threshold, so the only way to
# learn that audit was dropping rows was to already suspect it and go grep.
def _writer_done(task: asyncio.Task) -> None:
"""A row enqueued between the writer's last empty-queue check and this callback saw `_pending`
still occupied and did not start a writer; start one for it here or it waits for the next call."""
_pending.discard(task)
if _queue and not _pending:
_schedule(_flush())
def _enqueue(model, fields: dict) -> None:
"""Queue one row and make sure a writer is running. The shed check is on the QUEUE, which is
where rows wait now; `_pending` holds at most the one writer task."""
if len(_queue) >= _MAX_PENDING:
_shed_one()
return
_queue.append((model, fields))
if not _pending:
_schedule(_flush())
def _shed_one() -> None:
global _shed
_shed += 1
if _shed == 1 or _shed % 1000 == 0:
logging.getLogger("treg.audit").error(
"audit back-pressure: %d row(s) dropped this process (pending at %d)",
_shed, _MAX_PENDING)
async def _flush() -> None:
"""Drain the queue in batches until it is empty, then exit. One connection for the whole run.
A batch that fails is retried row by row, so one row the database refuses (a value out of
range, a constraint) costs that row and not the 199 around it - the failure evidence of a
burst is exactly what such a burst must not take down with it.
"""
async with _get_sem():
while _queue:
batch = [_queue.popleft() for _ in range(min(_BATCH, len(_queue)))]
if not await _write_batch(batch):
for row in batch:
await _write_batch([row])
async def _write_batch(rows: list[tuple[type, dict]]) -> bool:
try:
async with background_session_maker() as session:
session.add_all([model(**fields) for model, fields in rows])
await session.commit()
return True
except Exception: # noqa: BLE001 — audit must never surface into a call's result
# Swallowed on purpose, but neither silent nor local: the row is lost — that is the
# contract — and ERROR is what puts that loss in front of someone. At WARNING it stayed
# in the container's stdout, below the fault handler's threshold, so the only way to
# learn that audit was dropping rows was to already suspect it and go grep.
if len(rows) == 1:
logging.getLogger("treg.audit").error(
"audit write dropped for %s", model.__name__, exc_info=True)
"audit write dropped for %s", rows[0][0].__name__, exc_info=True)
return False
async def drain() -> None:
@@ -139,7 +183,16 @@ async def drain() -> None:
# suspends, so a loop keyed only on the callback spins synchronously forever while that callback
# (and every timer on the loop) starves. Latent since the first import; a CI Postgres runner hit
# the window deterministically and wedged whole 15-minute jobs on it.
while _pending:
#
# Whatever is still queued once no writer is running is flushed HERE, inline, not through
# `_schedule`: a drain that depends on scheduling a task to make progress spins forever the
# moment scheduling is stubbed out (a test kills the pipeline exactly that way).
while True:
tasks = list(_pending)
await asyncio.gather(*tasks, return_exceptions=True)
_pending.difference_update(tasks)
if tasks:
await asyncio.gather(*tasks, return_exceptions=True)
_pending.difference_update(tasks)
continue
if not _queue:
return
await _flush()
+43
View File
@@ -3,6 +3,8 @@
from __future__ import annotations
import asyncio
import logging
import time
from collections.abc import Sequence
from contextlib import asynccontextmanager
from copy import copy
@@ -413,6 +415,44 @@ def _route_manifest(routes: Sequence[BaseRoute]) -> list[str]:
return result
_POOL_GAUGE_SAMPLE_S = 1.0
_POOL_GAUGE_EMIT_S = 60.0
async def pool_gauge(*, sample_s: float = _POOL_GAUGE_SAMPLE_S,
emit_s: float = _POOL_GAUGE_EMIT_S) -> None:
"""Every minute, one `db_pool_gauge` event: the peak connections each pool had checked out in
that minute, next to its capacity. The reading behind `TREG_DB_POOL_OVERRIDES`: a pool whose
peak sits at capacity is one whose waiters are timing out (api: `503 treg_saturated`;
background: an audit or archive row dropped after `pool_timeout`), and a pool whose peak never
nears it is holding connections nothing uses. Sampling is a counter read, no I/O; a bad pass
never kills the loop. Telemetry, not a database consumer - so it is not in
`ROLE_BACKGROUND_TASKS` and runs in every role."""
from .infra.db import fold_pool_peaks, pool_snapshot
peaks: dict[str, int] = {}
samples = 0
opened = time.monotonic()
while True:
try:
fold_pool_peaks(peaks, pool_snapshot())
samples += 1
if time.monotonic() - opened >= emit_s:
snapshot = pool_snapshot()
props: dict = {"samples": samples, "window_s": round(time.monotonic() - opened)}
for name, row in snapshot.items():
props[f"{name}_peak"] = peaks.get(name, 0)
props[f"{name}_capacity"] = row["capacity"]
props[f"{name}_headroom"] = row["capacity"] - peaks.get(name, 0)
if snapshot:
analytics.capture(analytics.SERVER_DISTINCT_ID, "db_pool_gauge", props)
peaks, samples, opened = {}, 0, time.monotonic()
except asyncio.CancelledError:
raise
except Exception: # noqa: BLE001 - a gauge must never take the service down
logging.getLogger("treg").exception("db pool gauge pass failed")
await asyncio.sleep(sample_s)
def _lifespan(role: AppRole):
@asynccontextmanager
async def lifespan(app: FastAPI):
@@ -428,6 +468,7 @@ def _lifespan(role: AppRole):
if ROLE_BACKGROUND_TASKS[role] and adsconv.enabled()
else None
)
gauge_task = asyncio.create_task(pool_gauge()) if analytics.enabled() else None
# The archive's refresh worker (docs/context/architecture/archive.md): serve mode only,
# and a zero daily cap disables it without touching serving. Same discipline as the ads
# task — in-process, cancelled on shutdown, a bad pass never kills the loop.
@@ -460,6 +501,8 @@ def _lifespan(role: AppRole):
yield
finally:
try:
if gauge_task is not None:
gauge_task.cancel()
if ads_task is not None:
ads_task.cancel()
if archive_task is not None:
+28
View File
@@ -31,6 +31,18 @@ def platform_setting_name(provider: str) -> str:
return "platform_key_" + (provider or "").lower().replace("-", "_")
@lru_cache
def _blocked_email_domains(raw: str) -> frozenset[str]:
"""Parse `TREG_BLOCKED_EMAIL_DOMAINS` once per distinct value, not per request: split on commas,
trim, drop a leading `@` or `.` (operators paste both spellings), lowercase, drop empties. A
dotless entry (`com`) is dropped too: the classifier walks parent domains, so a bare public
suffix would refuse every address on earth from one typo in a dashboard field."""
return frozenset(
d for d in (part.strip().lstrip("@.").rstrip(".").lower() for part in raw.split(","))
if "." in d
)
class Settings(BaseSettings):
model_config = SettingsConfigDict(env_file=".env", env_prefix="TREG_", extra="ignore")
@@ -425,6 +437,22 @@ class Settings(BaseSettings):
# must be explicitly enabled (TREG_EMAIL_DEV_MODE=true) for local testing without a mail sender.
email_dev_mode: bool = False
# The OPS tier of the email-domain blocklist (TREG_BLOCKED_EMAIL_DOMAINS), comma-separated:
# "newfarm.io,other-farm.net". ADDED to the code tier in `domain/identity/access.py` (treg's
# confirmed farm roots and the throwaway-mail keyword rules), never replacing it. A listed domain
# blocks itself AND every subdomain, case-insensitively, at every sign-up and sign-in door and at
# the two doors that mint a promo-funded team (POST /users, POST /orgs). It exists because a
# signup-grant farm moves to a new root in minutes and the answer has to be a dashboard edit, not
# a deploy. Empty (the default) adds nothing. Existing accounts on a listed domain are suspended
# out of band, so listing a domain strands nobody legitimate. A blocklist, deliberately: no
# allowlist, no table, no admin UI.
blocked_email_domains: str = ""
@property
def blocked_email_domain_set(self) -> frozenset[str]:
"""The normalised `TREG_BLOCKED_EMAIL_DOMAINS` entries; empty = the code tier alone."""
return _blocked_email_domains(self.blocked_email_domains)
# Frictionless local mode: `curl … | sh` brings up a server you are already signed into, with no
# account, email or password. Only takes effect when `single_user_ok` allows it (see below).
single_user: bool = False
+59 -8
View File
@@ -1,15 +1,18 @@
"""Per-member usage policy shared by call and run surfaces."""
import logging
from datetime import datetime
from sqlalchemy import func
from sqlalchemy import case, func, update
from sqlalchemy.ext.asyncio import AsyncSession
from sqlmodel import select
from ...models import CallRecord, RunRecord
from ...models import CallRecord, Membership, RunRecord
from ...timeutil import utcnow_naive
from ..identity.access import Caller
log = logging.getLogger("treg.usage")
class UsagePolicyError(Exception):
"""A member exhausted their configured daily usage allowance."""
@@ -25,8 +28,15 @@ def _day_start_utc() -> datetime:
async def count_today(db: AsyncSession, org_id: int | None, user_email: str) -> int:
"""How many usage events this user has produced in this org since midnight UTC: proxy calls +
local-run grants (both `CallRecord`) plus server runs (`RunRecord`). Two indexed COUNTs."""
"""How many usage events this user has produced in this org since midnight UTC, from the JOURNAL:
proxy calls + local-run grants (both `CallRecord`) plus server runs (`RunRecord`).
NOT the cap's number and NOT on the call path. This is what the roster, `/usage/me` and the
usage report show, and what `seed_counter` copies when a cap is first set; the gate itself
reads `Membership.calls_today` (`take_daily_slot`). Two range COUNTs over
`(org_id, user_email, created_at)` - O(this member's rows today), which for a member making
100k calls a day was 2.8 s per call when the gate still ran it (2026-09-06).
"""
since = _day_start_utc()
calls = (await db.execute(select(func.count()).select_from(CallRecord).where(
CallRecord.org_id == org_id, CallRecord.user_email == user_email, CallRecord.created_at >= since,
@@ -37,15 +47,56 @@ async def count_today(db: AsyncSession, org_id: int | None, user_email: str) ->
return calls + runs
async def take_daily_slot(db: AsyncSession, membership_id: int, cap: int) -> bool:
"""Admit one usage event against the member's cap, or say no. ONE conditional UPDATE.
The WHERE is the check and the SET is the count, so N concurrent calls cannot each read a
compliant figure and together overshoot - the same idiom as `money.reserve`. The first event
of a new UTC day resets the counter to 1 instead of adding to yesterday. A refused event is not
counted: the counter is "admitted today", so a capped member hammering the gate reads exactly
`cap`, not a runaway number. Does not commit; the caller's transaction owns it.
"""
today = utcnow_naive().date()
# "Used today" is the counter only if it belongs to today; a stale or never-set day is 0. The
# same expression gates and counts, so a cap of 0 refuses even the day's first event.
used_today = case((Membership.calls_today_day == today, Membership.calls_today), else_=0)
result = await db.execute(
update(Membership)
.where(Membership.id == membership_id, used_today < cap)
.values(calls_today=used_today + 1, calls_today_day=today)
)
return result.rowcount == 1
async def seed_counter(db: AsyncSession, membership: Membership, user_email: str) -> None:
"""Start the counter from today's journal - called when a cap is set on a member who was
unlimited until now. Only capped members are counted on the call path, so without this a
member who already made 50k calls today would get a fresh allowance the moment they were capped.
One journal count, at cap-setting time, never per call."""
membership.calls_today = await count_today(db, membership.org_id, user_email)
membership.calls_today_day = utcnow_naive().date()
async def enforce_daily_cap(caller: Caller, db: AsyncSession, *, sandbox: bool) -> None:
"""Refuse a call/run once the caller has used their per-user daily cap for this org. `-1` (the
default) = unlimited, so unmetered members pay ZERO extra queries. The sandbox has its own limiter
and is exempt. Soft by design: the count reads best-effort `CallRecord`s, so under heavy load it
can lag slightly and fail OPEN (a few extra slip through) — never closed. See docs/USAGE-METERING-PLAN.md."""
and is exempt.
One conditional UPDATE of the member's row (`take_daily_slot`), exact under concurrency. Fails
OPEN if that statement cannot run: a cap is a courtesy limit an admin set on a colleague, and
the database being unavailable is not the colleague's fault - the money gates behind this one
are the ones that fail closed. See docs/USAGE-METERING-PLAN.md.
"""
cap = caller.membership.daily_call_cap
if cap < 0 or sandbox:
return
used = await count_today(db, caller.org_id, caller.email)
if used >= cap:
try:
admitted = await take_daily_slot(db, caller.membership.id, cap)
except Exception as exc: # noqa: BLE001 - fail open, see docstring
log.warning("daily-cap check failed for membership %s: %s", caller.membership.id, exc)
return
if not admitted:
used = (await db.execute(
select(Membership.calls_today).where(Membership.id == caller.membership.id))).scalar() or 0
raise UsagePolicyError(
f"daily usage limit reached ({used}/{cap}) — ask an admin to raise your cap")
+55
View File
@@ -241,3 +241,58 @@ def _is_agent_email(email: str) -> bool:
def _is_machine_email(email: str) -> bool:
"""An identity minted by an admin for a machine — never a person who can sign in."""
return _is_agent_email(email) or _norm_email(email).endswith(f"@{PUBLIC_DEMO_DOMAIN}")
# ---- the email-domain blocklist: throwaway mail and abusive signup domains ----------------------
# A new team is created with a promotional balance (`application.signup._grant_signup_promo`), which
# makes bulk registration on throwaway addresses worth someone's while. Two tiers, one classifier.
# Tier 1 is CODE: domains confirmed abusive in our own data, plus substring rules that catch
# throwaway-mail providers no static list has seen yet. Tier 2 is OPS:
# `TREG_BLOCKED_EMAIL_DOMAINS`, unioned in, so the next domain is a dashboard edit made the minute it
# appears, not a deploy. Three rules, each of which exists because the obvious implementation is
# wrong:
# - match the DOMAIN only, never the whole address. Matching the address false-flags real users
# whose USERNAME happens to contain a keyword (`tempmail@gmail.com` is a real person).
# - walk parent domains, whole labels off the front only and never the bare last label, because
# registering `<random>.<blocked-root>` is otherwise a one-line bypass. The walk is safe because
# no entry is a bare public suffix, which `config._blocked_email_domains` enforces for the ops
# tier by dropping dotless entries.
# - a PURE classifier: refusing, logging and skipping a perk are the caller's decisions
# (`application.signup.blocked_email`).
BLOCKED_EMAIL_DOMAINS: frozenset[str] = frozenset({
# Confirmed abusive in our own data: bulk registration only, no legitimate account on any of them.
"uberip.com",
"westcast-systems.com",
"mailfox.win",
"yopmail.com",
# Free `.my.id` subdomains are handed out publicly. Listed as the parent so the walk catches
# `<anything>.my.id`.
"my.id",
})
# Substring rules on the domain: throwaway-mail providers name themselves.
BLOCKED_EMAIL_KEYWORDS: tuple[str, ...] = (
"tempmail", "temp-mail", "mailinator", "guerrilla", "throwaway", "10minute", "trashmail",
"yopmail", "sharklasers", "dispostable", "getnada", "maildrop", "moakt", "mohmal",
"emailondeck", "fakemail",
)
def _email_domain(email: str) -> str:
"""The lowercased domain part of an address, "" when there is none. The ONLY part of an address
the blocklist ever looks at."""
return _norm_email(email).rpartition("@")[2]
def _is_blocked_email(email: str) -> bool:
"""Pure classifier: is this address on a blocked domain, on a subdomain of one, or on a domain
that names itself a throwaway? An empty ops list leaves the code tier alone in force."""
domain = _email_domain(email)
if not domain:
return False
ops = get_settings().blocked_email_domain_set
labels = domain.split(".")
for i in range(len(labels) - 1): # every parent domain, never the bare last label
candidate = ".".join(labels[i:])
if candidate in BLOCKED_EMAIL_DOMAINS or candidate in ops:
return True
return any(keyword in domain for keyword in BLOCKED_EMAIL_KEYWORDS)
+67 -13
View File
@@ -46,7 +46,7 @@ import uuid
from datetime import datetime, timedelta, timezone
from typing import NamedTuple
from sqlalchemy import delete, func, update
from sqlalchemy import case, delete, func, update
from sqlalchemy.exc import IntegrityError
from sqlalchemy.ext.asyncio import AsyncSession
from sqlmodel import select
@@ -125,11 +125,38 @@ async def _entry(
return row
async def _add_balance(db: AsyncSession, org_id: int, delta_micro: int) -> None:
"""Unconditional balance move (grants, refunds). The CONDITIONAL one lives in `reserve`."""
await db.execute(
update(Org).where(Org.id == org_id).values(balance_micro=Org.balance_micro + delta_micro)
)
def _day_start() -> datetime:
return _now().replace(hour=0, minute=0, second=0, microsecond=0)
def _spent_today_values(delta_micro: int) -> dict:
"""SET clause that folds `delta_micro` into the org's daily-spend counter.
The counter is "committed since midnight UTC": settled today plus still held from today. One
CASE keeps it honest across the day boundary - the first movement of a new UTC day resets it
to that movement instead of adding to yesterday's total. Always applied inside the statement
that already holds the org row (the balance UPDATE), so it costs no extra lock and cannot
disagree with the balance about which transaction it belongs to.
"""
today = _now().date()
return {
"spent_today_micro": case(
(Org.spent_today_day == today, Org.spent_today_micro + delta_micro), else_=delta_micro),
"spent_today_day": today,
}
async def _add_balance(db: AsyncSession, org_id: int, delta_micro: int, *,
spent_delta_micro: int = 0) -> None:
"""Unconditional balance move (grants, refunds). The CONDITIONAL one lives in `reserve`.
`spent_delta_micro` rides in the same UPDATE: settle and release move the balance AND change
what counts as committed today, and the two must land in one statement.
"""
values: dict = {"balance_micro": Org.balance_micro + delta_micro}
if spent_delta_micro:
values.update(_spent_today_values(spent_delta_micro))
await db.execute(update(Org).where(Org.id == org_id).values(**values))
# ---- funding -----------------------------------------------------------------------------------
@@ -265,7 +292,7 @@ async def reserve_in_transaction(
# callers cannot both pass it. rowcount 0 = the balance was not there.
update(Org)
.where(Org.id == org_id, Org.balance_micro >= charged)
.values(balance_micro=Org.balance_micro - charged)
.values(balance_micro=Org.balance_micro - charged, **_spent_today_values(charged))
)
if result.rowcount != 1:
balance = (await db.execute(select(Org.balance_micro).where(Org.id == org_id))).scalar() or 0
@@ -295,6 +322,7 @@ class _ClaimedHold(NamedTuple):
amount_micro: int
org_id: int
endpoint_id: str
created_at: datetime
async def _claim_hold(db: AsyncSession, call_id: str) -> _ClaimedHold | None:
@@ -312,7 +340,7 @@ async def _claim_hold(db: AsyncSession, call_id: str) -> _ClaimedHold | None:
hold = await db.get(Hold, call_id)
if hold is None:
return None
claimed = _ClaimedHold(hold.amount_micro, hold.org_id, hold.endpoint_id)
claimed = _ClaimedHold(hold.amount_micro, hold.org_id, hold.endpoint_id, hold.created_at)
result = await db.execute(delete(Hold).where(Hold.id == call_id))
if result.rowcount != 1:
# Lost the claim: somebody else is closing this hold. Deliberately NO rollback — the DELETE
@@ -365,8 +393,11 @@ async def _settle_in_transaction(
# The hold came out of the balance at reserve time; give back whatever the call didn't use. If the
# observed cost overran the estimate the delta is negative, which correctly takes MORE balance —
# the next reserve is the gate that stops an overrun from compounding.
if reserved != consumed:
await _add_balance(db, hold.org_id, reserved - consumed)
# The daily counter: what settled today goes in; the hold it replaces comes out, but only if
# that hold was counted today (a hold opened yesterday was yesterday's commitment).
spent_delta = consumed - (reserved if hold.created_at >= _day_start() else 0)
if reserved != consumed or spent_delta:
await _add_balance(db, hold.org_id, reserved - consumed, spent_delta_micro=spent_delta)
await _entry(
db, org_id=hold.org_id, kind="settle", amount_micro=-consumed, call_id=call_id,
endpoint_id=hold.endpoint_id, created_at=settled_at,
@@ -416,7 +447,9 @@ async def _release_in_transaction(
if hold is None:
return 0, False
amount = hold.amount_micro
await _add_balance(db, hold.org_id, amount)
# Same rule as settle: a hold counted today leaves today's counter; an older one never was in it.
spent_delta = -amount if hold.created_at >= _day_start() else 0
await _add_balance(db, hold.org_id, amount, spent_delta_micro=spent_delta)
await _entry(db, org_id=hold.org_id, kind="release", amount_micro=amount, call_id=call_id,
endpoint_id=hold.endpoint_id, meta={**(meta or {}), "reason": reason})
# Nothing was billable, so nothing is attributable: the tag rows go with the hold. Leaving them
@@ -485,12 +518,33 @@ async def reap_stale_holds(db: AsyncSession, *, org_id: int | None = None, limit
# ---- reads -------------------------------------------------------------------------------------
async def spent_today(db: AsyncSession, org_id: int) -> int:
"""Micro-USD this org has committed since midnight UTC: everything SETTLED today plus everything
still HELD from today. Two indexed aggregates, and the number a daily spend cap is checked against.
still HELD from today - the number a daily spend cap is checked against.
Runs on EVERY metered call, inside the reserve transaction, on an api-pool connection, so its
cost is the platform's throughput: ONE primary-key read of the org row. The counter is kept by
reserve/settle/release inside the balance UPDATE (`_spent_today_values`); it was an aggregate
over `ledgerentry` until 2026-09-06, when that aggregate - O(rows the platform wrote today) for
a heavy org, whatever the index - was what emptied the API pool. `spent_today_from_ledger` is
the same number from the journal, for reconciliation.
Deliberately not "sum of reserve entries": a reserve is refunded at settle, so counting both would
double-charge every call. Settled + still-open is exactly the money that is gone or promised.
"""
since = _now().replace(hour=0, minute=0, second=0, microsecond=0)
row = (await db.execute(
select(Org.spent_today_micro, Org.spent_today_day).where(Org.id == org_id))).one_or_none()
if row is None or row[1] != _now().date():
return 0 # nothing has moved for this org today; the first movement will stamp the day
return int(row[0])
async def spent_today_from_ledger(db: AsyncSession, org_id: int) -> int:
"""The same number computed from the journal - two range aggregates over `(org_id, created_at)`.
NOT on the call path. This is the reconciliation view of the counter: what `spent_today` must
agree with, and what a test asserts it against. `reconcile.py` and an operator who distrusts
the counter read this; a bigger cap check does not.
"""
since = _day_start()
settled = (await db.execute(
select(func.coalesce(func.sum(LedgerEntry.amount_micro), 0)).where(
LedgerEntry.org_id == org_id, LedgerEntry.kind == "settle", LedgerEntry.created_at >= since)
+67 -4
View File
@@ -3,6 +3,7 @@
from __future__ import annotations
import logging
import os
from collections.abc import AsyncIterator
from functools import cache
from importlib import import_module
@@ -38,8 +39,9 @@ _is_sqlite = "sqlite" in _db_url
# whole pool when this was 2 — see `_purge_expired_error_evidence`, now on `background` and
# single-flighted so concurrent readers cannot multiply it.
#
# Sizes are PER INSTANCE and a rolling deploy runs two, so the SUM is what must stay under the
# database plan's ~100 ceiling — see ops/deploy.md, and the guard test.
# Sizes are PER PROCESS, the web service runs two uvicorn workers, and a rolling deploy runs two
# instances, so `per_process × 2 × 2` is what must stay under the database plan's 103 ceiling —
# see `connection_budget` below and ops/deploy.md.
#
# These numbers can only be validated in production: too small and real traffic gets 503s, too large
# and the bulkhead is decorative, and no test can tell you which. `TREG_DB_POOL_OVERRIDES` makes a
@@ -48,8 +50,8 @@ _is_sqlite = "sqlite" in _db_url
# Everything that can hold a `background` slot at the same moment. Keep this in step with reality:
# it is what sizes the pool, and a consumer missing from it is a row silently dropped under load.
BACKGROUND_CONSUMERS: dict[str, int] = {
"audit._write": 4, # bounded by audit._MAX_CONCURRENT_WRITES
"archive._store/_touch": 4, # bounded by archive._MAX_CONCURRENT_WRITES (one shared semaphore)
"audit._flush": 1, # one batching writer per process (audit._MAX_CONCURRENT_WRITES)
"archive._store/_touch": 2, # bounded by archive._MAX_CONCURRENT_WRITES (one shared semaphore)
"adsconv.worker": 1, # holds its slot across two Google round trips — see follow-ups
"archive.prune_worker": 1, # holds one across a whole sweep
"archive.refresh_worker": 1,
@@ -100,6 +102,36 @@ if _overrides := get_settings().db_pool_overrides:
POOL_SPECS = _apply_overrides(POOL_SPECS, _overrides)
def connection_budget(workers: int | None = None) -> dict[str, int]:
"""How many connections these specs can open, at the three scopes that matter.
Every number in `POOL_SPECS` is PER PROCESS, and the reference deployment runs TWO: Render sets
`WEB_CONCURRENCY=2` on the 2c-4g plan and uvicorn honors it (`Started server process` twice in
the boot log). A rolling deploy then runs two instances for a minute. So the ceiling the specs
must clear is `per_process × workers × 2` against Postgres's `max_connections` (103 on the
1c-2g plan) - the arithmetic that, taken per instance, let the 2026-09-04 defaults (31) open
124 connections at every deploy until the dashboard override cut them to 21 (84).
"""
per_process = sum(spec["pool_size"] + spec["max_overflow"] for spec in POOL_SPECS.values())
if workers is None:
try:
workers = max(1, int(os.environ.get("WEB_CONCURRENCY", "1")))
except ValueError:
workers = 1
return {"per_process": per_process, "workers": workers,
"per_instance": per_process * workers, "deploy_peak": per_process * workers * 2}
_budget = connection_budget()
logging.getLogger("treg").info(
"db pools per process: api %d+%d, admin %d+%d, background %d+%d = %d; x%d workers = %d per "
"instance, %d at a rolling deploy (Postgres max_connections on the reference plan: 103)",
POOL_SPECS["api"]["pool_size"], POOL_SPECS["api"]["max_overflow"],
POOL_SPECS["admin"]["pool_size"], POOL_SPECS["admin"]["max_overflow"],
POOL_SPECS["background"]["pool_size"], POOL_SPECS["background"]["max_overflow"],
_budget["per_process"], _budget["workers"], _budget["per_instance"], _budget["deploy_peak"])
def _new_engine(name: str):
"""One pooled engine per `POOL_SPECS` entry.
@@ -128,6 +160,37 @@ if _is_sqlite:
_admin_engine = _background_engine = _engine
_engines = (_engine, _admin_engine, _background_engine)
_POOL_NAMES = ("api", "admin", "background")
def pool_snapshot() -> dict[str, dict[str, int]]:
"""What each pool holds RIGHT NOW: connections checked out, of how many it may hand out.
The sizing question `POOL_SPECS` answers by arithmetic ("13 is the sum of the semaphores") can
only be settled by measurement; this is the measurement. `checked_out` counts slots in use
(persistent and overflow alike), `capacity` is `pool_size + max_overflow`. SQLite has no pool
worth reading and reports nothing. A pure read of SQLAlchemy's counters - no lock, no I/O.
"""
if _is_sqlite:
return {}
out: dict[str, dict[str, int]] = {}
for name, engine in zip(_POOL_NAMES, _engines):
pool = engine.sync_engine.pool
checked_out = getattr(pool, "checkedout", None)
if checked_out is None:
continue
spec = POOL_SPECS[name]
out[name] = {"checked_out": int(checked_out()),
"capacity": spec["pool_size"] + spec["max_overflow"]}
return out
def fold_pool_peaks(peaks: dict[str, int], snapshot: dict[str, dict[str, int]]) -> dict[str, int]:
"""Keep the per-pool maximum of `checked_out` seen across samples (the gauge's whole job)."""
for name, row in snapshot.items():
if row["checked_out"] > peaks.get(name, 0):
peaks[name] = row["checked_out"]
return peaks
# The API pool: every request handler, through `get_session` or directly.
session_maker = async_sessionmaker(_engine, class_=AsyncSession, expire_on_commit=False)
+34 -2
View File
@@ -8,7 +8,7 @@ so every list/call/mutation is scoped to the caller's org. See docs/MULTI-TENANC
from __future__ import annotations
from datetime import datetime, timezone
from datetime import date, datetime, timezone
from sqlalchemy import BigInteger, JSON, Column, Index, Integer, UniqueConstraint, text
from sqlmodel import Field, SQLModel
@@ -44,6 +44,14 @@ class Org(SQLModel, table=True):
# conditional UPDATE against this integer is what stops concurrent agent calls racing past zero
# (see ledger.reserve). Only `domain/money` may write it.
balance_micro: int = Field(default=0)
# "Committed since midnight UTC": everything settled today plus everything still held from
# today - the number the fail-closed daily cap is checked against on EVERY metered call. Kept
# here, in the same UPDATE that moves the balance, because the equivalent aggregate over
# `ledgerentry` cost a scan of the platform's whole day per call and emptied the API pool
# (revision 0022). Written ONLY by domain/money; `spent_today_day` says which UTC day the
# counter belongs to, and a movement on a later day resets it.
spent_today_micro: int = Field(default=0, sa_column=Column("spent_today_micro", BigInteger, nullable=False, server_default="0"))
spent_today_day: date | None = Field(default=None)
# ---- Stripe billing (see billing.py; NO card data ever lands here) ----------------------------
# The org's Stripe Customer. Created lazily on the first top-up and reused forever after, because
@@ -166,8 +174,15 @@ class Membership(SQLModel, table=True):
promoted_from: str = Field(default="")
webhook_url: str | None = Field(default=None) # health alerts for this member's org POST here
# Per-user, per-day usage cap for this org (counts proxy calls + local + server runs). -1 = unlimited
# (the default — nobody is capped until an admin sets a limit). See api._enforce_daily_cap.
# (the default — nobody is capped until an admin sets a limit). See governance/usage.enforce_daily_cap.
daily_call_cap: int = Field(default=-1)
# The number that cap is checked against: usage events the gate has admitted since midnight
# UTC, and which UTC day it belongs to. Taken with ONE conditional UPDATE per capped call, so
# the check costs no count over `callrecord` (which had read a member's whole history per
# call - revision 0024) and the cap is exact rather than "a few extra slip through". Only
# capped members are counted; the roster's `used_today` still reads the journal.
calls_today: int = Field(default=0, sa_column=Column("calls_today", Integer, nullable=False, server_default="0"))
calls_today_day: date | None = Field(default=None)
# Per-member tool ACL: NULL = ALL tools in the org (the default — no restriction, no regression); a
# JSON list of tool NAMES = the ONLY tools this member may call or run. See api._require_tool_access.
tool_access: list | None = Field(default=None, sa_column=Column("tool_access", JSON, nullable=True))
@@ -248,6 +263,12 @@ class CallRecord(SQLModel, table=True):
# every 3 ms request query queue behind it until `pool_timeout` fires and
# callers get `503 treg_saturated`. Sizing the pool cannot fix a scan.
Index("ix_callrecord_endpoint_id_created_at", "endpoint_id", "created_at"),
# The per-user daily call cap (`governance/usage.count_today`, on every
# capped call): "this org, this member, since midnight". Without the triple
# the planner BitmapAnd-ed the member's WHOLE history through
# `ix_callrecord_user_email` - measured 2.6 s of 3.0 s on prod 2026-09-06 for
# a member with 287k rows. Revision 0023 builds it concurrently.
Index("ix_callrecord_org_id_user_email_created_at", "org_id", "user_email", "created_at"),
Index("ix_callrecord_org_id_created_at", "org_id", "created_at"),)
id: int | None = Field(default=None, primary_key=True)
@@ -670,6 +691,17 @@ class LedgerEntry(SQLModel, table=True):
reserve/settle is negative. `call_id` correlates the reserve→settle / reserve→release pair.
"""
# `(org_id, created_at)` is what EVERY metered call pays for: `ledger.spent_today` (the
# fail-closed daily cap, inside the reserve transaction on an api-pool connection) asks
# "this org, since midnight". With only single-column indexes the planner walked the whole
# platform's day and filtered the org in memory - measured on prod 2026-09-06 at 4.38M rows:
# 322k rows discarded and 381k buffer touches per call, 56-106 s once the day's pages were
# cold, each one holding an api-pool slot. That was the `503 treg_saturated` mechanism, and
# this table (not `callrecord`) was the largest IO consumer in the database. The pair also
# serves `entries_of` (`/billing`), which had walked the whole `created_at` index backward.
# Revision 0021 builds it concurrently.
__table_args__ = (Index("ix_ledgerentry_org_id_created_at", "org_id", "created_at"),)
id: str = Field(primary_key=True) # uuid4 hex
org_id: int = Field(foreign_key="org.id", index=True)
block_id: str | None = Field(default=None, index=True)
+16 -1
View File
@@ -22,6 +22,7 @@ import time
from datetime import datetime, timedelta, timezone
from sqlalchemy import delete
from sqlalchemy.dialects import postgresql, sqlite
from sqlalchemy.ext.asyncio import AsyncSession
from .models import Ephemeral
@@ -57,8 +58,12 @@ async def kv_put(db: AsyncSession, ns: str, k: str, v: dict, ttl_s: float | None
the code's lifetime)."""
row = await db.get(Ephemeral, (ns, k))
if row is None:
# An INSERT that another process may be racing: two uvicorn workers striking the same
# provider write the same `capacity:lock` key within the same millisecond, and the loser
# of a plain INSERT died on `ephemeral_pkey` (prod, 2026-09-06). ON CONFLICT makes the
# last writer win, which is what a key/value put means.
exp = _utcnow_naive() + timedelta(seconds=ttl_s or 0)
db.add(Ephemeral(ns=ns, k=k, v=v, expires_at=exp))
await db.execute(_upsert(db.get_bind().dialect.name, ns=ns, k=k, v=v, expires_at=exp))
return
row.v = v # reassign (not in-place) so SQLAlchemy marks the JSON column dirty
if ttl_s is not None:
@@ -66,6 +71,16 @@ async def kv_put(db: AsyncSession, ns: str, k: str, v: dict, ttl_s: float | None
db.add(row)
def _upsert(dialect: str, **values):
"""`INSERT ... ON CONFLICT (ns, k) DO UPDATE` for the dialect at hand - both backends spell it
the same way, but SQLAlchemy exposes it per dialect."""
insert = postgresql.insert if dialect == "postgresql" else sqlite.insert
stmt = insert(Ephemeral).values(**values)
return stmt.on_conflict_do_update(
index_elements=[Ephemeral.ns, Ephemeral.k],
set_={"v": stmt.excluded.v, "expires_at": stmt.excluded.expires_at})
async def kv_pop(db: AsyncSession, ns: str, k: str) -> dict | None:
"""Read-and-delete (ns, k) atomically-ish within this session. Returns the value if it was still
live (not expired), else None. The row is always removed."""
+8 -1
View File
@@ -57,6 +57,9 @@ class EmailVerifyIn(BaseModel):
_EMAIL_HTTP_ERRORS = {
"demo_address": (400, "that's a demo address — pick a real email"),
"machine_identity": (403, "this address cannot be used to sign in"),
# Same words as machine_identity on purpose: the caller learns neither that a list exists nor
# what is on it.
"blocked_domain": (403, "this address cannot be used to sign in"),
"rate_limited": (429, "too many code requests — please wait a few minutes"),
"invalid_code": (401, "invalid code"),
"suspended": (403, "account suspended"),
@@ -108,7 +111,7 @@ async def _find_or_create_user(db: AsyncSession, email: str) -> User:
Caller commits."""
try:
return await signup.find_or_create_user(db, email)
except signup.MachineIdentityError as exc:
except (signup.MachineIdentityError, signup.BlockedEmailError) as exc:
raise HTTPException(status_code=403, detail="this address cannot be used to sign in") from exc
@@ -191,6 +194,8 @@ _SOCIAL_PAGE_ERRORS = {
"google_unverified_email": ("Login failed", "Your Google email isn't verified.", False, 400),
"callback_failed": ("Login failed", "Something went wrong. Please try again.", False, 502),
"suspended": ("Account suspended", "This account has been suspended.", False, 403),
# A page, like `suspended`: a human is in the browser. Names no list and no domain.
"blocked_domain": ("Sign-in refused", "This address cannot be used to sign in.", False, 403),
}
@@ -674,6 +679,8 @@ async def auth_invite_signin_confirm(request: Request):
raise HTTPException(status_code=403, detail="this address cannot be used to sign in") from exc
if exc.kind == "suspended":
return _auth_page("Account suspended", "This account has been suspended.", ok=False, status=403)
if exc.kind == "blocked_domain":
return _auth_page("Sign-in refused", "This address cannot be used to sign in.", ok=False, status=403)
raise
resp = RedirectResponse(proof.destination, status_code=303)
resp.set_cookie(sess.COOKIE, proof.session_cookie, httponly=True,
+11 -12
View File
@@ -41,11 +41,11 @@ from ..config import get_settings
from ..domain.catalog import store as catalog_store
from ..domain.governance import access as access_policy
from ..domain.governance import publicdemo as publicdemo_policy
from ..domain.governance import usage as usage_policy
from ..domain.identity.access import Caller, require_member
from ..infra.db import get_session
from ..models import Tool
from .auth import _client_ip
from .orgs import count_today
# The app alias preserves the moved handlers' decorator text byte-for-byte.
@@ -122,17 +122,16 @@ async def _enforce_public_demo_ip_cap(request: Request, db: AsyncSession) -> Non
async def _enforce_daily_cap(caller: Caller, db: AsyncSession) -> None:
"""Refuse a call/run once the caller has used their per-user daily cap for this org. `-1` (the
default) = unlimited, so unmetered members pay ZERO extra queries. The sandbox has its own limiter
and is exempt. Soft by design: the count reads best-effort `CallRecord`s, so under heavy load it
can lag slightly and fail OPEN (a few extra slip through) — never closed. See docs/USAGE-METERING-PLAN.md."""
cap = caller.membership.daily_call_cap
if cap < 0 or demo_sandbox.is_sandbox(caller.org):
return
used = await count_today(db, caller.org_id, caller.email)
if used >= cap:
raise HTTPException(status_code=429, detail=(
f"daily usage limit reached ({used}/{cap}) — ask an admin to raise your cap"))
"""The run surfaces' door to the per-user daily cap - the SAME gate `/call/` goes through
(`application/call/authorize.py`), so a member cannot dodge the cap by switching path. Commits,
because the gate takes the slot with a write and this request's session is the only owner."""
try:
await usage_policy.enforce_daily_cap(
caller, db, sandbox=demo_sandbox.is_sandbox(caller.org))
except usage_policy.UsagePolicyError as exc:
await db.commit()
raise HTTPException(status_code=429, detail=exc.detail) from exc
await db.commit()
def _translate_call_failure(exc: CallFailure) -> HTTPException:
+8
View File
@@ -346,6 +346,7 @@ def _deny_view(r: DenyRule) -> dict:
_SIGNUP_HTTP_ERRORS = {
"machine_identity": (403, "this address cannot be used to sign in"),
"blocked_domain": (403, "this address cannot be used to sign in"), # same words: leaks no list
"unsafe_webhook": (422, "webhook_url must be a public http(s) URL"),
"email_exists": (409, "email already registered"),
"sandbox_user": (403, (
@@ -499,6 +500,8 @@ async def accept_invite(body: AcceptIn, db: AsyncSession = Depends(get_session))
org = await db.get(Org, invite.org_id)
if org is not None and org.suspended: # don't let anyone join a platform-locked org
raise HTTPException(status_code=403, detail="org suspended")
if signup_use_cases.blocked_email(email, "invite_code"): # creates a User directly: guards itself
raise HTTPException(status_code=403, detail="this address cannot be used to sign in")
user = (await db.execute(select(User).where(User.email == email))).scalar_one_or_none()
if user is not None and user.suspended: # a banned user must not accrue new memberships
raise HTTPException(status_code=403, detail="account suspended")
@@ -709,6 +712,11 @@ async def set_member_cap(
)).scalar_one_or_none()
if membership is None:
raise HTTPException(status_code=404, detail="not a member of this org")
if body.daily_call_cap >= 0 and membership.daily_call_cap < 0:
# Unlimited members are not counted on the call path; give the counter today's journal so a
# cap set mid-day starts from what they already used, not from zero.
user = await db.get(User, user_id)
await usage_policy.seed_counter(db, membership, user.email if user else "")
membership.daily_call_cap = body.daily_call_cap
await db.commit()
return {"user_id": user_id, "org_id": org_id, "daily_call_cap": body.daily_call_cap}
+4 -2
View File
@@ -1972,7 +1972,9 @@ details.tl li.more a{color:var(--link);text-decoration:none}
@app.get("/tools/{service}", include_in_schema=False)
async def tools_provider(service: str, db: AsyncSession = Depends(get_session)):
async def tools_provider(service: str, db: AsyncSession = Depends(get_session),
observations: endpoint_stats.EndpointObservationReader = Depends(
_endpoint_observation_reader)):
"""One provider's public page, in the use-case pages' skin (usecase.css): hero on the two
measured terms — "{provider} api pricing" (what Search Console shows people typing) and
"{provider} mcp" — the agent->treg->provider flow, setup (agent one-liner first), a prompt
@@ -2032,7 +2034,7 @@ async def tools_provider(service: str, db: AsyncSession = Depends(get_session)):
badge = "YOUR ACCOUNT" if is_oauth else "NO SIGNUP"
# The measured line: what treg.to has actually observed calling this provider. It is the one
# thing a vendor's own pricing page cannot print, and it goes above the fold for that reason.
obs = await _observed_or_empty(db, [e["id"] for e in eps])
obs = await _observed_or_empty(observations, [e["id"] for e in eps])
o_samples = sum(int(o.get("samples") or 0) for o in obs.values())
# The provider-wide rate weights each endpoint's published rate by the calls that DECIDED it
# (2xx + 5xx). `samples` still counts callers' 4xx, so weighting by it would let one team's
+90
View File
@@ -1079,3 +1079,93 @@ async def test_same_key_recordings_allocate_distinct_versions(clients: AsyncClie
keys, snaps = await _rows()
assert len(keys) == 1
assert [snap.version for snap in snaps] == list(range(1, 13))
# ---- the 2026-09-07 OOM regression test: memory-bounded pending work ---------------------------
# _MAX_PENDING_BYTES caps total body bytes held by pending tasks. Without it, 512 pending tasks ×
# 8 MB bodies = 4 GB worst case — the exact OOM that killed production at 2026-09-07T00:43:06Z.
async def test_pending_body_bytes_are_bounded_and_excess_is_shed(monkeypatch):
"""The bytes bound sheds recordings before the count bound would — the 2026-09-07 OOM fix.
With _MAX_CONCURRENT_WRITES=2 and heavy traffic, pending tasks holding large bodies can
accumulate faster than they drain. The bytes cap ensures total memory held by pending work
never exceeds a threshold, regardless of how many tasks fit under _MAX_PENDING.
"""
import asyncio as aio
release = aio.Event()
async def blocked_store(**kw):
await release.wait()
monkeypatch.setattr(archive, "_store_locked", blocked_store)
monkeypatch.setattr(archive, "_sem", None)
monkeypatch.setattr(archive, "_key_locks", None)
monkeypatch.setattr(archive, "_pending_bytes", 0)
archive._pending.clear()
original_max_bytes = archive._MAX_PENDING_BYTES
monkeypatch.setattr(archive, "_MAX_PENDING_BYTES", 1000)
common = dict(method="GET", endpoint_id=EP, provider="tikhub", caller_body=b"",
headers={}, status_code=200, media_type="application/json")
try:
archive.record(url="https://api.example/1", body=b"x" * 400, **common)
assert len(archive._pending) == 1
assert archive._pending_bytes == 400
archive.record(url="https://api.example/2", body=b"y" * 400, **common)
assert len(archive._pending) == 2
assert archive._pending_bytes == 800
archive.record(url="https://api.example/3", body=b"z" * 300, **common)
assert len(archive._pending) == 2, "third recording should be shed (800 + 300 > 1000)"
assert archive._pending_bytes == 800
archive.record(url="https://api.example/4", body=b"w" * 150, **common)
assert len(archive._pending) == 3, "fourth recording should fit (800 + 150 <= 1000)"
assert archive._pending_bytes == 950
finally:
release.set()
monkeypatch.setattr(archive, "_MAX_PENDING_BYTES", original_max_bytes)
await aio.gather(*archive._pending, return_exceptions=True)
archive._pending.clear()
monkeypatch.setattr(archive, "_pending_bytes", 0)
async def test_pending_bytes_released_when_task_completes(monkeypatch):
"""The done callback releases body bytes so they can be reused by new recordings."""
import asyncio as aio
release = aio.Event()
entered = aio.Event()
async def blocking_store(**kw):
entered.set()
await release.wait()
monkeypatch.setattr(archive, "_store_locked", blocking_store)
monkeypatch.setattr(archive, "_sem", None)
monkeypatch.setattr(archive, "_key_locks", None)
monkeypatch.setattr(archive, "_pending_bytes", 0)
archive._pending.clear()
common = dict(method="GET", endpoint_id=EP, provider="tikhub", caller_body=b"",
headers={}, status_code=200, media_type="application/json")
try:
archive.record(url="https://api.example/1", body=b"x" * 500, **common)
await aio.wait_for(entered.wait(), timeout=1)
assert archive._pending_bytes == 500
release.set()
await aio.gather(*archive._pending, return_exceptions=True)
await aio.sleep(0)
assert archive._pending_bytes == 0, "bytes should be released when task completes"
assert len(archive._pending) == 0
finally:
release.set()
await aio.gather(*archive._pending, return_exceptions=True)
archive._pending.clear()
monkeypatch.setattr(archive, "_pending_bytes", 0)
+8 -3
View File
@@ -167,12 +167,17 @@ async def test_queued_worker_rows_are_not_claimed_before_a_poll_slot(
await tick
@pytest.mark.parametrize("deadline", ["POLL_TIMEOUT_S", "PROCESS_TIMEOUT_S"])
# POLL_TIMEOUT_S wraps only the upstream poll, so it can be near-zero. PROCESS_TIMEOUT_S also
# wraps the claim's own DB round trip: at 10 ms a loaded CI runner fires it BEFORE the claim,
# which is correctly reported as backed_off with the row untouched - and then this test, which
# wants the post-claim path, fails on `consecutive_failures == 1`. Give the claim room; the hung
# poll is what the deadline must cut, and it never returns regardless.
@pytest.mark.parametrize("deadline, seconds", [("POLL_TIMEOUT_S", 0.01), ("PROCESS_TIMEOUT_S", 0.5)])
async def test_worker_bounds_whole_poll_and_keeps_hold_on_timeout(
clients: AsyncClient, monkeypatch, replicate_platform, deadline,
clients: AsyncClient, monkeypatch, replicate_platform, deadline, seconds,
):
call_id = await _due_submission(clients, monkeypatch, {})
monkeypatch.setattr(task_app, deadline, 0.01)
monkeypatch.setattr(task_app, deadline, seconds)
async def hangs(row, client):
await asyncio.Event().wait()
monkeypatch.setattr(task_app, "_poll", hangs)
+103
View File
@@ -58,3 +58,106 @@ def test_drain_livelock_regression_fails_instead_of_wedging(name):
result = subprocess.run(
[sys.executable, "-c", program], timeout=30, capture_output=True, text=True)
assert result.returncode == 0, result.stderr[-500:]
# ---- the batching writer (2026-09-07) ----------------------------------------------------------
# One writer per process, `_BATCH` rows per INSERT. What the pool budget bought must not cost rows:
# every row enqueued lands, a row the database refuses costs only itself, and shedding is the one
# loss - on the queue, where rows wait now.
from treg.infra.db import reset_db, session_maker # noqa: E402
from treg.models import SearchMiss # noqa: E402
async def _misses() -> list[str]:
from sqlalchemy import select
async with session_maker() as db:
return sorted((await db.execute(select(SearchMiss.query))).scalars().all())
@pytest.fixture
async def clean_audit():
await reset_db()
await audit.drain()
audit._queue.clear()
yield
await audit.drain()
async def test_a_burst_lands_in_batches_on_one_writer(clean_audit, monkeypatch):
"""More rows than one batch, enqueued at once: every one lands, and never more than one writer
task existed to do it (the whole point - one `background` slot per process, not four)."""
monkeypatch.setattr(audit, "_BATCH", 7)
peak = 0
real = audit._schedule
def counting(coro):
nonlocal peak
real(coro)
peak = max(peak, len(audit._pending))
monkeypatch.setattr(audit, "_schedule", counting)
for i in range(50):
audit.record_search_miss(query=f"q{i:02d}", source="test")
await audit.drain()
assert await _misses() == [f"q{i:02d}" for i in range(50)]
assert peak == 1
async def test_a_refused_row_costs_only_itself(clean_audit, monkeypatch):
"""A batch with one row the database rejects (NULL into a NOT NULL column) is retried row by
row, so the 199 rows around it still land - the failure evidence of a burst must survive the
burst."""
monkeypatch.setattr(audit, "_BATCH", 10)
for i in range(5):
audit.record_search_miss(query=f"ok{i}", source="test")
audit._enqueue(SearchMiss, dict(query=None, source="test")) # the bad row, mid-batch
for i in range(5, 9):
audit.record_search_miss(query=f"ok{i}", source="test")
await audit.drain()
assert await _misses() == [f"ok{i}" for i in range(9)]
async def test_rows_past_the_queue_bound_are_shed_not_queued(clean_audit, monkeypatch):
monkeypatch.setattr(audit, "_MAX_PENDING", 3)
monkeypatch.setattr(audit, "_schedule", lambda coro: coro.close()) # nothing drains meanwhile
before = audit._shed
for i in range(5):
audit.record_search_miss(query=f"s{i}", source="test")
assert len(audit._queue) == 3
assert audit._shed - before == 2
audit._queue.clear()
async def test_a_row_enqueued_as_the_writer_exits_still_lands(clean_audit, monkeypatch):
"""The gap between the writer's last empty-queue check and its done-callback: a row enqueued
there sees `_pending` occupied and starts nothing. The callback must start a writer for it, or
the row waits for the next call - forever, on a quiet server.
The writer is made to finish inside its first step (a batch writer that never suspends), so a
`call_soon` queued right behind that step runs after the task completed and BEFORE its done
callbacks - exactly the gap.
"""
landed: list[str] = []
async def instant(rows):
landed.extend(fields["query"] for _, fields in rows)
return True
monkeypatch.setattr(audit, "_write_batch", instant)
audit.record_search_miss(query="first", source="test")
(writer,) = audit._pending
seen: dict = {}
def late():
audit.record_search_miss(query="late", source="test")
seen["pending"] = set(audit._pending)
seen["queued"] = len(audit._queue)
asyncio.get_running_loop().call_soon(late)
await writer # resumes after the writer's done callbacks, `late` having run before them
assert seen == {"pending": {writer}, "queued": 1}, "the late row did not hit the gap"
# No drain: the done callback alone must have started a writer for the late row.
assert audit._pending and writer not in audit._pending
await asyncio.gather(*audit._pending)
assert landed == ["first", "late"]
+337
View File
@@ -0,0 +1,337 @@
"""The email-domain blocklist: throwaway mail and abusive signup domains, refused at every door.
A new team is created with a promotional balance, which makes bulk registration on throwaway
addresses worth someone's while. Two tiers: a CODE tier (domains confirmed abusive in our own data,
plus throwaway-mail keyword rules) and an OPS tier (`TREG_BLOCKED_EMAIL_DOMAINS`, additive, a
dashboard edit so a new domain needs no deploy). Each rule is pinned here because the obvious
implementation gets it wrong: match the DOMAIN only, walk parent domains but never the bare TLD,
refuse sign-in as well as sign-up, cover BOTH doors that mint a promo-funded team, reveal nothing to
the caller, count every block in the log, and fail open.
"""
from __future__ import annotations
import logging
import pytest
from fastapi import FastAPI
from httpx import ASGITransport, AsyncClient
from sqlmodel import select
from treg.api import app
from treg.application import signup
from treg.config import get_settings
from treg.domain.identity.access import _is_blocked_email
from treg.infra.db import reset_db, session_maker
from treg.models import Org, User
# The ops tier under test. Deliberately NOT the code tier's domains, so these tests prove the env
# var itself works end to end; `.example` is reserved and can never be a real user's domain.
OPS = "farm-a.example, Farm-B.example ,@farm-c.example,.farm-d.example"
REFUSAL = "this address cannot be used to sign in"
@pytest.fixture
def ops(monkeypatch):
"""Set the ops tier on the live Settings object (the shape conftest uses for `posthog_key`)."""
def _set(raw: str = OPS) -> None:
monkeypatch.setattr(get_settings(), "blocked_email_domains", raw, raising=False)
return _set
@pytest.fixture
async def client():
await reset_db()
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as c:
yield c
async def _user_count(email: str) -> int:
async with session_maker() as s:
return len((await s.execute(select(User).where(User.email == email))).scalars().all())
async def _otp_start(c: AsyncClient, email: str):
return await c.post("/auth/email/start", json={"email": email})
async def _otp_login(c: AsyncClient, email: str) -> str:
code = (await _otp_start(c, email)).json()["dev_code"]
r = await c.post("/auth/email/verify", json={"email": email, "code": code})
assert r.status_code == 200, r.text
return r.json()["token"]
# ---- the classifier ------------------------------------------------------------------------------
def test_code_tier_is_in_force_with_no_setting_at_all(ops):
ops("")
assert get_settings().blocked_email_domain_set == frozenset()
for email in ("a@uberip.com", "a@westcast-systems.com", "a@mailfox.win", "a@yopmail.com",
"a@mail.uberip.com", # subdomain of a confirmed root
"a@txtfromrizkirmdhn.my.id", # any `.my.id`, via the parent walk
"a@tempmail-fresh.xyz", "a@guerrillamail.info", "a@x.10minutemail.net"): # keywords
assert _is_blocked_email(email), email
def test_match_is_on_the_domain_only_never_the_local_part(ops):
"""The single most important rule. Matching the whole address false-flags real users whose
USERNAME happens to contain a keyword, which is how a blocklist starts refusing customers."""
ops()
assert not _is_blocked_email("tempmail@gmail.com")
assert not _is_blocked_email("yopmail.fan@company.dev")
assert not _is_blocked_email("farm-a.example@company.dev")
def test_ops_tier_parses_case_whitespace_and_leading_marks(ops):
ops()
assert get_settings().blocked_email_domain_set == frozenset(
{"farm-a.example", "farm-b.example", "farm-c.example", "farm-d.example"})
def test_ops_tier_matches_domain_and_subdomains_and_adds_to_the_code_tier(ops):
ops()
assert _is_blocked_email("a@farm-a.example")
assert _is_blocked_email("A@FARM-B.EXAMPLE")
assert _is_blocked_email("a@deep.mail.farm-c.example") # the subdomain bypass that must not work
assert _is_blocked_email("a@farm-d.example") # listed as ".farm-d.example"
assert _is_blocked_email("a@uberip.com") # the code tier is still there
def test_walk_strips_whole_labels_off_the_front_only(ops):
ops()
assert not _is_blocked_email("a@notfarm-a.example") # a string suffix, not a subdomain
assert not _is_blocked_email("a@farm-a.example.org") # the listed domain in the middle
assert not _is_blocked_email("a@company.dev")
assert not _is_blocked_email("a@uberip.co")
def test_a_bare_public_suffix_can_never_be_an_entry(ops):
"""`com` in the dashboard field must not refuse every address on earth."""
ops("com, net, , @, .")
assert get_settings().blocked_email_domain_set == frozenset()
assert not _is_blocked_email("a@company.com")
assert not _is_blocked_email("a@id") # the walk never tests the last label alone
def test_the_decision_logs_one_countable_line_per_block(ops, caplog):
ops()
with caplog.at_level(logging.WARNING, logger="treg.auth"):
assert signup.blocked_email("Farm@Mail.Farm-A.example", "otp_start")
assert not signup.blocked_email("ok@company.dev", "otp_start")
lines = [r.getMessage() for r in caplog.records if "signup_blocked_domain" in r.getMessage()]
assert lines == ["event=signup_blocked_domain door=otp_start domain=mail.farm-a.example"]
def test_the_decision_fails_open_on_a_classifier_error(monkeypatch, caplog):
def boom(email):
raise RuntimeError("bad blocklist")
monkeypatch.setattr(signup, "_is_blocked_email", boom)
with caplog.at_level(logging.ERROR, logger="treg.auth"):
assert not signup.blocked_email("a@uberip.com", "otp_start") # the door stays open
assert any("blocklist_error" in r.getMessage() for r in caplog.records)
# ---- the email OTP door --------------------------------------------------------------------------
async def test_otp_start_refuses_both_tiers_and_mints_no_code(client, ops):
ops()
for email in ("farm@farm-a.example", "farm@mail.farm-a.example", "Farm@UBERIP.com",
"farm@abc.my.id", "farm@tempmail-fresh.xyz"):
r = await _otp_start(client, email)
assert r.status_code == 403, (email, r.text)
assert r.json()["detail"] == REFUSAL
assert "dev_code" not in r.json() # no code minted, nothing to verify
async def test_otp_refusal_names_no_list_and_no_domain(client, ops):
ops()
body = (await _otp_start(client, "farm@farm-a.example")).text.lower()
for word in ("farm-a", "farm-b", "uberip", "block", "list", "domain"):
assert word not in body
async def test_otp_verify_refuses_a_code_minted_before_the_domain_was_listed(client, ops):
email = "farm@farm-b.example"
code = (await _otp_start(client, email)).json()["dev_code"] # not yet listed: code issued
ops()
r = await client.post("/auth/email/verify", json={"email": email, "code": code})
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
assert "treg_session" not in r.headers.get("set-cookie", "")
assert await _user_count(email) == 0 # refused BEFORE the row is created
async def test_otp_refuses_sign_in_of_an_account_that_predates_the_listing(client, ops):
"""Sign-in, not only sign-up: an existing account on a listed domain gets no new session."""
email = "early@farm-d.example"
await _otp_login(client, email)
assert await _user_count(email) == 1
ops()
assert (await _otp_start(client, email)).status_code == 403
async def test_otp_still_works_for_an_unlisted_domain_while_the_list_is_set(client, ops):
ops()
assert await _otp_login(client, "real@company.dev")
assert await _otp_login(client, "tempmail@company.dev") # local part is never looked at
# ---- open registration (POST /users: user + org + the $1 promo in one call) -----------------------
async def test_open_registration_refuses_a_blocked_domain_and_creates_nothing(client, ops):
ops()
for email in ("farm@sub.farm-a.example", "farm@sub.uberip.com"):
r = await client.post("/users", json={"email": email})
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
assert await _user_count(email) == 0
async with session_maker() as s:
assert (await s.execute(select(Org))).scalars().all() == [] # no team, so no grant
async def test_open_registration_is_unchanged_for_an_unlisted_domain(client, ops):
ops("")
r = await client.post("/users", json={"email": "someone@farm-a.example"})
assert r.status_code == 200, r.text
assert r.json()["email"] == "someone@farm-a.example"
# ---- creating a team (POST /orgs: the other promo door, for an already-registered identity) -------
async def test_create_org_refuses_an_identity_registered_before_its_domain_was_listed(client, ops):
tok = await _otp_login(client, "early@farm-a.example")
ops()
r = await client.post("/orgs", json={"name": "Farm 1"}, headers={"X-Treg-Token": tok})
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
async with session_maker() as s:
assert (await s.execute(select(Org))).scalars().all() == []
# ---- the social doors (GitHub, Google) -----------------------------------------------------------
def _github_idp(email: str) -> FastAPI:
g = FastAPI()
@g.post("/login/oauth/access_token")
async def token() -> dict:
return {"access_token": "gho_test", "token_type": "bearer"}
@g.get("/user")
async def user() -> dict:
return {"login": "farm", "email": email}
return g
def _google_idp(email: str) -> FastAPI:
g = FastAPI()
@g.post("/token")
async def token() -> dict:
return {"access_token": "goog_test", "token_type": "bearer"}
@g.get("/userinfo")
async def userinfo() -> dict:
return {"email": email, "email_verified": True}
return g
@pytest.fixture
async def social(monkeypatch):
"""Both social doors configured against in-process fake identity providers. The ops tier goes
in through the environment, as an operator would set it."""
monkeypatch.setenv("TREG_GITHUB_CLIENT_ID", "cid")
monkeypatch.setenv("TREG_GITHUB_CLIENT_SECRET", "csec")
monkeypatch.setenv("TREG_GITHUB_TOKEN_URL", "http://idp/login/oauth/access_token")
monkeypatch.setenv("TREG_GITHUB_API_URL", "http://idp")
monkeypatch.setenv("TREG_GOOGLE_CLIENT_ID", "gid")
monkeypatch.setenv("TREG_GOOGLE_CLIENT_SECRET", "gsec")
monkeypatch.setenv("TREG_GOOGLE_TOKEN_URL", "http://idp/token")
monkeypatch.setenv("TREG_GOOGLE_USERINFO_URL", "http://idp/userinfo")
monkeypatch.setenv("TREG_SESSION_SECRET", "test-session-secret")
monkeypatch.setenv("TREG_BLOCKED_EMAIL_DOMAINS", OPS)
get_settings.cache_clear()
await reset_db()
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as c:
yield c
if getattr(app.state, "http", None) is not None:
await app.state.http.aclose()
get_settings.cache_clear()
async def _social_callback(c: AsyncClient, door: str, idp: FastAPI):
app.state.http = AsyncClient(transport=ASGITransport(app=idp), base_url="http://idp")
r = await c.get(f"/auth/{door}", follow_redirects=False)
assert r.status_code == 302
state = c.cookies.get("treg_oauth_state")
return await c.get(f"/auth/{door}/callback?code=abc&state={state}", follow_redirects=False)
@pytest.mark.parametrize("door,email", [
("github", "farm@farm-a.example"), # ops tier
("google", "farm@mail.uberip.com"), # code tier, subdomain
])
async def test_social_login_on_a_blocked_domain_gets_a_refusal_page_and_no_session(social, door, email):
idp = _github_idp(email) if door == "github" else _google_idp(email)
cb = await _social_callback(social, door, idp)
assert cb.status_code == 403, cb.text
assert "cannot be used to sign in" in cb.text
assert "farm-a" not in cb.text and "uberip" not in cb.text
assert "treg_session" not in cb.headers.get("set-cookie", "")
assert (await social.get("/auth/me")).status_code == 401
assert await _user_count(email) == 0
async def test_social_login_on_an_unlisted_domain_still_signs_in(social):
cb = await _social_callback(social, "google", _google_idp("ok@company.dev"))
assert cb.status_code == 302 and cb.headers["location"] == "/app"
assert (await social.get("/auth/me")).json()["email"] == "ok@company.dev"
# ---- invites: the emailed link (mints a session) and the code (mints a membership token) ----------
@pytest.fixture
def sent_invites(monkeypatch):
from treg import email as email_mod
sent = []
async def _capture(email, inviter, org_name, role, code, email_token, expires_at="", link_base="",
shared=""):
sent.append({"email": email, "code": code, "email_token": email_token})
return True
monkeypatch.setattr(email_mod, "send_invite", _capture)
return sent
async def _team_with_invite(c: AsyncClient, owner: str, invitee: str) -> dict:
tok = await _otp_login(c, owner)
org = (await c.post("/orgs", json={"name": "Real Team"}, headers={"X-Treg-Token": tok})).json()
r = await c.post(f"/orgs/{org['org_id']}/invites", json={"email": invitee, "role": "member"},
headers={"X-Treg-Token": tok, "X-Treg-Org": org["org"]})
assert r.status_code == 200, r.text
return org
async def test_emailed_invite_link_refuses_a_blocked_domain(client, ops, sent_invites):
await _team_with_invite(client, "owner@company.dev", "farm@farm-a.example")
ops()
t = sent_invites[0]["email_token"]
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as visitor:
r = await visitor.post("/auth/invite-signin", content=f"t={t}",
headers={"content-type": "application/x-www-form-urlencoded"},
follow_redirects=False)
assert r.status_code == 403 and "cannot be used to sign in" in r.text
assert "treg_session" not in r.headers.get("set-cookie", "")
assert (await visitor.get("/invites/mine")).status_code == 401
assert await _user_count("farm@farm-a.example") == 0
async def test_invite_code_accept_refuses_a_blocked_domain(client, ops, sent_invites):
await _team_with_invite(client, "owner@company.dev", "farm@sub.farm-b.example")
ops()
r = await client.post("/invites/accept",
json={"code": sent_invites[0]["code"], "email": "farm@sub.farm-b.example"})
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
assert "token" not in r.json()
assert await _user_count("farm@sub.farm-b.example") == 0
+12 -1
View File
@@ -14,6 +14,7 @@ from treg.application import billing
from treg.application.call import authorize, overflow, reserve, service, settle
from treg.domain import money
from treg.domain.capacity import marks as capacity_marks
from treg.domain.governance import usage as usage_policy
_SRC = Path(__file__).parents[1] / "src" / "treg"
@@ -64,6 +65,12 @@ _DATAPLANE_DERIVED_WRITES = {
"async_resource_ownership": (
(service._execute_call, "async_task_app.remember_platform_resources"),
),
# The per-user daily cap takes its slot with one conditional UPDATE of the member's row
# (revision 0024) instead of counting the member's callrecord rows per call.
"member_daily_cap_slot": (
(authorize.authorize_call, "usage_policy.enforce_daily_cap"),
(usage_policy.enforce_daily_cap, "take_daily_slot"),
),
}
_EXPECTED_DATAPLANE_WRITES = frozenset({
"auto_topup_task",
@@ -76,12 +83,14 @@ _EXPECTED_DATAPLANE_WRITES = frozenset({
"overflow_budget_reservation",
"async_result_ownership",
"async_resource_ownership",
"member_daily_cap_slot",
})
_DERIVED_WRITE_FILES = {
_SRC / "application" / "billing.py": {"loop.create_task"},
_SRC / "application" / "call" / "authorize.py": {
"publicdemo_policy.enforce_public_demo_ip_cap",
"publicdemo_policy.enforce_public_demo_ip_cap", "usage_policy.enforce_daily_cap",
},
_SRC / "domain" / "governance" / "usage.py": {"take_daily_slot"},
_SRC / "application" / "call" / "reserve.py": {"billing.maybe_schedule_autotopup"},
_SRC / "application" / "call" / "settle.py": {
"adsconv.queue", "capacity_marks.strike", "capacity_marks.clear",
@@ -108,6 +117,8 @@ _EXPECTED_DERIVED_WRITE_SITES = {
"publicdemo_policy.enforce_public_demo_ip_cap"),
("application/call/authorize.py", "enforce_public_demo_limit",
"publicdemo_policy.enforce_public_demo_ip_cap"),
("application/call/authorize.py", "authorize_call", "usage_policy.enforce_daily_cap"),
("domain/governance/usage.py", "enforce_daily_cap", "take_daily_slot"),
("application/call/reserve.py", "_platform_reserve",
"billing.maybe_schedule_autotopup"),
("application/call/settle.py", "_record_first_call", "adsconv.queue"),
+131
View File
@@ -0,0 +1,131 @@
"""`Org.spent_today_micro` is the daily cap's number and must agree with the journal.
The counter exists so the fail-closed daily cap costs one primary-key read on every metered call
instead of an aggregate over the platform's whole day (revision 0022). These tests pin the
semantics the journal view (`spent_today_from_ledger`) has always had: settled today plus still
held from today, reset at the UTC day boundary, with a hold opened yesterday belonging to yesterday.
"""
from datetime import date, timedelta
import pytest
from httpx import ASGITransport, AsyncClient
from sqlalchemy import update
from treg.api import app
from treg.domain import money as ledger
from treg.infra.db import reset_db, session_maker
from treg.models import Hold, Org
from conftest import make_upstream
EP = "acme.thing.get"
@pytest.fixture
async def c():
await reset_db()
app.state.http = AsyncClient(transport=ASGITransport(app=make_upstream()), base_url="http://upstream")
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as client:
yield client
await app.state.http.aclose()
async def _org(c: AsyncClient) -> int:
r = await c.post("/users", json={"email": "counter@superdesign.dev"})
assert r.status_code == 200, r.text
return r.json()["org_id"]
async def _both(org_id: int) -> tuple[int, int]:
"""(counter, journal) — every assertion below checks them against each other too."""
async with session_maker() as db:
return await ledger.spent_today(db, org_id), await ledger.spent_today_from_ledger(db, org_id)
async def test_counter_follows_reserve_settle_and_release(c: AsyncClient):
org_id = await _org(c)
assert await _both(org_id) == (0, 0)
async with session_maker() as db:
a = await ledger.reserve(db, org_id, EP, 1_000)
b = await ledger.reserve(db, org_id, EP, 2_000)
held_a, held_b = ledger.with_margin(1_000), ledger.with_margin(2_000)
assert await _both(org_id) == (held_a + held_b, held_a + held_b)
# Settling below the estimate: the hold leaves, what was consumed stays.
async with session_maker() as db:
consumed = await ledger.settle(db, a, 400)
assert consumed == ledger.with_margin(400)
assert await _both(org_id) == (consumed + held_b, consumed + held_b)
# Releasing: the hold leaves and nothing replaces it.
async with session_maker() as db:
await ledger.release(db, b, reason="upstream 503")
assert await _both(org_id) == (consumed, consumed)
# A second settle of a closed hold is a no-op for the counter, as for the balance.
async with session_maker() as db:
assert await ledger.settle(db, a, 999) == 0
assert await _both(org_id) == (consumed, consumed)
async def test_settle_overrun_counts_what_was_actually_consumed(c: AsyncClient):
org_id = await _org(c)
async with session_maker() as db:
call = await ledger.reserve(db, org_id, EP, 1_000)
consumed = await ledger.settle(db, call, 5_000) # the provider charged more than estimated
assert consumed == ledger.with_margin(5_000)
assert await _both(org_id) == (consumed, consumed)
async def test_counter_resets_on_a_new_utc_day(c: AsyncClient):
org_id = await _org(c)
async with session_maker() as db:
call = await ledger.reserve(db, org_id, EP, 1_000)
await ledger.settle(db, call, 1_000)
spent = ledger.with_margin(1_000)
assert await _both(org_id) == (spent, spent)
# Move the counter to "yesterday" - as the day rolling over would leave it - and it reads 0.
yesterday = date.today() - timedelta(days=1)
async with session_maker() as db:
await db.execute(update(Org).where(Org.id == org_id).values(spent_today_day=yesterday))
await db.commit()
async with session_maker() as db:
assert await ledger.spent_today(db, org_id) == 0
# The first movement of the new day starts from that movement, not from yesterday's total.
async with session_maker() as db:
await ledger.reserve(db, org_id, EP, 300)
async with session_maker() as db:
assert await ledger.spent_today(db, org_id) == ledger.with_margin(300)
async def test_a_hold_opened_yesterday_belongs_to_yesterday(c: AsyncClient):
"""Settling or releasing an older hold must not subtract from today what today never counted."""
org_id = await _org(c)
async with session_maker() as db:
old = await ledger.reserve(db, org_id, EP, 1_000)
await ledger.reserve(db, org_id, EP, 2_000)
# Age the first hold: its row says yesterday, and the counter no longer includes it.
async with session_maker() as db:
await db.execute(update(Hold).where(Hold.id == old).values(
created_at=ledger._now() - timedelta(days=1)))
await db.execute(update(Org).where(Org.id == org_id).values(
spent_today_micro=ledger.with_margin(2_000)))
await db.commit()
assert await _both(org_id) == (ledger.with_margin(2_000), ledger.with_margin(2_000))
async with session_maker() as db:
consumed = await ledger.settle(db, old, 500) # settled TODAY, so what it consumed counts today
expect = ledger.with_margin(2_000) + consumed
assert await _both(org_id) == (expect, expect)
async with session_maker() as db:
older = await ledger.reserve(db, org_id, EP, 700)
await db.execute(update(Hold).where(Hold.id == older).values(
created_at=ledger._now() - timedelta(days=1)))
await db.execute(update(Org).where(Org.id == org_id).values(spent_today_micro=expect))
await db.commit()
async with session_maker() as db:
await ledger.release(db, older, reason="stale") # nothing counted today, nothing leaves
assert await _both(org_id) == (expect, expect)
+67
View File
@@ -0,0 +1,67 @@
"""The pool gauge is the reading behind `TREG_DB_POOL_OVERRIDES`: per-minute peak checked-out
connections per pool, next to capacity. Sizing by arithmetic got both minor pools wrong once each
(ops/deploy.md § Three pools); this is what settles the number."""
import asyncio
import pytest
from treg import analytics, bootstrap
from treg.infra import db as infra_db
def test_snapshot_reports_nothing_on_sqlite_and_a_pool_row_per_engine_otherwise():
snap = infra_db.pool_snapshot()
if infra_db._is_sqlite:
assert snap == {}
return
assert set(snap) == {"api", "admin", "background"}
for name, row in snap.items():
spec = infra_db.POOL_SPECS[name]
assert row["capacity"] == spec["pool_size"] + spec["max_overflow"]
assert 0 <= row["checked_out"] <= row["capacity"]
def test_fold_keeps_the_per_pool_maximum():
peaks: dict[str, int] = {}
infra_db.fold_pool_peaks(peaks, {"api": {"checked_out": 3, "capacity": 15}})
infra_db.fold_pool_peaks(peaks, {"api": {"checked_out": 9, "capacity": 15},
"background": {"checked_out": 2, "capacity": 13}})
infra_db.fold_pool_peaks(peaks, {"api": {"checked_out": 1, "capacity": 15}})
assert peaks == {"api": 9, "background": 2}
async def test_gauge_emits_one_event_per_window_with_peak_capacity_and_headroom(monkeypatch):
samples = iter([
{"api": {"checked_out": 4, "capacity": 15}, "background": {"checked_out": 1, "capacity": 13}},
{"api": {"checked_out": 11, "capacity": 15}, "background": {"checked_out": 13, "capacity": 13}},
{"api": {"checked_out": 2, "capacity": 15}, "background": {"checked_out": 0, "capacity": 13}},
])
last = {"api": {"checked_out": 0, "capacity": 15}, "background": {"checked_out": 0, "capacity": 13}}
monkeypatch.setattr(infra_db, "pool_snapshot", lambda: next(samples, last))
captured: list[tuple[str, str, dict]] = []
monkeypatch.setattr(analytics, "capture", lambda d, e, p=None, **kw: captured.append((d, e, p)))
task = asyncio.create_task(bootstrap.pool_gauge(sample_s=0.01, emit_s=0.05))
try:
for _ in range(100):
await asyncio.sleep(0.01)
if captured:
break
finally:
task.cancel()
with pytest.raises(asyncio.CancelledError):
await task
assert captured, "no gauge event within the window"
distinct_id, event, props = captured[0]
assert (distinct_id, event) == (analytics.SERVER_DISTINCT_ID, "db_pool_gauge")
assert props["api_peak"] == 11 and props["api_capacity"] == 15 and props["api_headroom"] == 4
assert props["background_peak"] == 13 and props["background_headroom"] == 0
assert props["samples"] >= 3
def test_connection_budget_multiplies_per_process_by_workers_and_the_deploy_overlap():
per_process = sum(s["pool_size"] + s["max_overflow"] for s in infra_db.POOL_SPECS.values())
one = infra_db.connection_budget(workers=1)
two = infra_db.connection_budget(workers=2)
assert one == {"per_process": per_process, "workers": 1,
"per_instance": per_process, "deploy_peak": per_process * 2}
assert two["per_instance"] == per_process * 2 and two["deploy_peak"] == per_process * 4
+1 -1
View File
@@ -118,7 +118,7 @@ def test_the_background_pool_fits_every_consumer_not_just_the_throttled_ones():
from treg import archive, audit
consumers = infra_db.BACKGROUND_CONSUMERS
assert consumers["audit._write"] == audit._MAX_CONCURRENT_WRITES
assert consumers["audit._flush"] == audit._MAX_CONCURRENT_WRITES
assert consumers["archive._store/_touch"] == archive._MAX_CONCURRENT_WRITES
assert infra_db.POOL_SPECS["background"]["pool_size"] >= sum(consumers.values())
+11
View File
@@ -82,3 +82,14 @@ async def test_sitemap_and_catalog_link_every_provider_page(clients: AsyncClient
assert f"/tools/{s}<" in sm, s
assert f'href="/tools/{s}"' in cat, s
assert "/pricing<" in sm
async def test_provider_page_reads_the_observation_reader_not_the_session(clients, caplog):
"""`/tools/{service}` once handed `_observed_or_empty` the request's AsyncSession instead of the
app's observation reader; the measured line silently came up empty and every page view logged
an AttributeError traceback (prod, 2026-09-06)."""
import logging
with caplog.at_level(logging.WARNING, logger="treg.catalog"):
r = await clients.get("/tools/dataforseo")
assert r.status_code == 200
assert "endpoint stats unavailable" not in caplog.text
+26
View File
@@ -7,6 +7,8 @@ SEPARATE session — if the state were still a per-process dict, the second sess
from __future__ import annotations
from sqlalchemy.dialects import postgresql, sqlite
from treg import ratestore
from treg.infra.db import reset_db, session_maker
from treg.models import Ephemeral
@@ -94,3 +96,27 @@ async def test_sweep_drops_expired_rows_only():
await db.commit()
assert await db.get(Ephemeral, ("sandbox_hit", "dead")) is None
assert await db.get(Ephemeral, ("sandbox_hit", "live")) is not None
async def test_kv_put_of_a_missing_key_is_an_upsert_on_both_backends():
"""Two processes striking the same provider write the same lock key in the same millisecond; the
loser of a plain INSERT died on `ephemeral_pkey` (prod, 2026-09-06). The statement must carry
ON CONFLICT on both dialects so the last writer wins instead of raising."""
from datetime import datetime
for dialect in ("postgresql", "sqlite"):
stmt = ratestore._upsert(dialect, ns="capacity:lock", k="p", v={"a": 1},
expires_at=datetime(2026, 1, 1))
sql = str(stmt.compile(dialect=(postgresql.dialect() if dialect == "postgresql" else sqlite.dialect())))
assert "ON CONFLICT (ns, k) DO UPDATE" in sql, sql
async def test_kv_put_overwrites_a_row_written_by_another_session():
async with session_maker() as other:
await ratestore.kv_put(other, "ns", "k", {"n": 1}, ttl_s=600)
await other.commit()
async with session_maker() as db:
# This session has never loaded (ns, k); its `get` sees the committed row and updates it.
await ratestore.kv_put(db, "ns", "k", {"n": 2}, ttl_s=600)
await db.commit()
async with session_maker() as db:
assert (await ratestore.kv_get(db, "ns", "k")) == {"n": 2}
+5 -4
View File
@@ -677,10 +677,11 @@ async def test_a_pin_cannot_smuggle_what_the_header_cannot(clients: AsyncClient)
# ---- the invariants an invoice actually rests on -------------------------------------------------
async def test_usage_survives_a_dead_audit_pipeline(clients: AsyncClient, platform_on, monkeypatch):
"""THE test that proves an invoice never depends on a lossy table. `audit._schedule` sheds rows
past its queue bound and swallows every exception — precisely under the load a successful builder
generates. With the audit pipeline entirely dead, the money must still be complete."""
monkeypatch.setattr(audit, "_schedule", lambda coro: coro.close())
"""THE test that proves an invoice never depends on a lossy table. `audit._enqueue` sheds rows
past its queue bound and the writer swallows every exception — precisely under the load a
successful builder generates. With the audit pipeline entirely dead, the money must still be
complete."""
monkeypatch.setattr(audit, "_enqueue", lambda model, fields: None)
org_id = await _org_id(clients)
for who in ("cust_A", "cust_B"):
r = await clients.get(f"/call/{EP}?aweme_id=7", headers={"X-Treg-Meta": f"customer={who}"})
+61 -10
View File
@@ -6,7 +6,8 @@ Soft by design (best-effort audit → fails open), so these tests seed records d
from __future__ import annotations
from datetime import timedelta
from datetime import date, timedelta
from pathlib import Path
from httpx import AsyncClient
from sqlmodel import select
@@ -64,22 +65,42 @@ async def test_unlimited_by_default(clients: AsyncClient):
assert (await clients.get("/call/echo/anything")).status_code == 200
async def test_runs_and_grants_count_toward_the_same_cap(clients: AsyncClient):
"""A member can't dodge the cap by switching path: CallRecord (call+local) AND RunRecord (server)
both count. Seed one of each to reach cap=2, then a proxy call is refused."""
async def _counter(org_id: int, email: str) -> tuple[int, date | None]:
async with session_maker() as s:
uid = (await s.execute(select(User.id).where(User.email == email))).scalar_one()
m = (await s.execute(select(Membership).where(
Membership.user_id == uid, Membership.org_id == org_id))).scalar_one()
return m.calls_today, m.calls_today_day
async def _set_counter(org_id: int, email: str, n: int, day: date) -> None:
async with session_maker() as s:
uid = (await s.execute(select(User.id).where(User.email == email))).scalar_one()
m = (await s.execute(select(Membership).where(
Membership.user_id == uid, Membership.org_id == org_id))).scalar_one()
m.calls_today, m.calls_today_day = n, day
await s.commit()
async def test_runs_and_calls_share_one_gate(clients: AsyncClient):
"""A member can't dodge the cap by switching path: both run handlers in api.py go through the
same `_enforce_daily_cap` door as `/call/` (authorize.py), and that door is what moves the
counter. Pinned statically because the run surfaces need a bundle to exercise end to end."""
src = (Path(__file__).parents[1] / "src" / "treg" / "api.py").read_text()
assert src.count("await _enforce_daily_cap(caller, db)") == 2 # local-run grant + server run
await _mk_echo_tool(clients)
org_id = await _set_cap(2)
await _seed_call(org_id, "tim@superdesign.dev") # 1 (a prior proxy/local event)
await _seed_run(org_id, "tim@superdesign.dev") # 2 (a prior server run)
await _set_counter(org_id, "tim@superdesign.dev", 2, date.today()) # two prior events today
blocked = await clients.get("/call/echo/anything")
assert blocked.status_code == 429 # used=2 (call+run) >= cap=2
assert blocked.status_code == 429 and "2/2" in blocked.json()["detail"]
assert await _counter(org_id, "tim@superdesign.dev") == (2, date.today()) # refused = not counted
async def test_cap_is_per_member_not_global(clients: AsyncClient):
"""Capping one member must not affect another in the same org."""
await _mk_echo_tool(clients)
org_id = await _set_cap(1)
await _seed_call(org_id, "tim@superdesign.dev") # tim is now at his cap
assert (await clients.get("/call/echo/anything")).status_code == 200 # tim is now at his cap
# invite bob into the SAME org (default cap -1)
code = (await clients.post(f"/orgs/{org_id}/invites", json={"email": "bob@x.io", "role": "member"})).json()["code"]
btok = (await clients.post("/invites/accept", json={"code": code, "email": "bob@x.io"})).json()["token"]
@@ -92,8 +113,38 @@ async def test_cap_is_per_member_not_global(clients: AsyncClient):
async def test_yesterdays_usage_does_not_count_today(clients: AsyncClient):
await _mk_echo_tool(clients)
org_id = await _set_cap(1)
await _seed_call(org_id, "tim@superdesign.dev", days_ago=1) # yesterday — outside today's window
assert (await clients.get("/call/echo/anything")).status_code == 200 # today's count is still 0
await _set_counter(org_id, "tim@superdesign.dev", 1, date.today() - timedelta(days=1)) # yesterday's
assert (await clients.get("/call/echo/anything")).status_code == 200 # a new day starts from 0
assert await _counter(org_id, "tim@superdesign.dev") == (1, date.today()) # ...and this was its first
assert (await clients.get("/call/echo/anything")).status_code == 429
async def test_setting_a_cap_seeds_the_counter_from_todays_journal(clients: AsyncClient):
"""Unlimited members are not counted on the call path, so a cap set mid-day starts from what the
journal says they already used - not from zero."""
await _mk_echo_tool(clients)
org_id = await _get_org_id()
for _ in range(3):
assert (await clients.get("/call/echo/anything")).status_code == 200
await audit.drain()
await _seed_run(org_id, "tim@superdesign.dev") # a server run counts too: 4 events in the journal
uid = [x["user_id"] for x in (await clients.get(f"/orgs/{org_id}/members")).json()
if x["email"] == "tim@superdesign.dev"][0]
assert (await clients.patch(f"/orgs/{org_id}/members/{uid}/cap", json={"daily_call_cap": 5})).status_code == 200
assert await _counter(org_id, "tim@superdesign.dev") == (4, date.today())
assert (await clients.get("/call/echo/anything")).status_code == 200 # 5th
assert (await clients.get("/call/echo/anything")).status_code == 429 # 6th
async def test_counter_agrees_with_the_journal_after_real_calls(clients: AsyncClient):
await _mk_echo_tool(clients)
org_id = await _set_cap(10)
for _ in range(4):
assert (await clients.get("/call/echo/anything")).status_code == 200
await audit.drain()
async with session_maker() as s:
journal = await count_today(s, org_id, "tim@superdesign.dev")
assert await _counter(org_id, "tim@superdesign.dev") == (journal, date.today()) == (4, date.today())
async def _get_org_id(email: str = "tim@superdesign.dev") -> int: