mirror of
https://github.com/superdesigndev/treg.git
synced 2026-10-02 03:24:35 +08:00
chore: merge main into adtrack landing fix
This commit is contained in:
@@ -32,7 +32,9 @@ Regenerate via `scripts/build-map.py`.
|
||||
| `scripts/build_plugin.py` | interface/skill.md |
|
||||
| `scripts/catalog_drift.py` | architecture/catalog.md |
|
||||
| `scripts/catalog_ingest.py` | architecture/catalog.md, architecture/instagram-oauth.md |
|
||||
| `scripts/catalog_replicate_prices.py` | architecture/catalog.md |
|
||||
| `scripts/catalog_validate.py` | architecture/catalog.md |
|
||||
| `scripts/catalog_verify_extended.py` | architecture/catalog.md |
|
||||
| `scripts/dev-local.sh` | ops/deploy.md |
|
||||
| `scripts/dump_surface.py` | architecture/composition.md |
|
||||
| `scripts/indexnow_submit.py` | interface/seo.md |
|
||||
@@ -58,6 +60,11 @@ Regenerate via `scripts/build-map.py`.
|
||||
| `src/treg/alembic/versions/0017_async_task_record.py` | architecture/data-model.md, architecture/money.md, architecture/multi-tenancy.md |
|
||||
| `src/treg/alembic/versions/0018_async_resource_ownership.py` | architecture/data-model.md, architecture/money.md, architecture/multi-tenancy.md |
|
||||
| `src/treg/alembic/versions/0019_async_poll_failures.py` | architecture/data-model.md, architecture/money.md |
|
||||
| `src/treg/alembic/versions/0020_callrecord_created_at_indexes.py` | architecture/data-model.md |
|
||||
| `src/treg/alembic/versions/0021_ledgerentry_org_created_at_index.py` | architecture/data-model.md |
|
||||
| `src/treg/alembic/versions/0022_org_spent_today_counter.py` | architecture/data-model.md |
|
||||
| `src/treg/alembic/versions/0023_callrecord_org_user_created_at_index.py` | architecture/data-model.md |
|
||||
| `src/treg/alembic/versions/0024_membership_calls_today_counter.py` | architecture/data-model.md |
|
||||
| `src/treg/analytics.py` | architecture/data-model.md |
|
||||
| `src/treg/api.py` | architecture/archive.md, architecture/money.md, architecture/multi-tenancy.md, architecture/proxy-model.md, architecture/super-admin.md, interface/api.md, interface/dashboard.md, interface/landing-sandbox.md, interface/seo.md |
|
||||
| `src/treg/application/__init__.py` | architecture/import-boundaries.md |
|
||||
@@ -318,9 +325,9 @@ Regenerate via `scripts/build-map.py`.
|
||||
| `architecture/ads-conversions.md` | `adsconv.py`, `signup.py`, `adtrack.js`, `gtag.js` |
|
||||
| `architecture/archive.md` | `archive.py`, `0002_archive_tables.py`, `0003_callrecord_cached.py`, `0004_archivekey_request_shape.py`, `0011_callrecord_archive_link.py`, `service.py`, `backfill_call_archive_links.py`, `api.py`, `bootstrap.py`, `admin.py`, `asynctasks.py` |
|
||||
| `architecture/auth-secrets.md` | `injectors.py`, `ssrf.py`, `crypto.py`, `oauth.py`, `__init__.py`, `authorization.py`, `oauth_flow.py`, `refresh.py`, `oauth_exchange.py`, `oauth_refresh.py`, `oauth_providers.py`, `health.py`, `connect.py`, `connections.py`, `resources.py`, `__init__.py`, `bindings.py`, `bundles.py`, `test_oauth_refresh.py`, `config.py` |
|
||||
| `architecture/catalog.md` | `contracts.yaml`, `adapters.yaml`, `findymail.search.business-profile.json`, `__init__.py`, `contracts.py`, `paths.py`, `plan.py`, `synthetic.py`, `route.py`, `test_routing.py`, `catalog-drift.yml`, `catalog_drift.py`, `catalog_ingest.py`, `catalog_validate.py`, `aliases.yaml`, `fx.yaml`, `aviato.yaml`, `crustdata.yaml`, `aviato.companies.acquisitions.json`, `aviato.companies.employees.json`, `aviato.companies.enrich.bulk.json`, `aviato.companies.enrich.json`, `aviato.companies.founders.json`, `aviato.companies.funding_rounds.json`, `aviato.companies.investments.json`, `aviato.companies.outbound_investments.json`, `aviato.companies.search.json`, `aviato.linkedin.company.posts.json`, `aviato.linkedin.post.comments.json`, `aviato.linkedin.post.reactions.json`, `aviato.linkedin.post.reposts.json`, `aviato.linkedin.user.posts.json`, `aviato.people.contact.get.json`, `aviato.people.email.find.json`, `aviato.people.enrich.bulk.json`, `aviato.people.enrich.json`, `aviato.people.phone.find.json`, `aviato.people.search.json`, `aviato.people.search.simple.json`, `crustdata.companies.autocomplete.json`, `crustdata.companies.enrich.json`, `crustdata.companies.identify.json`, `crustdata.companies.jobs.search.json`, `crustdata.companies.search.json`, `crustdata.people.autocomplete.json`, `crustdata.people.enrich.json`, `crustdata.people.search.json`, `google-search-console.yaml`, `google-search-console.extended.yaml`, `google-tag-manager.yaml`, `google-tag-manager.extended.yaml`, `instagram.yaml`, `instagram.extended.yaml`, `justoneapi.extended.yaml`, `minimax.yaml`, `apify.yaml`, `brightdata.yaml`, `companyenrich.yaml`, `oceanio.yaml`, `akta.extended.yaml`, `dataforseo.extended.yaml`, `tikhub.extended.yaml`, `minimax.video-gen.result.retrieve.json`, `minimax.video-gen.from_image.json`, `minimax.video-gen.task.status.json`, `openrouter.yaml`, `openrouter.extended.yaml`, `openrouter.x.alibaba-wan-3-0.json`, `replicate.yaml`, `replicate.extended.yaml`, `replicate.image-gen.flux-schnell.json`, `__init__.py`, `store.py`, `settlement.py`, `stats.py`, `catalog_observations.py`, `catalog.py`, `test_aigc_pr_b.py`, `test_catalog_api.py`, `test_catalog_validate.py` |
|
||||
| `architecture/catalog.md` | `contracts.yaml`, `adapters.yaml`, `findymail.search.business-profile.json`, `__init__.py`, `contracts.py`, `paths.py`, `plan.py`, `synthetic.py`, `route.py`, `test_routing.py`, `catalog-drift.yml`, `catalog_drift.py`, `catalog_ingest.py`, `catalog_replicate_prices.py`, `catalog_validate.py`, `catalog_verify_extended.py`, `aliases.yaml`, `fx.yaml`, `aviato.yaml`, `crustdata.yaml`, `aviato.companies.acquisitions.json`, `aviato.companies.employees.json`, `aviato.companies.enrich.bulk.json`, `aviato.companies.enrich.json`, `aviato.companies.founders.json`, `aviato.companies.funding_rounds.json`, `aviato.companies.investments.json`, `aviato.companies.outbound_investments.json`, `aviato.companies.search.json`, `aviato.linkedin.company.posts.json`, `aviato.linkedin.post.comments.json`, `aviato.linkedin.post.reactions.json`, `aviato.linkedin.post.reposts.json`, `aviato.linkedin.user.posts.json`, `aviato.people.contact.get.json`, `aviato.people.email.find.json`, `aviato.people.enrich.bulk.json`, `aviato.people.enrich.json`, `aviato.people.phone.find.json`, `aviato.people.search.json`, `aviato.people.search.simple.json`, `crustdata.companies.autocomplete.json`, `crustdata.companies.enrich.json`, `crustdata.companies.identify.json`, `crustdata.companies.jobs.search.json`, `crustdata.companies.search.json`, `crustdata.people.autocomplete.json`, `crustdata.people.enrich.json`, `crustdata.people.search.json`, `google-search-console.yaml`, `google-search-console.extended.yaml`, `google-tag-manager.yaml`, `google-tag-manager.extended.yaml`, `instagram.yaml`, `instagram.extended.yaml`, `justoneapi.extended.yaml`, `minimax.yaml`, `apify.yaml`, `brightdata.yaml`, `companyenrich.yaml`, `oceanio.yaml`, `akta.extended.yaml`, `dataforseo.extended.yaml`, `tikhub.extended.yaml`, `minimax.video-gen.result.retrieve.json`, `minimax.video-gen.from_image.json`, `minimax.video-gen.task.status.json`, `openrouter.yaml`, `openrouter.extended.yaml`, `openrouter.x.alibaba-wan-3-0.json`, `replicate.yaml`, `replicate.extended.yaml`, `replicate.image-gen.flux-schnell.json`, `__init__.py`, `store.py`, `settlement.py`, `stats.py`, `catalog_observations.py`, `catalog.py`, `test_aigc_pr_b.py`, `test_catalog_api.py`, `test_catalog_validate.py` |
|
||||
| `architecture/composition.md` | `bootstrap.py`, `bootstrap_handlers.py`, `bootstrap_http.py`, `call_surface.py`, `connect.py`, `mcp_oauth.py`, `session.py`, `admin.py`, `auth.py`, `billing.py`, `call.py`, `connections.py`, `onboard.py`, `orgs.py`, `resources.py`, `referrals.py`, `web.py`, `dump_surface.py`, `test_app_roles.py` |
|
||||
| `architecture/data-model.md` | `alembic.ini`, `env.py`, `0001_baseline_current_schema.py`, `0002_archive_tables.py`, `0003_callrecord_cached.py`, `0004_archivekey_request_shape.py`, `0005_capacity_policy_snapshot.py`, `0006_overflow_route.py`, `0007_overflow_spend.py`, `0008_org_platform_overflow_disabled.py`, `0009_callrecord_hit.py`, `0017_async_task_record.py`, `0018_async_resource_ownership.py`, `0019_async_poll_failures.py`, `0011_callrecord_archive_link.py`, `0015_idempotentcall_membership_cascade.py`, `maintenance.py`, `sitetrack.js`, `models.py`, `timeutil.py`, `db.py`, `referrals.py`, `audit.py`, `analytics.py`, `bootstrap_handlers.py`, `ratestore.py`, `auth.py`, `test_postgres_reset.py`, `test_alembic_expand_safety.py` |
|
||||
| `architecture/data-model.md` | `alembic.ini`, `env.py`, `0001_baseline_current_schema.py`, `0002_archive_tables.py`, `0003_callrecord_cached.py`, `0004_archivekey_request_shape.py`, `0005_capacity_policy_snapshot.py`, `0006_overflow_route.py`, `0007_overflow_spend.py`, `0008_org_platform_overflow_disabled.py`, `0009_callrecord_hit.py`, `0017_async_task_record.py`, `0018_async_resource_ownership.py`, `0019_async_poll_failures.py`, `0020_callrecord_created_at_indexes.py`, `0021_ledgerentry_org_created_at_index.py`, `0022_org_spent_today_counter.py`, `0023_callrecord_org_user_created_at_index.py`, `0024_membership_calls_today_counter.py`, `0011_callrecord_archive_link.py`, `0015_idempotentcall_membership_cascade.py`, `maintenance.py`, `sitetrack.js`, `models.py`, `timeutil.py`, `db.py`, `referrals.py`, `audit.py`, `analytics.py`, `bootstrap_handlers.py`, `ratestore.py`, `auth.py`, `test_postgres_reset.py`, `test_alembic_expand_safety.py` |
|
||||
| `architecture/import-boundaries.md` | `pyproject.toml`, `ci.yml`, `__init__.py`, `__init__.py`, `access.py`, `authorize.py`, `idempotency.py`, `overflow.py`, `route.py`, `__init__.py`, `intake.py`, `resolve.py`, `reserve.py`, `settle.py`, `evidence.py`, `service.py`, `types.py`, `client_identity.py`, `__init__.py`, `__init__.py`, `access.py`, `budgets.py`, `publicdemo.py`, `teams.py`, `usage.py`, `__init__.py`, `__init__.py`, `authorization.py`, `oauth_flow.py`, `refresh.py`, `__init__.py`, `__init__.py`, `__init__.py`, `__init__.py`, `injectors.py`, `relay.py`, `__init__.py`, `limiter.py`, `test_call_architecture.py`, `test_import_lightness.py` |
|
||||
| `architecture/instagram-oauth.md` | `catalog_ingest.py`, `access.py`, `resolve.py`, `service.py`, `instagram.yaml`, `instagram.extended.yaml`, `cli.py`, `store.py`, `authorization.py`, `oauth_flow.py`, `oauth_exchange.py`, `mcp.py`, `call.py`, `index.html`, `0010_oauth_authorization_method.py`, `test_instagram_oauth_architecture.py` |
|
||||
| `architecture/local-proxy.md` | `localproxy.py`, `server.js` |
|
||||
|
||||
@@ -72,14 +72,15 @@ agents then built against a constitution that was wrong.
|
||||
commit by design; a few other domain commits remain. Do not add another; move one out when you
|
||||
touch it.
|
||||
- **Table ownership.** One writer module per table; cross-domain reads are fine. Three recorded
|
||||
exceptions: only money writes `org.balance_micro` and the auto-top-up fields; the call runtime
|
||||
may persist an OAuth token refresh into `secret`; audit writes `callrecord`, domains only read it.
|
||||
exceptions: only money writes `org.balance_micro`, the daily-spend counter (`spent_today_*`) and
|
||||
the auto-top-up fields; the call runtime may persist an OAuth token refresh into `secret`; audit
|
||||
writes `callrecord`, domains only read it.
|
||||
- **The call runtime is self-contained.** `src/treg/application/call/` depends on no management
|
||||
code (routes, login, OAuth consent, Stripe top-up), reads only membership, deny rules,
|
||||
credentials, catalog prices and balances, and writes only what `tests/test_call_architecture.py`
|
||||
allowlists (the ledger entries, idempotency claims, OAuth refresh, audit and telemetry, first-call
|
||||
markers, tag budgets, capacity marks, overflow spend). Extend the test's allowlist in the same PR
|
||||
as any new write, and expect the reviewer to ask why.
|
||||
markers, tag budgets, capacity marks, overflow spend, the member's daily-cap slot). Extend the
|
||||
test's allowlist in the same PR as any new write, and expect the reviewer to ask why.
|
||||
- **Money.** Everything is **integer micro-USD** - never floats, never cents. The Stripe SDK lives
|
||||
only in `infra/stripe.py`, orchestration in `application/billing.py`, and `reconcile.py` is
|
||||
read-only. See `docs/context/architecture/money.md`.
|
||||
|
||||
@@ -298,6 +298,7 @@ Environment variables (prefix `TREG_`, read from `.env`):
|
||||
| `TREG_META_CLIENT_ID` / `_SECRET` | *(empty)* | Meta app credentials for Facebook Pages, Meta Ads, and optional Instagram `page-tools` |
|
||||
| `TREG_OAUTH_REVIEW_PENDING` | `instagram-login,page-messages` | Registry review keys awaiting production access. Remove `page-messages` after Page messaging approval; set empty after direct Instagram approval. |
|
||||
| `TREG_RESEND_API_KEY` / `TREG_EMAIL_FROM` | *(empty)* | transactional email via Resend (OTP codes + invites); From must be a Resend-verified sender |
|
||||
| `TREG_BLOCKED_EMAIL_DOMAINS` | *(empty)* | comma-separated email domains added to the built-in throwaway/farm blocklist, refused at every sign-up/sign-in door and at team creation (subdomains included, case-insensitive) |
|
||||
| `TREG_ADMIN_TOKEN` | *(empty)* | cross-tenant **super-admin** bearer; authorizes every `/admin/*` endpoint. Empty disables the env path (only `is_superadmin` users reach `/admin`). Keep it long + secret. |
|
||||
| `TREG_EMAIL_DEV_MODE` | `false` | when true, `/auth/email/start` returns the OTP in its response (no mail sender needed) — **dev/local only**, never in prod. |
|
||||
|
||||
|
||||
@@ -30,6 +30,13 @@ environment) and enforcement happens server-side and in the operating system.
|
||||
- **Server runs are resource-limited.** `treg run --server` executes each CLI with a scrubbed environment
|
||||
(treg's own secrets removed), a per-run throwaway home, an allow-list of runnable commands, output
|
||||
redaction, and POSIX resource limits (CPU, file size, no core dumps).
|
||||
- **Signup abuse has a brake.** Every new team gets a small promotional balance, which makes
|
||||
throwaway-email farming worth an attacker's time. An email-domain blocklist (throwaway-mail rules
|
||||
and confirmed farm roots in code, plus `TREG_BLOCKED_EMAIL_DOMAINS` for a new root without a
|
||||
redeploy; subdomains included, case-insensitive, domain only) refuses the address at every sign-up
|
||||
and sign-in door and at both team-creating endpoints. The refusal names neither the list nor the
|
||||
domain, every block is logged, and a classifier failure lets the sign-in through rather than
|
||||
breaking real signups.
|
||||
|
||||
## Known limitations (by design, documented on purpose)
|
||||
|
||||
|
||||
@@ -287,20 +287,27 @@ worker whenever the archive records, `archive_prune_batch` (500) bodies per pass
|
||||
`archive_prune_interval_s` (3600); batch 0 disables. Rollup counters (bodies_kept, kept_bytes)
|
||||
move atomically with each strip.
|
||||
|
||||
## Recorder throttle (2026-09-03)
|
||||
## Recorder throttle (2026-09-03, memory-bounded 2026-09-07)
|
||||
|
||||
At most `_MAX_CONCURRENT_WRITES` (4) recordings touch the database at once — audit's exact
|
||||
loop-bound-semaphore pattern. Before it, a burst could put up to 512 concurrent short sessions in
|
||||
front of the API's 15-slot pool (SToneX's pool-pressure report); those writes now land on the
|
||||
BACKGROUND pool instead (`ops/deploy.md` § Three pools), so the semaphore is the inner bound rather
|
||||
than the only one — a third module reaching for the wrong maker no longer needs its author to have
|
||||
read this section. Queued recordings wait inside
|
||||
their fire-and-forget task, so the caller is unaffected; the 30s bound covers wait+write, so a
|
||||
stuck queue still sheds rather than wedges. Throttled, not shed: the burst test proves all 12
|
||||
concurrent recordings land while peak DB concurrency stays ≤4.
|
||||
At most `_MAX_CONCURRENT_WRITES` (2) recordings touch the database at once — audit's exact
|
||||
loop-bound-semaphore pattern (four until 2026-09-07; every slot is paid per uvicorn worker and
|
||||
again per rolling-deploy instance, and a recording is one INSERT of a body already in memory).
|
||||
Before it, a burst could put up to 512 concurrent short sessions in front of the API's 15-slot pool
|
||||
(SToneX's pool-pressure report); those writes now land on the BACKGROUND pool instead
|
||||
(`ops/deploy.md` § Three pools), so the semaphore is the inner bound rather than the only one.
|
||||
Queued recordings wait inside their fire-and-forget task, so the caller is unaffected; the 30s
|
||||
bound covers wait+write, so a stuck queue still sheds rather than wedges. Throttled, not shed: the
|
||||
burst test proves all 12 concurrent recordings land while peak DB concurrency stays ≤2.
|
||||
|
||||
**Memory bound (2026-09-07 OOM fix).** Each pending task holds its `body` bytes in a closure — up to
|
||||
`_MAX_PENDING` (512) tasks × `archive_max_body_bytes` (2 MB) = 1 GB worst case. After #363 reduced
|
||||
concurrent writes from 4 to 2, backlog built faster under heavy traffic and the 2026-09-07T00:43:06Z
|
||||
OOM killed production at 4 GB. `_MAX_PENDING_BYTES` (256 MB) now caps total body bytes held by
|
||||
pending work: `record()` sheds when EITHER the task count OR the bytes threshold is exceeded. The
|
||||
done callback releases bytes when a task completes, keeping the budget accurate.
|
||||
|
||||
The semaphore is process-local, while production runs multiple processes. An exact in-process key
|
||||
lock is acquired before the semaphore, so duplicate recordings queue without consuming all four
|
||||
lock is acquired before the semaphore, so duplicate recordings queue without consuming both
|
||||
database-write slots and unrelated keys keep moving; weak references discard inactive locks. Once
|
||||
admitted, the writer locks and refreshes the matching `ArchiveKey` row before reading the newest
|
||||
snapshot and allocating version N+1. The refresh matters because the earlier unlocked lookup
|
||||
|
||||
@@ -17,6 +17,10 @@ sources:
|
||||
- src/treg/alembic/versions/0018_async_resource_ownership.py
|
||||
- src/treg/alembic/versions/0019_async_poll_failures.py
|
||||
- src/treg/alembic/versions/0020_callrecord_created_at_indexes.py
|
||||
- src/treg/alembic/versions/0021_ledgerentry_org_created_at_index.py
|
||||
- src/treg/alembic/versions/0022_org_spent_today_counter.py
|
||||
- src/treg/alembic/versions/0023_callrecord_org_user_created_at_index.py
|
||||
- src/treg/alembic/versions/0024_membership_calls_today_counter.py
|
||||
- src/treg/alembic/versions/0011_callrecord_archive_link.py
|
||||
- src/treg/alembic/versions/0015_idempotentcall_membership_cascade.py
|
||||
- src/treg/maintenance.py
|
||||
@@ -92,8 +96,10 @@ uses this metadata, never the encrypted token's shape.
|
||||
- **`Membership`** - links a user to an org: `user_id`, `org_id`, `role` (owner|admin|member),
|
||||
`token_hash` (SHA-256 of the bearer token, shown once), `webhook_url` (health alerts POST here),
|
||||
`daily_call_cap` (per-user, per-day usage cap; **-1 = unlimited**, the default - see
|
||||
`api._enforce_daily_cap`); unique `(user_id, org_id)`. **A token = a `(user, org)` pair.** `ROLE_RANK`
|
||||
orders the roles.
|
||||
`governance/usage.enforce_daily_cap`) with `calls_today` / `calls_today_day`, the counter that cap
|
||||
is checked against (one conditional UPDATE per capped event, revision 0024; only capped members are
|
||||
counted, the roster reads the journal); unique `(user_id, org_id)`. **A token = a `(user, org)`
|
||||
pair.** `ROLE_RANK` orders the roles.
|
||||
- **`Invite`** - a one-time join code: `org_id, email, role, code_hash (idx), status`
|
||||
(pending|accepted|revoked), `invited_by`. Carries a SECOND split secret, `email_token_hash (idx,
|
||||
nullable)` - the inbox-only sign-in token embedded ONLY in the invite email's link (the
|
||||
@@ -152,6 +158,27 @@ uses this metadata, never the encrypted token's shape.
|
||||
here queues every other query and the API pool empties into `503 treg_saturated` - see
|
||||
[deploy](../ops/deploy.md) § Three pools. The table has no retention sweep yet, so it only grows.
|
||||
|
||||
**`LedgerEntry` is the other one, and it was the larger.** It is append-only and never pruned
|
||||
(4.38M rows / 2.3 GB on prod 2026-09-06, ~400k rows a day), and `ledger.spent_today` - the
|
||||
fail-closed daily cap - reads it on EVERY metered call, inside the reserve transaction, on an
|
||||
api-pool connection. With only single-column indexes the planner walked the whole platform's day
|
||||
through `ix_ledgerentry_created_at` and filtered the org in memory: 322k rows discarded and 381k
|
||||
buffer touches per call, 56-106 s once the day's pages had been evicted from a 512 MB cache, and
|
||||
the heap had read 6.5 BILLION blocks - four times `callrecord`. Revision 0021 adds
|
||||
`(org_id, created_at)`, which also serves `entries_of` (the `/billing` page, previously a
|
||||
backward walk of the whole `created_at` index). That fixed light orgs and `/billing` but not the
|
||||
two orgs writing half the day - their rows are on every page of the day, and the planner kept
|
||||
walking it (395k buffer touches per call after 0021). So the cap no longer reads this table at
|
||||
all: revision 0022 adds `Org.spent_today_micro` / `spent_today_day`, kept by `domain/money`
|
||||
inside the balance UPDATE and read with one primary-key lookup; the journal aggregate survives
|
||||
as `spent_today_from_ledger` for reconciliation. The same shape on `callrecord` - the per-user
|
||||
daily call cap, `count_today`, which BitmapAnd-ed a member's whole history through
|
||||
`ix_callrecord_user_email` (2.6 s of 3.0 s for a 287k-row member) - gets
|
||||
`(org_id, user_email, created_at)` in revision 0023, and then the same answer as the ledger: the
|
||||
index-only scan still fetched the heap for today's not-yet-vacuumed pages (110k heap fetches,
|
||||
2.8 s), so revision 0024 moves the gate to `Membership.calls_today` and the journal count is
|
||||
left to the roster and `/usage/me`.
|
||||
|
||||
`refused_by` distinguishes a treg refusal (`auth`, `policy`, `balance`, `cap`, `resolution`,
|
||||
`request`, and other mechanism-specific values) from an upstream answer, where it is null.
|
||||
In-handler audits mark their own outcome; the shared exception handler records earlier
|
||||
@@ -374,7 +401,7 @@ key, including first sightings that never recurred to carry their own count out.
|
||||
the shared queue is genuinely backing up (`_FAULT_QUEUE_SHARE` of `_MAX_PENDING`) - the congestion the
|
||||
throttle was ever meant to prevent, rather than a wall-clock rate that fired against an empty queue.
|
||||
|
||||
**Losing data is ERROR, not WARNING.** `audit._write`, `audit._schedule`'s back-pressure shed, and
|
||||
**Losing data is ERROR, not WARNING.** `audit._write_batch`, `audit._enqueue`'s back-pressure shed, and
|
||||
`archive`'s `_store`/`_touch_write` drops all log at ERROR, because `FaultCaptureHandler` starts at ERROR:
|
||||
below it the loss reaches container stdout and nothing else, so it can neither be alerted on nor found
|
||||
without already suspecting it. Degradations that cost nothing (an archive lookup falling back to a live
|
||||
|
||||
@@ -623,9 +623,12 @@ a compliant figure and together exceed the cap. Overshoot is bounded by `concurr
|
||||
estimate`, and that is acceptable **only** because the hard gates sit behind it - the org balance and
|
||||
the per-org daily cap.
|
||||
|
||||
Making it exact would need a second materialized authority on spend: reset daily, decremented on
|
||||
release, corrected on settle divergence. Four new ways to disagree with `domain/money`, which is the one
|
||||
module allowed to move money. Not worth it. Never document these caps to builders as hard limits.
|
||||
Making it exact would need a second materialized authority on spend per tag: reset daily, decremented
|
||||
on release, corrected on settle divergence. Four new ways to disagree with `domain/money`, which is the
|
||||
one module allowed to move money. Not worth it for tag caps. (The per-org daily cap DOES have exactly
|
||||
such a counter, `Org.spent_today_micro`, since 2026-09-06 - kept by `domain/money` itself, inside the
|
||||
balance UPDATE, so there is no second writer to disagree with; see below.) Never document these caps
|
||||
to builders as hard limits.
|
||||
|
||||
### Refusal bodies are not the org's
|
||||
|
||||
@@ -661,6 +664,24 @@ and the deployment's `platform_daily_cap_usd` ceiling (default $500/day). The te
|
||||
limit and inspect it through `GET /orgs/{id}/settings`. A request above the platform ceiling is
|
||||
refused, not silently clamped.
|
||||
|
||||
The check itself, `ledger.spent_today`, is the most-run query on the platform: every metered call,
|
||||
inside the reserve transaction, on an api-pool connection, fail-closed. Its cost is therefore the
|
||||
platform's throughput, and it is ONE primary-key read of `Org.spent_today_micro` /
|
||||
`spent_today_day` (revision 0022). `domain/money` keeps that counter inside the same UPDATE that
|
||||
moves the balance: reserve adds the charged estimate, settle adds what was consumed and removes
|
||||
the estimate it replaces, release removes the estimate - in each case only when the hold was
|
||||
opened today, because a hold opened yesterday was yesterday's commitment. The first movement of a
|
||||
new UTC day resets the counter to that movement (`_spent_today_values`, one CASE expression).
|
||||
`spent_today_from_ledger` computes the same number from the journal over `(org_id, created_at)`
|
||||
(revision 0021) for reconciliation; `tests/test_daily_spend_counter.py` asserts the two agree
|
||||
through reserve, settle (under and over the estimate), release and the day boundary.
|
||||
|
||||
Why a counter and not an index: until 2026-09-06 the check was that journal aggregate, and for an
|
||||
org that writes a large share of the platform's day its rows sit on nearly every heap page of the
|
||||
day, so no index makes the aggregate cheaper than reading the day - measured 395k buffer touches
|
||||
per call, 56-171 s once those pages were cold, holding an api-pool slot throughout. That was the
|
||||
API-pool saturation (see [deploy](../ops/deploy.md) § Three pools).
|
||||
|
||||
## Referrals
|
||||
|
||||
`domain/referrals.py` owns policy; credit moves only through `ledger.grant`. Rewards are flat,
|
||||
|
||||
@@ -114,6 +114,41 @@ pair, so every list/create/mutation and the proxy are scoped to the caller's org
|
||||
Every identity door is blocked at the shared choke point `_find_or_create_user`, plus `register_user`
|
||||
(which predates it and creates a `User` directly) and `auth_email_start` (refuse early, mint no code).
|
||||
`list_members` carries `is_agent` so one roster can show people and machines apart.
|
||||
- **Email-domain blocklist.** The same choke points, for throwaway mail and domains used for bulk
|
||||
registration. A new team is created with a promotional balance, which is what makes registering in
|
||||
bulk on throwaway addresses worth someone's while. **Two tiers, one classifier**
|
||||
(`_is_blocked_email` in `domain/identity/access.py`, pure: it only answers). The CODE tier is
|
||||
`BLOCKED_EMAIL_DOMAINS` (domains confirmed abusive in our own data, and `my.id` so every free
|
||||
`.my.id` subdomain falls to the walk) plus `BLOCKED_EMAIL_KEYWORDS`, substring rules on the domain
|
||||
(`tempmail`, `mailinator`, `guerrilla`, `10minute`, ...) that catch domains no static list has
|
||||
seen. The OPS tier is `TREG_BLOCKED_EMAIL_DOMAINS`, comma-separated, **added to** the code tier
|
||||
and parsed once per distinct value in `config.py` (trim, drop a leading `@`/`.`, lowercase, and
|
||||
drop any dotless entry so a typed `com` cannot refuse the world): the next domain is a **dashboard
|
||||
edit, no redeploy**. The rules, each of which exists because the obvious implementation is wrong:
|
||||
match the **domain only**, never the whole address (matching the address false-flags real users
|
||||
whose username happens to contain a keyword); **walk parent domains**, whole labels off the front
|
||||
and never the bare last label, because registering `<random>.<blocked-root>` is otherwise a
|
||||
one-line bypass; **sign-in as well as sign-up** (an account that predates the listing gets no
|
||||
new session; existing accounts are suspended out of band). The DECISION lives in the application
|
||||
layer, `signup.blocked_email(email, door)`: it refuses, writes one structured line per block
|
||||
(`event=signup_blocked_domain door=<door> domain=<domain>` — the refusal reveals nothing, so the
|
||||
log is the only detection a burst has), and **fails open**, logging `event=blocklist_error`
|
||||
and letting the sign-in through if the classifier ever raises, because a misconfiguration must
|
||||
never break a real sign-in. The doors: `start_email_login` (before the rate window, so no code and
|
||||
no mail), `find_or_create_user` (so OTP verify, the GitHub and Google callbacks and the emailed
|
||||
invite link `POST /auth/invite-signin` refuse before the row lookup, raising
|
||||
`signup.BlockedEmailError` which each door translates to a `blocked_domain` kind), `register_user`
|
||||
(`POST /users` mints user + team + promo in one call), `create_org` (`POST /orgs`, the other promo
|
||||
door, reachable with a token minted before the listing) and the code-based `POST /invites/accept`
|
||||
(which constructs a `User` directly, so it guards itself). Every refusal is the `machine_identity`
|
||||
sibling's exact 403 `this address cannot be used to sign in` (a brand page on the browser doors,
|
||||
like `suspended`): the caller learns neither that a list exists nor what is on it. Deliberately a
|
||||
blocklist and nothing more: no allowlist, no table, no admin UI. Not covered: a session or identity
|
||||
token already live when the domain was listed keeps working until suspension or expiry (the
|
||||
out-of-band suspension); the promo grant and referral bonus are not separately gated, since with
|
||||
the doors closed no promo-funded team on a blocked domain can come into existence; and vendoring a
|
||||
full public disposable-domain list is a follow-up (megabytes of package data in the base wheel,
|
||||
which also ships the light CLI, and not yet checked against real users).
|
||||
**A rotate replaces the TOKEN, never the limits.** Because rotate is the same endpoint as create, an
|
||||
absent optional field used to fall back to its permissive default — and the dashboard's Rotate button
|
||||
sends only `{name, role, daily_call_cap}`, so a scoped agent silently became unrestricted
|
||||
|
||||
@@ -246,10 +246,17 @@ validated before resolving the shared HTTP client. `/auth/logout` remains an HTT
|
||||
(`PATCH /orgs/{id}/members/{user}/cap`, admin+) sets `Membership.daily_call_cap` (`-1` = unlimited,
|
||||
rejects `< -1`). `my_usage` (`GET /usage/me`, any member) returns the caller's own `used_today` + `cap`.
|
||||
`list_members` also returns each member's `daily_call_cap` + `used_today`. **Enforcement:**
|
||||
`_enforce_daily_cap` runs at the top of `call_tool`, `run_tool_server`, and `grant_local_run` (so no
|
||||
path dodges the cap); `count_today` = today's `CallRecord` + `RunRecord` for the user. `-1` (default)
|
||||
skips the count entirely (zero overhead); the sandbox is exempt. **Soft by design** - it counts the
|
||||
best-effort `CallRecord`, so under load it fails *open*, never closed.
|
||||
`governance/usage.enforce_daily_cap` runs in `/call/`'s authorization gate and, through
|
||||
`routers/call._enforce_daily_cap`, at the top of `run_tool_server` and `grant_local_run` (so no path
|
||||
dodges the cap). It takes the slot with ONE conditional UPDATE of `Membership.calls_today` /
|
||||
`calls_today_day` (`take_daily_slot`, revision 0024): the WHERE is the check and the SET is the
|
||||
count, so the cap is exact under concurrency, a refused event is not counted, and the first event of
|
||||
a new UTC day starts from 1. `-1` (default) skips it entirely (zero overhead); the sandbox is
|
||||
exempt; a database error fails *open* (a courtesy limit, not a money gate). `used_today` in the
|
||||
roster and `/usage/me` still comes from the journal (`count_today` = today's `CallRecord` +
|
||||
`RunRecord`), and `set_member_cap` copies that journal count onto the counter when a member goes
|
||||
from unlimited to capped, so a cap set mid-day does not start from zero. Until 2026-09-06 the gate
|
||||
itself ran that journal count - 2.8 s per call for a member with 110k rows that day.
|
||||
- **Super-admin (cross-tenant, `require_superadmin`):** `/admin/stats|orgs|orgs/{id}|users|tools|calls|
|
||||
errors|health` (reads - `errors` is failed calls across every credential tier with captured,
|
||||
admin-only request/response evidence, supports a `tier` filter, and runs the 14-day retention pass;
|
||||
@@ -356,7 +363,12 @@ validated before resolving the shared HTTP client. `/auth/logout` remains an HTT
|
||||
|
||||
- **Identity doors:** GitHub, Google and email OTP share first-proof user provisioning. They create
|
||||
a user without an automatic org; new users name their first team through onboarding or the CLI
|
||||
login picker. Suspended users are refused at every door.
|
||||
login picker. Suspended users are refused at every door, and so is any address on a blocked email
|
||||
domain (throwaway-mail rules and confirmed farm roots in code, plus `TREG_BLOCKED_EMAIL_DOMAINS`;
|
||||
subdomains included): the OTP start and verify, both social callbacks, the emailed invite link,
|
||||
plus `POST /users`, `POST /orgs` and `POST /invites/accept`, all with the same 403 `this address
|
||||
cannot be used to sign in` the machine-identity guard uses. See
|
||||
[multi-tenancy](../architecture/multi-tenancy.md).
|
||||
|
||||
- GitHub/Google: `GET /auth/{provider}[/callback]`, optional `?cli=<id>`; callbacks validate
|
||||
state before resolving the shared HTTP client, require a proven email, and set the session
|
||||
|
||||
+101
-12
@@ -72,7 +72,7 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
|
||||
|---|---|---|---|
|
||||
| `api` | `session_maker` | 5 + 10 | every request handler, via `get_session` or directly |
|
||||
| `admin` | `admin_session_maker` | 3 + **0** | `/admin/*` only, via `get_admin_session` |
|
||||
| `background` | `background_session_maker` | 13 + **0** | audit, archive writes, ads worker, the observation reader, the error-evidence sweep |
|
||||
| `background` | `background_session_maker` | 8 + **0** | audit (one batching writer), archive writes (two), ads worker, the observation reader, the error-evidence sweep |
|
||||
|
||||
Each class of work can exhaust only its own slots. Before this there was ONE pool of 15, and on
|
||||
2026-09-03 a single admin browser tab polling `/admin/archive/panel` (every 5 s, no in-flight
|
||||
@@ -82,6 +82,31 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
|
||||
(`audit.py` did, `archive.py` did not), while a pool bounds every module routed to it. Overflow
|
||||
is **0** on both minor pools for the same reason: it is the escape hatch a bulkhead must not have.
|
||||
|
||||
**Every number above is PER PROCESS, and the reference deployment runs two.** Render sets
|
||||
`WEB_CONCURRENCY=2` on the web service's 2c-4g plan (not a dashboard variable - injected at
|
||||
runtime, and absent on the crons) and uvicorn honors it: the boot log shows `Started parent
|
||||
process` then two `Started server process` lines. Each worker opens its own three pools, runs its
|
||||
own copy of every in-process background task (ads, archive refresh, prune, the gauge), and a
|
||||
rolling deploy runs two instances for about a minute. So the budget is
|
||||
`per_process × 2 workers × 2 instances` against `max_connections` (103 on the 1c-2g plan), and
|
||||
`infra/db.connection_budget` logs it at boot:
|
||||
|
||||
| specs | per process | per instance | deploy peak | 103? |
|
||||
|---|---|---|---|---|
|
||||
| code defaults 15 + 3 + 13 (until 2026-09-07) | 31 | 62 | **124** | over |
|
||||
| code defaults 15 + 3 + 8 (since 2026-09-07) | 26 | 52 | 104 | over by one |
|
||||
| dashboard override 15 + 2 + 4 | 21 | 42 | 84 | fits |
|
||||
|
||||
That is the post-mortem of the 2026-09-04 defaults: `background = 13` did not overload the
|
||||
database, it opened 124 connections at every deploy and restart until the override cut it to 84.
|
||||
Every earlier passage in this file that multiplied by two instances only was counting half the
|
||||
connections. Any resize must clear the deploy-peak column first; within it there are 2 spare
|
||||
per process today (23 → 92). Batching the audit writer (4 → 1) and halving the archive semaphore
|
||||
(4 → 2) on 2026-09-07 cut the derived `background` from 13 to 8, so a pool of 6 now serves every
|
||||
consumer but two archive writers at once and still fits (15 + 2 + 6 = 23 → 92); the way to more
|
||||
is `WEB_CONCURRENCY=1`, a larger database plan, or a pooler -
|
||||
not a bigger number in the override.
|
||||
|
||||
Two sizing rules, both learned by getting them wrong first:
|
||||
|
||||
- **`background` is derived, not chosen.** `BACKGROUND_CONSUMERS` lists everything that can hold
|
||||
@@ -116,11 +141,36 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
|
||||
deployment ran `admin.pool_size=2,background.pool_size=4` from before the 2026-09-04 bulkhead
|
||||
work until 2026-09-05 — pinning both minor pools BELOW the defaults that work had just raised
|
||||
(`admin` to 3, `background` to the derived 13), including the exact `admin=2` whose post-mortem
|
||||
is two bullets up. A `background` of 4 against 7 consumers needing 13 does not 503; it silently
|
||||
drops audit rows. The knob being a dashboard edit rather than a deploy is what makes it useful
|
||||
is two bullets up. A `background` of 4 against 7 consumers needing 13 (8 since 2026-09-07) does
|
||||
not 503; it silently drops audit rows. The knob being a dashboard edit rather than a deploy is what makes it useful
|
||||
mid-incident and what lets it survive the fix. Today it reads
|
||||
`api.pool_size=10,admin.pool_size=2,background.pool_size=4`; the two stale entries are still
|
||||
there deliberately, so that the `api` change could be observed on its own.
|
||||
`admin.pool_size=2,background.pool_size=4` - and those two entries are no longer "stale": with
|
||||
two uvicorn workers (§ above) they are what keeps a rolling deploy at 84 connections instead of
|
||||
124, so removing them is not a cleanup, it is the 2026-09-04 outage again. `api.pool_size=10` was
|
||||
added on 2026-09-05 and removed on 2026-09-06: against a database that is waiting on DISK (below),
|
||||
five more slots meant five more readers of the same cold pages, and the worst hour on record
|
||||
(2,136 pool faults at 11:00, on a third of the previous day's traffic) followed.
|
||||
- **The pools are measured, not argued about: `db_pool_gauge`.** `bootstrap.pool_gauge` samples
|
||||
`infra/db.pool_snapshot()` once a second and emits one PostHog event a minute per instance:
|
||||
`<pool>_peak` (most connections that pool had checked out in the minute), `<pool>_capacity`
|
||||
(`pool_size + max_overflow`) and `<pool>_headroom`. Telemetry, not a database consumer, so it is
|
||||
not in `ROLE_BACKGROUND_TASKS` and runs in every role. Read it like this: a pool whose peak sits
|
||||
at capacity is one whose waiters are timing out (`api`: `503 treg_saturated`; `background`: an
|
||||
audit or archive row dropped after `pool_timeout`); a pool whose peak never nears capacity is
|
||||
holding connections nothing uses. **Resize from the gauge, never from the arithmetic** - the
|
||||
arithmetic got both minor pools wrong once each (above), and the 2026-09-05 `api` raise made the
|
||||
saturation it meant to fix worse. The protocol: one pool at a time, one override at a time, each
|
||||
setting across at least one full daily peak (the 01:00-04:00 UTC batch window), judged by the same
|
||||
hour on consecutive days on three numbers - db_pool faults, `/call/` 503 rate, and the gap between
|
||||
`tool_called` events and `callrecord` rows (dropped audit). A change that raises the 503 rate at
|
||||
equal traffic is reverted, not tuned around.
|
||||
|
||||
```
|
||||
SELECT toStartOfHour(timestamp) h, max(toFloat(properties.background_peak)) bg_peak,
|
||||
any(properties.background_capacity) bg_cap, max(toFloat(properties.api_peak)) api_peak
|
||||
FROM events WHERE event = 'db_pool_gauge' AND timestamp > now() - INTERVAL 2 DAY
|
||||
GROUP BY h ORDER BY h DESC
|
||||
```
|
||||
- **No statement timeout yet.** The pools bound how many connections a class of work can hold, not
|
||||
how long a query may run; `alembic/env.py` still has the only timeouts in the app. Adding per-pool
|
||||
`statement_timeout` is deliberately a SEPARATE change: it is a behavior change on every query,
|
||||
@@ -147,6 +197,23 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
|
||||
through `ix_callrecord_endpoint_id_id` because no index carried `created_at` (revision 0020
|
||||
adds the pairs). **Reach for a pool size only after ruling out a scan;** raising it buys headroom
|
||||
and hides the cause.
|
||||
|
||||
That diagnosis was half right. Re-measured 2026-09-06 with wait events instead of response
|
||||
times: 88 % of active backends sat in `IO DataFileRead` / `IPC BufferIO` (waiting for a page, or
|
||||
for ANOTHER backend reading the same page), 18 of 879 samples were on CPU. The database is not
|
||||
CPU-bound; it is a 512 MB buffer cache in front of 35 GB, and the query holding the pool was
|
||||
not on `callrecord` at all: 674 of 879 active samples were `ledger.spent_today` on
|
||||
`ledgerentry`, the fail-closed daily cap that runs inside EVERY metered call's reserve
|
||||
transaction on an api-pool connection, scanning the whole platform's day because no index paired
|
||||
`org_id` with `created_at` (revision 0021 adds it; `ledgerentry` had read 6.5 BILLION heap
|
||||
blocks, four times `callrecord`). Whenever a large scan evicts the day's ledger pages - the
|
||||
30-day observation refresh, the `/billing` page's backward index walk, the per-call
|
||||
`idempotentcall` sweep, a concurrent index build - every in-flight `spent_today` stalls together
|
||||
for tens of seconds, and 20 slots are gone. 0021 fixed light orgs and `/billing` only: the two
|
||||
orgs writing half the day sit on every page of the day and the planner kept walking it, so
|
||||
revision 0022 moved the cap to a counter on the org row (one primary-key read) and 0023 gave the
|
||||
per-user cap its triple on `callrecord`. Two lessons: **sample `wait_event_type`, not
|
||||
latency**, and on a disk-bound database a bigger pool is more contention, not more throughput.
|
||||
- **SQLite aliases all three to one engine.** It has no pool to protect and file-level write locks
|
||||
it cannot share, so three engines against one file would only manufacture "database is locked".
|
||||
Tests therefore pin the ROUTING (which maker each module reaches for), not the isolation.
|
||||
@@ -241,6 +308,18 @@ bound to a closed maintenance loop. Calling `maintenance.upgrade()` directly doe
|
||||
code is exposed only through `Settings.expose_dev_code`, which requires `email_dev_mode` **and** a
|
||||
**local sqlite** `database_url` — so even a stray `TREG_EMAIL_DEV_MODE=true` on Postgres (a real deploy)
|
||||
can never leak a login code.
|
||||
- `blocked_email_domains` (`TREG_BLOCKED_EMAIL_DOMAINS`, default empty) - the OPS tier of the
|
||||
email-domain blocklist: comma-separated domains ADDED to the code tier (treg's confirmed farm
|
||||
roots and the throwaway-mail keyword rules in `domain/identity/access.py`), refused at every
|
||||
identity door and at both team-creating doors (`POST /users` and `POST /orgs`). Example:
|
||||
`newfarm.io,other-farm.net`. Case-insensitive; a listed domain also blocks its subdomains; a
|
||||
leading `@` or `.` and surrounding whitespace are tolerated; a dotless entry (`com`) is ignored.
|
||||
The signup-grant-farm brake: edit it in the Render dashboard the moment a new root appears, no
|
||||
redeploy; promote a root into the code tier in the next PR. Empty adds nothing (the code tier
|
||||
stays in force). Existing accounts on a listed domain must be suspended separately (`/admin`);
|
||||
the list only stops new sessions and new teams. Each block writes one
|
||||
`event=signup_blocked_domain door=... domain=...` log line, so a wave is countable. See
|
||||
[multi-tenancy](../architecture/multi-tenancy.md).
|
||||
- `run_proof` (`TREG_RUN_PROOF`) — the **isolated-runner proof** for `treg run --local`. A local run whose
|
||||
grant would return a secret the caller does **not** own (a shared-key tool a member may run but not read)
|
||||
must present this value in the `X-Treg-Run-Proof` header — a value held **only** by the root-installed
|
||||
@@ -317,13 +396,15 @@ UTC (SQLite is lax and hid this; it only bites on Postgres — the deploy target
|
||||
and in the serial Postgres CI migration set. `env.py` bounds Postgres lock and statement wait time so
|
||||
a contended migration fails before it queues the serving database behind DDL.
|
||||
|
||||
**Audit back-pressure (`audit.py`).** Audit rows are written off the request path (fire-and-forget), and
|
||||
each write opens a DB connection — from the **background** pool since 2026-09-03, so a burst here can no
|
||||
longer starve real requests, only other background work. Two limits still apply inside it: a loop-bound
|
||||
semaphore caps concurrent audit writes at `_MAX_CONCURRENT_WRITES` (queueing in-process rather than
|
||||
holding a pooled connection, and keeping `drain()` deterministic on SQLite, where all three makers share
|
||||
one engine), and under an extreme burst the writer **sheds** load — it drops any audit row past
|
||||
`_MAX_PENDING` rather than let the pending set grow without bound. Audit must never OOM or wedge the
|
||||
**Audit back-pressure (`audit.py`).** Audit rows are written off the request path (fire-and-forget):
|
||||
`record_call` appends to an in-process queue and ONE writer task per process drains it `_BATCH` rows
|
||||
per INSERT on a **background**-pool connection (since 2026-09-07; before that four writers each took
|
||||
one row per session, which cost four slots per process for millisecond inserts). A burst can therefore
|
||||
never starve real requests, only other background work. Two limits still apply: a loop-bound semaphore
|
||||
holds the writer to `_MAX_CONCURRENT_WRITES` (1), which keeps `drain()` deterministic on SQLite, where
|
||||
all three makers share one engine, and under an extreme burst `_enqueue` **sheds** load — it drops any
|
||||
audit row past `_MAX_PENDING` queued rows rather than let the queue grow without bound. A batch the
|
||||
database refuses is retried row by row, so one bad row costs one row. Audit must never OOM or wedge the
|
||||
server. Shedding is the *only* loss that should ever happen: `record_call` splats its telemetry dict
|
||||
into `CallRecord(**fields)`, so a key with no matching column used to raise inside `_write`, where the
|
||||
except swallowed it, and the whole row disappeared — a telemetry field deployed one commit ahead of its
|
||||
@@ -333,6 +414,14 @@ is the whole point: `FaultCaptureHandler` starts at ERROR, so at WARNING a lost
|
||||
stdout and nothing else, and the only way to learn audit was dropping was to already suspect it and go
|
||||
grep. **A quiet audit table is now a bug you can alert on**, not one you find out about weeks later.
|
||||
|
||||
**Archive memory bound (`archive.py`).** Each pending archive recording holds its `body` bytes in a
|
||||
task closure — up to `_MAX_PENDING` (512) tasks × `archive_max_body_bytes` (2 MB) = 1 GB worst case.
|
||||
After #363 reduced `_MAX_CONCURRENT_WRITES` from 4 to 2, backlog built faster than it drained under
|
||||
heavy `/call` + MCP traffic, and the 2026-09-07T00:43:06Z OOM killed the web service at 4 GB.
|
||||
`_MAX_PENDING_BYTES` (256 MB) now caps total body bytes in pending work: `record()` sheds when
|
||||
EITHER the task count OR the bytes threshold is exceeded. The done callback releases bytes when a
|
||||
task completes; a regression test pins the bound.
|
||||
|
||||
The proxy is thin and IO-bound (a relay, low CPU/memory), so cheap machines scale it.
|
||||
|
||||
## The per-org daily spend cap
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
"""composite (org_id, created_at) on ledgerentry — the per-call daily-cap check stops scanning the
|
||||
whole platform's day
|
||||
|
||||
Revision ID: 0021
|
||||
Revises: 0020
|
||||
Create Date: 2026-09-06
|
||||
|
||||
`ledger.spent_today` runs inside EVERY metered call's reserve transaction, on an api-pool
|
||||
connection: `sum(amount_micro) WHERE org_id = ? AND kind = 'settle' AND created_at >= today`.
|
||||
`ledgerentry` carried only single-column indexes, so the planner had two bad choices and took one
|
||||
per org: walk `ix_ledgerentry_created_at` over the WHOLE platform's day and filter `org_id` in
|
||||
memory (heavy orgs), or BitmapAnd the org's ENTIRE history against the day (light orgs). Measured
|
||||
on prod 2026-09-06 at 4.38M rows / 2.3 GB, with ~400k rows written per day:
|
||||
|
||||
heaviest org, warm cache: Rows Removed by Filter: 322,638
|
||||
Buffers: shared hit=367,715 read=13,777 (539 ms)
|
||||
same query, cold cache: 56–106 s, holding an api-pool slot the whole time
|
||||
|
||||
ix_ledgerentry_org_id 1,858,056 scans 90,870,787,481 tuples read
|
||||
ix_ledgerentry_created_at 1,295,504 scans 157,269,530,757 tuples read
|
||||
ledgerentry heap 6.5 BILLION blocks read — 4× `callrecord`, the largest consumer
|
||||
|
||||
That is the API-pool saturation. The database is a 1 vCPU / 2 GB instance with a 512 MB buffer
|
||||
cache in front of 35 GB; a 30-second activity sample showed 88 % of active backends in
|
||||
`IO DataFileRead` / `IPC BufferIO` (waiting for a page, or for ANOTHER backend reading the same
|
||||
page) and 674 of 879 active samples were this one query. Whenever a large scan evicts the day's
|
||||
ledger pages, every in-flight `spent_today` stalls together on disk for tens of seconds, each one
|
||||
holding an api-pool connection, and the pool empties into `503 treg_saturated`. Raising the pool
|
||||
(15 → 20 on 2026-09-05) made it worse: more concurrent readers of the same cold pages.
|
||||
|
||||
`(org_id, created_at)` turns both plans into one tight range: the org's rows since midnight,
|
||||
nothing else. It also serves `ledger.entries_of` (the `/billing` page), which walked the whole
|
||||
`created_at` index BACKWARD filtering `org_id` — measured 57 s for a quiet org. `kind` is
|
||||
deliberately not a third column: the pair serves both queries, a triple would serve only one, and
|
||||
every extra index on a 400k-rows/day table is paid on every write.
|
||||
|
||||
Built with the 0020 discipline — see that revision's docstring for why raising `lock_timeout` for
|
||||
a CONCURRENT build does not break the 2026-08-15 rule (its lock conflicts with neither reads nor
|
||||
writes and a statement waiting for it blocks nobody), and why each index is inspected for INVALID
|
||||
debris before building. The expand-safety linter counts the autocommit escape as non-additive, so
|
||||
this revision declares a rollback floor pro forma: the operation is one additive index.
|
||||
"""
|
||||
from collections.abc import Sequence
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0021"
|
||||
down_revision: str | Sequence[str] | None = "0020"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
contract = True # pro forma — see the rollback floor note; the operation is one additive index
|
||||
|
||||
_TABLE = "ledgerentry"
|
||||
_INDEXES = (
|
||||
("ix_ledgerentry_org_id_created_at", ["org_id", "created_at"]),
|
||||
)
|
||||
|
||||
# Long enough to outlast an autovacuum pass on a 2.3 GB table; see 0020 for why waiting on THIS
|
||||
# lock is safe. `env.py`'s values are restored before the autocommit block ends.
|
||||
_LOCK_TIMEOUT = "180s"
|
||||
_STATEMENT_TIMEOUT = "600s"
|
||||
_ENV_LOCK_TIMEOUT = "5s"
|
||||
_ENV_STATEMENT_TIMEOUT = "120s"
|
||||
|
||||
_VALIDITY = sa.text(
|
||||
"SELECT i.indisvalid FROM pg_class c JOIN pg_index i ON i.indexrelid = c.oid "
|
||||
"WHERE c.relname = :name")
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
if op.get_bind().dialect.name != "postgresql":
|
||||
for name, columns in _INDEXES: # SQLite: no concurrent mode, no traffic to block
|
||||
op.create_index(name, _TABLE, columns)
|
||||
return
|
||||
# CONCURRENTLY cannot run inside a transaction; alembic opens one by default.
|
||||
with op.get_context().autocommit_block():
|
||||
bind = op.get_bind()
|
||||
bind.execute(sa.text(f"SET lock_timeout = '{_LOCK_TIMEOUT}'"))
|
||||
bind.execute(sa.text(f"SET statement_timeout = '{_STATEMENT_TIMEOUT}'"))
|
||||
try:
|
||||
for name, columns in _INDEXES:
|
||||
valid = bind.execute(_VALIDITY, {"name": name}).scalar()
|
||||
if valid is True:
|
||||
continue
|
||||
if valid is False: # debris from a killed build — unusable, and never repaired
|
||||
op.drop_index(name, table_name=_TABLE, postgresql_concurrently=True)
|
||||
op.create_index(name, _TABLE, columns, postgresql_concurrently=True)
|
||||
finally:
|
||||
bind.execute(sa.text(f"SET lock_timeout = '{_ENV_LOCK_TIMEOUT}'"))
|
||||
bind.execute(sa.text(f"SET statement_timeout = '{_ENV_STATEMENT_TIMEOUT}'"))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
for name, _ in _INDEXES:
|
||||
op.drop_index(name, table_name=_TABLE)
|
||||
@@ -0,0 +1,81 @@
|
||||
"""org.spent_today_micro / spent_today_day — the daily cap reads a counter, not the journal
|
||||
|
||||
Revision ID: 0022
|
||||
Revises: 0021
|
||||
Create Date: 2026-09-06
|
||||
|
||||
`ledger.spent_today` is the fail-closed per-org daily cap and runs inside EVERY metered call's
|
||||
reserve transaction on an api-pool connection. Until now it was two aggregates over `ledgerentry`
|
||||
and `hold` since midnight. Revision 0021 gave the ledger half an `(org_id, created_at)` index and
|
||||
that fixed light orgs and `/billing`, but not the two orgs that write half the platform's day:
|
||||
their rows sit on nearly every heap page of the day, so the planner (correctly) keeps walking the
|
||||
whole day's `created_at` range — measured after 0021 on prod 2026-09-06:
|
||||
|
||||
org 5430 Rows Removed by Filter: 166,760 Buffers: shared hit=394,248 439 ms warm
|
||||
org 4645 Rows Removed by Filter: 385,118 Buffers: shared hit=395,506 1,797 ms warm
|
||||
cold (day's pages evicted by another scan): 56–171 s, holding an api-pool slot throughout
|
||||
|
||||
No index can make "sum of this org's rows today" cheaper than "this org's pages today" — for a
|
||||
heavy org that IS the day. The counter makes it one primary-key read: `domain/money` folds every
|
||||
reserve, settle and release into `org.spent_today_micro` inside the UPDATE that already moves the
|
||||
balance, and `spent_today_day` says which UTC day it belongs to (the first movement of a new day
|
||||
resets it). `spent_today_from_ledger` keeps the journal view for reconciliation.
|
||||
|
||||
**Backfill, in its own autocommit step.** The ALTER takes ACCESS EXCLUSIVE on `org`, the hottest
|
||||
row-updated table (every reserve); adding a column with a constant default is metadata-only on this
|
||||
Postgres and holds that lock for milliseconds. The backfill — one range aggregate over today's
|
||||
ledger and holds — would hold it for seconds if it ran in the same transaction, which is exactly
|
||||
the 2026-08-15 shape. So the ALTER commits first, then the two UPDATEs run autocommit, taking
|
||||
only row locks on the orgs that moved money today. Calls that reserve between the backfill and the
|
||||
new code starting are missed by at most that window; the cap is a blast-radius guard, not a bill.
|
||||
|
||||
Rollback floor: `spent_today_micro` is NOT NULL with a server default kept (the model carries the
|
||||
same default), so older code still inserts orgs; downgrading drops both columns. The expand-safety
|
||||
linter counts the autocommit escape as non-additive, hence the contract marker.
|
||||
"""
|
||||
from collections.abc import Sequence
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0022"
|
||||
down_revision: str | Sequence[str] | None = "0021"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
contract = True
|
||||
|
||||
# `timezone('UTC', now())` because the app writes NAIVE UTC timestamps and dates (models._now);
|
||||
# a session in another zone would otherwise draw the day boundary in the wrong place.
|
||||
_BACKFILL_SETTLED = sa.text("""
|
||||
UPDATE org SET spent_today_micro = x.v, spent_today_day = (timezone('UTC', now()))::date
|
||||
FROM (SELECT org_id, -sum(amount_micro) AS v FROM ledgerentry
|
||||
WHERE kind = 'settle' AND created_at >= date_trunc('day', timezone('UTC', now()))
|
||||
GROUP BY org_id) x
|
||||
WHERE org.id = x.org_id
|
||||
""")
|
||||
_BACKFILL_HELD = sa.text("""
|
||||
UPDATE org SET spent_today_micro = spent_today_micro + x.v,
|
||||
spent_today_day = (timezone('UTC', now()))::date
|
||||
FROM (SELECT org_id, sum(amount_micro) AS v FROM hold
|
||||
WHERE created_at >= date_trunc('day', timezone('UTC', now()))
|
||||
GROUP BY org_id) x
|
||||
WHERE org.id = x.org_id
|
||||
""")
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("org", sa.Column("spent_today_micro", sa.BigInteger(), nullable=False,
|
||||
server_default="0"))
|
||||
op.add_column("org", sa.Column("spent_today_day", sa.Date(), nullable=True))
|
||||
if op.get_bind().dialect.name != "postgresql":
|
||||
return # SQLite deployments are dev databases with no day of history worth carrying over
|
||||
with op.get_context().autocommit_block():
|
||||
bind = op.get_bind()
|
||||
bind.execute(_BACKFILL_SETTLED)
|
||||
bind.execute(_BACKFILL_HELD)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
with op.batch_alter_table("org") as batch:
|
||||
batch.drop_column("spent_today_day")
|
||||
batch.drop_column("spent_today_micro")
|
||||
@@ -0,0 +1,74 @@
|
||||
"""composite (org_id, user_email, created_at) on callrecord — the per-user daily cap stops reading
|
||||
a member's whole history
|
||||
|
||||
Revision ID: 0023
|
||||
Revises: 0022
|
||||
Create Date: 2026-09-06
|
||||
|
||||
`governance/usage.count_today` backs the per-user daily call cap and runs on every capped call:
|
||||
`count(*) WHERE org_id = ? AND user_email = ? AND created_at >= today`. With 0020's
|
||||
`(org_id, created_at)` and the single-column `ix_callrecord_user_email` the planner BitmapAnd-ed
|
||||
the two, and the `user_email` half read the member's WHOLE history. Measured on prod 2026-09-06,
|
||||
after 0021, for the busiest member (287k rows, 104k of them today):
|
||||
|
||||
Bitmap Index Scan on ix_callrecord_user_email rows=287,527 2,621 ms of 3,016 ms
|
||||
seen in the activity sample 96 times, longest 45.8 s cold
|
||||
|
||||
The triple is one tight range: this org, this member, since midnight. Built with the 0020
|
||||
discipline (raised `lock_timeout` for the CONCURRENT build only, INVALID-debris check, `env.py`
|
||||
timeouts restored) — see that revision for why waiting on this lock blocks nobody. The expand-safety
|
||||
linter counts the autocommit escape as non-additive, so this revision declares a rollback floor
|
||||
pro forma: the operation is one additive index.
|
||||
"""
|
||||
from collections.abc import Sequence
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0023"
|
||||
down_revision: str | Sequence[str] | None = "0022"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
contract = True # pro forma — see the rollback floor note; the operation is one additive index
|
||||
|
||||
_TABLE = "callrecord"
|
||||
_INDEXES = (
|
||||
("ix_callrecord_org_id_user_email_created_at", ["org_id", "user_email", "created_at"]),
|
||||
)
|
||||
|
||||
_LOCK_TIMEOUT = "180s"
|
||||
_STATEMENT_TIMEOUT = "600s"
|
||||
_ENV_LOCK_TIMEOUT = "5s"
|
||||
_ENV_STATEMENT_TIMEOUT = "120s"
|
||||
|
||||
_VALIDITY = sa.text(
|
||||
"SELECT i.indisvalid FROM pg_class c JOIN pg_index i ON i.indexrelid = c.oid "
|
||||
"WHERE c.relname = :name")
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
if op.get_bind().dialect.name != "postgresql":
|
||||
for name, columns in _INDEXES: # SQLite: no concurrent mode, no traffic to block
|
||||
op.create_index(name, _TABLE, columns)
|
||||
return
|
||||
# CONCURRENTLY cannot run inside a transaction; alembic opens one by default.
|
||||
with op.get_context().autocommit_block():
|
||||
bind = op.get_bind()
|
||||
bind.execute(sa.text(f"SET lock_timeout = '{_LOCK_TIMEOUT}'"))
|
||||
bind.execute(sa.text(f"SET statement_timeout = '{_STATEMENT_TIMEOUT}'"))
|
||||
try:
|
||||
for name, columns in _INDEXES:
|
||||
valid = bind.execute(_VALIDITY, {"name": name}).scalar()
|
||||
if valid is True:
|
||||
continue
|
||||
if valid is False: # debris from a killed build — unusable, and never repaired
|
||||
op.drop_index(name, table_name=_TABLE, postgresql_concurrently=True)
|
||||
op.create_index(name, _TABLE, columns, postgresql_concurrently=True)
|
||||
finally:
|
||||
bind.execute(sa.text(f"SET lock_timeout = '{_ENV_LOCK_TIMEOUT}'"))
|
||||
bind.execute(sa.text(f"SET statement_timeout = '{_ENV_STATEMENT_TIMEOUT}'"))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
for name, _ in _INDEXES:
|
||||
op.drop_index(name, table_name=_TABLE)
|
||||
@@ -0,0 +1,72 @@
|
||||
"""membership.calls_today / calls_today_day — the per-user daily cap takes a slot, not a count
|
||||
|
||||
Revision ID: 0024
|
||||
Revises: 0023
|
||||
Create Date: 2026-09-06
|
||||
|
||||
`governance/usage.enforce_daily_cap` runs on every call and run a capped member makes. Until now it
|
||||
counted the member's `callrecord` and `runrecord` rows since midnight. Revision 0023's
|
||||
`(org_id, user_email, created_at)` index turned that into an index-only scan, but today's pages are
|
||||
not yet all-visible until autovacuum reaches them, so every row still fetched the heap — measured
|
||||
on prod 2026-09-06 for the busiest member (110k rows today):
|
||||
|
||||
Index Only Scan ... Heap Fetches: 110,166 Buffers: shared hit=81,738 read=14,436 2,777 ms
|
||||
|
||||
O(this member's rows today), per call, on an api-pool connection. Same disease as `spent_today`
|
||||
(revision 0022), same cure: a counter on the row the gate already has. `take_daily_slot` is one
|
||||
conditional UPDATE — the WHERE is the check and the SET is the count, so the cap is exact under
|
||||
concurrency and a refused call is not counted. Only capped members are counted; the roster and
|
||||
`/usage/me` keep reading the journal (`count_today`), and `seed_counter` copies today's journal
|
||||
onto the row when a cap is first set, so a member capped mid-day does not start from zero.
|
||||
|
||||
**Backfill, in its own autocommit step, capped members only.** 123 of 9,239 memberships carry a
|
||||
cap; each backfill row is one journal count over 0023's index — the heaviest ~3 s, the rest
|
||||
milliseconds. It runs after the ALTER commits, so the ACCESS EXCLUSIVE lock on `membership`
|
||||
(read on every request) lasts milliseconds, not the length of those counts.
|
||||
|
||||
Rollback floor: `calls_today` is NOT NULL with a server default kept (the model carries the same
|
||||
default), so older code still inserts memberships; downgrading drops both columns. The
|
||||
expand-safety linter counts the autocommit escape as non-additive, hence the contract marker.
|
||||
"""
|
||||
from collections.abc import Sequence
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0024"
|
||||
down_revision: str | Sequence[str] | None = "0023"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
contract = True
|
||||
|
||||
# `timezone('UTC', now())` because the app writes NAIVE UTC timestamps and dates; a session in
|
||||
# another zone would draw the day boundary in the wrong place.
|
||||
_BACKFILL = sa.text("""
|
||||
UPDATE membership SET calls_today = x.n, calls_today_day = (timezone('UTC', now()))::date
|
||||
FROM (SELECT m.id,
|
||||
(SELECT count(*) FROM callrecord r
|
||||
WHERE r.org_id = m.org_id AND r.user_email = u.email
|
||||
AND r.created_at >= date_trunc('day', timezone('UTC', now())))
|
||||
+ (SELECT count(*) FROM runrecord r
|
||||
WHERE r.org_id = m.org_id AND r.user_email = u.email
|
||||
AND r.created_at >= date_trunc('day', timezone('UTC', now()))) AS n
|
||||
FROM membership m JOIN "user" u ON u.id = m.user_id
|
||||
WHERE m.daily_call_cap >= 0) x
|
||||
WHERE membership.id = x.id
|
||||
""")
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("membership", sa.Column("calls_today", sa.Integer(), nullable=False,
|
||||
server_default="0"))
|
||||
op.add_column("membership", sa.Column("calls_today_day", sa.Date(), nullable=True))
|
||||
if op.get_bind().dialect.name != "postgresql":
|
||||
return # SQLite deployments are dev databases with no day of history worth carrying over
|
||||
with op.get_context().autocommit_block():
|
||||
op.get_bind().execute(_BACKFILL)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
with op.batch_alter_table("membership") as batch:
|
||||
batch.drop_column("calls_today_day")
|
||||
batch.drop_column("calls_today")
|
||||
@@ -54,7 +54,8 @@ _FLUSH_INTERVAL_S = 2.0 # max staleness before a flush
|
||||
|
||||
_flusher: asyncio.Task | None = None
|
||||
|
||||
_SERVER_DISTINCT_ID = "treg-server"
|
||||
SERVER_DISTINCT_ID = "treg-server"
|
||||
_SERVER_DISTINCT_ID = SERVER_DISTINCT_ID # older name, kept for callers
|
||||
_FAULT_VALUE_MAX = 500
|
||||
_FAULT_WINDOW_S = 10.0 # one event per (fault type, site) per window; the rest are counted
|
||||
_FAULT_MAX_KEYS = 500 # bound the ledger — it is what caps the cost, so it must be finite
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta, timezone
|
||||
@@ -207,6 +209,8 @@ async def start_email_login(email: str, client_ip: str) -> dict:
|
||||
raise EmailAuthError("demo_address")
|
||||
if _is_machine_email(email):
|
||||
raise EmailAuthError("machine_identity")
|
||||
if signup.blocked_email(email, "otp_start"): # refuse early: no code, no mail, no rate window
|
||||
raise EmailAuthError("blocked_domain")
|
||||
|
||||
async with database.session_maker() as db:
|
||||
await ratestore.sweep(db, OTP_START_NS)
|
||||
@@ -251,9 +255,11 @@ async def verify_email_login(email: str, code: str) -> VerifiedEmail:
|
||||
raise EmailAuthError("invalid_code")
|
||||
await ratestore.kv_pop(db, OTP_NS, email)
|
||||
try:
|
||||
user = await signup.find_or_create_user(db, email)
|
||||
user = await signup.find_or_create_user(db, email, door="otp_verify")
|
||||
except signup.MachineIdentityError as exc:
|
||||
raise EmailAuthError("machine_identity") from exc
|
||||
except signup.BlockedEmailError as exc: # a code minted before the domain was listed
|
||||
raise EmailAuthError("blocked_domain") from exc
|
||||
if user.suspended:
|
||||
raise EmailAuthError("suspended")
|
||||
await db.commit()
|
||||
@@ -425,12 +431,14 @@ def start_google_login(cli: str, callback_base: Callable[[], str]) -> SocialLogi
|
||||
return SocialLoginStart(state=state, url=url)
|
||||
|
||||
|
||||
async def _provision_social_user(email: str, state: str) -> SocialLoginProof:
|
||||
async def _provision_social_user(email: str, state: str, door: str) -> SocialLoginProof:
|
||||
async with database.session_maker() as db:
|
||||
try:
|
||||
user = await signup.find_or_create_user(db, email) # first login = registration (user only; no auto org)
|
||||
user = await signup.find_or_create_user(db, email, door=door) # first login = registration (user only; no auto org)
|
||||
except signup.MachineIdentityError as exc:
|
||||
raise SocialLoginError("machine_identity") from exc
|
||||
except signup.BlockedEmailError as exc: # a Google/GitHub account on a listed domain
|
||||
raise SocialLoginError("blocked_domain") from exc
|
||||
if user.suspended: # a banned account may prove its email but must not receive a live session
|
||||
raise SocialLoginError("suspended")
|
||||
await db.commit()
|
||||
@@ -470,7 +478,7 @@ async def complete_github_login(
|
||||
except Exception as exc: # noqa: BLE001
|
||||
print(f"[auth] github callback error: {exc}") # keep internals server-side, not in the response
|
||||
raise SocialLoginError("callback_failed") from exc
|
||||
return await _provision_social_user(email, state)
|
||||
return await _provision_social_user(email, state, "github")
|
||||
|
||||
|
||||
async def complete_google_login(
|
||||
@@ -507,7 +515,7 @@ async def complete_google_login(
|
||||
except Exception as exc: # noqa: BLE001
|
||||
print(f"[auth] google callback error: {exc}") # keep internals server-side, not in the response
|
||||
raise SocialLoginError("callback_failed") from exc
|
||||
return await _provision_social_user(email, state)
|
||||
return await _provision_social_user(email, state, "google")
|
||||
|
||||
|
||||
async def current_identity(x_treg_token: str, session_cookie: str) -> CurrentIdentity:
|
||||
@@ -600,9 +608,11 @@ async def confirm_invite_signin(email_token: str) -> InviteSigninProof:
|
||||
if invite is None: # consumed / expired / revoked / suspended org → the SPA's expired banner
|
||||
raise InviteSigninError("expired")
|
||||
try:
|
||||
user = await signup.find_or_create_user(db, invite.email) # first click = registration (user only, no auto org)
|
||||
user = await signup.find_or_create_user(db, invite.email, door="invite_link") # first click = registration (user only, no auto org)
|
||||
except signup.MachineIdentityError as exc:
|
||||
raise InviteSigninError("machine_identity") from exc
|
||||
except signup.BlockedEmailError as exc: # an invite to a listed domain must not become a session
|
||||
raise InviteSigninError("blocked_domain") from exc
|
||||
if user is None or user.suspended: # a banned account may hold the link but must not get a session
|
||||
raise InviteSigninError("suspended")
|
||||
invite.email_token_hash = None # consume: one sign-in per emailed link
|
||||
@@ -871,9 +881,13 @@ async def _refresh_grant(*, refresh_token: str, client_id: str, resource: str) -
|
||||
# cost of being wrong is one sign-in; the cost of the other mistake is somebody's balance.
|
||||
killed = await _revoke_refresh_family(row.family_id, "reuse detected", db)
|
||||
await db.commit()
|
||||
# `CallRecord` has no column for the family or the kill count; audit drops unknown
|
||||
# telemetry keys with a warning on every occurrence, so they go to the log instead.
|
||||
logging.getLogger("treg.auth").warning(
|
||||
"refresh token reuse: family %s revoked (%s grants)", row.family_id, killed)
|
||||
audit.record_call(org_id=row.org_id, user_email="", tool_name="oauth.refresh_reuse",
|
||||
method="POST", path="/oauth/token", status_code=400, client="",
|
||||
telemetry={"family": row.family_id, "revoked": killed})
|
||||
refused_by="auth")
|
||||
raise OAuthServerError(
|
||||
"invalid_grant",
|
||||
"this refresh token was already used — the grant has been revoked, sign in again",
|
||||
|
||||
@@ -12,7 +12,7 @@ from .. import adsconv, health, sandbox as demo_sandbox
|
||||
from ..domain import money as ledger
|
||||
from ..domain import referrals
|
||||
from ..domain.governance.teams import _make_org_membership, _slugify
|
||||
from ..domain.identity.access import _is_machine_email, _norm_email
|
||||
from ..domain.identity.access import _email_domain, _is_blocked_email, _is_machine_email, _norm_email
|
||||
from ..infra.db import session_maker
|
||||
from ..models import Org, User
|
||||
from ..timeutil import utcnow_naive as _utcnow_naive
|
||||
@@ -30,7 +30,28 @@ class MachineIdentityError(Exception):
|
||||
"""A machine identity reached a human identity-provisioning command."""
|
||||
|
||||
|
||||
async def find_or_create_user(db: AsyncSession, email: str) -> User:
|
||||
class BlockedEmailError(Exception):
|
||||
"""An address on a blocked email domain reached an identity door."""
|
||||
|
||||
|
||||
def blocked_email(email: str, door: str) -> bool:
|
||||
"""The blocklist DECISION for the identity doors, over the pure classifier `_is_blocked_email`:
|
||||
True means refuse. One structured line per block, so a burst of refusals is countable from the
|
||||
logs — the refusal itself tells the caller nothing, so the log is the only detection. Fails OPEN:
|
||||
a classifier error is logged and the door stays open, because a misconfiguration must never break
|
||||
a real sign-in. `door` names the endpoint for the log only; it never reaches the caller."""
|
||||
log = logging.getLogger("treg.auth")
|
||||
try:
|
||||
if not _is_blocked_email(email):
|
||||
return False
|
||||
except Exception as exc: # noqa: BLE001 - fail open, by design
|
||||
log.error("event=blocklist_error door=%s error=%s", door, exc)
|
||||
return False
|
||||
log.warning("event=signup_blocked_domain door=%s domain=%s", door, _email_domain(email))
|
||||
return True
|
||||
|
||||
|
||||
async def find_or_create_user(db: AsyncSession, email: str, *, door: str = "login") -> User:
|
||||
"""Find a user by email, else register them — the user ONLY, **no auto personal org**. The shared
|
||||
core of every identity door (GitHub / Google / email OTP). A brand-new user therefore lands with
|
||||
zero teams and is asked to NAME + CREATE their first team (the dashboard's mandatory welcome, or
|
||||
@@ -44,6 +65,10 @@ async def find_or_create_user(db: AsyncSession, email: str) -> User:
|
||||
# (The domains are unroutable, so a code could never be delivered anyway; this makes it explicit.)
|
||||
if _is_machine_email(email):
|
||||
raise MachineIdentityError
|
||||
# Same choke point for the domain blocklist, and BEFORE the lookup on purpose: a blocked domain
|
||||
# gets no session whether or not it already has a row (sign-in, not just sign-up).
|
||||
if blocked_email(email, door):
|
||||
raise BlockedEmailError
|
||||
user = (await db.execute(select(User).where(User.email == email))).scalar_one_or_none()
|
||||
if user is None:
|
||||
user = User(email=email)
|
||||
@@ -176,6 +201,8 @@ async def register_user(
|
||||
# otherwise a caller could squat an agent address before an admin mints that agent.
|
||||
if _is_machine_email(email):
|
||||
raise SignupError("machine_identity")
|
||||
if blocked_email(email, "register"): # mints a promo-funded team in one call; refuse first
|
||||
raise SignupError("blocked_domain")
|
||||
if webhook_url and not health.safe_webhook_url(webhook_url): # SSRF guard on the alert URL
|
||||
raise SignupError("unsafe_webhook")
|
||||
if (await db.execute(select(User).where(User.email == email))).scalar_one_or_none():
|
||||
@@ -222,6 +249,10 @@ async def create_org(
|
||||
async with session_maker() as db:
|
||||
if demo_sandbox.is_sandbox_user(user): # anonymous sandbox visitors cannot mint real teams
|
||||
raise SignupError("sandbox_user")
|
||||
# An identity registered BEFORE its domain was listed still holds a live token; every team it
|
||||
# creates is another promo grant, so the blocklist covers this door too, not only sign-in.
|
||||
if blocked_email(user.email, "create_org"):
|
||||
raise SignupError("blocked_domain")
|
||||
click_field, gclid, landing = _ad_attribution_from(ad_cookie)
|
||||
# A browser sign-in reaches this door instead of /users, so both doors must read attribution.
|
||||
for _ in range(3): # a concurrent create can claim the slug before commit; retry a fresh lookup
|
||||
|
||||
+28
-7
@@ -203,11 +203,19 @@ from datetime import datetime, timedelta, timezone
|
||||
_log = logging.getLogger("treg.archive")
|
||||
_pending: set[asyncio.Task] = set()
|
||||
_MAX_PENDING = 512
|
||||
# Memory bound: each pending task holds its body bytes in a closure. 512 tasks × 8 MB = 4 GB in the
|
||||
# worst case — the 2026-09-07 OOM. This cap sheds recordings when total pending body bytes exceeds
|
||||
# the threshold, BEFORE the task count would shed them. 256 MB is generous for a 4 GB container and
|
||||
# still allows ~128 concurrent recordings of typical 2 MB bodies.
|
||||
_MAX_PENDING_BYTES = 256 * 1024 * 1024
|
||||
_pending_bytes = 0
|
||||
# At most this many recordings TOUCH THE DATABASE at once (audit's discipline, and its exact
|
||||
# loop-bound pattern). Without it a traffic burst put up to 512 concurrent short sessions in
|
||||
# front of the API's 15-slot pool — SToneX's pool-pressure report, 2026-09-03. Queued recordings
|
||||
# wait INSIDE their task; the caller's response left long ago either way.
|
||||
_MAX_CONCURRENT_WRITES = 4
|
||||
# wait INSIDE their task; the caller's response left long ago either way. Two, not four: every
|
||||
# slot here is paid twice (two uvicorn workers) and again at every deploy against the database's
|
||||
# 103-connection ceiling, and a recording is one INSERT of a body that is already in memory.
|
||||
_MAX_CONCURRENT_WRITES = 2
|
||||
|
||||
_sem: asyncio.Semaphore | None = None
|
||||
_sem_loop = None
|
||||
@@ -285,23 +293,36 @@ def record(
|
||||
bytes (`/calls/{id}/result`). Computed here rather than in `_store` so they are computed
|
||||
ONCE (the store reuses them) and are true whether or not the write lands: a shed recording
|
||||
still names the answer the caller received."""
|
||||
global _pending_bytes
|
||||
kh = cache_key(method, endpoint_id, url, caller_body, headers)
|
||||
ch = content_hash(body)
|
||||
if len(_pending) >= _MAX_PENDING: # shed load; the stream self-heals on the next call
|
||||
body_len = len(body)
|
||||
# Shed on EITHER count OR bytes — whichever bound bites first. The bytes bound prevents OOM
|
||||
# when a few large bodies queue while the semaphore is full; the count bound is the legacy
|
||||
# backstop for many small bodies (archive_max_body_bytes is 2 MB, so 512 × 2 MB = 1 GB).
|
||||
if len(_pending) >= _MAX_PENDING or _pending_bytes + body_len > _MAX_PENDING_BYTES:
|
||||
return kh, ch
|
||||
_pending_bytes += body_len
|
||||
task = asyncio.create_task(asyncio.wait_for(_store(
|
||||
method=method, endpoint_id=endpoint_id, provider=provider, url=url,
|
||||
caller_body=caller_body, headers=headers, status_code=status_code,
|
||||
media_type=media_type, body=body, origin=origin, key_hash=kh, body_hash=ch),
|
||||
timeout=_STORE_TIMEOUT_S))
|
||||
_pending.add(task)
|
||||
# NOT redundant with drain()'s own removal: on a running server drain() never fires, and this
|
||||
# callback is the only exit from `_pending` — without it the set fills to _MAX_PENDING and
|
||||
# record() sheds every recording from then on.
|
||||
task.add_done_callback(_pending.discard)
|
||||
# Release bytes AND task when done. NOT redundant with drain()'s own removal: on a running
|
||||
# server drain() never fires, and this callback is the only exit from `_pending` — without it
|
||||
# the set fills to _MAX_PENDING and record() sheds every recording from then on.
|
||||
task.add_done_callback(lambda t: _task_done(t, body_len))
|
||||
return kh, ch
|
||||
|
||||
|
||||
def _task_done(task: asyncio.Task, body_len: int) -> None:
|
||||
"""Release the task and its body bytes from the pending budget."""
|
||||
global _pending_bytes
|
||||
_pending.discard(task)
|
||||
_pending_bytes -= body_len
|
||||
|
||||
|
||||
async def store_terminal_response(
|
||||
call_id: str, provider: str, endpoint_id: str, status_code: int, body: bytes,
|
||||
) -> None:
|
||||
|
||||
+98
-45
@@ -1,29 +1,38 @@
|
||||
"""Audit writes — deferred and fire-and-forget (rule #2: never block the proxied response).
|
||||
|
||||
`record_call` schedules an insert on its own session and returns immediately; the response
|
||||
streams without waiting. A strong reference to each task is held until it finishes (otherwise
|
||||
the event loop may GC a bare create_task). Failures are swallowed: an audit hiccup must never
|
||||
break a real call. `drain()` flushes pending writes on shutdown / in tests.
|
||||
`record_call` queues a row and returns immediately; the response streams without waiting. One
|
||||
writer task per process drains the queue in batches on one connection (a strong reference to it
|
||||
is held until it finishes, otherwise the event loop may GC a bare create_task). Failures are
|
||||
swallowed: an audit hiccup must never break a real call. `drain()` flushes pending writes on
|
||||
shutdown / in tests.
|
||||
|
||||
Back-pressure (why this matters): each write opens a DB connection, from the BACKGROUND pool
|
||||
(db.py), so a burst here can starve other background work but never real calls. The semaphore stays
|
||||
as the inner, cheaper bound — it queues in-process instead of holding a pooled connection, and it is
|
||||
what keeps `drain()` deterministic on sqlite, where all three makers share one engine. Under an
|
||||
extreme burst we DROP audit rows past `_MAX_PENDING` rather than grow without bound — audit is
|
||||
best-effort; never OOM or wedge the server for it.
|
||||
Back-pressure (why this matters): the writer's connection comes from the BACKGROUND pool (db.py),
|
||||
so a burst here can starve other background work but never real calls. Rows queue in-process, not
|
||||
as pooled connections: one writer per process takes them off the queue `_BATCH` at a time and lands
|
||||
each batch in one INSERT round trip, which is what keeps `drain()` deterministic on sqlite, where
|
||||
all three makers share one engine. Under an extreme burst we DROP audit rows past `_MAX_PENDING`
|
||||
rather than grow without bound — audit is best-effort; never OOM or wedge the server for it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from collections import deque
|
||||
|
||||
from .infra.db import background_session_maker
|
||||
from .models import CallRecord, RunRecord, SearchMiss
|
||||
|
||||
_pending: set[asyncio.Task] = set()
|
||||
_MAX_CONCURRENT_WRITES = 4 # cap on audit writes holding a DB connection at once (protect the request pool)
|
||||
# ONE writer per process, and it writes in batches. Audit rows are single-row inserts that cost
|
||||
# milliseconds each, so four concurrent writers bought nothing but four `background` slots - and
|
||||
# every slot is paid twice (two uvicorn workers) and again at every deploy against the database's
|
||||
# 103-connection ceiling (ops/deploy.md). One writer draining a queue in batches of `_BATCH` rows
|
||||
# lands the same rows in fewer round trips and holds one connection.
|
||||
_MAX_CONCURRENT_WRITES = 1
|
||||
_MAX_PENDING = 5000 # shed load past this: drop the audit row rather than grow unbounded
|
||||
_BATCH = 200 # rows per INSERT round trip; a failed batch retries row by row
|
||||
_queue: deque[tuple[type, dict]] = deque()
|
||||
|
||||
_sem: asyncio.Semaphore | None = None
|
||||
_sem_loop = None
|
||||
@@ -48,7 +57,7 @@ def record_call(
|
||||
stay NULL. It is still fire-and-forget: the money landed in the ledger synchronously, so losing a
|
||||
row here costs analytics, not accounting. `refused_by` marks a call TREG refused before anything
|
||||
went upstream (see models.CallRecord) — NULL whenever the provider actually answered."""
|
||||
_schedule(_write(CallRecord,
|
||||
_enqueue(CallRecord, dict(
|
||||
org_id=org_id, user_email=user_email, tool_name=tool_name,
|
||||
method=method, path=path, status_code=status_code, client=client, refused_by=refused_by,
|
||||
**_known_fields(CallRecord, telemetry),
|
||||
@@ -78,14 +87,14 @@ def record_search_miss(*, query: str, source: str) -> None:
|
||||
"""A catalog search that matched nothing — logged so the misses can steer ingest (see
|
||||
models.SearchMiss). Same contract as every write here: fire-and-forget, and a dropped row
|
||||
under load costs a data point, never a search response."""
|
||||
_schedule(_write(SearchMiss, query=query[:300], source=source))
|
||||
_enqueue(SearchMiss, dict(query=query[:300], source=source))
|
||||
|
||||
|
||||
def record_run(
|
||||
*, org_id: int | None = None, user_email: str, bundle_name: str, argv: list, exit_code: int,
|
||||
duration_ms: int, client: str = ""
|
||||
) -> None:
|
||||
_schedule(_write(RunRecord,
|
||||
_enqueue(RunRecord, dict(
|
||||
org_id=org_id, user_email=user_email, bundle_name=bundle_name,
|
||||
argv=argv, exit_code=exit_code, duration_ms=duration_ms, client=client,
|
||||
))
|
||||
@@ -95,39 +104,74 @@ _shed = 0 # audit rows dropped by back-pressure this process; only ever grows
|
||||
|
||||
|
||||
def _schedule(coro) -> None:
|
||||
if len(_pending) >= _MAX_PENDING: # shed load — audit is best-effort, never OOM the server
|
||||
coro.close()
|
||||
# Say so. Shedding bypasses `_write` entirely, so the warning there never fires for it, and
|
||||
# a shed row is invisible: the audit table simply has less in it. For a table whose job is
|
||||
# to record what happened, "quiet" and "quietly broken" must not look identical — and the
|
||||
# failure-evidence columns ride this same path, so a burst silently loses exactly the errors
|
||||
# someone would go looking for. Logged on the first drop and then every 1,000th, so a long
|
||||
# incident cannot itself flood the log.
|
||||
global _shed
|
||||
_shed += 1
|
||||
if _shed == 1 or _shed % 1000 == 0:
|
||||
logging.getLogger("treg.audit").error(
|
||||
"audit back-pressure: %d row(s) dropped this process (pending at %d)",
|
||||
_shed, _MAX_PENDING)
|
||||
return
|
||||
"""Run the writer as a tracked task. Shedding happens in `_enqueue`, on the queue: a shed row is
|
||||
invisible - the audit table simply has less in it - so for a table whose job is to record what
|
||||
happened, "quiet" and "quietly broken" must not look identical. The failure-evidence columns
|
||||
ride this same path, so a burst would otherwise silently lose exactly the errors someone would
|
||||
go looking for. Logged on the first drop and then every 1,000th."""
|
||||
task = asyncio.create_task(coro)
|
||||
_pending.add(task)
|
||||
task.add_done_callback(_pending.discard)
|
||||
task.add_done_callback(_writer_done)
|
||||
|
||||
|
||||
async def _write(model, **fields) -> None:
|
||||
async with _get_sem(): # cap concurrent DB connections held by audit — never starve the request pool
|
||||
try:
|
||||
async with background_session_maker() as session:
|
||||
session.add(model(**fields))
|
||||
await session.commit()
|
||||
except Exception: # noqa: BLE001 — audit must never surface into a call's result
|
||||
# Swallowed on purpose, but neither silent nor local: the row is lost — that is the
|
||||
# contract — and ERROR is what puts that loss in front of someone. At WARNING it stayed
|
||||
# in the container's stdout, below the fault handler's threshold, so the only way to
|
||||
# learn that audit was dropping rows was to already suspect it and go grep.
|
||||
def _writer_done(task: asyncio.Task) -> None:
|
||||
"""A row enqueued between the writer's last empty-queue check and this callback saw `_pending`
|
||||
still occupied and did not start a writer; start one for it here or it waits for the next call."""
|
||||
_pending.discard(task)
|
||||
if _queue and not _pending:
|
||||
_schedule(_flush())
|
||||
|
||||
|
||||
def _enqueue(model, fields: dict) -> None:
|
||||
"""Queue one row and make sure a writer is running. The shed check is on the QUEUE, which is
|
||||
where rows wait now; `_pending` holds at most the one writer task."""
|
||||
if len(_queue) >= _MAX_PENDING:
|
||||
_shed_one()
|
||||
return
|
||||
_queue.append((model, fields))
|
||||
if not _pending:
|
||||
_schedule(_flush())
|
||||
|
||||
|
||||
def _shed_one() -> None:
|
||||
global _shed
|
||||
_shed += 1
|
||||
if _shed == 1 or _shed % 1000 == 0:
|
||||
logging.getLogger("treg.audit").error(
|
||||
"audit back-pressure: %d row(s) dropped this process (pending at %d)",
|
||||
_shed, _MAX_PENDING)
|
||||
|
||||
|
||||
async def _flush() -> None:
|
||||
"""Drain the queue in batches until it is empty, then exit. One connection for the whole run.
|
||||
|
||||
A batch that fails is retried row by row, so one row the database refuses (a value out of
|
||||
range, a constraint) costs that row and not the 199 around it - the failure evidence of a
|
||||
burst is exactly what such a burst must not take down with it.
|
||||
"""
|
||||
async with _get_sem():
|
||||
while _queue:
|
||||
batch = [_queue.popleft() for _ in range(min(_BATCH, len(_queue)))]
|
||||
if not await _write_batch(batch):
|
||||
for row in batch:
|
||||
await _write_batch([row])
|
||||
|
||||
|
||||
async def _write_batch(rows: list[tuple[type, dict]]) -> bool:
|
||||
try:
|
||||
async with background_session_maker() as session:
|
||||
session.add_all([model(**fields) for model, fields in rows])
|
||||
await session.commit()
|
||||
return True
|
||||
except Exception: # noqa: BLE001 — audit must never surface into a call's result
|
||||
# Swallowed on purpose, but neither silent nor local: the row is lost — that is the
|
||||
# contract — and ERROR is what puts that loss in front of someone. At WARNING it stayed
|
||||
# in the container's stdout, below the fault handler's threshold, so the only way to
|
||||
# learn that audit was dropping rows was to already suspect it and go grep.
|
||||
if len(rows) == 1:
|
||||
logging.getLogger("treg.audit").error(
|
||||
"audit write dropped for %s", model.__name__, exc_info=True)
|
||||
"audit write dropped for %s", rows[0][0].__name__, exc_info=True)
|
||||
return False
|
||||
|
||||
|
||||
async def drain() -> None:
|
||||
@@ -139,7 +183,16 @@ async def drain() -> None:
|
||||
# suspends, so a loop keyed only on the callback spins synchronously forever while that callback
|
||||
# (and every timer on the loop) starves. Latent since the first import; a CI Postgres runner hit
|
||||
# the window deterministically and wedged whole 15-minute jobs on it.
|
||||
while _pending:
|
||||
#
|
||||
# Whatever is still queued once no writer is running is flushed HERE, inline, not through
|
||||
# `_schedule`: a drain that depends on scheduling a task to make progress spins forever the
|
||||
# moment scheduling is stubbed out (a test kills the pipeline exactly that way).
|
||||
while True:
|
||||
tasks = list(_pending)
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
_pending.difference_update(tasks)
|
||||
if tasks:
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
_pending.difference_update(tasks)
|
||||
continue
|
||||
if not _queue:
|
||||
return
|
||||
await _flush()
|
||||
|
||||
@@ -3,6 +3,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import time
|
||||
from collections.abc import Sequence
|
||||
from contextlib import asynccontextmanager
|
||||
from copy import copy
|
||||
@@ -413,6 +415,44 @@ def _route_manifest(routes: Sequence[BaseRoute]) -> list[str]:
|
||||
return result
|
||||
|
||||
|
||||
_POOL_GAUGE_SAMPLE_S = 1.0
|
||||
_POOL_GAUGE_EMIT_S = 60.0
|
||||
|
||||
|
||||
async def pool_gauge(*, sample_s: float = _POOL_GAUGE_SAMPLE_S,
|
||||
emit_s: float = _POOL_GAUGE_EMIT_S) -> None:
|
||||
"""Every minute, one `db_pool_gauge` event: the peak connections each pool had checked out in
|
||||
that minute, next to its capacity. The reading behind `TREG_DB_POOL_OVERRIDES`: a pool whose
|
||||
peak sits at capacity is one whose waiters are timing out (api: `503 treg_saturated`;
|
||||
background: an audit or archive row dropped after `pool_timeout`), and a pool whose peak never
|
||||
nears it is holding connections nothing uses. Sampling is a counter read, no I/O; a bad pass
|
||||
never kills the loop. Telemetry, not a database consumer - so it is not in
|
||||
`ROLE_BACKGROUND_TASKS` and runs in every role."""
|
||||
from .infra.db import fold_pool_peaks, pool_snapshot
|
||||
peaks: dict[str, int] = {}
|
||||
samples = 0
|
||||
opened = time.monotonic()
|
||||
while True:
|
||||
try:
|
||||
fold_pool_peaks(peaks, pool_snapshot())
|
||||
samples += 1
|
||||
if time.monotonic() - opened >= emit_s:
|
||||
snapshot = pool_snapshot()
|
||||
props: dict = {"samples": samples, "window_s": round(time.monotonic() - opened)}
|
||||
for name, row in snapshot.items():
|
||||
props[f"{name}_peak"] = peaks.get(name, 0)
|
||||
props[f"{name}_capacity"] = row["capacity"]
|
||||
props[f"{name}_headroom"] = row["capacity"] - peaks.get(name, 0)
|
||||
if snapshot:
|
||||
analytics.capture(analytics.SERVER_DISTINCT_ID, "db_pool_gauge", props)
|
||||
peaks, samples, opened = {}, 0, time.monotonic()
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception: # noqa: BLE001 - a gauge must never take the service down
|
||||
logging.getLogger("treg").exception("db pool gauge pass failed")
|
||||
await asyncio.sleep(sample_s)
|
||||
|
||||
|
||||
def _lifespan(role: AppRole):
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
@@ -428,6 +468,7 @@ def _lifespan(role: AppRole):
|
||||
if ROLE_BACKGROUND_TASKS[role] and adsconv.enabled()
|
||||
else None
|
||||
)
|
||||
gauge_task = asyncio.create_task(pool_gauge()) if analytics.enabled() else None
|
||||
# The archive's refresh worker (docs/context/architecture/archive.md): serve mode only,
|
||||
# and a zero daily cap disables it without touching serving. Same discipline as the ads
|
||||
# task — in-process, cancelled on shutdown, a bad pass never kills the loop.
|
||||
@@ -460,6 +501,8 @@ def _lifespan(role: AppRole):
|
||||
yield
|
||||
finally:
|
||||
try:
|
||||
if gauge_task is not None:
|
||||
gauge_task.cancel()
|
||||
if ads_task is not None:
|
||||
ads_task.cancel()
|
||||
if archive_task is not None:
|
||||
|
||||
@@ -31,6 +31,18 @@ def platform_setting_name(provider: str) -> str:
|
||||
return "platform_key_" + (provider or "").lower().replace("-", "_")
|
||||
|
||||
|
||||
@lru_cache
|
||||
def _blocked_email_domains(raw: str) -> frozenset[str]:
|
||||
"""Parse `TREG_BLOCKED_EMAIL_DOMAINS` once per distinct value, not per request: split on commas,
|
||||
trim, drop a leading `@` or `.` (operators paste both spellings), lowercase, drop empties. A
|
||||
dotless entry (`com`) is dropped too: the classifier walks parent domains, so a bare public
|
||||
suffix would refuse every address on earth from one typo in a dashboard field."""
|
||||
return frozenset(
|
||||
d for d in (part.strip().lstrip("@.").rstrip(".").lower() for part in raw.split(","))
|
||||
if "." in d
|
||||
)
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
model_config = SettingsConfigDict(env_file=".env", env_prefix="TREG_", extra="ignore")
|
||||
|
||||
@@ -425,6 +437,22 @@ class Settings(BaseSettings):
|
||||
# must be explicitly enabled (TREG_EMAIL_DEV_MODE=true) for local testing without a mail sender.
|
||||
email_dev_mode: bool = False
|
||||
|
||||
# The OPS tier of the email-domain blocklist (TREG_BLOCKED_EMAIL_DOMAINS), comma-separated:
|
||||
# "newfarm.io,other-farm.net". ADDED to the code tier in `domain/identity/access.py` (treg's
|
||||
# confirmed farm roots and the throwaway-mail keyword rules), never replacing it. A listed domain
|
||||
# blocks itself AND every subdomain, case-insensitively, at every sign-up and sign-in door and at
|
||||
# the two doors that mint a promo-funded team (POST /users, POST /orgs). It exists because a
|
||||
# signup-grant farm moves to a new root in minutes and the answer has to be a dashboard edit, not
|
||||
# a deploy. Empty (the default) adds nothing. Existing accounts on a listed domain are suspended
|
||||
# out of band, so listing a domain strands nobody legitimate. A blocklist, deliberately: no
|
||||
# allowlist, no table, no admin UI.
|
||||
blocked_email_domains: str = ""
|
||||
|
||||
@property
|
||||
def blocked_email_domain_set(self) -> frozenset[str]:
|
||||
"""The normalised `TREG_BLOCKED_EMAIL_DOMAINS` entries; empty = the code tier alone."""
|
||||
return _blocked_email_domains(self.blocked_email_domains)
|
||||
|
||||
# Frictionless local mode: `curl … | sh` brings up a server you are already signed into, with no
|
||||
# account, email or password. Only takes effect when `single_user_ok` allows it (see below).
|
||||
single_user: bool = False
|
||||
|
||||
@@ -1,15 +1,18 @@
|
||||
"""Per-member usage policy shared by call and run surfaces."""
|
||||
|
||||
import logging
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import func
|
||||
from sqlalchemy import case, func, update
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
from sqlmodel import select
|
||||
|
||||
from ...models import CallRecord, RunRecord
|
||||
from ...models import CallRecord, Membership, RunRecord
|
||||
from ...timeutil import utcnow_naive
|
||||
from ..identity.access import Caller
|
||||
|
||||
log = logging.getLogger("treg.usage")
|
||||
|
||||
|
||||
class UsagePolicyError(Exception):
|
||||
"""A member exhausted their configured daily usage allowance."""
|
||||
@@ -25,8 +28,15 @@ def _day_start_utc() -> datetime:
|
||||
|
||||
|
||||
async def count_today(db: AsyncSession, org_id: int | None, user_email: str) -> int:
|
||||
"""How many usage events this user has produced in this org since midnight UTC: proxy calls +
|
||||
local-run grants (both `CallRecord`) plus server runs (`RunRecord`). Two indexed COUNTs."""
|
||||
"""How many usage events this user has produced in this org since midnight UTC, from the JOURNAL:
|
||||
proxy calls + local-run grants (both `CallRecord`) plus server runs (`RunRecord`).
|
||||
|
||||
NOT the cap's number and NOT on the call path. This is what the roster, `/usage/me` and the
|
||||
usage report show, and what `seed_counter` copies when a cap is first set; the gate itself
|
||||
reads `Membership.calls_today` (`take_daily_slot`). Two range COUNTs over
|
||||
`(org_id, user_email, created_at)` - O(this member's rows today), which for a member making
|
||||
100k calls a day was 2.8 s per call when the gate still ran it (2026-09-06).
|
||||
"""
|
||||
since = _day_start_utc()
|
||||
calls = (await db.execute(select(func.count()).select_from(CallRecord).where(
|
||||
CallRecord.org_id == org_id, CallRecord.user_email == user_email, CallRecord.created_at >= since,
|
||||
@@ -37,15 +47,56 @@ async def count_today(db: AsyncSession, org_id: int | None, user_email: str) ->
|
||||
return calls + runs
|
||||
|
||||
|
||||
async def take_daily_slot(db: AsyncSession, membership_id: int, cap: int) -> bool:
|
||||
"""Admit one usage event against the member's cap, or say no. ONE conditional UPDATE.
|
||||
|
||||
The WHERE is the check and the SET is the count, so N concurrent calls cannot each read a
|
||||
compliant figure and together overshoot - the same idiom as `money.reserve`. The first event
|
||||
of a new UTC day resets the counter to 1 instead of adding to yesterday. A refused event is not
|
||||
counted: the counter is "admitted today", so a capped member hammering the gate reads exactly
|
||||
`cap`, not a runaway number. Does not commit; the caller's transaction owns it.
|
||||
"""
|
||||
today = utcnow_naive().date()
|
||||
# "Used today" is the counter only if it belongs to today; a stale or never-set day is 0. The
|
||||
# same expression gates and counts, so a cap of 0 refuses even the day's first event.
|
||||
used_today = case((Membership.calls_today_day == today, Membership.calls_today), else_=0)
|
||||
result = await db.execute(
|
||||
update(Membership)
|
||||
.where(Membership.id == membership_id, used_today < cap)
|
||||
.values(calls_today=used_today + 1, calls_today_day=today)
|
||||
)
|
||||
return result.rowcount == 1
|
||||
|
||||
|
||||
async def seed_counter(db: AsyncSession, membership: Membership, user_email: str) -> None:
|
||||
"""Start the counter from today's journal - called when a cap is set on a member who was
|
||||
unlimited until now. Only capped members are counted on the call path, so without this a
|
||||
member who already made 50k calls today would get a fresh allowance the moment they were capped.
|
||||
One journal count, at cap-setting time, never per call."""
|
||||
membership.calls_today = await count_today(db, membership.org_id, user_email)
|
||||
membership.calls_today_day = utcnow_naive().date()
|
||||
|
||||
|
||||
async def enforce_daily_cap(caller: Caller, db: AsyncSession, *, sandbox: bool) -> None:
|
||||
"""Refuse a call/run once the caller has used their per-user daily cap for this org. `-1` (the
|
||||
default) = unlimited, so unmetered members pay ZERO extra queries. The sandbox has its own limiter
|
||||
and is exempt. Soft by design: the count reads best-effort `CallRecord`s, so under heavy load it
|
||||
can lag slightly and fail OPEN (a few extra slip through) — never closed. See docs/USAGE-METERING-PLAN.md."""
|
||||
and is exempt.
|
||||
|
||||
One conditional UPDATE of the member's row (`take_daily_slot`), exact under concurrency. Fails
|
||||
OPEN if that statement cannot run: a cap is a courtesy limit an admin set on a colleague, and
|
||||
the database being unavailable is not the colleague's fault - the money gates behind this one
|
||||
are the ones that fail closed. See docs/USAGE-METERING-PLAN.md.
|
||||
"""
|
||||
cap = caller.membership.daily_call_cap
|
||||
if cap < 0 or sandbox:
|
||||
return
|
||||
used = await count_today(db, caller.org_id, caller.email)
|
||||
if used >= cap:
|
||||
try:
|
||||
admitted = await take_daily_slot(db, caller.membership.id, cap)
|
||||
except Exception as exc: # noqa: BLE001 - fail open, see docstring
|
||||
log.warning("daily-cap check failed for membership %s: %s", caller.membership.id, exc)
|
||||
return
|
||||
if not admitted:
|
||||
used = (await db.execute(
|
||||
select(Membership.calls_today).where(Membership.id == caller.membership.id))).scalar() or 0
|
||||
raise UsagePolicyError(
|
||||
f"daily usage limit reached ({used}/{cap}) — ask an admin to raise your cap")
|
||||
|
||||
@@ -241,3 +241,58 @@ def _is_agent_email(email: str) -> bool:
|
||||
def _is_machine_email(email: str) -> bool:
|
||||
"""An identity minted by an admin for a machine — never a person who can sign in."""
|
||||
return _is_agent_email(email) or _norm_email(email).endswith(f"@{PUBLIC_DEMO_DOMAIN}")
|
||||
|
||||
|
||||
# ---- the email-domain blocklist: throwaway mail and abusive signup domains ----------------------
|
||||
# A new team is created with a promotional balance (`application.signup._grant_signup_promo`), which
|
||||
# makes bulk registration on throwaway addresses worth someone's while. Two tiers, one classifier.
|
||||
# Tier 1 is CODE: domains confirmed abusive in our own data, plus substring rules that catch
|
||||
# throwaway-mail providers no static list has seen yet. Tier 2 is OPS:
|
||||
# `TREG_BLOCKED_EMAIL_DOMAINS`, unioned in, so the next domain is a dashboard edit made the minute it
|
||||
# appears, not a deploy. Three rules, each of which exists because the obvious implementation is
|
||||
# wrong:
|
||||
# - match the DOMAIN only, never the whole address. Matching the address false-flags real users
|
||||
# whose USERNAME happens to contain a keyword (`tempmail@gmail.com` is a real person).
|
||||
# - walk parent domains, whole labels off the front only and never the bare last label, because
|
||||
# registering `<random>.<blocked-root>` is otherwise a one-line bypass. The walk is safe because
|
||||
# no entry is a bare public suffix, which `config._blocked_email_domains` enforces for the ops
|
||||
# tier by dropping dotless entries.
|
||||
# - a PURE classifier: refusing, logging and skipping a perk are the caller's decisions
|
||||
# (`application.signup.blocked_email`).
|
||||
BLOCKED_EMAIL_DOMAINS: frozenset[str] = frozenset({
|
||||
# Confirmed abusive in our own data: bulk registration only, no legitimate account on any of them.
|
||||
"uberip.com",
|
||||
"westcast-systems.com",
|
||||
"mailfox.win",
|
||||
"yopmail.com",
|
||||
# Free `.my.id` subdomains are handed out publicly. Listed as the parent so the walk catches
|
||||
# `<anything>.my.id`.
|
||||
"my.id",
|
||||
})
|
||||
# Substring rules on the domain: throwaway-mail providers name themselves.
|
||||
BLOCKED_EMAIL_KEYWORDS: tuple[str, ...] = (
|
||||
"tempmail", "temp-mail", "mailinator", "guerrilla", "throwaway", "10minute", "trashmail",
|
||||
"yopmail", "sharklasers", "dispostable", "getnada", "maildrop", "moakt", "mohmal",
|
||||
"emailondeck", "fakemail",
|
||||
)
|
||||
|
||||
|
||||
def _email_domain(email: str) -> str:
|
||||
"""The lowercased domain part of an address, "" when there is none. The ONLY part of an address
|
||||
the blocklist ever looks at."""
|
||||
return _norm_email(email).rpartition("@")[2]
|
||||
|
||||
|
||||
def _is_blocked_email(email: str) -> bool:
|
||||
"""Pure classifier: is this address on a blocked domain, on a subdomain of one, or on a domain
|
||||
that names itself a throwaway? An empty ops list leaves the code tier alone in force."""
|
||||
domain = _email_domain(email)
|
||||
if not domain:
|
||||
return False
|
||||
ops = get_settings().blocked_email_domain_set
|
||||
labels = domain.split(".")
|
||||
for i in range(len(labels) - 1): # every parent domain, never the bare last label
|
||||
candidate = ".".join(labels[i:])
|
||||
if candidate in BLOCKED_EMAIL_DOMAINS or candidate in ops:
|
||||
return True
|
||||
return any(keyword in domain for keyword in BLOCKED_EMAIL_KEYWORDS)
|
||||
|
||||
@@ -46,7 +46,7 @@ import uuid
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import NamedTuple
|
||||
|
||||
from sqlalchemy import delete, func, update
|
||||
from sqlalchemy import case, delete, func, update
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
from sqlmodel import select
|
||||
@@ -125,11 +125,38 @@ async def _entry(
|
||||
return row
|
||||
|
||||
|
||||
async def _add_balance(db: AsyncSession, org_id: int, delta_micro: int) -> None:
|
||||
"""Unconditional balance move (grants, refunds). The CONDITIONAL one lives in `reserve`."""
|
||||
await db.execute(
|
||||
update(Org).where(Org.id == org_id).values(balance_micro=Org.balance_micro + delta_micro)
|
||||
)
|
||||
def _day_start() -> datetime:
|
||||
return _now().replace(hour=0, minute=0, second=0, microsecond=0)
|
||||
|
||||
|
||||
def _spent_today_values(delta_micro: int) -> dict:
|
||||
"""SET clause that folds `delta_micro` into the org's daily-spend counter.
|
||||
|
||||
The counter is "committed since midnight UTC": settled today plus still held from today. One
|
||||
CASE keeps it honest across the day boundary - the first movement of a new UTC day resets it
|
||||
to that movement instead of adding to yesterday's total. Always applied inside the statement
|
||||
that already holds the org row (the balance UPDATE), so it costs no extra lock and cannot
|
||||
disagree with the balance about which transaction it belongs to.
|
||||
"""
|
||||
today = _now().date()
|
||||
return {
|
||||
"spent_today_micro": case(
|
||||
(Org.spent_today_day == today, Org.spent_today_micro + delta_micro), else_=delta_micro),
|
||||
"spent_today_day": today,
|
||||
}
|
||||
|
||||
|
||||
async def _add_balance(db: AsyncSession, org_id: int, delta_micro: int, *,
|
||||
spent_delta_micro: int = 0) -> None:
|
||||
"""Unconditional balance move (grants, refunds). The CONDITIONAL one lives in `reserve`.
|
||||
|
||||
`spent_delta_micro` rides in the same UPDATE: settle and release move the balance AND change
|
||||
what counts as committed today, and the two must land in one statement.
|
||||
"""
|
||||
values: dict = {"balance_micro": Org.balance_micro + delta_micro}
|
||||
if spent_delta_micro:
|
||||
values.update(_spent_today_values(spent_delta_micro))
|
||||
await db.execute(update(Org).where(Org.id == org_id).values(**values))
|
||||
|
||||
|
||||
# ---- funding -----------------------------------------------------------------------------------
|
||||
@@ -265,7 +292,7 @@ async def reserve_in_transaction(
|
||||
# callers cannot both pass it. rowcount 0 = the balance was not there.
|
||||
update(Org)
|
||||
.where(Org.id == org_id, Org.balance_micro >= charged)
|
||||
.values(balance_micro=Org.balance_micro - charged)
|
||||
.values(balance_micro=Org.balance_micro - charged, **_spent_today_values(charged))
|
||||
)
|
||||
if result.rowcount != 1:
|
||||
balance = (await db.execute(select(Org.balance_micro).where(Org.id == org_id))).scalar() or 0
|
||||
@@ -295,6 +322,7 @@ class _ClaimedHold(NamedTuple):
|
||||
amount_micro: int
|
||||
org_id: int
|
||||
endpoint_id: str
|
||||
created_at: datetime
|
||||
|
||||
|
||||
async def _claim_hold(db: AsyncSession, call_id: str) -> _ClaimedHold | None:
|
||||
@@ -312,7 +340,7 @@ async def _claim_hold(db: AsyncSession, call_id: str) -> _ClaimedHold | None:
|
||||
hold = await db.get(Hold, call_id)
|
||||
if hold is None:
|
||||
return None
|
||||
claimed = _ClaimedHold(hold.amount_micro, hold.org_id, hold.endpoint_id)
|
||||
claimed = _ClaimedHold(hold.amount_micro, hold.org_id, hold.endpoint_id, hold.created_at)
|
||||
result = await db.execute(delete(Hold).where(Hold.id == call_id))
|
||||
if result.rowcount != 1:
|
||||
# Lost the claim: somebody else is closing this hold. Deliberately NO rollback — the DELETE
|
||||
@@ -365,8 +393,11 @@ async def _settle_in_transaction(
|
||||
# The hold came out of the balance at reserve time; give back whatever the call didn't use. If the
|
||||
# observed cost overran the estimate the delta is negative, which correctly takes MORE balance —
|
||||
# the next reserve is the gate that stops an overrun from compounding.
|
||||
if reserved != consumed:
|
||||
await _add_balance(db, hold.org_id, reserved - consumed)
|
||||
# The daily counter: what settled today goes in; the hold it replaces comes out, but only if
|
||||
# that hold was counted today (a hold opened yesterday was yesterday's commitment).
|
||||
spent_delta = consumed - (reserved if hold.created_at >= _day_start() else 0)
|
||||
if reserved != consumed or spent_delta:
|
||||
await _add_balance(db, hold.org_id, reserved - consumed, spent_delta_micro=spent_delta)
|
||||
await _entry(
|
||||
db, org_id=hold.org_id, kind="settle", amount_micro=-consumed, call_id=call_id,
|
||||
endpoint_id=hold.endpoint_id, created_at=settled_at,
|
||||
@@ -416,7 +447,9 @@ async def _release_in_transaction(
|
||||
if hold is None:
|
||||
return 0, False
|
||||
amount = hold.amount_micro
|
||||
await _add_balance(db, hold.org_id, amount)
|
||||
# Same rule as settle: a hold counted today leaves today's counter; an older one never was in it.
|
||||
spent_delta = -amount if hold.created_at >= _day_start() else 0
|
||||
await _add_balance(db, hold.org_id, amount, spent_delta_micro=spent_delta)
|
||||
await _entry(db, org_id=hold.org_id, kind="release", amount_micro=amount, call_id=call_id,
|
||||
endpoint_id=hold.endpoint_id, meta={**(meta or {}), "reason": reason})
|
||||
# Nothing was billable, so nothing is attributable: the tag rows go with the hold. Leaving them
|
||||
@@ -485,12 +518,33 @@ async def reap_stale_holds(db: AsyncSession, *, org_id: int | None = None, limit
|
||||
# ---- reads -------------------------------------------------------------------------------------
|
||||
async def spent_today(db: AsyncSession, org_id: int) -> int:
|
||||
"""Micro-USD this org has committed since midnight UTC: everything SETTLED today plus everything
|
||||
still HELD from today. Two indexed aggregates, and the number a daily spend cap is checked against.
|
||||
still HELD from today - the number a daily spend cap is checked against.
|
||||
|
||||
Runs on EVERY metered call, inside the reserve transaction, on an api-pool connection, so its
|
||||
cost is the platform's throughput: ONE primary-key read of the org row. The counter is kept by
|
||||
reserve/settle/release inside the balance UPDATE (`_spent_today_values`); it was an aggregate
|
||||
over `ledgerentry` until 2026-09-06, when that aggregate - O(rows the platform wrote today) for
|
||||
a heavy org, whatever the index - was what emptied the API pool. `spent_today_from_ledger` is
|
||||
the same number from the journal, for reconciliation.
|
||||
|
||||
Deliberately not "sum of reserve entries": a reserve is refunded at settle, so counting both would
|
||||
double-charge every call. Settled + still-open is exactly the money that is gone or promised.
|
||||
"""
|
||||
since = _now().replace(hour=0, minute=0, second=0, microsecond=0)
|
||||
row = (await db.execute(
|
||||
select(Org.spent_today_micro, Org.spent_today_day).where(Org.id == org_id))).one_or_none()
|
||||
if row is None or row[1] != _now().date():
|
||||
return 0 # nothing has moved for this org today; the first movement will stamp the day
|
||||
return int(row[0])
|
||||
|
||||
|
||||
async def spent_today_from_ledger(db: AsyncSession, org_id: int) -> int:
|
||||
"""The same number computed from the journal - two range aggregates over `(org_id, created_at)`.
|
||||
|
||||
NOT on the call path. This is the reconciliation view of the counter: what `spent_today` must
|
||||
agree with, and what a test asserts it against. `reconcile.py` and an operator who distrusts
|
||||
the counter read this; a bigger cap check does not.
|
||||
"""
|
||||
since = _day_start()
|
||||
settled = (await db.execute(
|
||||
select(func.coalesce(func.sum(LedgerEntry.amount_micro), 0)).where(
|
||||
LedgerEntry.org_id == org_id, LedgerEntry.kind == "settle", LedgerEntry.created_at >= since)
|
||||
|
||||
+67
-4
@@ -3,6 +3,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from collections.abc import AsyncIterator
|
||||
from functools import cache
|
||||
from importlib import import_module
|
||||
@@ -38,8 +39,9 @@ _is_sqlite = "sqlite" in _db_url
|
||||
# whole pool when this was 2 — see `_purge_expired_error_evidence`, now on `background` and
|
||||
# single-flighted so concurrent readers cannot multiply it.
|
||||
#
|
||||
# Sizes are PER INSTANCE and a rolling deploy runs two, so the SUM is what must stay under the
|
||||
# database plan's ~100 ceiling — see ops/deploy.md, and the guard test.
|
||||
# Sizes are PER PROCESS, the web service runs two uvicorn workers, and a rolling deploy runs two
|
||||
# instances, so `per_process × 2 × 2` is what must stay under the database plan's 103 ceiling —
|
||||
# see `connection_budget` below and ops/deploy.md.
|
||||
#
|
||||
# These numbers can only be validated in production: too small and real traffic gets 503s, too large
|
||||
# and the bulkhead is decorative, and no test can tell you which. `TREG_DB_POOL_OVERRIDES` makes a
|
||||
@@ -48,8 +50,8 @@ _is_sqlite = "sqlite" in _db_url
|
||||
# Everything that can hold a `background` slot at the same moment. Keep this in step with reality:
|
||||
# it is what sizes the pool, and a consumer missing from it is a row silently dropped under load.
|
||||
BACKGROUND_CONSUMERS: dict[str, int] = {
|
||||
"audit._write": 4, # bounded by audit._MAX_CONCURRENT_WRITES
|
||||
"archive._store/_touch": 4, # bounded by archive._MAX_CONCURRENT_WRITES (one shared semaphore)
|
||||
"audit._flush": 1, # one batching writer per process (audit._MAX_CONCURRENT_WRITES)
|
||||
"archive._store/_touch": 2, # bounded by archive._MAX_CONCURRENT_WRITES (one shared semaphore)
|
||||
"adsconv.worker": 1, # holds its slot across two Google round trips — see follow-ups
|
||||
"archive.prune_worker": 1, # holds one across a whole sweep
|
||||
"archive.refresh_worker": 1,
|
||||
@@ -100,6 +102,36 @@ if _overrides := get_settings().db_pool_overrides:
|
||||
POOL_SPECS = _apply_overrides(POOL_SPECS, _overrides)
|
||||
|
||||
|
||||
def connection_budget(workers: int | None = None) -> dict[str, int]:
|
||||
"""How many connections these specs can open, at the three scopes that matter.
|
||||
|
||||
Every number in `POOL_SPECS` is PER PROCESS, and the reference deployment runs TWO: Render sets
|
||||
`WEB_CONCURRENCY=2` on the 2c-4g plan and uvicorn honors it (`Started server process` twice in
|
||||
the boot log). A rolling deploy then runs two instances for a minute. So the ceiling the specs
|
||||
must clear is `per_process × workers × 2` against Postgres's `max_connections` (103 on the
|
||||
1c-2g plan) - the arithmetic that, taken per instance, let the 2026-09-04 defaults (31) open
|
||||
124 connections at every deploy until the dashboard override cut them to 21 (84).
|
||||
"""
|
||||
per_process = sum(spec["pool_size"] + spec["max_overflow"] for spec in POOL_SPECS.values())
|
||||
if workers is None:
|
||||
try:
|
||||
workers = max(1, int(os.environ.get("WEB_CONCURRENCY", "1")))
|
||||
except ValueError:
|
||||
workers = 1
|
||||
return {"per_process": per_process, "workers": workers,
|
||||
"per_instance": per_process * workers, "deploy_peak": per_process * workers * 2}
|
||||
|
||||
|
||||
_budget = connection_budget()
|
||||
logging.getLogger("treg").info(
|
||||
"db pools per process: api %d+%d, admin %d+%d, background %d+%d = %d; x%d workers = %d per "
|
||||
"instance, %d at a rolling deploy (Postgres max_connections on the reference plan: 103)",
|
||||
POOL_SPECS["api"]["pool_size"], POOL_SPECS["api"]["max_overflow"],
|
||||
POOL_SPECS["admin"]["pool_size"], POOL_SPECS["admin"]["max_overflow"],
|
||||
POOL_SPECS["background"]["pool_size"], POOL_SPECS["background"]["max_overflow"],
|
||||
_budget["per_process"], _budget["workers"], _budget["per_instance"], _budget["deploy_peak"])
|
||||
|
||||
|
||||
def _new_engine(name: str):
|
||||
"""One pooled engine per `POOL_SPECS` entry.
|
||||
|
||||
@@ -128,6 +160,37 @@ if _is_sqlite:
|
||||
_admin_engine = _background_engine = _engine
|
||||
|
||||
_engines = (_engine, _admin_engine, _background_engine)
|
||||
_POOL_NAMES = ("api", "admin", "background")
|
||||
|
||||
|
||||
def pool_snapshot() -> dict[str, dict[str, int]]:
|
||||
"""What each pool holds RIGHT NOW: connections checked out, of how many it may hand out.
|
||||
|
||||
The sizing question `POOL_SPECS` answers by arithmetic ("13 is the sum of the semaphores") can
|
||||
only be settled by measurement; this is the measurement. `checked_out` counts slots in use
|
||||
(persistent and overflow alike), `capacity` is `pool_size + max_overflow`. SQLite has no pool
|
||||
worth reading and reports nothing. A pure read of SQLAlchemy's counters - no lock, no I/O.
|
||||
"""
|
||||
if _is_sqlite:
|
||||
return {}
|
||||
out: dict[str, dict[str, int]] = {}
|
||||
for name, engine in zip(_POOL_NAMES, _engines):
|
||||
pool = engine.sync_engine.pool
|
||||
checked_out = getattr(pool, "checkedout", None)
|
||||
if checked_out is None:
|
||||
continue
|
||||
spec = POOL_SPECS[name]
|
||||
out[name] = {"checked_out": int(checked_out()),
|
||||
"capacity": spec["pool_size"] + spec["max_overflow"]}
|
||||
return out
|
||||
|
||||
|
||||
def fold_pool_peaks(peaks: dict[str, int], snapshot: dict[str, dict[str, int]]) -> dict[str, int]:
|
||||
"""Keep the per-pool maximum of `checked_out` seen across samples (the gauge's whole job)."""
|
||||
for name, row in snapshot.items():
|
||||
if row["checked_out"] > peaks.get(name, 0):
|
||||
peaks[name] = row["checked_out"]
|
||||
return peaks
|
||||
|
||||
# The API pool: every request handler, through `get_session` or directly.
|
||||
session_maker = async_sessionmaker(_engine, class_=AsyncSession, expire_on_commit=False)
|
||||
|
||||
+34
-2
@@ -8,7 +8,7 @@ so every list/call/mutation is scoped to the caller's org. See docs/MULTI-TENANC
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from datetime import date, datetime, timezone
|
||||
|
||||
from sqlalchemy import BigInteger, JSON, Column, Index, Integer, UniqueConstraint, text
|
||||
from sqlmodel import Field, SQLModel
|
||||
@@ -44,6 +44,14 @@ class Org(SQLModel, table=True):
|
||||
# conditional UPDATE against this integer is what stops concurrent agent calls racing past zero
|
||||
# (see ledger.reserve). Only `domain/money` may write it.
|
||||
balance_micro: int = Field(default=0)
|
||||
# "Committed since midnight UTC": everything settled today plus everything still held from
|
||||
# today - the number the fail-closed daily cap is checked against on EVERY metered call. Kept
|
||||
# here, in the same UPDATE that moves the balance, because the equivalent aggregate over
|
||||
# `ledgerentry` cost a scan of the platform's whole day per call and emptied the API pool
|
||||
# (revision 0022). Written ONLY by domain/money; `spent_today_day` says which UTC day the
|
||||
# counter belongs to, and a movement on a later day resets it.
|
||||
spent_today_micro: int = Field(default=0, sa_column=Column("spent_today_micro", BigInteger, nullable=False, server_default="0"))
|
||||
spent_today_day: date | None = Field(default=None)
|
||||
|
||||
# ---- Stripe billing (see billing.py; NO card data ever lands here) ----------------------------
|
||||
# The org's Stripe Customer. Created lazily on the first top-up and reused forever after, because
|
||||
@@ -166,8 +174,15 @@ class Membership(SQLModel, table=True):
|
||||
promoted_from: str = Field(default="")
|
||||
webhook_url: str | None = Field(default=None) # health alerts for this member's org POST here
|
||||
# Per-user, per-day usage cap for this org (counts proxy calls + local + server runs). -1 = unlimited
|
||||
# (the default — nobody is capped until an admin sets a limit). See api._enforce_daily_cap.
|
||||
# (the default — nobody is capped until an admin sets a limit). See governance/usage.enforce_daily_cap.
|
||||
daily_call_cap: int = Field(default=-1)
|
||||
# The number that cap is checked against: usage events the gate has admitted since midnight
|
||||
# UTC, and which UTC day it belongs to. Taken with ONE conditional UPDATE per capped call, so
|
||||
# the check costs no count over `callrecord` (which had read a member's whole history per
|
||||
# call - revision 0024) and the cap is exact rather than "a few extra slip through". Only
|
||||
# capped members are counted; the roster's `used_today` still reads the journal.
|
||||
calls_today: int = Field(default=0, sa_column=Column("calls_today", Integer, nullable=False, server_default="0"))
|
||||
calls_today_day: date | None = Field(default=None)
|
||||
# Per-member tool ACL: NULL = ALL tools in the org (the default — no restriction, no regression); a
|
||||
# JSON list of tool NAMES = the ONLY tools this member may call or run. See api._require_tool_access.
|
||||
tool_access: list | None = Field(default=None, sa_column=Column("tool_access", JSON, nullable=True))
|
||||
@@ -248,6 +263,12 @@ class CallRecord(SQLModel, table=True):
|
||||
# every 3 ms request query queue behind it until `pool_timeout` fires and
|
||||
# callers get `503 treg_saturated`. Sizing the pool cannot fix a scan.
|
||||
Index("ix_callrecord_endpoint_id_created_at", "endpoint_id", "created_at"),
|
||||
# The per-user daily call cap (`governance/usage.count_today`, on every
|
||||
# capped call): "this org, this member, since midnight". Without the triple
|
||||
# the planner BitmapAnd-ed the member's WHOLE history through
|
||||
# `ix_callrecord_user_email` - measured 2.6 s of 3.0 s on prod 2026-09-06 for
|
||||
# a member with 287k rows. Revision 0023 builds it concurrently.
|
||||
Index("ix_callrecord_org_id_user_email_created_at", "org_id", "user_email", "created_at"),
|
||||
Index("ix_callrecord_org_id_created_at", "org_id", "created_at"),)
|
||||
|
||||
id: int | None = Field(default=None, primary_key=True)
|
||||
@@ -670,6 +691,17 @@ class LedgerEntry(SQLModel, table=True):
|
||||
reserve/settle is negative. `call_id` correlates the reserve→settle / reserve→release pair.
|
||||
"""
|
||||
|
||||
# `(org_id, created_at)` is what EVERY metered call pays for: `ledger.spent_today` (the
|
||||
# fail-closed daily cap, inside the reserve transaction on an api-pool connection) asks
|
||||
# "this org, since midnight". With only single-column indexes the planner walked the whole
|
||||
# platform's day and filtered the org in memory - measured on prod 2026-09-06 at 4.38M rows:
|
||||
# 322k rows discarded and 381k buffer touches per call, 56-106 s once the day's pages were
|
||||
# cold, each one holding an api-pool slot. That was the `503 treg_saturated` mechanism, and
|
||||
# this table (not `callrecord`) was the largest IO consumer in the database. The pair also
|
||||
# serves `entries_of` (`/billing`), which had walked the whole `created_at` index backward.
|
||||
# Revision 0021 builds it concurrently.
|
||||
__table_args__ = (Index("ix_ledgerentry_org_id_created_at", "org_id", "created_at"),)
|
||||
|
||||
id: str = Field(primary_key=True) # uuid4 hex
|
||||
org_id: int = Field(foreign_key="org.id", index=True)
|
||||
block_id: str | None = Field(default=None, index=True)
|
||||
|
||||
+16
-1
@@ -22,6 +22,7 @@ import time
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
from sqlalchemy import delete
|
||||
from sqlalchemy.dialects import postgresql, sqlite
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
|
||||
from .models import Ephemeral
|
||||
@@ -57,8 +58,12 @@ async def kv_put(db: AsyncSession, ns: str, k: str, v: dict, ttl_s: float | None
|
||||
the code's lifetime)."""
|
||||
row = await db.get(Ephemeral, (ns, k))
|
||||
if row is None:
|
||||
# An INSERT that another process may be racing: two uvicorn workers striking the same
|
||||
# provider write the same `capacity:lock` key within the same millisecond, and the loser
|
||||
# of a plain INSERT died on `ephemeral_pkey` (prod, 2026-09-06). ON CONFLICT makes the
|
||||
# last writer win, which is what a key/value put means.
|
||||
exp = _utcnow_naive() + timedelta(seconds=ttl_s or 0)
|
||||
db.add(Ephemeral(ns=ns, k=k, v=v, expires_at=exp))
|
||||
await db.execute(_upsert(db.get_bind().dialect.name, ns=ns, k=k, v=v, expires_at=exp))
|
||||
return
|
||||
row.v = v # reassign (not in-place) so SQLAlchemy marks the JSON column dirty
|
||||
if ttl_s is not None:
|
||||
@@ -66,6 +71,16 @@ async def kv_put(db: AsyncSession, ns: str, k: str, v: dict, ttl_s: float | None
|
||||
db.add(row)
|
||||
|
||||
|
||||
def _upsert(dialect: str, **values):
|
||||
"""`INSERT ... ON CONFLICT (ns, k) DO UPDATE` for the dialect at hand - both backends spell it
|
||||
the same way, but SQLAlchemy exposes it per dialect."""
|
||||
insert = postgresql.insert if dialect == "postgresql" else sqlite.insert
|
||||
stmt = insert(Ephemeral).values(**values)
|
||||
return stmt.on_conflict_do_update(
|
||||
index_elements=[Ephemeral.ns, Ephemeral.k],
|
||||
set_={"v": stmt.excluded.v, "expires_at": stmt.excluded.expires_at})
|
||||
|
||||
|
||||
async def kv_pop(db: AsyncSession, ns: str, k: str) -> dict | None:
|
||||
"""Read-and-delete (ns, k) atomically-ish within this session. Returns the value if it was still
|
||||
live (not expired), else None. The row is always removed."""
|
||||
|
||||
@@ -57,6 +57,9 @@ class EmailVerifyIn(BaseModel):
|
||||
_EMAIL_HTTP_ERRORS = {
|
||||
"demo_address": (400, "that's a demo address — pick a real email"),
|
||||
"machine_identity": (403, "this address cannot be used to sign in"),
|
||||
# Same words as machine_identity on purpose: the caller learns neither that a list exists nor
|
||||
# what is on it.
|
||||
"blocked_domain": (403, "this address cannot be used to sign in"),
|
||||
"rate_limited": (429, "too many code requests — please wait a few minutes"),
|
||||
"invalid_code": (401, "invalid code"),
|
||||
"suspended": (403, "account suspended"),
|
||||
@@ -108,7 +111,7 @@ async def _find_or_create_user(db: AsyncSession, email: str) -> User:
|
||||
Caller commits."""
|
||||
try:
|
||||
return await signup.find_or_create_user(db, email)
|
||||
except signup.MachineIdentityError as exc:
|
||||
except (signup.MachineIdentityError, signup.BlockedEmailError) as exc:
|
||||
raise HTTPException(status_code=403, detail="this address cannot be used to sign in") from exc
|
||||
|
||||
|
||||
@@ -191,6 +194,8 @@ _SOCIAL_PAGE_ERRORS = {
|
||||
"google_unverified_email": ("Login failed", "Your Google email isn't verified.", False, 400),
|
||||
"callback_failed": ("Login failed", "Something went wrong. Please try again.", False, 502),
|
||||
"suspended": ("Account suspended", "This account has been suspended.", False, 403),
|
||||
# A page, like `suspended`: a human is in the browser. Names no list and no domain.
|
||||
"blocked_domain": ("Sign-in refused", "This address cannot be used to sign in.", False, 403),
|
||||
}
|
||||
|
||||
|
||||
@@ -674,6 +679,8 @@ async def auth_invite_signin_confirm(request: Request):
|
||||
raise HTTPException(status_code=403, detail="this address cannot be used to sign in") from exc
|
||||
if exc.kind == "suspended":
|
||||
return _auth_page("Account suspended", "This account has been suspended.", ok=False, status=403)
|
||||
if exc.kind == "blocked_domain":
|
||||
return _auth_page("Sign-in refused", "This address cannot be used to sign in.", ok=False, status=403)
|
||||
raise
|
||||
resp = RedirectResponse(proof.destination, status_code=303)
|
||||
resp.set_cookie(sess.COOKIE, proof.session_cookie, httponly=True,
|
||||
|
||||
+11
-12
@@ -41,11 +41,11 @@ from ..config import get_settings
|
||||
from ..domain.catalog import store as catalog_store
|
||||
from ..domain.governance import access as access_policy
|
||||
from ..domain.governance import publicdemo as publicdemo_policy
|
||||
from ..domain.governance import usage as usage_policy
|
||||
from ..domain.identity.access import Caller, require_member
|
||||
from ..infra.db import get_session
|
||||
from ..models import Tool
|
||||
from .auth import _client_ip
|
||||
from .orgs import count_today
|
||||
|
||||
|
||||
# The app alias preserves the moved handlers' decorator text byte-for-byte.
|
||||
@@ -122,17 +122,16 @@ async def _enforce_public_demo_ip_cap(request: Request, db: AsyncSession) -> Non
|
||||
|
||||
|
||||
async def _enforce_daily_cap(caller: Caller, db: AsyncSession) -> None:
|
||||
"""Refuse a call/run once the caller has used their per-user daily cap for this org. `-1` (the
|
||||
default) = unlimited, so unmetered members pay ZERO extra queries. The sandbox has its own limiter
|
||||
and is exempt. Soft by design: the count reads best-effort `CallRecord`s, so under heavy load it
|
||||
can lag slightly and fail OPEN (a few extra slip through) — never closed. See docs/USAGE-METERING-PLAN.md."""
|
||||
cap = caller.membership.daily_call_cap
|
||||
if cap < 0 or demo_sandbox.is_sandbox(caller.org):
|
||||
return
|
||||
used = await count_today(db, caller.org_id, caller.email)
|
||||
if used >= cap:
|
||||
raise HTTPException(status_code=429, detail=(
|
||||
f"daily usage limit reached ({used}/{cap}) — ask an admin to raise your cap"))
|
||||
"""The run surfaces' door to the per-user daily cap - the SAME gate `/call/` goes through
|
||||
(`application/call/authorize.py`), so a member cannot dodge the cap by switching path. Commits,
|
||||
because the gate takes the slot with a write and this request's session is the only owner."""
|
||||
try:
|
||||
await usage_policy.enforce_daily_cap(
|
||||
caller, db, sandbox=demo_sandbox.is_sandbox(caller.org))
|
||||
except usage_policy.UsagePolicyError as exc:
|
||||
await db.commit()
|
||||
raise HTTPException(status_code=429, detail=exc.detail) from exc
|
||||
await db.commit()
|
||||
|
||||
|
||||
def _translate_call_failure(exc: CallFailure) -> HTTPException:
|
||||
|
||||
@@ -346,6 +346,7 @@ def _deny_view(r: DenyRule) -> dict:
|
||||
|
||||
_SIGNUP_HTTP_ERRORS = {
|
||||
"machine_identity": (403, "this address cannot be used to sign in"),
|
||||
"blocked_domain": (403, "this address cannot be used to sign in"), # same words: leaks no list
|
||||
"unsafe_webhook": (422, "webhook_url must be a public http(s) URL"),
|
||||
"email_exists": (409, "email already registered"),
|
||||
"sandbox_user": (403, (
|
||||
@@ -499,6 +500,8 @@ async def accept_invite(body: AcceptIn, db: AsyncSession = Depends(get_session))
|
||||
org = await db.get(Org, invite.org_id)
|
||||
if org is not None and org.suspended: # don't let anyone join a platform-locked org
|
||||
raise HTTPException(status_code=403, detail="org suspended")
|
||||
if signup_use_cases.blocked_email(email, "invite_code"): # creates a User directly: guards itself
|
||||
raise HTTPException(status_code=403, detail="this address cannot be used to sign in")
|
||||
user = (await db.execute(select(User).where(User.email == email))).scalar_one_or_none()
|
||||
if user is not None and user.suspended: # a banned user must not accrue new memberships
|
||||
raise HTTPException(status_code=403, detail="account suspended")
|
||||
@@ -709,6 +712,11 @@ async def set_member_cap(
|
||||
)).scalar_one_or_none()
|
||||
if membership is None:
|
||||
raise HTTPException(status_code=404, detail="not a member of this org")
|
||||
if body.daily_call_cap >= 0 and membership.daily_call_cap < 0:
|
||||
# Unlimited members are not counted on the call path; give the counter today's journal so a
|
||||
# cap set mid-day starts from what they already used, not from zero.
|
||||
user = await db.get(User, user_id)
|
||||
await usage_policy.seed_counter(db, membership, user.email if user else "")
|
||||
membership.daily_call_cap = body.daily_call_cap
|
||||
await db.commit()
|
||||
return {"user_id": user_id, "org_id": org_id, "daily_call_cap": body.daily_call_cap}
|
||||
|
||||
@@ -1972,7 +1972,9 @@ details.tl li.more a{color:var(--link);text-decoration:none}
|
||||
|
||||
|
||||
@app.get("/tools/{service}", include_in_schema=False)
|
||||
async def tools_provider(service: str, db: AsyncSession = Depends(get_session)):
|
||||
async def tools_provider(service: str, db: AsyncSession = Depends(get_session),
|
||||
observations: endpoint_stats.EndpointObservationReader = Depends(
|
||||
_endpoint_observation_reader)):
|
||||
"""One provider's public page, in the use-case pages' skin (usecase.css): hero on the two
|
||||
measured terms — "{provider} api pricing" (what Search Console shows people typing) and
|
||||
"{provider} mcp" — the agent->treg->provider flow, setup (agent one-liner first), a prompt
|
||||
@@ -2032,7 +2034,7 @@ async def tools_provider(service: str, db: AsyncSession = Depends(get_session)):
|
||||
badge = "YOUR ACCOUNT" if is_oauth else "NO SIGNUP"
|
||||
# The measured line: what treg.to has actually observed calling this provider. It is the one
|
||||
# thing a vendor's own pricing page cannot print, and it goes above the fold for that reason.
|
||||
obs = await _observed_or_empty(db, [e["id"] for e in eps])
|
||||
obs = await _observed_or_empty(observations, [e["id"] for e in eps])
|
||||
o_samples = sum(int(o.get("samples") or 0) for o in obs.values())
|
||||
# The provider-wide rate weights each endpoint's published rate by the calls that DECIDED it
|
||||
# (2xx + 5xx). `samples` still counts callers' 4xx, so weighting by it would let one team's
|
||||
|
||||
@@ -1079,3 +1079,93 @@ async def test_same_key_recordings_allocate_distinct_versions(clients: AsyncClie
|
||||
keys, snaps = await _rows()
|
||||
assert len(keys) == 1
|
||||
assert [snap.version for snap in snaps] == list(range(1, 13))
|
||||
|
||||
|
||||
# ---- the 2026-09-07 OOM regression test: memory-bounded pending work ---------------------------
|
||||
# _MAX_PENDING_BYTES caps total body bytes held by pending tasks. Without it, 512 pending tasks ×
|
||||
# 8 MB bodies = 4 GB worst case — the exact OOM that killed production at 2026-09-07T00:43:06Z.
|
||||
|
||||
|
||||
async def test_pending_body_bytes_are_bounded_and_excess_is_shed(monkeypatch):
|
||||
"""The bytes bound sheds recordings before the count bound would — the 2026-09-07 OOM fix.
|
||||
|
||||
With _MAX_CONCURRENT_WRITES=2 and heavy traffic, pending tasks holding large bodies can
|
||||
accumulate faster than they drain. The bytes cap ensures total memory held by pending work
|
||||
never exceeds a threshold, regardless of how many tasks fit under _MAX_PENDING.
|
||||
"""
|
||||
import asyncio as aio
|
||||
|
||||
release = aio.Event()
|
||||
|
||||
async def blocked_store(**kw):
|
||||
await release.wait()
|
||||
|
||||
monkeypatch.setattr(archive, "_store_locked", blocked_store)
|
||||
monkeypatch.setattr(archive, "_sem", None)
|
||||
monkeypatch.setattr(archive, "_key_locks", None)
|
||||
monkeypatch.setattr(archive, "_pending_bytes", 0)
|
||||
archive._pending.clear()
|
||||
original_max_bytes = archive._MAX_PENDING_BYTES
|
||||
monkeypatch.setattr(archive, "_MAX_PENDING_BYTES", 1000)
|
||||
|
||||
common = dict(method="GET", endpoint_id=EP, provider="tikhub", caller_body=b"",
|
||||
headers={}, status_code=200, media_type="application/json")
|
||||
try:
|
||||
archive.record(url="https://api.example/1", body=b"x" * 400, **common)
|
||||
assert len(archive._pending) == 1
|
||||
assert archive._pending_bytes == 400
|
||||
|
||||
archive.record(url="https://api.example/2", body=b"y" * 400, **common)
|
||||
assert len(archive._pending) == 2
|
||||
assert archive._pending_bytes == 800
|
||||
|
||||
archive.record(url="https://api.example/3", body=b"z" * 300, **common)
|
||||
assert len(archive._pending) == 2, "third recording should be shed (800 + 300 > 1000)"
|
||||
assert archive._pending_bytes == 800
|
||||
|
||||
archive.record(url="https://api.example/4", body=b"w" * 150, **common)
|
||||
assert len(archive._pending) == 3, "fourth recording should fit (800 + 150 <= 1000)"
|
||||
assert archive._pending_bytes == 950
|
||||
finally:
|
||||
release.set()
|
||||
monkeypatch.setattr(archive, "_MAX_PENDING_BYTES", original_max_bytes)
|
||||
await aio.gather(*archive._pending, return_exceptions=True)
|
||||
archive._pending.clear()
|
||||
monkeypatch.setattr(archive, "_pending_bytes", 0)
|
||||
|
||||
|
||||
async def test_pending_bytes_released_when_task_completes(monkeypatch):
|
||||
"""The done callback releases body bytes so they can be reused by new recordings."""
|
||||
import asyncio as aio
|
||||
|
||||
release = aio.Event()
|
||||
entered = aio.Event()
|
||||
|
||||
async def blocking_store(**kw):
|
||||
entered.set()
|
||||
await release.wait()
|
||||
|
||||
monkeypatch.setattr(archive, "_store_locked", blocking_store)
|
||||
monkeypatch.setattr(archive, "_sem", None)
|
||||
monkeypatch.setattr(archive, "_key_locks", None)
|
||||
monkeypatch.setattr(archive, "_pending_bytes", 0)
|
||||
archive._pending.clear()
|
||||
|
||||
common = dict(method="GET", endpoint_id=EP, provider="tikhub", caller_body=b"",
|
||||
headers={}, status_code=200, media_type="application/json")
|
||||
try:
|
||||
archive.record(url="https://api.example/1", body=b"x" * 500, **common)
|
||||
await aio.wait_for(entered.wait(), timeout=1)
|
||||
assert archive._pending_bytes == 500
|
||||
|
||||
release.set()
|
||||
await aio.gather(*archive._pending, return_exceptions=True)
|
||||
await aio.sleep(0)
|
||||
|
||||
assert archive._pending_bytes == 0, "bytes should be released when task completes"
|
||||
assert len(archive._pending) == 0
|
||||
finally:
|
||||
release.set()
|
||||
await aio.gather(*archive._pending, return_exceptions=True)
|
||||
archive._pending.clear()
|
||||
monkeypatch.setattr(archive, "_pending_bytes", 0)
|
||||
|
||||
@@ -167,12 +167,17 @@ async def test_queued_worker_rows_are_not_claimed_before_a_poll_slot(
|
||||
await tick
|
||||
|
||||
|
||||
@pytest.mark.parametrize("deadline", ["POLL_TIMEOUT_S", "PROCESS_TIMEOUT_S"])
|
||||
# POLL_TIMEOUT_S wraps only the upstream poll, so it can be near-zero. PROCESS_TIMEOUT_S also
|
||||
# wraps the claim's own DB round trip: at 10 ms a loaded CI runner fires it BEFORE the claim,
|
||||
# which is correctly reported as backed_off with the row untouched - and then this test, which
|
||||
# wants the post-claim path, fails on `consecutive_failures == 1`. Give the claim room; the hung
|
||||
# poll is what the deadline must cut, and it never returns regardless.
|
||||
@pytest.mark.parametrize("deadline, seconds", [("POLL_TIMEOUT_S", 0.01), ("PROCESS_TIMEOUT_S", 0.5)])
|
||||
async def test_worker_bounds_whole_poll_and_keeps_hold_on_timeout(
|
||||
clients: AsyncClient, monkeypatch, replicate_platform, deadline,
|
||||
clients: AsyncClient, monkeypatch, replicate_platform, deadline, seconds,
|
||||
):
|
||||
call_id = await _due_submission(clients, monkeypatch, {})
|
||||
monkeypatch.setattr(task_app, deadline, 0.01)
|
||||
monkeypatch.setattr(task_app, deadline, seconds)
|
||||
async def hangs(row, client):
|
||||
await asyncio.Event().wait()
|
||||
monkeypatch.setattr(task_app, "_poll", hangs)
|
||||
|
||||
@@ -58,3 +58,106 @@ def test_drain_livelock_regression_fails_instead_of_wedging(name):
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", program], timeout=30, capture_output=True, text=True)
|
||||
assert result.returncode == 0, result.stderr[-500:]
|
||||
|
||||
|
||||
# ---- the batching writer (2026-09-07) ----------------------------------------------------------
|
||||
# One writer per process, `_BATCH` rows per INSERT. What the pool budget bought must not cost rows:
|
||||
# every row enqueued lands, a row the database refuses costs only itself, and shedding is the one
|
||||
# loss - on the queue, where rows wait now.
|
||||
|
||||
from treg.infra.db import reset_db, session_maker # noqa: E402
|
||||
from treg.models import SearchMiss # noqa: E402
|
||||
|
||||
|
||||
async def _misses() -> list[str]:
|
||||
from sqlalchemy import select
|
||||
async with session_maker() as db:
|
||||
return sorted((await db.execute(select(SearchMiss.query))).scalars().all())
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def clean_audit():
|
||||
await reset_db()
|
||||
await audit.drain()
|
||||
audit._queue.clear()
|
||||
yield
|
||||
await audit.drain()
|
||||
|
||||
|
||||
async def test_a_burst_lands_in_batches_on_one_writer(clean_audit, monkeypatch):
|
||||
"""More rows than one batch, enqueued at once: every one lands, and never more than one writer
|
||||
task existed to do it (the whole point - one `background` slot per process, not four)."""
|
||||
monkeypatch.setattr(audit, "_BATCH", 7)
|
||||
peak = 0
|
||||
real = audit._schedule
|
||||
|
||||
def counting(coro):
|
||||
nonlocal peak
|
||||
real(coro)
|
||||
peak = max(peak, len(audit._pending))
|
||||
monkeypatch.setattr(audit, "_schedule", counting)
|
||||
|
||||
for i in range(50):
|
||||
audit.record_search_miss(query=f"q{i:02d}", source="test")
|
||||
await audit.drain()
|
||||
assert await _misses() == [f"q{i:02d}" for i in range(50)]
|
||||
assert peak == 1
|
||||
|
||||
|
||||
async def test_a_refused_row_costs_only_itself(clean_audit, monkeypatch):
|
||||
"""A batch with one row the database rejects (NULL into a NOT NULL column) is retried row by
|
||||
row, so the 199 rows around it still land - the failure evidence of a burst must survive the
|
||||
burst."""
|
||||
monkeypatch.setattr(audit, "_BATCH", 10)
|
||||
for i in range(5):
|
||||
audit.record_search_miss(query=f"ok{i}", source="test")
|
||||
audit._enqueue(SearchMiss, dict(query=None, source="test")) # the bad row, mid-batch
|
||||
for i in range(5, 9):
|
||||
audit.record_search_miss(query=f"ok{i}", source="test")
|
||||
await audit.drain()
|
||||
assert await _misses() == [f"ok{i}" for i in range(9)]
|
||||
|
||||
|
||||
async def test_rows_past_the_queue_bound_are_shed_not_queued(clean_audit, monkeypatch):
|
||||
monkeypatch.setattr(audit, "_MAX_PENDING", 3)
|
||||
monkeypatch.setattr(audit, "_schedule", lambda coro: coro.close()) # nothing drains meanwhile
|
||||
before = audit._shed
|
||||
for i in range(5):
|
||||
audit.record_search_miss(query=f"s{i}", source="test")
|
||||
assert len(audit._queue) == 3
|
||||
assert audit._shed - before == 2
|
||||
audit._queue.clear()
|
||||
|
||||
|
||||
async def test_a_row_enqueued_as_the_writer_exits_still_lands(clean_audit, monkeypatch):
|
||||
"""The gap between the writer's last empty-queue check and its done-callback: a row enqueued
|
||||
there sees `_pending` occupied and starts nothing. The callback must start a writer for it, or
|
||||
the row waits for the next call - forever, on a quiet server.
|
||||
|
||||
The writer is made to finish inside its first step (a batch writer that never suspends), so a
|
||||
`call_soon` queued right behind that step runs after the task completed and BEFORE its done
|
||||
callbacks - exactly the gap.
|
||||
"""
|
||||
landed: list[str] = []
|
||||
|
||||
async def instant(rows):
|
||||
landed.extend(fields["query"] for _, fields in rows)
|
||||
return True
|
||||
monkeypatch.setattr(audit, "_write_batch", instant)
|
||||
|
||||
audit.record_search_miss(query="first", source="test")
|
||||
(writer,) = audit._pending
|
||||
seen: dict = {}
|
||||
|
||||
def late():
|
||||
audit.record_search_miss(query="late", source="test")
|
||||
seen["pending"] = set(audit._pending)
|
||||
seen["queued"] = len(audit._queue)
|
||||
asyncio.get_running_loop().call_soon(late)
|
||||
|
||||
await writer # resumes after the writer's done callbacks, `late` having run before them
|
||||
assert seen == {"pending": {writer}, "queued": 1}, "the late row did not hit the gap"
|
||||
# No drain: the done callback alone must have started a writer for the late row.
|
||||
assert audit._pending and writer not in audit._pending
|
||||
await asyncio.gather(*audit._pending)
|
||||
assert landed == ["first", "late"]
|
||||
|
||||
@@ -0,0 +1,337 @@
|
||||
"""The email-domain blocklist: throwaway mail and abusive signup domains, refused at every door.
|
||||
|
||||
A new team is created with a promotional balance, which makes bulk registration on throwaway
|
||||
addresses worth someone's while. Two tiers: a CODE tier (domains confirmed abusive in our own data,
|
||||
plus throwaway-mail keyword rules) and an OPS tier (`TREG_BLOCKED_EMAIL_DOMAINS`, additive, a
|
||||
dashboard edit so a new domain needs no deploy). Each rule is pinned here because the obvious
|
||||
implementation gets it wrong: match the DOMAIN only, walk parent domains but never the bare TLD,
|
||||
refuse sign-in as well as sign-up, cover BOTH doors that mint a promo-funded team, reveal nothing to
|
||||
the caller, count every block in the log, and fail open.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
import pytest
|
||||
from fastapi import FastAPI
|
||||
from httpx import ASGITransport, AsyncClient
|
||||
from sqlmodel import select
|
||||
|
||||
from treg.api import app
|
||||
from treg.application import signup
|
||||
from treg.config import get_settings
|
||||
from treg.domain.identity.access import _is_blocked_email
|
||||
from treg.infra.db import reset_db, session_maker
|
||||
from treg.models import Org, User
|
||||
|
||||
# The ops tier under test. Deliberately NOT the code tier's domains, so these tests prove the env
|
||||
# var itself works end to end; `.example` is reserved and can never be a real user's domain.
|
||||
OPS = "farm-a.example, Farm-B.example ,@farm-c.example,.farm-d.example"
|
||||
REFUSAL = "this address cannot be used to sign in"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ops(monkeypatch):
|
||||
"""Set the ops tier on the live Settings object (the shape conftest uses for `posthog_key`)."""
|
||||
def _set(raw: str = OPS) -> None:
|
||||
monkeypatch.setattr(get_settings(), "blocked_email_domains", raw, raising=False)
|
||||
return _set
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def client():
|
||||
await reset_db()
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as c:
|
||||
yield c
|
||||
|
||||
|
||||
async def _user_count(email: str) -> int:
|
||||
async with session_maker() as s:
|
||||
return len((await s.execute(select(User).where(User.email == email))).scalars().all())
|
||||
|
||||
|
||||
async def _otp_start(c: AsyncClient, email: str):
|
||||
return await c.post("/auth/email/start", json={"email": email})
|
||||
|
||||
|
||||
async def _otp_login(c: AsyncClient, email: str) -> str:
|
||||
code = (await _otp_start(c, email)).json()["dev_code"]
|
||||
r = await c.post("/auth/email/verify", json={"email": email, "code": code})
|
||||
assert r.status_code == 200, r.text
|
||||
return r.json()["token"]
|
||||
|
||||
|
||||
# ---- the classifier ------------------------------------------------------------------------------
|
||||
|
||||
def test_code_tier_is_in_force_with_no_setting_at_all(ops):
|
||||
ops("")
|
||||
assert get_settings().blocked_email_domain_set == frozenset()
|
||||
for email in ("a@uberip.com", "a@westcast-systems.com", "a@mailfox.win", "a@yopmail.com",
|
||||
"a@mail.uberip.com", # subdomain of a confirmed root
|
||||
"a@txtfromrizkirmdhn.my.id", # any `.my.id`, via the parent walk
|
||||
"a@tempmail-fresh.xyz", "a@guerrillamail.info", "a@x.10minutemail.net"): # keywords
|
||||
assert _is_blocked_email(email), email
|
||||
|
||||
|
||||
def test_match_is_on_the_domain_only_never_the_local_part(ops):
|
||||
"""The single most important rule. Matching the whole address false-flags real users whose
|
||||
USERNAME happens to contain a keyword, which is how a blocklist starts refusing customers."""
|
||||
ops()
|
||||
assert not _is_blocked_email("tempmail@gmail.com")
|
||||
assert not _is_blocked_email("yopmail.fan@company.dev")
|
||||
assert not _is_blocked_email("farm-a.example@company.dev")
|
||||
|
||||
|
||||
def test_ops_tier_parses_case_whitespace_and_leading_marks(ops):
|
||||
ops()
|
||||
assert get_settings().blocked_email_domain_set == frozenset(
|
||||
{"farm-a.example", "farm-b.example", "farm-c.example", "farm-d.example"})
|
||||
|
||||
|
||||
def test_ops_tier_matches_domain_and_subdomains_and_adds_to_the_code_tier(ops):
|
||||
ops()
|
||||
assert _is_blocked_email("a@farm-a.example")
|
||||
assert _is_blocked_email("A@FARM-B.EXAMPLE")
|
||||
assert _is_blocked_email("a@deep.mail.farm-c.example") # the subdomain bypass that must not work
|
||||
assert _is_blocked_email("a@farm-d.example") # listed as ".farm-d.example"
|
||||
assert _is_blocked_email("a@uberip.com") # the code tier is still there
|
||||
|
||||
|
||||
def test_walk_strips_whole_labels_off_the_front_only(ops):
|
||||
ops()
|
||||
assert not _is_blocked_email("a@notfarm-a.example") # a string suffix, not a subdomain
|
||||
assert not _is_blocked_email("a@farm-a.example.org") # the listed domain in the middle
|
||||
assert not _is_blocked_email("a@company.dev")
|
||||
assert not _is_blocked_email("a@uberip.co")
|
||||
|
||||
|
||||
def test_a_bare_public_suffix_can_never_be_an_entry(ops):
|
||||
"""`com` in the dashboard field must not refuse every address on earth."""
|
||||
ops("com, net, , @, .")
|
||||
assert get_settings().blocked_email_domain_set == frozenset()
|
||||
assert not _is_blocked_email("a@company.com")
|
||||
assert not _is_blocked_email("a@id") # the walk never tests the last label alone
|
||||
|
||||
|
||||
def test_the_decision_logs_one_countable_line_per_block(ops, caplog):
|
||||
ops()
|
||||
with caplog.at_level(logging.WARNING, logger="treg.auth"):
|
||||
assert signup.blocked_email("Farm@Mail.Farm-A.example", "otp_start")
|
||||
assert not signup.blocked_email("ok@company.dev", "otp_start")
|
||||
lines = [r.getMessage() for r in caplog.records if "signup_blocked_domain" in r.getMessage()]
|
||||
assert lines == ["event=signup_blocked_domain door=otp_start domain=mail.farm-a.example"]
|
||||
|
||||
|
||||
def test_the_decision_fails_open_on_a_classifier_error(monkeypatch, caplog):
|
||||
def boom(email):
|
||||
raise RuntimeError("bad blocklist")
|
||||
monkeypatch.setattr(signup, "_is_blocked_email", boom)
|
||||
with caplog.at_level(logging.ERROR, logger="treg.auth"):
|
||||
assert not signup.blocked_email("a@uberip.com", "otp_start") # the door stays open
|
||||
assert any("blocklist_error" in r.getMessage() for r in caplog.records)
|
||||
|
||||
|
||||
# ---- the email OTP door --------------------------------------------------------------------------
|
||||
|
||||
async def test_otp_start_refuses_both_tiers_and_mints_no_code(client, ops):
|
||||
ops()
|
||||
for email in ("farm@farm-a.example", "farm@mail.farm-a.example", "Farm@UBERIP.com",
|
||||
"farm@abc.my.id", "farm@tempmail-fresh.xyz"):
|
||||
r = await _otp_start(client, email)
|
||||
assert r.status_code == 403, (email, r.text)
|
||||
assert r.json()["detail"] == REFUSAL
|
||||
assert "dev_code" not in r.json() # no code minted, nothing to verify
|
||||
|
||||
|
||||
async def test_otp_refusal_names_no_list_and_no_domain(client, ops):
|
||||
ops()
|
||||
body = (await _otp_start(client, "farm@farm-a.example")).text.lower()
|
||||
for word in ("farm-a", "farm-b", "uberip", "block", "list", "domain"):
|
||||
assert word not in body
|
||||
|
||||
|
||||
async def test_otp_verify_refuses_a_code_minted_before_the_domain_was_listed(client, ops):
|
||||
email = "farm@farm-b.example"
|
||||
code = (await _otp_start(client, email)).json()["dev_code"] # not yet listed: code issued
|
||||
ops()
|
||||
r = await client.post("/auth/email/verify", json={"email": email, "code": code})
|
||||
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
|
||||
assert "treg_session" not in r.headers.get("set-cookie", "")
|
||||
assert await _user_count(email) == 0 # refused BEFORE the row is created
|
||||
|
||||
|
||||
async def test_otp_refuses_sign_in_of_an_account_that_predates_the_listing(client, ops):
|
||||
"""Sign-in, not only sign-up: an existing account on a listed domain gets no new session."""
|
||||
email = "early@farm-d.example"
|
||||
await _otp_login(client, email)
|
||||
assert await _user_count(email) == 1
|
||||
ops()
|
||||
assert (await _otp_start(client, email)).status_code == 403
|
||||
|
||||
|
||||
async def test_otp_still_works_for_an_unlisted_domain_while_the_list_is_set(client, ops):
|
||||
ops()
|
||||
assert await _otp_login(client, "real@company.dev")
|
||||
assert await _otp_login(client, "tempmail@company.dev") # local part is never looked at
|
||||
|
||||
|
||||
# ---- open registration (POST /users: user + org + the $1 promo in one call) -----------------------
|
||||
|
||||
async def test_open_registration_refuses_a_blocked_domain_and_creates_nothing(client, ops):
|
||||
ops()
|
||||
for email in ("farm@sub.farm-a.example", "farm@sub.uberip.com"):
|
||||
r = await client.post("/users", json={"email": email})
|
||||
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
|
||||
assert await _user_count(email) == 0
|
||||
async with session_maker() as s:
|
||||
assert (await s.execute(select(Org))).scalars().all() == [] # no team, so no grant
|
||||
|
||||
|
||||
async def test_open_registration_is_unchanged_for_an_unlisted_domain(client, ops):
|
||||
ops("")
|
||||
r = await client.post("/users", json={"email": "someone@farm-a.example"})
|
||||
assert r.status_code == 200, r.text
|
||||
assert r.json()["email"] == "someone@farm-a.example"
|
||||
|
||||
|
||||
# ---- creating a team (POST /orgs: the other promo door, for an already-registered identity) -------
|
||||
|
||||
async def test_create_org_refuses_an_identity_registered_before_its_domain_was_listed(client, ops):
|
||||
tok = await _otp_login(client, "early@farm-a.example")
|
||||
ops()
|
||||
r = await client.post("/orgs", json={"name": "Farm 1"}, headers={"X-Treg-Token": tok})
|
||||
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
|
||||
async with session_maker() as s:
|
||||
assert (await s.execute(select(Org))).scalars().all() == []
|
||||
|
||||
|
||||
# ---- the social doors (GitHub, Google) -----------------------------------------------------------
|
||||
|
||||
def _github_idp(email: str) -> FastAPI:
|
||||
g = FastAPI()
|
||||
|
||||
@g.post("/login/oauth/access_token")
|
||||
async def token() -> dict:
|
||||
return {"access_token": "gho_test", "token_type": "bearer"}
|
||||
|
||||
@g.get("/user")
|
||||
async def user() -> dict:
|
||||
return {"login": "farm", "email": email}
|
||||
|
||||
return g
|
||||
|
||||
|
||||
def _google_idp(email: str) -> FastAPI:
|
||||
g = FastAPI()
|
||||
|
||||
@g.post("/token")
|
||||
async def token() -> dict:
|
||||
return {"access_token": "goog_test", "token_type": "bearer"}
|
||||
|
||||
@g.get("/userinfo")
|
||||
async def userinfo() -> dict:
|
||||
return {"email": email, "email_verified": True}
|
||||
|
||||
return g
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def social(monkeypatch):
|
||||
"""Both social doors configured against in-process fake identity providers. The ops tier goes
|
||||
in through the environment, as an operator would set it."""
|
||||
monkeypatch.setenv("TREG_GITHUB_CLIENT_ID", "cid")
|
||||
monkeypatch.setenv("TREG_GITHUB_CLIENT_SECRET", "csec")
|
||||
monkeypatch.setenv("TREG_GITHUB_TOKEN_URL", "http://idp/login/oauth/access_token")
|
||||
monkeypatch.setenv("TREG_GITHUB_API_URL", "http://idp")
|
||||
monkeypatch.setenv("TREG_GOOGLE_CLIENT_ID", "gid")
|
||||
monkeypatch.setenv("TREG_GOOGLE_CLIENT_SECRET", "gsec")
|
||||
monkeypatch.setenv("TREG_GOOGLE_TOKEN_URL", "http://idp/token")
|
||||
monkeypatch.setenv("TREG_GOOGLE_USERINFO_URL", "http://idp/userinfo")
|
||||
monkeypatch.setenv("TREG_SESSION_SECRET", "test-session-secret")
|
||||
monkeypatch.setenv("TREG_BLOCKED_EMAIL_DOMAINS", OPS)
|
||||
get_settings.cache_clear()
|
||||
await reset_db()
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as c:
|
||||
yield c
|
||||
if getattr(app.state, "http", None) is not None:
|
||||
await app.state.http.aclose()
|
||||
get_settings.cache_clear()
|
||||
|
||||
|
||||
async def _social_callback(c: AsyncClient, door: str, idp: FastAPI):
|
||||
app.state.http = AsyncClient(transport=ASGITransport(app=idp), base_url="http://idp")
|
||||
r = await c.get(f"/auth/{door}", follow_redirects=False)
|
||||
assert r.status_code == 302
|
||||
state = c.cookies.get("treg_oauth_state")
|
||||
return await c.get(f"/auth/{door}/callback?code=abc&state={state}", follow_redirects=False)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("door,email", [
|
||||
("github", "farm@farm-a.example"), # ops tier
|
||||
("google", "farm@mail.uberip.com"), # code tier, subdomain
|
||||
])
|
||||
async def test_social_login_on_a_blocked_domain_gets_a_refusal_page_and_no_session(social, door, email):
|
||||
idp = _github_idp(email) if door == "github" else _google_idp(email)
|
||||
cb = await _social_callback(social, door, idp)
|
||||
assert cb.status_code == 403, cb.text
|
||||
assert "cannot be used to sign in" in cb.text
|
||||
assert "farm-a" not in cb.text and "uberip" not in cb.text
|
||||
assert "treg_session" not in cb.headers.get("set-cookie", "")
|
||||
assert (await social.get("/auth/me")).status_code == 401
|
||||
assert await _user_count(email) == 0
|
||||
|
||||
|
||||
async def test_social_login_on_an_unlisted_domain_still_signs_in(social):
|
||||
cb = await _social_callback(social, "google", _google_idp("ok@company.dev"))
|
||||
assert cb.status_code == 302 and cb.headers["location"] == "/app"
|
||||
assert (await social.get("/auth/me")).json()["email"] == "ok@company.dev"
|
||||
|
||||
|
||||
# ---- invites: the emailed link (mints a session) and the code (mints a membership token) ----------
|
||||
|
||||
@pytest.fixture
|
||||
def sent_invites(monkeypatch):
|
||||
from treg import email as email_mod
|
||||
sent = []
|
||||
|
||||
async def _capture(email, inviter, org_name, role, code, email_token, expires_at="", link_base="",
|
||||
shared=""):
|
||||
sent.append({"email": email, "code": code, "email_token": email_token})
|
||||
return True
|
||||
|
||||
monkeypatch.setattr(email_mod, "send_invite", _capture)
|
||||
return sent
|
||||
|
||||
|
||||
async def _team_with_invite(c: AsyncClient, owner: str, invitee: str) -> dict:
|
||||
tok = await _otp_login(c, owner)
|
||||
org = (await c.post("/orgs", json={"name": "Real Team"}, headers={"X-Treg-Token": tok})).json()
|
||||
r = await c.post(f"/orgs/{org['org_id']}/invites", json={"email": invitee, "role": "member"},
|
||||
headers={"X-Treg-Token": tok, "X-Treg-Org": org["org"]})
|
||||
assert r.status_code == 200, r.text
|
||||
return org
|
||||
|
||||
|
||||
async def test_emailed_invite_link_refuses_a_blocked_domain(client, ops, sent_invites):
|
||||
await _team_with_invite(client, "owner@company.dev", "farm@farm-a.example")
|
||||
ops()
|
||||
t = sent_invites[0]["email_token"]
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as visitor:
|
||||
r = await visitor.post("/auth/invite-signin", content=f"t={t}",
|
||||
headers={"content-type": "application/x-www-form-urlencoded"},
|
||||
follow_redirects=False)
|
||||
assert r.status_code == 403 and "cannot be used to sign in" in r.text
|
||||
assert "treg_session" not in r.headers.get("set-cookie", "")
|
||||
assert (await visitor.get("/invites/mine")).status_code == 401
|
||||
assert await _user_count("farm@farm-a.example") == 0
|
||||
|
||||
|
||||
async def test_invite_code_accept_refuses_a_blocked_domain(client, ops, sent_invites):
|
||||
await _team_with_invite(client, "owner@company.dev", "farm@sub.farm-b.example")
|
||||
ops()
|
||||
r = await client.post("/invites/accept",
|
||||
json={"code": sent_invites[0]["code"], "email": "farm@sub.farm-b.example"})
|
||||
assert r.status_code == 403 and r.json()["detail"] == REFUSAL
|
||||
assert "token" not in r.json()
|
||||
assert await _user_count("farm@sub.farm-b.example") == 0
|
||||
@@ -14,6 +14,7 @@ from treg.application import billing
|
||||
from treg.application.call import authorize, overflow, reserve, service, settle
|
||||
from treg.domain import money
|
||||
from treg.domain.capacity import marks as capacity_marks
|
||||
from treg.domain.governance import usage as usage_policy
|
||||
|
||||
|
||||
_SRC = Path(__file__).parents[1] / "src" / "treg"
|
||||
@@ -64,6 +65,12 @@ _DATAPLANE_DERIVED_WRITES = {
|
||||
"async_resource_ownership": (
|
||||
(service._execute_call, "async_task_app.remember_platform_resources"),
|
||||
),
|
||||
# The per-user daily cap takes its slot with one conditional UPDATE of the member's row
|
||||
# (revision 0024) instead of counting the member's callrecord rows per call.
|
||||
"member_daily_cap_slot": (
|
||||
(authorize.authorize_call, "usage_policy.enforce_daily_cap"),
|
||||
(usage_policy.enforce_daily_cap, "take_daily_slot"),
|
||||
),
|
||||
}
|
||||
_EXPECTED_DATAPLANE_WRITES = frozenset({
|
||||
"auto_topup_task",
|
||||
@@ -76,12 +83,14 @@ _EXPECTED_DATAPLANE_WRITES = frozenset({
|
||||
"overflow_budget_reservation",
|
||||
"async_result_ownership",
|
||||
"async_resource_ownership",
|
||||
"member_daily_cap_slot",
|
||||
})
|
||||
_DERIVED_WRITE_FILES = {
|
||||
_SRC / "application" / "billing.py": {"loop.create_task"},
|
||||
_SRC / "application" / "call" / "authorize.py": {
|
||||
"publicdemo_policy.enforce_public_demo_ip_cap",
|
||||
"publicdemo_policy.enforce_public_demo_ip_cap", "usage_policy.enforce_daily_cap",
|
||||
},
|
||||
_SRC / "domain" / "governance" / "usage.py": {"take_daily_slot"},
|
||||
_SRC / "application" / "call" / "reserve.py": {"billing.maybe_schedule_autotopup"},
|
||||
_SRC / "application" / "call" / "settle.py": {
|
||||
"adsconv.queue", "capacity_marks.strike", "capacity_marks.clear",
|
||||
@@ -108,6 +117,8 @@ _EXPECTED_DERIVED_WRITE_SITES = {
|
||||
"publicdemo_policy.enforce_public_demo_ip_cap"),
|
||||
("application/call/authorize.py", "enforce_public_demo_limit",
|
||||
"publicdemo_policy.enforce_public_demo_ip_cap"),
|
||||
("application/call/authorize.py", "authorize_call", "usage_policy.enforce_daily_cap"),
|
||||
("domain/governance/usage.py", "enforce_daily_cap", "take_daily_slot"),
|
||||
("application/call/reserve.py", "_platform_reserve",
|
||||
"billing.maybe_schedule_autotopup"),
|
||||
("application/call/settle.py", "_record_first_call", "adsconv.queue"),
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
"""`Org.spent_today_micro` is the daily cap's number and must agree with the journal.
|
||||
|
||||
The counter exists so the fail-closed daily cap costs one primary-key read on every metered call
|
||||
instead of an aggregate over the platform's whole day (revision 0022). These tests pin the
|
||||
semantics the journal view (`spent_today_from_ledger`) has always had: settled today plus still
|
||||
held from today, reset at the UTC day boundary, with a hold opened yesterday belonging to yesterday.
|
||||
"""
|
||||
from datetime import date, timedelta
|
||||
|
||||
import pytest
|
||||
from httpx import ASGITransport, AsyncClient
|
||||
from sqlalchemy import update
|
||||
|
||||
from treg.api import app
|
||||
from treg.domain import money as ledger
|
||||
from treg.infra.db import reset_db, session_maker
|
||||
from treg.models import Hold, Org
|
||||
from conftest import make_upstream
|
||||
|
||||
EP = "acme.thing.get"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def c():
|
||||
await reset_db()
|
||||
app.state.http = AsyncClient(transport=ASGITransport(app=make_upstream()), base_url="http://upstream")
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://registry") as client:
|
||||
yield client
|
||||
await app.state.http.aclose()
|
||||
|
||||
|
||||
async def _org(c: AsyncClient) -> int:
|
||||
r = await c.post("/users", json={"email": "counter@superdesign.dev"})
|
||||
assert r.status_code == 200, r.text
|
||||
return r.json()["org_id"]
|
||||
|
||||
|
||||
async def _both(org_id: int) -> tuple[int, int]:
|
||||
"""(counter, journal) — every assertion below checks them against each other too."""
|
||||
async with session_maker() as db:
|
||||
return await ledger.spent_today(db, org_id), await ledger.spent_today_from_ledger(db, org_id)
|
||||
|
||||
|
||||
async def test_counter_follows_reserve_settle_and_release(c: AsyncClient):
|
||||
org_id = await _org(c)
|
||||
assert await _both(org_id) == (0, 0)
|
||||
|
||||
async with session_maker() as db:
|
||||
a = await ledger.reserve(db, org_id, EP, 1_000)
|
||||
b = await ledger.reserve(db, org_id, EP, 2_000)
|
||||
held_a, held_b = ledger.with_margin(1_000), ledger.with_margin(2_000)
|
||||
assert await _both(org_id) == (held_a + held_b, held_a + held_b)
|
||||
|
||||
# Settling below the estimate: the hold leaves, what was consumed stays.
|
||||
async with session_maker() as db:
|
||||
consumed = await ledger.settle(db, a, 400)
|
||||
assert consumed == ledger.with_margin(400)
|
||||
assert await _both(org_id) == (consumed + held_b, consumed + held_b)
|
||||
|
||||
# Releasing: the hold leaves and nothing replaces it.
|
||||
async with session_maker() as db:
|
||||
await ledger.release(db, b, reason="upstream 503")
|
||||
assert await _both(org_id) == (consumed, consumed)
|
||||
|
||||
# A second settle of a closed hold is a no-op for the counter, as for the balance.
|
||||
async with session_maker() as db:
|
||||
assert await ledger.settle(db, a, 999) == 0
|
||||
assert await _both(org_id) == (consumed, consumed)
|
||||
|
||||
|
||||
async def test_settle_overrun_counts_what_was_actually_consumed(c: AsyncClient):
|
||||
org_id = await _org(c)
|
||||
async with session_maker() as db:
|
||||
call = await ledger.reserve(db, org_id, EP, 1_000)
|
||||
consumed = await ledger.settle(db, call, 5_000) # the provider charged more than estimated
|
||||
assert consumed == ledger.with_margin(5_000)
|
||||
assert await _both(org_id) == (consumed, consumed)
|
||||
|
||||
|
||||
async def test_counter_resets_on_a_new_utc_day(c: AsyncClient):
|
||||
org_id = await _org(c)
|
||||
async with session_maker() as db:
|
||||
call = await ledger.reserve(db, org_id, EP, 1_000)
|
||||
await ledger.settle(db, call, 1_000)
|
||||
spent = ledger.with_margin(1_000)
|
||||
assert await _both(org_id) == (spent, spent)
|
||||
|
||||
# Move the counter to "yesterday" - as the day rolling over would leave it - and it reads 0.
|
||||
yesterday = date.today() - timedelta(days=1)
|
||||
async with session_maker() as db:
|
||||
await db.execute(update(Org).where(Org.id == org_id).values(spent_today_day=yesterday))
|
||||
await db.commit()
|
||||
async with session_maker() as db:
|
||||
assert await ledger.spent_today(db, org_id) == 0
|
||||
|
||||
# The first movement of the new day starts from that movement, not from yesterday's total.
|
||||
async with session_maker() as db:
|
||||
await ledger.reserve(db, org_id, EP, 300)
|
||||
async with session_maker() as db:
|
||||
assert await ledger.spent_today(db, org_id) == ledger.with_margin(300)
|
||||
|
||||
|
||||
async def test_a_hold_opened_yesterday_belongs_to_yesterday(c: AsyncClient):
|
||||
"""Settling or releasing an older hold must not subtract from today what today never counted."""
|
||||
org_id = await _org(c)
|
||||
async with session_maker() as db:
|
||||
old = await ledger.reserve(db, org_id, EP, 1_000)
|
||||
await ledger.reserve(db, org_id, EP, 2_000)
|
||||
# Age the first hold: its row says yesterday, and the counter no longer includes it.
|
||||
async with session_maker() as db:
|
||||
await db.execute(update(Hold).where(Hold.id == old).values(
|
||||
created_at=ledger._now() - timedelta(days=1)))
|
||||
await db.execute(update(Org).where(Org.id == org_id).values(
|
||||
spent_today_micro=ledger.with_margin(2_000)))
|
||||
await db.commit()
|
||||
assert await _both(org_id) == (ledger.with_margin(2_000), ledger.with_margin(2_000))
|
||||
|
||||
async with session_maker() as db:
|
||||
consumed = await ledger.settle(db, old, 500) # settled TODAY, so what it consumed counts today
|
||||
expect = ledger.with_margin(2_000) + consumed
|
||||
assert await _both(org_id) == (expect, expect)
|
||||
|
||||
async with session_maker() as db:
|
||||
older = await ledger.reserve(db, org_id, EP, 700)
|
||||
await db.execute(update(Hold).where(Hold.id == older).values(
|
||||
created_at=ledger._now() - timedelta(days=1)))
|
||||
await db.execute(update(Org).where(Org.id == org_id).values(spent_today_micro=expect))
|
||||
await db.commit()
|
||||
async with session_maker() as db:
|
||||
await ledger.release(db, older, reason="stale") # nothing counted today, nothing leaves
|
||||
assert await _both(org_id) == (expect, expect)
|
||||
@@ -0,0 +1,67 @@
|
||||
"""The pool gauge is the reading behind `TREG_DB_POOL_OVERRIDES`: per-minute peak checked-out
|
||||
connections per pool, next to capacity. Sizing by arithmetic got both minor pools wrong once each
|
||||
(ops/deploy.md § Three pools); this is what settles the number."""
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
|
||||
from treg import analytics, bootstrap
|
||||
from treg.infra import db as infra_db
|
||||
|
||||
|
||||
def test_snapshot_reports_nothing_on_sqlite_and_a_pool_row_per_engine_otherwise():
|
||||
snap = infra_db.pool_snapshot()
|
||||
if infra_db._is_sqlite:
|
||||
assert snap == {}
|
||||
return
|
||||
assert set(snap) == {"api", "admin", "background"}
|
||||
for name, row in snap.items():
|
||||
spec = infra_db.POOL_SPECS[name]
|
||||
assert row["capacity"] == spec["pool_size"] + spec["max_overflow"]
|
||||
assert 0 <= row["checked_out"] <= row["capacity"]
|
||||
|
||||
|
||||
def test_fold_keeps_the_per_pool_maximum():
|
||||
peaks: dict[str, int] = {}
|
||||
infra_db.fold_pool_peaks(peaks, {"api": {"checked_out": 3, "capacity": 15}})
|
||||
infra_db.fold_pool_peaks(peaks, {"api": {"checked_out": 9, "capacity": 15},
|
||||
"background": {"checked_out": 2, "capacity": 13}})
|
||||
infra_db.fold_pool_peaks(peaks, {"api": {"checked_out": 1, "capacity": 15}})
|
||||
assert peaks == {"api": 9, "background": 2}
|
||||
|
||||
|
||||
async def test_gauge_emits_one_event_per_window_with_peak_capacity_and_headroom(monkeypatch):
|
||||
samples = iter([
|
||||
{"api": {"checked_out": 4, "capacity": 15}, "background": {"checked_out": 1, "capacity": 13}},
|
||||
{"api": {"checked_out": 11, "capacity": 15}, "background": {"checked_out": 13, "capacity": 13}},
|
||||
{"api": {"checked_out": 2, "capacity": 15}, "background": {"checked_out": 0, "capacity": 13}},
|
||||
])
|
||||
last = {"api": {"checked_out": 0, "capacity": 15}, "background": {"checked_out": 0, "capacity": 13}}
|
||||
monkeypatch.setattr(infra_db, "pool_snapshot", lambda: next(samples, last))
|
||||
captured: list[tuple[str, str, dict]] = []
|
||||
monkeypatch.setattr(analytics, "capture", lambda d, e, p=None, **kw: captured.append((d, e, p)))
|
||||
task = asyncio.create_task(bootstrap.pool_gauge(sample_s=0.01, emit_s=0.05))
|
||||
try:
|
||||
for _ in range(100):
|
||||
await asyncio.sleep(0.01)
|
||||
if captured:
|
||||
break
|
||||
finally:
|
||||
task.cancel()
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await task
|
||||
assert captured, "no gauge event within the window"
|
||||
distinct_id, event, props = captured[0]
|
||||
assert (distinct_id, event) == (analytics.SERVER_DISTINCT_ID, "db_pool_gauge")
|
||||
assert props["api_peak"] == 11 and props["api_capacity"] == 15 and props["api_headroom"] == 4
|
||||
assert props["background_peak"] == 13 and props["background_headroom"] == 0
|
||||
assert props["samples"] >= 3
|
||||
|
||||
|
||||
def test_connection_budget_multiplies_per_process_by_workers_and_the_deploy_overlap():
|
||||
per_process = sum(s["pool_size"] + s["max_overflow"] for s in infra_db.POOL_SPECS.values())
|
||||
one = infra_db.connection_budget(workers=1)
|
||||
two = infra_db.connection_budget(workers=2)
|
||||
assert one == {"per_process": per_process, "workers": 1,
|
||||
"per_instance": per_process, "deploy_peak": per_process * 2}
|
||||
assert two["per_instance"] == per_process * 2 and two["deploy_peak"] == per_process * 4
|
||||
@@ -118,7 +118,7 @@ def test_the_background_pool_fits_every_consumer_not_just_the_throttled_ones():
|
||||
from treg import archive, audit
|
||||
|
||||
consumers = infra_db.BACKGROUND_CONSUMERS
|
||||
assert consumers["audit._write"] == audit._MAX_CONCURRENT_WRITES
|
||||
assert consumers["audit._flush"] == audit._MAX_CONCURRENT_WRITES
|
||||
assert consumers["archive._store/_touch"] == archive._MAX_CONCURRENT_WRITES
|
||||
assert infra_db.POOL_SPECS["background"]["pool_size"] >= sum(consumers.values())
|
||||
|
||||
|
||||
@@ -82,3 +82,14 @@ async def test_sitemap_and_catalog_link_every_provider_page(clients: AsyncClient
|
||||
assert f"/tools/{s}<" in sm, s
|
||||
assert f'href="/tools/{s}"' in cat, s
|
||||
assert "/pricing<" in sm
|
||||
|
||||
|
||||
async def test_provider_page_reads_the_observation_reader_not_the_session(clients, caplog):
|
||||
"""`/tools/{service}` once handed `_observed_or_empty` the request's AsyncSession instead of the
|
||||
app's observation reader; the measured line silently came up empty and every page view logged
|
||||
an AttributeError traceback (prod, 2026-09-06)."""
|
||||
import logging
|
||||
with caplog.at_level(logging.WARNING, logger="treg.catalog"):
|
||||
r = await clients.get("/tools/dataforseo")
|
||||
assert r.status_code == 200
|
||||
assert "endpoint stats unavailable" not in caplog.text
|
||||
|
||||
@@ -7,6 +7,8 @@ SEPARATE session — if the state were still a per-process dict, the second sess
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from sqlalchemy.dialects import postgresql, sqlite
|
||||
|
||||
from treg import ratestore
|
||||
from treg.infra.db import reset_db, session_maker
|
||||
from treg.models import Ephemeral
|
||||
@@ -94,3 +96,27 @@ async def test_sweep_drops_expired_rows_only():
|
||||
await db.commit()
|
||||
assert await db.get(Ephemeral, ("sandbox_hit", "dead")) is None
|
||||
assert await db.get(Ephemeral, ("sandbox_hit", "live")) is not None
|
||||
|
||||
|
||||
async def test_kv_put_of_a_missing_key_is_an_upsert_on_both_backends():
|
||||
"""Two processes striking the same provider write the same lock key in the same millisecond; the
|
||||
loser of a plain INSERT died on `ephemeral_pkey` (prod, 2026-09-06). The statement must carry
|
||||
ON CONFLICT on both dialects so the last writer wins instead of raising."""
|
||||
from datetime import datetime
|
||||
for dialect in ("postgresql", "sqlite"):
|
||||
stmt = ratestore._upsert(dialect, ns="capacity:lock", k="p", v={"a": 1},
|
||||
expires_at=datetime(2026, 1, 1))
|
||||
sql = str(stmt.compile(dialect=(postgresql.dialect() if dialect == "postgresql" else sqlite.dialect())))
|
||||
assert "ON CONFLICT (ns, k) DO UPDATE" in sql, sql
|
||||
|
||||
|
||||
async def test_kv_put_overwrites_a_row_written_by_another_session():
|
||||
async with session_maker() as other:
|
||||
await ratestore.kv_put(other, "ns", "k", {"n": 1}, ttl_s=600)
|
||||
await other.commit()
|
||||
async with session_maker() as db:
|
||||
# This session has never loaded (ns, k); its `get` sees the committed row and updates it.
|
||||
await ratestore.kv_put(db, "ns", "k", {"n": 2}, ttl_s=600)
|
||||
await db.commit()
|
||||
async with session_maker() as db:
|
||||
assert (await ratestore.kv_get(db, "ns", "k")) == {"n": 2}
|
||||
|
||||
@@ -677,10 +677,11 @@ async def test_a_pin_cannot_smuggle_what_the_header_cannot(clients: AsyncClient)
|
||||
|
||||
# ---- the invariants an invoice actually rests on -------------------------------------------------
|
||||
async def test_usage_survives_a_dead_audit_pipeline(clients: AsyncClient, platform_on, monkeypatch):
|
||||
"""THE test that proves an invoice never depends on a lossy table. `audit._schedule` sheds rows
|
||||
past its queue bound and swallows every exception — precisely under the load a successful builder
|
||||
generates. With the audit pipeline entirely dead, the money must still be complete."""
|
||||
monkeypatch.setattr(audit, "_schedule", lambda coro: coro.close())
|
||||
"""THE test that proves an invoice never depends on a lossy table. `audit._enqueue` sheds rows
|
||||
past its queue bound and the writer swallows every exception — precisely under the load a
|
||||
successful builder generates. With the audit pipeline entirely dead, the money must still be
|
||||
complete."""
|
||||
monkeypatch.setattr(audit, "_enqueue", lambda model, fields: None)
|
||||
org_id = await _org_id(clients)
|
||||
for who in ("cust_A", "cust_B"):
|
||||
r = await clients.get(f"/call/{EP}?aweme_id=7", headers={"X-Treg-Meta": f"customer={who}"})
|
||||
|
||||
+61
-10
@@ -6,7 +6,8 @@ Soft by design (best-effort audit → fails open), so these tests seed records d
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import timedelta
|
||||
from datetime import date, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
from httpx import AsyncClient
|
||||
from sqlmodel import select
|
||||
@@ -64,22 +65,42 @@ async def test_unlimited_by_default(clients: AsyncClient):
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200
|
||||
|
||||
|
||||
async def test_runs_and_grants_count_toward_the_same_cap(clients: AsyncClient):
|
||||
"""A member can't dodge the cap by switching path: CallRecord (call+local) AND RunRecord (server)
|
||||
both count. Seed one of each to reach cap=2, then a proxy call is refused."""
|
||||
async def _counter(org_id: int, email: str) -> tuple[int, date | None]:
|
||||
async with session_maker() as s:
|
||||
uid = (await s.execute(select(User.id).where(User.email == email))).scalar_one()
|
||||
m = (await s.execute(select(Membership).where(
|
||||
Membership.user_id == uid, Membership.org_id == org_id))).scalar_one()
|
||||
return m.calls_today, m.calls_today_day
|
||||
|
||||
|
||||
async def _set_counter(org_id: int, email: str, n: int, day: date) -> None:
|
||||
async with session_maker() as s:
|
||||
uid = (await s.execute(select(User.id).where(User.email == email))).scalar_one()
|
||||
m = (await s.execute(select(Membership).where(
|
||||
Membership.user_id == uid, Membership.org_id == org_id))).scalar_one()
|
||||
m.calls_today, m.calls_today_day = n, day
|
||||
await s.commit()
|
||||
|
||||
|
||||
async def test_runs_and_calls_share_one_gate(clients: AsyncClient):
|
||||
"""A member can't dodge the cap by switching path: both run handlers in api.py go through the
|
||||
same `_enforce_daily_cap` door as `/call/` (authorize.py), and that door is what moves the
|
||||
counter. Pinned statically because the run surfaces need a bundle to exercise end to end."""
|
||||
src = (Path(__file__).parents[1] / "src" / "treg" / "api.py").read_text()
|
||||
assert src.count("await _enforce_daily_cap(caller, db)") == 2 # local-run grant + server run
|
||||
await _mk_echo_tool(clients)
|
||||
org_id = await _set_cap(2)
|
||||
await _seed_call(org_id, "tim@superdesign.dev") # 1 (a prior proxy/local event)
|
||||
await _seed_run(org_id, "tim@superdesign.dev") # 2 (a prior server run)
|
||||
await _set_counter(org_id, "tim@superdesign.dev", 2, date.today()) # two prior events today
|
||||
blocked = await clients.get("/call/echo/anything")
|
||||
assert blocked.status_code == 429 # used=2 (call+run) >= cap=2
|
||||
assert blocked.status_code == 429 and "2/2" in blocked.json()["detail"]
|
||||
assert await _counter(org_id, "tim@superdesign.dev") == (2, date.today()) # refused = not counted
|
||||
|
||||
|
||||
async def test_cap_is_per_member_not_global(clients: AsyncClient):
|
||||
"""Capping one member must not affect another in the same org."""
|
||||
await _mk_echo_tool(clients)
|
||||
org_id = await _set_cap(1)
|
||||
await _seed_call(org_id, "tim@superdesign.dev") # tim is now at his cap
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200 # tim is now at his cap
|
||||
# invite bob into the SAME org (default cap -1)
|
||||
code = (await clients.post(f"/orgs/{org_id}/invites", json={"email": "bob@x.io", "role": "member"})).json()["code"]
|
||||
btok = (await clients.post("/invites/accept", json={"code": code, "email": "bob@x.io"})).json()["token"]
|
||||
@@ -92,8 +113,38 @@ async def test_cap_is_per_member_not_global(clients: AsyncClient):
|
||||
async def test_yesterdays_usage_does_not_count_today(clients: AsyncClient):
|
||||
await _mk_echo_tool(clients)
|
||||
org_id = await _set_cap(1)
|
||||
await _seed_call(org_id, "tim@superdesign.dev", days_ago=1) # yesterday — outside today's window
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200 # today's count is still 0
|
||||
await _set_counter(org_id, "tim@superdesign.dev", 1, date.today() - timedelta(days=1)) # yesterday's
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200 # a new day starts from 0
|
||||
assert await _counter(org_id, "tim@superdesign.dev") == (1, date.today()) # ...and this was its first
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 429
|
||||
|
||||
|
||||
async def test_setting_a_cap_seeds_the_counter_from_todays_journal(clients: AsyncClient):
|
||||
"""Unlimited members are not counted on the call path, so a cap set mid-day starts from what the
|
||||
journal says they already used - not from zero."""
|
||||
await _mk_echo_tool(clients)
|
||||
org_id = await _get_org_id()
|
||||
for _ in range(3):
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200
|
||||
await audit.drain()
|
||||
await _seed_run(org_id, "tim@superdesign.dev") # a server run counts too: 4 events in the journal
|
||||
uid = [x["user_id"] for x in (await clients.get(f"/orgs/{org_id}/members")).json()
|
||||
if x["email"] == "tim@superdesign.dev"][0]
|
||||
assert (await clients.patch(f"/orgs/{org_id}/members/{uid}/cap", json={"daily_call_cap": 5})).status_code == 200
|
||||
assert await _counter(org_id, "tim@superdesign.dev") == (4, date.today())
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200 # 5th
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 429 # 6th
|
||||
|
||||
|
||||
async def test_counter_agrees_with_the_journal_after_real_calls(clients: AsyncClient):
|
||||
await _mk_echo_tool(clients)
|
||||
org_id = await _set_cap(10)
|
||||
for _ in range(4):
|
||||
assert (await clients.get("/call/echo/anything")).status_code == 200
|
||||
await audit.drain()
|
||||
async with session_maker() as s:
|
||||
journal = await count_today(s, org_id, "tim@superdesign.dev")
|
||||
assert await _counter(org_id, "tim@superdesign.dev") == (journal, date.today()) == (4, date.today())
|
||||
|
||||
|
||||
async def _get_org_id(email: str = "tim@superdesign.dev") -> int:
|
||||
|
||||
Reference in New Issue
Block a user