feat(fast): /fast ultrafast for GPT-6 Astra, priced at the served tier

OpenAI's Ultrafast service tier (service_tier="ultrafast") is broadly
available for GPT-6 Astra in the API and Codex (Pro 500 / Enterprise),
billed at 6x Standard. GPT-6.1 Sol Ultrafast is announced as coming soon.

- One tier table: agent.fast_mode.parse_service_tier / STATIC_TIERS now
  back the CLI, gateway and TUI config loaders (three hand-copied parsers).
- resolve_fast_mode_overrides(tier="ultrafast") sends the tier only for
  Ultrafast models on api.openai.com / chatgpt.com; anything else gets
  nothing rather than a silent swap to another paid tier.
- /fast ultrafast on CLI, gateway (typed + picker when supported) and TUI;
  static-tier attach sites pass the tier through.
- Pricing: the response's SERVED service_tier is folded into usage (Codex
  stream assembler keeps it); served ultrafast prices from a published
  6x row (272K whole-request tier included). A request asking for
  Ultrafast but served at default bills at Standard.
- i18n: 4 new keys in all 17 locales; option lists mention ultrafast.
This commit is contained in:
kshitijk4poor
2026-09-29 23:54:34 +05:30
parent ebf1be6527
commit bdde674296
54 changed files with 381 additions and 118 deletions
+2 -2
View File
@@ -64,9 +64,9 @@ def record_aux_usage(
if raw_usage is None:
return
from agent.usage_pricing import estimate_usage_cost, normalize_usage
from agent.usage_pricing import estimate_usage_cost, normalize_usage, with_served_service_tier
usage = normalize_usage(raw_usage, provider=provider)
usage = with_served_service_tier(normalize_usage(raw_usage, provider=provider), response)
if not (
usage.input_tokens or usage.output_tokens
or usage.cache_read_tokens or usage.cache_write_tokens
+4 -2
View File
@@ -779,7 +779,7 @@ def _output_text_of(item: Any) -> str:
class _CodexResponseAssembler:
"""Assemble a Response-shaped ``SimpleNamespace`` from raw Responses SSE events.
Only ``usage`` / ``status`` / ``id`` are read from the terminal frame — never ``response.output``. Output
Only ``usage`` / ``status`` / ``id`` / ``service_tier`` are read from the terminal frame — never ``response.output``. Output
items come from ``output_item.done``, or are synthesized from text deltas, or settled from function calls
announced via ``output_item.added`` but never confirmed (some backends omit per-item done events on success)."""
@@ -790,6 +790,7 @@ class _CodexResponseAssembler:
active_summary_index: Any = None
terminal_status: str = "completed"
terminal_usage = terminal_response_id = terminal_incomplete_details = terminal_error = None
terminal_service_tier = None # the tier the backend SERVED (may differ from the one requested)
# terminal_status defaults to "completed", so settlement needs an explicitly observed response.completed frame.
saw_response_completed = False
@@ -911,6 +912,7 @@ class _CodexResponseAssembler:
resp_obj = _event_field(event, "response")
if resp_obj is not None:
self.terminal_usage, self.terminal_response_id = _event_field(resp_obj, "usage"), _event_field(resp_obj, "id")
self.terminal_service_tier = _event_field(resp_obj, "service_tier")
rstatus = _event_field(resp_obj, "status")
if isinstance(rstatus, str):
self.terminal_status = rstatus
@@ -978,7 +980,7 @@ class _CodexResponseAssembler:
return SimpleNamespace(
output=output, output_text="".join(self.text_deltas), usage=self.terminal_usage, status=self.terminal_status,
id=self.terminal_response_id, model=self.model, incomplete_details=self.terminal_incomplete_details,
error=self.terminal_error)
error=self.terminal_error, service_tier=self.terminal_service_tier)
def _consume_codex_event_stream(
+23 -2
View File
@@ -1,7 +1,7 @@
"""Bounded fast-mode windows (``/fast auto`` and ``/fast cold``).
``agent.service_tier``: ``None`` (normal), ``"priority"`` (static fast, pinned into
``agent.request_overrides`` at build time), ``"auto"`` (every user turn opens a
``agent.service_tier``: ``None`` (normal), ``"priority"`` / ``"ultrafast"`` (static tiers,
pinned into ``agent.request_overrides`` at build time), ``"auto"`` (every user turn opens a
window of ``agent.fast_auto_seconds``) or ``"cold"`` (only a session's first turn,
no prior history, opens it). The provider's fast override is layered onto request
kwargs only while the window is open; only per-request params (``service_tier`` /
@@ -20,6 +20,27 @@ DEFAULT_WINDOW_SECONDS = 60
# Documented fast-mode rate-limit headers; a limit of 0 means the organization has no fast
# capacity for the model (https://platform.claude.com/docs/en/build-with-claude/fast-mode).
_FAST_LIMIT_HEADERS = ("anthropic-fast-input-tokens-limit", "anthropic-fast-output-tokens-limit")
#: Tiers sent on every request of the session (OpenAI ``service_tier`` values; ``priority`` also
#: selects Anthropic/xAI fast mode). Ultrafast is OpenAI-only and gated per model.
STATIC_TIERS = frozenset({"priority", "ultrafast"})
NORMAL_TIER_WORDS = frozenset({"", "normal", "default", "standard", "off", "none"})
# User/config word -> agent.service_tier. The single table every surface (config loaders, /fast
# on CLI / gateway / TUI) parses through, so a new tier is one edit.
SERVICE_TIER_WORDS: dict[str, str] = {
"fast": "priority", "priority": "priority", "on": "priority",
"ultrafast": "ultrafast", "auto": "auto", "cold": "cold",
}
def parse_service_tier(raw: Any) -> str | None:
"""``agent.service_tier`` for a user/config word; None for normal and for unknown words."""
value = str(raw or "").strip().lower()
return None if value in NORMAL_TIER_WORDS else SERVICE_TIER_WORDS.get(value)
def service_tier_word(tier: Any) -> str:
"""The user-facing word for a stored tier (``priority`` -> ``fast``, None/"" -> ``normal``)."""
return {"priority": "fast", None: "normal", "": "normal"}.get(tier, tier)
def begin_turn(agent: Any, conversation_history: Any) -> None:
+3 -2
View File
@@ -17,7 +17,7 @@ from typing import Any, Dict, List
from agent.image_token_cost import calibrate_from_usage
from agent.usage_anchor import capture_usage_anchor, set_usage_anchor
from agent.usage_pricing import estimate_usage_cost, normalize_usage
from agent.usage_pricing import estimate_usage_cost, normalize_usage, with_served_service_tier
logger = logging.getLogger("agent.conversation_loop")
@@ -98,7 +98,8 @@ def record_response_usage(
)
return ResponseUsageOutcome(compression_attempts=compression_attempts, rearmed=rearmed)
canonical_usage = normalize_usage(response.usage, provider=agent.provider, api_mode=agent.api_mode)
canonical_usage = with_served_service_tier(
normalize_usage(response.usage, provider=agent.provider, api_mode=agent.api_mode), response)
# Aggregator-only usage kept for pricing: advisor tokens are priced at each advisor's
# OWN model rate and added as dollars below.
aggregator_usage = canonical_usage
+34 -1
View File
@@ -2,7 +2,7 @@ from __future__ import annotations
import logging
import re
from dataclasses import dataclass, fields
from dataclasses import dataclass, fields, replace
from datetime import datetime, timezone
from decimal import Decimal
from typing import Any, Dict, Literal, Optional
@@ -290,6 +290,22 @@ for _slug, _inp, _out, _read, _write, _inp_above, _out_above, _read_above, _writ
)
del _slug, _inp, _out, _read, _write, _inp_above, _out_above, _read_above, _write_above
# OpenAI Ultrafast (``service_tier: "ultrafast"``): 6x Standard on every bucket, same 272K
# whole-request tier. Selected by the tier the response reports it was SERVED at (a request asking
# for Ultrafast can be served at ``default``, and is then billed at Standard).
_OPENAI_ULTRAFAST_PRICING: Dict[str, PricingEntry] = {
"gpt-6-astra": _snap(
"60.00", "300.00", "6.00", "75.00",
url="https://developers.openai.com/api/docs/pricing?latest-pricing=ultrafast",
version="openai-ultrafast-2026-09",
tier_threshold_tokens=272_000,
input_cost_per_million_above=Decimal("120.00"),
output_cost_per_million_above=Decimal("450.00"),
cache_read_cost_per_million_above=Decimal("12.00"),
cache_write_cost_per_million_above=Decimal("150.00"),
),
}
# Context-tiered Gemini Pro: above 200k prompt tokens the *_above rates apply to
# the whole request (see PricingEntry).
_OFFICIAL_DOCS_PRICING[("google", "gemini-3.1-pro")] = _snap(
@@ -449,6 +465,19 @@ def _lookup_official_docs_pricing(route: BillingRoute) -> Optional[PricingEntry]
return _OFFICIAL_DOCS_PRICING.get((route.provider, normalized)) if normalized != model else None
def with_served_service_tier(usage: CanonicalUsage, response: Any) -> CanonicalUsage:
"""``usage`` with the response's served ``service_tier`` folded into ``raw_usage``. OpenAI reports
the tier on the response, not inside ``usage``, and pricing reads it from ``raw_usage``."""
tier = getattr(response, "service_tier", None)
if not isinstance(tier, str) or not tier.strip():
return usage
return replace(usage, raw_usage={**(usage.raw_usage or {}), "service_tier": tier.strip().lower()})
def _served_openai_tier(usage: CanonicalUsage) -> Optional[str]:
return usage.raw_usage.get("service_tier") if isinstance(usage.raw_usage, dict) else None
def _served_fast(usage: CanonicalUsage) -> bool:
"""Anthropic names the speed that served a fast-mode request in ``usage.speed``."""
return isinstance(usage.raw_usage, dict) and usage.raw_usage.get("speed") == "fast"
@@ -653,6 +682,10 @@ def estimate_usage_cost(
entry = _anthropic_fast_mode_entry(route.model)
if not entry:
return _unknown_cost("official_docs_snapshot", "fast-mode pricing unavailable for model")
if route.provider == "openai" and _served_openai_tier(usage) == "ultrafast":
entry = _OPENAI_ULTRAFAST_PRICING.get(route.model)
if not entry:
return _unknown_cost("official_docs_snapshot", "ultrafast pricing unavailable for model")
if not entry:
return _unknown_cost("none")
@@ -84,10 +84,10 @@ export function ModelSettingsSkeleton({ subpage }: Pick<ModelSettingsProps, 'sub
)
}
// agent.service_tier stores "fast"/"priority"/"on" for fast; anything else is
// normal (mirrors tui_gateway _load_service_tier).
// agent.service_tier stores "fast"/"priority"/"on" for fast and "ultrafast" for OpenAI
// Ultrafast; anything else is normal (mirrors agent.fast_mode.parse_service_tier).
const isFastTier = (tier: unknown): boolean =>
['fast', 'priority', 'on'].includes(
['fast', 'priority', 'on', 'ultrafast'].includes(
String(tier ?? '')
.trim()
.toLowerCase()
+4 -11
View File
@@ -222,17 +222,10 @@ class GatewayConfigLoadersMixin:
@classmethod
def _load_service_tier(cls) -> str | None:
"""``agent.service_tier``: fast/priority/on => "priority"; normal/off => None; None when unset/unknown."""
raw = cls._cfg_str("agent", "service_tier")
value = raw.lower()
if not value or value in {"normal", "default", "standard", "off", "none"}:
return None
if value in {"fast", "priority", "on"}:
return "priority"
if value in {"auto", "cold"}:
return value
logger.warning("Unknown service_tier '%s', ignoring", raw)
return None
"""``agent.service_tier`` parsed like the CLI (``hermes_cli.cli_config_load``); None when unset/unknown."""
from hermes_cli.cli_config_load import _parse_service_tier_config
return _parse_service_tier_config(cls._cfg_str("agent", "service_tier"))
@staticmethod
def _load_show_reasoning() -> bool:
+4 -2
View File
@@ -309,6 +309,7 @@ class GatewayTurnMixin:
"""Effective model/runtime config for one turn. With `/fast` priority on, fast-mode
``request_overrides`` are deep-merged OVER the per-provider ones so both reach the model."""
from gateway.run import _deep_merge_request_overrides
from agent.fast_mode import STATIC_TIERS
from hermes_cli.models import resolve_fast_mode_overrides
# Tests bind this method onto bare namespaces, so no class-level tables here.
runtime = {
@@ -328,13 +329,14 @@ class GatewayTurnMixin:
runtime["api_mode"], runtime["command"], tuple(runtime["args"]),
),
}
if getattr(self, "_service_tier", None) != "priority":
tier = getattr(self, "_service_tier", None)
if tier not in STATIC_TIERS:
# None / auto / cold: the bounded window is applied per request by agent.fast_mode.
route["request_overrides"] = base_request_overrides
return route
try:
overrides = resolve_fast_mode_overrides(
route["model"], provider=runtime["provider"], base_url=runtime["base_url"],
route["model"], provider=runtime["provider"], base_url=runtime["base_url"], tier=tier,
)
except Exception:
overrides = None
+10 -4
View File
@@ -29,6 +29,7 @@ _FAST_SELECTIONS = {
"off": (None, "normal", "gateway.fast.label_normal"),
"auto": ("auto", "auto", None),
"cold": ("cold", "cold", None),
"ultrafast": ("ultrafast", "ultrafast", None),
}
# /reasoning display-toggle arguments -> show_reasoning value.
@@ -804,18 +805,23 @@ class GatewayModelCommandsMixin:
async def _handle_fast_command(self, event: MessageEvent) -> Optional[str]:
"""Handle /fast — the CLI Priority Processing toggle; session-scoped unless ``--global``
(persists agent.service_tier, parity with /model)."""
from agent.fast_mode import service_tier_word
from gateway.run import _load_gateway_config, _resolve_gateway_model
from hermes_cli.models import model_supports_fast_mode
from hermes_cli.models import model_supports_fast_mode, model_supports_ultrafast
# The /reasoning parser strips --global (any position) and normalizes unicode dashes.
args, persist_global = self._parse_reasoning_command_args(event.get_command_args().strip().lower())
session_key = self._session_key_for_source(event.source)
self._service_tier = self._resolve_session_service_tier(session_key=session_key)
if not model_supports_fast_mode(_resolve_gateway_model(_load_gateway_config())):
model = _resolve_gateway_model(_load_gateway_config())
if not model_supports_fast_mode(model):
return t("gateway.fast.not_supported")
ultrafast = model_supports_ultrafast(model)
if args == "ultrafast" and not ultrafast:
return t("gateway.fast.ultrafast_not_supported", model=model)
if args and args != "status":
return self._apply_fast_selection(session_key, args, persist=persist_global)
mode = "fast" if self._service_tier == "priority" else (self._service_tier or "normal")
mode = service_tier_word(self._service_tier)
status = {"fast": t("gateway.fast.status_fast"), "normal": t("gateway.fast.status_normal")}.get(mode, mode)
async def _on_fast_choice(_chat_id: str, value: str) -> str:
@@ -827,7 +833,7 @@ class GatewayModelCommandsMixin:
title=t("gateway.fast.picker_title", mode=status),
choices=[
{"value": v, "label": t(f"gateway.fast.choice_{v}"), "is_current": mode == v}
for v in ("fast", "normal", "auto", "cold")
for v in ("fast", "normal", "auto", "cold", *(("ultrafast",) if ultrafast else ()))
],
on_choice_selected=_on_fast_choice,
)
+5 -3
View File
@@ -523,16 +523,18 @@ class CLIAgentSetupMixin:
def _resolve_turn_agent_config(self, user_message: str) -> dict:
"""Effective model/runtime config for one turn — always the session's primary
provider. With `/fast` on (service_tier == "priority") attach request_overrides;
provider. With a static `/fast` tier (fast / ultrafast) attach request_overrides;
auto/cold tiers are applied per request by agent.fast_mode instead."""
from agent.fast_mode import STATIC_TIERS
from hermes_cli.models import resolve_fast_mode_overrides
runtime = _current_runtime(self)
route = {"model": self.model, "runtime": runtime, "signature": _route_signature(self.model, runtime)}
overrides = None
if getattr(self, "service_tier", None) == "priority":
tier = getattr(self, "service_tier", None)
if tier in STATIC_TIERS:
try:
overrides = resolve_fast_mode_overrides(
route["model"], provider=runtime["provider"], base_url=runtime["base_url"])
route["model"], provider=runtime["provider"], base_url=runtime["base_url"], tier=tier)
except Exception:
pass
route["request_overrides"] = overrides
+8 -2
View File
@@ -191,7 +191,8 @@ _BUSY_MODES = ("queue", "steer", "interrupt")
# /fast argument -> (service_tier value, persisted config value)
_FAST_TIERS = {
"fast": ("priority", "fast"), "on": ("priority", "fast"), "normal": (None, "normal"),
"off": (None, "normal"), "auto": ("auto", "auto"), "cold": ("cold", "cold")}
"off": (None, "normal"), "auto": ("auto", "auto"), "cold": ("cold", "cold"),
"ultrafast": ("ultrafast", "ultrafast")}
# /reasoning display toggles: arg -> (attr, value, headline key, follow-up note key or None);
# the keys resolve under ``cli.commands.reasoning.*`` at call time.
@@ -2638,11 +2639,16 @@ class CLICommandsMixin:
raw = _command_arg(cmd)
usage = _dim_line(_t("fast.usage"))
if not raw or raw.lower() == "status":
status = {"priority": "fast", None: "normal"}.get(self.service_tier, self.service_tier)
from agent.fast_mode import service_tier_word
status = service_tier_word(self.service_tier)
return _cp(_accent_line(_t("fast.status", feature=feature_name, status=status)), usage)
arg, explicit_global = _split_scope_flags(raw)
if arg not in _FAST_TIERS:
return _cp(_dim_line(_t("shared.unknown_argument", arg=arg)), usage)
if arg == "ultrafast":
if not _probe("hermes_cli.models", "model_supports_ultrafast", False, model):
return _cp(_dim_line(_t("fast.ultrafast_not_supported", model=model or "?")), usage)
feature_name = _t("fast.feature_ultrafast")
self.service_tier, saved_value = _FAST_TIERS[arg]
_retire_agent(self) # Force agent re-init with new service-tier config
saved = explicit_global and _save("agent.service_tier", saved_value)
+7 -10
View File
@@ -69,16 +69,13 @@ def _parse_reasoning_config(effort) -> dict | None:
def _parse_service_tier_config(raw: str) -> str | None:
"""Parse a persisted fast-mode preference: None, "priority", "auto", or "cold"."""
value = str(raw or "").strip().lower()
if not value or value in {"normal", "default", "standard", "off", "none"}:
return None
if value in {"fast", "priority", "on"}:
return "priority"
if value in {"auto", "cold"}:
return value
logger.warning("Unknown service_tier '%s', ignoring", raw)
return None
"""Parse a persisted fast-mode preference: None, "priority", "ultrafast", "auto", or "cold"."""
from agent.fast_mode import NORMAL_TIER_WORDS, parse_service_tier
tier = parse_service_tier(raw)
if tier is None and str(raw or "").strip().lower() not in NORMAL_TIER_WORDS:
logger.warning("Unknown service_tier '%s', ignoring", raw)
return tier
# terminal.<key> -> TERMINAL_<KEY> env var. Container-resource keys apply to docker,
+15 -1
View File
@@ -42,6 +42,7 @@ from hermes_cli.models_catalog_static import (
_LIVE_FIRST_PICKER_PROVIDERS,
_MODELS_DEV_PREFERRED,
_OPENAI_FAST_MODE_PREFIXES,
_OPENAI_ULTRAFAST_MODELS,
_PROVIDER_ALIASES,
_PROVIDER_LABELS,
_PROVIDER_MODELS,
@@ -1185,17 +1186,30 @@ def _fast_mode_route_supported(
return not host or host in allowed.values()
def model_supports_ultrafast(model_id: Optional[str]) -> bool:
"""OpenAI Ultrafast (``service_tier: "ultrafast"``) is published per model, not per family."""
from agent.model_metadata import strip_codex_context_variant_suffix
base = _strip_vendor_prefix(strip_codex_context_variant_suffix(str(model_id or ""))).split(":")[0]
return base in _OPENAI_ULTRAFAST_MODELS
def resolve_fast_mode_overrides(
model_id: Optional[str], *, provider: Optional[str] = None, base_url: Optional[str] = None
model_id: Optional[str], *, provider: Optional[str] = None, base_url: Optional[str] = None,
tier: Optional[str] = None,
) -> dict[str, Any] | None:
"""Fast/priority request_overrides — ``{"speed": "fast"}`` (Anthropic Fast Mode) or
``{"service_tier": "priority"}`` (OpenAI / xAI Priority Processing) — or None if unsupported.
``tier="ultrafast"`` asks for OpenAI Ultrafast instead: ``{"service_tier": "ultrafast"}`` on an
Ultrafast model, None elsewhere (never a silent downgrade to a different paid tier).
With ``provider``/``base_url`` the route is gated too (``_fast_mode_route_supported``) so proxies
never see the params. Single fast-mode gate for ``/fast`` and ``agent.fast_mode`` windows."""
if not model_supports_fast_mode(model_id):
return None
if (provider or base_url) and not _fast_mode_route_supported(model_id, provider, base_url):
return None
if tier == "ultrafast":
return {"service_tier": "ultrafast"} if model_supports_ultrafast(model_id) else None
return {"speed": "fast"} if _is_anthropic_fast_model(model_id) else {"service_tier": "priority"}
+4
View File
@@ -560,6 +560,10 @@ _LIVE_FIRST_PICKER_PROVIDERS: frozenset[str] = frozenset({"opencode-zen", "openc
# positives are harmless. Codex-series models are excluded — the Codex Responses API doesn't
# expose service_tier.
_OPENAI_FAST_MODE_PREFIXES: tuple[str, ...] = ("gpt-", "o1", "o3", "o4")
# OpenAI Ultrafast (service_tier="ultrafast", 6x Standard): broadly available for GPT-6 Astra only
# (developers.openai.com/api/docs/guides/ultrafast-mode, 2026-09-29); GPT-6.1 Sol "coming soon".
# Exact wire slugs, matched after stripping the vendor prefix and the Hermes-side ``-900k`` alias.
_OPENAI_ULTRAFAST_MODELS: frozenset[str] = frozenset({"gpt-6-astra"})
# Providers where models.dev is authoritative: the curated list is an offline fallback plus custom
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} boodskappe saamgepers"
slashCmd.session.compress.nothing: "niks om saam te pers nie"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "vinnige modus: {0}"
slashCmd.session.fast.usage: "gebruik: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "gebruik: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "aanwyser: {0}"
slashCmd.session.indicator.switched: "aanwyser → {0}"
slashCmd.session.indicator.usage: "gebruik: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — standaardverwerking"
choice_auto: "auto — vinnig vir die eerste sekondes van elke beurt"
choice_cold: "cold — vinnig slegs vir die eerste beurt van 'n sessie"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6x prys)"
ultrafast_not_supported: "⚡ Ultrafast is slegs beskikbaar vir GPT-6 Astra (huidige model: `{model}`)."
footer:
status: "📎 Looptyd-voetstuk: **{state}**\nVelde: `{fields}`\nPlatform: `{platform}`"
usage: "Gebruik: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Bestuur redenering-inspanning en -vertoon"
fast:
description: "Vinnige modus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Vinnige modus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Wys of verander die vertoon-skin/tema"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Vinnige modus"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Gebruik: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast is slegs beskikbaar vir GPT-6 Astra (huidige model: {model})."
usage: "Gebruik: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} gestel op {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "تم ضغط {0} من الرسائل"
slashCmd.session.compress.nothing: "لا يوجد ما يُضغط"
slashCmd.session.compress.tokSuffix: " · {0} رمزًا"
slashCmd.session.fast.mode: "الوضع السريع: {0}"
slashCmd.session.fast.usage: "الاستخدام: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "الاستخدام: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "المؤشر: {0}"
slashCmd.session.indicator.switched: "المؤشر → {0}"
slashCmd.session.indicator.usage: "الاستخدام: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — المعالجة القياسية"
choice_auto: "auto — سريع في الثواني الأولى من كل دور"
choice_cold: "cold — سريع في الدور الأول من الجلسة فقط"
choice_ultrafast: "ultrafast — ‏OpenAI Ultrafast ‏(GPT-6 Astra، السعر ×6)"
ultrafast_not_supported: "⚡ ‏Ultrafast متاح فقط لـ GPT-6 Astra (النموذج الحالي: `{model}`)."
footer:
status: "📎 تذييل التشغيل: **{state}**\nالحقول: `{fields}`\nالمنصّة: `{platform}`"
usage: "الاستخدام: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "إدارة مستوى الاستدلال وطريقة عرضه"
fast:
description: "الوضع السريع — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "الوضع السريع — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "عرض سمة العرض أو تغييرها"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "الوضع السريع"
feature_anthropic: "الوضع السريع من Anthropic"
feature_openai: "المعالجة ذات الأولوية"
usage: "الاستخدام: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) ‏Ultrafast متاح فقط لـ GPT-6 Astra (النموذج الحالي: {model})."
usage: "الاستخدام: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ تم ضبط {feature} على {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} Nachrichten komprimiert"
slashCmd.session.compress.nothing: "nichts zu komprimieren"
slashCmd.session.compress.tokSuffix: " · {0} Tok"
slashCmd.session.fast.mode: "Fast-Modus: {0}"
slashCmd.session.fast.usage: "Verwendung: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "Verwendung: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "Indikator: {0}"
slashCmd.session.indicator.switched: "Indikator → {0}"
slashCmd.session.indicator.usage: "Verwendung: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — Standardverarbeitung"
choice_auto: "auto — schnell in den ersten Sekunden jedes Zugs"
choice_cold: "cold — schnell nur im ersten Zug einer Sitzung"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6-facher Preis)"
ultrafast_not_supported: "⚡ Ultrafast ist nur für GPT-6 Astra verfügbar (aktuelles Modell: `{model}`)."
footer:
status: "📎 Laufzeit-Fußzeile: **{state}**\nFelder: `{fields}`\nPlattform: `{platform}`"
usage: "Verwendung: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Reasoning-Stärke und -Anzeige verwalten"
fast:
description: "Schnellmodus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Schnellmodus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Anzeige-Skin/Theme anzeigen oder ändern"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Schnellmodus"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Verwendung: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast ist nur für GPT-6 Astra verfügbar (aktuelles Modell: {model})."
usage: "Verwendung: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} auf {value} gesetzt {scope}"
hatch:
+8 -4
View File
@@ -174,8 +174,8 @@ gateway:
denied_reason_plural: "❌ Commands denied ({count} commands). Reason relayed to the agent: \"{reason}\""
fast:
not_supported: "⚡ /fast is only available for OpenAI models that support Priority Processing."
status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast <normal|fast|auto|cold|status>`"
unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, auto, cold, status"
status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast <normal|fast|auto|cold|ultrafast|status>`"
unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, auto, cold, ultrafast, status"
saved: "⚡ ✓ Priority Processing: **{label}** (saved to config)\n_(takes effect on next message)_"
session_only: "⚡ ✓ Priority Processing: **{label}** (this session only)"
label_fast: "FAST"
@@ -187,6 +187,8 @@ gateway:
choice_normal: "normal — standard processing"
choice_auto: "auto — fast for the first seconds of every turn"
choice_cold: "cold — fast for the first turn of a session only"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6x price)"
ultrafast_not_supported: "⚡ Ultrafast is only available for GPT-6 Astra (current model: `{model}`)."
footer:
status: "📎 Runtime footer: **{state}**\nFields: `{fields}`\nPlatform: `{platform}`"
usage: "Usage: `/footer [on|off|status]`"
@@ -1669,7 +1671,7 @@ slash:
reasoning:
description: "Manage reasoning effort and display"
fast:
description: "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Show or change the display skin/theme"
indicator:
@@ -2751,7 +2753,9 @@ cli:
feature_generic: "Fast mode"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Usage: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast is only available for GPT-6 Astra (current model: {model})."
usage: "Usage: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} set to {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} mensajes comprimidos"
slashCmd.session.compress.nothing: "nada que comprimir"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "modo rápido: {0}"
slashCmd.session.fast.usage: "uso: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "uso: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "indicador: {0}"
slashCmd.session.indicator.switched: "indicador → {0}"
slashCmd.session.indicator.usage: "uso: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — procesamiento estándar"
choice_auto: "auto — rápido en los primeros segundos de cada turno"
choice_cold: "cold — rápido solo en el primer turno de una sesión"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, precio x6)"
ultrafast_not_supported: "⚡ Ultrafast solo está disponible para GPT-6 Astra (modelo actual: `{model}`)."
footer:
status: "📎 Pie de ejecución: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`"
usage: "Uso: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Gestionar el esfuerzo de razonamiento y su visualización"
fast:
description: "Modo rápido: OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Modo rápido: OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Mostrar o cambiar el skin/tema de la interfaz"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Modo rápido"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Uso: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast solo está disponible para GPT-6 Astra (modelo actual: {model})."
usage: "Uso: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} establecido en {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} messages compressés"
slashCmd.session.compress.nothing: "rien à compresser"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "mode rapide : {0}"
slashCmd.session.fast.usage: "usage : /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "usage : /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "indicateur : {0}"
slashCmd.session.indicator.switched: "indicateur → {0}"
slashCmd.session.indicator.usage: "usage : /indicator [{0}]"
+6 -2
View File
@@ -210,6 +210,8 @@ gateway:
choice_normal: "normal — traitement standard"
choice_auto: "auto — rapide pendant les premières secondes de chaque tour"
choice_cold: "cold — rapide uniquement au premier tour d'une session"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, prix x6)"
ultrafast_not_supported: "⚡ Ultrafast n'est disponible que pour GPT-6 Astra (modèle actuel : `{model}`)."
footer:
status: "📎 Pied de page d'exécution : **{state}**\nChamps : `{fields}`\nPlateforme : `{platform}`"
usage: "Usage : `/footer [on|off|status]`"
@@ -1850,7 +1852,7 @@ slash:
reasoning:
description: "Gérer l'effort de raisonnement et son affichage"
fast:
description: "Mode rapide — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Mode rapide — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Afficher ou changer le skin/thème d'affichage"
indicator:
@@ -2929,7 +2931,9 @@ cli:
feature_generic: "Mode rapide"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Usage : /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast n'est disponible que pour GPT-6 Astra (modèle actuel : {model})."
usage: "Usage : /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature} : {status}"
set_to: "✓ {feature} défini sur {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "dlúthaíodh {0} teachtaireacht"
slashCmd.session.compress.nothing: "níl aon rud le dlúthú"
slashCmd.session.compress.tokSuffix: " · {0} comhartha"
slashCmd.session.fast.mode: "mód tapa: {0}"
slashCmd.session.fast.usage: "úsáid: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "úsáid: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "táscaire: {0}"
slashCmd.session.indicator.switched: "táscaire → {0}"
slashCmd.session.indicator.usage: "úsáid: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — gnáthphróiseáil"
choice_auto: "auto — tapa do na chéad soicindí de gach seal"
choice_cold: "cold — tapa don chéad seal de sheisiún amháin"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, praghas x6)"
ultrafast_not_supported: "⚡ Níl Ultrafast ar fáil ach do GPT-6 Astra (samhail reatha: `{model}`)."
footer:
status: "📎 Buntásc rite: **{state}**\nRéimsí: `{fields}`\nArdán: `{platform}`"
usage: "Úsáid: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Bainistigh iarracht agus taispeáint na réasúnaíochta"
fast:
description: "Mód tapa — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Mód tapa — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Taispeáin nó athraigh craiceann/téama na taispeána"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Mód tapa"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Úsáid: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Níl Ultrafast ar fáil ach do GPT-6 Astra (samhail reatha: {model})."
usage: "Úsáid: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} socraithe go {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} üzenet tömörítve"
slashCmd.session.compress.nothing: "nincs mit tömöríteni"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "gyors mód: {0}"
slashCmd.session.fast.usage: "használat: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "használat: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "indikátor: {0}"
slashCmd.session.indicator.switched: "indikátor → {0}"
slashCmd.session.indicator.usage: "használat: /indicator [{0}]"
+6 -2
View File
@@ -187,6 +187,8 @@ gateway:
choice_normal: "normal — normál feldolgozás"
choice_auto: "auto — gyors minden kör első másodperceiben"
choice_cold: "cold — gyors csak a munkamenet első körében"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, hatszoros ár)"
ultrafast_not_supported: "⚡ Az Ultrafast csak a GPT-6 Astra modellhez érhető el (jelenlegi modell: `{model}`)."
footer:
status: "📎 Futási idejű lábléc: **{state}**\nMezők: `{fields}`\nPlatform: `{platform}`"
usage: "Használat: `/footer [on|off|status]`"
@@ -1615,7 +1617,7 @@ slash:
reasoning:
description: "Gondolkodási erőfeszítés és megjelenítés kezelése"
fast:
description: "Gyors mód — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Gyors mód — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "A megjelenítési skin/téma megjelenítése vagy módosítása"
indicator:
@@ -2668,7 +2670,9 @@ cli:
feature_generic: "Gyors mód"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Használat: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Az Ultrafast csak a GPT-6 Astra modellhez érhető el (jelenlegi modell: {model})."
usage: "Használat: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} beállítva: {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "compressi {0} messaggi"
slashCmd.session.compress.nothing: "niente da comprimere"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "modalità veloce: {0}"
slashCmd.session.fast.usage: "uso: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "uso: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "indicatore: {0}"
slashCmd.session.indicator.switched: "indicatore → {0}"
slashCmd.session.indicator.usage: "uso: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — elaborazione standard"
choice_auto: "auto — veloce nei primi secondi di ogni turno"
choice_cold: "cold — veloce solo nel primo turno di una sessione"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, prezzo x6)"
ultrafast_not_supported: "⚡ Ultrafast è disponibile solo per GPT-6 Astra (modello attuale: `{model}`)."
footer:
status: "📎 Footer di runtime: **{state}**\nCampi: `{fields}`\nPiattaforma: `{platform}`"
usage: "Uso: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Gestisci lo sforzo di ragionamento e la sua visualizzazione"
fast:
description: "Modalità veloce — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Modalità veloce — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Mostra o cambia la skin/tema di visualizzazione"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Modalità veloce"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Uso: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast è disponibile solo per GPT-6 Astra (modello attuale: {model})."
usage: "Uso: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} impostato su {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} 件のメッセージを圧縮
slashCmd.session.compress.nothing: "圧縮するものがありません"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "高速モード: {0}"
slashCmd.session.fast.usage: "使い方: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "使い方: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "インジケーター: {0}"
slashCmd.session.indicator.switched: "インジケーター → {0}"
slashCmd.session.indicator.usage: "使い方: /indicator [{0}]"
+5 -1
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — 標準処理"
choice_auto: "auto — 各ターンの最初の数秒間だけ高速"
choice_cold: "cold — セッションの最初のターンのみ高速"
choice_ultrafast: "ultrafast — OpenAI Ultrafast(GPT-6 Astra、料金 6 倍)"
ultrafast_not_supported: "⚡ Ultrafast は GPT-6 Astra でのみ利用できます(現在のモデル: `{model}`)。"
footer:
status: "📎 ランタイムフッター: **{state}**\nフィールド: `{fields}`\nプラットフォーム: `{platform}`"
usage: "使い方: `/footer [on|off|status]`"
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "高速モード"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "使い方: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast は GPT-6 Astra でのみ利用できます(現在のモデル: {model})。"
usage: "使い方: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} を {value} に設定しました {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "메시지 {0}개 압축함"
slashCmd.session.compress.nothing: "압축할 내용이 없어요"
slashCmd.session.compress.tokSuffix: " · {0} 토큰"
slashCmd.session.fast.mode: "빠른 모드: {0}"
slashCmd.session.fast.usage: "사용법: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "사용법: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "인디케이터: {0}"
slashCmd.session.indicator.switched: "인디케이터 → {0}"
slashCmd.session.indicator.usage: "사용법: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — 표준 처리"
choice_auto: "auto — 매 턴의 처음 몇 초 동안 빠름"
choice_cold: "cold — 세션의 첫 턴에만 빠름"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 가격 6배)"
ultrafast_not_supported: "⚡ Ultrafast는 GPT-6 Astra에서만 사용할 수 있습니다 (현재 모델: `{model}`)."
footer:
status: "📎 런타임 푸터: **{state}**\n필드: `{fields}`\n플랫폼: `{platform}`"
usage: "사용법: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "추론 강도와 표시 관리"
fast:
description: "빠른 모드 — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "빠른 모드 — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "화면 스킨/테마 표시 또는 변경"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "빠른 모드"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "사용법: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast는 GPT-6 Astra에서만 사용할 수 있습니다 (현재 모델: {model})."
usage: "사용법: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature}을(를) {value}(으)로 설정했어요 {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} mensagens compactadas"
slashCmd.session.compress.nothing: "nada para compactar"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "modo rápido: {0}"
slashCmd.session.fast.usage: "uso: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "uso: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "indicador: {0}"
slashCmd.session.indicator.switched: "indicador → {0}"
slashCmd.session.indicator.usage: "uso: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — processamento padrão"
choice_auto: "auto — rápido nos primeiros segundos de cada turno"
choice_cold: "cold — rápido apenas no primeiro turno de uma sessão"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, preço 6x)"
ultrafast_not_supported: "⚡ O Ultrafast só está disponível para o GPT-6 Astra (modelo atual: `{model}`)."
footer:
status: "📎 Rodapé de execução: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`"
usage: "Uso: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Gerenciar o esforço e a exibição do raciocínio"
fast:
description: "Modo rápido — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Modo rápido — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Mostrar ou alterar a skin/tema de exibição"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Modo rápido"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Uso: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) O Ultrafast só está disponível para o GPT-6 Astra (modelo atual: {model})."
usage: "Uso: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} definido como {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "сжато сообщений: {0}"
slashCmd.session.compress.nothing: "нечего сжимать"
slashCmd.session.compress.tokSuffix: " · {0} ток."
slashCmd.session.fast.mode: "быстрый режим: {0}"
slashCmd.session.fast.usage: "Использование: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "Использование: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "индикатор: {0}"
slashCmd.session.indicator.switched: "индикатор → {0}"
slashCmd.session.indicator.usage: "Использование: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — стандартная обработка"
choice_auto: "auto — быстро в первые секунды каждого хода"
choice_cold: "cold — быстро только на первом ходе сессии"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, цена ×6)"
ultrafast_not_supported: "⚡ Ultrafast доступен только для GPT-6 Astra (текущая модель: `{model}`)."
footer:
status: "📎 Нижний колонтитул среды выполнения: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`"
usage: "Использование: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Управлять уровнем и отображением рассуждений"
fast:
description: "Быстрый режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Быстрый режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Показать или сменить скин/тему оформления"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Быстрый режим"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Использование: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast доступен только для GPT-6 Astra (текущая модель: {model})."
usage: "Использование: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature}: установлено {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} mesaj sıkıştırıldı"
slashCmd.session.compress.nothing: "sıkıştırılacak bir şey yok"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "hızlı mod: {0}"
slashCmd.session.fast.usage: "kullanım: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "kullanım: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "gösterge: {0}"
slashCmd.session.indicator.switched: "gösterge → {0}"
slashCmd.session.indicator.usage: "kullanım: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — standart işleme"
choice_auto: "auto — her turun ilk saniyelerinde hızlı"
choice_cold: "cold — yalnızca oturumun ilk turunda hızlı"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6 kat fiyat)"
ultrafast_not_supported: "⚡ Ultrafast yalnızca GPT-6 Astra için kullanılabilir (geçerli model: `{model}`)."
footer:
status: "📎 Çalışma zamanı altbilgisi: **{state}**\nAlanlar: `{fields}`\nPlatform: `{platform}`"
usage: "Kullanım: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Akıl yürütme düzeyini ve gösterimini yönet"
fast:
description: "Hızlı mod — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Hızlı mod — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Görüntü görünümünü/temasını göster veya değiştir"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Hızlı mod"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Kullanım: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast yalnızca GPT-6 Astra için kullanılabilir (geçerli model: {model})."
usage: "Kullanım: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature} {value} olarak ayarlandı {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "стиснуто повідомле
slashCmd.session.compress.nothing: "нічого стискати"
slashCmd.session.compress.tokSuffix: " · {0} ток"
slashCmd.session.fast.mode: "швидкий режим: {0}"
slashCmd.session.fast.usage: "використання: /fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "використання: /fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "індикатор: {0}"
slashCmd.session.indicator.switched: "індикатор → {0}"
slashCmd.session.indicator.usage: "використання: /indicator [{0}]"
+6 -2
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — стандартна обробка"
choice_auto: "auto — швидко в перші секунди кожного ходу"
choice_cold: "cold — швидко лише на першому ході сесії"
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, ціна ×6)"
ultrafast_not_supported: "⚡ Ultrafast доступний лише для GPT-6 Astra (поточна модель: `{model}`)."
footer:
status: "📎 Нижній колонтитул середовища: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`"
usage: "Використання: `/footer [on|off|status]`"
@@ -1598,7 +1600,7 @@ slash:
reasoning:
description: "Керувати рівнем зусиль міркування та його показом"
fast:
description: "Швидкий режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
description: "Швидкий режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
skin:
description: "Показати або змінити оформлення (тему) інтерфейсу"
indicator:
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "Швидкий режим"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "Використання: /fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast доступний лише для GPT-6 Astra (поточна модель: {model})."
usage: "Використання: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}: {status}"
set_to: "✓ {feature}: встановлено {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "已壓縮 {0} 則訊息"
slashCmd.session.compress.nothing: "沒有可壓縮的內容"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "快速模式:{0}"
slashCmd.session.fast.usage: "用法:/fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "用法:/fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "指示器:{0}"
slashCmd.session.indicator.switched: "指示器 → {0}"
slashCmd.session.indicator.usage: "用法:/indicator [{0}]"
+5 -1
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — 標準處理"
choice_auto: "auto — 每輪的前幾秒快速"
choice_cold: "cold — 僅會話的第一輪快速"
choice_ultrafast: "ultrafast — OpenAI Ultrafast(GPT-6 Astra,6 倍價格)"
ultrafast_not_supported: "⚡ Ultrafast 僅適用於 GPT-6 Astra(目前模型:`{model}`)。"
footer:
status: "📎 執行階段頁尾:**{state}**\n欄位:`{fields}`\n平台:`{platform}`"
usage: "用法:`/footer [on|off|status]`"
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "快速模式"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "用法:/fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast 僅適用於 GPT-6 Astra(目前模型:{model})。"
usage: "用法:/fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}:{status}"
set_to: "✓ {feature} 已設為 {value} {scope}"
hatch:
+1 -1
View File
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "已压缩 {0} 条消息"
slashCmd.session.compress.nothing: "没有可压缩的内容"
slashCmd.session.compress.tokSuffix: " · {0} tok"
slashCmd.session.fast.mode: "快速模式:{0}"
slashCmd.session.fast.usage: "用法:/fast [normal|fast|status|on|off|toggle]"
slashCmd.session.fast.usage: "用法:/fast [normal|fast|ultrafast|status|on|off|toggle]"
slashCmd.session.indicator.current: "指示器:{0}"
slashCmd.session.indicator.switched: "指示器 → {0}"
slashCmd.session.indicator.usage: "用法:/indicator [{0}]"
+5 -1
View File
@@ -184,6 +184,8 @@ gateway:
choice_normal: "normal — 标准处理"
choice_auto: "auto — 每轮的前几秒快速"
choice_cold: "cold — 仅会话的第一轮快速"
choice_ultrafast: "ultrafast — OpenAI Ultrafast(GPT-6 Astra,6 倍价格)"
ultrafast_not_supported: "⚡ Ultrafast 仅适用于 GPT-6 Astra(当前模型:`{model}`)。"
footer:
status: "📎 运行时页脚:**{state}**\n字段:`{fields}`\n平台:`{platform}`"
usage: "用法:`/footer [on|off|status]`"
@@ -2651,7 +2653,9 @@ cli:
feature_generic: "快速模式"
feature_anthropic: "Anthropic Fast Mode"
feature_openai: "Priority Processing"
usage: "用法:/fast [normal|fast|auto|cold|status] [--global]"
feature_ultrafast: "OpenAI Ultrafast"
ultrafast_not_supported: "(._.) Ultrafast 仅适用于 GPT-6 Astra(当前模型:{model})。"
usage: "用法:/fast [normal|fast|auto|cold|ultrafast|status] [--global]"
status: "{feature}:{status}"
set_to: "✓ {feature} 已设为 {value} {scope}"
hatch:
+106
View File
@@ -0,0 +1,106 @@
"""OpenAI Ultrafast (``service_tier: "ultrafast"``): one tier word table, a per-model request gate,
and pricing from the tier the response was SERVED at. Relationship tests, no catalog snapshots."""
from types import SimpleNamespace
import pytest
from agent.fast_mode import SERVICE_TIER_WORDS, STATIC_TIERS, parse_service_tier
from agent.usage_pricing import (
_OPENAI_ULTRAFAST_PRICING,
CanonicalUsage,
estimate_usage_cost,
with_served_service_tier,
)
from hermes_cli.models import model_supports_ultrafast, resolve_fast_mode_overrides
ASTRA_SPELLINGS = ("gpt-6-astra", "openai/gpt-6-astra", "gpt-6-astra-900k")
def test_every_config_loader_parses_tiers_through_the_same_table(monkeypatch):
from gateway.run import GatewayRunner
from hermes_cli.cli_config_load import _parse_service_tier_config
import tui_gateway.server as tui
for word, tier in {**SERVICE_TIER_WORDS, "normal": None, "off": None, "bogus": None}.items():
monkeypatch.setattr(GatewayRunner, "_cfg_str", classmethod(lambda cls, *_k, _w=word: _w))
monkeypatch.setattr(tui, "_load_cfg", lambda _w=word: {"agent": {"service_tier": _w}})
assert parse_service_tier(word) == tier
assert _parse_service_tier_config(word) == tier, word
assert GatewayRunner._load_service_tier() == tier, word
assert tui._load_service_tier() == tier, word
assert "ultrafast" in STATIC_TIERS
@pytest.mark.parametrize("provider,base_url", [("openai", "https://api.openai.com/v1"),
("openai-codex", "https://chatgpt.com/backend-api/codex")])
def test_ultrafast_is_requested_only_for_ultrafast_models_on_first_party_routes(provider, base_url):
for model in ASTRA_SPELLINGS:
assert model_supports_ultrafast(model), model
assert resolve_fast_mode_overrides(model, provider=provider, base_url=base_url, tier="ultrafast") == {
"service_tier": "ultrafast"}
# A Priority-capable model without Ultrafast gets nothing, never a silent swap to another paid tier.
assert resolve_fast_mode_overrides("gpt-6-sol", provider=provider, base_url=base_url) == {"service_tier": "priority"}
assert resolve_fast_mode_overrides("gpt-6-sol", provider=provider, base_url=base_url, tier="ultrafast") is None
# Proxies never see the tier.
assert resolve_fast_mode_overrides("openai/gpt-6-astra", provider="openrouter",
base_url="https://openrouter.ai/api/v1", tier="ultrafast") is None
def test_cli_and_gateway_turn_routes_send_the_static_tier():
import cli as cli_mod
from gateway.run import GatewayRunner
stub = SimpleNamespace(model="gpt-6-astra", api_key="k", base_url="https://api.openai.com/v1", provider="openai",
api_mode="codex_responses", acp_command=None, acp_args=[], _credential_pool=None,
service_tier="ultrafast")
assert cli_mod.HermesCLI._resolve_turn_agent_config(stub, "hi")["request_overrides"] == {"service_tier": "ultrafast"}
runner = object.__new__(GatewayRunner)
runner._service_tier = "ultrafast"
rk = {"api_key": "k", "base_url": "https://api.openai.com/v1", "provider": "openai", "api_mode": "codex_responses",
"command": None, "args": [], "credential_pool": None, "max_tokens": None}
assert runner._resolve_turn_agent_config("hi", "gpt-6-astra", rk)["request_overrides"] == {"service_tier": "ultrafast"}
def test_cli_refuses_ultrafast_on_a_model_without_it(monkeypatch):
import cli as cli_mod
from unittest.mock import MagicMock
stub = SimpleNamespace(service_tier="priority", model="gpt-6-sol", agent=MagicMock(model="gpt-6-sol"),
_fast_command_available=lambda: True)
monkeypatch.setattr(cli_mod, "_cprint", lambda *a, **k: None)
cli_mod.HermesCLI._handle_fast_command(stub, "/fast ultrafast")
assert stub.service_tier == "priority"
stub.model = stub.agent.model = "gpt-6-astra"
cli_mod.HermesCLI._handle_fast_command(stub, "/fast ultrafast")
assert stub.service_tier == "ultrafast"
def _usage(prompt_uncached: int, served_tier=None) -> CanonicalUsage:
usage = CanonicalUsage(input_tokens=prompt_uncached, output_tokens=10_000, cache_read_tokens=20_000,
cache_write_tokens=5_000)
return with_served_service_tier(usage, SimpleNamespace(service_tier=served_tier))
@pytest.mark.parametrize("prompt_uncached", [50_000, 400_000]) # below / above the 272K whole-request tier
def test_served_ultrafast_bills_at_the_ultrafast_row_and_requested_only_does_not(prompt_uncached):
for model in _OPENAI_ULTRAFAST_PRICING:
standard = estimate_usage_cost(model, _usage(prompt_uncached), provider="openai-api")
served_default = estimate_usage_cost(model, _usage(prompt_uncached, "default"), provider="openai-api")
ultra = estimate_usage_cost(model, _usage(prompt_uncached, "ultrafast"), provider="openai-api")
assert served_default.amount_usd == standard.amount_usd # asked for Ultrafast, served at Standard
assert ultra.pricing_version == _OPENAI_ULTRAFAST_PRICING[model].pricing_version
assert ultra.amount_usd == standard.amount_usd * 6 # every bucket, both context tiers
def test_served_ultrafast_on_a_model_without_a_published_rate_is_unknown():
assert estimate_usage_cost("gpt-6-sol", _usage(1_000, "ultrafast"), provider="openai-api").status == "unknown"
def test_codex_stream_assembler_keeps_the_served_tier():
from agent.codex_runtime import _consume_codex_event_stream
done = {"type": "response.completed", "response": {"id": "r1", "status": "completed", "service_tier": "default",
"usage": {"input_tokens": 1, "output_tokens": 1}}}
events = [{"type": "response.output_text.delta", "delta": "PASS"}, done]
assert _consume_codex_event_stream(iter(events), model="gpt-6-astra").service_tier == "default"
+1 -1
View File
@@ -233,7 +233,7 @@ def _cfg_get_fast(params):
else session.get("create_service_tier_override"))
if tier is None:
tier = _load_service_tier()
return {"value": "fast" if tier == "priority" else "normal"}
return {"value": {"priority": "fast", "ultrafast": "ultrafast"}.get(tier, "normal")}
def _cfg_get_thinking_mode(params):
+8 -6
View File
@@ -164,7 +164,7 @@ def _set_model(rid, params, key, value, session):
_FAST_WORDS = {"fast": "fast", "on": "fast", "normal": "normal", "off": "normal",
"auto": "auto", "cold": "cold"}
"auto": "auto", "cold": "cold", "ultrafast": "ultrafast"}
def _set_fast(rid, params, key, value, session):
@@ -176,13 +176,14 @@ def _set_fast(rid, params, key, value, session):
current_tier = session["create_service_tier_override"] or None # pre-build pin beats global
else:
current_tier = _load_service_tier()
from agent.fast_mode import STATIC_TIERS, service_tier_word
if raw == "status":
return _kv(rid, key, {"priority": "fast", None: "normal", "": "normal"}.get(current_tier, current_tier))
nv = _FAST_WORDS.get(raw, ("normal" if current_tier == "priority" else "fast") if raw in {"", "toggle"} else None)
return _kv(rid, key, service_tier_word(current_tier))
nv = _FAST_WORDS.get(raw, ("normal" if current_tier in STATIC_TIERS else "fast") if raw in {"", "toggle"} else None)
if nv is None:
return _err(rid, 4002, f"unknown fast mode: {value}")
overrides = None
if nv == "fast":
if nv in ("fast", "ultrafast"):
from hermes_cli.models import resolve_fast_mode_overrides
if agent is not None:
target_model = getattr(agent, "model", None)
@@ -192,9 +193,10 @@ def _set_fast(rid, params, key, value, session):
if not target_model:
return _err(rid, 4002, "fast mode is not available without a selected model")
overrides = resolve_fast_mode_overrides(target_model, provider=getattr(agent, "provider", None),
base_url=getattr(agent, "base_url", None))
base_url=getattr(agent, "base_url", None),
tier="ultrafast" if nv == "ultrafast" else None)
if overrides is None:
return _err(rid, 4002, "fast mode is not available for this model")
return _err(rid, 4002, f"{nv} mode is not available for this model")
if session is not None:
# Session-scoped like `reasoning` (global = `--global` / Settings → Model): writing config.yaml
# here flipped fast mode for every surface. The create override survives rebuilds; "" pins normal.
+2 -1
View File
@@ -323,7 +323,8 @@ def _mirror_prompt(sid, session, agent, arg) -> None:
agent._cached_system_prompt = None
_FAST_TIERS = {"fast": "priority", "on": "priority", "normal": None, "off": None, "auto": "auto", "cold": "cold"}
_FAST_TIERS = {"fast": "priority", "on": "priority", "normal": None, "off": None, "auto": "auto", "cold": "cold",
"ultrafast": "ultrafast"}
def _mirror_fast(sid, session, agent, arg) -> None:
+8 -9
View File
@@ -29,6 +29,7 @@ from hermes_cli.env_loader import load_hermes_dotenv
from utils import file_signature, is_truthy_value
from hermes_state_ids import new_session_id
from tools.environments.local import hermes_subprocess_env
from agent.fast_mode import STATIC_TIERS
from agent.replay_cleanup import canonicalize_replay_history
from agent.reasoning_effort import clamp_effort, route_supported_efforts
from agent.compaction_display import project_compaction_message_for_display # noqa: F401
@@ -1870,12 +1871,10 @@ def _load_reasoning_config(model: str = "") -> dict | None:
return resolve_reasoning_config(_load_cfg(), model)
_SERVICE_TIER_ALIASES = {"fast": "priority", "priority": "priority", "on": "priority", "auto": "auto", "cold": "cold"}
def _load_service_tier() -> str | None:
raw = str((_load_cfg().get("agent") or {}).get("service_tier", "") or "").strip().lower()
return _SERVICE_TIER_ALIASES.get(raw)
from agent.fast_mode import parse_service_tier
return parse_service_tier((_load_cfg().get("agent") or {}).get("service_tier", ""))
def _load_provider_routing() -> dict:
@@ -2262,7 +2261,7 @@ def _live_session_identity(session: dict) -> tuple[str, str]:
return str(model), str(provider or "")
def _fast_tier_applies(agent, model: str, provider: str, *, route_known: bool) -> bool:
def _fast_tier_applies(agent, model: str, provider: str, *, route_known: bool, tier: str | None = None) -> bool:
"""Whether a priority tier reaches this session's route. Every request builder asks the same gate, so a
profile-wide ``service_tier: fast`` sends nothing to a local server or a proxy, and the session must not
report Fast there either. ``route_known`` is False while a switch is pending: the agent's base URL still
@@ -2274,7 +2273,7 @@ def _fast_tier_applies(agent, model: str, provider: str, *, route_known: bool) -
base_url = getattr(agent, "_anthropic_base_url", None)
base_url = base_url or getattr(agent, "base_url", None)
try:
return resolve_fast_mode_overrides(model, provider=provider or None, base_url=base_url) is not None
return resolve_fast_mode_overrides(model, provider=provider or None, base_url=base_url, tier=tier) is not None
except Exception:
return False
@@ -2325,8 +2324,8 @@ def _session_info(agent, session: dict | None = None) -> dict:
"provider": pending_provider or provider,
"reasoning_effort": reasoning_effort, "reasoning_effort_wire": reasoning_effort_wire,
"service_tier": service_tier,
"fast": service_tier == "priority" and _fast_tier_applies(agent, model, pending_provider or provider,
route_known=not pending_provider),
"fast": service_tier in STATIC_TIERS and _fast_tier_applies(agent, model, pending_provider or provider,
route_known=not pending_provider, tier=service_tier),
"yolo": yolo, "approval_mode": approval_mode,
"tools": dict(mirror.get("tools") or {}) if isinstance(mirror.get("tools"), dict) else {},
"skills": dict(mirror.get("skills") or {}) if isinstance(mirror.get("skills"), dict) else {},
+12 -6
View File
@@ -26,6 +26,12 @@ const TUI_SESSION_MODEL_RE = new RegExp(`(?:^|\\s)${TUI_SESSION_MODEL_FLAG}(?:\\
const REASONING_SESSION_FLAGS = new Set(['--session'])
const REASONING_GLOBAL_FLAGS = new Set(['--global'])
type FastModeWord = 'fast' | 'normal' | 'ultrafast'
// `config.get/set fast` answer fast | ultrafast | normal (auto/cold windows read as normal here).
const fastModeWord = (value: unknown): FastModeWord =>
value === 'fast' || value === 'ultrafast' ? value : 'normal'
const modelValueForConfigSet = (arg: string) => {
const trimmed = arg.trim()
@@ -606,11 +612,11 @@ export const sessionCommands: SlashCommand[] = [
},
{
help: 'toggle fast mode [normal|fast|status|on|off|toggle]',
help: 'toggle fast mode [normal|fast|ultrafast|status|on|off|toggle]',
name: 'fast',
run: (arg, ctx) => {
const mode = arg.trim().toLowerCase()
const valid = new Set(['', 'status', 'normal', 'fast', 'on', 'off', 'toggle'])
const valid = new Set(['', 'status', 'normal', 'fast', 'ultrafast', 'on', 'off', 'toggle'])
if (!valid.has(mode)) {
return ctx.transcript.sys(t('slashCmd.session.fast.usage'))
@@ -621,7 +627,7 @@ export const sessionCommands: SlashCommand[] = [
.rpc<ConfigGetValueResponse>('config.get', { key: 'fast', session_id: ctx.sid })
.then(
ctx.guarded<ConfigGetValueResponse>(r =>
ctx.transcript.sys(t('slashCmd.session.fast.mode', r.value === 'fast' ? 'fast' : 'normal'))
ctx.transcript.sys(t('slashCmd.session.fast.mode', fastModeWord(r.value)))
)
)
.catch(ctx.guardedErr)
@@ -631,15 +637,15 @@ export const sessionCommands: SlashCommand[] = [
.rpc<ConfigSetResponse>('config.set', { key: 'fast', session_id: ctx.sid, value: mode })
.then(
ctx.guarded<ConfigSetResponse>(r => {
const next = r.value === 'fast' ? 'fast' : 'normal'
const next = fastModeWord(r.value)
ctx.transcript.sys(t('slashCmd.session.fast.mode', next))
patchUiState(state => ({
...state,
info: state.info
? {
...state.info,
fast: next === 'fast',
service_tier: next === 'fast' ? 'priority' : ''
fast: next !== 'normal',
service_tier: { fast: 'priority', normal: '', ultrafast: 'ultrafast' }[next]
}
: state.info
}))
+1 -1
View File
@@ -30,7 +30,7 @@ export const slashCmdSessionEn = {
},
fast: {
mode: (mode: string) => `fast mode: ${mode}`,
usage: 'usage: /fast [normal|fast|status|on|off|toggle]'
usage: 'usage: /fast [normal|fast|ultrafast|status|on|off|toggle]'
},
indicator: {
current: (style: string) => `indicator: ${style}`,