feat(fast): /fast ultrafast for GPT-6 Astra, priced at the served tier
OpenAI's Ultrafast service tier (service_tier="ultrafast") is broadly available for GPT-6 Astra in the API and Codex (Pro 500 / Enterprise), billed at 6x Standard. GPT-6.1 Sol Ultrafast is announced as coming soon. - One tier table: agent.fast_mode.parse_service_tier / STATIC_TIERS now back the CLI, gateway and TUI config loaders (three hand-copied parsers). - resolve_fast_mode_overrides(tier="ultrafast") sends the tier only for Ultrafast models on api.openai.com / chatgpt.com; anything else gets nothing rather than a silent swap to another paid tier. - /fast ultrafast on CLI, gateway (typed + picker when supported) and TUI; static-tier attach sites pass the tier through. - Pricing: the response's SERVED service_tier is folded into usage (Codex stream assembler keeps it); served ultrafast prices from a published 6x row (272K whole-request tier included). A request asking for Ultrafast but served at default bills at Standard. - i18n: 4 new keys in all 17 locales; option lists mention ultrafast.
This commit is contained in:
@@ -64,9 +64,9 @@ def record_aux_usage(
|
||||
if raw_usage is None:
|
||||
return
|
||||
|
||||
from agent.usage_pricing import estimate_usage_cost, normalize_usage
|
||||
from agent.usage_pricing import estimate_usage_cost, normalize_usage, with_served_service_tier
|
||||
|
||||
usage = normalize_usage(raw_usage, provider=provider)
|
||||
usage = with_served_service_tier(normalize_usage(raw_usage, provider=provider), response)
|
||||
if not (
|
||||
usage.input_tokens or usage.output_tokens
|
||||
or usage.cache_read_tokens or usage.cache_write_tokens
|
||||
|
||||
@@ -779,7 +779,7 @@ def _output_text_of(item: Any) -> str:
|
||||
class _CodexResponseAssembler:
|
||||
"""Assemble a Response-shaped ``SimpleNamespace`` from raw Responses SSE events.
|
||||
|
||||
Only ``usage`` / ``status`` / ``id`` are read from the terminal frame — never ``response.output``. Output
|
||||
Only ``usage`` / ``status`` / ``id`` / ``service_tier`` are read from the terminal frame — never ``response.output``. Output
|
||||
items come from ``output_item.done``, or are synthesized from text deltas, or settled from function calls
|
||||
announced via ``output_item.added`` but never confirmed (some backends omit per-item done events on success)."""
|
||||
|
||||
@@ -790,6 +790,7 @@ class _CodexResponseAssembler:
|
||||
active_summary_index: Any = None
|
||||
terminal_status: str = "completed"
|
||||
terminal_usage = terminal_response_id = terminal_incomplete_details = terminal_error = None
|
||||
terminal_service_tier = None # the tier the backend SERVED (may differ from the one requested)
|
||||
# terminal_status defaults to "completed", so settlement needs an explicitly observed response.completed frame.
|
||||
saw_response_completed = False
|
||||
|
||||
@@ -911,6 +912,7 @@ class _CodexResponseAssembler:
|
||||
resp_obj = _event_field(event, "response")
|
||||
if resp_obj is not None:
|
||||
self.terminal_usage, self.terminal_response_id = _event_field(resp_obj, "usage"), _event_field(resp_obj, "id")
|
||||
self.terminal_service_tier = _event_field(resp_obj, "service_tier")
|
||||
rstatus = _event_field(resp_obj, "status")
|
||||
if isinstance(rstatus, str):
|
||||
self.terminal_status = rstatus
|
||||
@@ -978,7 +980,7 @@ class _CodexResponseAssembler:
|
||||
return SimpleNamespace(
|
||||
output=output, output_text="".join(self.text_deltas), usage=self.terminal_usage, status=self.terminal_status,
|
||||
id=self.terminal_response_id, model=self.model, incomplete_details=self.terminal_incomplete_details,
|
||||
error=self.terminal_error)
|
||||
error=self.terminal_error, service_tier=self.terminal_service_tier)
|
||||
|
||||
|
||||
def _consume_codex_event_stream(
|
||||
|
||||
+23
-2
@@ -1,7 +1,7 @@
|
||||
"""Bounded fast-mode windows (``/fast auto`` and ``/fast cold``).
|
||||
|
||||
``agent.service_tier``: ``None`` (normal), ``"priority"`` (static fast, pinned into
|
||||
``agent.request_overrides`` at build time), ``"auto"`` (every user turn opens a
|
||||
``agent.service_tier``: ``None`` (normal), ``"priority"`` / ``"ultrafast"`` (static tiers,
|
||||
pinned into ``agent.request_overrides`` at build time), ``"auto"`` (every user turn opens a
|
||||
window of ``agent.fast_auto_seconds``) or ``"cold"`` (only a session's first turn,
|
||||
no prior history, opens it). The provider's fast override is layered onto request
|
||||
kwargs only while the window is open; only per-request params (``service_tier`` /
|
||||
@@ -20,6 +20,27 @@ DEFAULT_WINDOW_SECONDS = 60
|
||||
# Documented fast-mode rate-limit headers; a limit of 0 means the organization has no fast
|
||||
# capacity for the model (https://platform.claude.com/docs/en/build-with-claude/fast-mode).
|
||||
_FAST_LIMIT_HEADERS = ("anthropic-fast-input-tokens-limit", "anthropic-fast-output-tokens-limit")
|
||||
#: Tiers sent on every request of the session (OpenAI ``service_tier`` values; ``priority`` also
|
||||
#: selects Anthropic/xAI fast mode). Ultrafast is OpenAI-only and gated per model.
|
||||
STATIC_TIERS = frozenset({"priority", "ultrafast"})
|
||||
NORMAL_TIER_WORDS = frozenset({"", "normal", "default", "standard", "off", "none"})
|
||||
# User/config word -> agent.service_tier. The single table every surface (config loaders, /fast
|
||||
# on CLI / gateway / TUI) parses through, so a new tier is one edit.
|
||||
SERVICE_TIER_WORDS: dict[str, str] = {
|
||||
"fast": "priority", "priority": "priority", "on": "priority",
|
||||
"ultrafast": "ultrafast", "auto": "auto", "cold": "cold",
|
||||
}
|
||||
|
||||
|
||||
def parse_service_tier(raw: Any) -> str | None:
|
||||
"""``agent.service_tier`` for a user/config word; None for normal and for unknown words."""
|
||||
value = str(raw or "").strip().lower()
|
||||
return None if value in NORMAL_TIER_WORDS else SERVICE_TIER_WORDS.get(value)
|
||||
|
||||
|
||||
def service_tier_word(tier: Any) -> str:
|
||||
"""The user-facing word for a stored tier (``priority`` -> ``fast``, None/"" -> ``normal``)."""
|
||||
return {"priority": "fast", None: "normal", "": "normal"}.get(tier, tier)
|
||||
|
||||
|
||||
def begin_turn(agent: Any, conversation_history: Any) -> None:
|
||||
|
||||
+3
-2
@@ -17,7 +17,7 @@ from typing import Any, Dict, List
|
||||
|
||||
from agent.image_token_cost import calibrate_from_usage
|
||||
from agent.usage_anchor import capture_usage_anchor, set_usage_anchor
|
||||
from agent.usage_pricing import estimate_usage_cost, normalize_usage
|
||||
from agent.usage_pricing import estimate_usage_cost, normalize_usage, with_served_service_tier
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
@@ -98,7 +98,8 @@ def record_response_usage(
|
||||
)
|
||||
return ResponseUsageOutcome(compression_attempts=compression_attempts, rearmed=rearmed)
|
||||
|
||||
canonical_usage = normalize_usage(response.usage, provider=agent.provider, api_mode=agent.api_mode)
|
||||
canonical_usage = with_served_service_tier(
|
||||
normalize_usage(response.usage, provider=agent.provider, api_mode=agent.api_mode), response)
|
||||
# Aggregator-only usage kept for pricing: advisor tokens are priced at each advisor's
|
||||
# OWN model rate and added as dollars below.
|
||||
aggregator_usage = canonical_usage
|
||||
|
||||
+34
-1
@@ -2,7 +2,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, fields
|
||||
from dataclasses import dataclass, fields, replace
|
||||
from datetime import datetime, timezone
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, Literal, Optional
|
||||
@@ -290,6 +290,22 @@ for _slug, _inp, _out, _read, _write, _inp_above, _out_above, _read_above, _writ
|
||||
)
|
||||
del _slug, _inp, _out, _read, _write, _inp_above, _out_above, _read_above, _write_above
|
||||
|
||||
# OpenAI Ultrafast (``service_tier: "ultrafast"``): 6x Standard on every bucket, same 272K
|
||||
# whole-request tier. Selected by the tier the response reports it was SERVED at (a request asking
|
||||
# for Ultrafast can be served at ``default``, and is then billed at Standard).
|
||||
_OPENAI_ULTRAFAST_PRICING: Dict[str, PricingEntry] = {
|
||||
"gpt-6-astra": _snap(
|
||||
"60.00", "300.00", "6.00", "75.00",
|
||||
url="https://developers.openai.com/api/docs/pricing?latest-pricing=ultrafast",
|
||||
version="openai-ultrafast-2026-09",
|
||||
tier_threshold_tokens=272_000,
|
||||
input_cost_per_million_above=Decimal("120.00"),
|
||||
output_cost_per_million_above=Decimal("450.00"),
|
||||
cache_read_cost_per_million_above=Decimal("12.00"),
|
||||
cache_write_cost_per_million_above=Decimal("150.00"),
|
||||
),
|
||||
}
|
||||
|
||||
# Context-tiered Gemini Pro: above 200k prompt tokens the *_above rates apply to
|
||||
# the whole request (see PricingEntry).
|
||||
_OFFICIAL_DOCS_PRICING[("google", "gemini-3.1-pro")] = _snap(
|
||||
@@ -449,6 +465,19 @@ def _lookup_official_docs_pricing(route: BillingRoute) -> Optional[PricingEntry]
|
||||
return _OFFICIAL_DOCS_PRICING.get((route.provider, normalized)) if normalized != model else None
|
||||
|
||||
|
||||
def with_served_service_tier(usage: CanonicalUsage, response: Any) -> CanonicalUsage:
|
||||
"""``usage`` with the response's served ``service_tier`` folded into ``raw_usage``. OpenAI reports
|
||||
the tier on the response, not inside ``usage``, and pricing reads it from ``raw_usage``."""
|
||||
tier = getattr(response, "service_tier", None)
|
||||
if not isinstance(tier, str) or not tier.strip():
|
||||
return usage
|
||||
return replace(usage, raw_usage={**(usage.raw_usage or {}), "service_tier": tier.strip().lower()})
|
||||
|
||||
|
||||
def _served_openai_tier(usage: CanonicalUsage) -> Optional[str]:
|
||||
return usage.raw_usage.get("service_tier") if isinstance(usage.raw_usage, dict) else None
|
||||
|
||||
|
||||
def _served_fast(usage: CanonicalUsage) -> bool:
|
||||
"""Anthropic names the speed that served a fast-mode request in ``usage.speed``."""
|
||||
return isinstance(usage.raw_usage, dict) and usage.raw_usage.get("speed") == "fast"
|
||||
@@ -653,6 +682,10 @@ def estimate_usage_cost(
|
||||
entry = _anthropic_fast_mode_entry(route.model)
|
||||
if not entry:
|
||||
return _unknown_cost("official_docs_snapshot", "fast-mode pricing unavailable for model")
|
||||
if route.provider == "openai" and _served_openai_tier(usage) == "ultrafast":
|
||||
entry = _OPENAI_ULTRAFAST_PRICING.get(route.model)
|
||||
if not entry:
|
||||
return _unknown_cost("official_docs_snapshot", "ultrafast pricing unavailable for model")
|
||||
if not entry:
|
||||
return _unknown_cost("none")
|
||||
|
||||
|
||||
@@ -84,10 +84,10 @@ export function ModelSettingsSkeleton({ subpage }: Pick<ModelSettingsProps, 'sub
|
||||
)
|
||||
}
|
||||
|
||||
// agent.service_tier stores "fast"/"priority"/"on" for fast; anything else is
|
||||
// normal (mirrors tui_gateway _load_service_tier).
|
||||
// agent.service_tier stores "fast"/"priority"/"on" for fast and "ultrafast" for OpenAI
|
||||
// Ultrafast; anything else is normal (mirrors agent.fast_mode.parse_service_tier).
|
||||
const isFastTier = (tier: unknown): boolean =>
|
||||
['fast', 'priority', 'on'].includes(
|
||||
['fast', 'priority', 'on', 'ultrafast'].includes(
|
||||
String(tier ?? '')
|
||||
.trim()
|
||||
.toLowerCase()
|
||||
|
||||
@@ -222,17 +222,10 @@ class GatewayConfigLoadersMixin:
|
||||
|
||||
@classmethod
|
||||
def _load_service_tier(cls) -> str | None:
|
||||
"""``agent.service_tier``: fast/priority/on => "priority"; normal/off => None; None when unset/unknown."""
|
||||
raw = cls._cfg_str("agent", "service_tier")
|
||||
value = raw.lower()
|
||||
if not value or value in {"normal", "default", "standard", "off", "none"}:
|
||||
return None
|
||||
if value in {"fast", "priority", "on"}:
|
||||
return "priority"
|
||||
if value in {"auto", "cold"}:
|
||||
return value
|
||||
logger.warning("Unknown service_tier '%s', ignoring", raw)
|
||||
return None
|
||||
"""``agent.service_tier`` parsed like the CLI (``hermes_cli.cli_config_load``); None when unset/unknown."""
|
||||
from hermes_cli.cli_config_load import _parse_service_tier_config
|
||||
|
||||
return _parse_service_tier_config(cls._cfg_str("agent", "service_tier"))
|
||||
|
||||
@staticmethod
|
||||
def _load_show_reasoning() -> bool:
|
||||
|
||||
+4
-2
@@ -309,6 +309,7 @@ class GatewayTurnMixin:
|
||||
"""Effective model/runtime config for one turn. With `/fast` priority on, fast-mode
|
||||
``request_overrides`` are deep-merged OVER the per-provider ones so both reach the model."""
|
||||
from gateway.run import _deep_merge_request_overrides
|
||||
from agent.fast_mode import STATIC_TIERS
|
||||
from hermes_cli.models import resolve_fast_mode_overrides
|
||||
# Tests bind this method onto bare namespaces, so no class-level tables here.
|
||||
runtime = {
|
||||
@@ -328,13 +329,14 @@ class GatewayTurnMixin:
|
||||
runtime["api_mode"], runtime["command"], tuple(runtime["args"]),
|
||||
),
|
||||
}
|
||||
if getattr(self, "_service_tier", None) != "priority":
|
||||
tier = getattr(self, "_service_tier", None)
|
||||
if tier not in STATIC_TIERS:
|
||||
# None / auto / cold: the bounded window is applied per request by agent.fast_mode.
|
||||
route["request_overrides"] = base_request_overrides
|
||||
return route
|
||||
try:
|
||||
overrides = resolve_fast_mode_overrides(
|
||||
route["model"], provider=runtime["provider"], base_url=runtime["base_url"],
|
||||
route["model"], provider=runtime["provider"], base_url=runtime["base_url"], tier=tier,
|
||||
)
|
||||
except Exception:
|
||||
overrides = None
|
||||
|
||||
@@ -29,6 +29,7 @@ _FAST_SELECTIONS = {
|
||||
"off": (None, "normal", "gateway.fast.label_normal"),
|
||||
"auto": ("auto", "auto", None),
|
||||
"cold": ("cold", "cold", None),
|
||||
"ultrafast": ("ultrafast", "ultrafast", None),
|
||||
}
|
||||
|
||||
# /reasoning display-toggle arguments -> show_reasoning value.
|
||||
@@ -804,18 +805,23 @@ class GatewayModelCommandsMixin:
|
||||
async def _handle_fast_command(self, event: MessageEvent) -> Optional[str]:
|
||||
"""Handle /fast — the CLI Priority Processing toggle; session-scoped unless ``--global``
|
||||
(persists agent.service_tier, parity with /model)."""
|
||||
from agent.fast_mode import service_tier_word
|
||||
from gateway.run import _load_gateway_config, _resolve_gateway_model
|
||||
from hermes_cli.models import model_supports_fast_mode
|
||||
from hermes_cli.models import model_supports_fast_mode, model_supports_ultrafast
|
||||
|
||||
# The /reasoning parser strips --global (any position) and normalizes unicode dashes.
|
||||
args, persist_global = self._parse_reasoning_command_args(event.get_command_args().strip().lower())
|
||||
session_key = self._session_key_for_source(event.source)
|
||||
self._service_tier = self._resolve_session_service_tier(session_key=session_key)
|
||||
if not model_supports_fast_mode(_resolve_gateway_model(_load_gateway_config())):
|
||||
model = _resolve_gateway_model(_load_gateway_config())
|
||||
if not model_supports_fast_mode(model):
|
||||
return t("gateway.fast.not_supported")
|
||||
ultrafast = model_supports_ultrafast(model)
|
||||
if args == "ultrafast" and not ultrafast:
|
||||
return t("gateway.fast.ultrafast_not_supported", model=model)
|
||||
if args and args != "status":
|
||||
return self._apply_fast_selection(session_key, args, persist=persist_global)
|
||||
mode = "fast" if self._service_tier == "priority" else (self._service_tier or "normal")
|
||||
mode = service_tier_word(self._service_tier)
|
||||
status = {"fast": t("gateway.fast.status_fast"), "normal": t("gateway.fast.status_normal")}.get(mode, mode)
|
||||
|
||||
async def _on_fast_choice(_chat_id: str, value: str) -> str:
|
||||
@@ -827,7 +833,7 @@ class GatewayModelCommandsMixin:
|
||||
title=t("gateway.fast.picker_title", mode=status),
|
||||
choices=[
|
||||
{"value": v, "label": t(f"gateway.fast.choice_{v}"), "is_current": mode == v}
|
||||
for v in ("fast", "normal", "auto", "cold")
|
||||
for v in ("fast", "normal", "auto", "cold", *(("ultrafast",) if ultrafast else ()))
|
||||
],
|
||||
on_choice_selected=_on_fast_choice,
|
||||
)
|
||||
|
||||
@@ -523,16 +523,18 @@ class CLIAgentSetupMixin:
|
||||
|
||||
def _resolve_turn_agent_config(self, user_message: str) -> dict:
|
||||
"""Effective model/runtime config for one turn — always the session's primary
|
||||
provider. With `/fast` on (service_tier == "priority") attach request_overrides;
|
||||
provider. With a static `/fast` tier (fast / ultrafast) attach request_overrides;
|
||||
auto/cold tiers are applied per request by agent.fast_mode instead."""
|
||||
from agent.fast_mode import STATIC_TIERS
|
||||
from hermes_cli.models import resolve_fast_mode_overrides
|
||||
runtime = _current_runtime(self)
|
||||
route = {"model": self.model, "runtime": runtime, "signature": _route_signature(self.model, runtime)}
|
||||
overrides = None
|
||||
if getattr(self, "service_tier", None) == "priority":
|
||||
tier = getattr(self, "service_tier", None)
|
||||
if tier in STATIC_TIERS:
|
||||
try:
|
||||
overrides = resolve_fast_mode_overrides(
|
||||
route["model"], provider=runtime["provider"], base_url=runtime["base_url"])
|
||||
route["model"], provider=runtime["provider"], base_url=runtime["base_url"], tier=tier)
|
||||
except Exception:
|
||||
pass
|
||||
route["request_overrides"] = overrides
|
||||
|
||||
@@ -191,7 +191,8 @@ _BUSY_MODES = ("queue", "steer", "interrupt")
|
||||
# /fast argument -> (service_tier value, persisted config value)
|
||||
_FAST_TIERS = {
|
||||
"fast": ("priority", "fast"), "on": ("priority", "fast"), "normal": (None, "normal"),
|
||||
"off": (None, "normal"), "auto": ("auto", "auto"), "cold": ("cold", "cold")}
|
||||
"off": (None, "normal"), "auto": ("auto", "auto"), "cold": ("cold", "cold"),
|
||||
"ultrafast": ("ultrafast", "ultrafast")}
|
||||
|
||||
# /reasoning display toggles: arg -> (attr, value, headline key, follow-up note key or None);
|
||||
# the keys resolve under ``cli.commands.reasoning.*`` at call time.
|
||||
@@ -2638,11 +2639,16 @@ class CLICommandsMixin:
|
||||
raw = _command_arg(cmd)
|
||||
usage = _dim_line(_t("fast.usage"))
|
||||
if not raw or raw.lower() == "status":
|
||||
status = {"priority": "fast", None: "normal"}.get(self.service_tier, self.service_tier)
|
||||
from agent.fast_mode import service_tier_word
|
||||
status = service_tier_word(self.service_tier)
|
||||
return _cp(_accent_line(_t("fast.status", feature=feature_name, status=status)), usage)
|
||||
arg, explicit_global = _split_scope_flags(raw)
|
||||
if arg not in _FAST_TIERS:
|
||||
return _cp(_dim_line(_t("shared.unknown_argument", arg=arg)), usage)
|
||||
if arg == "ultrafast":
|
||||
if not _probe("hermes_cli.models", "model_supports_ultrafast", False, model):
|
||||
return _cp(_dim_line(_t("fast.ultrafast_not_supported", model=model or "?")), usage)
|
||||
feature_name = _t("fast.feature_ultrafast")
|
||||
self.service_tier, saved_value = _FAST_TIERS[arg]
|
||||
_retire_agent(self) # Force agent re-init with new service-tier config
|
||||
saved = explicit_global and _save("agent.service_tier", saved_value)
|
||||
|
||||
@@ -69,16 +69,13 @@ def _parse_reasoning_config(effort) -> dict | None:
|
||||
|
||||
|
||||
def _parse_service_tier_config(raw: str) -> str | None:
|
||||
"""Parse a persisted fast-mode preference: None, "priority", "auto", or "cold"."""
|
||||
value = str(raw or "").strip().lower()
|
||||
if not value or value in {"normal", "default", "standard", "off", "none"}:
|
||||
return None
|
||||
if value in {"fast", "priority", "on"}:
|
||||
return "priority"
|
||||
if value in {"auto", "cold"}:
|
||||
return value
|
||||
logger.warning("Unknown service_tier '%s', ignoring", raw)
|
||||
return None
|
||||
"""Parse a persisted fast-mode preference: None, "priority", "ultrafast", "auto", or "cold"."""
|
||||
from agent.fast_mode import NORMAL_TIER_WORDS, parse_service_tier
|
||||
|
||||
tier = parse_service_tier(raw)
|
||||
if tier is None and str(raw or "").strip().lower() not in NORMAL_TIER_WORDS:
|
||||
logger.warning("Unknown service_tier '%s', ignoring", raw)
|
||||
return tier
|
||||
|
||||
|
||||
# terminal.<key> -> TERMINAL_<KEY> env var. Container-resource keys apply to docker,
|
||||
|
||||
+15
-1
@@ -42,6 +42,7 @@ from hermes_cli.models_catalog_static import (
|
||||
_LIVE_FIRST_PICKER_PROVIDERS,
|
||||
_MODELS_DEV_PREFERRED,
|
||||
_OPENAI_FAST_MODE_PREFIXES,
|
||||
_OPENAI_ULTRAFAST_MODELS,
|
||||
_PROVIDER_ALIASES,
|
||||
_PROVIDER_LABELS,
|
||||
_PROVIDER_MODELS,
|
||||
@@ -1185,17 +1186,30 @@ def _fast_mode_route_supported(
|
||||
return not host or host in allowed.values()
|
||||
|
||||
|
||||
def model_supports_ultrafast(model_id: Optional[str]) -> bool:
|
||||
"""OpenAI Ultrafast (``service_tier: "ultrafast"``) is published per model, not per family."""
|
||||
from agent.model_metadata import strip_codex_context_variant_suffix
|
||||
|
||||
base = _strip_vendor_prefix(strip_codex_context_variant_suffix(str(model_id or ""))).split(":")[0]
|
||||
return base in _OPENAI_ULTRAFAST_MODELS
|
||||
|
||||
|
||||
def resolve_fast_mode_overrides(
|
||||
model_id: Optional[str], *, provider: Optional[str] = None, base_url: Optional[str] = None
|
||||
model_id: Optional[str], *, provider: Optional[str] = None, base_url: Optional[str] = None,
|
||||
tier: Optional[str] = None,
|
||||
) -> dict[str, Any] | None:
|
||||
"""Fast/priority request_overrides — ``{"speed": "fast"}`` (Anthropic Fast Mode) or
|
||||
``{"service_tier": "priority"}`` (OpenAI / xAI Priority Processing) — or None if unsupported.
|
||||
``tier="ultrafast"`` asks for OpenAI Ultrafast instead: ``{"service_tier": "ultrafast"}`` on an
|
||||
Ultrafast model, None elsewhere (never a silent downgrade to a different paid tier).
|
||||
With ``provider``/``base_url`` the route is gated too (``_fast_mode_route_supported``) so proxies
|
||||
never see the params. Single fast-mode gate for ``/fast`` and ``agent.fast_mode`` windows."""
|
||||
if not model_supports_fast_mode(model_id):
|
||||
return None
|
||||
if (provider or base_url) and not _fast_mode_route_supported(model_id, provider, base_url):
|
||||
return None
|
||||
if tier == "ultrafast":
|
||||
return {"service_tier": "ultrafast"} if model_supports_ultrafast(model_id) else None
|
||||
return {"speed": "fast"} if _is_anthropic_fast_model(model_id) else {"service_tier": "priority"}
|
||||
|
||||
|
||||
|
||||
@@ -560,6 +560,10 @@ _LIVE_FIRST_PICKER_PROVIDERS: frozenset[str] = frozenset({"opencode-zen", "openc
|
||||
# positives are harmless. Codex-series models are excluded — the Codex Responses API doesn't
|
||||
# expose service_tier.
|
||||
_OPENAI_FAST_MODE_PREFIXES: tuple[str, ...] = ("gpt-", "o1", "o3", "o4")
|
||||
# OpenAI Ultrafast (service_tier="ultrafast", 6x Standard): broadly available for GPT-6 Astra only
|
||||
# (developers.openai.com/api/docs/guides/ultrafast-mode, 2026-09-29); GPT-6.1 Sol "coming soon".
|
||||
# Exact wire slugs, matched after stripping the vendor prefix and the Hermes-side ``-900k`` alias.
|
||||
_OPENAI_ULTRAFAST_MODELS: frozenset[str] = frozenset({"gpt-6-astra"})
|
||||
|
||||
|
||||
# Providers where models.dev is authoritative: the curated list is an offline fallback plus custom
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} boodskappe saamgepers"
|
||||
slashCmd.session.compress.nothing: "niks om saam te pers nie"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "vinnige modus: {0}"
|
||||
slashCmd.session.fast.usage: "gebruik: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "gebruik: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "aanwyser: {0}"
|
||||
slashCmd.session.indicator.switched: "aanwyser → {0}"
|
||||
slashCmd.session.indicator.usage: "gebruik: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — standaardverwerking"
|
||||
choice_auto: "auto — vinnig vir die eerste sekondes van elke beurt"
|
||||
choice_cold: "cold — vinnig slegs vir die eerste beurt van 'n sessie"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6x prys)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast is slegs beskikbaar vir GPT-6 Astra (huidige model: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Looptyd-voetstuk: **{state}**\nVelde: `{fields}`\nPlatform: `{platform}`"
|
||||
usage: "Gebruik: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Bestuur redenering-inspanning en -vertoon"
|
||||
fast:
|
||||
description: "Vinnige modus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Vinnige modus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Wys of verander die vertoon-skin/tema"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Vinnige modus"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Gebruik: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast is slegs beskikbaar vir GPT-6 Astra (huidige model: {model})."
|
||||
usage: "Gebruik: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} gestel op {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "تم ضغط {0} من الرسائل"
|
||||
slashCmd.session.compress.nothing: "لا يوجد ما يُضغط"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} رمزًا"
|
||||
slashCmd.session.fast.mode: "الوضع السريع: {0}"
|
||||
slashCmd.session.fast.usage: "الاستخدام: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "الاستخدام: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "المؤشر: {0}"
|
||||
slashCmd.session.indicator.switched: "المؤشر → {0}"
|
||||
slashCmd.session.indicator.usage: "الاستخدام: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — المعالجة القياسية"
|
||||
choice_auto: "auto — سريع في الثواني الأولى من كل دور"
|
||||
choice_cold: "cold — سريع في الدور الأول من الجلسة فقط"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra، السعر ×6)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast متاح فقط لـ GPT-6 Astra (النموذج الحالي: `{model}`)."
|
||||
footer:
|
||||
status: "📎 تذييل التشغيل: **{state}**\nالحقول: `{fields}`\nالمنصّة: `{platform}`"
|
||||
usage: "الاستخدام: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "إدارة مستوى الاستدلال وطريقة عرضه"
|
||||
fast:
|
||||
description: "الوضع السريع — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "الوضع السريع — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "عرض سمة العرض أو تغييرها"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "الوضع السريع"
|
||||
feature_anthropic: "الوضع السريع من Anthropic"
|
||||
feature_openai: "المعالجة ذات الأولوية"
|
||||
usage: "الاستخدام: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast متاح فقط لـ GPT-6 Astra (النموذج الحالي: {model})."
|
||||
usage: "الاستخدام: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ تم ضبط {feature} على {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} Nachrichten komprimiert"
|
||||
slashCmd.session.compress.nothing: "nichts zu komprimieren"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} Tok"
|
||||
slashCmd.session.fast.mode: "Fast-Modus: {0}"
|
||||
slashCmd.session.fast.usage: "Verwendung: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "Verwendung: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "Indikator: {0}"
|
||||
slashCmd.session.indicator.switched: "Indikator → {0}"
|
||||
slashCmd.session.indicator.usage: "Verwendung: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — Standardverarbeitung"
|
||||
choice_auto: "auto — schnell in den ersten Sekunden jedes Zugs"
|
||||
choice_cold: "cold — schnell nur im ersten Zug einer Sitzung"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6-facher Preis)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast ist nur für GPT-6 Astra verfügbar (aktuelles Modell: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Laufzeit-Fußzeile: **{state}**\nFelder: `{fields}`\nPlattform: `{platform}`"
|
||||
usage: "Verwendung: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Reasoning-Stärke und -Anzeige verwalten"
|
||||
fast:
|
||||
description: "Schnellmodus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Schnellmodus — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Anzeige-Skin/Theme anzeigen oder ändern"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Schnellmodus"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Verwendung: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast ist nur für GPT-6 Astra verfügbar (aktuelles Modell: {model})."
|
||||
usage: "Verwendung: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} auf {value} gesetzt {scope}"
|
||||
hatch:
|
||||
|
||||
+8
-4
@@ -174,8 +174,8 @@ gateway:
|
||||
denied_reason_plural: "❌ Commands denied ({count} commands). Reason relayed to the agent: \"{reason}\""
|
||||
fast:
|
||||
not_supported: "⚡ /fast is only available for OpenAI models that support Priority Processing."
|
||||
status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast <normal|fast|auto|cold|status>`"
|
||||
unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, auto, cold, status"
|
||||
status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast <normal|fast|auto|cold|ultrafast|status>`"
|
||||
unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, auto, cold, ultrafast, status"
|
||||
saved: "⚡ ✓ Priority Processing: **{label}** (saved to config)\n_(takes effect on next message)_"
|
||||
session_only: "⚡ ✓ Priority Processing: **{label}** (this session only)"
|
||||
label_fast: "FAST"
|
||||
@@ -187,6 +187,8 @@ gateway:
|
||||
choice_normal: "normal — standard processing"
|
||||
choice_auto: "auto — fast for the first seconds of every turn"
|
||||
choice_cold: "cold — fast for the first turn of a session only"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6x price)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast is only available for GPT-6 Astra (current model: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Runtime footer: **{state}**\nFields: `{fields}`\nPlatform: `{platform}`"
|
||||
usage: "Usage: `/footer [on|off|status]`"
|
||||
@@ -1669,7 +1671,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Manage reasoning effort and display"
|
||||
fast:
|
||||
description: "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Show or change the display skin/theme"
|
||||
indicator:
|
||||
@@ -2751,7 +2753,9 @@ cli:
|
||||
feature_generic: "Fast mode"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Usage: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast is only available for GPT-6 Astra (current model: {model})."
|
||||
usage: "Usage: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} set to {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} mensajes comprimidos"
|
||||
slashCmd.session.compress.nothing: "nada que comprimir"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "modo rápido: {0}"
|
||||
slashCmd.session.fast.usage: "uso: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "uso: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "indicador: {0}"
|
||||
slashCmd.session.indicator.switched: "indicador → {0}"
|
||||
slashCmd.session.indicator.usage: "uso: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — procesamiento estándar"
|
||||
choice_auto: "auto — rápido en los primeros segundos de cada turno"
|
||||
choice_cold: "cold — rápido solo en el primer turno de una sesión"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, precio x6)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast solo está disponible para GPT-6 Astra (modelo actual: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Pie de ejecución: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`"
|
||||
usage: "Uso: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Gestionar el esfuerzo de razonamiento y su visualización"
|
||||
fast:
|
||||
description: "Modo rápido: OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Modo rápido: OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Mostrar o cambiar el skin/tema de la interfaz"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Modo rápido"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Uso: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast solo está disponible para GPT-6 Astra (modelo actual: {model})."
|
||||
usage: "Uso: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} establecido en {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} messages compressés"
|
||||
slashCmd.session.compress.nothing: "rien à compresser"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "mode rapide : {0}"
|
||||
slashCmd.session.fast.usage: "usage : /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "usage : /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "indicateur : {0}"
|
||||
slashCmd.session.indicator.switched: "indicateur → {0}"
|
||||
slashCmd.session.indicator.usage: "usage : /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -210,6 +210,8 @@ gateway:
|
||||
choice_normal: "normal — traitement standard"
|
||||
choice_auto: "auto — rapide pendant les premières secondes de chaque tour"
|
||||
choice_cold: "cold — rapide uniquement au premier tour d'une session"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, prix x6)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast n'est disponible que pour GPT-6 Astra (modèle actuel : `{model}`)."
|
||||
footer:
|
||||
status: "📎 Pied de page d'exécution : **{state}**\nChamps : `{fields}`\nPlateforme : `{platform}`"
|
||||
usage: "Usage : `/footer [on|off|status]`"
|
||||
@@ -1850,7 +1852,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Gérer l'effort de raisonnement et son affichage"
|
||||
fast:
|
||||
description: "Mode rapide — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Mode rapide — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Afficher ou changer le skin/thème d'affichage"
|
||||
indicator:
|
||||
@@ -2929,7 +2931,9 @@ cli:
|
||||
feature_generic: "Mode rapide"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Usage : /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast n'est disponible que pour GPT-6 Astra (modèle actuel : {model})."
|
||||
usage: "Usage : /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature} : {status}"
|
||||
set_to: "✓ {feature} défini sur {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "dlúthaíodh {0} teachtaireacht"
|
||||
slashCmd.session.compress.nothing: "níl aon rud le dlúthú"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} comhartha"
|
||||
slashCmd.session.fast.mode: "mód tapa: {0}"
|
||||
slashCmd.session.fast.usage: "úsáid: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "úsáid: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "táscaire: {0}"
|
||||
slashCmd.session.indicator.switched: "táscaire → {0}"
|
||||
slashCmd.session.indicator.usage: "úsáid: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — gnáthphróiseáil"
|
||||
choice_auto: "auto — tapa do na chéad soicindí de gach seal"
|
||||
choice_cold: "cold — tapa don chéad seal de sheisiún amháin"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, praghas x6)"
|
||||
ultrafast_not_supported: "⚡ Níl Ultrafast ar fáil ach do GPT-6 Astra (samhail reatha: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Buntásc rite: **{state}**\nRéimsí: `{fields}`\nArdán: `{platform}`"
|
||||
usage: "Úsáid: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Bainistigh iarracht agus taispeáint na réasúnaíochta"
|
||||
fast:
|
||||
description: "Mód tapa — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Mód tapa — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Taispeáin nó athraigh craiceann/téama na taispeána"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Mód tapa"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Úsáid: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Níl Ultrafast ar fáil ach do GPT-6 Astra (samhail reatha: {model})."
|
||||
usage: "Úsáid: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} socraithe go {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} üzenet tömörítve"
|
||||
slashCmd.session.compress.nothing: "nincs mit tömöríteni"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "gyors mód: {0}"
|
||||
slashCmd.session.fast.usage: "használat: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "használat: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "indikátor: {0}"
|
||||
slashCmd.session.indicator.switched: "indikátor → {0}"
|
||||
slashCmd.session.indicator.usage: "használat: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -187,6 +187,8 @@ gateway:
|
||||
choice_normal: "normal — normál feldolgozás"
|
||||
choice_auto: "auto — gyors minden kör első másodperceiben"
|
||||
choice_cold: "cold — gyors csak a munkamenet első körében"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, hatszoros ár)"
|
||||
ultrafast_not_supported: "⚡ Az Ultrafast csak a GPT-6 Astra modellhez érhető el (jelenlegi modell: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Futási idejű lábléc: **{state}**\nMezők: `{fields}`\nPlatform: `{platform}`"
|
||||
usage: "Használat: `/footer [on|off|status]`"
|
||||
@@ -1615,7 +1617,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Gondolkodási erőfeszítés és megjelenítés kezelése"
|
||||
fast:
|
||||
description: "Gyors mód — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Gyors mód — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "A megjelenítési skin/téma megjelenítése vagy módosítása"
|
||||
indicator:
|
||||
@@ -2668,7 +2670,9 @@ cli:
|
||||
feature_generic: "Gyors mód"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Használat: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Az Ultrafast csak a GPT-6 Astra modellhez érhető el (jelenlegi modell: {model})."
|
||||
usage: "Használat: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} beállítva: {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "compressi {0} messaggi"
|
||||
slashCmd.session.compress.nothing: "niente da comprimere"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "modalità veloce: {0}"
|
||||
slashCmd.session.fast.usage: "uso: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "uso: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "indicatore: {0}"
|
||||
slashCmd.session.indicator.switched: "indicatore → {0}"
|
||||
slashCmd.session.indicator.usage: "uso: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — elaborazione standard"
|
||||
choice_auto: "auto — veloce nei primi secondi di ogni turno"
|
||||
choice_cold: "cold — veloce solo nel primo turno di una sessione"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, prezzo x6)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast è disponibile solo per GPT-6 Astra (modello attuale: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Footer di runtime: **{state}**\nCampi: `{fields}`\nPiattaforma: `{platform}`"
|
||||
usage: "Uso: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Gestisci lo sforzo di ragionamento e la sua visualizzazione"
|
||||
fast:
|
||||
description: "Modalità veloce — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Modalità veloce — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Mostra o cambia la skin/tema di visualizzazione"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Modalità veloce"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Uso: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast è disponibile solo per GPT-6 Astra (modello attuale: {model})."
|
||||
usage: "Uso: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} impostato su {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} 件のメッセージを圧縮
|
||||
slashCmd.session.compress.nothing: "圧縮するものがありません"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "高速モード: {0}"
|
||||
slashCmd.session.fast.usage: "使い方: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "使い方: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "インジケーター: {0}"
|
||||
slashCmd.session.indicator.switched: "インジケーター → {0}"
|
||||
slashCmd.session.indicator.usage: "使い方: /indicator [{0}]"
|
||||
|
||||
+5
-1
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — 標準処理"
|
||||
choice_auto: "auto — 各ターンの最初の数秒間だけ高速"
|
||||
choice_cold: "cold — セッションの最初のターンのみ高速"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast(GPT-6 Astra、料金 6 倍)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast は GPT-6 Astra でのみ利用できます(現在のモデル: `{model}`)。"
|
||||
footer:
|
||||
status: "📎 ランタイムフッター: **{state}**\nフィールド: `{fields}`\nプラットフォーム: `{platform}`"
|
||||
usage: "使い方: `/footer [on|off|status]`"
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "高速モード"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "使い方: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast は GPT-6 Astra でのみ利用できます(現在のモデル: {model})。"
|
||||
usage: "使い方: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} を {value} に設定しました {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "메시지 {0}개 압축함"
|
||||
slashCmd.session.compress.nothing: "압축할 내용이 없어요"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} 토큰"
|
||||
slashCmd.session.fast.mode: "빠른 모드: {0}"
|
||||
slashCmd.session.fast.usage: "사용법: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "사용법: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "인디케이터: {0}"
|
||||
slashCmd.session.indicator.switched: "인디케이터 → {0}"
|
||||
slashCmd.session.indicator.usage: "사용법: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — 표준 처리"
|
||||
choice_auto: "auto — 매 턴의 처음 몇 초 동안 빠름"
|
||||
choice_cold: "cold — 세션의 첫 턴에만 빠름"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 가격 6배)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast는 GPT-6 Astra에서만 사용할 수 있습니다 (현재 모델: `{model}`)."
|
||||
footer:
|
||||
status: "📎 런타임 푸터: **{state}**\n필드: `{fields}`\n플랫폼: `{platform}`"
|
||||
usage: "사용법: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "추론 강도와 표시 관리"
|
||||
fast:
|
||||
description: "빠른 모드 — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "빠른 모드 — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "화면 스킨/테마 표시 또는 변경"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "빠른 모드"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "사용법: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast는 GPT-6 Astra에서만 사용할 수 있습니다 (현재 모델: {model})."
|
||||
usage: "사용법: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature}을(를) {value}(으)로 설정했어요 {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} mensagens compactadas"
|
||||
slashCmd.session.compress.nothing: "nada para compactar"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "modo rápido: {0}"
|
||||
slashCmd.session.fast.usage: "uso: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "uso: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "indicador: {0}"
|
||||
slashCmd.session.indicator.switched: "indicador → {0}"
|
||||
slashCmd.session.indicator.usage: "uso: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — processamento padrão"
|
||||
choice_auto: "auto — rápido nos primeiros segundos de cada turno"
|
||||
choice_cold: "cold — rápido apenas no primeiro turno de uma sessão"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, preço 6x)"
|
||||
ultrafast_not_supported: "⚡ O Ultrafast só está disponível para o GPT-6 Astra (modelo atual: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Rodapé de execução: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`"
|
||||
usage: "Uso: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Gerenciar o esforço e a exibição do raciocínio"
|
||||
fast:
|
||||
description: "Modo rápido — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Modo rápido — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Mostrar ou alterar a skin/tema de exibição"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Modo rápido"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Uso: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) O Ultrafast só está disponível para o GPT-6 Astra (modelo atual: {model})."
|
||||
usage: "Uso: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} definido como {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "сжато сообщений: {0}"
|
||||
slashCmd.session.compress.nothing: "нечего сжимать"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} ток."
|
||||
slashCmd.session.fast.mode: "быстрый режим: {0}"
|
||||
slashCmd.session.fast.usage: "Использование: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "Использование: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "индикатор: {0}"
|
||||
slashCmd.session.indicator.switched: "индикатор → {0}"
|
||||
slashCmd.session.indicator.usage: "Использование: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — стандартная обработка"
|
||||
choice_auto: "auto — быстро в первые секунды каждого хода"
|
||||
choice_cold: "cold — быстро только на первом ходе сессии"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, цена ×6)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast доступен только для GPT-6 Astra (текущая модель: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Нижний колонтитул среды выполнения: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`"
|
||||
usage: "Использование: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Управлять уровнем и отображением рассуждений"
|
||||
fast:
|
||||
description: "Быстрый режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Быстрый режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Показать или сменить скин/тему оформления"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Быстрый режим"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Использование: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast доступен только для GPT-6 Astra (текущая модель: {model})."
|
||||
usage: "Использование: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature}: установлено {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "{0} mesaj sıkıştırıldı"
|
||||
slashCmd.session.compress.nothing: "sıkıştırılacak bir şey yok"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "hızlı mod: {0}"
|
||||
slashCmd.session.fast.usage: "kullanım: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "kullanım: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "gösterge: {0}"
|
||||
slashCmd.session.indicator.switched: "gösterge → {0}"
|
||||
slashCmd.session.indicator.usage: "kullanım: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — standart işleme"
|
||||
choice_auto: "auto — her turun ilk saniyelerinde hızlı"
|
||||
choice_cold: "cold — yalnızca oturumun ilk turunda hızlı"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, 6 kat fiyat)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast yalnızca GPT-6 Astra için kullanılabilir (geçerli model: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Çalışma zamanı altbilgisi: **{state}**\nAlanlar: `{fields}`\nPlatform: `{platform}`"
|
||||
usage: "Kullanım: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Akıl yürütme düzeyini ve gösterimini yönet"
|
||||
fast:
|
||||
description: "Hızlı mod — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Hızlı mod — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Görüntü görünümünü/temasını göster veya değiştir"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Hızlı mod"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Kullanım: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast yalnızca GPT-6 Astra için kullanılabilir (geçerli model: {model})."
|
||||
usage: "Kullanım: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature} {value} olarak ayarlandı {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "стиснуто повідомле
|
||||
slashCmd.session.compress.nothing: "нічого стискати"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} ток"
|
||||
slashCmd.session.fast.mode: "швидкий режим: {0}"
|
||||
slashCmd.session.fast.usage: "використання: /fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "використання: /fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "індикатор: {0}"
|
||||
slashCmd.session.indicator.switched: "індикатор → {0}"
|
||||
slashCmd.session.indicator.usage: "використання: /indicator [{0}]"
|
||||
|
||||
+6
-2
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — стандартна обробка"
|
||||
choice_auto: "auto — швидко в перші секунди кожного ходу"
|
||||
choice_cold: "cold — швидко лише на першому ході сесії"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast (GPT-6 Astra, ціна ×6)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast доступний лише для GPT-6 Astra (поточна модель: `{model}`)."
|
||||
footer:
|
||||
status: "📎 Нижній колонтитул середовища: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`"
|
||||
usage: "Використання: `/footer [on|off|status]`"
|
||||
@@ -1598,7 +1600,7 @@ slash:
|
||||
reasoning:
|
||||
description: "Керувати рівнем зусиль міркування та його показом"
|
||||
fast:
|
||||
description: "Швидкий режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)"
|
||||
description: "Швидкий режим — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold/ultrafast)"
|
||||
skin:
|
||||
description: "Показати або змінити оформлення (тему) інтерфейсу"
|
||||
indicator:
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "Швидкий режим"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "Використання: /fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast доступний лише для GPT-6 Astra (поточна модель: {model})."
|
||||
usage: "Використання: /fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}: {status}"
|
||||
set_to: "✓ {feature}: встановлено {value} {scope}"
|
||||
hatch:
|
||||
|
||||
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "已壓縮 {0} 則訊息"
|
||||
slashCmd.session.compress.nothing: "沒有可壓縮的內容"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "快速模式:{0}"
|
||||
slashCmd.session.fast.usage: "用法:/fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "用法:/fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "指示器:{0}"
|
||||
slashCmd.session.indicator.switched: "指示器 → {0}"
|
||||
slashCmd.session.indicator.usage: "用法:/indicator [{0}]"
|
||||
|
||||
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — 標準處理"
|
||||
choice_auto: "auto — 每輪的前幾秒快速"
|
||||
choice_cold: "cold — 僅會話的第一輪快速"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast(GPT-6 Astra,6 倍價格)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast 僅適用於 GPT-6 Astra(目前模型:`{model}`)。"
|
||||
footer:
|
||||
status: "📎 執行階段頁尾:**{state}**\n欄位:`{fields}`\n平台:`{platform}`"
|
||||
usage: "用法:`/footer [on|off|status]`"
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "快速模式"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "用法:/fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast 僅適用於 GPT-6 Astra(目前模型:{model})。"
|
||||
usage: "用法:/fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}:{status}"
|
||||
set_to: "✓ {feature} 已設為 {value} {scope}"
|
||||
hatch:
|
||||
|
||||
+1
-1
@@ -933,7 +933,7 @@ slashCmd.session.compress.compressedOther: "已压缩 {0} 条消息"
|
||||
slashCmd.session.compress.nothing: "没有可压缩的内容"
|
||||
slashCmd.session.compress.tokSuffix: " · {0} tok"
|
||||
slashCmd.session.fast.mode: "快速模式:{0}"
|
||||
slashCmd.session.fast.usage: "用法:/fast [normal|fast|status|on|off|toggle]"
|
||||
slashCmd.session.fast.usage: "用法:/fast [normal|fast|ultrafast|status|on|off|toggle]"
|
||||
slashCmd.session.indicator.current: "指示器:{0}"
|
||||
slashCmd.session.indicator.switched: "指示器 → {0}"
|
||||
slashCmd.session.indicator.usage: "用法:/indicator [{0}]"
|
||||
|
||||
+5
-1
@@ -184,6 +184,8 @@ gateway:
|
||||
choice_normal: "normal — 标准处理"
|
||||
choice_auto: "auto — 每轮的前几秒快速"
|
||||
choice_cold: "cold — 仅会话的第一轮快速"
|
||||
choice_ultrafast: "ultrafast — OpenAI Ultrafast(GPT-6 Astra,6 倍价格)"
|
||||
ultrafast_not_supported: "⚡ Ultrafast 仅适用于 GPT-6 Astra(当前模型:`{model}`)。"
|
||||
footer:
|
||||
status: "📎 运行时页脚:**{state}**\n字段:`{fields}`\n平台:`{platform}`"
|
||||
usage: "用法:`/footer [on|off|status]`"
|
||||
@@ -2651,7 +2653,9 @@ cli:
|
||||
feature_generic: "快速模式"
|
||||
feature_anthropic: "Anthropic Fast Mode"
|
||||
feature_openai: "Priority Processing"
|
||||
usage: "用法:/fast [normal|fast|auto|cold|status] [--global]"
|
||||
feature_ultrafast: "OpenAI Ultrafast"
|
||||
ultrafast_not_supported: "(._.) Ultrafast 仅适用于 GPT-6 Astra(当前模型:{model})。"
|
||||
usage: "用法:/fast [normal|fast|auto|cold|ultrafast|status] [--global]"
|
||||
status: "{feature}:{status}"
|
||||
set_to: "✓ {feature} 已设为 {value} {scope}"
|
||||
hatch:
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
"""OpenAI Ultrafast (``service_tier: "ultrafast"``): one tier word table, a per-model request gate,
|
||||
and pricing from the tier the response was SERVED at. Relationship tests, no catalog snapshots."""
|
||||
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.fast_mode import SERVICE_TIER_WORDS, STATIC_TIERS, parse_service_tier
|
||||
from agent.usage_pricing import (
|
||||
_OPENAI_ULTRAFAST_PRICING,
|
||||
CanonicalUsage,
|
||||
estimate_usage_cost,
|
||||
with_served_service_tier,
|
||||
)
|
||||
from hermes_cli.models import model_supports_ultrafast, resolve_fast_mode_overrides
|
||||
|
||||
ASTRA_SPELLINGS = ("gpt-6-astra", "openai/gpt-6-astra", "gpt-6-astra-900k")
|
||||
|
||||
|
||||
def test_every_config_loader_parses_tiers_through_the_same_table(monkeypatch):
|
||||
from gateway.run import GatewayRunner
|
||||
from hermes_cli.cli_config_load import _parse_service_tier_config
|
||||
import tui_gateway.server as tui
|
||||
|
||||
for word, tier in {**SERVICE_TIER_WORDS, "normal": None, "off": None, "bogus": None}.items():
|
||||
monkeypatch.setattr(GatewayRunner, "_cfg_str", classmethod(lambda cls, *_k, _w=word: _w))
|
||||
monkeypatch.setattr(tui, "_load_cfg", lambda _w=word: {"agent": {"service_tier": _w}})
|
||||
assert parse_service_tier(word) == tier
|
||||
assert _parse_service_tier_config(word) == tier, word
|
||||
assert GatewayRunner._load_service_tier() == tier, word
|
||||
assert tui._load_service_tier() == tier, word
|
||||
assert "ultrafast" in STATIC_TIERS
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider,base_url", [("openai", "https://api.openai.com/v1"),
|
||||
("openai-codex", "https://chatgpt.com/backend-api/codex")])
|
||||
def test_ultrafast_is_requested_only_for_ultrafast_models_on_first_party_routes(provider, base_url):
|
||||
for model in ASTRA_SPELLINGS:
|
||||
assert model_supports_ultrafast(model), model
|
||||
assert resolve_fast_mode_overrides(model, provider=provider, base_url=base_url, tier="ultrafast") == {
|
||||
"service_tier": "ultrafast"}
|
||||
# A Priority-capable model without Ultrafast gets nothing, never a silent swap to another paid tier.
|
||||
assert resolve_fast_mode_overrides("gpt-6-sol", provider=provider, base_url=base_url) == {"service_tier": "priority"}
|
||||
assert resolve_fast_mode_overrides("gpt-6-sol", provider=provider, base_url=base_url, tier="ultrafast") is None
|
||||
# Proxies never see the tier.
|
||||
assert resolve_fast_mode_overrides("openai/gpt-6-astra", provider="openrouter",
|
||||
base_url="https://openrouter.ai/api/v1", tier="ultrafast") is None
|
||||
|
||||
|
||||
def test_cli_and_gateway_turn_routes_send_the_static_tier():
|
||||
import cli as cli_mod
|
||||
from gateway.run import GatewayRunner
|
||||
|
||||
stub = SimpleNamespace(model="gpt-6-astra", api_key="k", base_url="https://api.openai.com/v1", provider="openai",
|
||||
api_mode="codex_responses", acp_command=None, acp_args=[], _credential_pool=None,
|
||||
service_tier="ultrafast")
|
||||
assert cli_mod.HermesCLI._resolve_turn_agent_config(stub, "hi")["request_overrides"] == {"service_tier": "ultrafast"}
|
||||
runner = object.__new__(GatewayRunner)
|
||||
runner._service_tier = "ultrafast"
|
||||
rk = {"api_key": "k", "base_url": "https://api.openai.com/v1", "provider": "openai", "api_mode": "codex_responses",
|
||||
"command": None, "args": [], "credential_pool": None, "max_tokens": None}
|
||||
assert runner._resolve_turn_agent_config("hi", "gpt-6-astra", rk)["request_overrides"] == {"service_tier": "ultrafast"}
|
||||
|
||||
|
||||
def test_cli_refuses_ultrafast_on_a_model_without_it(monkeypatch):
|
||||
import cli as cli_mod
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
stub = SimpleNamespace(service_tier="priority", model="gpt-6-sol", agent=MagicMock(model="gpt-6-sol"),
|
||||
_fast_command_available=lambda: True)
|
||||
monkeypatch.setattr(cli_mod, "_cprint", lambda *a, **k: None)
|
||||
cli_mod.HermesCLI._handle_fast_command(stub, "/fast ultrafast")
|
||||
assert stub.service_tier == "priority"
|
||||
stub.model = stub.agent.model = "gpt-6-astra"
|
||||
cli_mod.HermesCLI._handle_fast_command(stub, "/fast ultrafast")
|
||||
assert stub.service_tier == "ultrafast"
|
||||
|
||||
|
||||
def _usage(prompt_uncached: int, served_tier=None) -> CanonicalUsage:
|
||||
usage = CanonicalUsage(input_tokens=prompt_uncached, output_tokens=10_000, cache_read_tokens=20_000,
|
||||
cache_write_tokens=5_000)
|
||||
return with_served_service_tier(usage, SimpleNamespace(service_tier=served_tier))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("prompt_uncached", [50_000, 400_000]) # below / above the 272K whole-request tier
|
||||
def test_served_ultrafast_bills_at_the_ultrafast_row_and_requested_only_does_not(prompt_uncached):
|
||||
for model in _OPENAI_ULTRAFAST_PRICING:
|
||||
standard = estimate_usage_cost(model, _usage(prompt_uncached), provider="openai-api")
|
||||
served_default = estimate_usage_cost(model, _usage(prompt_uncached, "default"), provider="openai-api")
|
||||
ultra = estimate_usage_cost(model, _usage(prompt_uncached, "ultrafast"), provider="openai-api")
|
||||
assert served_default.amount_usd == standard.amount_usd # asked for Ultrafast, served at Standard
|
||||
assert ultra.pricing_version == _OPENAI_ULTRAFAST_PRICING[model].pricing_version
|
||||
assert ultra.amount_usd == standard.amount_usd * 6 # every bucket, both context tiers
|
||||
|
||||
|
||||
def test_served_ultrafast_on_a_model_without_a_published_rate_is_unknown():
|
||||
assert estimate_usage_cost("gpt-6-sol", _usage(1_000, "ultrafast"), provider="openai-api").status == "unknown"
|
||||
|
||||
|
||||
def test_codex_stream_assembler_keeps_the_served_tier():
|
||||
from agent.codex_runtime import _consume_codex_event_stream
|
||||
|
||||
done = {"type": "response.completed", "response": {"id": "r1", "status": "completed", "service_tier": "default",
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1}}}
|
||||
events = [{"type": "response.output_text.delta", "delta": "PASS"}, done]
|
||||
assert _consume_codex_event_stream(iter(events), model="gpt-6-astra").service_tier == "default"
|
||||
@@ -233,7 +233,7 @@ def _cfg_get_fast(params):
|
||||
else session.get("create_service_tier_override"))
|
||||
if tier is None:
|
||||
tier = _load_service_tier()
|
||||
return {"value": "fast" if tier == "priority" else "normal"}
|
||||
return {"value": {"priority": "fast", "ultrafast": "ultrafast"}.get(tier, "normal")}
|
||||
|
||||
|
||||
def _cfg_get_thinking_mode(params):
|
||||
|
||||
@@ -164,7 +164,7 @@ def _set_model(rid, params, key, value, session):
|
||||
|
||||
|
||||
_FAST_WORDS = {"fast": "fast", "on": "fast", "normal": "normal", "off": "normal",
|
||||
"auto": "auto", "cold": "cold"}
|
||||
"auto": "auto", "cold": "cold", "ultrafast": "ultrafast"}
|
||||
|
||||
|
||||
def _set_fast(rid, params, key, value, session):
|
||||
@@ -176,13 +176,14 @@ def _set_fast(rid, params, key, value, session):
|
||||
current_tier = session["create_service_tier_override"] or None # pre-build pin beats global
|
||||
else:
|
||||
current_tier = _load_service_tier()
|
||||
from agent.fast_mode import STATIC_TIERS, service_tier_word
|
||||
if raw == "status":
|
||||
return _kv(rid, key, {"priority": "fast", None: "normal", "": "normal"}.get(current_tier, current_tier))
|
||||
nv = _FAST_WORDS.get(raw, ("normal" if current_tier == "priority" else "fast") if raw in {"", "toggle"} else None)
|
||||
return _kv(rid, key, service_tier_word(current_tier))
|
||||
nv = _FAST_WORDS.get(raw, ("normal" if current_tier in STATIC_TIERS else "fast") if raw in {"", "toggle"} else None)
|
||||
if nv is None:
|
||||
return _err(rid, 4002, f"unknown fast mode: {value}")
|
||||
overrides = None
|
||||
if nv == "fast":
|
||||
if nv in ("fast", "ultrafast"):
|
||||
from hermes_cli.models import resolve_fast_mode_overrides
|
||||
if agent is not None:
|
||||
target_model = getattr(agent, "model", None)
|
||||
@@ -192,9 +193,10 @@ def _set_fast(rid, params, key, value, session):
|
||||
if not target_model:
|
||||
return _err(rid, 4002, "fast mode is not available without a selected model")
|
||||
overrides = resolve_fast_mode_overrides(target_model, provider=getattr(agent, "provider", None),
|
||||
base_url=getattr(agent, "base_url", None))
|
||||
base_url=getattr(agent, "base_url", None),
|
||||
tier="ultrafast" if nv == "ultrafast" else None)
|
||||
if overrides is None:
|
||||
return _err(rid, 4002, "fast mode is not available for this model")
|
||||
return _err(rid, 4002, f"{nv} mode is not available for this model")
|
||||
if session is not None:
|
||||
# Session-scoped like `reasoning` (global = `--global` / Settings → Model): writing config.yaml
|
||||
# here flipped fast mode for every surface. The create override survives rebuilds; "" pins normal.
|
||||
|
||||
@@ -323,7 +323,8 @@ def _mirror_prompt(sid, session, agent, arg) -> None:
|
||||
agent._cached_system_prompt = None
|
||||
|
||||
|
||||
_FAST_TIERS = {"fast": "priority", "on": "priority", "normal": None, "off": None, "auto": "auto", "cold": "cold"}
|
||||
_FAST_TIERS = {"fast": "priority", "on": "priority", "normal": None, "off": None, "auto": "auto", "cold": "cold",
|
||||
"ultrafast": "ultrafast"}
|
||||
|
||||
|
||||
def _mirror_fast(sid, session, agent, arg) -> None:
|
||||
|
||||
@@ -29,6 +29,7 @@ from hermes_cli.env_loader import load_hermes_dotenv
|
||||
from utils import file_signature, is_truthy_value
|
||||
from hermes_state_ids import new_session_id
|
||||
from tools.environments.local import hermes_subprocess_env
|
||||
from agent.fast_mode import STATIC_TIERS
|
||||
from agent.replay_cleanup import canonicalize_replay_history
|
||||
from agent.reasoning_effort import clamp_effort, route_supported_efforts
|
||||
from agent.compaction_display import project_compaction_message_for_display # noqa: F401
|
||||
@@ -1870,12 +1871,10 @@ def _load_reasoning_config(model: str = "") -> dict | None:
|
||||
return resolve_reasoning_config(_load_cfg(), model)
|
||||
|
||||
|
||||
_SERVICE_TIER_ALIASES = {"fast": "priority", "priority": "priority", "on": "priority", "auto": "auto", "cold": "cold"}
|
||||
|
||||
|
||||
def _load_service_tier() -> str | None:
|
||||
raw = str((_load_cfg().get("agent") or {}).get("service_tier", "") or "").strip().lower()
|
||||
return _SERVICE_TIER_ALIASES.get(raw)
|
||||
from agent.fast_mode import parse_service_tier
|
||||
|
||||
return parse_service_tier((_load_cfg().get("agent") or {}).get("service_tier", ""))
|
||||
|
||||
|
||||
def _load_provider_routing() -> dict:
|
||||
@@ -2262,7 +2261,7 @@ def _live_session_identity(session: dict) -> tuple[str, str]:
|
||||
return str(model), str(provider or "")
|
||||
|
||||
|
||||
def _fast_tier_applies(agent, model: str, provider: str, *, route_known: bool) -> bool:
|
||||
def _fast_tier_applies(agent, model: str, provider: str, *, route_known: bool, tier: str | None = None) -> bool:
|
||||
"""Whether a priority tier reaches this session's route. Every request builder asks the same gate, so a
|
||||
profile-wide ``service_tier: fast`` sends nothing to a local server or a proxy, and the session must not
|
||||
report Fast there either. ``route_known`` is False while a switch is pending: the agent's base URL still
|
||||
@@ -2274,7 +2273,7 @@ def _fast_tier_applies(agent, model: str, provider: str, *, route_known: bool) -
|
||||
base_url = getattr(agent, "_anthropic_base_url", None)
|
||||
base_url = base_url or getattr(agent, "base_url", None)
|
||||
try:
|
||||
return resolve_fast_mode_overrides(model, provider=provider or None, base_url=base_url) is not None
|
||||
return resolve_fast_mode_overrides(model, provider=provider or None, base_url=base_url, tier=tier) is not None
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@@ -2325,8 +2324,8 @@ def _session_info(agent, session: dict | None = None) -> dict:
|
||||
"provider": pending_provider or provider,
|
||||
"reasoning_effort": reasoning_effort, "reasoning_effort_wire": reasoning_effort_wire,
|
||||
"service_tier": service_tier,
|
||||
"fast": service_tier == "priority" and _fast_tier_applies(agent, model, pending_provider or provider,
|
||||
route_known=not pending_provider),
|
||||
"fast": service_tier in STATIC_TIERS and _fast_tier_applies(agent, model, pending_provider or provider,
|
||||
route_known=not pending_provider, tier=service_tier),
|
||||
"yolo": yolo, "approval_mode": approval_mode,
|
||||
"tools": dict(mirror.get("tools") or {}) if isinstance(mirror.get("tools"), dict) else {},
|
||||
"skills": dict(mirror.get("skills") or {}) if isinstance(mirror.get("skills"), dict) else {},
|
||||
|
||||
@@ -26,6 +26,12 @@ const TUI_SESSION_MODEL_RE = new RegExp(`(?:^|\\s)${TUI_SESSION_MODEL_FLAG}(?:\\
|
||||
const REASONING_SESSION_FLAGS = new Set(['--session'])
|
||||
const REASONING_GLOBAL_FLAGS = new Set(['--global'])
|
||||
|
||||
type FastModeWord = 'fast' | 'normal' | 'ultrafast'
|
||||
|
||||
// `config.get/set fast` answer fast | ultrafast | normal (auto/cold windows read as normal here).
|
||||
const fastModeWord = (value: unknown): FastModeWord =>
|
||||
value === 'fast' || value === 'ultrafast' ? value : 'normal'
|
||||
|
||||
const modelValueForConfigSet = (arg: string) => {
|
||||
const trimmed = arg.trim()
|
||||
|
||||
@@ -606,11 +612,11 @@ export const sessionCommands: SlashCommand[] = [
|
||||
},
|
||||
|
||||
{
|
||||
help: 'toggle fast mode [normal|fast|status|on|off|toggle]',
|
||||
help: 'toggle fast mode [normal|fast|ultrafast|status|on|off|toggle]',
|
||||
name: 'fast',
|
||||
run: (arg, ctx) => {
|
||||
const mode = arg.trim().toLowerCase()
|
||||
const valid = new Set(['', 'status', 'normal', 'fast', 'on', 'off', 'toggle'])
|
||||
const valid = new Set(['', 'status', 'normal', 'fast', 'ultrafast', 'on', 'off', 'toggle'])
|
||||
|
||||
if (!valid.has(mode)) {
|
||||
return ctx.transcript.sys(t('slashCmd.session.fast.usage'))
|
||||
@@ -621,7 +627,7 @@ export const sessionCommands: SlashCommand[] = [
|
||||
.rpc<ConfigGetValueResponse>('config.get', { key: 'fast', session_id: ctx.sid })
|
||||
.then(
|
||||
ctx.guarded<ConfigGetValueResponse>(r =>
|
||||
ctx.transcript.sys(t('slashCmd.session.fast.mode', r.value === 'fast' ? 'fast' : 'normal'))
|
||||
ctx.transcript.sys(t('slashCmd.session.fast.mode', fastModeWord(r.value)))
|
||||
)
|
||||
)
|
||||
.catch(ctx.guardedErr)
|
||||
@@ -631,15 +637,15 @@ export const sessionCommands: SlashCommand[] = [
|
||||
.rpc<ConfigSetResponse>('config.set', { key: 'fast', session_id: ctx.sid, value: mode })
|
||||
.then(
|
||||
ctx.guarded<ConfigSetResponse>(r => {
|
||||
const next = r.value === 'fast' ? 'fast' : 'normal'
|
||||
const next = fastModeWord(r.value)
|
||||
ctx.transcript.sys(t('slashCmd.session.fast.mode', next))
|
||||
patchUiState(state => ({
|
||||
...state,
|
||||
info: state.info
|
||||
? {
|
||||
...state.info,
|
||||
fast: next === 'fast',
|
||||
service_tier: next === 'fast' ? 'priority' : ''
|
||||
fast: next !== 'normal',
|
||||
service_tier: { fast: 'priority', normal: '', ultrafast: 'ultrafast' }[next]
|
||||
}
|
||||
: state.info
|
||||
}))
|
||||
|
||||
@@ -30,7 +30,7 @@ export const slashCmdSessionEn = {
|
||||
},
|
||||
fast: {
|
||||
mode: (mode: string) => `fast mode: ${mode}`,
|
||||
usage: 'usage: /fast [normal|fast|status|on|off|toggle]'
|
||||
usage: 'usage: /fast [normal|fast|ultrafast|status|on|off|toggle]'
|
||||
},
|
||||
indicator: {
|
||||
current: (style: string) => `indicator: ${style}`,
|
||||
|
||||
Reference in New Issue
Block a user