OR-341: Alert waiting agents and the dashboard when Slurm monitoring is lost (#475)

* Surface lost Slurm monitoring to waiting agents and the dashboard

Keep supervisor stderr in run-logs/<id>.supervisor.log, alert a session's
pending wake-up once per monitoring outage, and show monitoringError on live runs.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Debounce Slurm monitoring errors and combine per-session alerts

Report monitoringError only after 60s of failed polls, send one alert per session
listing every newly unmonitored run, dedupe repeated ssh log-stream errors, and
word the alert without promising a wake-up that needs monitoring back.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Keep monitoring tests in their temp store and log each poll failure

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Log monitoring recovery and pin alert formatting in tests

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Exercise the Slurm monitoring grace period in the supervisor integration test

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Drop stray bytecode and anchor the grace-period check on the first poll

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Keep remote ssh text out of monitoring alerts and bound the supervisor log

Name unmonitored runs by local ids and point the agent at orx exp status, show
a warning from any live run on the dashboard, and restart the supervisor log
past 1 MiB.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* Show every live run's monitoring reason and roll the supervisor log over

orx exp status now prints the monitoring error of older live runs, the dashboard
takes the newest live run's error through one helper, and a full supervisor log
is renamed rather than truncated under a running supervisor.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Myles Anderson
2026-09-29 16:39:35 -07:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 85b6f61d0a
commit ddb603173b
14 changed files with 601 additions and 245 deletions
+4 -2
View File
@@ -115,8 +115,10 @@ orx exp wait <expId> --interval 10 --timeout 3600
After launching, use `orx exp wake <expId>` when you want to end the turn and
resume after that run succeeds or fails. Wake-up is opt-in, fires only for
`done` or `failed`, and waits behind queued user messages. Use either wait or
wake for a run, not both.
`done` or `failed`, and waits behind queued user messages. If orx cannot
monitor a live Slurm run for over a minute, it tells you once per outage; the
final wake-up comes once orx can see the run finish. Use either wait or wake
for a run, not both.
## Sizing compute
+15 -2
View File
@@ -172,7 +172,9 @@ pub(crate) fn default_hf_image(flavor: &str) -> String {
}
}
/// Spawn `orx supervise <runId>` fully detached (own process group, no stdio),
const SUPERVISOR_LOG_MAX_BYTES: u64 = 1024 * 1024;
/// Spawn `orx supervise <runId>` fully detached (own process group, stderr to a log),
/// so it outlives this command and any SSH session that launched it.
pub(crate) fn spawn_detached_supervise(run_id: &str) -> Result<()> {
let exe = crate::paths::spawnable_exe().map_err(|e| {
@@ -181,13 +183,24 @@ pub(crate) fn spawn_detached_supervise(run_id: &str) -> Result<()> {
e
)
})?;
// Supervisor diagnostics (retries, transitions) exist only on stderr; keep them per run.
let path = crate::store::log_path(run_id).with_extension("supervisor.log");
if std::fs::metadata(&path).is_ok_and(|meta| meta.len() >= SUPERVISOR_LOG_MAX_BYTES) {
// Rename rather than truncate: a still-running supervisor keeps writing to its handle.
let _ = std::fs::rename(&path, path.with_extension("log.1"));
}
let stderr = std::fs::OpenOptions::new()
.create(true)
.append(true)
.open(path)
.map_or_else(|_| std::process::Stdio::null(), std::process::Stdio::from);
// A long-lived `orx up` may be running a replaced binary; spawn the new file at its path.
let mut cmd = std::process::Command::new(exe);
cmd.arg("supervise")
.arg(run_id)
.stdin(std::process::Stdio::null())
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null());
.stderr(stderr);
// The supervisor re-resolves its directories from its own environment, so
// without this a run launched from the macOS app is tracked in a different
// store than the app is reading.
+36 -6
View File
@@ -24,6 +24,8 @@ use crate::store::{log_path, now_ms, RunStatus, Store};
const POLL_INTERVAL: Duration = Duration::from_secs(5);
/// How long a silent log stream is held before re-checking job state.
const LOG_IDLE: Duration = Duration::from_secs(30);
/// How long monitoring must keep failing before it is reported on the run.
const MONITORING_GRACE: Duration = Duration::from_secs(60);
fn open_supervisor_lock(path: &std::path::Path) -> Result<fd_lock::RwLock<std::fs::File>> {
let file = std::fs::OpenOptions::new()
@@ -712,13 +714,24 @@ async fn tail_logs_ssh(
}
};
let mut seen = 0u64;
let mut last_error = None;
loop {
let mut sink = |line: &str| {
let _ = writeln!(log_file, "{line}");
};
match ssh::stream_logs(&target, &dir, seen, LOG_IDLE, &mut sink).await {
Ok(s) => seen = s,
Err(err) => eprintln!("supervise {run_id}: log stream error (will retry): {err}"),
Ok(s) => {
seen = s;
last_error = None;
}
// Retries every 2s through an outage; log each distinct failure once.
Err(err) => {
let err = err.to_string();
if last_error.as_ref() != Some(&err) {
eprintln!("supervise {run_id}: log stream error (will retry): {err}");
last_error = Some(err);
}
}
}
let _ = log_file.flush();
if *done.borrow() {
@@ -1101,6 +1114,8 @@ async fn run_slurm(
let mut last_status = status_of(&stored)?;
let mut cancel_sent = descriptor.cancellation_accepted;
let mut failing_since = None;
let mut last_error = None;
loop {
if !cancel_sent && local_cancel_requested(&store, &run_id) {
@@ -1117,12 +1132,27 @@ async fn run_slurm(
Ok(job) if matches!(job.stage.as_str(), "GONE" | "UNAVAILABLE") => Some("Monitoring unavailable: no exit status or scheduler record. Check cluster/accounting availability; the job has not been declared stopped.".into()),
_ => None,
};
if error != descriptor.monitoring_error || cancel_sent != descriptor.cancellation_accepted {
if let Some(error) = &error {
eprintln!("supervise {run_id}: {error}");
match (&error, &last_error) {
(Some(message), last) if last.as_ref() != Some(message) => {
eprintln!("supervise {run_id}: {message}");
}
(None, Some(_)) => eprintln!("supervise {run_id}: monitoring restored"),
_ => {}
}
last_error = error.clone();
failing_since = error
.as_ref()
.map(|_| failing_since.unwrap_or_else(std::time::Instant::now));
// Brief outages recover on their own; a restarted supervisor keeps an existing report.
let reported = error.clone().filter(|_| {
descriptor.monitoring_error.is_some()
|| failing_since.is_some_and(|since| since.elapsed() >= MONITORING_GRACE)
});
if reported != descriptor.monitoring_error
|| cancel_sent != descriptor.cancellation_accepted
{
let mut updated = descriptor.clone();
updated.monitoring_error = error.clone();
updated.monitoring_error = reported;
updated.cancellation_accepted = cancel_sent;
match store.set_backend_json(&run_id, &updated.to_json()) {
Ok(()) => descriptor = updated,
+173
View File
@@ -7786,6 +7786,26 @@ fn run_wakeup_text(run: &crate::store::StoredRun) -> Option<String> {
})
}
// Only local identifiers here: the error text includes remote ssh stderr, so the agent reads it
// as command output instead of inside an `[orx]` instruction.
fn run_monitoring_text(runs: &[crate::store::StoredRun]) -> String {
let lines = runs
.iter()
.map(|run| {
format!(
"- run `{}` of experiment `{}` (still **{}**)",
run.id, run.experiment_id, run.status
)
})
.collect::<Vec<_>>()
.join("\n");
format!(
"[orx] orx can no longer monitor these live runs:\n{lines}\nRun `orx exp status <expId>` \
for the reason and tell the user so they can fix it. orx wakes you once it can see a run \
finish; until then each run keeps its current status."
)
}
fn first_wakeup_per_session(wakeups: Vec<crate::store::RunWakeup>) -> Vec<crate::store::RunWakeup> {
let mut seen_sessions = HashSet::new();
wakeups
@@ -7861,6 +7881,70 @@ async fn process_run_wakeups(
}
}
}
process_run_monitoring_alerts(chat, store, data_dir_move_in_progress).await
}
/// Tells each waiting session, once per outage, which of its live runs can no
/// longer be monitored, leaving their terminal wake-ups pending.
async fn process_run_monitoring_alerts(
chat: &Arc<ChatHost>,
store: Store,
data_dir_move_in_progress: Option<&std::sync::atomic::AtomicBool>,
) -> Result<()> {
let mut unalerted = Vec::new();
for wakeup in store.list_active_run_wakeups()? {
let unmonitored = crate::jobs::BackendDescriptor::parse(&wakeup.run.backend_json)
.is_ok_and(|descriptor| descriptor.monitoring_error.is_some());
if unmonitored && !wakeup.monitoring_alerted {
unalerted.push(wakeup);
} else if !unmonitored && wakeup.monitoring_alerted {
store.set_run_wakeup_monitoring_alerted(
&wakeup.run.id,
&wakeup.chat_session_id,
false,
)?;
}
}
let mut by_session: HashMap<String, Vec<crate::store::RunWakeup>> = HashMap::new();
for wakeup in unalerted {
by_session
.entry(wakeup.chat_session_id.clone())
.or_default()
.push(wakeup);
}
for (session_id, alerts) in by_session {
let Some(mut guard) = TurnGuard::claim_hidden(chat, &session_id).await else {
continue;
};
if data_dir_move_in_progress
.is_some_and(|flag| flag.load(std::sync::atomic::Ordering::SeqCst))
{
guard.release().await;
return Ok(());
}
let mut claimed = Vec::new();
for wakeup in alerts {
if store.set_run_wakeup_monitoring_alerted(&wakeup.run.id, &session_id, true)? {
claimed.push(wakeup.run);
}
}
if claimed.is_empty() {
guard.release().await;
continue;
}
let text = run_monitoring_text(&claimed);
let started = chat.send_hidden_message(&session_id, text, guard).await;
if !matches!(started, Ok(TurnSubmission::Started(_))) {
for run in &claimed {
store.set_run_wakeup_monitoring_alerted(&run.id, &session_id, false)?;
}
if let Err(err) = started {
if !chat.is_busy(&session_id).await {
eprintln!("orx up: run watcher: {err}");
}
}
}
}
Ok(())
}
@@ -10332,11 +10416,13 @@ with other project runs using `orx runs p1` and inspect the file located by `orx
run: first,
chat_session_id: "owner".into(),
state: "pending".into(),
monitoring_alerted: false,
},
crate::store::RunWakeup {
run: second,
chat_session_id: "owner".into(),
state: "pending".into(),
monitoring_alerted: false,
},
]);
@@ -10506,6 +10592,93 @@ with other project runs using `orx runs p1` and inspect the file located by `orx
drop(store);
let _ = std::fs::remove_dir_all(dir);
}
fn unmonitored_run() -> StoredRun {
StoredRun {
backend_json:
r#"{"kind":"slurm_job","monitoringError":"Monitoring unavailable: ssh failed."}"#
.into(),
..run("starting")
}
}
#[test]
fn monitoring_message_names_runs_without_remote_text() {
let first = StoredRun {
backend_json: r#"{"kind":"slurm_job","monitoringError":"REMOTE_SENTINEL"}"#.into(),
..run("starting")
};
let mut second = run("running");
second.id = "run_y".into();
let text = run_monitoring_text(&[first, second]);
assert!(!text.contains("REMOTE_SENTINEL"));
assert_eq!(
text,
"[orx] orx can no longer monitor these live runs:\n\
- run `run_x` of experiment `exp_1` (still **starting**)\n\
- run `run_y` of experiment `exp_1` (still **running**)\n\
Run `orx exp status <expId>` for the reason and tell the user so they can fix it. orx wakes you \
once it can see a run finish; until then each run keeps its current status."
);
}
#[tokio::test]
async fn busy_session_leaves_monitoring_alert_unclaimed() {
let (store, dir) = temp_store("alert-busy");
session(&store, "owner");
store.upsert_run(&unmonitored_run()).unwrap();
store.register_run_wakeup("run_x", "owner").unwrap();
let host = Arc::new(ChatHost::new(
Arc::new(crate::local::opencode::AgentHost::new(None)),
Arc::new(crate::local::codex::CodexHost::new()),
Arc::new(crate::local::claude::ClaudeHost::new()),
));
host.turns
.lock()
.await
.insert("owner".into(), TurnState::Draining);
drop(store);
process_run_wakeups(&host, Store::open_at(dir.clone()).unwrap(), None)
.await
.unwrap();
let store = Store::open_at(dir.clone()).unwrap();
assert!(!store.list_active_run_wakeups().unwrap()[0].monitoring_alerted);
assert!(matches!(
host.turns.lock().await.get("owner"),
Some(TurnState::Draining)
));
drop(store);
let _ = std::fs::remove_dir_all(dir);
}
#[tokio::test]
async fn restored_monitoring_rearms_the_alert() {
let (store, dir) = temp_store("alert-restored");
session(&store, "owner");
store.upsert_run(&run("running")).unwrap();
store.register_run_wakeup("run_x", "owner").unwrap();
store
.set_run_wakeup_monitoring_alerted("run_x", "owner", true)
.unwrap();
let host = Arc::new(ChatHost::new(
Arc::new(crate::local::opencode::AgentHost::new(None)),
Arc::new(crate::local::codex::CodexHost::new()),
Arc::new(crate::local::claude::ClaudeHost::new()),
));
drop(store);
process_run_wakeups(&host, Store::open_at(dir.clone()).unwrap(), None)
.await
.unwrap();
let store = Store::open_at(dir.clone()).unwrap();
assert!(!store.list_active_run_wakeups().unwrap()[0].monitoring_alerted);
assert!(!host.is_busy("owner").await);
drop(store);
let _ = std::fs::remove_dir_all(dir);
}
}
#[cfg(test)]
+12
View File
@@ -206,6 +206,18 @@ impl LocalPlane {
}
None => println!(" last run: — (never run)"),
}
// A forced relaunch can leave an older run live; monitoring alerts send agents here.
for older in store.list_runs_by_experiment(&exp.id)?.iter().skip(1) {
if !matches!(older.status.as_str(), "starting" | "running") {
continue;
}
let error = crate::jobs::BackendDescriptor::parse(&older.backend_json)
.ok()
.and_then(|backend| backend.monitoring_error);
if let Some(error) = error {
println!(" older live run {}: {error}", older.id);
}
}
Ok(())
}
+99 -5
View File
@@ -337,6 +337,7 @@ pub struct RunWakeup {
pub run: StoredRun,
pub chat_session_id: String,
pub state: String,
pub monitoring_alerted: bool,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
@@ -554,6 +555,7 @@ impl Store {
claim_token TEXT,
claimed_at INTEGER,
delivered_at INTEGER,
monitoring_alerted INTEGER NOT NULL DEFAULT 0,
PRIMARY KEY(run_id, chat_session_id)
);
CREATE INDEX IF NOT EXISTS idx_chat_run_wakeups_requested
@@ -639,6 +641,7 @@ impl Store {
"ALTER TABLE chat_run_wakeups ADD COLUMN claim_token TEXT",
"ALTER TABLE chat_run_wakeups ADD COLUMN claimed_at INTEGER",
"ALTER TABLE chat_run_wakeups ADD COLUMN delivered_at INTEGER",
"ALTER TABLE chat_run_wakeups ADD COLUMN monitoring_alerted INTEGER NOT NULL DEFAULT 0",
"ALTER TABLE chat_messages ADD COLUMN parent_id TEXT",
"ALTER TABLE chat_messages ADD COLUMN base_native_session_id TEXT",
"ALTER TABLE chat_messages ADD COLUMN result_native_session_id TEXT",
@@ -1172,26 +1175,58 @@ impl Store {
}
pub fn list_ready_run_wakeups(&self) -> Result<Vec<RunWakeup>> {
let mut stmt = self.conn.prepare(
self.list_run_wakeups(
"w.state IN ('pending', 'claimed') AND r.status IN ('done', 'failed')
ORDER BY COALESCE(r.ended_at, r.updated_at), w.requested_at, r.id",
)
}
/// Pending wake-ups whose run is still live, for monitoring alerts.
pub fn list_active_run_wakeups(&self) -> Result<Vec<RunWakeup>> {
self.list_run_wakeups(
"w.state = 'pending' AND r.status IN ('starting', 'running')
ORDER BY w.requested_at, r.id",
)
}
fn list_run_wakeups(&self, filter: &'static str) -> Result<Vec<RunWakeup>> {
let mut stmt = self.conn.prepare(&format!(
"SELECT r.id, r.experiment_id, r.project_id, r.status, r.backend_json, r.command,
r.created_at, r.updated_at, r.ended_at, r.exit_code,
r.commit_sha, r.result_markdown, r.cancel_requested, r.chat_session_id,
w.chat_session_id, w.state
w.chat_session_id, w.state, w.monitoring_alerted
FROM chat_run_wakeups w
JOIN runs r ON r.id = w.run_id
WHERE w.state IN ('pending', 'claimed') AND r.status IN ('done', 'failed')
ORDER BY COALESCE(r.ended_at, r.updated_at), w.requested_at, r.id",
)?;
WHERE {filter}"
))?;
let rows = stmt.query_map([], |row| {
Ok(RunWakeup {
run: row_to_run(row)?,
chat_session_id: row.get(14)?,
state: row.get(15)?,
monitoring_alerted: row.get(16)?,
})
})?;
Ok(rows.collect::<std::result::Result<Vec<_>, _>>()?)
}
/// Flips a pending wake-up's monitoring alert flag; `true` when this call changed it,
/// which makes setting it an atomic claim across `orx up` processes.
pub fn set_run_wakeup_monitoring_alerted(
&self,
run_id: &str,
chat_session_id: &str,
alerted: bool,
) -> Result<bool> {
let changed = self.conn.execute(
"UPDATE chat_run_wakeups SET monitoring_alerted = ?3
WHERE run_id = ?1 AND chat_session_id = ?2
AND state = 'pending' AND monitoring_alerted != ?3",
params![run_id, chat_session_id, alerted],
)?;
Ok(changed == 1)
}
pub fn claim_run_wakeup(&self, run_id: &str, chat_session_id: &str) -> Result<Option<String>> {
let token = uuid::Uuid::new_v4().to_string();
let claimed = self.conn.execute(
@@ -5000,6 +5035,65 @@ mod tests {
let _ = std::fs::remove_dir_all(dir);
}
#[test]
fn monitoring_alert_claims_once_and_keeps_the_terminal_wakeup() {
let dir = std::env::temp_dir().join(format!("orx-store-alert-{}", uuid::Uuid::new_v4()));
let first = Store::open_at(dir.clone()).unwrap();
first
.create_chat_session(&chat_session_fixture("chat_A"))
.unwrap();
for (id, status) in [("run_live", "starting"), ("run_done", "done")] {
first
.upsert_run(&run_fixture(id, status, Some("chat_A")))
.unwrap();
first.register_run_wakeup(id, "chat_A").unwrap();
}
let second = Store::open_at(dir.clone()).unwrap();
let active = first.list_active_run_wakeups().unwrap();
assert_eq!(active.len(), 1);
assert_eq!(active[0].run.id, "run_live");
assert!(!active[0].monitoring_alerted);
assert!(first
.set_run_wakeup_monitoring_alerted("run_live", "chat_A", true)
.unwrap());
assert!(!second
.set_run_wakeup_monitoring_alerted("run_live", "chat_A", true)
.unwrap());
assert!(second.list_active_run_wakeups().unwrap()[0].monitoring_alerted);
assert!(second
.set_run_wakeup_monitoring_alerted("run_live", "chat_A", false)
.unwrap());
assert!(!first
.set_run_wakeup_monitoring_alerted("run_live", "chat_A", false)
.unwrap());
first
.set_run_wakeup_monitoring_alerted("run_live", "chat_A", true)
.unwrap();
assert!(first
.update_status("run_live", RunStatus::Done, Some(2), Some(0))
.unwrap());
assert!(first.list_active_run_wakeups().unwrap().is_empty());
assert!(first
.list_ready_run_wakeups()
.unwrap()
.iter()
.any(|wakeup| wakeup.run.id == "run_live" && wakeup.state == "pending"));
first
.claim_run_wakeup("run_live", "chat_A")
.unwrap()
.unwrap();
assert!(!first
.set_run_wakeup_monitoring_alerted("run_live", "chat_A", false)
.unwrap());
drop(second);
drop(first);
let _ = std::fs::remove_dir_all(dir);
}
#[test]
fn run_wakeup_claim_is_atomic_across_store_connections_and_recoverable() {
let dir = std::env::temp_dir().join(format!("orx-store-claim-{}", uuid::Uuid::new_v4()));
+16 -11
View File
@@ -218,7 +218,7 @@ esac
.unwrap();
std::fs::set_permissions(&ssh, std::fs::Permissions::from_mode(0o755)).unwrap();
let phase = sandbox.0.join("phase");
std::fs::write(&phase, "outage").unwrap();
std::fs::write(&phase, "missing").unwrap();
struct ChildGuard(std::process::Child);
impl Drop for ChildGuard {
fn drop(&mut self) {
@@ -244,8 +244,8 @@ esac
.unwrap();
serde_json::from_str(&raw).unwrap()
};
let wait = |condition: &dyn Fn() -> bool| {
let deadline = Instant::now() + Duration::from_secs(15);
let wait_for = |timeout: Duration, condition: &dyn Fn() -> bool| {
let deadline = Instant::now() + timeout;
while !condition() {
assert!(
Instant::now() < deadline,
@@ -254,19 +254,24 @@ esac
std::thread::sleep(Duration::from_millis(100));
}
};
let wait = |condition: &dyn Fn() -> bool| wait_for(Duration::from_secs(15), condition);
// A missing scheduler record is reported only after about a minute, without failing the run.
wait(&|| {
std::fs::read_to_string(sandbox.0.join("calls")).is_ok_and(|c| c.contains("exit_code"))
});
std::thread::sleep(Duration::from_secs(10));
assert!(metadata()["monitoringError"].is_null());
wait_for(Duration::from_secs(75), &|| {
metadata()["monitoringError"]
.as_str()
.is_some_and(|s| s.contains("scheduler record"))
});
std::fs::write(&phase, "outage").unwrap();
wait(&|| {
metadata()["monitoringError"]
.as_str()
.is_some_and(|s| s.contains("Reconnect"))
});
std::fs::write(&phase, "missing").unwrap();
wait(&|| {
metadata()["monitoringError"]
.as_str()
.is_some_and(|s| s.contains("scheduler record"))
});
// Exceeds the old one-minute GONE-to-failed threshold.
std::thread::sleep(Duration::from_secs(61));
let status: String = db
.query_row("SELECT status FROM runs WHERE id='run'", [], |r| r.get(0))
.unwrap();
@@ -1,4 +1,4 @@
var a1=Object.defineProperty;var cw=h=>{throw TypeError(h)};var o1=(h,e,t)=>e in h?a1(h,e,{enumerable:!0,configurable:!0,writable:!0,value:t}):h[e]=t;var L=(h,e,t)=>o1(h,typeof e!="symbol"?e+"":e,t),By=(h,e,t)=>e.has(h)||cw("Cannot "+t);var a=(h,e,t)=>(By(h,e,"read from private field"),t?t.call(h):e.get(h)),b=(h,e,t)=>e.has(h)?cw("Cannot add the same private member more than once"):e instanceof WeakSet?e.add(h):e.set(h,t),g=(h,e,t,n)=>(By(h,e,"write to private field"),n?n.call(h,t):e.set(h,t),t),y=(h,e,t)=>(By(h,e,"access private method"),t);var Zt=(h,e,t,n)=>({set _(i){g(h,e,i,t)},get _(){return a(h,e,n)}});import{g as vy,c as LA,r as gi,M as l1,j as fe,f as ku,I as Bu,a as c1,i as hw,P as h1,b as dw,S as d1,d as u1,C as f1,e as p1,h as g1}from"./index-pYxvwZLG.js";const m1=()=>"Find in document",v1=()=>"在文档中查找",y1=()=>"جست‌وجو در سند",b1=()=>"البحث في المستند",E1=()=>"Buscar en el documento",w1=()=>"दस्तावेज़ में खोजें",uw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?v1():t==="fa"?y1():t==="ar"?b1():t==="es"?E1():t==="hi"?w1():m1()}),A1=()=>"Fit to width",x1=()=>"适应宽度",S1=()=>"متناسب با عرض",T1=()=>"ملاءمة العرض",C1=()=>"Ajustar al ancho",P1=()=>"चौड़ाई में फ़िट करें",fw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?x1():t==="fa"?S1():t==="ar"?T1():t==="es"?C1():t==="hi"?P1():A1()}),I1=()=>"Next match",R1=()=>"下一个匹配项",O1=()=>"مورد بعدی",M1=()=>"التطابق التالي",D1=()=>"Coincidencia siguiente",L1=()=>"अगला मिलान",pw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?R1():t==="fa"?O1():t==="ar"?M1():t==="es"?D1():t==="hi"?L1():I1()}),N1=()=>"Previous match",F1=()=>"上一个匹配项",k1=()=>"مورد قبلی",B1=()=>"التطابق السابق",j1=()=>"Coincidencia anterior",U1=()=>"पिछला मिलान",gw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?F1():t==="fa"?k1():t==="ar"?B1():t==="es"?j1():t==="hi"?U1():N1()});/**
var a1=Object.defineProperty;var cw=h=>{throw TypeError(h)};var o1=(h,e,t)=>e in h?a1(h,e,{enumerable:!0,configurable:!0,writable:!0,value:t}):h[e]=t;var L=(h,e,t)=>o1(h,typeof e!="symbol"?e+"":e,t),By=(h,e,t)=>e.has(h)||cw("Cannot "+t);var a=(h,e,t)=>(By(h,e,"read from private field"),t?t.call(h):e.get(h)),b=(h,e,t)=>e.has(h)?cw("Cannot add the same private member more than once"):e instanceof WeakSet?e.add(h):e.set(h,t),g=(h,e,t,n)=>(By(h,e,"write to private field"),n?n.call(h,t):e.set(h,t),t),y=(h,e,t)=>(By(h,e,"access private method"),t);var Zt=(h,e,t,n)=>({set _(i){g(h,e,i,t)},get _(){return a(h,e,n)}});import{g as vy,c as LA,r as gi,M as l1,j as fe,f as ku,I as Bu,a as c1,i as hw,P as h1,b as dw,S as d1,d as u1,C as f1,e as p1,h as g1}from"./index-CHipQ6B8.js";const m1=()=>"Find in document",v1=()=>"在文档中查找",y1=()=>"جست‌وجو در سند",b1=()=>"البحث في المستند",E1=()=>"Buscar en el documento",w1=()=>"दस्तावेज़ में खोजें",uw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?v1():t==="fa"?y1():t==="ar"?b1():t==="es"?E1():t==="hi"?w1():m1()}),A1=()=>"Fit to width",x1=()=>"适应宽度",S1=()=>"متناسب با عرض",T1=()=>"ملاءمة العرض",C1=()=>"Ajustar al ancho",P1=()=>"चौड़ाई में फ़िट करें",fw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?x1():t==="fa"?S1():t==="ar"?T1():t==="es"?C1():t==="hi"?P1():A1()}),I1=()=>"Next match",R1=()=>"下一个匹配项",O1=()=>"مورد بعدی",M1=()=>"التطابق التالي",D1=()=>"Coincidencia siguiente",L1=()=>"अगला मिलान",pw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?R1():t==="fa"?O1():t==="ar"?M1():t==="es"?D1():t==="hi"?L1():I1()}),N1=()=>"Previous match",F1=()=>"上一个匹配项",k1=()=>"مورد قبلی",B1=()=>"التطابق السابق",j1=()=>"Coincidencia anterior",U1=()=>"पिछला मिलान",gw=((h={},e={})=>{const t=e.locale??vy();return t==="zh-CN"?F1():t==="fa"?k1():t==="ar"?B1():t==="es"?j1():t==="hi"?U1():N1()});/**
* @license lucide-react v1.23.0 - ISC
*
* This source code is licensed under the ISC license.
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+2 -2
View File
@@ -49,8 +49,8 @@
html { background: #ffffff; }
html[data-theme="dark"] { background: #0e0c0c; }
</style>
<script type="module" crossorigin src="/assets/index-pYxvwZLG.js"></script>
<link rel="stylesheet" crossorigin href="/assets/index-c6tqUdfa.css">
<script type="module" crossorigin src="/assets/index-CHipQ6B8.js"></script>
<link rel="stylesheet" crossorigin href="/assets/index-CtHX_xm8.css">
</head>
<body>
<div
+13
View File
@@ -113,6 +113,19 @@ export function runDisplayStatus(run: Pick<Run, "status" | "cancelRequested">):
return live && run.cancelRequested ? "cancelling" : run.status;
}
/** Why the supervisor can't currently observe a live run, if it can't. */
export function runMonitoringError(run: Pick<Run, "status" | "backend">): string | null {
if (run.status !== "running" && run.status !== "starting") return null;
const error = run.backend?.monitoringError;
return typeof error === "string" && error ? error : null;
}
/** The newest live run's monitoring error: a forced relaunch can leave an older run live. */
export function experimentMonitoringError(runs: Pick<Run, "status" | "backend" | "createdAt">[]): string | null {
const newestFirst = [...runs].sort((a, b) => b.createdAt - a.createdAt);
return newestFirst.map(runMonitoringError).find((error) => error !== null) ?? null;
}
const writeScopes = new WeakMap<Response, ReturnType<typeof workspaceScope>>();
async function json<T>(res: Response): Promise<T> {
+8 -1
View File
@@ -17,6 +17,7 @@ import { FolderTree, GitBranch, Terminal } from "lucide-react";
import { parseDiff, type FileData } from "react-diff-view";
import {
backendKind,
experimentMonitoringError,
fmtDuration,
fmtNumber,
runDisplayStatus,
@@ -205,6 +206,7 @@ export function ExpHoverCard({
const failureNote =
latestRun?.status === "failed" && latestRun.resultMarkdown ? latestRun.resultMarkdown : null;
const body = exp.description || (failureNote ? null : latestRun?.resultMarkdown) || null;
const monitoringError = experimentMonitoringError(runs);
// Clamped by default; "Show more" appears when the clamp actually hides
// content and stays while expanded so "Show less" remains reachable.
@@ -223,7 +225,7 @@ export function ExpHoverCard({
return createPortal(
<div
ref={measure.ref}
className="exp-hover-card fixed z-60 bg-background border border-border rounded-lg shadow-menu py-3.5 px-4 text-sm text-text [&_.hc-head]:flex [&_.hc-head]:items-baseline [&_.hc-head]:justify-between [&_.hc-head]:gap-2.5 [&_.hc-slug]:text-sm [&_.hc-slug]:font-semibold [&_.hc-slug]:min-w-0 [&_.hc-slug]:overflow-hidden [&_.hc-slug]:text-ellipsis [&_.hc-slug]:whitespace-nowrap [&_.hc-title]:mt-[3px] [&_.hc-title]:text-text [&_.hc-actions]:flex [&_.hc-actions]:items-center [&_.hc-actions]:gap-1.5 [&_.hc-actions]:mt-2.5 [&_.hc-actions_button]:inline-flex [&_.hc-actions_button]:items-center [&_.hc-actions_button]:justify-center [&_.hc-actions_button]:gap-[5px] [&_.hc-actions_button]:min-w-21 [&_.hc-actions_button]:py-1.5 [&_.hc-actions_button]:px-2.5 [&_.hc-actions_button]:border [&_.hc-actions_button]:border-border [&_.hc-actions_button]:rounded-md [&_.hc-actions_button]:bg-background [&_.hc-actions_button]:text-text [&_.hc-actions_button]:text-sm [&_.hc-actions_button]:font-medium [&_.hc-actions_button:hover]:border-border-hover-strong [&_.hc-actions_button:hover]:bg-canvas [&_.hc-body]:mt-2.5 [&_.hc-body]:border-t [&_.hc-body]:border-t-border-variant [&_.hc-body]:pt-2.5 [&_.hc-body]:leading-[1.6] [&_.hc-body]:whitespace-pre-line [&_.hc-body]:line-clamp-10 [&_.hc-body.expanded]:block [&_.hc-body.expanded]:line-clamp-none [&_.hc-body.expanded]:max-h-[45vh] [&_.hc-body.expanded]:overflow-y-auto [&_.hc-body.expanded]:overflow-x-hidden [&_.hc-body.expanded]:pb-1 [&_.hc-toggle]:mt-1 [&_.hc-toggle]:text-sm [&_.hc-toggle]:font-medium [&_.hc-toggle]:text-muted [&_.hc-toggle:hover]:text-text [&_.hc-failure]:mt-2 [&_.hc-failure]:text-accent-red [&_.hc-failure]:line-clamp-3 [&_.hc-stats]:mt-2.5 [&_.hc-stats]:border-t [&_.hc-stats]:border-t-border-variant [&_.hc-stats]:pt-2.5 [&_.hc-stats]:flex [&_.hc-stats]:items-center [&_.hc-stats]:gap-3 [&_.hc-stats]:flex-wrap [&_.hc-stats]:text-xs [&_.hc-stats]:text-text [&_.hc-git]:mt-2.5 [&_.hc-git]:pt-2 [&_.hc-git]:border-t [&_.hc-git]:border-t-border-variant [&_.hc-git]:text-xs [&_.hc-git]:text-text [&_.hc-git]:flex [&_.hc-git]:flex-col [&_.hc-git]:gap-1 [&_.hc-git-row]:flex [&_.hc-git-row]:items-center [&_.hc-git-row]:gap-2.5 [&_.hc-git-row]:flex-wrap [&_.hc-git-row]:min-w-0 [&_.hc-branch]:inline-flex [&_.hc-branch]:items-center [&_.hc-branch]:gap-1 [&_.hc-branch]:min-w-0 [&_.hc-branch]:overflow-hidden [&_.hc-branch]:text-ellipsis [&_.hc-branch]:whitespace-nowrap [&_.hc-foot]:mt-2 [&_.hc-foot]:flex [&_.hc-foot]:items-center [&_.hc-foot]:justify-between [&_.hc-foot]:gap-2.5 [&_.hc-foot]:text-xs [&_.hc-foot]:text-muted [&_.hc-foot_.hc-command]:min-w-0 [&_.hc-foot_.hc-command]:overflow-hidden [&_.hc-foot_.hc-command]:text-ellipsis [&_.hc-foot_.hc-command]:whitespace-nowrap"
className="exp-hover-card fixed z-60 bg-background border border-border rounded-lg shadow-menu py-3.5 px-4 text-sm text-text [&_.hc-head]:flex [&_.hc-head]:items-baseline [&_.hc-head]:justify-between [&_.hc-head]:gap-2.5 [&_.hc-slug]:text-sm [&_.hc-slug]:font-semibold [&_.hc-slug]:min-w-0 [&_.hc-slug]:overflow-hidden [&_.hc-slug]:text-ellipsis [&_.hc-slug]:whitespace-nowrap [&_.hc-title]:mt-[3px] [&_.hc-title]:text-text [&_.hc-actions]:flex [&_.hc-actions]:items-center [&_.hc-actions]:gap-1.5 [&_.hc-actions]:mt-2.5 [&_.hc-actions_button]:inline-flex [&_.hc-actions_button]:items-center [&_.hc-actions_button]:justify-center [&_.hc-actions_button]:gap-[5px] [&_.hc-actions_button]:min-w-21 [&_.hc-actions_button]:py-1.5 [&_.hc-actions_button]:px-2.5 [&_.hc-actions_button]:border [&_.hc-actions_button]:border-border [&_.hc-actions_button]:rounded-md [&_.hc-actions_button]:bg-background [&_.hc-actions_button]:text-text [&_.hc-actions_button]:text-sm [&_.hc-actions_button]:font-medium [&_.hc-actions_button:hover]:border-border-hover-strong [&_.hc-actions_button:hover]:bg-canvas [&_.hc-body]:mt-2.5 [&_.hc-body]:border-t [&_.hc-body]:border-t-border-variant [&_.hc-body]:pt-2.5 [&_.hc-body]:leading-[1.6] [&_.hc-body]:whitespace-pre-line [&_.hc-body]:line-clamp-10 [&_.hc-body.expanded]:block [&_.hc-body.expanded]:line-clamp-none [&_.hc-body.expanded]:max-h-[45vh] [&_.hc-body.expanded]:overflow-y-auto [&_.hc-body.expanded]:overflow-x-hidden [&_.hc-body.expanded]:pb-1 [&_.hc-toggle]:mt-1 [&_.hc-toggle]:text-sm [&_.hc-toggle]:font-medium [&_.hc-toggle]:text-muted [&_.hc-toggle:hover]:text-text [&_.hc-failure]:mt-2 [&_.hc-failure]:text-accent-red [&_.hc-failure]:line-clamp-3 [&_.hc-monitoring]:mt-2 [&_.hc-monitoring]:text-accent-amber [&_.hc-monitoring]:line-clamp-3 [&_.hc-stats]:mt-2.5 [&_.hc-stats]:border-t [&_.hc-stats]:border-t-border-variant [&_.hc-stats]:pt-2.5 [&_.hc-stats]:flex [&_.hc-stats]:items-center [&_.hc-stats]:gap-3 [&_.hc-stats]:flex-wrap [&_.hc-stats]:text-xs [&_.hc-stats]:text-text [&_.hc-git]:mt-2.5 [&_.hc-git]:pt-2 [&_.hc-git]:border-t [&_.hc-git]:border-t-border-variant [&_.hc-git]:text-xs [&_.hc-git]:text-text [&_.hc-git]:flex [&_.hc-git]:flex-col [&_.hc-git]:gap-1 [&_.hc-git-row]:flex [&_.hc-git-row]:items-center [&_.hc-git-row]:gap-2.5 [&_.hc-git-row]:flex-wrap [&_.hc-git-row]:min-w-0 [&_.hc-branch]:inline-flex [&_.hc-branch]:items-center [&_.hc-branch]:gap-1 [&_.hc-branch]:min-w-0 [&_.hc-branch]:overflow-hidden [&_.hc-branch]:text-ellipsis [&_.hc-branch]:whitespace-nowrap [&_.hc-foot]:mt-2 [&_.hc-foot]:flex [&_.hc-foot]:items-center [&_.hc-foot]:justify-between [&_.hc-foot]:gap-2.5 [&_.hc-foot]:text-xs [&_.hc-foot]:text-muted [&_.hc-foot_.hc-command]:min-w-0 [&_.hc-foot_.hc-command]:overflow-hidden [&_.hc-foot_.hc-command]:text-ellipsis [&_.hc-foot_.hc-command]:whitespace-nowrap"
style={{
width: CARD_W,
left: x,
@@ -268,6 +270,11 @@ export function ExpHoverCard({
</button>
)}
{failureNote && <div className="hc-failure">{failureNote}</div>}
{monitoringError && (
<div className="hc-monitoring" title={monitoringError}>
{monitoringError}
</div>
)}
<div className="hc-stats">
<span>
{new Intl.ListFormat(getLocale(), { style: "short" }).format([
+7
View File
@@ -9,6 +9,7 @@ import {
} from "lucide-react";
import { useEffect, useState } from "react";
import {
experimentMonitoringError,
fmtDuration,
runDisplayStatus,
timeAgo,
@@ -64,6 +65,7 @@ export function ExperimentOverview({
onOpenCode: (intent: TabOpenIntent) => void;
}) {
const latestRun = runs[0] ?? null;
const monitoringError = experimentMonitoringError(runs);
const hasLiveRun = runs.some(
(run) => run.status === "running" || run.status === "starting",
);
@@ -143,6 +145,11 @@ export function ExperimentOverview({
{latestRun.command && (
<code className={EXPERIMENT_OVERVIEW_COMMAND_CLASS_NAME}>$ {latestRun.command}</code>
)}
{monitoringError && (
<p className="experiment-overview-monitoring mt-4 text-accent-amber text-sm wrap-anywhere">
{monitoringError}
</p>
)}
{latestRun.resultMarkdown && (
<div
className={`experiment-overview-result mt-4 [&.failed]:text-accent-red ${latestRun.status === "failed" ? "failed" : ""}`}