Synced from monorepo
Synced from monorepo Changes: - Shell: accept target response id on rewind execute - Shell: stamp response id on chat user message chunks - Worktree: optional rebuild and stale git registration cleanup in auto-GC - Worktree: kind-aware auto-GC TTLs and config knobs - Worktree: macOS process CWD scan and Unix PID liveness for GC guards - Worktree: automatic throttled GC on startup (Linux age-based; non-Linux dead-only) - Pager: add `[ui].combine_queued_prompts` to batch queued follow-ups - Shell: stop overwriting user skills - Tools: read markdown in `skills/` directories untruncated - `/usage` shows per-session token and dollar usage in the TUI - Security: prompt on environment-dumping `ps` variants - Security: always-safe `kubectl` no longer runs arbitrary kubeconfig credential plugins without permission - Tools: make scheduler deletion durable - Shell: add relocation storage primitives - Shell: give side model calls their own conversation ids - Fix five workflow-runtime bugs (budget, pause, cancel, reconnect) - Security: peel `env -S` / `--split-string` operands in the Bash permission gate (managed deny/ask) - Pager: expose doctor in the TUI - Security: block unauthorized RCE via abused safe commands - Pager idle watcher cue: "1 subagent still running" instead of "watching · 1 subagent" - Security: block `rg --pre` arbitrary code execution in auto-mode - Voice: diagnose silent-mic failures (macOS permission) and add doctor/terminal-setup Voice section - App builder deployer: `allow_forking` and `show_built_with_grok` - Pager: stop stacking duplicate "Worked for" markers on parked turns - Shell: support `max` as a distinct reasoning effort tier - Tools: serialize background `/loop` fires on the whole work unit - Shell: add working-directory relocation state primitives - Proto: `ClientToolResult` and `ChatConfig` client-side tools - Shell: model providers - Chat: select App Builder product on the Build path - Shell: attach author identity to feedback when the deployment opts in - Doctor: fix for SSH wrap setup - Workflow authoring skills: create-workflow and import-claude-workflow docs - Add read-only grok doctor - Sandbox: apply Landlock without a controlling TTY - Pager: recover image paste over grok wrap on headless remotes - Pager: make actions screen-mode aware - Shell: resume sessions when the working directory moves - Pager: centralize terminal diagnostics - Workspace: gate inline shell file access - Pager: centralize terminal probes - Pager: edit minimal prompts in an external editor - Pager: standardize backgrounding on Ctrl+B - Shell: recap rides the parent turn's prompt cache - Tools: add scheduler lifecycle version clock Source-Revision: 0f4d7c91b8b2b408333f6de1e8a76cb8eaa71899
This commit is contained in:
parent
a881e6703f
commit
3af4d5d398
556 changed files with 56609 additions and 21892 deletions
|
|
@ -7,12 +7,14 @@
|
|||
|
||||
use std::fs;
|
||||
use std::path::{Component, Path, PathBuf};
|
||||
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
|
||||
|
||||
use anyhow::{Context, Result, anyhow, bail};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::{ContentController, PtyHarness, StyledLine, pager_binary, parse_keys};
|
||||
use crate::{
|
||||
AgentTurnExpectation, ContentController, PtyHarness, StyledLine, pager_binary, parse_keys,
|
||||
};
|
||||
|
||||
const SGR_LEFT_BUTTON: u16 = 0;
|
||||
const SGR_MIDDLE_BUTTON: u16 = 1;
|
||||
|
|
@ -26,6 +28,7 @@ pub const SGR_SCROLL_DOWN: u16 = 65;
|
|||
const DEFAULT_ROWS: u16 = 50;
|
||||
const DEFAULT_COLS: u16 = 120;
|
||||
const DEFAULT_WAIT_TIMEOUT_MS: u64 = 15_000;
|
||||
const EXPECTATION_SETTLE_TIMEOUT: Duration = Duration::from_secs(10);
|
||||
|
||||
/// Declarative scenario consumed by [`ScriptedScenarioRunner`].
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
|
|
@ -172,11 +175,12 @@ pub struct WorkspaceConfig {
|
|||
pub struct MockConfig {
|
||||
#[serde(default = "default_mock_response")]
|
||||
pub response: String,
|
||||
/// Optional per-agent-turn responses, consumed FIFO (one per real agent
|
||||
/// turn; aux requests don't consume one — see
|
||||
/// `MockInferenceServer::set_agent_turns`). Lets a scenario give each
|
||||
/// turn a distinct sentinel, e.g. to prove a transcript tail was
|
||||
/// truncated and re-generated. Falls back to `response` when exhausted.
|
||||
/// Required per-agent-turn responses, registered as ordered foreground
|
||||
/// expectations on both supported pager inference backends. Every listed
|
||||
/// turn must be satisfied before the runner reports success. Lets a
|
||||
/// scenario give each turn a distinct sentinel, e.g. to prove a transcript
|
||||
/// tail was truncated and re-generated. Falls back to `response` when
|
||||
/// exhausted.
|
||||
#[serde(default)]
|
||||
pub turns: Vec<String>,
|
||||
#[serde(default)]
|
||||
|
|
@ -531,9 +535,15 @@ impl ScriptedScenarioRunner {
|
|||
.await
|
||||
.context("start mock content")?;
|
||||
content.set_response(&scenario.mock.response);
|
||||
if !scenario.mock.turns.is_empty() {
|
||||
content.set_turns(scenario.mock.turns.iter().cloned());
|
||||
}
|
||||
let turn_expectations: Vec<_> = scenario
|
||||
.mock
|
||||
.turns
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, turn)| {
|
||||
content.expect_agent_turn(format!("scenario turn {}", index + 1), turn)
|
||||
})
|
||||
.collect();
|
||||
|
||||
if let Some(config_toml) = &scenario.environment.config_toml {
|
||||
let grok_home = content.home().join(".grok");
|
||||
|
|
@ -636,6 +646,34 @@ impl ScriptedScenarioRunner {
|
|||
report.status = ScriptedRunStatus::Failed;
|
||||
}
|
||||
|
||||
if report.status == ScriptedRunStatus::Running {
|
||||
let settle_deadline = Instant::now() + EXPECTATION_SETTLE_TIMEOUT;
|
||||
while turn_expectations
|
||||
.iter()
|
||||
.any(|expectation| !expectation.is_satisfied())
|
||||
&& Instant::now() < settle_deadline
|
||||
{
|
||||
harness.update(Duration::from_millis(100));
|
||||
}
|
||||
}
|
||||
let unsatisfied_turns: Vec<_> = turn_expectations
|
||||
.iter()
|
||||
.filter(|expectation| !expectation.is_satisfied())
|
||||
.map(AgentTurnExpectation::diagnostic)
|
||||
.collect();
|
||||
if !unsatisfied_turns.is_empty() {
|
||||
report.bugs.push(BugFinding {
|
||||
step: scenario.steps.len(),
|
||||
severity: BugSeverity::Bug,
|
||||
message: format!(
|
||||
"required mock.turns expectations were not satisfied:\n- {}",
|
||||
unsatisfied_turns.join("\n- ")
|
||||
),
|
||||
screen_text: harness.screen_contents(),
|
||||
});
|
||||
report.status = ScriptedRunStatus::Failed;
|
||||
}
|
||||
|
||||
let _ = harness.quit();
|
||||
if report.status == ScriptedRunStatus::Running {
|
||||
report.status = ScriptedRunStatus::Passed;
|
||||
|
|
|
|||
Loading…
Reference in a new issue