grok-build-upstream-mirror/crates/codegen/xai-grok-test-support/src/headless.rs
grokkybara[bot] a5727c5960 Synced from monorepo
Changes:
- Non-blocking coding-data sharing upsell banner
- Consolidate remediation in Doctor
- Auto mode defers fail-closed gate asks to the classifier
- Coalesce marketplace list fetches
- Allow removing a marketplace source by name
- Contain hung git marketplace sources (timeouts, non-blocking refresh, unbrick modal)
- Label failed workspace RPCs with error_kind
- Drop redundant explicit tonic/prost deps from xai-grok-shell
- Report real exit codes for completed background shells
- Narrow the date-rollover reminder to date-bearing templates
- Wire toolOverrides through the session and agent
- Security: Bash(git:*) allowlist matches whole command chain by prefix
- Split prompt-trigger telemetry and record classifier provenance
- Raise connectors-manager timeout to 60s
- Auto classifier honors recorded approvals for repeat actions
- Apply doctor fixes in the TUI
- Auto-mode classifier timeouts prompt instead of silently denying
- Scope subagent completion drains to the owning session
- Add the toolOverrides wire types
- Set client_identifier=grok-agent-sdk
- Accept both spellings of the workspace-teleport kill switch
- Persist one-shot occurrence journal
- Stop turns that poll the exact same tool call 16x in a row
- Copy compaction checkpoint files when forking sessions
- Auto-focus permission prompt from scrollback
- Esc cancels the running turn in non-vim and minimal modes
- List Ctrl+Z undo and redo in keyboard shortcuts
- Out-of-process macOS mic capture
- Show active auth mode on session-info
- Install the npm binary under $GROK_HOME
- Remove hover/click dead zones between dashboard items
- Route startup warnings to doctor
- Document [feedback.user] author identity config
- Extend bang command timeout
- Close combine-queued edit-hold race
- Integrate relocation recovery
- Expose privacy notice rollout flag
- Break harness discovery ref cycle so connections can idle-evict
- Shift/Alt+Enter inserts newline when editing a queued prompt
- Gate project Claude permissions on folder trust
- Echo response.create.event_id on response.created
- Toast when session creation fails from disk full
- Add shared test process lifecycle
- Enable dynamic workflows by default
- Add relocation transaction state machine
- Add shared test sandbox
- Surface auth failures on model-switch compact
- Persist durable scheduler expiry
- Confirm before removing extensions-modal items
- Re-run compact and prompt after login when compact hit expired auth
- Recap sends hosted tools under backend search
2026-07-22 19:22:27 +01:00

361 lines
11 KiB
Rust

//! Headless mode (`grok -p`) test runner.
//!
//! Runs the grok binary as a subprocess with the mock server, captures output.
use std::path::{Path, PathBuf};
use std::process::ExitStatus;
use std::time::Duration;
use tokio::io::AsyncReadExt as _;
use crate::env::grok_binary;
use crate::mock_server::MockInferenceServer;
use crate::process::{TestOutput, TestProcess, TestProcessConfig};
use crate::sandbox::TestSandbox;
pub struct HeadlessResult {
pub status: ExitStatus,
pub stdout: String,
pub stderr: String,
pub timed_out: bool,
/// Wall time of the headless command invocation; logged so CI timeout
/// budgets can be tuned against observed durations.
pub elapsed: Duration,
}
/// Timeout for one headless grok invocation: 60 seconds, multiplied by
/// [`crate::scaled`]'s `GROK_TEST_TIMEOUT_SCALE`.
fn headless_timeout() -> Duration {
crate::scaled(Duration::from_secs(60))
}
const HEADLESS_DRAIN_TIMEOUT: Duration = Duration::from_secs(2);
/// Run `grok` with the given args against the mock server, bounded by the
/// scaled headless timeout. Uses an isolated HOME and disables telemetry.
pub async fn run_headless(
server: &MockInferenceServer,
args: &[&str],
cwd: &Path,
) -> HeadlessResult {
run_headless_with_env(server, args, cwd, &[]).await
}
/// Like [`run_headless`], but with extra environment variables applied after the
/// sandbox baseline so they take precedence — e.g. to re-enable a feature the
/// baseline turns off.
pub async fn run_headless_with_env(
server: &MockInferenceServer,
args: &[&str],
cwd: &Path,
env: &[(&str, &str)],
) -> HeadlessResult {
let sandbox = TestSandbox::builder().mock_url(server.url()).build();
let mut cmd = tokio::process::Command::new(grok_binary());
cmd.args(args).current_dir(cwd);
run_headless_with_cmd_and_sandbox(cmd, &sandbox, env).await
}
/// Apply and retain one [`TestSandbox`] while running a custom headless command.
pub async fn run_headless_in_sandbox(
cmd: tokio::process::Command,
sandbox: TestSandbox,
) -> HeadlessResult {
run_headless_in_sandbox_with_env(cmd, sandbox, &[]).await
}
pub async fn run_headless_in_sandbox_with_env(
cmd: tokio::process::Command,
sandbox: TestSandbox,
overrides: &[(&str, &str)],
) -> HeadlessResult {
run_headless_in_sandbox_borrowed_with_env(cmd, &sandbox, overrides).await
}
/// Run a custom headless command while leaving the caller's sandbox available
/// for post-run artifact inspection.
pub async fn run_headless_in_sandbox_borrowed(
cmd: tokio::process::Command,
sandbox: &TestSandbox,
) -> HeadlessResult {
run_headless_in_sandbox_borrowed_with_env(cmd, sandbox, &[]).await
}
pub async fn run_headless_in_sandbox_borrowed_with_env(
cmd: tokio::process::Command,
sandbox: &TestSandbox,
overrides: &[(&str, &str)],
) -> HeadlessResult {
run_headless_with_cmd_and_sandbox(cmd, sandbox, overrides).await
}
async fn run_headless_with_cmd_and_sandbox(
cmd: tokio::process::Command,
sandbox: &TestSandbox,
overrides: &[(&str, &str)],
) -> HeadlessResult {
let program = PathBuf::from(cmd.as_std().get_program());
let started = std::time::Instant::now();
let mut process = TestProcess::spawn(
cmd,
sandbox,
TestProcessConfig::new()
.label(format!("headless command {}", program.display()))
.stdout(TestOutput::Piped)
.stderr(TestOutput::Piped)
.envs(overrides.iter().copied()),
)
.unwrap_or_else(|error| {
panic!(
"failed to spawn headless command at {}: {error}\n{}",
program.display(),
sandbox.diagnostic_summary(),
)
});
let mut stdout = process.take_stdout().expect("child stdout missing");
let stdout_handle = tokio::spawn(async move {
let mut bytes = Vec::new();
stdout.read_to_end(&mut bytes).await?;
Ok::<Vec<u8>, std::io::Error>(bytes)
});
let mut stderr = process.take_stderr().expect("child stderr missing");
let stderr_handle = tokio::spawn(async move {
let mut bytes = Vec::new();
stderr.read_to_end(&mut bytes).await?;
Ok::<Vec<u8>, std::io::Error>(bytes)
});
let (status, timed_out) = match process
.wait_with_deadline(headless_timeout())
.await
.unwrap_or_else(|error| {
panic!(
"failed to wait for headless command {}: {error}\n{}",
program.display(),
process.diagnostic_summary(),
)
}) {
Some(status) => (status, false),
None => {
let status = process.kill().await.unwrap_or_else(|error| {
panic!(
"failed to kill timed out headless command {}: {error}\n{}",
program.display(),
process.diagnostic_summary(),
)
});
(status, true)
}
};
let stdout = finish_output_drain(
stdout_handle,
process.stdout_tail().text,
"stdout",
&program,
&process,
)
.await;
let stderr = finish_output_drain(
stderr_handle,
process.stderr_tail().text,
"stderr",
&program,
&process,
)
.await;
let elapsed = started.elapsed();
// Timing breadcrumb for tuning CI timeout budgets against observed
// durations (visible with --nocapture).
eprintln!(
"[harness-timing] headless command {}: {elapsed:?} (timed_out={timed_out})",
program.display()
);
HeadlessResult {
status,
stdout,
stderr,
timed_out,
elapsed,
}
}
async fn finish_output_drain(
mut handle: tokio::task::JoinHandle<std::io::Result<Vec<u8>>>,
partial_tail: String,
stream: &str,
program: &Path,
process: &TestProcess,
) -> String {
match tokio::time::timeout(HEADLESS_DRAIN_TIMEOUT, &mut handle).await {
Ok(Ok(Ok(bytes))) => String::from_utf8_lossy(&bytes).into_owned(),
Ok(Ok(Err(error))) => {
tracing::warn!(
%stream,
program = %program.display(),
%error,
"headless output drain failed; returning captured partial tail"
);
partial_tail
}
Ok(Err(error)) => {
tracing::warn!(
%stream,
program = %program.display(),
%error,
"headless output task failed; returning captured partial tail"
);
partial_tail
}
Err(_) => {
handle.abort();
let _ = handle.await;
tracing::warn!(
%stream,
program = %program.display(),
diagnostics = %process.diagnostic_summary(),
"headless output drain timed out; returning captured partial tail"
);
partial_tail
}
}
}
const CRASH_PATTERNS: &[&str] = &[
"panicked at",
"SIGSEGV",
"segfault",
"undefined symbol",
"SIGABRT",
"cannot open shared object",
];
/// Diagnostic helper: format the tail of stderr for assertion messages.
pub fn stderr_tail(stderr: &str, max_chars: usize) -> &str {
&stderr[stderr.len().saturating_sub(max_chars)..]
}
/// Assert that a headless run succeeded (non-timeout, zero exit code).
pub fn assert_headless_success(
result: &HeadlessResult,
label: &str,
server: Option<&MockInferenceServer>,
) {
assert!(
!result.timed_out,
"{label}: timed out after {:?}\nstderr tail:\n{}",
headless_timeout(),
stderr_tail(&result.stderr, 500)
);
assert!(
result.status.success(),
"{label}: exited with {:?}\nstderr tail:\n{}\n{}",
result.status.code(),
stderr_tail(&result.stderr, 1000),
server
.map(|s| format!("request log:\n{}", s.request_log_summary()))
.unwrap_or_default()
);
}
/// Panic if stderr contains any crash/linking-failure indicators.
pub fn assert_no_crashes(stderr: &str) {
let lower = stderr.to_lowercase();
for pattern in CRASH_PATTERNS {
assert!(
!lower.contains(&pattern.to_lowercase()),
"stderr contains crash indicator '{pattern}':\n{}",
stderr_tail(stderr, 500)
);
}
}
#[cfg(test)]
mod tests {
use super::*;
#[cfg(unix)]
#[tokio::test]
async fn borrowed_runner_keeps_sandbox_artifacts_available() {
let sandbox = TestSandbox::new();
let artifact = sandbox.temp_dir().join("borrowed-headless.txt");
let script = format!("printf kept > '{}'", artifact.display());
let mut cmd = tokio::process::Command::new("/bin/sh");
cmd.args(["-c", &script]);
let result = run_headless_in_sandbox_borrowed(cmd, &sandbox).await;
assert!(result.status.success(), "stderr: {}", result.stderr);
assert_eq!(std::fs::read_to_string(artifact).unwrap(), "kept");
}
#[cfg(unix)]
#[tokio::test]
async fn output_drain_timeout_returns_partial_capture() {
let (tx, rx) = tokio::sync::oneshot::channel::<()>();
let handle = tokio::spawn(async move {
let _keep_open = tx;
let _ = rx.await;
Ok::<Vec<u8>, std::io::Error>(b"complete".to_vec())
});
let sandbox = TestSandbox::new();
let mut command = tokio::process::Command::new("/bin/sh");
command.args(["-c", "exit 0"]);
let mut process = TestProcess::spawn(
command,
&sandbox,
TestProcessConfig::new().label("headless-drain-test"),
)
.expect("spawn drain fixture");
process
.wait_with_deadline(Duration::from_secs(2))
.await
.expect("wait fixture")
.expect("fixture exits");
let output = finish_output_drain(
handle,
"partial".to_owned(),
"stdout",
Path::new("fixture"),
&process,
)
.await;
assert_eq!(output, "partial");
}
#[cfg(unix)]
#[tokio::test]
async fn custom_headless_env_is_explicit_and_wins_after_sandbox_baseline() {
let sandbox = TestSandbox::new();
let mut cmd = tokio::process::Command::new("/bin/sh");
cmd.args([
"-c",
"printf '%s|%s|%s|%s' \"${AMBIENT_ONLY-unset}\" \"$GROK_PROMPT_SUGGESTIONS\" \"$FEATURE_TEST_VAR\" \"$HOME\"",
])
.env("AMBIENT_ONLY", "discarded")
.env("GROK_PROMPT_SUGGESTIONS", "command-level-discarded");
let result = run_headless_in_sandbox_borrowed_with_env(
cmd,
&sandbox,
&[
("GROK_PROMPT_SUGGESTIONS", "explicit-override"),
("FEATURE_TEST_VAR", "enabled"),
],
)
.await;
assert!(result.status.success(), "stderr: {}", result.stderr);
assert_eq!(
result.stdout,
format!(
"unset|explicit-override|enabled|{}",
sandbox.home().display()
)
);
}
}