Synced from monorepo

Synced from monorepo

Changes:
- Shell: accept target response id on rewind execute
- Shell: stamp response id on chat user message chunks
- Worktree: optional rebuild and stale git registration cleanup in auto-GC
- Worktree: kind-aware auto-GC TTLs and config knobs
- Worktree: macOS process CWD scan and Unix PID liveness for GC guards
- Worktree: automatic throttled GC on startup (Linux age-based; non-Linux dead-only)
- Pager: add `[ui].combine_queued_prompts` to batch queued follow-ups
- Shell: stop overwriting user skills
- Tools: read markdown in `skills/` directories untruncated
- `/usage` shows per-session token and dollar usage in the TUI
- Security: prompt on environment-dumping `ps` variants
- Security: always-safe `kubectl` no longer runs arbitrary kubeconfig credential plugins without permission
- Tools: make scheduler deletion durable
- Shell: add relocation storage primitives
- Shell: give side model calls their own conversation ids
- Fix five workflow-runtime bugs (budget, pause, cancel, reconnect)
- Security: peel `env -S` / `--split-string` operands in the Bash permission gate (managed deny/ask)
- Pager: expose doctor in the TUI
- Security: block unauthorized RCE via abused safe commands
- Pager idle watcher cue: "1 subagent still running" instead of "watching · 1 subagent"
- Security: block `rg --pre` arbitrary code execution in auto-mode
- Voice: diagnose silent-mic failures (macOS permission) and add doctor/terminal-setup Voice section
- App builder deployer: `allow_forking` and `show_built_with_grok`
- Pager: stop stacking duplicate "Worked for" markers on parked turns
- Shell: support `max` as a distinct reasoning effort tier
- Tools: serialize background `/loop` fires on the whole work unit
- Shell: add working-directory relocation state primitives
- Proto: `ClientToolResult` and `ChatConfig` client-side tools
- Shell: model providers
- Chat: select App Builder product on the Build path
- Shell: attach author identity to feedback when the deployment opts in
- Doctor: fix for SSH wrap setup
- Workflow authoring skills: create-workflow and import-claude-workflow docs
- Add read-only grok doctor
- Sandbox: apply Landlock without a controlling TTY
- Pager: recover image paste over grok wrap on headless remotes
- Pager: make actions screen-mode aware
- Shell: resume sessions when the working directory moves
- Pager: centralize terminal diagnostics
- Workspace: gate inline shell file access
- Pager: centralize terminal probes
- Pager: edit minimal prompts in an external editor
- Pager: standardize backgrounding on Ctrl+B
- Shell: recap rides the parent turn's prompt cache
- Tools: add scheduler lifecycle version clock

Source-Revision: 0f4d7c91b8b2b408333f6de1e8a76cb8eaa71899
This commit is contained in:
grokkybara[bot] 2026-07-21 18:10:23 +00:00
commit 3af4d5d398
556 changed files with 56609 additions and 21892 deletions

View file

@ -6,9 +6,9 @@ pub(crate) use serde_json::json;
pub(crate) use std::path::Path;
pub(crate) use std::time::{Duration, Instant};
pub(crate) use xai_grok_pager_pty_harness::{
ContentController, InferenceEndpoint, InferenceRequestMatcher, MockModel, PtyHarness,
ScriptedResponse, SseEvent, keys, oauth_env_for_pager, pager_binary, seed_fake_oauth, sse,
wait_for_labels_absent, wait_for_model_via_new_sessions,
AgentTurnExpectation, ContentController, MockModel, PtyHarness, ScriptedResponse, SseEvent,
keys, oauth_env_for_pager, pager_binary, seed_fake_oauth, sse, wait_for_labels_absent,
wait_for_model_via_new_sessions,
};
/// Default PTY size used by every e2e test. Large enough to render the
@ -22,6 +22,17 @@ pub(crate) const DEFAULT_COLS: u16 = 120;
/// which can take a few seconds on cold build directories.
pub(crate) const WELCOME_TIMEOUT: Duration = Duration::from_secs(20);
/// Wait budget for a `--continue` / resume to replay the prior transcript back
/// into scrollback. Resume is strictly heavier than a cold start: it runs
/// `session/load` (MCP startup, git chores, a full `updates.jsonl` replay, and
/// session spawn) on the agent's single-threaded runtime, and the client-side
/// `acp_send` has no timeout — so under the fully-parallel pty_e2e suite the
/// starved agent thread can push this well past the 20s `WELCOME_TIMEOUT`
/// (leaving the "Loading session…" placeholder up). Sized generously for the
/// same contention reason as `WRAP_TIMEOUT`, not because resume is slow when
/// run alone.
pub(crate) const RESUME_TIMEOUT: Duration = Duration::from_secs(60);
/// Substring we wait for on the welcome screen. Matches the menu label `"Quit"`
/// (`render_welcome_done` / gate menus); case-sensitive, so it does **not**
/// match the lowercase `"quit"` hint line during `AuthState::Authenticating`.
@ -582,11 +593,6 @@ pub(crate) fn responses_api_tool_call_events(
events
}
/// Chat Completions SSE stream with a single tool_call (fallback endpoint).
pub(crate) fn chat_completions_tool_call_events(name: &str, arguments: &str) -> Vec<SseEvent> {
chat_completions_tool_call_events_with_id("call_read_hdr", name, arguments)
}
/// [`chat_completions_tool_call_events`] with an explicit `tool_call` id, for
/// tests scripting several calls into ONE conversation (a reused id would
/// alias distinct calls in history and confuse dangling-call bookkeeping).
@ -643,100 +649,6 @@ pub(crate) fn chat_completions_tool_call_events_with_id(
]
}
/// Responses API SSE stream that emits a single assistant text message —
/// the FIFO counterpart of `set_response` for tests scripting DISTINCT text
/// replies per turn (e.g. one per auto-wake).
pub(crate) fn responses_api_message_events(text: &str) -> Vec<SseEvent> {
vec![
SseEvent::data(
json!({
"type": "response.created",
"sequence_number": 0,
"response": {
"id": "resp_text",
"object": "response",
"created_at": 1234567890,
"model": "test-model",
"status": "in_progress",
"output": []
}
})
.to_string(),
),
SseEvent::data(
json!({
"type": "response.output_text.delta",
"sequence_number": 1,
"item_id": "item_text",
"output_index": 0,
"content_index": 0,
"delta": text
})
.to_string(),
),
SseEvent::data(
json!({
"type": "response.completed",
"sequence_number": 2,
"response": {
"id": "resp_text",
"object": "response",
"created_at": 1234567890,
"model": "test-model",
"status": "completed",
"output": [{
"type": "message",
"id": "msg_text",
"role": "assistant",
"status": "completed",
"content": [{
"type": "output_text",
"text": text,
"annotations": []
}]
}],
"usage": {
"input_tokens": 10,
"output_tokens": 10,
"total_tokens": 20,
"input_tokens_details": { "cached_tokens": 0 },
"output_tokens_details": { "reasoning_tokens": 0 }
}
}
})
.to_string(),
),
SseEvent::data("[DONE]".to_string()),
]
}
/// Chat Completions SSE stream with a single assistant text message
/// (fallback endpoint counterpart of [`responses_api_message_events`]).
pub(crate) fn chat_completions_message_events(text: &str) -> Vec<SseEvent> {
vec![
SseEvent::data(
json!({
"id": "chatcmpl-text",
"object": "chat.completion.chunk",
"created": 1234567890,
"model": "test-model",
"choices": [{
"index": 0,
"delta": { "role": "assistant", "content": text },
"finish_reason": "stop"
}],
"usage": {
"prompt_tokens": 10,
"completion_tokens": 10,
"total_tokens": 20
}
})
.to_string(),
),
SseEvent::data("[DONE]".to_string()),
]
}
/// Poll the raw PTY stream until at least one OSC 52 clipboard payload has
/// been flushed (or `timeout` elapses), then return everything decoded so
/// far. A copy lands asynchronously after the triggering input, so a fixed
@ -820,22 +732,20 @@ pub(crate) fn locate_screen_text(screen: &str, needle: &str) -> Option<(u16, u16
None
}
/// Queue one scripted tool-call turn on both inference endpoints (only the
/// endpoint the agent actually uses drains its FIFO; the other stays parked).
pub(crate) fn enqueue_tool_turn(
/// Register one named scripted tool-call turn on both inference endpoints.
pub(crate) fn expect_tool_turn(
content: &ContentController,
call_id: &str,
name: &str,
args: String,
) {
content.enqueue_response(
"/v1/responses",
) -> AgentTurnExpectation {
content.expect_agent_turn_with_responses(
format!("tool turn {call_id}"),
ScriptedResponse::sse(responses_api_tool_call_events(call_id, name, &args)),
);
content.enqueue_response(
"/v1/chat/completions",
ScriptedResponse::sse(chat_completions_tool_call_events(name, &args)),
);
ScriptedResponse::sse(chat_completions_tool_call_events_with_id(
call_id, name, &args,
)),
)
}
/// Responses API SSE stream whose `response.completed` output carries one
@ -974,7 +884,7 @@ pub(crate) fn chat_completions_parallel_tool_call_events(
}
/// Queue one scripted turn with parallel tool calls on both inference
/// endpoints (see [`enqueue_tool_turn`]).
/// endpoints (see [`expect_tool_turn`]).
pub(crate) fn enqueue_parallel_tool_turn(
content: &ContentController,
calls: &[(&str, &str, String)],
@ -991,23 +901,15 @@ pub(crate) fn enqueue_parallel_tool_turn(
/// Seed a target file under the isolated HOME and queue a scripted `read_file`
/// tool call (Responses + Chat Completions) so the pager renders a Read header.
pub(crate) fn seed_read_file_tool_call(content: &ContentController, abs_path: &Path) {
pub(crate) fn seed_read_file_tool_call(
content: &ContentController,
abs_path: &Path,
) -> AgentTurnExpectation {
let args = json!({ "target_file": abs_path.to_string_lossy() }).to_string();
// Prefer Responses API (primary agent path); also queue Chat Completions.
content.enqueue_response(
"/v1/responses",
ScriptedResponse::sse(responses_api_tool_call_events(
"call_read_hdr",
"read_file",
&args,
)),
);
content.enqueue_response(
"/v1/chat/completions",
ScriptedResponse::sse(chat_completions_tool_call_events("read_file", &args)),
);
let turn = expect_tool_turn(content, "call_read_hdr", "read_file", args);
// Follow-up turn after tool result: plain completion so the session settles.
content.set_response(READ_HDR_SENTINEL);
turn
}
// ── Minimal (scrollback-native) mode e2e helpers ────────────────────────