From f85bf314bdc92a499b9f7686eb8c7d7cb13bb9c2 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Thu, 8 Oct 2026 23:33:38 -0700 Subject: [PATCH 01/15] feat(rust): add resume skill reload opt-out Allow callers that own skill reconciliation to omit the SDK's automatic post-resume reload. Preserve awaited best-effort behavior by default and leave mandatory resume readiness and wire settings unchanged. Cover request omission, retained default behavior, callbacks, and cleanup with framed transport controls. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- rust/README.md | 15 ++ rust/src/ahp_host/factory_tests.rs | 18 +++ rust/src/session.rs | 56 +++---- rust/src/types.rs | 22 +++ rust/src/types/tests.rs | 25 ++++ rust/tests/prepared_session_test.rs | 219 ++++++++++++++++++++++++++++ rust/tests/session_test.rs | 56 ++++--- rust/tests/skill_provider_test.rs | 49 +++++++ 8 files changed, 414 insertions(+), 46 deletions(-) diff --git a/rust/README.md b/rust/README.md index 87484defec..38e6a155cc 100644 --- a/rust/README.md +++ b/rust/README.md @@ -110,6 +110,17 @@ When allowed and performed, `session.transcript_recovery()` returns the planned backup path, invalid line numbers, and whether `session.start` was moved; the backup is written on the next append rather than during resume. +Rust resumes await a best-effort `session.skills.reload` by default: returned +errors are logged, but an unanswered request delays resume completion. +`ResumeSessionConfig::with_reload_skills(false)` omits that automatic request +when the application owns required skill reconciliation before skill-dependent +work. It is SDK-local, does not disable skills, and does not prove the catalog +or disabled-skill preferences are current. Explicit +`session.rpc().skills().reload().await` remains available. This option does not +gate continued pending work or work already running on another connection; +applications must account for those paths in their own reconciliation barrier. +It does not cancel or settle a reload that was already issued. + After `Client::start` succeeds, inspect its startup cost without parsing logs: ```rust,ignore @@ -1456,6 +1467,10 @@ larger results return a filesystem error before encoding or decoding. ### Rust-only API +Rust also exposes `ResumeSessionConfig::with_reload_skills` to control its +automatic post-resume skill reload; other SDKs such as Node and .NET do not +issue that automatic request. The Rust default is unchanged. + A handful of conveniences exist only on the Rust SDK as of 0.1.0. These are surface areas where Rust idiom (newtypes, enums, trait objects) gives a clearly nicer shape than Node/Python/Go/.NET currently expose. Rust diff --git a/rust/src/ahp_host/factory_tests.rs b/rust/src/ahp_host/factory_tests.rs index bb328c7acd..894f06131c 100644 --- a/rust/src/ahp_host/factory_tests.rs +++ b/rust/src/ahp_host/factory_tests.rs @@ -677,6 +677,24 @@ fn host_settings_round_trip_without_callbacks_or_losing_constraints() { assert!(wire.get("mcpOauthTokenStorage").is_none()); } +#[test] +fn host_resume_settings_do_not_forward_local_skill_reload_policy() { + let config = crate::ResumeSessionConfig::new(crate::SessionId::from("skills")); + let expected = resume_config_for_host(&config).unwrap(); + for reload in [false, true] { + assert_eq!( + resume_config_for_host(&config.clone().with_reload_skills(reload)).unwrap(), + expected + ); + } + let decoded = + resume_config_from_host(&serde_json::from_value(expected.clone()).unwrap()).unwrap(); + assert_eq!(decoded.reload_skills, None); + let mut unsupported = expected; + unsupported["reloadSkills"] = json!(false); + assert!(resume_config_from_host(&serde_json::from_value(unsupported).unwrap()).is_err()); +} + #[tokio::test] async fn release_before_request_poll_prevents_factory_invocation() { let (client, mut peer) = fixture(); diff --git a/rust/src/session.rs b/rust/src/session.rs index 1ba8325095..ca2b6fe14b 100644 --- a/rust/src/session.rs +++ b/rust/src/session.rs @@ -1701,8 +1701,10 @@ impl Client { /// Resume an existing session on the CLI. /// - /// Sends `session.resume` and `session.skills.reload`, registers the - /// session on the router, and spawns the event loop. + /// Sends `session.resume`, registers the session on the router, and + /// spawns the event loop. By default, also awaits a best-effort + /// `session.skills.reload`; [`ResumeSessionConfig::with_reload_skills(false)`](ResumeSessionConfig::with_reload_skills) + /// omits that request when the caller owns skill reconciliation. /// /// All callbacks (event handler, hooks, transform) are configured /// via [`ResumeSessionConfig`] using its `with_*` builder methods. @@ -2084,8 +2086,10 @@ impl Client { /// Resume an existing session on the CLI. /// - /// Sends `session.resume` and `session.skills.reload`, registers the - /// session on the router, and spawns the event loop. + /// Sends `session.resume`, registers the session on the router, and + /// spawns the event loop. By default, also awaits a best-effort + /// `session.skills.reload` unless [`ResumeSessionConfig::reload_skills`] + /// is explicitly `false`. /// /// All callbacks (event handler, hooks, transform) are configured /// via [`ResumeSessionConfig`] using its `with_*` builder methods. @@ -2111,6 +2115,7 @@ impl Client { None }; let session_id = config.session_id.clone(); + let reload_skills = config.reload_skills.unwrap_or(true); if config.hooks_handler.is_some() && config.hooks.is_none() { config.hooks = Some(true); } @@ -2327,27 +2332,28 @@ impl Client { registration.cleanup(event_loop).await; return Err(error); } - // Reload skills after resume (best-effort). - let skills_reload_start = Instant::now(); - if let Err(e) = self - .call( - "session.skills.reload", - Some(serde_json::json!({ "sessionId": session_id })), - ) - .await - { - warn!( - elapsed_ms = skills_reload_start.elapsed().as_millis(), - session_id = %session_id, - error = %e, - "Client::resume_session skills reload request failed" - ); - } else { - tracing::debug!( - elapsed_ms = skills_reload_start.elapsed().as_millis(), - session_id = %session_id, - "Client::resume_session skills reload request completed successfully" - ); + if reload_skills { + let skills_reload_start = Instant::now(); + if let Err(e) = self + .call( + "session.skills.reload", + Some(serde_json::json!({ "sessionId": session_id })), + ) + .await + { + warn!( + elapsed_ms = skills_reload_start.elapsed().as_millis(), + session_id = %session_id, + error = %e, + "Client::resume_session skills reload request failed" + ); + } else { + tracing::debug!( + elapsed_ms = skills_reload_start.elapsed().as_millis(), + session_id = %session_id, + "Client::resume_session skills reload request completed successfully" + ); + } } *capabilities.write() = resume_result.capabilities.unwrap_or_default(); diff --git a/rust/src/types.rs b/rust/src/types.rs index 9917693110..fbc58c2870 100644 --- a/rust/src/types.rs +++ b/rust/src/types.rs @@ -3604,6 +3604,20 @@ impl SessionConfig { pub struct ResumeSessionConfig { /// ID of the session to resume. pub session_id: SessionId, + /// Whether the SDK automatically reloads skills after the runtime resumes. + /// + /// Unset or `true` preserves the awaited, best-effort + /// `session.skills.reload` call: returned errors are logged, but an + /// unanswered request still delays resume completion. Set `false` only + /// when the caller owns any required skill reconciliation before + /// skill-dependent work. + /// + /// This SDK-local option is not sent to the runtime. It does not disable + /// skills or imply that the catalog or disabled-skill preferences are + /// current. It does not gate continued pending work or work already + /// running on another connection, nor cancel or settle an already-issued + /// reload. + pub reload_skills: Option, /// Model to use for this session (e.g. `"gpt-4"`, `"claude-sonnet-4"`). /// Can change the model when resuming. pub model: Option, @@ -3900,6 +3914,7 @@ impl std::fmt::Debug for ResumeSessionConfig { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.debug_struct("ResumeSessionConfig") .field("session_id", &self.session_id) + .field("reload_skills", &self.reload_skills) .field("model", &self.model) .field("allowed_models", &self.allowed_models) .field("client_name", &self.client_name) @@ -4206,6 +4221,7 @@ impl ResumeSessionConfig { pub fn new(session_id: SessionId) -> Self { Self { session_id, + reload_skills: None, model: None, allowed_models: None, client_name: None, @@ -4302,6 +4318,12 @@ impl ResumeSessionConfig { } } + /// Set [`Self::reload_skills`], controlling the SDK's automatic skill reload. + pub fn with_reload_skills(mut self, reload_skills: bool) -> Self { + self.reload_skills = Some(reload_skills); + self + } + /// Install a [`PermissionHandler`] for the resumed session. pub fn with_permission_handler(mut self, handler: Arc) -> Self { self.permission_handler = Some(handler); diff --git a/rust/src/types/tests.rs b/rust/src/types/tests.rs index 0749d946bd..215f1ffa63 100644 --- a/rust/src/types/tests.rs +++ b/rust/src/types/tests.rs @@ -1149,6 +1149,31 @@ fn resume_session_config_serializes_continue_pending_work_to_camel_case() { assert!(json.get("continuePendingWork").is_none()); } +#[test] +fn resume_reload_skills_is_local_only_and_preserves_skill_configuration() { + let config = ResumeSessionConfig::new(SessionId::from("skills")) + .with_enable_skills(true) + .with_enable_config_discovery(true) + .with_skill_directories([PathBuf::from("skills")]) + .with_disabled_skills(["disabled"]); + assert_eq!(config.reload_skills, None); + let (wire, _) = config.clone().into_wire().unwrap(); + let expected = serde_json::to_value(wire).unwrap(); + assert_eq!(expected["enableSkills"], true); + assert_eq!(expected["enableConfigDiscovery"], true); + assert_eq!(expected["skillDirectories"], json!(["skills"])); + assert_eq!(expected["disabledSkills"], json!(["disabled"])); + + for reload in [false, true] { + let configured = config.clone().with_reload_skills(reload); + assert_eq!(configured.reload_skills, Some(reload)); + assert_eq!(configured.clone().reload_skills, Some(reload)); + assert!(format!("{configured:?}").contains(&format!("reload_skills: Some({reload})"))); + let (wire, _) = configured.into_wire().unwrap(); + assert_eq!(serde_json::to_value(wire).unwrap(), expected); + } +} + #[test] fn resume_policy_and_recovery_report_round_trip() { let config = diff --git a/rust/tests/prepared_session_test.rs b/rust/tests/prepared_session_test.rs index 1ae60cbbe1..aec924cb37 100644 --- a/rust/tests/prepared_session_test.rs +++ b/rust/tests/prepared_session_test.rs @@ -213,6 +213,11 @@ impl McpAuthHandler for CancelMcpAuthHandler { } } +struct NoopHooks; + +#[async_trait] +impl github_copilot_sdk::hooks::SessionHooks for NoopHooks {} + fn make_client() -> (Client, FakeServer) { let (client_write, server_read) = duplex(1 << 20); let (server_write, client_read) = duplex(1 << 20); @@ -1077,6 +1082,220 @@ async fn resume_session_wrapper_keeps_rpc_sequence() { drop(session); } +#[tokio::test] +async fn resume_without_automatic_skill_reload() { + check_resume_without_automatic_skill_reload(false).await; +} + +#[tokio::test] +async fn prepared_resume_without_automatic_skill_reload() { + check_resume_without_automatic_skill_reload(true).await; +} + +async fn check_resume_without_automatic_skill_reload(prepared: bool) { + let (client, mut server) = make_client(); + let session_id = SessionId::new("caller-owned-skills"); + let published = Arc::new(tokio::sync::Notify::new()); + let config = ResumeSessionConfig::new(session_id.clone()) + .with_reload_skills(false) + .with_enable_skills(true) + .with_permission_handler(Arc::new(PublicationFence(published.clone()))) + .with_hooks(Arc::new(NoopHooks)) + .with_mcp_auth_handler(Arc::new(CancelMcpAuthHandler)) + .with_coauthor_enabled(false); + let prepared = prepared.then(|| client.prepare_resume_session(config.clone()).unwrap()); + let mut events = prepared.as_ref().map(|prepared| prepared.subscribe()); + let mut start = Box::pin(async { + match prepared { + Some(prepared) => prepared.start().await, + None => client.resume_session(config).await, + } + }); + + let resume = tokio::select! { + result = &mut start => panic!("resume finished before its request: {:?}", result.map(|_| ())), + request = server.read_request() => request, + }; + assert_eq!(resume["method"], "session.resume"); + assert_eq!(resume["params"]["enableSkills"], true); + assert_eq!(resume["params"]["hooks"], true); + assert!(resume["params"].get("reloadSkills").is_none()); + assert!(start.as_mut().now_or_never().is_none()); + server + .send_event(session_id.as_str(), "before-resume", "session.idle", true) + .await; + server + .await_publication(session_id.as_str(), &published) + .await; + server + .respond(&resume, json!({ "sessionId": session_id.as_str() })) + .await; + + let interest = tokio::select! { + result = &mut start => panic!("resume skipped MCP-auth readiness: {:?}", result.map(|_| ())), + request = server.read_request() => request, + }; + assert_eq!(interest["method"], "session.eventLog.registerInterest"); + assert!(start.as_mut().now_or_never().is_none()); + let hook = json!({ + "jsonrpc": "2.0", "id": 9020, "method": "hooks.invoke", + "params": { + "sessionId": session_id.as_str(), "hookType": "sessionEnd", + "input": { + "sessionId": session_id.as_str(), "timestamp": 1234567890, + "cwd": "/tmp", "reason": "complete" + } + } + }); + write_framed(&mut server.write, &serde_json::to_vec(&hook).unwrap()).await; + let response = server.read_request().await; + assert_eq!(response["id"], 9020); + assert_eq!(response["result"]["output"], json!({})); + server + .send_event(session_id.as_str(), "during-interest", "session.idle", true) + .await; + server + .await_publication(session_id.as_str(), &published) + .await; + server.respond(&interest, json!({})).await; + + let options = tokio::select! { + result = &mut start => panic!("resume skipped options readiness: {:?}", result.map(|_| ())), + request = server.read_request() => request, + }; + assert_eq!( + options["method"], "session.options.update", + "opting out must not issue session.skills.reload" + ); + assert!(start.as_mut().now_or_never().is_none()); + server + .send_event(session_id.as_str(), "during-options", "session.idle", true) + .await; + server + .await_publication(session_id.as_str(), &published) + .await; + server.respond(&options, json!({ "success": true })).await; + let session = timeout(TIMEOUT, start).await.unwrap().unwrap(); + let mut events = events.take().unwrap_or_else(|| session.subscribe()); + for id in [ + "before-resume", + "publication-fence", + "during-interest", + "publication-fence", + "during-options", + "publication-fence", + ] { + expect_event_id(&mut events, id).await; + } + + let skills = session.rpc().skills(); + let mut explicit_reload = Box::pin(skills.reload()); + let reload = tokio::select! { + result = &mut explicit_reload => panic!("explicit reload finished before its request: {result:?}"), + request = server.read_request() => request, + }; + assert_eq!(reload["method"], "session.skills.reload"); + assert!(explicit_reload.as_mut().now_or_never().is_none()); + server + .respond(&reload, json!({ "errors": [], "warnings": [] })) + .await; + let diagnostics = timeout(TIMEOUT, explicit_reload).await.unwrap().unwrap(); + assert!(diagnostics.errors.is_empty()); + assert!(diagnostics.warnings.is_empty()); + server.expect_quiet().await; + session.stop_event_loop().await; + drop(session); + expect_closed(&mut events).await; + await_no_registrations(&client).await; +} + +#[tokio::test] +async fn automatic_skill_reload_waits_for_settlement_and_tolerates_returned_errors() { + for reload_skills in [None, Some(true)] { + for reload_error in [false, true] { + let (client, mut server) = make_client(); + let session_id = SessionId::new("automatic-skills"); + let mut config = ResumeSessionConfig::new(session_id.clone()); + config.reload_skills = reload_skills; + let mut start = Box::pin(client.resume_session(config)); + let resume = tokio::select! { + result = &mut start => panic!("resume finished before its request: {:?}", result.map(|_| ())), + request = server.read_request() => request, + }; + assert_eq!(resume["method"], "session.resume"); + server + .respond(&resume, json!({ "sessionId": session_id.as_str() })) + .await; + let reload = tokio::select! { + result = &mut start => panic!("resume did not await its reload: {:?}", result.map(|_| ())), + request = server.read_request() => request, + }; + assert_eq!(reload["method"], "session.skills.reload"); + assert!(start.as_mut().now_or_never().is_none()); + if reload_error { + server + .respond_error(&reload, -32003, "skills reload failed") + .await; + } else { + server.respond(&reload, json!({})).await; + } + let session = timeout(TIMEOUT, start).await.unwrap().unwrap(); + assert_eq!(session.id(), &session_id); + server.expect_quiet().await; + session.stop_event_loop().await; + drop(session); + await_no_registrations(&client).await; + } + } +} + +#[tokio::test] +async fn resume_without_skill_reload_cleans_up_on_owner_connection_loss() { + for phase in [ + "session.resume", + "session.eventLog.registerInterest", + "session.options.update", + ] { + let (client, mut server) = make_client(); + let session_id = SessionId::new("skills-owner-loss"); + let prepared = client + .prepare_resume_session( + ResumeSessionConfig::new(session_id.clone()) + .with_reload_skills(false) + .with_mcp_auth_handler(Arc::new(CancelMcpAuthHandler)) + .with_coauthor_enabled(false), + ) + .unwrap(); + let mut events = prepared.subscribe(); + let start = tokio::spawn(prepared.start()); + for expected in [ + "session.resume", + "session.eventLog.registerInterest", + "session.options.update", + ] { + let request = server.read_request().await; + assert_eq!(request["method"], expected); + if expected == phase { + break; + } + server + .respond( + &request, + json!({ "sessionId": session_id.as_str(), "success": true }), + ) + .await; + } + client.force_stop(); + let error = expect_error(timeout(TIMEOUT, start).await.unwrap().unwrap()); + assert_eq!( + error.kind(), + &ErrorKind::Protocol(github_copilot_sdk::ProtocolErrorKind::RequestCancelled), + ); + expect_closed(&mut events).await; + await_no_registrations(&client).await; + } +} + #[tokio::test] async fn resume_bootstrap_retains_all_startup_phases_and_keeps_later_subscribers_live() { let (client, mut server) = make_client(); diff --git a/rust/tests/session_test.rs b/rust/tests/session_test.rs index 40c455549c..53c5d03221 100644 --- a/rust/tests/session_test.rs +++ b/rust/tests/session_test.rs @@ -8636,6 +8636,7 @@ enum StartupRoute { Local, Deferred, Resume, + ResumeWithoutSkillReload, } #[derive(Clone, Copy, Debug)] @@ -8662,17 +8663,22 @@ async fn check_startup_callback_ownership( write: server_write, session_id: "startup-callbacks".into(), }; - let prepared = if route == StartupRoute::Resume { - client - .prepare_resume_session( - github_copilot_sdk::ResumeSessionConfig::new(SessionId::new(&server.session_id)) - .with_permission_handler(permission.clone()) - .with_session_fs_provider(fs.clone()) - .with_elicitation_handler(elicitation.clone()) - .with_mcp_auth_handler(mcp_auth.clone()) - .with_coauthor_enabled(false), - ) - .unwrap() + let resume = matches!( + route, + StartupRoute::Resume | StartupRoute::ResumeWithoutSkillReload + ); + let prepared = if resume { + let mut config = + github_copilot_sdk::ResumeSessionConfig::new(SessionId::new(&server.session_id)) + .with_permission_handler(permission.clone()) + .with_session_fs_provider(fs.clone()) + .with_elicitation_handler(elicitation.clone()) + .with_mcp_auth_handler(mcp_auth.clone()) + .with_coauthor_enabled(false); + if route == StartupRoute::ResumeWithoutSkillReload { + config = config.with_reload_skills(false); + } + client.prepare_resume_session(config).unwrap() } else { let mut config = SessionConfig::default() .with_permission_handler(permission.clone()) @@ -8694,7 +8700,7 @@ async fn check_startup_callback_ownership( let mut request = timeout(TIMEOUT, server.read_request()).await.unwrap(); assert_eq!( request["method"], - if route == StartupRoute::Resume { + if resume { "session.resume" } else { "session.create" @@ -8875,20 +8881,26 @@ async fn check_startup_callback_ownership( #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn startup_callbacks_owned_until_resume_response() { - for outcome in [StartupOutcome::Failure, StartupOutcome::Cancellation] { - Box::pin(check_startup_callback_ownership( - StartupRoute::Resume, - "session.resume", - outcome, - true, - )) - .await; + for route in [StartupRoute::Resume, StartupRoute::ResumeWithoutSkillReload] { + for outcome in [StartupOutcome::Failure, StartupOutcome::Cancellation] { + Box::pin(check_startup_callback_ownership( + route, + "session.resume", + outcome, + true, + )) + .await; + } } } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn startup_callbacks_owned_through_mcp_interest() { - for route in [StartupRoute::Local, StartupRoute::Resume] { + for route in [ + StartupRoute::Local, + StartupRoute::Resume, + StartupRoute::ResumeWithoutSkillReload, + ] { for outcome in [StartupOutcome::Failure, StartupOutcome::Cancellation] { for before_response in [false, true] { Box::pin(check_startup_callback_ownership( @@ -8922,6 +8934,7 @@ async fn startup_callbacks_owned_through_mode_patch() { StartupRoute::Local, StartupRoute::Deferred, StartupRoute::Resume, + StartupRoute::ResumeWithoutSkillReload, ] { for outcome in [StartupOutcome::Failure, StartupOutcome::Cancellation] { Box::pin(check_startup_callback_ownership( @@ -8954,6 +8967,7 @@ async fn startup_callbacks_survive_successful_setup() { StartupRoute::Local, StartupRoute::Deferred, StartupRoute::Resume, + StartupRoute::ResumeWithoutSkillReload, ] { Box::pin(check_startup_callback_ownership( route, diff --git a/rust/tests/skill_provider_test.rs b/rust/tests/skill_provider_test.rs index 71494ee65c..c85cc68d71 100644 --- a/rust/tests/skill_provider_test.rs +++ b/rust/tests/skill_provider_test.rs @@ -520,6 +520,55 @@ async fn serves_early_create_and_resume_callbacks() { let _ = timeout(TIMEOUT, resume).await.unwrap().unwrap(); } +#[tokio::test] +async fn serves_skill_provider_callbacks_without_automatic_reload() { + let (client, mut server) = make_client(); + let provider = Arc::new(RecordingProvider::new()); + let calls = provider.calls.clone(); + let resume = tokio::spawn({ + let client = client.clone(); + async move { + client + .resume_session( + ResumeSessionConfig::new(SessionId::new("caller-owned-provider")) + .with_reload_skills(false) + .with_skill_provider(provider), + ) + .await + .unwrap() + } + }); + let request = server.read_request().await; + assert_eq!(request["method"], "session.resume"); + assert_eq!(request["params"]["hasSkillProvider"], true); + server + .send_request( + 12, + "skillProvider.list", + json!({ "sessionId": "caller-owned-provider" }), + ) + .await; + let list = server.read_response().await; + assert_eq!(list["id"], 12); + assert_eq!(list["result"]["skills"][0]["name"], "review"); + server + .respond(&request, json!({ "sessionId": "caller-owned-provider" })) + .await; + let session = timeout(TIMEOUT, resume).await.unwrap().unwrap(); + server + .send_request( + 13, + "skillProvider.read", + json!({ "sessionId": "caller-owned-provider", "name": "review" }), + ) + .await; + let read = server.read_response().await; + assert_eq!(read["id"], 13); + assert_eq!(read["result"]["markdown"], "Review carefully."); + assert_eq!(calls.snapshot(), ["list", "read:review"]); + session.stop_event_loop().await; +} + #[tokio::test] async fn cloud_sessions_reject_provider_before_rpc() { let (client, mut server) = make_client(); From e799329dcbe9514dcda30b586eee7ca4a64bb3b8 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 12:47:39 -0700 Subject: [PATCH 02/15] fix(test): resolve Rust codegen schemas across SDK layouts Use the generator's existing schema resolver in the selected-schema discriminator regression instead of assuming a nested runtime checkout. Preserve the same discriminator assertions for explicit checkout schemas and the pinned published package. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- nodejs/test/rust-codegen.test.ts | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/nodejs/test/rust-codegen.test.ts b/nodejs/test/rust-codegen.test.ts index 5f34d804b1..80720d5877 100644 --- a/nodejs/test/rust-codegen.test.ts +++ b/nodejs/test/rust-codegen.test.ts @@ -1,5 +1,6 @@ import type { ApiSchema } from "../../scripts/codegen/utils.ts"; import { + getApiSchemaPath, normalizeSchemaBrandCasing, postProcessSchema, propagateInternalVisibility, @@ -976,18 +977,13 @@ describe("Rust x-legacy-parameters", () => { ).toThrow(/Rust string enum Kind is requested for different values/); }); - it("keeps every const discriminator of the committed API schema distinct in Rust", () => { + it("keeps every const discriminator of the selected API schema distinct in Rust", async () => { // Mirror the generator's own schema preparation so emission order matches. const schema = propagateInternalVisibility( postProcessSchema( stripBooleanLiterals( normalizeSchemaBrandCasing( - JSON.parse( - readFileSync( - new URL("../../../../generated/api.schema.json", import.meta.url), - "utf8" - ) - ) as ApiSchema + JSON.parse(readFileSync(await getApiSchemaPath(), "utf8")) as ApiSchema ) ) as JSONSchema7 ) From 47c21069d6f68f9a066c3d46f4bcc9bd9603c09d Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 13:47:08 -0700 Subject: [PATCH 03/15] fix(ci): isolate published-runtime device policy fixtures Run the existing strict device-policy controls in the Linux CAPI lane using disposable containers and the documented production device channel. Preserve pinned runtimes and source-artifact coverage, and fail on missing results or uncertain cleanup. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/actions/run-alpine-tests/action.yml | 1 + .github/workflows/sdk-nodejs.yml | 58 ++- .github/workflows/sdk-platform.yml | 4 + .github/workflows/sdk.yml | 4 + CONTRIBUTING.md | 21 + scripts/ci/device-policy-bootstrap.mjs | 85 ++++ scripts/ci/device-policy-fixture.mjs | 466 ++++++++++++++++++++ scripts/ci/device-policy-fixture.test.mjs | 207 +++++++++ scripts/ci/device-policy-legacy.js | 10 + scripts/ci/device-policy-native.js | 10 + 10 files changed, 862 insertions(+), 4 deletions(-) create mode 100644 scripts/ci/device-policy-bootstrap.mjs create mode 100644 scripts/ci/device-policy-fixture.mjs create mode 100644 scripts/ci/device-policy-fixture.test.mjs create mode 100644 scripts/ci/device-policy-legacy.js create mode 100644 scripts/ci/device-policy-native.js diff --git a/.github/actions/run-alpine-tests/action.yml b/.github/actions/run-alpine-tests/action.yml index a9842c11be..7808ca3f67 100644 --- a/.github/actions/run-alpine-tests/action.yml +++ b/.github/actions/run-alpine-tests/action.yml @@ -54,6 +54,7 @@ runs: --env COPILOT_SDK_ROOT="$ALPINE_TEST_SDK_ROOT" \ --env COPILOT_HMAC_KEY \ --env COPILOT_SDK_E2E_BACKEND \ + --env COPILOT_CI_RUNTIME_SOURCE \ --env BUNDLED_CLI_CACHE_DIR \ --env CARGO_TERM_COLOR \ --env RUST_BACKTRACE \ diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index 9fbbf97e76..eba1d72a4a 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -7,6 +7,7 @@ env: HUSKY: 0 POWERSHELL_UPDATECHECK: Off SDK_HOME: ${{ inputs.sdk-home }} + COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} SETUP_NODE_TIMEOUT_MINUTES: 10 on: @@ -19,6 +20,9 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string outputs: capi-result: description: "Standard-platform CAPI job result, independent of BYOK jobs" @@ -89,16 +93,48 @@ jobs: working-directory: ${{ inputs.sdk-home }}/scripts/corrections - if: runner.os == 'Windows' run: pwsh.exe -Command "Write-Host 'PowerShell ready'" + - name: Test device-policy CI plans and required-result guard + if: runner.os == 'Linux' + working-directory: . + run: node --test "$SDK_HOME/scripts/ci/device-policy-fixture.test.mjs" + - name: Prepare isolated production device-policy fixtures + id: device-policy + if: inputs.runtime-source == 'published' && runner.os == 'Linux' + working-directory: . + run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" prepare - name: Test out-of-process id: subprocess env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} run: | + test_args=() + pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") + if [[ -n "$pattern" ]]; then + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + test_args+=(--testNamePattern "$pattern") + fi if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ]; then - npm test -- --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" + npm test -- "${test_args[@]}" --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" else - npm test + npm test -- "${test_args[@]}" fi + - name: Test required production device-policy controls + if: ${{ !cancelled() && inputs.runtime-source == 'published' && runner.os == 'Linux' && (steps.subprocess.outcome == 'success' || steps.subprocess.outcome == 'failure') }} + env: + COPILOT_CLI_PATH: ${{ steps.device-policy.outputs.native-launcher }} + COPILOT_LEGACY_CLI_PATH: ${{ steps.device-policy.outputs.legacy-launcher }} + COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} + run: | + status=0 + npm test -- test/e2e/managed_plugin_progress.e2e.test.ts test/e2e/rpc_server.e2e.test.ts \ + --testNamePattern "$(node ../scripts/ci/device-policy-fixture.mjs required-pattern)" \ + --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-device-policy-results.json" || status=$? + node ../scripts/ci/device-policy-fixture.mjs verify "$RUNNER_TEMP/sdk-device-policy-results.json" + exit "$status" + - name: Clean up owned device-policy containers + if: always() && steps.device-policy.outcome == 'success' + working-directory: . + run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" cleanup - name: Upload Flake Finder Node.js test results if: always() && (github.event_name == 'merge_group' || github.event_name == 'pull_request') && runner.os == 'Linux' continue-on-error: true @@ -162,7 +198,14 @@ jobs: - name: Test out-of-process E2Es env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - run: npm test -- test/e2e + run: | + test_args=() + pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") + if [[ -n "$pattern" ]]; then + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + test_args+=(--testNamePattern "$pattern") + fi + npm test -- test/e2e "${test_args[@]}" nodejs-musl-x64: name: "Node.js (Alpine x64, CAPI)" @@ -182,6 +225,7 @@ jobs: - uses: ./.github/actions/run-alpine-tests env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} + COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} with: image: node:22-alpine sdk-root: /workspace/${{ inputs.sdk-home }} @@ -206,6 +250,12 @@ jobs: test -x "$COPILOT_CLI_PATH" test -f "$(dirname "$COPILOT_CLI_PATH")/runtime.node" status=0 - npm test || status=$? + pattern=$(node "$COPILOT_SDK_ROOT/scripts/ci/device-policy-fixture.mjs" portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") + if [ -n "$pattern" ]; then + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + npm test -- --testNamePattern "$pattern" || status=$? + else + npm test || status=$? + fi COPILOT_SDK_DEFAULT_CONNECTION=inprocess npm test -- test/e2e/inprocess_ffi.e2e.test.ts test/e2e/auth_host.e2e.test.ts || status=$? exit "$status" diff --git a/.github/workflows/sdk-platform.yml b/.github/workflows/sdk-platform.yml index f1ed4e6ed4..568afd4147 100644 --- a/.github/workflows/sdk-platform.yml +++ b/.github/workflows/sdk-platform.yml @@ -11,6 +11,9 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string outputs: nodejs-result: description: "Node.js SDK result, independent of other languages" @@ -29,6 +32,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit python: diff --git a/.github/workflows/sdk.yml b/.github/workflows/sdk.yml index 56c165db89..dbf255259e 100644 --- a/.github/workflows/sdk.yml +++ b/.github/workflows/sdk.yml @@ -446,6 +446,7 @@ jobs: with: platform: linux-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-x64: @@ -456,6 +457,7 @@ jobs: with: platform: linuxmusl-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-arm64: @@ -476,6 +478,7 @@ jobs: with: platform: darwin-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-win32-x64: @@ -487,6 +490,7 @@ jobs: with: platform: win32-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit # Only Linux CAPI gates this rollup; the full SDK aggregate still reports all coverage. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 74519107b4..15e5c5cc7c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -369,6 +369,27 @@ backends, and languages keep their existing scheduling and failure reporting; they do not gate this rollup. The full `SDK` aggregate still requires all scheduled coverage to succeed. +Standalone CI assigns the two managed-device fixture controls to a separate +mandatory Linux CAPI step. Each fixture-bearing CLI child runs the unchanged, +pinned published runtime in its own disposable container, with the fixture +installed at the [documented production device-policy location](https://docs.github.com/en/copilot/how-tos/administer-copilot/manage-for-enterprise/use-managed-settings/deploy-managed-settings). +The launcher consumes the temporary-file test hint, verifies isolation and +root file ownership, then restores the runner's identity before launching the +runtime. It never writes policy to the host. A result guard requires both +original plugin-lifecycle and sessionless device/model tests to pass; missing +or skipped controls fail the job. Other published-runtime profiles explicitly +delegate only those two cases to this step. Source-runtime profiles retain +their existing full test selection and launch setup. + +The CI-only container setup requires Linux, Docker, and Node.js 22.15 or newer. +Do not provision a machine-wide policy file on a developer workstation to +run these fixtures. Launcher plan, selection, and result-guard unit controls +run without Docker, a CLI, or dependency installation: + +```bash +node --test scripts/ci/device-policy-fixture.test.mjs +``` + The three BYOK backend sweeps run in separate Linux TypeScript jobs, alongside the normal CAPI job; they do not repeat unit tests, packaging, or static checks. After preparing the runtime as described above, run a sweep from the diff --git a/scripts/ci/device-policy-bootstrap.mjs b/scripts/ci/device-policy-bootstrap.mjs new file mode 100644 index 0000000000..48e30268e8 --- /dev/null +++ b/scripts/ci/device-policy-bootstrap.mjs @@ -0,0 +1,85 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { POLICY_FILE, verifyIdentity, verifyIsolation } from "./device-policy-fixture.mjs"; + +function start() { + const plan = JSON.parse(process.env.COPILOT_CI_DEVICE_POLICY_PLAN); + if (process.platform !== "linux" || typeof process.execve !== "function") + throw new Error("Unsupported device-policy container runtime"); + if (process.argv[2] === "--runtime") { + const policy = fs.lstatSync(POLICY_FILE); + verifyIdentity(plan, { + uid: process.getuid(), + gid: process.getgid(), + groups: process.getgroups(), + policy: { + isFile: policy.isFile(), + isSymbolicLink: policy.isSymbolicLink(), + uid: policy.uid, + gid: policy.gid, + mode: policy.mode, + }, + }); + process.execve(plan.executable, plan.argv, plan.env); + throw new Error("Runtime exec unexpectedly returned"); + } + verifyIsolation({ + originalNamespace: plan.originalNamespace, + namespace: fs.readlinkSync("/proc/self/ns/mnt"), + mountinfo: fs.readFileSync("/proc/self/mountinfo", "utf8"), + uid: process.getuid(), + gid: process.getgid(), + }); + if ( + !Number.isSafeInteger(plan.uid) || + plan.uid <= 0 || + !Number.isSafeInteger(plan.gid) || + plan.gid < 0 || + !plan.groups.every((group) => Number.isSafeInteger(group) && group >= 0) + ) + throw new Error("Invalid runner identity"); + const directory = path.dirname(POLICY_FILE); + if (fs.existsSync(directory)) throw new Error("The container already has a device-policy directory"); + fs.mkdirSync(directory, { mode: 0o755 }); + fs.writeFileSync(POLICY_FILE, plan.policy, { mode: 0o644, flag: "wx" }); + fs.chmodSync(POLICY_FILE, 0o644); + + const passwd = fs.readFileSync("/etc/passwd", "utf8"); + if (!passwd.split("\n").some((line) => Number(line.split(":")[2]) === plan.uid)) { + if (/[:\r\n]/.test(plan.home)) throw new Error("Invalid runner home"); + fs.appendFileSync("/etc/passwd", `copilot-sdk-ci:x:${plan.uid}:${plan.gid}::${plan.home}:/bin/sh\n`); + } + const group = fs.readFileSync("/etc/group", "utf8"); + if (!group.split("\n").some((line) => Number(line.split(":")[2]) === plan.gid)) { + fs.appendFileSync("/etc/group", `copilot-sdk-ci:x:${plan.gid}:\n`); + } + if (!fs.existsSync(plan.home)) { + fs.mkdirSync(plan.home, { recursive: true, mode: 0o700 }); + } + fs.chownSync(plan.home, plan.uid, plan.gid); + const args = [ + "/usr/bin/setpriv", + `--reuid=${plan.uid}`, + `--regid=${plan.gid}`, + ...(plan.groups.length ? [`--groups=${plan.groups.join(",")}`] : ["--clear-groups"]), + plan.nodeExecutable, + fileURLToPath(import.meta.url), + "--runtime", + ]; + process.execve("/usr/bin/setpriv", args, process.env); + throw new Error("Identity transition unexpectedly returned"); +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + try { + start(); + } catch (error) { + console.error(error.message); + process.exitCode = 1; + } +} diff --git a/scripts/ci/device-policy-fixture.mjs b/scripts/ci/device-policy-fixture.mjs new file mode 100644 index 0000000000..9a082a16f3 --- /dev/null +++ b/scripts/ci/device-policy-fixture.mjs @@ -0,0 +1,466 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import { spawn, spawnSync } from "node:child_process"; +import { randomUUID } from "node:crypto"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const SDK_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); +export const IMAGE = "node:22-bookworm"; +export const POLICY_FILE = "/etc/github-copilot/managed-settings.json"; +export const OWNER_LABEL = "com.github.copilot-sdk.device-policy"; +export const CLIENT_LABEL = `${OWNER_LABEL}.client`; +export const REQUIRED_CASES = [ + { + file: "managed_plugin_progress.e2e.test.ts", + title: "emits presentation-neutral completion after installing required plugins", + }, + { + file: "rpc_server.e2e.test.ts", + title: "should round trip sessionless managed settings", + }, +]; + +const escapeRegex = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +export const REQUIRED_PATTERN = REQUIRED_CASES.map(({ title }) => escapeRegex(title)).join("|"); +export const PORTABLE_PATTERN = `^(?!.*(?:${REQUIRED_PATTERN})).*$`; + +export function portableTestPattern(source) { + if (source === "checkout") return ""; + if (source !== "published") throw new Error("Unknown runtime artifact source"); + return PORTABLE_PATTERN; +} + +export function runtimeInvocation(kind, runtime, nodeExecutable, args, env) { + if (!["native", "legacy"].includes(kind)) throw new Error("Unknown device-policy entry point"); + return { + executable: kind === "native" ? runtime : nodeExecutable, + argv: kind === "native" ? [runtime, ...args] : [nodeExecutable, runtime, ...args], + env: { ...env }, + }; +} + +export function verifyRequiredResults(report) { + for (const expected of REQUIRED_CASES) { + const matches = (report.testResults ?? []) + .filter((suite) => suite.name.replaceAll("\\", "/").endsWith(`/${expected.file}`)) + .flatMap((suite) => suite.assertionResults ?? []) + .filter((test) => test.title === expected.title); + if (matches.length !== 1 || matches[0].status !== "passed") { + throw new Error(`Required device-policy control did not pass: ${expected.title}`); + } + } +} + +function inside(parent, child) { + const relative = path.posix.relative(parent, child); + return relative === "" || (relative !== ".." && !relative.startsWith("../") && !path.posix.isAbsolute(relative)); +} + +function safeBind(directory) { + if (!path.posix.isAbsolute(directory) || /[:,\r\n]/.test(directory)) { + throw new Error("Device-policy binds require unambiguous absolute Linux paths"); + } + if ( + ["/", "/etc", "/proc", "/sys", "/dev"].some((root) => + root === "/" ? directory === root : inside(root, directory), + ) + ) { + throw new Error("A device-policy container cannot bind host system or policy paths"); + } + return directory; +} + +export function fixtureDirectories({ cwd, env, temporaryDirectory }) { + const candidates = [ + cwd, + ...["COPILOT_HOME", "GH_CONFIG_DIR", "XDG_CONFIG_HOME", "XDG_STATE_HOME"] + .map((name) => env[name]) + .filter(Boolean), + ]; + const roots = new Set(); + for (const candidate of candidates) { + const relative = path.posix.relative(temporaryDirectory, candidate); + const first = relative.split("/")[0]; + if (!inside(temporaryDirectory, candidate) || !/^copilot-test-(work|home|config)-[^/]+$/.test(first)) { + throw new Error("Writable device-policy binds must belong to an SDK E2E fixture"); + } + roots.add(safeBind(path.posix.join(temporaryDirectory, first))); + } + return [...roots]; +} + +export function dockerArguments(plan, cidfile) { + const args = [ + "run", + "--rm", + "--init", + "--interactive", + "--network", + "host", + "--cidfile", + cidfile, + "--label", + `${OWNER_LABEL}=${plan.owner}`, + "--label", + `${CLIENT_LABEL}=${plan.client}`, + "--workdir", + plan.cwd, + "--env", + "COPILOT_CI_DEVICE_POLICY_PLAN", + ]; + const mounts = new Map(); + for (const directory of plan.readonly) mounts.set(safeBind(directory), "ro"); + for (const directory of plan.writable) { + safeBind(directory); + if (mounts.has(directory)) throw new Error("A device-policy bind cannot be both read-only and writable"); + mounts.set(directory, "rw"); + } + for (const [directory, mode] of mounts) args.push("--volume", `${directory}:${directory}:${mode}`); + args.push(IMAGE, "node", path.posix.join(plan.sdkRoot, "scripts/ci/device-policy-bootstrap.mjs")); + return args; +} + +export function verifyIsolation({ originalNamespace, namespace, mountinfo, uid, gid }) { + if (uid !== 0 || gid !== 0 || namespace === originalNamespace || !namespace || !originalNamespace) { + throw new Error("Device policy must be provisioned inside a separate root container"); + } + const mounts = mountinfo + .trim() + .split("\n") + .map((line) => { + const fields = line.split(" "); + const separator = fields.indexOf("-"); + return { target: fields[4], type: fields[separator + 1], propagation: fields.slice(6, separator) }; + }); + const policyMount = mounts + .filter(({ target }) => inside(target, POLICY_FILE)) + .sort((a, b) => b.target.length - a.target.length)[0]; + if ( + !policyMount || + policyMount.target !== "/" || + policyMount.type !== "overlay" || + policyMount.propagation.some((field) => /^(shared|master|propagate_from):/.test(field)) + ) { + throw new Error("Device policy requires a private container root, not a host bind"); + } +} + +export function verifyIdentity(plan, { uid, gid, groups, policy }) { + if ( + uid !== plan.uid || + gid !== plan.gid || + JSON.stringify([...groups].sort((a, b) => a - b)) !== JSON.stringify([...plan.groups].sort((a, b) => a - b)) + ) { + throw new Error("The device-policy runtime must retain the original runner identity"); + } + if ( + !policy.isFile || + policy.isSymbolicLink || + policy.uid !== 0 || + policy.gid !== 0 || + (policy.mode & 0o777) !== 0o644 + ) { + throw new Error("The device policy must be a regular root-owned mode-0644 file"); + } +} + +function command(command, args, options = {}) { + const result = spawnSync(command, args, { encoding: "utf8", ...options }); + if (result.error) throw result.error; + if (result.status !== 0) throw new Error(`${command} failed (${result.status}): ${result.stderr ?? ""}`); + return result.stdout; +} + +export function containerId(value) { + const id = value.trim(); + if (!/^[a-f0-9]{64}$/.test(id)) throw new Error("Invalid owned device-policy container ID"); + return id; +} + +export function verifyOwnership(label, owner) { + if (!owner || label.trim() !== owner) throw new Error("Refusing to mutate a container owned by another job"); +} + +export function pendingLaunches(entries) { + for (const entry of entries) { + if (!/^[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}\.(pending|cid)$/.test(entry)) + throw new Error("Unknown device-policy cleanup registry entry"); + const pending = entry.replace(/\.cid$/, ".pending"); + if (!entries.includes(pending)) throw new Error("Device-policy container identity has no launch receipt"); + } + return entries.filter((entry) => entry.endsWith(".pending")); +} + +function inspectOwned(id, owner) { + const inspected = spawnSync("docker", ["inspect", "--format", `{{index .Config.Labels "${OWNER_LABEL}"}}`, id], { + encoding: "utf8", + }); + if (inspected.status !== 0 && /No such (object|container)/i.test(inspected.stderr)) return false; + if (inspected.error || inspected.status !== 0) + throw inspected.error ?? new Error(`Cannot inspect device-policy container: ${inspected.stderr}`); + verifyOwnership(inspected.stdout, owner); + return true; +} + +function removeOwned(id, owner) { + if (!inspectOwned(id, owner)) return; + const removed = spawnSync("docker", ["rm", "--force", id], { encoding: "utf8" }); + if (removed.status !== 0 && /No such (object|container)/i.test(removed.stderr)) return; + if (removed.error || removed.status !== 0) + throw removed.error ?? new Error(`Cannot remove device-policy container: ${removed.stderr}`); +} + +function stopOwned(id, owner, signal) { + if (!inspectOwned(id, owner)) return; + const stopped = spawnSync("docker", ["stop", "--signal", signal, "--timeout", "5", id], { encoding: "utf8" }); + if (stopped.status !== 0 && /No such (object|container)/i.test(stopped.stderr)) return; + if (stopped.error || stopped.status !== 0) + throw stopped.error ?? new Error(`Cannot stop device-policy container: ${stopped.stderr}`); +} + +function ownedRegistry() { + const registry = process.env.COPILOT_CI_DEVICE_POLICY_REGISTRY; + if (!registry || !process.env.RUNNER_TEMP) throw new Error("Missing owned device-policy container registry"); + const info = fs.lstatSync(registry); + if ( + !inside(process.env.RUNNER_TEMP, registry) || + !path.basename(registry).startsWith("sdk-device-policy-") || + fs.realpathSync(registry) !== registry || + !info.isDirectory() || + info.isSymbolicLink() || + info.uid !== process.getuid() + ) { + throw new Error("Device-policy cidfiles must belong to this runner's temporary registry"); + } + return registry; +} + +export async function launchRuntime(kind) { + if (process.platform !== "linux" || typeof process.execve !== "function") { + throw new Error("The device-policy launcher requires Linux and Node.js 22.15 or newer"); + } + const runtime = + process.env[ + kind === "native" ? "COPILOT_CI_DEVICE_POLICY_NATIVE_PATH" : "COPILOT_CI_DEVICE_POLICY_LEGACY_PATH" + ]; + if (!runtime) throw new Error("Missing original device-policy runtime path"); + const invocation = runtimeInvocation(kind, runtime, process.execPath, process.argv.slice(2), process.env); + const originalEnv = invocation.env; + const fixture = process.env.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; + if (!fixture) { + process.execve(invocation.executable, invocation.argv, originalEnv); + throw new Error("Original runtime exec unexpectedly returned"); + } + + const fixtureInfo = fs.lstatSync(fixture); + if ( + !inside(process.cwd(), fixture) || + fs.realpathSync(fixture) !== fixture || + !fixtureInfo.isFile() || + fixtureInfo.isSymbolicLink() || + fixtureInfo.uid !== process.getuid() + ) + throw new Error("Device policy must be a regular file inside the test workspace"); + const policy = fs.readFileSync(fixture, "utf8"); + const parsed = JSON.parse(policy); + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) + throw new Error("Device policy must be a JSON object"); + // The launcher consumes the test-only path; the runtime must discover the production file. + delete originalEnv.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; + const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; + if (!owner) throw new Error("Missing device-policy job ownership"); + const registry = ownedRegistry(); + const writable = fixtureDirectories({ cwd: process.cwd(), env: originalEnv, temporaryDirectory: os.tmpdir() }); + for (const directory of writable) { + const info = fs.lstatSync(directory); + if (fs.realpathSync(directory) !== directory || !info.isDirectory() || info.uid !== process.getuid()) + throw new Error("A writable fixture bind must be a real directory owned by the runner"); + } + const readonly = [ + SDK_ROOT, + path.dirname(process.env.COPILOT_CI_DEVICE_POLICY_LEGACY_PATH), + path.dirname(process.execPath), + ]; + for (const name of [ + "NODE_EXTRA_CA_CERTS", + "SSL_CERT_FILE", + "REQUESTS_CA_BUNDLE", + "CURL_CA_BUNDLE", + "GIT_SSL_CAINFO", + ]) { + const certificate = originalEnv[name]; + if (certificate) { + if (!inside(os.tmpdir(), certificate) || !fs.statSync(certificate).isFile()) + throw new Error("Replay certificates must belong to the test temporary directory"); + readonly.push(certificate); + } + } + const plan = { + owner, + client: randomUUID(), + sdkRoot: SDK_ROOT, + readonly: [...new Set(readonly)], + writable, + policy, + ...invocation, + cwd: process.cwd(), + nodeExecutable: process.execPath, + uid: process.getuid(), + gid: process.getgid(), + groups: process.getgroups(), + home: os.homedir(), + originalNamespace: fs.readlinkSync("/proc/self/ns/mnt"), + }; + const cidfile = path.join(registry, `${plan.client}.cid`); + const pending = path.join(registry, `${plan.client}.pending`); + fs.writeFileSync(pending, "", { flag: "wx", mode: 0o600 }); + const child = spawn("docker", dockerArguments(plan, cidfile), { + stdio: "inherit", + env: { ...originalEnv, COPILOT_CI_DEVICE_POLICY_PLAN: JSON.stringify(plan) }, + }); + let stopping; + const stop = (signal) => { + if (stopping) return; + stopping = signal; + try { + if (fs.existsSync(cidfile)) stopOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner, signal); + else child.kill(signal); + } catch (error) { + console.error(error.message); + process.exitCode = 1; + child.kill("SIGTERM"); + } + }; + const onTerm = () => stop("SIGTERM"); + const onInt = () => stop("SIGINT"); + process.once("SIGTERM", onTerm); + process.once("SIGINT", onInt); + let result; + try { + result = await new Promise((resolve, reject) => { + child.once("error", reject); + child.once("close", (code, signal) => resolve({ code, signal })); + }); + } finally { + process.removeListener("SIGTERM", onTerm); + process.removeListener("SIGINT", onInt); + // A cancelled Docker attach can close before its cidfile is written. + const ids = command("docker", [ + "ps", + "--all", + "--quiet", + "--no-trunc", + "--filter", + `label=${OWNER_LABEL}=${owner}`, + "--filter", + `label=${CLIENT_LABEL}=${plan.client}`, + ]).trim(); + for (const id of ids ? ids.split("\n") : []) removeOwned(containerId(id), owner); + if (fs.existsSync(cidfile)) { + removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); + fs.unlinkSync(cidfile); + fs.unlinkSync(pending); + } else { + throw new Error("Docker launch has no container identity; cleanup cannot prove settlement"); + } + } + if (process.exitCode) return; + if (stopping || result.signal) process.kill(process.pid, stopping ?? result.signal); + else process.exitCode = result.code ?? 1; +} + +function prepare() { + if ( + process.platform !== "linux" || + !process.env.GITHUB_ENV || + !process.env.GITHUB_OUTPUT || + !process.env.RUNNER_TEMP + ) { + throw new Error("Device-policy containers can only be prepared in Linux CI"); + } + const native = fs.realpathSync(process.env.COPILOT_CLI_PATH); + const legacy = fs.realpathSync(process.env.COPILOT_LEGACY_CLI_PATH); + const packageRoot = path.dirname(legacy); + if ( + native !== path.join(packageRoot, "prebuilds/linux-x64/copilot-runtime") || + path.basename(legacy) !== "app.js" + ) { + throw new Error("Device-policy fixtures require the staged GNU package's two original entry points"); + } + const expected = JSON.parse(fs.readFileSync(path.join(SDK_ROOT, "nodejs/package.json"), "utf8")).copilotCliVersion; + const actual = JSON.parse(fs.readFileSync(path.join(packageRoot, "package.json"), "utf8")).version; + if (expected !== actual) throw new Error("Device-policy package differs from the pinned CLI version"); + if (fs.existsSync(POLICY_FILE)) throw new Error("The CI host already has a device policy"); + command("docker", ["pull", IMAGE], { stdio: "inherit" }); + const registry = fs.mkdtempSync(path.join(process.env.RUNNER_TEMP, "sdk-device-policy-")); + const owner = randomUUID(); + fs.appendFileSync( + process.env.GITHUB_ENV, + `COPILOT_CI_DEVICE_POLICY_NATIVE_PATH=${native}\nCOPILOT_CI_DEVICE_POLICY_LEGACY_PATH=${legacy}\nCOPILOT_CI_DEVICE_POLICY_OWNER=${owner}\nCOPILOT_CI_DEVICE_POLICY_REGISTRY=${registry}\n`, + ); + fs.appendFileSync( + process.env.GITHUB_OUTPUT, + `native-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-native.js")}\nlegacy-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-legacy.js")}\n`, + ); +} + +function cleanup() { + const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; + if (!owner) throw new Error("Missing device-policy job ownership"); + const ids = command("docker", [ + "ps", + "--all", + "--quiet", + "--no-trunc", + "--filter", + `label=${OWNER_LABEL}=${owner}`, + ]).trim(); + const errors = []; + for (const id of ids ? ids.split("\n") : []) { + try { + removeOwned(containerId(id), owner); + } catch (error) { + errors.push(error.message); + } + } + const registry = ownedRegistry(); + for (const entry of pendingLaunches(fs.readdirSync(registry))) { + const pending = path.join(registry, entry); + const cidfile = path.join(registry, entry.replace(/\.pending$/, ".cid")); + try { + if (!fs.existsSync(cidfile)) throw new Error("Unsettled Docker launch has no container identity"); + removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); + fs.unlinkSync(cidfile); + fs.unlinkSync(pending); + } catch (error) { + errors.push(error.message); + } + } + if (errors.length) throw new Error(`Device-policy cleanup did not settle: ${errors.join("; ")}`); + if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + try { + const [action, argument] = process.argv.slice(2); + if (action === "prepare") prepare(); + else if (action === "cleanup") cleanup(); + else if (action === "portable-pattern") console.log(portableTestPattern(argument)); + else if (action === "required-pattern") console.log(REQUIRED_PATTERN); + else if (action === "verify" && argument) { + verifyRequiredResults(JSON.parse(fs.readFileSync(argument, "utf8"))); + if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); + } else + throw new Error( + "Usage: device-policy-fixture.mjs ", + ); + } catch (error) { + console.error(error.message); + process.exitCode = 1; + } +} diff --git a/scripts/ci/device-policy-fixture.test.mjs b/scripts/ci/device-policy-fixture.test.mjs new file mode 100644 index 0000000000..8f33a412c3 --- /dev/null +++ b/scripts/ci/device-policy-fixture.test.mjs @@ -0,0 +1,207 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import assert from "node:assert/strict"; +import test from "node:test"; +import "./device-policy-bootstrap.mjs"; +import { + CLIENT_LABEL, + containerId, + dockerArguments, + fixtureDirectories, + OWNER_LABEL, + pendingLaunches, + POLICY_FILE, + portableTestPattern, + runtimeInvocation, + PORTABLE_PATTERN, + REQUIRED_CASES, + REQUIRED_PATTERN, + verifyIdentity, + verifyIsolation, + verifyOwnership, + verifyRequiredResults, +} from "./device-policy-fixture.mjs"; + +const report = () => ({ + testResults: REQUIRED_CASES.map(({ file, title }) => ({ + name: `/workspace/nodejs/test/e2e/${file}`, + assertionResults: [{ title, status: "passed" }], + })), +}); + +test("requires both unchanged strict controls to pass", () => { + verifyRequiredResults(report()); + for (const index of [0, 1]) { + for (const status of ["pending", "skipped", "failed", "todo"]) { + const result = report(); + result.testResults[index].assertionResults[0].status = status; + assert.throws(() => verifyRequiredResults(result), /did not pass/); + } + const missing = report(); + missing.testResults.splice(index, 1); + assert.throws(() => verifyRequiredResults(missing), /did not pass/); + const duplicate = report(); + duplicate.testResults.push(duplicate.testResults[index]); + assert.throws(() => verifyRequiredResults(duplicate), /did not pass/); + } + assert.throws(() => verifyRequiredResults({}), /did not pass/); +}); + +test("assigns only the two device-fixture cases to the mandatory Linux gate", () => { + assert.equal(portableTestPattern("checkout"), ""); + assert.equal(portableTestPattern("published"), PORTABLE_PATTERN); + assert.throws(() => portableTestPattern(""), /Unknown/); + const portable = new RegExp(PORTABLE_PATTERN); + const required = new RegExp(REQUIRED_PATTERN); + for (const { title } of REQUIRED_CASES) { + assert.equal(portable.test(`Suite ${title}`), false); + assert.equal(required.test(title), true); + } + for (const title of [ + "should clear the managed settings cache", + "should expose the managed settings schema", + "should list server sessions", + ]) { + assert.equal(portable.test(title), true); + assert.equal(required.test(title), false); + } +}); + +test("preserves both runtime entry points' original argv and environment", () => { + const args = ["--stdio", "--argument-with-spaces=one two", "--literal=;$()"]; + const env = { TOKEN: "private-value", COPILOT_HOME: "/tmp/copilot-test-home-one" }; + const native = runtimeInvocation("native", "/package/copilot-runtime", "/node/bin/node", args, env); + assert.equal(native.executable, "/package/copilot-runtime"); + assert.deepEqual(native.argv, ["/package/copilot-runtime", ...args]); + const legacy = runtimeInvocation("legacy", "/package/app.js", "/node/bin/node", args, env); + assert.equal(legacy.executable, "/node/bin/node"); + assert.deepEqual(legacy.argv, ["/node/bin/node", "/package/app.js", ...args]); + assert.deepEqual(native.env, env); + assert.notEqual(native.env, env); + assert.throws(() => runtimeInvocation("unknown", "", "", [], {}), /entry point/); +}); + +test("binds only explicitly owned E2E fixture directories for writing", () => { + const roots = fixtureDirectories({ + cwd: "/tmp/copilot-test-work-one", + env: { + COPILOT_HOME: "/tmp/copilot-test-home-one", + GH_CONFIG_DIR: "/tmp/copilot-test-config-one", + XDG_CONFIG_HOME: "/tmp/copilot-test-config-one", + XDG_STATE_HOME: "/tmp/copilot-test-work-one/local-home", + }, + temporaryDirectory: "/tmp", + }); + assert.deepEqual(roots, [ + "/tmp/copilot-test-work-one", + "/tmp/copilot-test-home-one", + "/tmp/copilot-test-config-one", + ]); + for (const cwd of ["/etc", "/home/runner", "/tmp", "/tmp/../../etc", "/tmp/not-a-fixture"]) { + assert.throws(() => fixtureDirectories({ cwd, env: {}, temporaryDirectory: "/tmp" }), /fixture/); + } +}); + +test("keeps policy and secret environment contents out of Docker arguments", () => { + const plan = { + owner: "job-one", + client: "client-one", + sdkRoot: "/workspace", + cwd: "/tmp/copilot-test-work-one", + readonly: ["/workspace", "/artifacts/package"], + writable: ["/tmp/copilot-test-work-one"], + env: { SECRET: "must-not-appear" }, + policy: '{"model":"secret-model"}', + }; + const args = dockerArguments(plan, "/registry/client-one.cid"); + assert.deepEqual(args.slice(0, 7), ["run", "--rm", "--init", "--interactive", "--network", "host", "--cidfile"]); + assert.equal(args.includes(`${OWNER_LABEL}=job-one`), true); + assert.equal(args.includes(`${CLIENT_LABEL}=client-one`), true); + assert.equal(args.includes("COPILOT_CI_DEVICE_POLICY_PLAN"), true); + assert.equal(args.includes("/artifacts/package:/artifacts/package:ro"), true); + assert.equal(args.includes("/tmp/copilot-test-work-one:/tmp/copilot-test-work-one:rw"), true); + assert.equal(args.join(" ").includes("must-not-appear"), false); + assert.equal(args.join(" ").includes("secret-model"), false); + const other = dockerArguments({ ...plan, client: "client-two" }, "/registry/client-two.cid"); + assert.notDeepEqual(args, other); + for (const directory of ["/", "/etc", "/etc/github-copilot", "/proc", "/sys", "/dev", "/path:ambiguous"]) { + assert.throws(() => dockerArguments({ ...plan, readonly: [directory] }, "/registry/client.cid")); + } + assert.throws(() => dockerArguments({ ...plan, readonly: plan.writable }, "/registry/client.cid"), /both/); +}); + +test("requires exact job ownership and full container IDs before cleanup", () => { + verifyOwnership("job-one\n", "job-one"); + for (const [label, owner] of [ + ["job-two", "job-one"], + ["", ""], + ["", "job-one"], + ]) { + assert.throws(() => verifyOwnership(label, owner), /another job/); + } + assert.equal(containerId(`${"a".repeat(64)}\n`), "a".repeat(64)); + for (const id of ["", "a".repeat(12), `${"a".repeat(64)} extra`, "G".repeat(64)]) { + assert.throws(() => containerId(id), /container ID/); + } +}); + +test("exposes interrupted or unknown launch receipts instead of claiming cleanup", () => { + const client = "12345678-1234-1234-1234-123456789abc"; + assert.deepEqual(pendingLaunches([]), []); + assert.deepEqual(pendingLaunches([`${client}.pending`, `${client}.cid`]), [`${client}.pending`]); + assert.deepEqual(pendingLaunches([`${client}.pending`]), [`${client}.pending`]); + assert.throws(() => pendingLaunches([`${client}.cid`]), /no launch receipt/); + assert.throws(() => pendingLaunches(["unknown"]), /Unknown/); +}); + +const isolated = () => ({ + originalNamespace: "mnt:[100]", + namespace: "mnt:[200]", + uid: 0, + gid: 0, + mountinfo: "1 0 0:1 / / rw - overlay overlay rw\n2 1 0:2 / /workspace ro - ext4 disk ro", +}); + +test("requires private container backing before any policy mutation", () => { + verifyIsolation(isolated()); + for (const change of [ + { namespace: "mnt:[100]" }, + { uid: 1001 }, + { gid: 1001 }, + { originalNamespace: "" }, + { mountinfo: "1 0 0:1 / / rw shared:1 - overlay overlay rw" }, + { mountinfo: "1 0 0:1 / / rw - ext4 disk rw" }, + { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc rw - ext4 host rw` }, + { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc/github-copilot rw - ext4 host rw` }, + ]) + assert.throws(() => verifyIsolation({ ...isolated(), ...change })); +}); + +test("retains runner uid, gid and groups with an ordinary root-owned device file", () => { + const plan = { uid: 1001, gid: 1001, groups: [1001, 118] }; + const runtime = { + uid: 1001, + gid: 1001, + groups: [118, 1001], + policy: { isFile: true, isSymbolicLink: false, uid: 0, gid: 0, mode: 0o100644 }, + }; + verifyIdentity(plan, runtime); + for (const change of [{ uid: 0 }, { gid: 0 }, { groups: [] }]) { + assert.throws(() => verifyIdentity(plan, { ...runtime, ...change }), /identity/); + } + for (const change of [ + { uid: 1001 }, + { gid: 1001 }, + { mode: 0o100666 }, + { isSymbolicLink: true }, + { isFile: false }, + ]) { + assert.throws( + () => verifyIdentity(plan, { ...runtime, policy: { ...runtime.policy, ...change } }), + /root-owned/, + ); + } + assert.equal(POLICY_FILE, "/etc/github-copilot/managed-settings.json"); +}); diff --git a/scripts/ci/device-policy-legacy.js b/scripts/ci/device-policy-legacy.js new file mode 100644 index 0000000000..806647cf11 --- /dev/null +++ b/scripts/ci/device-policy-legacy.js @@ -0,0 +1,10 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import("./device-policy-fixture.mjs") + .then(({ launchRuntime }) => launchRuntime("legacy")) + .catch((error) => { + console.error(error.message); + process.exitCode = 1; + }); diff --git a/scripts/ci/device-policy-native.js b/scripts/ci/device-policy-native.js new file mode 100644 index 0000000000..d269a07ea6 --- /dev/null +++ b/scripts/ci/device-policy-native.js @@ -0,0 +1,10 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import("./device-policy-fixture.mjs") + .then(({ launchRuntime }) => launchRuntime("native")) + .catch((error) => { + console.error(error.message); + process.exitCode = 1; + }); From 656affd3f065ffd75dee02f5f19a0b04afa9e1bb Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 13:50:05 -0700 Subject: [PATCH 04/15] fix(ci): retain main push-event test reporting Reconcile the device-fixture test selection with canonical main's existing reporting condition without changing runtime pins or assertions. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/workflows/sdk-nodejs.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index eba1d72a4a..3130aabbea 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -113,7 +113,7 @@ jobs: echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" test_args+=(--testNamePattern "$pattern") fi - if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ]; then + if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ] || [ "$GITHUB_EVENT_NAME" = "push" ]; then npm test -- "${test_args[@]}" --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" else npm test -- "${test_args[@]}" From af88ba0cfb061c9a6e112b8d2c799fd26aeafcf9 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 13:50:50 -0700 Subject: [PATCH 05/15] fix(ci): reconcile main Node workflow reporting Retain main's existing runner and cross-platform report upload configuration while preserving explicit device-fixture selection and the strict Linux CAPI result guard. No runtime pin or assertion changes. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/workflows/sdk-nodejs.yml | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index 3130aabbea..5b506d5d90 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -39,7 +39,7 @@ jobs: strategy: fail-fast: false matrix: - os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26","windows-latest"]') }} + os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} runs-on: ${{ matrix.os }} defaults: run: @@ -109,10 +109,13 @@ jobs: run: | test_args=() pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - if [[ -n "$pattern" ]]; then - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - test_args+=(--testNamePattern "$pattern") - fi + case "$pattern" in + "") ;; + *) + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + test_args+=(--testNamePattern "$pattern") + ;; + esac if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ] || [ "$GITHUB_EVENT_NAME" = "push" ]; then npm test -- "${test_args[@]}" --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" else @@ -136,11 +139,11 @@ jobs: working-directory: . run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" cleanup - name: Upload Flake Finder Node.js test results - if: always() && (github.event_name == 'merge_group' || github.event_name == 'pull_request') && runner.os == 'Linux' + if: always() && (github.event_name == 'merge_group' || github.event_name == 'pull_request' || github.event_name == 'push') continue-on-error: true uses: actions/upload-artifact@v7 with: - name: suspected-flakes-sdk-nodejs-${{ github.run_attempt }}-linux + name: suspected-flakes-sdk-nodejs-${{ github.run_attempt }}-${{ runner.os == 'macOS' && 'macos' || runner.os == 'Windows' && 'windows' || 'linux' }} path: ${{ runner.temp }}/sdk-nodejs-results.json if-no-files-found: warn retention-days: 7 From 16b0fdaef9b0a62d1622d83ec01261d968499305 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 14:22:30 -0700 Subject: [PATCH 06/15] fix(ci): validate and read policy through one file descriptor Open the owned fixture without following a final symlink and validate the same descriptor used for reading, eliminating the lstat-to-path-read race. Pure launch-plan controls remain separate from actual filesystem and runtime proof. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- scripts/ci/device-policy-fixture.mjs | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/scripts/ci/device-policy-fixture.mjs b/scripts/ci/device-policy-fixture.mjs index 9a082a16f3..7c65a71354 100644 --- a/scripts/ci/device-policy-fixture.mjs +++ b/scripts/ci/device-policy-fixture.mjs @@ -257,16 +257,18 @@ export async function launchRuntime(kind) { throw new Error("Original runtime exec unexpectedly returned"); } - const fixtureInfo = fs.lstatSync(fixture); - if ( - !inside(process.cwd(), fixture) || - fs.realpathSync(fixture) !== fixture || - !fixtureInfo.isFile() || - fixtureInfo.isSymbolicLink() || - fixtureInfo.uid !== process.getuid() - ) + if (!inside(process.cwd(), fixture) || fs.realpathSync(fixture) !== fixture) throw new Error("Device policy must be a regular file inside the test workspace"); - const policy = fs.readFileSync(fixture, "utf8"); + const descriptor = fs.openSync(fixture, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW); + let policy; + try { + const fixtureInfo = fs.fstatSync(descriptor); + if (!fixtureInfo.isFile() || fixtureInfo.uid !== process.getuid()) + throw new Error("Device policy must be a regular file owned by the runner"); + policy = fs.readFileSync(descriptor, "utf8"); + } finally { + fs.closeSync(descriptor); + } const parsed = JSON.parse(policy); if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) throw new Error("Device policy must be a JSON object"); From 5b3d50bc3f9ecbdb9cb56bf89c865bc20802315f Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 15:02:11 -0700 Subject: [PATCH 07/15] test: report Assisted contract failure boundaries Log only synthetic phase timings and provider call/error counts when the existing contract fails. Preserve all scenario assertions, selections, timeouts, and runtime behavior so automatic Windows CI can identify its earliest stalled boundary. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../assisted_autopilot_permission.e2e.test.ts | 53 ++++++++++++++++++- 1 file changed, 52 insertions(+), 1 deletion(-) diff --git a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts index 5577d03761..c1c0e0bb18 100644 --- a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts +++ b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts @@ -8,7 +8,7 @@ import { mkdir, realpath, rm, writeFile } from "node:fs/promises"; import { createServer } from "node:http"; import { join } from "node:path"; import { text } from "node:stream/consumers"; -import { describe, expect, it, onTestFinished } from "vitest"; +import { describe, expect, it, onTestFailed, onTestFinished } from "vitest"; import { createAttributedPermissionResult, type CopilotSession, @@ -50,6 +50,30 @@ describe("Assisted permission handling in Autopilot", async () => { const judgeOutputs: string[] = []; const authorizationJudgeRequests: string[] = []; const providerFailures: Error[] = []; + const started = Date.now(); + let phase = "setup"; + let phaseStarted = started; + const completedPhases: Array<{ phase: string; elapsedMs: number }> = []; + const enterPhase = (next: string) => { + const now = Date.now(); + completedPhases.push({ phase, elapsedMs: now - phaseStarted }); + phase = next; + phaseStarted = now; + }; + onTestFailed(() => { + console.error( + "Assisted contract failure boundary:", + JSON.stringify({ + phase, + phaseElapsedMs: Date.now() - phaseStarted, + elapsedMs: Date.now() - started, + completedPhases, + agentCalls, + judgeCalls: judgeOutputs.length, + providerFailures: providerFailures.length, + }) + ); + }); const outsideDir = `${workDir}-outside`; await mkdir(outsideDir, { recursive: true }); await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); @@ -454,6 +478,7 @@ describe("Assisted permission handling in Autopilot", async () => { { route: "human", lifecycle: "resume", decision: "deny" }, ] as const; for (const scenario of scenarios) { + const label = `initial:${scenario.route}:${scenario.lifecycle}:${scenario.decision}`; let permissionCallbacks = 0; const expectedRecommendation = scenario.decision === "judge" ? "approve" : "requireApproval"; @@ -527,19 +552,26 @@ describe("Assisted permission handling in Autopilot", async () => { } as const; let session: CopilotSession | undefined; try { + enterPhase(`${label}:create`); session = await client.createSession(sessionConfig); if (scenario.lifecycle === "resume") { const sessionId = session.sessionId; + enterPhase(`${label}:prime`); await session.sendAndWait({ prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", }); + enterPhase(`${label}:configure-before-resume`); await configureAssistedAutopilot(session, workDir); await expectAssistedAutopilotConfigured(session, workDir); + enterPhase(`${label}:disconnect-before-resume`); await session.disconnect(); session = undefined; + enterPhase(`${label}:resume`); session = await client.resumeSession(sessionId, sessionConfig); + enterPhase(`${label}:verify-resume`); await expectAssistedAutopilotConfigured(session, workDir); } else { + enterPhase(`${label}:configure`); await configureAssistedAutopilot(session, workDir); } session.on((event) => { @@ -584,6 +616,7 @@ describe("Assisted permission handling in Autopilot", async () => { try { const taskComplete = getNextEventOfType(session, "session.task_complete"); + enterPhase(`${label}:send`); await session.send({ prompt: scenario.route === "shell" @@ -592,11 +625,13 @@ describe("Assisted permission handling in Autopilot", async () => { ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, }); + enterPhase(`${label}:task-complete`); await taskComplete; } catch (error) { throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); } + enterPhase(`${label}:assert`); const expectedCompletions = scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); @@ -646,6 +681,7 @@ describe("Assisted permission handling in Autopilot", async () => { } } finally { if (session) { + enterPhase(`${label}:cleanup`); await session.abort(); await session.disconnect(); } @@ -672,6 +708,7 @@ describe("Assisted permission handling in Autopilot", async () => { }); let authorizationSession: CopilotSession | undefined; try { + enterPhase("authorization:create"); authorizationSession = await client.createSession({ model: "local/gpt-5.6-sol", providers, @@ -711,9 +748,11 @@ describe("Assisted permission handling in Autopilot", async () => { ); }, }); + enterPhase("authorization:prime"); await authorizationSession.sendAndWait({ prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, }); + enterPhase("authorization:configure"); await configureAssistedAutopilot(authorizationSession, workDir); authorizationSession.on((event) => { if (event.type === "permission.requested") { @@ -744,11 +783,14 @@ describe("Assisted permission handling in Autopilot", async () => { }); const taskComplete = getNextEventOfType(authorizationSession, "session.task_complete"); + enterPhase("authorization:send"); await authorizationSession.send({ prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, }); + enterPhase("authorization:task-complete"); await taskComplete; + enterPhase("authorization:assert"); expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); expect(authorizationJudgeRequests).toHaveLength(1); const authorizationRequest = authorizationJudgeRequests[0] ?? ""; @@ -775,6 +817,7 @@ describe("Assisted permission handling in Autopilot", async () => { expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); } finally { if (authorizationSession) { + enterPhase("authorization:cleanup"); await authorizationSession.abort(); await authorizationSession.disconnect(); } @@ -794,6 +837,7 @@ describe("Assisted permission handling in Autopilot", async () => { }[]; const contractJudgeStart = judgeOutputs.length; for (const scenario of contractScenarios) { + const label = `contract:${scenario.scope}:${scenario.recommendation}:${scenario.handler}`; let permissionCallbacks = 0; const recommendations: string[] = []; const recoveryStatuses: string[] = []; @@ -865,7 +909,9 @@ describe("Assisted permission handling in Autopilot", async () => { } as const; let session: CopilotSession | undefined; try { + enterPhase(`${label}:create`); session = await client.createSession(sessionConfig); + enterPhase(`${label}:configure`); await configureAssistedAutopilot(session, workDir); session.on((event) => { if (event.type === "permission.requested") { @@ -914,12 +960,15 @@ describe("Assisted permission handling in Autopilot", async () => { : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; try { const taskComplete = getNextEventOfType(session, "session.task_complete"); + enterPhase(`${label}:send`); await session.send({ prompt }); + enterPhase(`${label}:task-complete`); await taskComplete; } catch (error) { throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); } + enterPhase(`${label}:assert`); if (scenario.handler === "none") { expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); expect(recommendations, JSON.stringify(scenario)).toEqual([]); @@ -1018,12 +1067,14 @@ describe("Assisted permission handling in Autopilot", async () => { } } finally { if (session) { + enterPhase(`${label}:cleanup`); await session.abort(); await session.disconnect(); } } } + enterPhase("final-assertions"); expect(providerFailures).toEqual([]); expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ JUDGE_OUTPUT, From 4ae385d5db505375885ef4bf6bb38098678f9080 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 15:29:36 -0700 Subject: [PATCH 08/15] test: separate complete Assisted permission contracts Give lifecycle, authorization-ordering, and handler/subagent contracts separate existing Vitest budgets. Retain every scenario assertion and fully awaited abort/disconnect; avoid changing timeouts or weakening cleanup to fit fifteen scenarios into one Windows deadline. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../assisted_autopilot_permission.e2e.test.ts | 1958 +++++++++-------- 1 file changed, 1003 insertions(+), 955 deletions(-) diff --git a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts index c1c0e0bb18..ef67382bfc 100644 --- a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts +++ b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts @@ -45,181 +45,314 @@ function contractFixtureName(scope: ContractScope, recommendation: ContractRecom describe("Assisted permission handling in Autopilot", async () => { const { copilotClient: client, workDir } = await createSdkTestContext({ useStdio: true }); - it("resolves Assisted recommendations before Autopilot permission recovery", async () => { - let agentCalls = 0; - const judgeOutputs: string[] = []; - const authorizationJudgeRequests: string[] = []; - const providerFailures: Error[] = []; - const started = Date.now(); - let phase = "setup"; - let phaseStarted = started; - const completedPhases: Array<{ phase: string; elapsedMs: number }> = []; - const enterPhase = (next: string) => { - const now = Date.now(); - completedPhases.push({ phase, elapsedMs: now - phaseStarted }); - phase = next; - phaseStarted = now; - }; - onTestFailed(() => { - console.error( - "Assisted contract failure boundary:", - JSON.stringify({ - phase, - phaseElapsedMs: Date.now() - phaseStarted, - elapsedMs: Date.now() - started, - completedPhases, - agentCalls, - judgeCalls: judgeOutputs.length, - providerFailures: providerFailures.length, - }) - ); - }); - const outsideDir = `${workDir}-outside`; - await mkdir(outsideDir, { recursive: true }); - await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); - const modelServer = createServer((request, response) => { - void (async () => { - const body = JSON.parse(await text(request)) as { - model: string; - stream?: boolean; - messages: Array<{ role?: string; content?: unknown }>; - }; - const isJudge = body.model === "gpt-6-luna"; - const messages = JSON.stringify(body.messages); - const latestUserMessage = [...body.messages] - .reverse() - .find((entry) => entry.role === "user"); - const latestUserText = JSON.stringify(latestUserMessage?.content); - let message: - | { role: "assistant"; content: string } - | { - role: "assistant"; - content: string; - tool_calls: Array<{ - id: string; - type: "function"; - function: { name: string; arguments: string }; - }>; - }; - if (isJudge) { - const earlierRestrictionIndex = messages.lastIndexOf( - EARLIER_RESTRICTION_MARKER - ); - const laterAuthorizationIndex = messages.lastIndexOf( - LATER_AUTHORIZATION_MARKER - ); - const authorizationRoute = laterAuthorizationIndex !== -1; - if (authorizationRoute) { - authorizationJudgeRequests.push(messages); - } - const contractRequiresApproval = - messages.includes(contractFixtureName("root", "requireApproval")) || - messages.includes(contractFixtureName("subagent", "requireApproval")); - const output = authorizationRoute - ? earlierRestrictionIndex !== -1 && - earlierRestrictionIndex < laterAuthorizationIndex - ? JUDGE_OUTPUT - : HUMAN_REVIEW_OUTPUT - : messages.includes("assisted-human-ask") || contractRequiresApproval - ? HUMAN_REVIEW_OUTPUT - : JUDGE_OUTPUT; - judgeOutputs.push(output); - message = { role: "assistant", content: output }; - } else { - agentCalls++; - const toolResultCount = body.messages.filter( - (entry) => entry.role === "tool" - ).length; - const contractChildRoute = latestUserText.includes("ASSISTED_CONTRACT_CHILD_"); - const contractSubagentRoute = latestUserText.includes( - "ASSISTED_CONTRACT_SUBAGENT_" - ); - const contractRootRoute = latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); - const authorizationRestrictionRoute = latestUserText.includes( - EARLIER_RESTRICTION_MARKER - ); - const authorizationAllowRoute = latestUserText.includes( - LATER_AUTHORIZATION_MARKER - ); - if (authorizationRestrictionRoute) { - message = { role: "assistant", content: "restriction-recorded" }; - } else if (authorizationAllowRoute) { - message = - toolResultCount >= 2 - ? { role: "assistant", content: "authorization-updated" } - : toolResultCount === 1 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-task-complete", - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: "authorization-updated", - }), + it.each(["lifecycle", "authorization", "handler"] as const)( + "resolves Assisted recommendations before Autopilot permission recovery (%s contract)", + async (contract) => { + let agentCalls = 0; + const judgeOutputs: string[] = []; + const authorizationJudgeRequests: string[] = []; + const providerFailures: Error[] = []; + const started = Date.now(); + let phase = "setup"; + let phaseStarted = started; + const completedPhases: Array<{ phase: string; elapsedMs: number }> = []; + const enterPhase = (next: string) => { + const now = Date.now(); + completedPhases.push({ phase, elapsedMs: now - phaseStarted }); + phase = next; + phaseStarted = now; + }; + onTestFailed(() => { + console.error( + "Assisted contract failure boundary:", + JSON.stringify({ + phase, + phaseElapsedMs: Date.now() - phaseStarted, + elapsedMs: Date.now() - started, + completedPhases, + agentCalls, + judgeCalls: judgeOutputs.length, + providerFailures: providerFailures.length, + }) + ); + }); + const outsideDir = `${workDir}-outside`; + await mkdir(outsideDir, { recursive: true }); + await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); + const modelServer = createServer((request, response) => { + void (async () => { + const body = JSON.parse(await text(request)) as { + model: string; + stream?: boolean; + messages: Array<{ role?: string; content?: unknown }>; + }; + const isJudge = body.model === "gpt-6-luna"; + const messages = JSON.stringify(body.messages); + const latestUserMessage = [...body.messages] + .reverse() + .find((entry) => entry.role === "user"); + const latestUserText = JSON.stringify(latestUserMessage?.content); + let message: + | { role: "assistant"; content: string } + | { + role: "assistant"; + content: string; + tool_calls: Array<{ + id: string; + type: "function"; + function: { name: string; arguments: string }; + }>; + }; + if (isJudge) { + const earlierRestrictionIndex = messages.lastIndexOf( + EARLIER_RESTRICTION_MARKER + ); + const laterAuthorizationIndex = messages.lastIndexOf( + LATER_AUTHORIZATION_MARKER + ); + const authorizationRoute = laterAuthorizationIndex !== -1; + if (authorizationRoute) { + authorizationJudgeRequests.push(messages); + } + const contractRequiresApproval = + messages.includes(contractFixtureName("root", "requireApproval")) || + messages.includes(contractFixtureName("subagent", "requireApproval")); + const output = authorizationRoute + ? earlierRestrictionIndex !== -1 && + earlierRestrictionIndex < laterAuthorizationIndex + ? JUDGE_OUTPUT + : HUMAN_REVIEW_OUTPUT + : messages.includes("assisted-human-ask") || contractRequiresApproval + ? HUMAN_REVIEW_OUTPUT + : JUDGE_OUTPUT; + judgeOutputs.push(output); + message = { role: "assistant", content: output }; + } else { + agentCalls++; + const toolResultCount = body.messages.filter( + (entry) => entry.role === "tool" + ).length; + const contractChildRoute = latestUserText.includes( + "ASSISTED_CONTRACT_CHILD_" + ); + const contractSubagentRoute = latestUserText.includes( + "ASSISTED_CONTRACT_SUBAGENT_" + ); + const contractRootRoute = + latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); + const authorizationRestrictionRoute = latestUserText.includes( + EARLIER_RESTRICTION_MARKER + ); + const authorizationAllowRoute = latestUserText.includes( + LATER_AUTHORIZATION_MARKER + ); + if (authorizationRestrictionRoute) { + message = { role: "assistant", content: "restriction-recorded" }; + } else if (authorizationAllowRoute) { + message = + toolResultCount >= 2 + ? { role: "assistant", content: "authorization-updated" } + : toolResultCount === 1 + ? { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-authorization-task-complete", + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: "authorization-updated", + }), + }, }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-shell", - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: - process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` - : `mkdir ${AUTHORIZATION_FIXTURE}`, - description: - "Create the fixture authorized by the latest instruction", - }), + ], + } + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-authorization-shell", + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: + process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` + : `mkdir ${AUTHORIZATION_FIXTURE}`, + description: + "Create the fixture authorized by the latest instruction", + }), + }, }, + ], + }; + } else if ( + contractChildRoute || + contractSubagentRoute || + contractRootRoute + ) { + const recommendation: ContractRecommendation = latestUserText.includes( + "_REQUIRE" + ) + ? "requireApproval" + : latestUserText.includes("_EXCLUDED") + ? "excluded" + : "approve"; + const scope: ContractScope = + contractChildRoute || contractSubagentRoute ? "subagent" : "root"; + const finalText = `contract-${scope}-${ + recommendation === "approve" ? "approved" : "blocked" + }`; + if (contractSubagentRoute) { + message = + toolResultCount === 0 + ? { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-${recommendation}`, + type: "function", + function: { + name: "task", + arguments: JSON.stringify({ + name: "assisted-contract", + description: + "Exercise an Assisted permission request", + prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( + recommendation + )}: Run the requested shell command once and report whether it completed.`, + agent_type: "task", + mode: "sync", + }), + }, + }, + ], + } + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-complete-${recommendation}`, + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: finalText, + }), + }, + }, + ], + }; + } else if (contractChildRoute && toolResultCount > 0) { + message = { role: "assistant", content: finalText }; + } else if ( + (contractChildRoute || contractRootRoute) && + (toolResultCount === 0 || + (recommendation === "excluded" && toolResultCount === 1)) + ) { + const command = + recommendation === "excluded" + ? process.platform === "win32" + ? "Get-Content approval-probe.txt | & $runner" + : "printf payload | $runner" + : process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${contractFixtureName( + scope, + recommendation + )}` + : `mkdir ${contractFixtureName(scope, recommendation)}`; + message = { + role: "assistant", + content: "", + tool_calls: [ + { + id: + toolResultCount === 0 + ? contractShellCallId(scope, recommendation) + : `${contractShellCallId( + scope, + recommendation + )}-retry`, + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command, + description: + "Exercise Assisted permission routing", + }), }, - ], - }; - } else if (contractChildRoute || contractSubagentRoute || contractRootRoute) { - const recommendation: ContractRecommendation = latestUserText.includes( - "_REQUIRE" - ) - ? "requireApproval" - : latestUserText.includes("_EXCLUDED") - ? "excluded" - : "approve"; - const scope: ContractScope = - contractChildRoute || contractSubagentRoute ? "subagent" : "root"; - const finalText = `contract-${scope}-${ - recommendation === "approve" ? "approved" : "blocked" - }`; - if (contractSubagentRoute) { - message = - toolResultCount === 0 + }, + ], + }; + } else { + message = { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-complete-${recommendation}`, + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ summary: finalText }), + }, + }, + ], + }; + } + } else { + const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); + const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); + const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); + const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); + const primeRoute = + primeIndex > Math.max(shellIndex, pathIndex, humanIndex); + const humanRoute = + humanIndex > Math.max(primeIndex, shellIndex, pathIndex); + const shellRoute = + shellIndex > Math.max(primeIndex, pathIndex, humanIndex); + const resumedShellRoute = + messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; + const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); + const humanLifecycle = + messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || + messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") + ? "resume" + : "create"; + const finalText = humanRoute + ? humanApproved + ? "human-approved" + : "human-denied" + : shellRoute + ? "shell-approved" + : "approval-probe.txt"; + message = primeRoute + ? { role: "assistant", content: "prime-ready" } + : toolResultCount >= 2 + ? { + role: "assistant", + content: finalText, + } + : toolResultCount === 1 ? { role: "assistant", content: "", tool_calls: [ { - id: `assisted-contract-task-${recommendation}`, + id: "assisted-autopilot-task-complete", type: "function", function: { - name: "task", + name: "task_complete", arguments: JSON.stringify({ - name: "assisted-contract", - description: - "Exercise an Assisted permission request", - prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( - recommendation - )}: Run the requested shell command once and report whether it completed.`, - agent_type: "task", - mode: "sync", + summary: finalText, }), }, }, @@ -230,860 +363,775 @@ describe("Assisted permission handling in Autopilot", async () => { content: "", tool_calls: [ { - id: `assisted-contract-task-complete-${recommendation}`, + id: shellRoute + ? "assisted-autopilot-shell" + : humanRoute + ? "assisted-autopilot-human" + : "assisted-autopilot-glob", type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: finalText, - }), - }, + function: + shellRoute || humanRoute + ? { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: humanRoute + ? process.platform === + "win32" + ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : process.platform === + "win32" + ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` + : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, + description: humanRoute + ? "Create the human-reviewed SDK permission fixture" + : "Create the authorized SDK permission fixture", + }), + } + : { + name: "glob", + arguments: JSON.stringify({ + pattern: "*.txt", + path: outsideDir, + }), + }, }, ], }; - } else if (contractChildRoute && toolResultCount > 0) { - message = { role: "assistant", content: finalText }; - } else if ( - (contractChildRoute || contractRootRoute) && - (toolResultCount === 0 || - (recommendation === "excluded" && toolResultCount === 1)) - ) { - const command = - recommendation === "excluded" - ? process.platform === "win32" - ? "Get-Content approval-probe.txt | & $runner" - : "printf payload | $runner" - : process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${contractFixtureName( - scope, - recommendation - )}` - : `mkdir ${contractFixtureName(scope, recommendation)}`; - message = { - role: "assistant", - content: "", - tool_calls: [ - { - id: - toolResultCount === 0 - ? contractShellCallId(scope, recommendation) - : `${contractShellCallId( - scope, - recommendation - )}-retry`, - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command, - description: "Exercise Assisted permission routing", - }), - }, - }, - ], - }; - } else { - message = { - role: "assistant", - content: "", - tool_calls: [ + } + } + + const choice = { + index: 0, + message, + finish_reason: "tool_calls" in message ? "tool_calls" : "stop", + }; + const completion = { + id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, + object: "chat.completion", + created: 1, + model: body.model, + choices: [choice], + usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, + }; + if (body.stream) { + response.writeHead(200, { "content-type": "text/event-stream" }); + response.end( + `data: ${JSON.stringify({ + ...completion, + object: "chat.completion.chunk", + choices: [ { - id: `assisted-contract-task-complete-${recommendation}`, - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ summary: finalText }), + index: 0, + delta: { + ...message, + ...("tool_calls" in message + ? { + tool_calls: message.tool_calls.map( + (call, index) => ({ index, ...call }) + ), + } + : {}), }, + finish_reason: choice.finish_reason, }, ], - }; - } + })}\n\ndata: [DONE]\n\n` + ); } else { - const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); - const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); - const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); - const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); - const primeRoute = primeIndex > Math.max(shellIndex, pathIndex, humanIndex); - const humanRoute = humanIndex > Math.max(primeIndex, shellIndex, pathIndex); - const shellRoute = shellIndex > Math.max(primeIndex, pathIndex, humanIndex); - const resumedShellRoute = - messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; - const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); - const humanLifecycle = - messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || - messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") - ? "resume" - : "create"; - const finalText = humanRoute - ? humanApproved - ? "human-approved" - : "human-denied" - : shellRoute - ? "shell-approved" - : "approval-probe.txt"; - message = primeRoute - ? { role: "assistant", content: "prime-ready" } - : toolResultCount >= 2 - ? { - role: "assistant", - content: finalText, - } - : toolResultCount === 1 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-autopilot-task-complete", - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: finalText, - }), - }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: shellRoute - ? "assisted-autopilot-shell" - : humanRoute - ? "assisted-autopilot-human" - : "assisted-autopilot-glob", - type: "function", - function: - shellRoute || humanRoute - ? { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: humanRoute - ? process.platform === "win32" - ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : process.platform === "win32" - ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` - : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, - description: humanRoute - ? "Create the human-reviewed SDK permission fixture" - : "Create the authorized SDK permission fixture", - }), - } - : { - name: "glob", - arguments: JSON.stringify({ - pattern: "*.txt", - path: outsideDir, - }), - }, - }, - ], - }; + response.writeHead(200, { "content-type": "application/json" }); + response.end(JSON.stringify(completion)); } - } - - const choice = { - index: 0, - message, - finish_reason: "tool_calls" in message ? "tool_calls" : "stop", - }; - const completion = { - id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, - object: "chat.completion", - created: 1, - model: body.model, - choices: [choice], - usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, - }; - if (body.stream) { - response.writeHead(200, { "content-type": "text/event-stream" }); - response.end( - `data: ${JSON.stringify({ - ...completion, - object: "chat.completion.chunk", - choices: [ - { - index: 0, - delta: { - ...message, - ...("tool_calls" in message - ? { - tool_calls: message.tool_calls.map( - (call, index) => ({ index, ...call }) - ), - } - : {}), - }, - finish_reason: choice.finish_reason, - }, - ], - })}\n\ndata: [DONE]\n\n` + })().catch((error: unknown) => { + providerFailures.push( + error instanceof Error ? error : new Error(String(error)) ); - } else { - response.writeHead(200, { "content-type": "application/json" }); - response.end(JSON.stringify(completion)); + response.writeHead(500).end(); + }); + }); + await new Promise((resolve, reject) => { + modelServer.once("error", reject); + modelServer.listen(0, "127.0.0.1", resolve); + }); + onTestFinished(async () => { + modelServer.closeAllConnections(); + if (modelServer.listening) { + await new Promise((resolve, reject) => { + modelServer.close((error) => (error ? reject(error) : resolve())); + }); } - })().catch((error: unknown) => { - providerFailures.push(error instanceof Error ? error : new Error(String(error))); - response.writeHead(500).end(); }); - }); - await new Promise((resolve, reject) => { - modelServer.once("error", reject); - modelServer.listen(0, "127.0.0.1", resolve); - }); - onTestFinished(async () => { - modelServer.closeAllConnections(); - if (modelServer.listening) { - await new Promise((resolve, reject) => { - modelServer.close((error) => (error ? reject(error) : resolve())); - }); + const address = modelServer.address(); + if (!address || typeof address === "string") { + throw new Error("Missing local model server address"); } - }); - const address = modelServer.address(); - if (!address || typeof address === "string") { - throw new Error("Missing local model server address"); - } - execFileSync("git", ["init", "--quiet"], { cwd: workDir }); - await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + execFileSync("git", ["init", "--quiet"], { cwd: workDir }); + await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - const providers: NamedProviderConfig[] = [ - { - name: "local", - type: "openai", - baseUrl: `http://127.0.0.1:${address.port}/v1`, - apiKey: "synthetic-test-token", - wireApi: "completions", - }, - ]; - const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ - id, - provider: "local", - modelId: "gpt-4o", - wireModel: id, - })); - const scenarios = [ - { route: "shell", lifecycle: "create", decision: "judge" }, - { route: "path", lifecycle: "create", decision: "judge" }, - { route: "shell", lifecycle: "resume", decision: "judge" }, - { route: "path", lifecycle: "resume", decision: "judge" }, - { route: "human", lifecycle: "create", decision: "approve" }, - { route: "human", lifecycle: "create", decision: "deny" }, - { route: "human", lifecycle: "resume", decision: "approve" }, - { route: "human", lifecycle: "resume", decision: "deny" }, - ] as const; - for (const scenario of scenarios) { - const label = `initial:${scenario.route}:${scenario.lifecycle}:${scenario.decision}`; - let permissionCallbacks = 0; - const expectedRecommendation = - scenario.decision === "judge" ? "approve" : "requireApproval"; - const recommendationByToolCallId = new Map(); - const recommendationWaiters = new Map< - string, - (recommendation: string | undefined) => void - >(); - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; - const decisionSources: string[] = []; - const taskOutcomes: Array<{ - success?: boolean; - summary?: string; - outcome?: string; - reason?: string; - }> = []; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "glob", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, + const providers: NamedProviderConfig[] = [ + { + name: "local", + type: "openai", + baseUrl: `http://127.0.0.1:${address.port}/v1`, + apiKey: "synthetic-test-token", + wireApi: "completions", }, - onPermissionRequest: async (request: PermissionRequest) => { - permissionCallbacks++; - if (!request.toolCallId) { - throw new Error( - "Expected the permission request to identify its tool call" - ); - } - const recommendation = recommendationByToolCallId.has(request.toolCallId) - ? recommendationByToolCallId.get(request.toolCallId) - : await new Promise((resolve) => { - recommendationWaiters.set(request.toolCallId!, resolve); - }); - recommendationByToolCallId.delete(request.toolCallId); - if (recommendation !== expectedRecommendation) { - throw new Error( - `Expected Assisted recommendation ${expectedRecommendation}, got ${String( - recommendation - )}` - ); - } - if (scenario.decision === "judge") { + ]; + const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ + id, + provider: "local", + modelId: "gpt-4o", + wireModel: id, + })); + const scenarios = [ + { route: "shell", lifecycle: "create", decision: "judge" }, + { route: "path", lifecycle: "create", decision: "judge" }, + { route: "shell", lifecycle: "resume", decision: "judge" }, + { route: "path", lifecycle: "resume", decision: "judge" }, + { route: "human", lifecycle: "create", decision: "approve" }, + { route: "human", lifecycle: "create", decision: "deny" }, + { route: "human", lifecycle: "resume", decision: "approve" }, + { route: "human", lifecycle: "resume", decision: "deny" }, + ] as const; + for (const scenario of contract === "lifecycle" ? scenarios : []) { + const label = `initial:${scenario.route}:${scenario.lifecycle}:${scenario.decision}`; + let permissionCallbacks = 0; + const expectedRecommendation = + scenario.decision === "judge" ? "approve" : "requireApproval"; + const recommendationByToolCallId = new Map(); + const recommendationWaiters = new Map< + string, + (recommendation: string | undefined) => void + >(); + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; + const decisionSources: string[] = []; + const taskOutcomes: Array<{ + success?: boolean; + summary?: string; + outcome?: string; + reason?: string; + }> = []; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "glob", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + onPermissionRequest: async (request: PermissionRequest) => { + permissionCallbacks++; + if (!request.toolCallId) { + throw new Error( + "Expected the permission request to identify its tool call" + ); + } + const recommendation = recommendationByToolCallId.has(request.toolCallId) + ? recommendationByToolCallId.get(request.toolCallId) + : await new Promise((resolve) => { + recommendationWaiters.set(request.toolCallId!, resolve); + }); + recommendationByToolCallId.delete(request.toolCallId); + if (recommendation !== expectedRecommendation) { + throw new Error( + `Expected Assisted recommendation ${expectedRecommendation}, got ${String( + recommendation + )}` + ); + } + if (scenario.decision === "judge") { + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } + ); + } return createAttributedPermissionResult( - { kind: "approve-once" }, + { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, { - outcome: "auto_approved", - source: "assisted_approval", + outcome: "prompted_user", + source: "human_response", surface: "sdk", - responseCapability: "headless", + responseCapability: "interactive", } ); + }, + } as const; + let session: CopilotSession | undefined; + try { + enterPhase(`${label}:create`); + session = await client.createSession(sessionConfig); + if (scenario.lifecycle === "resume") { + const sessionId = session.sessionId; + enterPhase(`${label}:prime`); + await session.sendAndWait({ + prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", + }); + enterPhase(`${label}:configure-before-resume`); + await configureAssistedAutopilot(session, workDir); + await expectAssistedAutopilotConfigured(session, workDir); + enterPhase(`${label}:disconnect-before-resume`); + await session.disconnect(); + session = undefined; + enterPhase(`${label}:resume`); + session = await client.resumeSession(sessionId, sessionConfig); + enterPhase(`${label}:verify-resume`); + await expectAssistedAutopilotConfigured(session, workDir); + } else { + enterPhase(`${label}:configure`); + await configureAssistedAutopilot(session, workDir); } - return createAttributedPermissionResult( - { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, - { - outcome: "prompted_user", - source: "human_response", - surface: "sdk", - responseCapability: "interactive", + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; + } + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + const toolCallId = ( + event.data as { + permissionRequest?: { toolCallId?: string }; + } + ).permissionRequest?.toolCallId; + if (toolCallId) { + const waiter = recommendationWaiters.get(toolCallId); + if (waiter) { + recommendationWaiters.delete(toolCallId); + waiter(recommendation); + } else { + recommendationByToolCallId.set(toolCallId, recommendation); + } + } + if (recommendation) recommendations.push(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }) + .decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + recoveryStatuses.push((event.data as { status: string }).status); + } else if (event.type === "tool.execution_complete") { + toolResults.push(event.data); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); } + }); + + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + enterPhase(`${label}:send`); + await session.send({ + prompt: + scenario.route === "shell" + ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` + : scenario.route === "path" + ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` + : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, + }); + enterPhase(`${label}:task-complete`); + await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } + + enterPhase(`${label}:assert`); + const expectedCompletions = + scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill(expectedRecommendation) ); - }, - } as const; - let session: CopilotSession | undefined; - try { - enterPhase(`${label}:create`); - session = await client.createSession(sessionConfig); - if (scenario.lifecycle === "resume") { - const sessionId = session.sessionId; - enterPhase(`${label}:prime`); - await session.sendAndWait({ - prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", + expect(decisionSources, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill( + scenario.decision === "judge" ? "assisted_approval" : "human_response" + ) + ); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); + if (scenario.decision === "deny") { + expect( + toolResults.some((result) => result.success === false), + JSON.stringify(scenario) + ).toBe(true); + expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); + } else { + expect( + toolResults.every((result) => result.success === true), + JSON.stringify(scenario) + ).toBe(true); + } + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + success: true, }); - enterPhase(`${label}:configure-before-resume`); - await configureAssistedAutopilot(session, workDir); - await expectAssistedAutopilotConfigured(session, workDir); - enterPhase(`${label}:disconnect-before-resume`); - await session.disconnect(); - session = undefined; - enterPhase(`${label}:resume`); - session = await client.resumeSession(sessionId, sessionConfig); - enterPhase(`${label}:verify-resume`); - await expectAssistedAutopilotConfigured(session, workDir); - } else { - enterPhase(`${label}:configure`); - await configureAssistedAutopilot(session, workDir); + expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( + scenario.route === "shell" + ? "shell-approved" + : scenario.route === "path" + ? "approval-probe.txt" + : `human-${scenario.decision === "approve" ? "approved" : "denied"}` + ); + if (scenario.route === "human") { + expect( + existsSync( + join( + workDir, + `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` + ) + ), + JSON.stringify(scenario) + ).toBe(scenario.decision === "approve"); + } + } finally { + if (session) { + enterPhase(`${label}:cleanup`); + await session.abort(); + await session.disconnect(); + } } - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; + } + + if (contract === "lifecycle") { + expect(providerFailures).toEqual([]); + expect(judgeOutputs).toEqual([ + ...Array(4).fill(JUDGE_OUTPUT), + ...Array(4).fill(HUMAN_REVIEW_OUTPUT), + ]); + expect(agentCalls).toBe(scenarios.length * 2 + 4); + } + + if (contract === "authorization") { + const authorizationJudgeStart = judgeOutputs.length; + let authorizationPermissionCallbacks = 0; + const authorizationRecommendations: string[] = []; + const authorizationDecisionSources: string[] = []; + const authorizationRecoveryStatuses: string[] = []; + const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = + []; + const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = + []; + let resolveAuthorizationRecommendation: ( + recommendation: string | undefined + ) => void; + const authorizationRecommendation = new Promise((resolve) => { + resolveAuthorizationRecommendation = resolve; + }); + let authorizationSession: CopilotSession | undefined; + try { + enterPhase("authorization:create"); + authorizationSession = await client.createSession({ + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + onPermissionRequest: async (request) => { + authorizationPermissionCallbacks++; + if (request.toolCallId !== "assisted-authorization-shell") { + throw new Error( + `Expected authorization shell permission, got ${String( + request.toolCallId + )}` + ); } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - const toolCallId = ( - event.data as { - permissionRequest?: { toolCallId?: string }; + const recommendation = await authorizationRecommendation; + if (recommendation !== "approve") { + throw new Error( + `Expected later authorization to reach the judge as approve, got ${String( + recommendation + )}` + ); } - ).permissionRequest?.toolCallId; - if (toolCallId) { - const waiter = recommendationWaiters.get(toolCallId); - if (waiter) { - recommendationWaiters.delete(toolCallId); - waiter(recommendation); - } else { - recommendationByToolCallId.set(toolCallId, recommendation); + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } + ); + }, + }); + enterPhase("authorization:prime"); + await authorizationSession.sendAndWait({ + prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, + }); + enterPhase("authorization:configure"); + await configureAssistedAutopilot(authorizationSession, workDir); + authorizationSession.on((event) => { + if (event.type === "permission.requested") { + const recommendation = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + }; + } + ).promptRequest?.assistedApproval?.recommendation; + if (recommendation) { + authorizationRecommendations.push(recommendation); } + resolveAuthorizationRecommendation(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }) + .decisionSource; + if (source) authorizationDecisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + authorizationRecoveryStatuses.push( + (event.data as { status: string }).status + ); + } else if (event.type === "tool.execution_complete") { + authorizationToolResults.push({ + toolCallId: event.data.toolCallId, + success: event.data.success, + }); + } else if (event.type === "session.task_complete") { + authorizationTaskOutcomes.push(event.data); } - if (recommendation) recommendations.push(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }).decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - recoveryStatuses.push((event.data as { status: string }).status); - } else if (event.type === "tool.execution_complete") { - toolResults.push(event.data); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); - } - }); + }); - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - enterPhase(`${label}:send`); - await session.send({ - prompt: - scenario.route === "shell" - ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` - : scenario.route === "path" - ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` - : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, + const taskComplete = getNextEventOfType( + authorizationSession, + "session.task_complete" + ); + enterPhase("authorization:send"); + await authorizationSession.send({ + prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, }); - enterPhase(`${label}:task-complete`); + enterPhase("authorization:task-complete"); await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } - enterPhase(`${label}:assert`); - const expectedCompletions = - scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill(expectedRecommendation) - ); - expect(decisionSources, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill( - scenario.decision === "judge" ? "assisted_approval" : "human_response" - ) - ); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); - if (scenario.decision === "deny") { + enterPhase("authorization:assert"); + expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); + expect(authorizationJudgeRequests).toHaveLength(1); + const authorizationRequest = authorizationJudgeRequests[0] ?? ""; expect( - toolResults.some((result) => result.success === false), - JSON.stringify(scenario) - ).toBe(true); - expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); - } else { - expect( - toolResults.every((result) => result.success === true), - JSON.stringify(scenario) - ).toBe(true); - } - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - success: true, - }); - expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( - scenario.route === "shell" - ? "shell-approved" - : scenario.route === "path" - ? "approval-probe.txt" - : `human-${scenario.decision === "approve" ? "approved" : "denied"}` - ); - if (scenario.route === "human") { + authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) + ).toBeGreaterThanOrEqual(0); expect( - existsSync( - join( - workDir, - `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` - ) - ), - JSON.stringify(scenario) - ).toBe(scenario.decision === "approve"); - } - } finally { - if (session) { - enterPhase(`${label}:cleanup`); - await session.abort(); - await session.disconnect(); + authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER) + ).toBeGreaterThan(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)); + expect(authorizationPermissionCallbacks).toBe(1); + expect(authorizationRecommendations).toEqual(["approve"]); + expect(authorizationDecisionSources).toEqual(["assisted_approval"]); + expect(authorizationRecoveryStatuses).toEqual([]); + expect(authorizationToolResults).toContainEqual({ + toolCallId: "assisted-authorization-shell", + success: true, + }); + expect(authorizationTaskOutcomes).toEqual([ + expect.objectContaining({ + success: true, + summary: "authorization-updated", + }), + ]); + expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); + } finally { + if (authorizationSession) { + enterPhase("authorization:cleanup"); + await authorizationSession.abort(); + await authorizationSession.disconnect(); + } } } - } - expect(providerFailures).toEqual([]); - expect(judgeOutputs).toEqual([ - ...Array(4).fill(JUDGE_OUTPUT), - ...Array(4).fill(HUMAN_REVIEW_OUTPUT), - ]); - expect(agentCalls).toBe(scenarios.length * 2 + 4); - - const authorizationJudgeStart = judgeOutputs.length; - let authorizationPermissionCallbacks = 0; - const authorizationRecommendations: string[] = []; - const authorizationDecisionSources: string[] = []; - const authorizationRecoveryStatuses: string[] = []; - const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = []; - const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = []; - let resolveAuthorizationRecommendation: (recommendation: string | undefined) => void; - const authorizationRecommendation = new Promise((resolve) => { - resolveAuthorizationRecommendation = resolve; - }); - let authorizationSession: CopilotSession | undefined; - try { - enterPhase("authorization:create"); - authorizationSession = await client.createSession({ - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - onPermissionRequest: async (request) => { - authorizationPermissionCallbacks++; - if (request.toolCallId !== "assisted-authorization-shell") { - throw new Error( - `Expected authorization shell permission, got ${String( - request.toolCallId - )}` + const contractScenarios = [ + { scope: "root", recommendation: "approve", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "excluded", handler: "registered" }, + { scope: "subagent", recommendation: "approve", handler: "registered" }, + { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "none" }, + ] as const satisfies readonly { + scope: ContractScope; + recommendation: ContractRecommendation; + handler: "registered" | "none"; + }[]; + const contractJudgeStart = judgeOutputs.length; + for (const scenario of contract === "handler" ? contractScenarios : []) { + const label = `contract:${scenario.scope}:${scenario.recommendation}:${scenario.handler}`; + let permissionCallbacks = 0; + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const decisionSources: string[] = []; + const sequence: string[] = []; + const subagentIds = new Set(); + const permissionAgentIds: Array = []; + const toolResults: Array<{ + agentId?: string; + toolCallId?: string; + success?: boolean; + error?: { message?: string }; + result?: unknown; + }> = []; + const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; + const onPermissionRequest = () => { + permissionCallbacks++; + sequence.push("host"); + if (scenario.recommendation === "approve") { + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); } - const recommendation = await authorizationRecommendation; - if (recommendation !== "approve") { - throw new Error( - `Expected later authorization to reach the judge as approve, got ${String( - recommendation - )}` + if (scenario.recommendation === "requireApproval") { + return createAttributedPermissionResult( + { + kind: "reject", + feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), + }, + { + outcome: "autopilot_denied", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); } - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - }, - }); - enterPhase("authorization:prime"); - await authorizationSession.sendAndWait({ - prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, - }); - enterPhase("authorization:configure"); - await configureAssistedAutopilot(authorizationSession, workDir); - authorizationSession.on((event) => { - if (event.type === "permission.requested") { - const recommendation = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - }; - } - ).promptRequest?.assistedApproval?.recommendation; - if (recommendation) { - authorizationRecommendations.push(recommendation); + if (scenario.recommendation === "excluded") { + return createAttributedPermissionResult( + { kind: "user-not-available" }, + { + outcome: "autopilot_denied", + source: "unattended_fallback", + surface: "sdk", + responseCapability: "headless", + } + ); } - resolveAuthorizationRecommendation(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }).decisionSource; - if (source) authorizationDecisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - authorizationRecoveryStatuses.push((event.data as { status: string }).status); - } else if (event.type === "tool.execution_complete") { - authorizationToolResults.push({ - toolCallId: event.data.toolCallId, - success: event.data.success, + throw new Error("Unexpected Assisted permission recommendation"); + }; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), + } as const; + let session: CopilotSession | undefined; + try { + enterPhase(`${label}:create`); + session = await client.createSession(sessionConfig); + enterPhase(`${label}:configure`); + await configureAssistedAutopilot(session, workDir); + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; + } + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + if (recommendation) { + recommendations.push(recommendation); + sequence.push(`permission:${recommendation}`); + } + permissionAgentIds.push(event.agentId); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }) + .decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + const status = (event.data as { status: string }).status; + recoveryStatuses.push(status); + sequence.push(`recovery:${status}`); + } else if (event.type === "tool.execution_complete") { + toolResults.push({ + agentId: event.agentId, + toolCallId: event.data.toolCallId, + success: event.data.success, + error: event.data.error, + result: event.data.result, + }); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); + } else if (event.type === "subagent.started") { + subagentIds.add(event.agentId); + } }); - } else if (event.type === "session.task_complete") { - authorizationTaskOutcomes.push(event.data); - } - }); - - const taskComplete = getNextEventOfType(authorizationSession, "session.task_complete"); - enterPhase("authorization:send"); - await authorizationSession.send({ - prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, - }); - enterPhase("authorization:task-complete"); - await taskComplete; - enterPhase("authorization:assert"); - expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); - expect(authorizationJudgeRequests).toHaveLength(1); - const authorizationRequest = authorizationJudgeRequests[0] ?? ""; - expect(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)).toBeGreaterThanOrEqual( - 0 - ); - expect(authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER)).toBeGreaterThan( - authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) - ); - expect(authorizationPermissionCallbacks).toBe(1); - expect(authorizationRecommendations).toEqual(["approve"]); - expect(authorizationDecisionSources).toEqual(["assisted_approval"]); - expect(authorizationRecoveryStatuses).toEqual([]); - expect(authorizationToolResults).toContainEqual({ - toolCallId: "assisted-authorization-shell", - success: true, - }); - expect(authorizationTaskOutcomes).toEqual([ - expect.objectContaining({ - success: true, - summary: "authorization-updated", - }), - ]); - expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); - } finally { - if (authorizationSession) { - enterPhase("authorization:cleanup"); - await authorizationSession.abort(); - await authorizationSession.disconnect(); - } - } + const marker = contractMarker(scenario.recommendation); + const prompt = + scenario.scope === "root" + ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` + : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + enterPhase(`${label}:send`); + await session.send({ prompt }); + enterPhase(`${label}:task-complete`); + await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } - const contractScenarios = [ - { scope: "root", recommendation: "approve", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "excluded", handler: "registered" }, - { scope: "subagent", recommendation: "approve", handler: "registered" }, - { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "none" }, - ] as const satisfies readonly { - scope: ContractScope; - recommendation: ContractRecommendation; - handler: "registered" | "none"; - }[]; - const contractJudgeStart = judgeOutputs.length; - for (const scenario of contractScenarios) { - const label = `contract:${scenario.scope}:${scenario.recommendation}:${scenario.handler}`; - let permissionCallbacks = 0; - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const decisionSources: string[] = []; - const sequence: string[] = []; - const subagentIds = new Set(); - const permissionAgentIds: Array = []; - const toolResults: Array<{ - agentId?: string; - toolCallId?: string; - success?: boolean; - error?: { message?: string }; - result?: unknown; - }> = []; - const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; - const onPermissionRequest = () => { - permissionCallbacks++; - sequence.push("host"); - if (scenario.recommendation === "approve") { - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } + enterPhase(`${label}:assert`); + if (scenario.handler === "none") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); + expect(recommendations, JSON.stringify(scenario)).toEqual([]); + } else if (scenario.recommendation === "approve") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); + } else if (scenario.recommendation === "requireApproval") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual([ + "requireApproval", + ]); + } else { + expect( + permissionCallbacks, + JSON.stringify(scenario) + ).toBeGreaterThanOrEqual(1); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(permissionCallbacks).fill("excluded") + ); + } + if (scenario.handler === "none") { + expect(decisionSources, JSON.stringify(scenario)).toEqual([]); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); + } else { + expect(decisionSources, JSON.stringify(scenario)).toContain( + scenario.recommendation === "excluded" + ? "unattended_fallback" + : "assisted_approval" + ); + } + if (scenario.recommendation === "excluded") { + expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); + } else if (scenario.handler === "registered") { + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + } + const hostIndex = sequence.indexOf("host"); + const firstRecoveryIndex = sequence.findIndex((entry) => + entry.startsWith("recovery:") ); - } - if (scenario.recommendation === "requireApproval") { - return createAttributedPermissionResult( - { - kind: "reject", - feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), - }, - { - outcome: "autopilot_denied", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } + if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { + expect( + hostIndex, + JSON.stringify({ scenario, sequence }) + ).toBeGreaterThanOrEqual(0); + expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( + firstRecoveryIndex + ); + } + + const shellResult = toolResults.find( + (result) => + result.toolCallId === + contractShellCallId(scenario.scope, scenario.recommendation) ); - } - if (scenario.recommendation === "excluded") { - return createAttributedPermissionResult( - { kind: "user-not-available" }, - { - outcome: "autopilot_denied", - source: "unattended_fallback", - surface: "sdk", - responseCapability: "headless", - } + expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); + expect(shellResult?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation === "approve" && scenario.handler === "registered" ); - } - throw new Error("Unexpected Assisted permission recommendation"); - }; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), - } as const; - let session: CopilotSession | undefined; - try { - enterPhase(`${label}:create`); - session = await client.createSession(sessionConfig); - enterPhase(`${label}:configure`); - await configureAssistedAutopilot(session, workDir); - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; - } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - if (recommendation) { - recommendations.push(recommendation); - sequence.push(`permission:${recommendation}`); + if ( + scenario.recommendation === "requireApproval" && + scenario.handler === "registered" + ) { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Ask the registered human permission handler." + ); + } else if (scenario.handler === "none") { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Permission could not be granted automatically." + ); + } + if (scenario.scope === "subagent") { + expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); + expect( + subagentIds.has(shellResult?.agentId ?? ""), + JSON.stringify(scenario) + ).toBe(true); + for (const agentId of permissionAgentIds) { + expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe( + true + ); } - permissionAgentIds.push(event.agentId); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }).decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - const status = (event.data as { status: string }).status; - recoveryStatuses.push(status); - sequence.push(`recovery:${status}`); - } else if (event.type === "tool.execution_complete") { - toolResults.push({ - agentId: event.agentId, - toolCallId: event.data.toolCallId, - success: event.data.success, - error: event.data.error, - result: event.data.result, - }); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); - } else if (event.type === "subagent.started") { - subagentIds.add(event.agentId); } - }); - - const marker = contractMarker(scenario.recommendation); - const prompt = - scenario.scope === "root" - ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` - : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - enterPhase(`${label}:send`); - await session.send({ prompt }); - enterPhase(`${label}:task-complete`); - await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } - enterPhase(`${label}:assert`); - if (scenario.handler === "none") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); - expect(recommendations, JSON.stringify(scenario)).toEqual([]); - } else if (scenario.recommendation === "approve") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); - } else if (scenario.recommendation === "requireApproval") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual(["requireApproval"]); - } else { - expect(permissionCallbacks, JSON.stringify(scenario)).toBeGreaterThanOrEqual(1); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(permissionCallbacks).fill("excluded") - ); - } - if (scenario.handler === "none") { - expect(decisionSources, JSON.stringify(scenario)).toEqual([]); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); - } else { - expect(decisionSources, JSON.stringify(scenario)).toContain( - scenario.recommendation === "excluded" - ? "unattended_fallback" - : "assisted_approval" - ); - } - if (scenario.recommendation === "excluded") { - expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); - } else if (scenario.handler === "registered") { - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - } - const hostIndex = sequence.indexOf("host"); - const firstRecoveryIndex = sequence.findIndex((entry) => - entry.startsWith("recovery:") - ); - if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { expect( - hostIndex, - JSON.stringify({ scenario, sequence }) - ).toBeGreaterThanOrEqual(0); - expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( - firstRecoveryIndex - ); - } - - const shellResult = toolResults.find( - (result) => - result.toolCallId === - contractShellCallId(scenario.scope, scenario.recommendation) - ); - expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); - expect(shellResult?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation === "approve" && scenario.handler === "registered" - ); - if ( - scenario.recommendation === "requireApproval" && - scenario.handler === "registered" - ) { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Ask the registered human permission handler." - ); - } else if (scenario.handler === "none") { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Permission could not be granted automatically." + existsSync( + join( + workDir, + contractFixtureName(scenario.scope, scenario.recommendation) + ) + ), + JSON.stringify(scenario) + ).toBe( + scenario.recommendation === "approve" && scenario.handler === "registered" ); - } - if (scenario.scope === "subagent") { - expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); expect( - subagentIds.has(shellResult?.agentId ?? ""), + (await session.rpc.permissions.pendingRequests()).items, JSON.stringify(scenario) - ).toBe(true); - for (const agentId of permissionAgentIds) { - expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe(true); + ).toEqual([]); + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation !== "excluded" && scenario.handler === "registered" + ); + if (scenario.handler === "none") { + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + outcome: "continue", + reason: "Autopilot is still recovering from a required permission.", + }); + } + } finally { + if (session) { + enterPhase(`${label}:cleanup`); + await session.abort(); + await session.disconnect(); } } + } - expect( - existsSync( - join(workDir, contractFixtureName(scenario.scope, scenario.recommendation)) - ), - JSON.stringify(scenario) - ).toBe(scenario.recommendation === "approve" && scenario.handler === "registered"); - expect( - (await session.rpc.permissions.pendingRequests()).items, - JSON.stringify(scenario) - ).toEqual([]); - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation !== "excluded" && scenario.handler === "registered" - ); - if (scenario.handler === "none") { - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - outcome: "continue", - reason: "Autopilot is still recovering from a required permission.", - }); - } - } finally { - if (session) { - enterPhase(`${label}:cleanup`); - await session.abort(); - await session.disconnect(); - } + enterPhase("final-assertions"); + expect(providerFailures).toEqual([]); + if (contract === "handler") { + expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + HUMAN_REVIEW_OUTPUT, + ]); } } - - enterPhase("final-assertions"); - expect(providerFailures).toEqual([]); - expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - HUMAN_REVIEW_OUTPUT, - ]); - }); + ); }); async function configureAssistedAutopilot(session: CopilotSession, workDir: string): Promise { From 76fbc0ee3025a0937f37509bb589e138a208700f Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 15:30:32 -0700 Subject: [PATCH 09/15] test: record complete contract cleanup timings Provide synthetic-only completion counts and cumulative awaited-cleanup timing for each split contract so remote CI can verify the decomposition without inferring a detach-latency fix. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../e2e/assisted_autopilot_permission.e2e.test.ts | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts index ef67382bfc..a06e01ebe8 100644 --- a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts +++ b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts @@ -1130,6 +1130,19 @@ describe("Assisted permission handling in Autopilot", async () => { HUMAN_REVIEW_OUTPUT, ]); } + console.info( + "Assisted contract completed:", + JSON.stringify({ + contract, + elapsedMs: Date.now() - started, + cleanupMs: completedPhases + .filter(({ phase }) => /cleanup|disconnect-before-resume/.test(phase)) + .reduce((total, { elapsedMs }) => total + elapsedMs, 0), + agentCalls, + judgeCalls: judgeOutputs.length, + providerFailures: providerFailures.length, + }) + ); } ); }); From 6c0814dc0cc16d748d67a71dea5a0d2b6708d89a Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 15:54:57 -0700 Subject: [PATCH 10/15] test: report attached-shell queue failure metadata Capture only synthetic phase, counts, timings, and marker-presence booleans from existing polls/events on failure. Preserve assertions, timeouts, polling RPCs, and cancellation ownership. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../e2e/rpc_tasks_and_handlers.e2e.test.ts | 42 +++++++++++++++++-- 1 file changed, 38 insertions(+), 4 deletions(-) diff --git a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts index e6e6a18c8f..d23ca0446b 100644 --- a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts +++ b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts @@ -2,7 +2,7 @@ * Copyright (c) Microsoft Corporation. All rights reserved. *--------------------------------------------------------------------------------------------*/ -import { describe, expect, it } from "vitest"; +import { describe, expect, it, onTestFailed } from "vitest"; import { z } from "zod"; import { approveAll, CopilotRequestHandler } from "../../src/index.js"; import type { SessionEvent, CopilotRequestContext, CopilotSession } from "../../src/index.js"; @@ -390,25 +390,59 @@ describe("Session tasks RPC and pending handlers", async () => { { timeout: 120_000 }, async () => { await withRunningAttachedShell(async (session, shellId, eventTypes, replies) => { + const started = performance.now(); + let phase = "enqueue"; + let queuePolls = 0; + let pendingCount = 0; + let queuedMatch = false; + onTestFailed(() => { + console.error( + "Attached shell queue failure boundary:", + JSON.stringify({ + phase, + elapsedMs: Math.round(performance.now() - started), + queuePolls, + pendingCount, + queuedMatch, + replyCount: replies.length, + hasQueuedReply: replies.some((reply) => reply.includes("QUEUED_DONE")), + sessionIdleCount: eventTypes.filter((type) => type === "session.idle") + .length, + assistantIdleCount: eventTypes.filter( + (type) => type === "assistant.idle" + ).length, + }) + ); + }); // The running attached shell holds idle, so an enqueued message waits behind it. await session.send({ prompt: "Reply with exactly QUEUED_DONE.", mode: "enqueue" }); + phase = "pending-queue"; await waitForCondition( - async () => - (await session.rpc.queue.pendingItems()).items.some((item) => + async () => { + const { items } = await session.rpc.queue.pendingItems(); + queuePolls++; + pendingCount = items.length; + queuedMatch = items.some((item) => item.displayText.includes("QUEUED_DONE") - ), + ); + return queuedMatch; + }, { timeoutMessage: "The enqueued message was not parked behind the shell" } ); + phase = "assert-parked"; expect(replies.some((reply) => reply.includes("QUEUED_DONE"))).toBe(false); expect(eventTypes).not.toContain("session.idle"); + phase = "cancel-shell"; expect((await session.rpc.tasks.cancel({ id: shellId })).cancelled).toBe(true); + phase = "queued-reply"; await waitForCondition( () => replies.some((reply) => reply.includes("QUEUED_DONE")), { timeoutMessage: `The queued message never ran after cancelling shell ${shellId}`, } ); + phase = "session-idle"; await waitForCondition(() => eventTypes.includes("session.idle"), { timeoutMessage: "session.idle never followed the queued message", }); From 5af93d4538f47bfd9df1b660b2f988d459ca16f3 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 17:00:35 -0700 Subject: [PATCH 11/15] ci: use standard macOS runners for published SDK tests Keep source-runtime runner profiles, matrix selections and test contracts unchanged. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/workflows/sdk-dotnet.yml | 9 ++++++--- .github/workflows/sdk-go.yml | 7 +++++-- .github/workflows/sdk-java.yml | 7 +++++-- .github/workflows/sdk-nodejs.yml | 4 ++-- .github/workflows/sdk-platform.yml | 5 +++++ .github/workflows/sdk-python.yml | 7 +++++-- .github/workflows/sdk-rust.yml | 7 +++++-- .github/workflows/sdk.yml | 1 + CONTRIBUTING.md | 5 +++++ 9 files changed, 39 insertions(+), 13 deletions(-) diff --git a/.github/workflows/sdk-dotnet.yml b/.github/workflows/sdk-dotnet.yml index 14a4c3265b..f26c440437 100644 --- a/.github/workflows/sdk-dotnet.yml +++ b/.github/workflows/sdk-dotnet.yml @@ -19,6 +19,9 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read @@ -99,9 +102,9 @@ jobs: retention-days: 7 dotnet-darwin-arm64: - name: ".NET (macos-26-xlarge-agent-runtime, default, CAPI)" + name: ".NET (${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }}, default, CAPI)" if: inputs.platform == 'all' || inputs.platform == 'darwin-arm64' - runs-on: macos-26-xlarge-agent-runtime + runs-on: ${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }} timeout-minutes: 30 defaults: run: @@ -147,7 +150,7 @@ jobs: - if: failure() uses: actions/upload-artifact@v7 with: - name: dotnet-test-diagnostics-macos-26-xlarge-default-capi-${{ github.run_attempt }} + name: dotnet-test-diagnostics-${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge' }}-default-capi-${{ github.run_attempt }} path: ${{ inputs.sdk-home }}/dotnet/TestResults/ if-no-files-found: warn retention-days: 7 diff --git a/.github/workflows/sdk-go.yml b/.github/workflows/sdk-go.yml index b5593f26b1..25b63039a9 100644 --- a/.github/workflows/sdk-go.yml +++ b/.github/workflows/sdk-go.yml @@ -19,19 +19,22 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: go: - name: "Go (${{ matrix.os }})" + name: "Go (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-java.yml b/.github/workflows/sdk-java.yml index 011fef0898..3d022571e3 100644 --- a/.github/workflows/sdk-java.yml +++ b/.github/workflows/sdk-java.yml @@ -18,13 +18,16 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: java: - name: "Java (${{ matrix.os }}, JDK ${{ matrix.test-jdk }})" + name: "Java (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, JDK ${{ matrix.test-jdk }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false @@ -36,7 +39,7 @@ jobs: test-jdk: "17" - os: windows-latest test-jdk: "17" - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index 5b506d5d90..f94e00bfe4 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -34,13 +34,13 @@ permissions: jobs: nodejs: - name: "Node.js (${{ matrix.os }}, CAPI)" + name: "Node.js (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, CAPI)" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-platform.yml b/.github/workflows/sdk-platform.yml index 568afd4147..85937f783c 100644 --- a/.github/workflows/sdk-platform.yml +++ b/.github/workflows/sdk-platform.yml @@ -41,6 +41,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit go: @@ -49,6 +50,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit dotnet: @@ -57,6 +59,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit rust: @@ -65,6 +68,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit java: @@ -73,4 +77,5 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit diff --git a/.github/workflows/sdk-python.yml b/.github/workflows/sdk-python.yml index df9cbab1a9..e81bb63975 100644 --- a/.github/workflows/sdk-python.yml +++ b/.github/workflows/sdk-python.yml @@ -19,19 +19,22 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: python: - name: "Python (${{ matrix.os }})" + name: "Python (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} timeout-minutes: 20 defaults: run: diff --git a/.github/workflows/sdk-rust.yml b/.github/workflows/sdk-rust.yml index 04500c3489..12506d5cad 100644 --- a/.github/workflows/sdk-rust.yml +++ b/.github/workflows/sdk-rust.yml @@ -19,20 +19,23 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: rust: - name: "Rust (${{ matrix.os }})" + name: "Rust (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) timeout-minutes: 60 strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-xlarge-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-xlarge-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk.yml b/.github/workflows/sdk.yml index 1fa6969a1b..3c70b00d33 100644 --- a/.github/workflows/sdk.yml +++ b/.github/workflows/sdk.yml @@ -468,6 +468,7 @@ jobs: with: platform: linuxmusl-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-darwin-arm64: diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 15e5c5cc7c..b1df98a029 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -363,6 +363,11 @@ failure fails the job. Java uses JDK 25 on all four platforms, plus a Linux/glibc JDK 17 compatibility job using precompiled classes. Merge groups retain the reduced Linux TypeScript CAPI subprocess coverage. +Standalone published-runtime macOS jobs use the standard `macos-26` ARM64 +GitHub-hosted runner. Source-runtime jobs retain their configured runtime +runners, including the larger Rust and .NET profiles. Test selection and +timeouts are unchanged; failures on the standard runner still fail coverage. + The `sdk-typescript` required rollup checks only the Linux CAPI job, including its build, packaging, and applicable static checks. Other platforms, BYOK backends, and languages keep their existing scheduling and failure reporting; From 76fc111a4b81e503bf93a1a385250d0365d5276f Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 17:06:12 -0700 Subject: [PATCH 12/15] chore: restrict resume fix PR to Rust changes Remove author-added non-Rust test and CI follow-ups, restoring those paths to the accepted canonical main revision. Preserve all Rust changes and prior commits. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/actions/run-alpine-tests/action.yml | 1 - .github/workflows/sdk-dotnet.yml | 9 +- .github/workflows/sdk-go.yml | 7 +- .github/workflows/sdk-java.yml | 7 +- .github/workflows/sdk-nodejs.yml | 65 +- .github/workflows/sdk-platform.yml | 9 - .github/workflows/sdk-python.yml | 7 +- .github/workflows/sdk-rust.yml | 7 +- .github/workflows/sdk.yml | 5 - CONTRIBUTING.md | 26 - .../assisted_autopilot_permission.e2e.test.ts | 1922 ++++++++--------- .../e2e/rpc_tasks_and_handlers.e2e.test.ts | 42 +- nodejs/test/rust-codegen.test.ts | 10 +- scripts/ci/device-policy-bootstrap.mjs | 85 - scripts/ci/device-policy-fixture.mjs | 468 ---- scripts/ci/device-policy-fixture.test.mjs | 207 -- scripts/ci/device-policy-legacy.js | 10 - scripts/ci/device-policy-native.js | 10 - 18 files changed, 933 insertions(+), 1964 deletions(-) delete mode 100644 scripts/ci/device-policy-bootstrap.mjs delete mode 100644 scripts/ci/device-policy-fixture.mjs delete mode 100644 scripts/ci/device-policy-fixture.test.mjs delete mode 100644 scripts/ci/device-policy-legacy.js delete mode 100644 scripts/ci/device-policy-native.js diff --git a/.github/actions/run-alpine-tests/action.yml b/.github/actions/run-alpine-tests/action.yml index 7808ca3f67..a9842c11be 100644 --- a/.github/actions/run-alpine-tests/action.yml +++ b/.github/actions/run-alpine-tests/action.yml @@ -54,7 +54,6 @@ runs: --env COPILOT_SDK_ROOT="$ALPINE_TEST_SDK_ROOT" \ --env COPILOT_HMAC_KEY \ --env COPILOT_SDK_E2E_BACKEND \ - --env COPILOT_CI_RUNTIME_SOURCE \ --env BUNDLED_CLI_CACHE_DIR \ --env CARGO_TERM_COLOR \ --env RUST_BACKTRACE \ diff --git a/.github/workflows/sdk-dotnet.yml b/.github/workflows/sdk-dotnet.yml index f26c440437..14a4c3265b 100644 --- a/.github/workflows/sdk-dotnet.yml +++ b/.github/workflows/sdk-dotnet.yml @@ -19,9 +19,6 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read @@ -102,9 +99,9 @@ jobs: retention-days: 7 dotnet-darwin-arm64: - name: ".NET (${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }}, default, CAPI)" + name: ".NET (macos-26-xlarge-agent-runtime, default, CAPI)" if: inputs.platform == 'all' || inputs.platform == 'darwin-arm64' - runs-on: ${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }} + runs-on: macos-26-xlarge-agent-runtime timeout-minutes: 30 defaults: run: @@ -150,7 +147,7 @@ jobs: - if: failure() uses: actions/upload-artifact@v7 with: - name: dotnet-test-diagnostics-${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge' }}-default-capi-${{ github.run_attempt }} + name: dotnet-test-diagnostics-macos-26-xlarge-default-capi-${{ github.run_attempt }} path: ${{ inputs.sdk-home }}/dotnet/TestResults/ if-no-files-found: warn retention-days: 7 diff --git a/.github/workflows/sdk-go.yml b/.github/workflows/sdk-go.yml index 25b63039a9..b5593f26b1 100644 --- a/.github/workflows/sdk-go.yml +++ b/.github/workflows/sdk-go.yml @@ -19,22 +19,19 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: go: - name: "Go (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" + name: "Go (${{ matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-java.yml b/.github/workflows/sdk-java.yml index 3d022571e3..011fef0898 100644 --- a/.github/workflows/sdk-java.yml +++ b/.github/workflows/sdk-java.yml @@ -18,16 +18,13 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: java: - name: "Java (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, JDK ${{ matrix.test-jdk }})" + name: "Java (${{ matrix.os }}, JDK ${{ matrix.test-jdk }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false @@ -39,7 +36,7 @@ jobs: test-jdk: "17" - os: windows-latest test-jdk: "17" - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index f94e00bfe4..db14685c13 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -7,7 +7,6 @@ env: HUSKY: 0 POWERSHELL_UPDATECHECK: Off SDK_HOME: ${{ inputs.sdk-home }} - COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} SETUP_NODE_TIMEOUT_MINUTES: 10 on: @@ -20,9 +19,6 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string outputs: capi-result: description: "Standard-platform CAPI job result, independent of BYOK jobs" @@ -34,13 +30,13 @@ permissions: jobs: nodejs: - name: "Node.js (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, CAPI)" + name: "Node.js (${{ matrix.os }}, CAPI)" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash @@ -93,51 +89,16 @@ jobs: working-directory: ${{ inputs.sdk-home }}/scripts/corrections - if: runner.os == 'Windows' run: pwsh.exe -Command "Write-Host 'PowerShell ready'" - - name: Test device-policy CI plans and required-result guard - if: runner.os == 'Linux' - working-directory: . - run: node --test "$SDK_HOME/scripts/ci/device-policy-fixture.test.mjs" - - name: Prepare isolated production device-policy fixtures - id: device-policy - if: inputs.runtime-source == 'published' && runner.os == 'Linux' - working-directory: . - run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" prepare - name: Test out-of-process id: subprocess env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} run: | - test_args=() - pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - case "$pattern" in - "") ;; - *) - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - test_args+=(--testNamePattern "$pattern") - ;; - esac if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ] || [ "$GITHUB_EVENT_NAME" = "push" ]; then - npm test -- "${test_args[@]}" --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" + npm test -- --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" else - npm test -- "${test_args[@]}" + npm test fi - - name: Test required production device-policy controls - if: ${{ !cancelled() && inputs.runtime-source == 'published' && runner.os == 'Linux' && (steps.subprocess.outcome == 'success' || steps.subprocess.outcome == 'failure') }} - env: - COPILOT_CLI_PATH: ${{ steps.device-policy.outputs.native-launcher }} - COPILOT_LEGACY_CLI_PATH: ${{ steps.device-policy.outputs.legacy-launcher }} - COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - run: | - status=0 - npm test -- test/e2e/managed_plugin_progress.e2e.test.ts test/e2e/rpc_server.e2e.test.ts \ - --testNamePattern "$(node ../scripts/ci/device-policy-fixture.mjs required-pattern)" \ - --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-device-policy-results.json" || status=$? - node ../scripts/ci/device-policy-fixture.mjs verify "$RUNNER_TEMP/sdk-device-policy-results.json" - exit "$status" - - name: Clean up owned device-policy containers - if: always() && steps.device-policy.outcome == 'success' - working-directory: . - run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" cleanup - name: Upload Flake Finder Node.js test results if: always() && (github.event_name == 'merge_group' || github.event_name == 'pull_request' || github.event_name == 'push') continue-on-error: true @@ -201,14 +162,7 @@ jobs: - name: Test out-of-process E2Es env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - run: | - test_args=() - pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - if [[ -n "$pattern" ]]; then - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - test_args+=(--testNamePattern "$pattern") - fi - npm test -- test/e2e "${test_args[@]}" + run: npm test -- test/e2e nodejs-musl-x64: name: "Node.js (Alpine x64, CAPI)" @@ -228,7 +182,6 @@ jobs: - uses: ./.github/actions/run-alpine-tests env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} with: image: node:22-alpine sdk-root: /workspace/${{ inputs.sdk-home }} @@ -253,12 +206,6 @@ jobs: test -x "$COPILOT_CLI_PATH" test -f "$(dirname "$COPILOT_CLI_PATH")/runtime.node" status=0 - pattern=$(node "$COPILOT_SDK_ROOT/scripts/ci/device-policy-fixture.mjs" portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - if [ -n "$pattern" ]; then - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - npm test -- --testNamePattern "$pattern" || status=$? - else - npm test || status=$? - fi + npm test || status=$? COPILOT_SDK_DEFAULT_CONNECTION=inprocess npm test -- test/e2e/inprocess_ffi.e2e.test.ts test/e2e/auth_host.e2e.test.ts || status=$? exit "$status" diff --git a/.github/workflows/sdk-platform.yml b/.github/workflows/sdk-platform.yml index 85937f783c..f1ed4e6ed4 100644 --- a/.github/workflows/sdk-platform.yml +++ b/.github/workflows/sdk-platform.yml @@ -11,9 +11,6 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string outputs: nodejs-result: description: "Node.js SDK result, independent of other languages" @@ -32,7 +29,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit python: @@ -41,7 +37,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit go: @@ -50,7 +45,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit dotnet: @@ -59,7 +53,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit rust: @@ -68,7 +61,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit java: @@ -77,5 +69,4 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit diff --git a/.github/workflows/sdk-python.yml b/.github/workflows/sdk-python.yml index e81bb63975..df9cbab1a9 100644 --- a/.github/workflows/sdk-python.yml +++ b/.github/workflows/sdk-python.yml @@ -19,22 +19,19 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: python: - name: "Python (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" + name: "Python (${{ matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} timeout-minutes: 20 defaults: run: diff --git a/.github/workflows/sdk-rust.yml b/.github/workflows/sdk-rust.yml index 12506d5cad..04500c3489 100644 --- a/.github/workflows/sdk-rust.yml +++ b/.github/workflows/sdk-rust.yml @@ -19,23 +19,20 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: rust: - name: "Rust (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }})" + name: "Rust (${{ matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) timeout-minutes: 60 strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-xlarge-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-xlarge-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk.yml b/.github/workflows/sdk.yml index 3c70b00d33..602d02b027 100644 --- a/.github/workflows/sdk.yml +++ b/.github/workflows/sdk.yml @@ -446,7 +446,6 @@ jobs: with: platform: linux-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-x64: @@ -457,7 +456,6 @@ jobs: with: platform: linuxmusl-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-arm64: @@ -468,7 +466,6 @@ jobs: with: platform: linuxmusl-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-darwin-arm64: @@ -479,7 +476,6 @@ jobs: with: platform: darwin-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-win32-x64: @@ -491,7 +487,6 @@ jobs: with: platform: win32-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit # Only Linux CAPI gates this rollup; the full SDK aggregate still reports all coverage. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index b1df98a029..74519107b4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -363,38 +363,12 @@ failure fails the job. Java uses JDK 25 on all four platforms, plus a Linux/glibc JDK 17 compatibility job using precompiled classes. Merge groups retain the reduced Linux TypeScript CAPI subprocess coverage. -Standalone published-runtime macOS jobs use the standard `macos-26` ARM64 -GitHub-hosted runner. Source-runtime jobs retain their configured runtime -runners, including the larger Rust and .NET profiles. Test selection and -timeouts are unchanged; failures on the standard runner still fail coverage. - The `sdk-typescript` required rollup checks only the Linux CAPI job, including its build, packaging, and applicable static checks. Other platforms, BYOK backends, and languages keep their existing scheduling and failure reporting; they do not gate this rollup. The full `SDK` aggregate still requires all scheduled coverage to succeed. -Standalone CI assigns the two managed-device fixture controls to a separate -mandatory Linux CAPI step. Each fixture-bearing CLI child runs the unchanged, -pinned published runtime in its own disposable container, with the fixture -installed at the [documented production device-policy location](https://docs.github.com/en/copilot/how-tos/administer-copilot/manage-for-enterprise/use-managed-settings/deploy-managed-settings). -The launcher consumes the temporary-file test hint, verifies isolation and -root file ownership, then restores the runner's identity before launching the -runtime. It never writes policy to the host. A result guard requires both -original plugin-lifecycle and sessionless device/model tests to pass; missing -or skipped controls fail the job. Other published-runtime profiles explicitly -delegate only those two cases to this step. Source-runtime profiles retain -their existing full test selection and launch setup. - -The CI-only container setup requires Linux, Docker, and Node.js 22.15 or newer. -Do not provision a machine-wide policy file on a developer workstation to -run these fixtures. Launcher plan, selection, and result-guard unit controls -run without Docker, a CLI, or dependency installation: - -```bash -node --test scripts/ci/device-policy-fixture.test.mjs -``` - The three BYOK backend sweeps run in separate Linux TypeScript jobs, alongside the normal CAPI job; they do not repeat unit tests, packaging, or static checks. After preparing the runtime as described above, run a sweep from the diff --git a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts index a06e01ebe8..5577d03761 100644 --- a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts +++ b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts @@ -8,7 +8,7 @@ import { mkdir, realpath, rm, writeFile } from "node:fs/promises"; import { createServer } from "node:http"; import { join } from "node:path"; import { text } from "node:stream/consumers"; -import { describe, expect, it, onTestFailed, onTestFinished } from "vitest"; +import { describe, expect, it, onTestFinished } from "vitest"; import { createAttributedPermissionResult, type CopilotSession, @@ -45,314 +45,157 @@ function contractFixtureName(scope: ContractScope, recommendation: ContractRecom describe("Assisted permission handling in Autopilot", async () => { const { copilotClient: client, workDir } = await createSdkTestContext({ useStdio: true }); - it.each(["lifecycle", "authorization", "handler"] as const)( - "resolves Assisted recommendations before Autopilot permission recovery (%s contract)", - async (contract) => { - let agentCalls = 0; - const judgeOutputs: string[] = []; - const authorizationJudgeRequests: string[] = []; - const providerFailures: Error[] = []; - const started = Date.now(); - let phase = "setup"; - let phaseStarted = started; - const completedPhases: Array<{ phase: string; elapsedMs: number }> = []; - const enterPhase = (next: string) => { - const now = Date.now(); - completedPhases.push({ phase, elapsedMs: now - phaseStarted }); - phase = next; - phaseStarted = now; - }; - onTestFailed(() => { - console.error( - "Assisted contract failure boundary:", - JSON.stringify({ - phase, - phaseElapsedMs: Date.now() - phaseStarted, - elapsedMs: Date.now() - started, - completedPhases, - agentCalls, - judgeCalls: judgeOutputs.length, - providerFailures: providerFailures.length, - }) - ); - }); - const outsideDir = `${workDir}-outside`; - await mkdir(outsideDir, { recursive: true }); - await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); - const modelServer = createServer((request, response) => { - void (async () => { - const body = JSON.parse(await text(request)) as { - model: string; - stream?: boolean; - messages: Array<{ role?: string; content?: unknown }>; - }; - const isJudge = body.model === "gpt-6-luna"; - const messages = JSON.stringify(body.messages); - const latestUserMessage = [...body.messages] - .reverse() - .find((entry) => entry.role === "user"); - const latestUserText = JSON.stringify(latestUserMessage?.content); - let message: - | { role: "assistant"; content: string } - | { - role: "assistant"; - content: string; - tool_calls: Array<{ - id: string; - type: "function"; - function: { name: string; arguments: string }; - }>; - }; - if (isJudge) { - const earlierRestrictionIndex = messages.lastIndexOf( - EARLIER_RESTRICTION_MARKER - ); - const laterAuthorizationIndex = messages.lastIndexOf( - LATER_AUTHORIZATION_MARKER - ); - const authorizationRoute = laterAuthorizationIndex !== -1; - if (authorizationRoute) { - authorizationJudgeRequests.push(messages); - } - const contractRequiresApproval = - messages.includes(contractFixtureName("root", "requireApproval")) || - messages.includes(contractFixtureName("subagent", "requireApproval")); - const output = authorizationRoute - ? earlierRestrictionIndex !== -1 && - earlierRestrictionIndex < laterAuthorizationIndex - ? JUDGE_OUTPUT - : HUMAN_REVIEW_OUTPUT - : messages.includes("assisted-human-ask") || contractRequiresApproval - ? HUMAN_REVIEW_OUTPUT - : JUDGE_OUTPUT; - judgeOutputs.push(output); - message = { role: "assistant", content: output }; - } else { - agentCalls++; - const toolResultCount = body.messages.filter( - (entry) => entry.role === "tool" - ).length; - const contractChildRoute = latestUserText.includes( - "ASSISTED_CONTRACT_CHILD_" - ); - const contractSubagentRoute = latestUserText.includes( - "ASSISTED_CONTRACT_SUBAGENT_" - ); - const contractRootRoute = - latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); - const authorizationRestrictionRoute = latestUserText.includes( - EARLIER_RESTRICTION_MARKER - ); - const authorizationAllowRoute = latestUserText.includes( - LATER_AUTHORIZATION_MARKER - ); - if (authorizationRestrictionRoute) { - message = { role: "assistant", content: "restriction-recorded" }; - } else if (authorizationAllowRoute) { - message = - toolResultCount >= 2 - ? { role: "assistant", content: "authorization-updated" } - : toolResultCount === 1 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-task-complete", - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: "authorization-updated", - }), - }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-shell", - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: - process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` - : `mkdir ${AUTHORIZATION_FIXTURE}`, - description: - "Create the fixture authorized by the latest instruction", - }), - }, - }, - ], - }; - } else if ( - contractChildRoute || - contractSubagentRoute || - contractRootRoute - ) { - const recommendation: ContractRecommendation = latestUserText.includes( - "_REQUIRE" - ) - ? "requireApproval" - : latestUserText.includes("_EXCLUDED") - ? "excluded" - : "approve"; - const scope: ContractScope = - contractChildRoute || contractSubagentRoute ? "subagent" : "root"; - const finalText = `contract-${scope}-${ - recommendation === "approve" ? "approved" : "blocked" - }`; - if (contractSubagentRoute) { - message = - toolResultCount === 0 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: `assisted-contract-task-${recommendation}`, - type: "function", - function: { - name: "task", - arguments: JSON.stringify({ - name: "assisted-contract", - description: - "Exercise an Assisted permission request", - prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( - recommendation - )}: Run the requested shell command once and report whether it completed.`, - agent_type: "task", - mode: "sync", - }), - }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: `assisted-contract-task-complete-${recommendation}`, - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: finalText, - }), - }, - }, - ], - }; - } else if (contractChildRoute && toolResultCount > 0) { - message = { role: "assistant", content: finalText }; - } else if ( - (contractChildRoute || contractRootRoute) && - (toolResultCount === 0 || - (recommendation === "excluded" && toolResultCount === 1)) - ) { - const command = - recommendation === "excluded" - ? process.platform === "win32" - ? "Get-Content approval-probe.txt | & $runner" - : "printf payload | $runner" - : process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${contractFixtureName( - scope, - recommendation - )}` - : `mkdir ${contractFixtureName(scope, recommendation)}`; - message = { - role: "assistant", - content: "", - tool_calls: [ - { - id: - toolResultCount === 0 - ? contractShellCallId(scope, recommendation) - : `${contractShellCallId( - scope, - recommendation - )}-retry`, - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command, - description: - "Exercise Assisted permission routing", - }), - }, - }, - ], - }; - } else { - message = { - role: "assistant", - content: "", - tool_calls: [ - { - id: `assisted-contract-task-complete-${recommendation}`, - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ summary: finalText }), - }, - }, - ], - }; - } - } else { - const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); - const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); - const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); - const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); - const primeRoute = - primeIndex > Math.max(shellIndex, pathIndex, humanIndex); - const humanRoute = - humanIndex > Math.max(primeIndex, shellIndex, pathIndex); - const shellRoute = - shellIndex > Math.max(primeIndex, pathIndex, humanIndex); - const resumedShellRoute = - messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; - const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); - const humanLifecycle = - messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || - messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") - ? "resume" - : "create"; - const finalText = humanRoute - ? humanApproved - ? "human-approved" - : "human-denied" - : shellRoute - ? "shell-approved" - : "approval-probe.txt"; - message = primeRoute - ? { role: "assistant", content: "prime-ready" } - : toolResultCount >= 2 + it("resolves Assisted recommendations before Autopilot permission recovery", async () => { + let agentCalls = 0; + const judgeOutputs: string[] = []; + const authorizationJudgeRequests: string[] = []; + const providerFailures: Error[] = []; + const outsideDir = `${workDir}-outside`; + await mkdir(outsideDir, { recursive: true }); + await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); + const modelServer = createServer((request, response) => { + void (async () => { + const body = JSON.parse(await text(request)) as { + model: string; + stream?: boolean; + messages: Array<{ role?: string; content?: unknown }>; + }; + const isJudge = body.model === "gpt-6-luna"; + const messages = JSON.stringify(body.messages); + const latestUserMessage = [...body.messages] + .reverse() + .find((entry) => entry.role === "user"); + const latestUserText = JSON.stringify(latestUserMessage?.content); + let message: + | { role: "assistant"; content: string } + | { + role: "assistant"; + content: string; + tool_calls: Array<{ + id: string; + type: "function"; + function: { name: string; arguments: string }; + }>; + }; + if (isJudge) { + const earlierRestrictionIndex = messages.lastIndexOf( + EARLIER_RESTRICTION_MARKER + ); + const laterAuthorizationIndex = messages.lastIndexOf( + LATER_AUTHORIZATION_MARKER + ); + const authorizationRoute = laterAuthorizationIndex !== -1; + if (authorizationRoute) { + authorizationJudgeRequests.push(messages); + } + const contractRequiresApproval = + messages.includes(contractFixtureName("root", "requireApproval")) || + messages.includes(contractFixtureName("subagent", "requireApproval")); + const output = authorizationRoute + ? earlierRestrictionIndex !== -1 && + earlierRestrictionIndex < laterAuthorizationIndex + ? JUDGE_OUTPUT + : HUMAN_REVIEW_OUTPUT + : messages.includes("assisted-human-ask") || contractRequiresApproval + ? HUMAN_REVIEW_OUTPUT + : JUDGE_OUTPUT; + judgeOutputs.push(output); + message = { role: "assistant", content: output }; + } else { + agentCalls++; + const toolResultCount = body.messages.filter( + (entry) => entry.role === "tool" + ).length; + const contractChildRoute = latestUserText.includes("ASSISTED_CONTRACT_CHILD_"); + const contractSubagentRoute = latestUserText.includes( + "ASSISTED_CONTRACT_SUBAGENT_" + ); + const contractRootRoute = latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); + const authorizationRestrictionRoute = latestUserText.includes( + EARLIER_RESTRICTION_MARKER + ); + const authorizationAllowRoute = latestUserText.includes( + LATER_AUTHORIZATION_MARKER + ); + if (authorizationRestrictionRoute) { + message = { role: "assistant", content: "restriction-recorded" }; + } else if (authorizationAllowRoute) { + message = + toolResultCount >= 2 + ? { role: "assistant", content: "authorization-updated" } + : toolResultCount === 1 ? { role: "assistant", - content: finalText, + content: "", + tool_calls: [ + { + id: "assisted-authorization-task-complete", + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: "authorization-updated", + }), + }, + }, + ], } - : toolResultCount === 1 + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-authorization-shell", + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: + process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` + : `mkdir ${AUTHORIZATION_FIXTURE}`, + description: + "Create the fixture authorized by the latest instruction", + }), + }, + }, + ], + }; + } else if (contractChildRoute || contractSubagentRoute || contractRootRoute) { + const recommendation: ContractRecommendation = latestUserText.includes( + "_REQUIRE" + ) + ? "requireApproval" + : latestUserText.includes("_EXCLUDED") + ? "excluded" + : "approve"; + const scope: ContractScope = + contractChildRoute || contractSubagentRoute ? "subagent" : "root"; + const finalText = `contract-${scope}-${ + recommendation === "approve" ? "approved" : "blocked" + }`; + if (contractSubagentRoute) { + message = + toolResultCount === 0 ? { role: "assistant", content: "", tool_calls: [ { - id: "assisted-autopilot-task-complete", + id: `assisted-contract-task-${recommendation}`, type: "function", function: { - name: "task_complete", + name: "task", arguments: JSON.stringify({ - summary: finalText, + name: "assisted-contract", + description: + "Exercise an Assisted permission request", + prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( + recommendation + )}: Run the requested shell command once and report whether it completed.`, + agent_type: "task", + mode: "sync", }), }, }, @@ -363,788 +206,833 @@ describe("Assisted permission handling in Autopilot", async () => { content: "", tool_calls: [ { - id: shellRoute - ? "assisted-autopilot-shell" - : humanRoute - ? "assisted-autopilot-human" - : "assisted-autopilot-glob", + id: `assisted-contract-task-complete-${recommendation}`, type: "function", - function: - shellRoute || humanRoute - ? { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: humanRoute - ? process.platform === - "win32" - ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : process.platform === - "win32" - ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` - : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, - description: humanRoute - ? "Create the human-reviewed SDK permission fixture" - : "Create the authorized SDK permission fixture", - }), - } - : { - name: "glob", - arguments: JSON.stringify({ - pattern: "*.txt", - path: outsideDir, - }), - }, + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: finalText, + }), + }, }, ], }; - } - } - - const choice = { - index: 0, - message, - finish_reason: "tool_calls" in message ? "tool_calls" : "stop", - }; - const completion = { - id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, - object: "chat.completion", - created: 1, - model: body.model, - choices: [choice], - usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, - }; - if (body.stream) { - response.writeHead(200, { "content-type": "text/event-stream" }); - response.end( - `data: ${JSON.stringify({ - ...completion, - object: "chat.completion.chunk", - choices: [ + } else if (contractChildRoute && toolResultCount > 0) { + message = { role: "assistant", content: finalText }; + } else if ( + (contractChildRoute || contractRootRoute) && + (toolResultCount === 0 || + (recommendation === "excluded" && toolResultCount === 1)) + ) { + const command = + recommendation === "excluded" + ? process.platform === "win32" + ? "Get-Content approval-probe.txt | & $runner" + : "printf payload | $runner" + : process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${contractFixtureName( + scope, + recommendation + )}` + : `mkdir ${contractFixtureName(scope, recommendation)}`; + message = { + role: "assistant", + content: "", + tool_calls: [ { - index: 0, - delta: { - ...message, - ...("tool_calls" in message - ? { - tool_calls: message.tool_calls.map( - (call, index) => ({ index, ...call }) - ), - } - : {}), + id: + toolResultCount === 0 + ? contractShellCallId(scope, recommendation) + : `${contractShellCallId( + scope, + recommendation + )}-retry`, + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command, + description: "Exercise Assisted permission routing", + }), }, - finish_reason: choice.finish_reason, }, ], - })}\n\ndata: [DONE]\n\n` - ); + }; + } else { + message = { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-complete-${recommendation}`, + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ summary: finalText }), + }, + }, + ], + }; + } } else { - response.writeHead(200, { "content-type": "application/json" }); - response.end(JSON.stringify(completion)); + const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); + const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); + const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); + const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); + const primeRoute = primeIndex > Math.max(shellIndex, pathIndex, humanIndex); + const humanRoute = humanIndex > Math.max(primeIndex, shellIndex, pathIndex); + const shellRoute = shellIndex > Math.max(primeIndex, pathIndex, humanIndex); + const resumedShellRoute = + messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; + const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); + const humanLifecycle = + messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || + messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") + ? "resume" + : "create"; + const finalText = humanRoute + ? humanApproved + ? "human-approved" + : "human-denied" + : shellRoute + ? "shell-approved" + : "approval-probe.txt"; + message = primeRoute + ? { role: "assistant", content: "prime-ready" } + : toolResultCount >= 2 + ? { + role: "assistant", + content: finalText, + } + : toolResultCount === 1 + ? { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-autopilot-task-complete", + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: finalText, + }), + }, + }, + ], + } + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: shellRoute + ? "assisted-autopilot-shell" + : humanRoute + ? "assisted-autopilot-human" + : "assisted-autopilot-glob", + type: "function", + function: + shellRoute || humanRoute + ? { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: humanRoute + ? process.platform === "win32" + ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : process.platform === "win32" + ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` + : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, + description: humanRoute + ? "Create the human-reviewed SDK permission fixture" + : "Create the authorized SDK permission fixture", + }), + } + : { + name: "glob", + arguments: JSON.stringify({ + pattern: "*.txt", + path: outsideDir, + }), + }, + }, + ], + }; } - })().catch((error: unknown) => { - providerFailures.push( - error instanceof Error ? error : new Error(String(error)) + } + + const choice = { + index: 0, + message, + finish_reason: "tool_calls" in message ? "tool_calls" : "stop", + }; + const completion = { + id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, + object: "chat.completion", + created: 1, + model: body.model, + choices: [choice], + usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, + }; + if (body.stream) { + response.writeHead(200, { "content-type": "text/event-stream" }); + response.end( + `data: ${JSON.stringify({ + ...completion, + object: "chat.completion.chunk", + choices: [ + { + index: 0, + delta: { + ...message, + ...("tool_calls" in message + ? { + tool_calls: message.tool_calls.map( + (call, index) => ({ index, ...call }) + ), + } + : {}), + }, + finish_reason: choice.finish_reason, + }, + ], + })}\n\ndata: [DONE]\n\n` ); - response.writeHead(500).end(); - }); - }); - await new Promise((resolve, reject) => { - modelServer.once("error", reject); - modelServer.listen(0, "127.0.0.1", resolve); - }); - onTestFinished(async () => { - modelServer.closeAllConnections(); - if (modelServer.listening) { - await new Promise((resolve, reject) => { - modelServer.close((error) => (error ? reject(error) : resolve())); - }); + } else { + response.writeHead(200, { "content-type": "application/json" }); + response.end(JSON.stringify(completion)); } + })().catch((error: unknown) => { + providerFailures.push(error instanceof Error ? error : new Error(String(error))); + response.writeHead(500).end(); }); - const address = modelServer.address(); - if (!address || typeof address === "string") { - throw new Error("Missing local model server address"); + }); + await new Promise((resolve, reject) => { + modelServer.once("error", reject); + modelServer.listen(0, "127.0.0.1", resolve); + }); + onTestFinished(async () => { + modelServer.closeAllConnections(); + if (modelServer.listening) { + await new Promise((resolve, reject) => { + modelServer.close((error) => (error ? reject(error) : resolve())); + }); } + }); + const address = modelServer.address(); + if (!address || typeof address === "string") { + throw new Error("Missing local model server address"); + } - execFileSync("git", ["init", "--quiet"], { cwd: workDir }); - await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + execFileSync("git", ["init", "--quiet"], { cwd: workDir }); + await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - const providers: NamedProviderConfig[] = [ - { - name: "local", - type: "openai", - baseUrl: `http://127.0.0.1:${address.port}/v1`, - apiKey: "synthetic-test-token", - wireApi: "completions", + const providers: NamedProviderConfig[] = [ + { + name: "local", + type: "openai", + baseUrl: `http://127.0.0.1:${address.port}/v1`, + apiKey: "synthetic-test-token", + wireApi: "completions", + }, + ]; + const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ + id, + provider: "local", + modelId: "gpt-4o", + wireModel: id, + })); + const scenarios = [ + { route: "shell", lifecycle: "create", decision: "judge" }, + { route: "path", lifecycle: "create", decision: "judge" }, + { route: "shell", lifecycle: "resume", decision: "judge" }, + { route: "path", lifecycle: "resume", decision: "judge" }, + { route: "human", lifecycle: "create", decision: "approve" }, + { route: "human", lifecycle: "create", decision: "deny" }, + { route: "human", lifecycle: "resume", decision: "approve" }, + { route: "human", lifecycle: "resume", decision: "deny" }, + ] as const; + for (const scenario of scenarios) { + let permissionCallbacks = 0; + const expectedRecommendation = + scenario.decision === "judge" ? "approve" : "requireApproval"; + const recommendationByToolCallId = new Map(); + const recommendationWaiters = new Map< + string, + (recommendation: string | undefined) => void + >(); + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; + const decisionSources: string[] = []; + const taskOutcomes: Array<{ + success?: boolean; + summary?: string; + outcome?: string; + reason?: string; + }> = []; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "glob", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, }, - ]; - const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ - id, - provider: "local", - modelId: "gpt-4o", - wireModel: id, - })); - const scenarios = [ - { route: "shell", lifecycle: "create", decision: "judge" }, - { route: "path", lifecycle: "create", decision: "judge" }, - { route: "shell", lifecycle: "resume", decision: "judge" }, - { route: "path", lifecycle: "resume", decision: "judge" }, - { route: "human", lifecycle: "create", decision: "approve" }, - { route: "human", lifecycle: "create", decision: "deny" }, - { route: "human", lifecycle: "resume", decision: "approve" }, - { route: "human", lifecycle: "resume", decision: "deny" }, - ] as const; - for (const scenario of contract === "lifecycle" ? scenarios : []) { - const label = `initial:${scenario.route}:${scenario.lifecycle}:${scenario.decision}`; - let permissionCallbacks = 0; - const expectedRecommendation = - scenario.decision === "judge" ? "approve" : "requireApproval"; - const recommendationByToolCallId = new Map(); - const recommendationWaiters = new Map< - string, - (recommendation: string | undefined) => void - >(); - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; - const decisionSources: string[] = []; - const taskOutcomes: Array<{ - success?: boolean; - summary?: string; - outcome?: string; - reason?: string; - }> = []; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "glob", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - onPermissionRequest: async (request: PermissionRequest) => { - permissionCallbacks++; - if (!request.toolCallId) { - throw new Error( - "Expected the permission request to identify its tool call" - ); - } - const recommendation = recommendationByToolCallId.has(request.toolCallId) - ? recommendationByToolCallId.get(request.toolCallId) - : await new Promise((resolve) => { - recommendationWaiters.set(request.toolCallId!, resolve); - }); - recommendationByToolCallId.delete(request.toolCallId); - if (recommendation !== expectedRecommendation) { - throw new Error( - `Expected Assisted recommendation ${expectedRecommendation}, got ${String( - recommendation - )}` - ); - } - if (scenario.decision === "judge") { - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - } + onPermissionRequest: async (request: PermissionRequest) => { + permissionCallbacks++; + if (!request.toolCallId) { + throw new Error( + "Expected the permission request to identify its tool call" + ); + } + const recommendation = recommendationByToolCallId.has(request.toolCallId) + ? recommendationByToolCallId.get(request.toolCallId) + : await new Promise((resolve) => { + recommendationWaiters.set(request.toolCallId!, resolve); + }); + recommendationByToolCallId.delete(request.toolCallId); + if (recommendation !== expectedRecommendation) { + throw new Error( + `Expected Assisted recommendation ${expectedRecommendation}, got ${String( + recommendation + )}` + ); + } + if (scenario.decision === "judge") { return createAttributedPermissionResult( - { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, + { kind: "approve-once" }, { - outcome: "prompted_user", - source: "human_response", + outcome: "auto_approved", + source: "assisted_approval", surface: "sdk", - responseCapability: "interactive", + responseCapability: "headless", } ); - }, - } as const; - let session: CopilotSession | undefined; - try { - enterPhase(`${label}:create`); - session = await client.createSession(sessionConfig); - if (scenario.lifecycle === "resume") { - const sessionId = session.sessionId; - enterPhase(`${label}:prime`); - await session.sendAndWait({ - prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", - }); - enterPhase(`${label}:configure-before-resume`); - await configureAssistedAutopilot(session, workDir); - await expectAssistedAutopilotConfigured(session, workDir); - enterPhase(`${label}:disconnect-before-resume`); - await session.disconnect(); - session = undefined; - enterPhase(`${label}:resume`); - session = await client.resumeSession(sessionId, sessionConfig); - enterPhase(`${label}:verify-resume`); - await expectAssistedAutopilotConfigured(session, workDir); - } else { - enterPhase(`${label}:configure`); - await configureAssistedAutopilot(session, workDir); } - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; - } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - const toolCallId = ( - event.data as { - permissionRequest?: { toolCallId?: string }; - } - ).permissionRequest?.toolCallId; - if (toolCallId) { - const waiter = recommendationWaiters.get(toolCallId); - if (waiter) { - recommendationWaiters.delete(toolCallId); - waiter(recommendation); - } else { - recommendationByToolCallId.set(toolCallId, recommendation); - } - } - if (recommendation) recommendations.push(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }) - .decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - recoveryStatuses.push((event.data as { status: string }).status); - } else if (event.type === "tool.execution_complete") { - toolResults.push(event.data); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); + return createAttributedPermissionResult( + { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, + { + outcome: "prompted_user", + source: "human_response", + surface: "sdk", + responseCapability: "interactive", } - }); - - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - enterPhase(`${label}:send`); - await session.send({ - prompt: - scenario.route === "shell" - ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` - : scenario.route === "path" - ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` - : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, - }); - enterPhase(`${label}:task-complete`); - await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } - - enterPhase(`${label}:assert`); - const expectedCompletions = - scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill(expectedRecommendation) ); - expect(decisionSources, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill( - scenario.decision === "judge" ? "assisted_approval" : "human_response" - ) - ); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); - if (scenario.decision === "deny") { - expect( - toolResults.some((result) => result.success === false), - JSON.stringify(scenario) - ).toBe(true); - expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); - } else { - expect( - toolResults.every((result) => result.success === true), - JSON.stringify(scenario) - ).toBe(true); - } - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - success: true, + }, + } as const; + let session: CopilotSession | undefined; + try { + session = await client.createSession(sessionConfig); + if (scenario.lifecycle === "resume") { + const sessionId = session.sessionId; + await session.sendAndWait({ + prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", }); - expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( - scenario.route === "shell" - ? "shell-approved" - : scenario.route === "path" - ? "approval-probe.txt" - : `human-${scenario.decision === "approve" ? "approved" : "denied"}` - ); - if (scenario.route === "human") { - expect( - existsSync( - join( - workDir, - `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` - ) - ), - JSON.stringify(scenario) - ).toBe(scenario.decision === "approve"); - } - } finally { - if (session) { - enterPhase(`${label}:cleanup`); - await session.abort(); - await session.disconnect(); - } + await configureAssistedAutopilot(session, workDir); + await expectAssistedAutopilotConfigured(session, workDir); + await session.disconnect(); + session = undefined; + session = await client.resumeSession(sessionId, sessionConfig); + await expectAssistedAutopilotConfigured(session, workDir); + } else { + await configureAssistedAutopilot(session, workDir); } - } - - if (contract === "lifecycle") { - expect(providerFailures).toEqual([]); - expect(judgeOutputs).toEqual([ - ...Array(4).fill(JUDGE_OUTPUT), - ...Array(4).fill(HUMAN_REVIEW_OUTPUT), - ]); - expect(agentCalls).toBe(scenarios.length * 2 + 4); - } - - if (contract === "authorization") { - const authorizationJudgeStart = judgeOutputs.length; - let authorizationPermissionCallbacks = 0; - const authorizationRecommendations: string[] = []; - const authorizationDecisionSources: string[] = []; - const authorizationRecoveryStatuses: string[] = []; - const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = - []; - const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = - []; - let resolveAuthorizationRecommendation: ( - recommendation: string | undefined - ) => void; - const authorizationRecommendation = new Promise((resolve) => { - resolveAuthorizationRecommendation = resolve; - }); - let authorizationSession: CopilotSession | undefined; - try { - enterPhase("authorization:create"); - authorizationSession = await client.createSession({ - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - onPermissionRequest: async (request) => { - authorizationPermissionCallbacks++; - if (request.toolCallId !== "assisted-authorization-shell") { - throw new Error( - `Expected authorization shell permission, got ${String( - request.toolCallId - )}` - ); + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; } - const recommendation = await authorizationRecommendation; - if (recommendation !== "approve") { - throw new Error( - `Expected later authorization to reach the judge as approve, got ${String( - recommendation - )}` - ); + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + const toolCallId = ( + event.data as { + permissionRequest?: { toolCallId?: string }; } - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - }, - }); - enterPhase("authorization:prime"); - await authorizationSession.sendAndWait({ - prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, - }); - enterPhase("authorization:configure"); - await configureAssistedAutopilot(authorizationSession, workDir); - authorizationSession.on((event) => { - if (event.type === "permission.requested") { - const recommendation = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - }; - } - ).promptRequest?.assistedApproval?.recommendation; - if (recommendation) { - authorizationRecommendations.push(recommendation); + ).permissionRequest?.toolCallId; + if (toolCallId) { + const waiter = recommendationWaiters.get(toolCallId); + if (waiter) { + recommendationWaiters.delete(toolCallId); + waiter(recommendation); + } else { + recommendationByToolCallId.set(toolCallId, recommendation); } - resolveAuthorizationRecommendation(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }) - .decisionSource; - if (source) authorizationDecisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - authorizationRecoveryStatuses.push( - (event.data as { status: string }).status - ); - } else if (event.type === "tool.execution_complete") { - authorizationToolResults.push({ - toolCallId: event.data.toolCallId, - success: event.data.success, - }); - } else if (event.type === "session.task_complete") { - authorizationTaskOutcomes.push(event.data); } - }); + if (recommendation) recommendations.push(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }).decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + recoveryStatuses.push((event.data as { status: string }).status); + } else if (event.type === "tool.execution_complete") { + toolResults.push(event.data); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); + } + }); - const taskComplete = getNextEventOfType( - authorizationSession, - "session.task_complete" - ); - enterPhase("authorization:send"); - await authorizationSession.send({ - prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + await session.send({ + prompt: + scenario.route === "shell" + ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` + : scenario.route === "path" + ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` + : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, }); - enterPhase("authorization:task-complete"); await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } - enterPhase("authorization:assert"); - expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); - expect(authorizationJudgeRequests).toHaveLength(1); - const authorizationRequest = authorizationJudgeRequests[0] ?? ""; + const expectedCompletions = + scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill(expectedRecommendation) + ); + expect(decisionSources, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill( + scenario.decision === "judge" ? "assisted_approval" : "human_response" + ) + ); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); + if (scenario.decision === "deny") { expect( - authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) - ).toBeGreaterThanOrEqual(0); + toolResults.some((result) => result.success === false), + JSON.stringify(scenario) + ).toBe(true); + expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); + } else { expect( - authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER) - ).toBeGreaterThan(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)); - expect(authorizationPermissionCallbacks).toBe(1); - expect(authorizationRecommendations).toEqual(["approve"]); - expect(authorizationDecisionSources).toEqual(["assisted_approval"]); - expect(authorizationRecoveryStatuses).toEqual([]); - expect(authorizationToolResults).toContainEqual({ - toolCallId: "assisted-authorization-shell", - success: true, - }); - expect(authorizationTaskOutcomes).toEqual([ - expect.objectContaining({ - success: true, - summary: "authorization-updated", - }), - ]); - expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); - } finally { - if (authorizationSession) { - enterPhase("authorization:cleanup"); - await authorizationSession.abort(); - await authorizationSession.disconnect(); - } + toolResults.every((result) => result.success === true), + JSON.stringify(scenario) + ).toBe(true); + } + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + success: true, + }); + expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( + scenario.route === "shell" + ? "shell-approved" + : scenario.route === "path" + ? "approval-probe.txt" + : `human-${scenario.decision === "approve" ? "approved" : "denied"}` + ); + if (scenario.route === "human") { + expect( + existsSync( + join( + workDir, + `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` + ) + ), + JSON.stringify(scenario) + ).toBe(scenario.decision === "approve"); + } + } finally { + if (session) { + await session.abort(); + await session.disconnect(); } } + } - const contractScenarios = [ - { scope: "root", recommendation: "approve", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "excluded", handler: "registered" }, - { scope: "subagent", recommendation: "approve", handler: "registered" }, - { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "none" }, - ] as const satisfies readonly { - scope: ContractScope; - recommendation: ContractRecommendation; - handler: "registered" | "none"; - }[]; - const contractJudgeStart = judgeOutputs.length; - for (const scenario of contract === "handler" ? contractScenarios : []) { - const label = `contract:${scenario.scope}:${scenario.recommendation}:${scenario.handler}`; - let permissionCallbacks = 0; - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const decisionSources: string[] = []; - const sequence: string[] = []; - const subagentIds = new Set(); - const permissionAgentIds: Array = []; - const toolResults: Array<{ - agentId?: string; - toolCallId?: string; - success?: boolean; - error?: { message?: string }; - result?: unknown; - }> = []; - const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; - const onPermissionRequest = () => { - permissionCallbacks++; - sequence.push("host"); - if (scenario.recommendation === "approve") { - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - } - if (scenario.recommendation === "requireApproval") { - return createAttributedPermissionResult( - { - kind: "reject", - feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), - }, - { - outcome: "autopilot_denied", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } + expect(providerFailures).toEqual([]); + expect(judgeOutputs).toEqual([ + ...Array(4).fill(JUDGE_OUTPUT), + ...Array(4).fill(HUMAN_REVIEW_OUTPUT), + ]); + expect(agentCalls).toBe(scenarios.length * 2 + 4); + + const authorizationJudgeStart = judgeOutputs.length; + let authorizationPermissionCallbacks = 0; + const authorizationRecommendations: string[] = []; + const authorizationDecisionSources: string[] = []; + const authorizationRecoveryStatuses: string[] = []; + const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = []; + const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = []; + let resolveAuthorizationRecommendation: (recommendation: string | undefined) => void; + const authorizationRecommendation = new Promise((resolve) => { + resolveAuthorizationRecommendation = resolve; + }); + let authorizationSession: CopilotSession | undefined; + try { + authorizationSession = await client.createSession({ + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + onPermissionRequest: async (request) => { + authorizationPermissionCallbacks++; + if (request.toolCallId !== "assisted-authorization-shell") { + throw new Error( + `Expected authorization shell permission, got ${String( + request.toolCallId + )}` ); } - if (scenario.recommendation === "excluded") { - return createAttributedPermissionResult( - { kind: "user-not-available" }, - { - outcome: "autopilot_denied", - source: "unattended_fallback", - surface: "sdk", - responseCapability: "headless", - } + const recommendation = await authorizationRecommendation; + if (recommendation !== "approve") { + throw new Error( + `Expected later authorization to reach the judge as approve, got ${String( + recommendation + )}` ); } - throw new Error("Unexpected Assisted permission recommendation"); - }; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), - } as const; - let session: CopilotSession | undefined; - try { - enterPhase(`${label}:create`); - session = await client.createSession(sessionConfig); - enterPhase(`${label}:configure`); - await configureAssistedAutopilot(session, workDir); - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; - } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - if (recommendation) { - recommendations.push(recommendation); - sequence.push(`permission:${recommendation}`); - } - permissionAgentIds.push(event.agentId); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }) - .decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - const status = (event.data as { status: string }).status; - recoveryStatuses.push(status); - sequence.push(`recovery:${status}`); - } else if (event.type === "tool.execution_complete") { - toolResults.push({ - agentId: event.agentId, - toolCallId: event.data.toolCallId, - success: event.data.success, - error: event.data.error, - result: event.data.result, - }); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); - } else if (event.type === "subagent.started") { - subagentIds.add(event.agentId); + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } + ); + }, + }); + await authorizationSession.sendAndWait({ + prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, + }); + await configureAssistedAutopilot(authorizationSession, workDir); + authorizationSession.on((event) => { + if (event.type === "permission.requested") { + const recommendation = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + }; } + ).promptRequest?.assistedApproval?.recommendation; + if (recommendation) { + authorizationRecommendations.push(recommendation); + } + resolveAuthorizationRecommendation(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }).decisionSource; + if (source) authorizationDecisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + authorizationRecoveryStatuses.push((event.data as { status: string }).status); + } else if (event.type === "tool.execution_complete") { + authorizationToolResults.push({ + toolCallId: event.data.toolCallId, + success: event.data.success, }); + } else if (event.type === "session.task_complete") { + authorizationTaskOutcomes.push(event.data); + } + }); - const marker = contractMarker(scenario.recommendation); - const prompt = - scenario.scope === "root" - ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` - : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - enterPhase(`${label}:send`); - await session.send({ prompt }); - enterPhase(`${label}:task-complete`); - await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } + const taskComplete = getNextEventOfType(authorizationSession, "session.task_complete"); + await authorizationSession.send({ + prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, + }); + await taskComplete; - enterPhase(`${label}:assert`); - if (scenario.handler === "none") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); - expect(recommendations, JSON.stringify(scenario)).toEqual([]); - } else if (scenario.recommendation === "approve") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); - } else if (scenario.recommendation === "requireApproval") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual([ - "requireApproval", - ]); - } else { - expect( - permissionCallbacks, - JSON.stringify(scenario) - ).toBeGreaterThanOrEqual(1); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(permissionCallbacks).fill("excluded") - ); - } - if (scenario.handler === "none") { - expect(decisionSources, JSON.stringify(scenario)).toEqual([]); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); - } else { - expect(decisionSources, JSON.stringify(scenario)).toContain( - scenario.recommendation === "excluded" - ? "unattended_fallback" - : "assisted_approval" - ); - } - if (scenario.recommendation === "excluded") { - expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); - } else if (scenario.handler === "registered") { - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - } - const hostIndex = sequence.indexOf("host"); - const firstRecoveryIndex = sequence.findIndex((entry) => - entry.startsWith("recovery:") - ); - if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { - expect( - hostIndex, - JSON.stringify({ scenario, sequence }) - ).toBeGreaterThanOrEqual(0); - expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( - firstRecoveryIndex - ); - } + expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); + expect(authorizationJudgeRequests).toHaveLength(1); + const authorizationRequest = authorizationJudgeRequests[0] ?? ""; + expect(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)).toBeGreaterThanOrEqual( + 0 + ); + expect(authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER)).toBeGreaterThan( + authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) + ); + expect(authorizationPermissionCallbacks).toBe(1); + expect(authorizationRecommendations).toEqual(["approve"]); + expect(authorizationDecisionSources).toEqual(["assisted_approval"]); + expect(authorizationRecoveryStatuses).toEqual([]); + expect(authorizationToolResults).toContainEqual({ + toolCallId: "assisted-authorization-shell", + success: true, + }); + expect(authorizationTaskOutcomes).toEqual([ + expect.objectContaining({ + success: true, + summary: "authorization-updated", + }), + ]); + expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); + } finally { + if (authorizationSession) { + await authorizationSession.abort(); + await authorizationSession.disconnect(); + } + } - const shellResult = toolResults.find( - (result) => - result.toolCallId === - contractShellCallId(scenario.scope, scenario.recommendation) + const contractScenarios = [ + { scope: "root", recommendation: "approve", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "excluded", handler: "registered" }, + { scope: "subagent", recommendation: "approve", handler: "registered" }, + { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "none" }, + ] as const satisfies readonly { + scope: ContractScope; + recommendation: ContractRecommendation; + handler: "registered" | "none"; + }[]; + const contractJudgeStart = judgeOutputs.length; + for (const scenario of contractScenarios) { + let permissionCallbacks = 0; + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const decisionSources: string[] = []; + const sequence: string[] = []; + const subagentIds = new Set(); + const permissionAgentIds: Array = []; + const toolResults: Array<{ + agentId?: string; + toolCallId?: string; + success?: boolean; + error?: { message?: string }; + result?: unknown; + }> = []; + const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; + const onPermissionRequest = () => { + permissionCallbacks++; + sequence.push("host"); + if (scenario.recommendation === "approve") { + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); - expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); - expect(shellResult?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation === "approve" && scenario.handler === "registered" + } + if (scenario.recommendation === "requireApproval") { + return createAttributedPermissionResult( + { + kind: "reject", + feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), + }, + { + outcome: "autopilot_denied", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); - if ( - scenario.recommendation === "requireApproval" && - scenario.handler === "registered" - ) { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Ask the registered human permission handler." - ); - } else if (scenario.handler === "none") { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Permission could not be granted automatically." - ); - } - if (scenario.scope === "subagent") { - expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); - expect( - subagentIds.has(shellResult?.agentId ?? ""), - JSON.stringify(scenario) - ).toBe(true); - for (const agentId of permissionAgentIds) { - expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe( - true - ); + } + if (scenario.recommendation === "excluded") { + return createAttributedPermissionResult( + { kind: "user-not-available" }, + { + outcome: "autopilot_denied", + source: "unattended_fallback", + surface: "sdk", + responseCapability: "headless", } + ); + } + throw new Error("Unexpected Assisted permission recommendation"); + }; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), + } as const; + let session: CopilotSession | undefined; + try { + session = await client.createSession(sessionConfig); + await configureAssistedAutopilot(session, workDir); + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; + } + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + if (recommendation) { + recommendations.push(recommendation); + sequence.push(`permission:${recommendation}`); + } + permissionAgentIds.push(event.agentId); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }).decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + const status = (event.data as { status: string }).status; + recoveryStatuses.push(status); + sequence.push(`recovery:${status}`); + } else if (event.type === "tool.execution_complete") { + toolResults.push({ + agentId: event.agentId, + toolCallId: event.data.toolCallId, + success: event.data.success, + error: event.data.error, + result: event.data.result, + }); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); + } else if (event.type === "subagent.started") { + subagentIds.add(event.agentId); } + }); + + const marker = contractMarker(scenario.recommendation); + const prompt = + scenario.scope === "root" + ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` + : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + await session.send({ prompt }); + await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } + if (scenario.handler === "none") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); + expect(recommendations, JSON.stringify(scenario)).toEqual([]); + } else if (scenario.recommendation === "approve") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); + } else if (scenario.recommendation === "requireApproval") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual(["requireApproval"]); + } else { + expect(permissionCallbacks, JSON.stringify(scenario)).toBeGreaterThanOrEqual(1); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(permissionCallbacks).fill("excluded") + ); + } + if (scenario.handler === "none") { + expect(decisionSources, JSON.stringify(scenario)).toEqual([]); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); + } else { + expect(decisionSources, JSON.stringify(scenario)).toContain( + scenario.recommendation === "excluded" + ? "unattended_fallback" + : "assisted_approval" + ); + } + if (scenario.recommendation === "excluded") { + expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); + } else if (scenario.handler === "registered") { + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + } + const hostIndex = sequence.indexOf("host"); + const firstRecoveryIndex = sequence.findIndex((entry) => + entry.startsWith("recovery:") + ); + if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { expect( - existsSync( - join( - workDir, - contractFixtureName(scenario.scope, scenario.recommendation) - ) - ), - JSON.stringify(scenario) - ).toBe( - scenario.recommendation === "approve" && scenario.handler === "registered" + hostIndex, + JSON.stringify({ scenario, sequence }) + ).toBeGreaterThanOrEqual(0); + expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( + firstRecoveryIndex + ); + } + + const shellResult = toolResults.find( + (result) => + result.toolCallId === + contractShellCallId(scenario.scope, scenario.recommendation) + ); + expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); + expect(shellResult?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation === "approve" && scenario.handler === "registered" + ); + if ( + scenario.recommendation === "requireApproval" && + scenario.handler === "registered" + ) { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Ask the registered human permission handler." + ); + } else if (scenario.handler === "none") { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Permission could not be granted automatically." ); + } + if (scenario.scope === "subagent") { + expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); expect( - (await session.rpc.permissions.pendingRequests()).items, + subagentIds.has(shellResult?.agentId ?? ""), JSON.stringify(scenario) - ).toEqual([]); - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation !== "excluded" && scenario.handler === "registered" - ); - if (scenario.handler === "none") { - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - outcome: "continue", - reason: "Autopilot is still recovering from a required permission.", - }); - } - } finally { - if (session) { - enterPhase(`${label}:cleanup`); - await session.abort(); - await session.disconnect(); + ).toBe(true); + for (const agentId of permissionAgentIds) { + expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe(true); } } - } - enterPhase("final-assertions"); - expect(providerFailures).toEqual([]); - if (contract === "handler") { - expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - HUMAN_REVIEW_OUTPUT, - ]); + expect( + existsSync( + join(workDir, contractFixtureName(scenario.scope, scenario.recommendation)) + ), + JSON.stringify(scenario) + ).toBe(scenario.recommendation === "approve" && scenario.handler === "registered"); + expect( + (await session.rpc.permissions.pendingRequests()).items, + JSON.stringify(scenario) + ).toEqual([]); + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation !== "excluded" && scenario.handler === "registered" + ); + if (scenario.handler === "none") { + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + outcome: "continue", + reason: "Autopilot is still recovering from a required permission.", + }); + } + } finally { + if (session) { + await session.abort(); + await session.disconnect(); + } } - console.info( - "Assisted contract completed:", - JSON.stringify({ - contract, - elapsedMs: Date.now() - started, - cleanupMs: completedPhases - .filter(({ phase }) => /cleanup|disconnect-before-resume/.test(phase)) - .reduce((total, { elapsedMs }) => total + elapsedMs, 0), - agentCalls, - judgeCalls: judgeOutputs.length, - providerFailures: providerFailures.length, - }) - ); } - ); + + expect(providerFailures).toEqual([]); + expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + HUMAN_REVIEW_OUTPUT, + ]); + }); }); async function configureAssistedAutopilot(session: CopilotSession, workDir: string): Promise { diff --git a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts index d23ca0446b..e6e6a18c8f 100644 --- a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts +++ b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts @@ -2,7 +2,7 @@ * Copyright (c) Microsoft Corporation. All rights reserved. *--------------------------------------------------------------------------------------------*/ -import { describe, expect, it, onTestFailed } from "vitest"; +import { describe, expect, it } from "vitest"; import { z } from "zod"; import { approveAll, CopilotRequestHandler } from "../../src/index.js"; import type { SessionEvent, CopilotRequestContext, CopilotSession } from "../../src/index.js"; @@ -390,59 +390,25 @@ describe("Session tasks RPC and pending handlers", async () => { { timeout: 120_000 }, async () => { await withRunningAttachedShell(async (session, shellId, eventTypes, replies) => { - const started = performance.now(); - let phase = "enqueue"; - let queuePolls = 0; - let pendingCount = 0; - let queuedMatch = false; - onTestFailed(() => { - console.error( - "Attached shell queue failure boundary:", - JSON.stringify({ - phase, - elapsedMs: Math.round(performance.now() - started), - queuePolls, - pendingCount, - queuedMatch, - replyCount: replies.length, - hasQueuedReply: replies.some((reply) => reply.includes("QUEUED_DONE")), - sessionIdleCount: eventTypes.filter((type) => type === "session.idle") - .length, - assistantIdleCount: eventTypes.filter( - (type) => type === "assistant.idle" - ).length, - }) - ); - }); // The running attached shell holds idle, so an enqueued message waits behind it. await session.send({ prompt: "Reply with exactly QUEUED_DONE.", mode: "enqueue" }); - phase = "pending-queue"; await waitForCondition( - async () => { - const { items } = await session.rpc.queue.pendingItems(); - queuePolls++; - pendingCount = items.length; - queuedMatch = items.some((item) => + async () => + (await session.rpc.queue.pendingItems()).items.some((item) => item.displayText.includes("QUEUED_DONE") - ); - return queuedMatch; - }, + ), { timeoutMessage: "The enqueued message was not parked behind the shell" } ); - phase = "assert-parked"; expect(replies.some((reply) => reply.includes("QUEUED_DONE"))).toBe(false); expect(eventTypes).not.toContain("session.idle"); - phase = "cancel-shell"; expect((await session.rpc.tasks.cancel({ id: shellId })).cancelled).toBe(true); - phase = "queued-reply"; await waitForCondition( () => replies.some((reply) => reply.includes("QUEUED_DONE")), { timeoutMessage: `The queued message never ran after cancelling shell ${shellId}`, } ); - phase = "session-idle"; await waitForCondition(() => eventTypes.includes("session.idle"), { timeoutMessage: "session.idle never followed the queued message", }); diff --git a/nodejs/test/rust-codegen.test.ts b/nodejs/test/rust-codegen.test.ts index 71c31fed5f..53c471b104 100644 --- a/nodejs/test/rust-codegen.test.ts +++ b/nodejs/test/rust-codegen.test.ts @@ -1,6 +1,5 @@ import type { ApiSchema } from "../../scripts/codegen/utils.ts"; import { - getApiSchemaPath, normalizeSchemaBrandCasing, postProcessSchema, propagateInternalVisibility, @@ -1016,13 +1015,18 @@ describe("Rust x-legacy-parameters", () => { ).toThrow(/Rust string enum Kind is requested for different values/); }); - it("keeps every const discriminator of the selected API schema distinct in Rust", async () => { + it("keeps every const discriminator of the committed API schema distinct in Rust", () => { // Mirror the generator's own schema preparation so emission order matches. const schema = propagateInternalVisibility( postProcessSchema( stripBooleanLiterals( normalizeSchemaBrandCasing( - JSON.parse(readFileSync(await getApiSchemaPath(), "utf8")) as ApiSchema + JSON.parse( + readFileSync( + new URL("../../../../generated/api.schema.json", import.meta.url), + "utf8" + ) + ) as ApiSchema ) ) as JSONSchema7 ) diff --git a/scripts/ci/device-policy-bootstrap.mjs b/scripts/ci/device-policy-bootstrap.mjs deleted file mode 100644 index 48e30268e8..0000000000 --- a/scripts/ci/device-policy-bootstrap.mjs +++ /dev/null @@ -1,85 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import fs from "node:fs"; -import path from "node:path"; -import { fileURLToPath } from "node:url"; -import { POLICY_FILE, verifyIdentity, verifyIsolation } from "./device-policy-fixture.mjs"; - -function start() { - const plan = JSON.parse(process.env.COPILOT_CI_DEVICE_POLICY_PLAN); - if (process.platform !== "linux" || typeof process.execve !== "function") - throw new Error("Unsupported device-policy container runtime"); - if (process.argv[2] === "--runtime") { - const policy = fs.lstatSync(POLICY_FILE); - verifyIdentity(plan, { - uid: process.getuid(), - gid: process.getgid(), - groups: process.getgroups(), - policy: { - isFile: policy.isFile(), - isSymbolicLink: policy.isSymbolicLink(), - uid: policy.uid, - gid: policy.gid, - mode: policy.mode, - }, - }); - process.execve(plan.executable, plan.argv, plan.env); - throw new Error("Runtime exec unexpectedly returned"); - } - verifyIsolation({ - originalNamespace: plan.originalNamespace, - namespace: fs.readlinkSync("/proc/self/ns/mnt"), - mountinfo: fs.readFileSync("/proc/self/mountinfo", "utf8"), - uid: process.getuid(), - gid: process.getgid(), - }); - if ( - !Number.isSafeInteger(plan.uid) || - plan.uid <= 0 || - !Number.isSafeInteger(plan.gid) || - plan.gid < 0 || - !plan.groups.every((group) => Number.isSafeInteger(group) && group >= 0) - ) - throw new Error("Invalid runner identity"); - const directory = path.dirname(POLICY_FILE); - if (fs.existsSync(directory)) throw new Error("The container already has a device-policy directory"); - fs.mkdirSync(directory, { mode: 0o755 }); - fs.writeFileSync(POLICY_FILE, plan.policy, { mode: 0o644, flag: "wx" }); - fs.chmodSync(POLICY_FILE, 0o644); - - const passwd = fs.readFileSync("/etc/passwd", "utf8"); - if (!passwd.split("\n").some((line) => Number(line.split(":")[2]) === plan.uid)) { - if (/[:\r\n]/.test(plan.home)) throw new Error("Invalid runner home"); - fs.appendFileSync("/etc/passwd", `copilot-sdk-ci:x:${plan.uid}:${plan.gid}::${plan.home}:/bin/sh\n`); - } - const group = fs.readFileSync("/etc/group", "utf8"); - if (!group.split("\n").some((line) => Number(line.split(":")[2]) === plan.gid)) { - fs.appendFileSync("/etc/group", `copilot-sdk-ci:x:${plan.gid}:\n`); - } - if (!fs.existsSync(plan.home)) { - fs.mkdirSync(plan.home, { recursive: true, mode: 0o700 }); - } - fs.chownSync(plan.home, plan.uid, plan.gid); - const args = [ - "/usr/bin/setpriv", - `--reuid=${plan.uid}`, - `--regid=${plan.gid}`, - ...(plan.groups.length ? [`--groups=${plan.groups.join(",")}`] : ["--clear-groups"]), - plan.nodeExecutable, - fileURLToPath(import.meta.url), - "--runtime", - ]; - process.execve("/usr/bin/setpriv", args, process.env); - throw new Error("Identity transition unexpectedly returned"); -} - -if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { - try { - start(); - } catch (error) { - console.error(error.message); - process.exitCode = 1; - } -} diff --git a/scripts/ci/device-policy-fixture.mjs b/scripts/ci/device-policy-fixture.mjs deleted file mode 100644 index 7c65a71354..0000000000 --- a/scripts/ci/device-policy-fixture.mjs +++ /dev/null @@ -1,468 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import { spawn, spawnSync } from "node:child_process"; -import { randomUUID } from "node:crypto"; -import fs from "node:fs"; -import os from "node:os"; -import path from "node:path"; -import { fileURLToPath } from "node:url"; - -const SDK_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); -export const IMAGE = "node:22-bookworm"; -export const POLICY_FILE = "/etc/github-copilot/managed-settings.json"; -export const OWNER_LABEL = "com.github.copilot-sdk.device-policy"; -export const CLIENT_LABEL = `${OWNER_LABEL}.client`; -export const REQUIRED_CASES = [ - { - file: "managed_plugin_progress.e2e.test.ts", - title: "emits presentation-neutral completion after installing required plugins", - }, - { - file: "rpc_server.e2e.test.ts", - title: "should round trip sessionless managed settings", - }, -]; - -const escapeRegex = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); -export const REQUIRED_PATTERN = REQUIRED_CASES.map(({ title }) => escapeRegex(title)).join("|"); -export const PORTABLE_PATTERN = `^(?!.*(?:${REQUIRED_PATTERN})).*$`; - -export function portableTestPattern(source) { - if (source === "checkout") return ""; - if (source !== "published") throw new Error("Unknown runtime artifact source"); - return PORTABLE_PATTERN; -} - -export function runtimeInvocation(kind, runtime, nodeExecutable, args, env) { - if (!["native", "legacy"].includes(kind)) throw new Error("Unknown device-policy entry point"); - return { - executable: kind === "native" ? runtime : nodeExecutable, - argv: kind === "native" ? [runtime, ...args] : [nodeExecutable, runtime, ...args], - env: { ...env }, - }; -} - -export function verifyRequiredResults(report) { - for (const expected of REQUIRED_CASES) { - const matches = (report.testResults ?? []) - .filter((suite) => suite.name.replaceAll("\\", "/").endsWith(`/${expected.file}`)) - .flatMap((suite) => suite.assertionResults ?? []) - .filter((test) => test.title === expected.title); - if (matches.length !== 1 || matches[0].status !== "passed") { - throw new Error(`Required device-policy control did not pass: ${expected.title}`); - } - } -} - -function inside(parent, child) { - const relative = path.posix.relative(parent, child); - return relative === "" || (relative !== ".." && !relative.startsWith("../") && !path.posix.isAbsolute(relative)); -} - -function safeBind(directory) { - if (!path.posix.isAbsolute(directory) || /[:,\r\n]/.test(directory)) { - throw new Error("Device-policy binds require unambiguous absolute Linux paths"); - } - if ( - ["/", "/etc", "/proc", "/sys", "/dev"].some((root) => - root === "/" ? directory === root : inside(root, directory), - ) - ) { - throw new Error("A device-policy container cannot bind host system or policy paths"); - } - return directory; -} - -export function fixtureDirectories({ cwd, env, temporaryDirectory }) { - const candidates = [ - cwd, - ...["COPILOT_HOME", "GH_CONFIG_DIR", "XDG_CONFIG_HOME", "XDG_STATE_HOME"] - .map((name) => env[name]) - .filter(Boolean), - ]; - const roots = new Set(); - for (const candidate of candidates) { - const relative = path.posix.relative(temporaryDirectory, candidate); - const first = relative.split("/")[0]; - if (!inside(temporaryDirectory, candidate) || !/^copilot-test-(work|home|config)-[^/]+$/.test(first)) { - throw new Error("Writable device-policy binds must belong to an SDK E2E fixture"); - } - roots.add(safeBind(path.posix.join(temporaryDirectory, first))); - } - return [...roots]; -} - -export function dockerArguments(plan, cidfile) { - const args = [ - "run", - "--rm", - "--init", - "--interactive", - "--network", - "host", - "--cidfile", - cidfile, - "--label", - `${OWNER_LABEL}=${plan.owner}`, - "--label", - `${CLIENT_LABEL}=${plan.client}`, - "--workdir", - plan.cwd, - "--env", - "COPILOT_CI_DEVICE_POLICY_PLAN", - ]; - const mounts = new Map(); - for (const directory of plan.readonly) mounts.set(safeBind(directory), "ro"); - for (const directory of plan.writable) { - safeBind(directory); - if (mounts.has(directory)) throw new Error("A device-policy bind cannot be both read-only and writable"); - mounts.set(directory, "rw"); - } - for (const [directory, mode] of mounts) args.push("--volume", `${directory}:${directory}:${mode}`); - args.push(IMAGE, "node", path.posix.join(plan.sdkRoot, "scripts/ci/device-policy-bootstrap.mjs")); - return args; -} - -export function verifyIsolation({ originalNamespace, namespace, mountinfo, uid, gid }) { - if (uid !== 0 || gid !== 0 || namespace === originalNamespace || !namespace || !originalNamespace) { - throw new Error("Device policy must be provisioned inside a separate root container"); - } - const mounts = mountinfo - .trim() - .split("\n") - .map((line) => { - const fields = line.split(" "); - const separator = fields.indexOf("-"); - return { target: fields[4], type: fields[separator + 1], propagation: fields.slice(6, separator) }; - }); - const policyMount = mounts - .filter(({ target }) => inside(target, POLICY_FILE)) - .sort((a, b) => b.target.length - a.target.length)[0]; - if ( - !policyMount || - policyMount.target !== "/" || - policyMount.type !== "overlay" || - policyMount.propagation.some((field) => /^(shared|master|propagate_from):/.test(field)) - ) { - throw new Error("Device policy requires a private container root, not a host bind"); - } -} - -export function verifyIdentity(plan, { uid, gid, groups, policy }) { - if ( - uid !== plan.uid || - gid !== plan.gid || - JSON.stringify([...groups].sort((a, b) => a - b)) !== JSON.stringify([...plan.groups].sort((a, b) => a - b)) - ) { - throw new Error("The device-policy runtime must retain the original runner identity"); - } - if ( - !policy.isFile || - policy.isSymbolicLink || - policy.uid !== 0 || - policy.gid !== 0 || - (policy.mode & 0o777) !== 0o644 - ) { - throw new Error("The device policy must be a regular root-owned mode-0644 file"); - } -} - -function command(command, args, options = {}) { - const result = spawnSync(command, args, { encoding: "utf8", ...options }); - if (result.error) throw result.error; - if (result.status !== 0) throw new Error(`${command} failed (${result.status}): ${result.stderr ?? ""}`); - return result.stdout; -} - -export function containerId(value) { - const id = value.trim(); - if (!/^[a-f0-9]{64}$/.test(id)) throw new Error("Invalid owned device-policy container ID"); - return id; -} - -export function verifyOwnership(label, owner) { - if (!owner || label.trim() !== owner) throw new Error("Refusing to mutate a container owned by another job"); -} - -export function pendingLaunches(entries) { - for (const entry of entries) { - if (!/^[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}\.(pending|cid)$/.test(entry)) - throw new Error("Unknown device-policy cleanup registry entry"); - const pending = entry.replace(/\.cid$/, ".pending"); - if (!entries.includes(pending)) throw new Error("Device-policy container identity has no launch receipt"); - } - return entries.filter((entry) => entry.endsWith(".pending")); -} - -function inspectOwned(id, owner) { - const inspected = spawnSync("docker", ["inspect", "--format", `{{index .Config.Labels "${OWNER_LABEL}"}}`, id], { - encoding: "utf8", - }); - if (inspected.status !== 0 && /No such (object|container)/i.test(inspected.stderr)) return false; - if (inspected.error || inspected.status !== 0) - throw inspected.error ?? new Error(`Cannot inspect device-policy container: ${inspected.stderr}`); - verifyOwnership(inspected.stdout, owner); - return true; -} - -function removeOwned(id, owner) { - if (!inspectOwned(id, owner)) return; - const removed = spawnSync("docker", ["rm", "--force", id], { encoding: "utf8" }); - if (removed.status !== 0 && /No such (object|container)/i.test(removed.stderr)) return; - if (removed.error || removed.status !== 0) - throw removed.error ?? new Error(`Cannot remove device-policy container: ${removed.stderr}`); -} - -function stopOwned(id, owner, signal) { - if (!inspectOwned(id, owner)) return; - const stopped = spawnSync("docker", ["stop", "--signal", signal, "--timeout", "5", id], { encoding: "utf8" }); - if (stopped.status !== 0 && /No such (object|container)/i.test(stopped.stderr)) return; - if (stopped.error || stopped.status !== 0) - throw stopped.error ?? new Error(`Cannot stop device-policy container: ${stopped.stderr}`); -} - -function ownedRegistry() { - const registry = process.env.COPILOT_CI_DEVICE_POLICY_REGISTRY; - if (!registry || !process.env.RUNNER_TEMP) throw new Error("Missing owned device-policy container registry"); - const info = fs.lstatSync(registry); - if ( - !inside(process.env.RUNNER_TEMP, registry) || - !path.basename(registry).startsWith("sdk-device-policy-") || - fs.realpathSync(registry) !== registry || - !info.isDirectory() || - info.isSymbolicLink() || - info.uid !== process.getuid() - ) { - throw new Error("Device-policy cidfiles must belong to this runner's temporary registry"); - } - return registry; -} - -export async function launchRuntime(kind) { - if (process.platform !== "linux" || typeof process.execve !== "function") { - throw new Error("The device-policy launcher requires Linux and Node.js 22.15 or newer"); - } - const runtime = - process.env[ - kind === "native" ? "COPILOT_CI_DEVICE_POLICY_NATIVE_PATH" : "COPILOT_CI_DEVICE_POLICY_LEGACY_PATH" - ]; - if (!runtime) throw new Error("Missing original device-policy runtime path"); - const invocation = runtimeInvocation(kind, runtime, process.execPath, process.argv.slice(2), process.env); - const originalEnv = invocation.env; - const fixture = process.env.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; - if (!fixture) { - process.execve(invocation.executable, invocation.argv, originalEnv); - throw new Error("Original runtime exec unexpectedly returned"); - } - - if (!inside(process.cwd(), fixture) || fs.realpathSync(fixture) !== fixture) - throw new Error("Device policy must be a regular file inside the test workspace"); - const descriptor = fs.openSync(fixture, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW); - let policy; - try { - const fixtureInfo = fs.fstatSync(descriptor); - if (!fixtureInfo.isFile() || fixtureInfo.uid !== process.getuid()) - throw new Error("Device policy must be a regular file owned by the runner"); - policy = fs.readFileSync(descriptor, "utf8"); - } finally { - fs.closeSync(descriptor); - } - const parsed = JSON.parse(policy); - if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) - throw new Error("Device policy must be a JSON object"); - // The launcher consumes the test-only path; the runtime must discover the production file. - delete originalEnv.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; - const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; - if (!owner) throw new Error("Missing device-policy job ownership"); - const registry = ownedRegistry(); - const writable = fixtureDirectories({ cwd: process.cwd(), env: originalEnv, temporaryDirectory: os.tmpdir() }); - for (const directory of writable) { - const info = fs.lstatSync(directory); - if (fs.realpathSync(directory) !== directory || !info.isDirectory() || info.uid !== process.getuid()) - throw new Error("A writable fixture bind must be a real directory owned by the runner"); - } - const readonly = [ - SDK_ROOT, - path.dirname(process.env.COPILOT_CI_DEVICE_POLICY_LEGACY_PATH), - path.dirname(process.execPath), - ]; - for (const name of [ - "NODE_EXTRA_CA_CERTS", - "SSL_CERT_FILE", - "REQUESTS_CA_BUNDLE", - "CURL_CA_BUNDLE", - "GIT_SSL_CAINFO", - ]) { - const certificate = originalEnv[name]; - if (certificate) { - if (!inside(os.tmpdir(), certificate) || !fs.statSync(certificate).isFile()) - throw new Error("Replay certificates must belong to the test temporary directory"); - readonly.push(certificate); - } - } - const plan = { - owner, - client: randomUUID(), - sdkRoot: SDK_ROOT, - readonly: [...new Set(readonly)], - writable, - policy, - ...invocation, - cwd: process.cwd(), - nodeExecutable: process.execPath, - uid: process.getuid(), - gid: process.getgid(), - groups: process.getgroups(), - home: os.homedir(), - originalNamespace: fs.readlinkSync("/proc/self/ns/mnt"), - }; - const cidfile = path.join(registry, `${plan.client}.cid`); - const pending = path.join(registry, `${plan.client}.pending`); - fs.writeFileSync(pending, "", { flag: "wx", mode: 0o600 }); - const child = spawn("docker", dockerArguments(plan, cidfile), { - stdio: "inherit", - env: { ...originalEnv, COPILOT_CI_DEVICE_POLICY_PLAN: JSON.stringify(plan) }, - }); - let stopping; - const stop = (signal) => { - if (stopping) return; - stopping = signal; - try { - if (fs.existsSync(cidfile)) stopOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner, signal); - else child.kill(signal); - } catch (error) { - console.error(error.message); - process.exitCode = 1; - child.kill("SIGTERM"); - } - }; - const onTerm = () => stop("SIGTERM"); - const onInt = () => stop("SIGINT"); - process.once("SIGTERM", onTerm); - process.once("SIGINT", onInt); - let result; - try { - result = await new Promise((resolve, reject) => { - child.once("error", reject); - child.once("close", (code, signal) => resolve({ code, signal })); - }); - } finally { - process.removeListener("SIGTERM", onTerm); - process.removeListener("SIGINT", onInt); - // A cancelled Docker attach can close before its cidfile is written. - const ids = command("docker", [ - "ps", - "--all", - "--quiet", - "--no-trunc", - "--filter", - `label=${OWNER_LABEL}=${owner}`, - "--filter", - `label=${CLIENT_LABEL}=${plan.client}`, - ]).trim(); - for (const id of ids ? ids.split("\n") : []) removeOwned(containerId(id), owner); - if (fs.existsSync(cidfile)) { - removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); - fs.unlinkSync(cidfile); - fs.unlinkSync(pending); - } else { - throw new Error("Docker launch has no container identity; cleanup cannot prove settlement"); - } - } - if (process.exitCode) return; - if (stopping || result.signal) process.kill(process.pid, stopping ?? result.signal); - else process.exitCode = result.code ?? 1; -} - -function prepare() { - if ( - process.platform !== "linux" || - !process.env.GITHUB_ENV || - !process.env.GITHUB_OUTPUT || - !process.env.RUNNER_TEMP - ) { - throw new Error("Device-policy containers can only be prepared in Linux CI"); - } - const native = fs.realpathSync(process.env.COPILOT_CLI_PATH); - const legacy = fs.realpathSync(process.env.COPILOT_LEGACY_CLI_PATH); - const packageRoot = path.dirname(legacy); - if ( - native !== path.join(packageRoot, "prebuilds/linux-x64/copilot-runtime") || - path.basename(legacy) !== "app.js" - ) { - throw new Error("Device-policy fixtures require the staged GNU package's two original entry points"); - } - const expected = JSON.parse(fs.readFileSync(path.join(SDK_ROOT, "nodejs/package.json"), "utf8")).copilotCliVersion; - const actual = JSON.parse(fs.readFileSync(path.join(packageRoot, "package.json"), "utf8")).version; - if (expected !== actual) throw new Error("Device-policy package differs from the pinned CLI version"); - if (fs.existsSync(POLICY_FILE)) throw new Error("The CI host already has a device policy"); - command("docker", ["pull", IMAGE], { stdio: "inherit" }); - const registry = fs.mkdtempSync(path.join(process.env.RUNNER_TEMP, "sdk-device-policy-")); - const owner = randomUUID(); - fs.appendFileSync( - process.env.GITHUB_ENV, - `COPILOT_CI_DEVICE_POLICY_NATIVE_PATH=${native}\nCOPILOT_CI_DEVICE_POLICY_LEGACY_PATH=${legacy}\nCOPILOT_CI_DEVICE_POLICY_OWNER=${owner}\nCOPILOT_CI_DEVICE_POLICY_REGISTRY=${registry}\n`, - ); - fs.appendFileSync( - process.env.GITHUB_OUTPUT, - `native-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-native.js")}\nlegacy-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-legacy.js")}\n`, - ); -} - -function cleanup() { - const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; - if (!owner) throw new Error("Missing device-policy job ownership"); - const ids = command("docker", [ - "ps", - "--all", - "--quiet", - "--no-trunc", - "--filter", - `label=${OWNER_LABEL}=${owner}`, - ]).trim(); - const errors = []; - for (const id of ids ? ids.split("\n") : []) { - try { - removeOwned(containerId(id), owner); - } catch (error) { - errors.push(error.message); - } - } - const registry = ownedRegistry(); - for (const entry of pendingLaunches(fs.readdirSync(registry))) { - const pending = path.join(registry, entry); - const cidfile = path.join(registry, entry.replace(/\.pending$/, ".cid")); - try { - if (!fs.existsSync(cidfile)) throw new Error("Unsettled Docker launch has no container identity"); - removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); - fs.unlinkSync(cidfile); - fs.unlinkSync(pending); - } catch (error) { - errors.push(error.message); - } - } - if (errors.length) throw new Error(`Device-policy cleanup did not settle: ${errors.join("; ")}`); - if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); -} - -if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { - try { - const [action, argument] = process.argv.slice(2); - if (action === "prepare") prepare(); - else if (action === "cleanup") cleanup(); - else if (action === "portable-pattern") console.log(portableTestPattern(argument)); - else if (action === "required-pattern") console.log(REQUIRED_PATTERN); - else if (action === "verify" && argument) { - verifyRequiredResults(JSON.parse(fs.readFileSync(argument, "utf8"))); - if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); - } else - throw new Error( - "Usage: device-policy-fixture.mjs ", - ); - } catch (error) { - console.error(error.message); - process.exitCode = 1; - } -} diff --git a/scripts/ci/device-policy-fixture.test.mjs b/scripts/ci/device-policy-fixture.test.mjs deleted file mode 100644 index 8f33a412c3..0000000000 --- a/scripts/ci/device-policy-fixture.test.mjs +++ /dev/null @@ -1,207 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import assert from "node:assert/strict"; -import test from "node:test"; -import "./device-policy-bootstrap.mjs"; -import { - CLIENT_LABEL, - containerId, - dockerArguments, - fixtureDirectories, - OWNER_LABEL, - pendingLaunches, - POLICY_FILE, - portableTestPattern, - runtimeInvocation, - PORTABLE_PATTERN, - REQUIRED_CASES, - REQUIRED_PATTERN, - verifyIdentity, - verifyIsolation, - verifyOwnership, - verifyRequiredResults, -} from "./device-policy-fixture.mjs"; - -const report = () => ({ - testResults: REQUIRED_CASES.map(({ file, title }) => ({ - name: `/workspace/nodejs/test/e2e/${file}`, - assertionResults: [{ title, status: "passed" }], - })), -}); - -test("requires both unchanged strict controls to pass", () => { - verifyRequiredResults(report()); - for (const index of [0, 1]) { - for (const status of ["pending", "skipped", "failed", "todo"]) { - const result = report(); - result.testResults[index].assertionResults[0].status = status; - assert.throws(() => verifyRequiredResults(result), /did not pass/); - } - const missing = report(); - missing.testResults.splice(index, 1); - assert.throws(() => verifyRequiredResults(missing), /did not pass/); - const duplicate = report(); - duplicate.testResults.push(duplicate.testResults[index]); - assert.throws(() => verifyRequiredResults(duplicate), /did not pass/); - } - assert.throws(() => verifyRequiredResults({}), /did not pass/); -}); - -test("assigns only the two device-fixture cases to the mandatory Linux gate", () => { - assert.equal(portableTestPattern("checkout"), ""); - assert.equal(portableTestPattern("published"), PORTABLE_PATTERN); - assert.throws(() => portableTestPattern(""), /Unknown/); - const portable = new RegExp(PORTABLE_PATTERN); - const required = new RegExp(REQUIRED_PATTERN); - for (const { title } of REQUIRED_CASES) { - assert.equal(portable.test(`Suite ${title}`), false); - assert.equal(required.test(title), true); - } - for (const title of [ - "should clear the managed settings cache", - "should expose the managed settings schema", - "should list server sessions", - ]) { - assert.equal(portable.test(title), true); - assert.equal(required.test(title), false); - } -}); - -test("preserves both runtime entry points' original argv and environment", () => { - const args = ["--stdio", "--argument-with-spaces=one two", "--literal=;$()"]; - const env = { TOKEN: "private-value", COPILOT_HOME: "/tmp/copilot-test-home-one" }; - const native = runtimeInvocation("native", "/package/copilot-runtime", "/node/bin/node", args, env); - assert.equal(native.executable, "/package/copilot-runtime"); - assert.deepEqual(native.argv, ["/package/copilot-runtime", ...args]); - const legacy = runtimeInvocation("legacy", "/package/app.js", "/node/bin/node", args, env); - assert.equal(legacy.executable, "/node/bin/node"); - assert.deepEqual(legacy.argv, ["/node/bin/node", "/package/app.js", ...args]); - assert.deepEqual(native.env, env); - assert.notEqual(native.env, env); - assert.throws(() => runtimeInvocation("unknown", "", "", [], {}), /entry point/); -}); - -test("binds only explicitly owned E2E fixture directories for writing", () => { - const roots = fixtureDirectories({ - cwd: "/tmp/copilot-test-work-one", - env: { - COPILOT_HOME: "/tmp/copilot-test-home-one", - GH_CONFIG_DIR: "/tmp/copilot-test-config-one", - XDG_CONFIG_HOME: "/tmp/copilot-test-config-one", - XDG_STATE_HOME: "/tmp/copilot-test-work-one/local-home", - }, - temporaryDirectory: "/tmp", - }); - assert.deepEqual(roots, [ - "/tmp/copilot-test-work-one", - "/tmp/copilot-test-home-one", - "/tmp/copilot-test-config-one", - ]); - for (const cwd of ["/etc", "/home/runner", "/tmp", "/tmp/../../etc", "/tmp/not-a-fixture"]) { - assert.throws(() => fixtureDirectories({ cwd, env: {}, temporaryDirectory: "/tmp" }), /fixture/); - } -}); - -test("keeps policy and secret environment contents out of Docker arguments", () => { - const plan = { - owner: "job-one", - client: "client-one", - sdkRoot: "/workspace", - cwd: "/tmp/copilot-test-work-one", - readonly: ["/workspace", "/artifacts/package"], - writable: ["/tmp/copilot-test-work-one"], - env: { SECRET: "must-not-appear" }, - policy: '{"model":"secret-model"}', - }; - const args = dockerArguments(plan, "/registry/client-one.cid"); - assert.deepEqual(args.slice(0, 7), ["run", "--rm", "--init", "--interactive", "--network", "host", "--cidfile"]); - assert.equal(args.includes(`${OWNER_LABEL}=job-one`), true); - assert.equal(args.includes(`${CLIENT_LABEL}=client-one`), true); - assert.equal(args.includes("COPILOT_CI_DEVICE_POLICY_PLAN"), true); - assert.equal(args.includes("/artifacts/package:/artifacts/package:ro"), true); - assert.equal(args.includes("/tmp/copilot-test-work-one:/tmp/copilot-test-work-one:rw"), true); - assert.equal(args.join(" ").includes("must-not-appear"), false); - assert.equal(args.join(" ").includes("secret-model"), false); - const other = dockerArguments({ ...plan, client: "client-two" }, "/registry/client-two.cid"); - assert.notDeepEqual(args, other); - for (const directory of ["/", "/etc", "/etc/github-copilot", "/proc", "/sys", "/dev", "/path:ambiguous"]) { - assert.throws(() => dockerArguments({ ...plan, readonly: [directory] }, "/registry/client.cid")); - } - assert.throws(() => dockerArguments({ ...plan, readonly: plan.writable }, "/registry/client.cid"), /both/); -}); - -test("requires exact job ownership and full container IDs before cleanup", () => { - verifyOwnership("job-one\n", "job-one"); - for (const [label, owner] of [ - ["job-two", "job-one"], - ["", ""], - ["", "job-one"], - ]) { - assert.throws(() => verifyOwnership(label, owner), /another job/); - } - assert.equal(containerId(`${"a".repeat(64)}\n`), "a".repeat(64)); - for (const id of ["", "a".repeat(12), `${"a".repeat(64)} extra`, "G".repeat(64)]) { - assert.throws(() => containerId(id), /container ID/); - } -}); - -test("exposes interrupted or unknown launch receipts instead of claiming cleanup", () => { - const client = "12345678-1234-1234-1234-123456789abc"; - assert.deepEqual(pendingLaunches([]), []); - assert.deepEqual(pendingLaunches([`${client}.pending`, `${client}.cid`]), [`${client}.pending`]); - assert.deepEqual(pendingLaunches([`${client}.pending`]), [`${client}.pending`]); - assert.throws(() => pendingLaunches([`${client}.cid`]), /no launch receipt/); - assert.throws(() => pendingLaunches(["unknown"]), /Unknown/); -}); - -const isolated = () => ({ - originalNamespace: "mnt:[100]", - namespace: "mnt:[200]", - uid: 0, - gid: 0, - mountinfo: "1 0 0:1 / / rw - overlay overlay rw\n2 1 0:2 / /workspace ro - ext4 disk ro", -}); - -test("requires private container backing before any policy mutation", () => { - verifyIsolation(isolated()); - for (const change of [ - { namespace: "mnt:[100]" }, - { uid: 1001 }, - { gid: 1001 }, - { originalNamespace: "" }, - { mountinfo: "1 0 0:1 / / rw shared:1 - overlay overlay rw" }, - { mountinfo: "1 0 0:1 / / rw - ext4 disk rw" }, - { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc rw - ext4 host rw` }, - { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc/github-copilot rw - ext4 host rw` }, - ]) - assert.throws(() => verifyIsolation({ ...isolated(), ...change })); -}); - -test("retains runner uid, gid and groups with an ordinary root-owned device file", () => { - const plan = { uid: 1001, gid: 1001, groups: [1001, 118] }; - const runtime = { - uid: 1001, - gid: 1001, - groups: [118, 1001], - policy: { isFile: true, isSymbolicLink: false, uid: 0, gid: 0, mode: 0o100644 }, - }; - verifyIdentity(plan, runtime); - for (const change of [{ uid: 0 }, { gid: 0 }, { groups: [] }]) { - assert.throws(() => verifyIdentity(plan, { ...runtime, ...change }), /identity/); - } - for (const change of [ - { uid: 1001 }, - { gid: 1001 }, - { mode: 0o100666 }, - { isSymbolicLink: true }, - { isFile: false }, - ]) { - assert.throws( - () => verifyIdentity(plan, { ...runtime, policy: { ...runtime.policy, ...change } }), - /root-owned/, - ); - } - assert.equal(POLICY_FILE, "/etc/github-copilot/managed-settings.json"); -}); diff --git a/scripts/ci/device-policy-legacy.js b/scripts/ci/device-policy-legacy.js deleted file mode 100644 index 806647cf11..0000000000 --- a/scripts/ci/device-policy-legacy.js +++ /dev/null @@ -1,10 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import("./device-policy-fixture.mjs") - .then(({ launchRuntime }) => launchRuntime("legacy")) - .catch((error) => { - console.error(error.message); - process.exitCode = 1; - }); diff --git a/scripts/ci/device-policy-native.js b/scripts/ci/device-policy-native.js deleted file mode 100644 index d269a07ea6..0000000000 --- a/scripts/ci/device-policy-native.js +++ /dev/null @@ -1,10 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import("./device-policy-fixture.mjs") - .then(({ launchRuntime }) => launchRuntime("native")) - .catch((error) => { - console.error(error.message); - process.exitCode = 1; - }); From 3e9fd49f71b8f955789be599962a0b3808522d4b Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 19:46:40 -0700 Subject: [PATCH 13/15] fix(ci): repair standalone SDK test setup without SDK changes Restore strict isolated device-policy coverage and portable schema lookup. Admit standalone macOS runners and isolate complete Assisted contracts and Windows framework suites without weakening assertions or increasing per-test or per-job limits. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/actions/run-alpine-tests/action.yml | 1 + .github/workflows/sdk-dotnet.yml | 18 +- .github/workflows/sdk-go.yml | 7 +- .github/workflows/sdk-java.yml | 7 +- .github/workflows/sdk-nodejs.yml | 65 +- .github/workflows/sdk-platform.yml | 9 + .github/workflows/sdk-python.yml | 7 +- .github/workflows/sdk-rust.yml | 7 +- .github/workflows/sdk.yml | 5 + CONTRIBUTING.md | 37 + .../assisted_autopilot_permission.e2e.test.ts | 1922 +++++++++-------- nodejs/test/rust-codegen.test.ts | 10 +- scripts/ci/device-policy-bootstrap.mjs | 85 + scripts/ci/device-policy-fixture.mjs | 468 ++++ scripts/ci/device-policy-fixture.test.mjs | 207 ++ scripts/ci/device-policy-legacy.js | 10 + scripts/ci/device-policy-native.js | 10 + scripts/ci/run-dotnet-tests.sh | 5 + 18 files changed, 1949 insertions(+), 931 deletions(-) create mode 100644 scripts/ci/device-policy-bootstrap.mjs create mode 100644 scripts/ci/device-policy-fixture.mjs create mode 100644 scripts/ci/device-policy-fixture.test.mjs create mode 100644 scripts/ci/device-policy-legacy.js create mode 100644 scripts/ci/device-policy-native.js diff --git a/.github/actions/run-alpine-tests/action.yml b/.github/actions/run-alpine-tests/action.yml index a9842c11be..7808ca3f67 100644 --- a/.github/actions/run-alpine-tests/action.yml +++ b/.github/actions/run-alpine-tests/action.yml @@ -54,6 +54,7 @@ runs: --env COPILOT_SDK_ROOT="$ALPINE_TEST_SDK_ROOT" \ --env COPILOT_HMAC_KEY \ --env COPILOT_SDK_E2E_BACKEND \ + --env COPILOT_CI_RUNTIME_SOURCE \ --env BUNDLED_CLI_CACHE_DIR \ --env CARGO_TERM_COLOR \ --env RUST_BACKTRACE \ diff --git a/.github/workflows/sdk-dotnet.yml b/.github/workflows/sdk-dotnet.yml index 14a4c3265b..7b2e426e81 100644 --- a/.github/workflows/sdk-dotnet.yml +++ b/.github/workflows/sdk-dotnet.yml @@ -19,6 +19,9 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read @@ -99,9 +102,9 @@ jobs: retention-days: 7 dotnet-darwin-arm64: - name: ".NET (macos-26-xlarge-agent-runtime, default, CAPI)" + name: ".NET (${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }}, default, CAPI)" if: inputs.platform == 'all' || inputs.platform == 'darwin-arm64' - runs-on: macos-26-xlarge-agent-runtime + runs-on: ${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }} timeout-minutes: 30 defaults: run: @@ -147,14 +150,18 @@ jobs: - if: failure() uses: actions/upload-artifact@v7 with: - name: dotnet-test-diagnostics-macos-26-xlarge-default-capi-${{ github.run_attempt }} + name: dotnet-test-diagnostics-${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge' }}-default-capi-${{ github.run_attempt }} path: ${{ inputs.sdk-home }}/dotnet/TestResults/ if-no-files-found: warn retention-days: 7 dotnet-win32-x64: - name: ".NET (windows-latest, default, CAPI)" + name: ".NET (windows-latest, ${{ matrix.framework == 'all' && 'default' || matrix.framework }}, CAPI)" if: inputs.platform == 'all' || inputs.platform == 'win32-x64' + strategy: + fail-fast: false + matrix: + framework: ${{ fromJSON(inputs.runtime-source == 'published' && '["net8.0","net472"]' || '["all"]') }} runs-on: windows-latest timeout-minutes: 30 defaults: @@ -164,6 +171,7 @@ jobs: env: COPILOT_SDK_E2E_BACKEND: capi DOTNET_TEST_RUNTIME: win-x64 + DOTNET_TEST_FRAMEWORK: ${{ matrix.framework != 'all' && matrix.framework || '' }} DOTNET_TEST_RESULTS_DIRECTORY: TestResults/subprocess steps: - uses: actions/checkout@v7 @@ -203,7 +211,7 @@ jobs: - if: failure() uses: actions/upload-artifact@v7 with: - name: dotnet-test-diagnostics-windows-latest-default-capi-${{ github.run_attempt }} + name: dotnet-test-diagnostics-windows-latest-${{ matrix.framework == 'all' && 'default' || matrix.framework }}-capi-${{ github.run_attempt }} path: ${{ inputs.sdk-home }}/dotnet/TestResults/ if-no-files-found: warn retention-days: 7 diff --git a/.github/workflows/sdk-go.yml b/.github/workflows/sdk-go.yml index b5593f26b1..25b63039a9 100644 --- a/.github/workflows/sdk-go.yml +++ b/.github/workflows/sdk-go.yml @@ -19,19 +19,22 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: go: - name: "Go (${{ matrix.os }})" + name: "Go (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-java.yml b/.github/workflows/sdk-java.yml index 011fef0898..3d022571e3 100644 --- a/.github/workflows/sdk-java.yml +++ b/.github/workflows/sdk-java.yml @@ -18,13 +18,16 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: java: - name: "Java (${{ matrix.os }}, JDK ${{ matrix.test-jdk }})" + name: "Java (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, JDK ${{ matrix.test-jdk }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false @@ -36,7 +39,7 @@ jobs: test-jdk: "17" - os: windows-latest test-jdk: "17" - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index db14685c13..f94e00bfe4 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -7,6 +7,7 @@ env: HUSKY: 0 POWERSHELL_UPDATECHECK: Off SDK_HOME: ${{ inputs.sdk-home }} + COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} SETUP_NODE_TIMEOUT_MINUTES: 10 on: @@ -19,6 +20,9 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string outputs: capi-result: description: "Standard-platform CAPI job result, independent of BYOK jobs" @@ -30,13 +34,13 @@ permissions: jobs: nodejs: - name: "Node.js (${{ matrix.os }}, CAPI)" + name: "Node.js (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, CAPI)" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash @@ -89,16 +93,51 @@ jobs: working-directory: ${{ inputs.sdk-home }}/scripts/corrections - if: runner.os == 'Windows' run: pwsh.exe -Command "Write-Host 'PowerShell ready'" + - name: Test device-policy CI plans and required-result guard + if: runner.os == 'Linux' + working-directory: . + run: node --test "$SDK_HOME/scripts/ci/device-policy-fixture.test.mjs" + - name: Prepare isolated production device-policy fixtures + id: device-policy + if: inputs.runtime-source == 'published' && runner.os == 'Linux' + working-directory: . + run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" prepare - name: Test out-of-process id: subprocess env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} run: | + test_args=() + pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") + case "$pattern" in + "") ;; + *) + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + test_args+=(--testNamePattern "$pattern") + ;; + esac if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ] || [ "$GITHUB_EVENT_NAME" = "push" ]; then - npm test -- --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" + npm test -- "${test_args[@]}" --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" else - npm test + npm test -- "${test_args[@]}" fi + - name: Test required production device-policy controls + if: ${{ !cancelled() && inputs.runtime-source == 'published' && runner.os == 'Linux' && (steps.subprocess.outcome == 'success' || steps.subprocess.outcome == 'failure') }} + env: + COPILOT_CLI_PATH: ${{ steps.device-policy.outputs.native-launcher }} + COPILOT_LEGACY_CLI_PATH: ${{ steps.device-policy.outputs.legacy-launcher }} + COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} + run: | + status=0 + npm test -- test/e2e/managed_plugin_progress.e2e.test.ts test/e2e/rpc_server.e2e.test.ts \ + --testNamePattern "$(node ../scripts/ci/device-policy-fixture.mjs required-pattern)" \ + --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-device-policy-results.json" || status=$? + node ../scripts/ci/device-policy-fixture.mjs verify "$RUNNER_TEMP/sdk-device-policy-results.json" + exit "$status" + - name: Clean up owned device-policy containers + if: always() && steps.device-policy.outcome == 'success' + working-directory: . + run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" cleanup - name: Upload Flake Finder Node.js test results if: always() && (github.event_name == 'merge_group' || github.event_name == 'pull_request' || github.event_name == 'push') continue-on-error: true @@ -162,7 +201,14 @@ jobs: - name: Test out-of-process E2Es env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - run: npm test -- test/e2e + run: | + test_args=() + pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") + if [[ -n "$pattern" ]]; then + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + test_args+=(--testNamePattern "$pattern") + fi + npm test -- test/e2e "${test_args[@]}" nodejs-musl-x64: name: "Node.js (Alpine x64, CAPI)" @@ -182,6 +228,7 @@ jobs: - uses: ./.github/actions/run-alpine-tests env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} + COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} with: image: node:22-alpine sdk-root: /workspace/${{ inputs.sdk-home }} @@ -206,6 +253,12 @@ jobs: test -x "$COPILOT_CLI_PATH" test -f "$(dirname "$COPILOT_CLI_PATH")/runtime.node" status=0 - npm test || status=$? + pattern=$(node "$COPILOT_SDK_ROOT/scripts/ci/device-policy-fixture.mjs" portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") + if [ -n "$pattern" ]; then + echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" + npm test -- --testNamePattern "$pattern" || status=$? + else + npm test || status=$? + fi COPILOT_SDK_DEFAULT_CONNECTION=inprocess npm test -- test/e2e/inprocess_ffi.e2e.test.ts test/e2e/auth_host.e2e.test.ts || status=$? exit "$status" diff --git a/.github/workflows/sdk-platform.yml b/.github/workflows/sdk-platform.yml index f1ed4e6ed4..85937f783c 100644 --- a/.github/workflows/sdk-platform.yml +++ b/.github/workflows/sdk-platform.yml @@ -11,6 +11,9 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string outputs: nodejs-result: description: "Node.js SDK result, independent of other languages" @@ -29,6 +32,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit python: @@ -37,6 +41,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit go: @@ -45,6 +50,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit dotnet: @@ -53,6 +59,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit rust: @@ -61,6 +68,7 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit java: @@ -69,4 +77,5 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} + runtime-source: ${{ inputs.runtime-source }} secrets: inherit diff --git a/.github/workflows/sdk-python.yml b/.github/workflows/sdk-python.yml index df9cbab1a9..e81bb63975 100644 --- a/.github/workflows/sdk-python.yml +++ b/.github/workflows/sdk-python.yml @@ -19,19 +19,22 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: python: - name: "Python (${{ matrix.os }})" + name: "Python (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} timeout-minutes: 20 defaults: run: diff --git a/.github/workflows/sdk-rust.yml b/.github/workflows/sdk-rust.yml index 04500c3489..12506d5cad 100644 --- a/.github/workflows/sdk-rust.yml +++ b/.github/workflows/sdk-rust.yml @@ -19,20 +19,23 @@ on: sdk-home: required: true type: string + runtime-source: + required: true + type: string permissions: contents: read jobs: rust: - name: "Rust (${{ matrix.os }})" + name: "Rust (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) timeout-minutes: 60 strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-xlarge-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-xlarge-agent-runtime","windows-latest"]') }} - runs-on: ${{ matrix.os }} + runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk.yml b/.github/workflows/sdk.yml index 602d02b027..3c70b00d33 100644 --- a/.github/workflows/sdk.yml +++ b/.github/workflows/sdk.yml @@ -446,6 +446,7 @@ jobs: with: platform: linux-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-x64: @@ -456,6 +457,7 @@ jobs: with: platform: linuxmusl-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-arm64: @@ -466,6 +468,7 @@ jobs: with: platform: linuxmusl-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-darwin-arm64: @@ -476,6 +479,7 @@ jobs: with: platform: darwin-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-win32-x64: @@ -487,6 +491,7 @@ jobs: with: platform: win32-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} + runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit # Only Linux CAPI gates this rollup; the full SDK aggregate still reports all coverage. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 74519107b4..ed83ef2a5a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -363,12 +363,49 @@ failure fails the job. Java uses JDK 25 on all four platforms, plus a Linux/glibc JDK 17 compatibility job using precompiled classes. Merge groups retain the reduced Linux TypeScript CAPI subprocess coverage. +Standalone published-runtime macOS jobs use the standard `macos-26` ARM64 +GitHub-hosted runner. Source-runtime jobs retain their configured runtime +runners, including the larger Rust and .NET profiles. Test selection and +timeouts are unchanged; failures on the standard runner still fail coverage. + +Standalone Windows .NET runs `net8.0` and `net472` in separate mandatory jobs, +each with full subprocess coverage, the existing in-process smoke step, and +the existing 30-minute job limit. This increases total allowed runner time, +not one aggregate deadline. Source-runtime Windows jobs retain both targets +in the original single job. + +The Assisted permission contract runs as three complete scenario groups: +lifecycle, authorization, and handler. All assertions and sequential awaited +cleanup remain; each group uses the existing test timeout. This increases +total permitted suite time rather than reducing runtime teardown latency. + The `sdk-typescript` required rollup checks only the Linux CAPI job, including its build, packaging, and applicable static checks. Other platforms, BYOK backends, and languages keep their existing scheduling and failure reporting; they do not gate this rollup. The full `SDK` aggregate still requires all scheduled coverage to succeed. +Standalone CI assigns the two managed-device fixture controls to a separate +mandatory Linux CAPI step. Each fixture-bearing CLI child runs the unchanged, +pinned published runtime in its own disposable container, with the fixture +installed at the [documented production device-policy location](https://docs.github.com/en/copilot/how-tos/administer-copilot/manage-for-enterprise/use-managed-settings/deploy-managed-settings). +The launcher consumes the temporary-file test hint, verifies isolation and +root file ownership, then restores the runner's identity before launching the +runtime. It never writes policy to the host. A result guard requires both +original plugin-lifecycle and sessionless device/model tests to pass; missing +or skipped controls fail the job. Other published-runtime profiles explicitly +delegate only those two cases to this step. Source-runtime profiles retain +their existing full test selection and launch setup. + +The CI-only container setup requires Linux, Docker, and Node.js 22.15 or newer. +Do not provision a machine-wide policy file on a developer workstation to +run these fixtures. Launcher plan, selection, and result-guard unit controls +run without Docker, a CLI, or dependency installation: + +```bash +node --test scripts/ci/device-policy-fixture.test.mjs +``` + The three BYOK backend sweeps run in separate Linux TypeScript jobs, alongside the normal CAPI job; they do not repeat unit tests, packaging, or static checks. After preparing the runtime as described above, run a sweep from the diff --git a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts index 5577d03761..a06e01ebe8 100644 --- a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts +++ b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts @@ -8,7 +8,7 @@ import { mkdir, realpath, rm, writeFile } from "node:fs/promises"; import { createServer } from "node:http"; import { join } from "node:path"; import { text } from "node:stream/consumers"; -import { describe, expect, it, onTestFinished } from "vitest"; +import { describe, expect, it, onTestFailed, onTestFinished } from "vitest"; import { createAttributedPermissionResult, type CopilotSession, @@ -45,157 +45,314 @@ function contractFixtureName(scope: ContractScope, recommendation: ContractRecom describe("Assisted permission handling in Autopilot", async () => { const { copilotClient: client, workDir } = await createSdkTestContext({ useStdio: true }); - it("resolves Assisted recommendations before Autopilot permission recovery", async () => { - let agentCalls = 0; - const judgeOutputs: string[] = []; - const authorizationJudgeRequests: string[] = []; - const providerFailures: Error[] = []; - const outsideDir = `${workDir}-outside`; - await mkdir(outsideDir, { recursive: true }); - await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); - const modelServer = createServer((request, response) => { - void (async () => { - const body = JSON.parse(await text(request)) as { - model: string; - stream?: boolean; - messages: Array<{ role?: string; content?: unknown }>; - }; - const isJudge = body.model === "gpt-6-luna"; - const messages = JSON.stringify(body.messages); - const latestUserMessage = [...body.messages] - .reverse() - .find((entry) => entry.role === "user"); - const latestUserText = JSON.stringify(latestUserMessage?.content); - let message: - | { role: "assistant"; content: string } - | { - role: "assistant"; - content: string; - tool_calls: Array<{ - id: string; - type: "function"; - function: { name: string; arguments: string }; - }>; - }; - if (isJudge) { - const earlierRestrictionIndex = messages.lastIndexOf( - EARLIER_RESTRICTION_MARKER - ); - const laterAuthorizationIndex = messages.lastIndexOf( - LATER_AUTHORIZATION_MARKER - ); - const authorizationRoute = laterAuthorizationIndex !== -1; - if (authorizationRoute) { - authorizationJudgeRequests.push(messages); - } - const contractRequiresApproval = - messages.includes(contractFixtureName("root", "requireApproval")) || - messages.includes(contractFixtureName("subagent", "requireApproval")); - const output = authorizationRoute - ? earlierRestrictionIndex !== -1 && - earlierRestrictionIndex < laterAuthorizationIndex - ? JUDGE_OUTPUT - : HUMAN_REVIEW_OUTPUT - : messages.includes("assisted-human-ask") || contractRequiresApproval - ? HUMAN_REVIEW_OUTPUT - : JUDGE_OUTPUT; - judgeOutputs.push(output); - message = { role: "assistant", content: output }; - } else { - agentCalls++; - const toolResultCount = body.messages.filter( - (entry) => entry.role === "tool" - ).length; - const contractChildRoute = latestUserText.includes("ASSISTED_CONTRACT_CHILD_"); - const contractSubagentRoute = latestUserText.includes( - "ASSISTED_CONTRACT_SUBAGENT_" - ); - const contractRootRoute = latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); - const authorizationRestrictionRoute = latestUserText.includes( - EARLIER_RESTRICTION_MARKER - ); - const authorizationAllowRoute = latestUserText.includes( - LATER_AUTHORIZATION_MARKER - ); - if (authorizationRestrictionRoute) { - message = { role: "assistant", content: "restriction-recorded" }; - } else if (authorizationAllowRoute) { - message = - toolResultCount >= 2 - ? { role: "assistant", content: "authorization-updated" } - : toolResultCount === 1 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-task-complete", - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: "authorization-updated", - }), + it.each(["lifecycle", "authorization", "handler"] as const)( + "resolves Assisted recommendations before Autopilot permission recovery (%s contract)", + async (contract) => { + let agentCalls = 0; + const judgeOutputs: string[] = []; + const authorizationJudgeRequests: string[] = []; + const providerFailures: Error[] = []; + const started = Date.now(); + let phase = "setup"; + let phaseStarted = started; + const completedPhases: Array<{ phase: string; elapsedMs: number }> = []; + const enterPhase = (next: string) => { + const now = Date.now(); + completedPhases.push({ phase, elapsedMs: now - phaseStarted }); + phase = next; + phaseStarted = now; + }; + onTestFailed(() => { + console.error( + "Assisted contract failure boundary:", + JSON.stringify({ + phase, + phaseElapsedMs: Date.now() - phaseStarted, + elapsedMs: Date.now() - started, + completedPhases, + agentCalls, + judgeCalls: judgeOutputs.length, + providerFailures: providerFailures.length, + }) + ); + }); + const outsideDir = `${workDir}-outside`; + await mkdir(outsideDir, { recursive: true }); + await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); + const modelServer = createServer((request, response) => { + void (async () => { + const body = JSON.parse(await text(request)) as { + model: string; + stream?: boolean; + messages: Array<{ role?: string; content?: unknown }>; + }; + const isJudge = body.model === "gpt-6-luna"; + const messages = JSON.stringify(body.messages); + const latestUserMessage = [...body.messages] + .reverse() + .find((entry) => entry.role === "user"); + const latestUserText = JSON.stringify(latestUserMessage?.content); + let message: + | { role: "assistant"; content: string } + | { + role: "assistant"; + content: string; + tool_calls: Array<{ + id: string; + type: "function"; + function: { name: string; arguments: string }; + }>; + }; + if (isJudge) { + const earlierRestrictionIndex = messages.lastIndexOf( + EARLIER_RESTRICTION_MARKER + ); + const laterAuthorizationIndex = messages.lastIndexOf( + LATER_AUTHORIZATION_MARKER + ); + const authorizationRoute = laterAuthorizationIndex !== -1; + if (authorizationRoute) { + authorizationJudgeRequests.push(messages); + } + const contractRequiresApproval = + messages.includes(contractFixtureName("root", "requireApproval")) || + messages.includes(contractFixtureName("subagent", "requireApproval")); + const output = authorizationRoute + ? earlierRestrictionIndex !== -1 && + earlierRestrictionIndex < laterAuthorizationIndex + ? JUDGE_OUTPUT + : HUMAN_REVIEW_OUTPUT + : messages.includes("assisted-human-ask") || contractRequiresApproval + ? HUMAN_REVIEW_OUTPUT + : JUDGE_OUTPUT; + judgeOutputs.push(output); + message = { role: "assistant", content: output }; + } else { + agentCalls++; + const toolResultCount = body.messages.filter( + (entry) => entry.role === "tool" + ).length; + const contractChildRoute = latestUserText.includes( + "ASSISTED_CONTRACT_CHILD_" + ); + const contractSubagentRoute = latestUserText.includes( + "ASSISTED_CONTRACT_SUBAGENT_" + ); + const contractRootRoute = + latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); + const authorizationRestrictionRoute = latestUserText.includes( + EARLIER_RESTRICTION_MARKER + ); + const authorizationAllowRoute = latestUserText.includes( + LATER_AUTHORIZATION_MARKER + ); + if (authorizationRestrictionRoute) { + message = { role: "assistant", content: "restriction-recorded" }; + } else if (authorizationAllowRoute) { + message = + toolResultCount >= 2 + ? { role: "assistant", content: "authorization-updated" } + : toolResultCount === 1 + ? { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-authorization-task-complete", + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: "authorization-updated", + }), + }, }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-shell", - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: - process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` - : `mkdir ${AUTHORIZATION_FIXTURE}`, - description: - "Create the fixture authorized by the latest instruction", - }), + ], + } + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-authorization-shell", + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: + process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` + : `mkdir ${AUTHORIZATION_FIXTURE}`, + description: + "Create the fixture authorized by the latest instruction", + }), + }, }, + ], + }; + } else if ( + contractChildRoute || + contractSubagentRoute || + contractRootRoute + ) { + const recommendation: ContractRecommendation = latestUserText.includes( + "_REQUIRE" + ) + ? "requireApproval" + : latestUserText.includes("_EXCLUDED") + ? "excluded" + : "approve"; + const scope: ContractScope = + contractChildRoute || contractSubagentRoute ? "subagent" : "root"; + const finalText = `contract-${scope}-${ + recommendation === "approve" ? "approved" : "blocked" + }`; + if (contractSubagentRoute) { + message = + toolResultCount === 0 + ? { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-${recommendation}`, + type: "function", + function: { + name: "task", + arguments: JSON.stringify({ + name: "assisted-contract", + description: + "Exercise an Assisted permission request", + prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( + recommendation + )}: Run the requested shell command once and report whether it completed.`, + agent_type: "task", + mode: "sync", + }), + }, + }, + ], + } + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-complete-${recommendation}`, + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: finalText, + }), + }, + }, + ], + }; + } else if (contractChildRoute && toolResultCount > 0) { + message = { role: "assistant", content: finalText }; + } else if ( + (contractChildRoute || contractRootRoute) && + (toolResultCount === 0 || + (recommendation === "excluded" && toolResultCount === 1)) + ) { + const command = + recommendation === "excluded" + ? process.platform === "win32" + ? "Get-Content approval-probe.txt | & $runner" + : "printf payload | $runner" + : process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${contractFixtureName( + scope, + recommendation + )}` + : `mkdir ${contractFixtureName(scope, recommendation)}`; + message = { + role: "assistant", + content: "", + tool_calls: [ + { + id: + toolResultCount === 0 + ? contractShellCallId(scope, recommendation) + : `${contractShellCallId( + scope, + recommendation + )}-retry`, + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command, + description: + "Exercise Assisted permission routing", + }), }, - ], - }; - } else if (contractChildRoute || contractSubagentRoute || contractRootRoute) { - const recommendation: ContractRecommendation = latestUserText.includes( - "_REQUIRE" - ) - ? "requireApproval" - : latestUserText.includes("_EXCLUDED") - ? "excluded" - : "approve"; - const scope: ContractScope = - contractChildRoute || contractSubagentRoute ? "subagent" : "root"; - const finalText = `contract-${scope}-${ - recommendation === "approve" ? "approved" : "blocked" - }`; - if (contractSubagentRoute) { - message = - toolResultCount === 0 + }, + ], + }; + } else { + message = { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-complete-${recommendation}`, + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ summary: finalText }), + }, + }, + ], + }; + } + } else { + const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); + const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); + const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); + const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); + const primeRoute = + primeIndex > Math.max(shellIndex, pathIndex, humanIndex); + const humanRoute = + humanIndex > Math.max(primeIndex, shellIndex, pathIndex); + const shellRoute = + shellIndex > Math.max(primeIndex, pathIndex, humanIndex); + const resumedShellRoute = + messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; + const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); + const humanLifecycle = + messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || + messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") + ? "resume" + : "create"; + const finalText = humanRoute + ? humanApproved + ? "human-approved" + : "human-denied" + : shellRoute + ? "shell-approved" + : "approval-probe.txt"; + message = primeRoute + ? { role: "assistant", content: "prime-ready" } + : toolResultCount >= 2 + ? { + role: "assistant", + content: finalText, + } + : toolResultCount === 1 ? { role: "assistant", content: "", tool_calls: [ { - id: `assisted-contract-task-${recommendation}`, + id: "assisted-autopilot-task-complete", type: "function", function: { - name: "task", + name: "task_complete", arguments: JSON.stringify({ - name: "assisted-contract", - description: - "Exercise an Assisted permission request", - prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( - recommendation - )}: Run the requested shell command once and report whether it completed.`, - agent_type: "task", - mode: "sync", + summary: finalText, }), }, }, @@ -206,833 +363,788 @@ describe("Assisted permission handling in Autopilot", async () => { content: "", tool_calls: [ { - id: `assisted-contract-task-complete-${recommendation}`, + id: shellRoute + ? "assisted-autopilot-shell" + : humanRoute + ? "assisted-autopilot-human" + : "assisted-autopilot-glob", type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: finalText, - }), - }, + function: + shellRoute || humanRoute + ? { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: humanRoute + ? process.platform === + "win32" + ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : process.platform === + "win32" + ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` + : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, + description: humanRoute + ? "Create the human-reviewed SDK permission fixture" + : "Create the authorized SDK permission fixture", + }), + } + : { + name: "glob", + arguments: JSON.stringify({ + pattern: "*.txt", + path: outsideDir, + }), + }, }, ], }; - } else if (contractChildRoute && toolResultCount > 0) { - message = { role: "assistant", content: finalText }; - } else if ( - (contractChildRoute || contractRootRoute) && - (toolResultCount === 0 || - (recommendation === "excluded" && toolResultCount === 1)) - ) { - const command = - recommendation === "excluded" - ? process.platform === "win32" - ? "Get-Content approval-probe.txt | & $runner" - : "printf payload | $runner" - : process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${contractFixtureName( - scope, - recommendation - )}` - : `mkdir ${contractFixtureName(scope, recommendation)}`; - message = { - role: "assistant", - content: "", - tool_calls: [ - { - id: - toolResultCount === 0 - ? contractShellCallId(scope, recommendation) - : `${contractShellCallId( - scope, - recommendation - )}-retry`, - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command, - description: "Exercise Assisted permission routing", - }), - }, - }, - ], - }; - } else { - message = { - role: "assistant", - content: "", - tool_calls: [ + } + } + + const choice = { + index: 0, + message, + finish_reason: "tool_calls" in message ? "tool_calls" : "stop", + }; + const completion = { + id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, + object: "chat.completion", + created: 1, + model: body.model, + choices: [choice], + usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, + }; + if (body.stream) { + response.writeHead(200, { "content-type": "text/event-stream" }); + response.end( + `data: ${JSON.stringify({ + ...completion, + object: "chat.completion.chunk", + choices: [ { - id: `assisted-contract-task-complete-${recommendation}`, - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ summary: finalText }), + index: 0, + delta: { + ...message, + ...("tool_calls" in message + ? { + tool_calls: message.tool_calls.map( + (call, index) => ({ index, ...call }) + ), + } + : {}), }, + finish_reason: choice.finish_reason, }, ], - }; - } + })}\n\ndata: [DONE]\n\n` + ); } else { - const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); - const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); - const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); - const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); - const primeRoute = primeIndex > Math.max(shellIndex, pathIndex, humanIndex); - const humanRoute = humanIndex > Math.max(primeIndex, shellIndex, pathIndex); - const shellRoute = shellIndex > Math.max(primeIndex, pathIndex, humanIndex); - const resumedShellRoute = - messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; - const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); - const humanLifecycle = - messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || - messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") - ? "resume" - : "create"; - const finalText = humanRoute - ? humanApproved - ? "human-approved" - : "human-denied" - : shellRoute - ? "shell-approved" - : "approval-probe.txt"; - message = primeRoute - ? { role: "assistant", content: "prime-ready" } - : toolResultCount >= 2 - ? { - role: "assistant", - content: finalText, - } - : toolResultCount === 1 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-autopilot-task-complete", - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: finalText, - }), - }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: shellRoute - ? "assisted-autopilot-shell" - : humanRoute - ? "assisted-autopilot-human" - : "assisted-autopilot-glob", - type: "function", - function: - shellRoute || humanRoute - ? { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: humanRoute - ? process.platform === "win32" - ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : process.platform === "win32" - ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` - : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, - description: humanRoute - ? "Create the human-reviewed SDK permission fixture" - : "Create the authorized SDK permission fixture", - }), - } - : { - name: "glob", - arguments: JSON.stringify({ - pattern: "*.txt", - path: outsideDir, - }), - }, - }, - ], - }; + response.writeHead(200, { "content-type": "application/json" }); + response.end(JSON.stringify(completion)); } - } - - const choice = { - index: 0, - message, - finish_reason: "tool_calls" in message ? "tool_calls" : "stop", - }; - const completion = { - id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, - object: "chat.completion", - created: 1, - model: body.model, - choices: [choice], - usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, - }; - if (body.stream) { - response.writeHead(200, { "content-type": "text/event-stream" }); - response.end( - `data: ${JSON.stringify({ - ...completion, - object: "chat.completion.chunk", - choices: [ - { - index: 0, - delta: { - ...message, - ...("tool_calls" in message - ? { - tool_calls: message.tool_calls.map( - (call, index) => ({ index, ...call }) - ), - } - : {}), - }, - finish_reason: choice.finish_reason, - }, - ], - })}\n\ndata: [DONE]\n\n` + })().catch((error: unknown) => { + providerFailures.push( + error instanceof Error ? error : new Error(String(error)) ); - } else { - response.writeHead(200, { "content-type": "application/json" }); - response.end(JSON.stringify(completion)); + response.writeHead(500).end(); + }); + }); + await new Promise((resolve, reject) => { + modelServer.once("error", reject); + modelServer.listen(0, "127.0.0.1", resolve); + }); + onTestFinished(async () => { + modelServer.closeAllConnections(); + if (modelServer.listening) { + await new Promise((resolve, reject) => { + modelServer.close((error) => (error ? reject(error) : resolve())); + }); } - })().catch((error: unknown) => { - providerFailures.push(error instanceof Error ? error : new Error(String(error))); - response.writeHead(500).end(); }); - }); - await new Promise((resolve, reject) => { - modelServer.once("error", reject); - modelServer.listen(0, "127.0.0.1", resolve); - }); - onTestFinished(async () => { - modelServer.closeAllConnections(); - if (modelServer.listening) { - await new Promise((resolve, reject) => { - modelServer.close((error) => (error ? reject(error) : resolve())); - }); + const address = modelServer.address(); + if (!address || typeof address === "string") { + throw new Error("Missing local model server address"); } - }); - const address = modelServer.address(); - if (!address || typeof address === "string") { - throw new Error("Missing local model server address"); - } - execFileSync("git", ["init", "--quiet"], { cwd: workDir }); - await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + execFileSync("git", ["init", "--quiet"], { cwd: workDir }); + await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - const providers: NamedProviderConfig[] = [ - { - name: "local", - type: "openai", - baseUrl: `http://127.0.0.1:${address.port}/v1`, - apiKey: "synthetic-test-token", - wireApi: "completions", - }, - ]; - const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ - id, - provider: "local", - modelId: "gpt-4o", - wireModel: id, - })); - const scenarios = [ - { route: "shell", lifecycle: "create", decision: "judge" }, - { route: "path", lifecycle: "create", decision: "judge" }, - { route: "shell", lifecycle: "resume", decision: "judge" }, - { route: "path", lifecycle: "resume", decision: "judge" }, - { route: "human", lifecycle: "create", decision: "approve" }, - { route: "human", lifecycle: "create", decision: "deny" }, - { route: "human", lifecycle: "resume", decision: "approve" }, - { route: "human", lifecycle: "resume", decision: "deny" }, - ] as const; - for (const scenario of scenarios) { - let permissionCallbacks = 0; - const expectedRecommendation = - scenario.decision === "judge" ? "approve" : "requireApproval"; - const recommendationByToolCallId = new Map(); - const recommendationWaiters = new Map< - string, - (recommendation: string | undefined) => void - >(); - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; - const decisionSources: string[] = []; - const taskOutcomes: Array<{ - success?: boolean; - summary?: string; - outcome?: string; - reason?: string; - }> = []; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "glob", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, + const providers: NamedProviderConfig[] = [ + { + name: "local", + type: "openai", + baseUrl: `http://127.0.0.1:${address.port}/v1`, + apiKey: "synthetic-test-token", + wireApi: "completions", }, - onPermissionRequest: async (request: PermissionRequest) => { - permissionCallbacks++; - if (!request.toolCallId) { - throw new Error( - "Expected the permission request to identify its tool call" - ); - } - const recommendation = recommendationByToolCallId.has(request.toolCallId) - ? recommendationByToolCallId.get(request.toolCallId) - : await new Promise((resolve) => { - recommendationWaiters.set(request.toolCallId!, resolve); - }); - recommendationByToolCallId.delete(request.toolCallId); - if (recommendation !== expectedRecommendation) { - throw new Error( - `Expected Assisted recommendation ${expectedRecommendation}, got ${String( - recommendation - )}` - ); - } - if (scenario.decision === "judge") { + ]; + const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ + id, + provider: "local", + modelId: "gpt-4o", + wireModel: id, + })); + const scenarios = [ + { route: "shell", lifecycle: "create", decision: "judge" }, + { route: "path", lifecycle: "create", decision: "judge" }, + { route: "shell", lifecycle: "resume", decision: "judge" }, + { route: "path", lifecycle: "resume", decision: "judge" }, + { route: "human", lifecycle: "create", decision: "approve" }, + { route: "human", lifecycle: "create", decision: "deny" }, + { route: "human", lifecycle: "resume", decision: "approve" }, + { route: "human", lifecycle: "resume", decision: "deny" }, + ] as const; + for (const scenario of contract === "lifecycle" ? scenarios : []) { + const label = `initial:${scenario.route}:${scenario.lifecycle}:${scenario.decision}`; + let permissionCallbacks = 0; + const expectedRecommendation = + scenario.decision === "judge" ? "approve" : "requireApproval"; + const recommendationByToolCallId = new Map(); + const recommendationWaiters = new Map< + string, + (recommendation: string | undefined) => void + >(); + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; + const decisionSources: string[] = []; + const taskOutcomes: Array<{ + success?: boolean; + summary?: string; + outcome?: string; + reason?: string; + }> = []; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "glob", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + onPermissionRequest: async (request: PermissionRequest) => { + permissionCallbacks++; + if (!request.toolCallId) { + throw new Error( + "Expected the permission request to identify its tool call" + ); + } + const recommendation = recommendationByToolCallId.has(request.toolCallId) + ? recommendationByToolCallId.get(request.toolCallId) + : await new Promise((resolve) => { + recommendationWaiters.set(request.toolCallId!, resolve); + }); + recommendationByToolCallId.delete(request.toolCallId); + if (recommendation !== expectedRecommendation) { + throw new Error( + `Expected Assisted recommendation ${expectedRecommendation}, got ${String( + recommendation + )}` + ); + } + if (scenario.decision === "judge") { + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } + ); + } return createAttributedPermissionResult( - { kind: "approve-once" }, + { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, { - outcome: "auto_approved", - source: "assisted_approval", + outcome: "prompted_user", + source: "human_response", surface: "sdk", - responseCapability: "headless", + responseCapability: "interactive", } ); + }, + } as const; + let session: CopilotSession | undefined; + try { + enterPhase(`${label}:create`); + session = await client.createSession(sessionConfig); + if (scenario.lifecycle === "resume") { + const sessionId = session.sessionId; + enterPhase(`${label}:prime`); + await session.sendAndWait({ + prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", + }); + enterPhase(`${label}:configure-before-resume`); + await configureAssistedAutopilot(session, workDir); + await expectAssistedAutopilotConfigured(session, workDir); + enterPhase(`${label}:disconnect-before-resume`); + await session.disconnect(); + session = undefined; + enterPhase(`${label}:resume`); + session = await client.resumeSession(sessionId, sessionConfig); + enterPhase(`${label}:verify-resume`); + await expectAssistedAutopilotConfigured(session, workDir); + } else { + enterPhase(`${label}:configure`); + await configureAssistedAutopilot(session, workDir); } - return createAttributedPermissionResult( - { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, - { - outcome: "prompted_user", - source: "human_response", - surface: "sdk", - responseCapability: "interactive", + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; + } + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + const toolCallId = ( + event.data as { + permissionRequest?: { toolCallId?: string }; + } + ).permissionRequest?.toolCallId; + if (toolCallId) { + const waiter = recommendationWaiters.get(toolCallId); + if (waiter) { + recommendationWaiters.delete(toolCallId); + waiter(recommendation); + } else { + recommendationByToolCallId.set(toolCallId, recommendation); + } + } + if (recommendation) recommendations.push(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }) + .decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + recoveryStatuses.push((event.data as { status: string }).status); + } else if (event.type === "tool.execution_complete") { + toolResults.push(event.data); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); } + }); + + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + enterPhase(`${label}:send`); + await session.send({ + prompt: + scenario.route === "shell" + ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` + : scenario.route === "path" + ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` + : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, + }); + enterPhase(`${label}:task-complete`); + await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } + + enterPhase(`${label}:assert`); + const expectedCompletions = + scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill(expectedRecommendation) ); - }, - } as const; - let session: CopilotSession | undefined; - try { - session = await client.createSession(sessionConfig); - if (scenario.lifecycle === "resume") { - const sessionId = session.sessionId; - await session.sendAndWait({ - prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", + expect(decisionSources, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill( + scenario.decision === "judge" ? "assisted_approval" : "human_response" + ) + ); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); + if (scenario.decision === "deny") { + expect( + toolResults.some((result) => result.success === false), + JSON.stringify(scenario) + ).toBe(true); + expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); + } else { + expect( + toolResults.every((result) => result.success === true), + JSON.stringify(scenario) + ).toBe(true); + } + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + success: true, }); - await configureAssistedAutopilot(session, workDir); - await expectAssistedAutopilotConfigured(session, workDir); - await session.disconnect(); - session = undefined; - session = await client.resumeSession(sessionId, sessionConfig); - await expectAssistedAutopilotConfigured(session, workDir); - } else { - await configureAssistedAutopilot(session, workDir); + expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( + scenario.route === "shell" + ? "shell-approved" + : scenario.route === "path" + ? "approval-probe.txt" + : `human-${scenario.decision === "approve" ? "approved" : "denied"}` + ); + if (scenario.route === "human") { + expect( + existsSync( + join( + workDir, + `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` + ) + ), + JSON.stringify(scenario) + ).toBe(scenario.decision === "approve"); + } + } finally { + if (session) { + enterPhase(`${label}:cleanup`); + await session.abort(); + await session.disconnect(); + } } - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; + } + + if (contract === "lifecycle") { + expect(providerFailures).toEqual([]); + expect(judgeOutputs).toEqual([ + ...Array(4).fill(JUDGE_OUTPUT), + ...Array(4).fill(HUMAN_REVIEW_OUTPUT), + ]); + expect(agentCalls).toBe(scenarios.length * 2 + 4); + } + + if (contract === "authorization") { + const authorizationJudgeStart = judgeOutputs.length; + let authorizationPermissionCallbacks = 0; + const authorizationRecommendations: string[] = []; + const authorizationDecisionSources: string[] = []; + const authorizationRecoveryStatuses: string[] = []; + const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = + []; + const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = + []; + let resolveAuthorizationRecommendation: ( + recommendation: string | undefined + ) => void; + const authorizationRecommendation = new Promise((resolve) => { + resolveAuthorizationRecommendation = resolve; + }); + let authorizationSession: CopilotSession | undefined; + try { + enterPhase("authorization:create"); + authorizationSession = await client.createSession({ + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + onPermissionRequest: async (request) => { + authorizationPermissionCallbacks++; + if (request.toolCallId !== "assisted-authorization-shell") { + throw new Error( + `Expected authorization shell permission, got ${String( + request.toolCallId + )}` + ); } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - const toolCallId = ( - event.data as { - permissionRequest?: { toolCallId?: string }; + const recommendation = await authorizationRecommendation; + if (recommendation !== "approve") { + throw new Error( + `Expected later authorization to reach the judge as approve, got ${String( + recommendation + )}` + ); } - ).permissionRequest?.toolCallId; - if (toolCallId) { - const waiter = recommendationWaiters.get(toolCallId); - if (waiter) { - recommendationWaiters.delete(toolCallId); - waiter(recommendation); - } else { - recommendationByToolCallId.set(toolCallId, recommendation); + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } + ); + }, + }); + enterPhase("authorization:prime"); + await authorizationSession.sendAndWait({ + prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, + }); + enterPhase("authorization:configure"); + await configureAssistedAutopilot(authorizationSession, workDir); + authorizationSession.on((event) => { + if (event.type === "permission.requested") { + const recommendation = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + }; + } + ).promptRequest?.assistedApproval?.recommendation; + if (recommendation) { + authorizationRecommendations.push(recommendation); } + resolveAuthorizationRecommendation(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }) + .decisionSource; + if (source) authorizationDecisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + authorizationRecoveryStatuses.push( + (event.data as { status: string }).status + ); + } else if (event.type === "tool.execution_complete") { + authorizationToolResults.push({ + toolCallId: event.data.toolCallId, + success: event.data.success, + }); + } else if (event.type === "session.task_complete") { + authorizationTaskOutcomes.push(event.data); } - if (recommendation) recommendations.push(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }).decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - recoveryStatuses.push((event.data as { status: string }).status); - } else if (event.type === "tool.execution_complete") { - toolResults.push(event.data); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); - } - }); + }); - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - await session.send({ - prompt: - scenario.route === "shell" - ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` - : scenario.route === "path" - ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` - : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, + const taskComplete = getNextEventOfType( + authorizationSession, + "session.task_complete" + ); + enterPhase("authorization:send"); + await authorizationSession.send({ + prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, }); + enterPhase("authorization:task-complete"); await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } - const expectedCompletions = - scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill(expectedRecommendation) - ); - expect(decisionSources, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill( - scenario.decision === "judge" ? "assisted_approval" : "human_response" - ) - ); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); - if (scenario.decision === "deny") { - expect( - toolResults.some((result) => result.success === false), - JSON.stringify(scenario) - ).toBe(true); - expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); - } else { + enterPhase("authorization:assert"); + expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); + expect(authorizationJudgeRequests).toHaveLength(1); + const authorizationRequest = authorizationJudgeRequests[0] ?? ""; expect( - toolResults.every((result) => result.success === true), - JSON.stringify(scenario) - ).toBe(true); - } - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - success: true, - }); - expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( - scenario.route === "shell" - ? "shell-approved" - : scenario.route === "path" - ? "approval-probe.txt" - : `human-${scenario.decision === "approve" ? "approved" : "denied"}` - ); - if (scenario.route === "human") { + authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) + ).toBeGreaterThanOrEqual(0); expect( - existsSync( - join( - workDir, - `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` - ) - ), - JSON.stringify(scenario) - ).toBe(scenario.decision === "approve"); - } - } finally { - if (session) { - await session.abort(); - await session.disconnect(); + authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER) + ).toBeGreaterThan(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)); + expect(authorizationPermissionCallbacks).toBe(1); + expect(authorizationRecommendations).toEqual(["approve"]); + expect(authorizationDecisionSources).toEqual(["assisted_approval"]); + expect(authorizationRecoveryStatuses).toEqual([]); + expect(authorizationToolResults).toContainEqual({ + toolCallId: "assisted-authorization-shell", + success: true, + }); + expect(authorizationTaskOutcomes).toEqual([ + expect.objectContaining({ + success: true, + summary: "authorization-updated", + }), + ]); + expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); + } finally { + if (authorizationSession) { + enterPhase("authorization:cleanup"); + await authorizationSession.abort(); + await authorizationSession.disconnect(); + } } } - } - expect(providerFailures).toEqual([]); - expect(judgeOutputs).toEqual([ - ...Array(4).fill(JUDGE_OUTPUT), - ...Array(4).fill(HUMAN_REVIEW_OUTPUT), - ]); - expect(agentCalls).toBe(scenarios.length * 2 + 4); - - const authorizationJudgeStart = judgeOutputs.length; - let authorizationPermissionCallbacks = 0; - const authorizationRecommendations: string[] = []; - const authorizationDecisionSources: string[] = []; - const authorizationRecoveryStatuses: string[] = []; - const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = []; - const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = []; - let resolveAuthorizationRecommendation: (recommendation: string | undefined) => void; - const authorizationRecommendation = new Promise((resolve) => { - resolveAuthorizationRecommendation = resolve; - }); - let authorizationSession: CopilotSession | undefined; - try { - authorizationSession = await client.createSession({ - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - onPermissionRequest: async (request) => { - authorizationPermissionCallbacks++; - if (request.toolCallId !== "assisted-authorization-shell") { - throw new Error( - `Expected authorization shell permission, got ${String( - request.toolCallId - )}` + const contractScenarios = [ + { scope: "root", recommendation: "approve", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "excluded", handler: "registered" }, + { scope: "subagent", recommendation: "approve", handler: "registered" }, + { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "none" }, + ] as const satisfies readonly { + scope: ContractScope; + recommendation: ContractRecommendation; + handler: "registered" | "none"; + }[]; + const contractJudgeStart = judgeOutputs.length; + for (const scenario of contract === "handler" ? contractScenarios : []) { + const label = `contract:${scenario.scope}:${scenario.recommendation}:${scenario.handler}`; + let permissionCallbacks = 0; + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const decisionSources: string[] = []; + const sequence: string[] = []; + const subagentIds = new Set(); + const permissionAgentIds: Array = []; + const toolResults: Array<{ + agentId?: string; + toolCallId?: string; + success?: boolean; + error?: { message?: string }; + result?: unknown; + }> = []; + const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; + const onPermissionRequest = () => { + permissionCallbacks++; + sequence.push("host"); + if (scenario.recommendation === "approve") { + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); } - const recommendation = await authorizationRecommendation; - if (recommendation !== "approve") { - throw new Error( - `Expected later authorization to reach the judge as approve, got ${String( - recommendation - )}` + if (scenario.recommendation === "requireApproval") { + return createAttributedPermissionResult( + { + kind: "reject", + feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), + }, + { + outcome: "autopilot_denied", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); } - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - }, - }); - await authorizationSession.sendAndWait({ - prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, - }); - await configureAssistedAutopilot(authorizationSession, workDir); - authorizationSession.on((event) => { - if (event.type === "permission.requested") { - const recommendation = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - }; - } - ).promptRequest?.assistedApproval?.recommendation; - if (recommendation) { - authorizationRecommendations.push(recommendation); + if (scenario.recommendation === "excluded") { + return createAttributedPermissionResult( + { kind: "user-not-available" }, + { + outcome: "autopilot_denied", + source: "unattended_fallback", + surface: "sdk", + responseCapability: "headless", + } + ); } - resolveAuthorizationRecommendation(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }).decisionSource; - if (source) authorizationDecisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - authorizationRecoveryStatuses.push((event.data as { status: string }).status); - } else if (event.type === "tool.execution_complete") { - authorizationToolResults.push({ - toolCallId: event.data.toolCallId, - success: event.data.success, + throw new Error("Unexpected Assisted permission recommendation"); + }; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), + } as const; + let session: CopilotSession | undefined; + try { + enterPhase(`${label}:create`); + session = await client.createSession(sessionConfig); + enterPhase(`${label}:configure`); + await configureAssistedAutopilot(session, workDir); + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; + } + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + if (recommendation) { + recommendations.push(recommendation); + sequence.push(`permission:${recommendation}`); + } + permissionAgentIds.push(event.agentId); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }) + .decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + const status = (event.data as { status: string }).status; + recoveryStatuses.push(status); + sequence.push(`recovery:${status}`); + } else if (event.type === "tool.execution_complete") { + toolResults.push({ + agentId: event.agentId, + toolCallId: event.data.toolCallId, + success: event.data.success, + error: event.data.error, + result: event.data.result, + }); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); + } else if (event.type === "subagent.started") { + subagentIds.add(event.agentId); + } }); - } else if (event.type === "session.task_complete") { - authorizationTaskOutcomes.push(event.data); - } - }); - - const taskComplete = getNextEventOfType(authorizationSession, "session.task_complete"); - await authorizationSession.send({ - prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, - }); - await taskComplete; - expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); - expect(authorizationJudgeRequests).toHaveLength(1); - const authorizationRequest = authorizationJudgeRequests[0] ?? ""; - expect(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)).toBeGreaterThanOrEqual( - 0 - ); - expect(authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER)).toBeGreaterThan( - authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) - ); - expect(authorizationPermissionCallbacks).toBe(1); - expect(authorizationRecommendations).toEqual(["approve"]); - expect(authorizationDecisionSources).toEqual(["assisted_approval"]); - expect(authorizationRecoveryStatuses).toEqual([]); - expect(authorizationToolResults).toContainEqual({ - toolCallId: "assisted-authorization-shell", - success: true, - }); - expect(authorizationTaskOutcomes).toEqual([ - expect.objectContaining({ - success: true, - summary: "authorization-updated", - }), - ]); - expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); - } finally { - if (authorizationSession) { - await authorizationSession.abort(); - await authorizationSession.disconnect(); - } - } + const marker = contractMarker(scenario.recommendation); + const prompt = + scenario.scope === "root" + ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` + : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + enterPhase(`${label}:send`); + await session.send({ prompt }); + enterPhase(`${label}:task-complete`); + await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } - const contractScenarios = [ - { scope: "root", recommendation: "approve", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "excluded", handler: "registered" }, - { scope: "subagent", recommendation: "approve", handler: "registered" }, - { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "none" }, - ] as const satisfies readonly { - scope: ContractScope; - recommendation: ContractRecommendation; - handler: "registered" | "none"; - }[]; - const contractJudgeStart = judgeOutputs.length; - for (const scenario of contractScenarios) { - let permissionCallbacks = 0; - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const decisionSources: string[] = []; - const sequence: string[] = []; - const subagentIds = new Set(); - const permissionAgentIds: Array = []; - const toolResults: Array<{ - agentId?: string; - toolCallId?: string; - success?: boolean; - error?: { message?: string }; - result?: unknown; - }> = []; - const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; - const onPermissionRequest = () => { - permissionCallbacks++; - sequence.push("host"); - if (scenario.recommendation === "approve") { - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } + enterPhase(`${label}:assert`); + if (scenario.handler === "none") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); + expect(recommendations, JSON.stringify(scenario)).toEqual([]); + } else if (scenario.recommendation === "approve") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); + } else if (scenario.recommendation === "requireApproval") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual([ + "requireApproval", + ]); + } else { + expect( + permissionCallbacks, + JSON.stringify(scenario) + ).toBeGreaterThanOrEqual(1); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(permissionCallbacks).fill("excluded") + ); + } + if (scenario.handler === "none") { + expect(decisionSources, JSON.stringify(scenario)).toEqual([]); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); + } else { + expect(decisionSources, JSON.stringify(scenario)).toContain( + scenario.recommendation === "excluded" + ? "unattended_fallback" + : "assisted_approval" + ); + } + if (scenario.recommendation === "excluded") { + expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); + } else if (scenario.handler === "registered") { + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + } + const hostIndex = sequence.indexOf("host"); + const firstRecoveryIndex = sequence.findIndex((entry) => + entry.startsWith("recovery:") ); - } - if (scenario.recommendation === "requireApproval") { - return createAttributedPermissionResult( - { - kind: "reject", - feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), - }, - { - outcome: "autopilot_denied", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } + if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { + expect( + hostIndex, + JSON.stringify({ scenario, sequence }) + ).toBeGreaterThanOrEqual(0); + expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( + firstRecoveryIndex + ); + } + + const shellResult = toolResults.find( + (result) => + result.toolCallId === + contractShellCallId(scenario.scope, scenario.recommendation) ); - } - if (scenario.recommendation === "excluded") { - return createAttributedPermissionResult( - { kind: "user-not-available" }, - { - outcome: "autopilot_denied", - source: "unattended_fallback", - surface: "sdk", - responseCapability: "headless", - } + expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); + expect(shellResult?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation === "approve" && scenario.handler === "registered" ); - } - throw new Error("Unexpected Assisted permission recommendation"); - }; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), - } as const; - let session: CopilotSession | undefined; - try { - session = await client.createSession(sessionConfig); - await configureAssistedAutopilot(session, workDir); - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; - } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - if (recommendation) { - recommendations.push(recommendation); - sequence.push(`permission:${recommendation}`); + if ( + scenario.recommendation === "requireApproval" && + scenario.handler === "registered" + ) { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Ask the registered human permission handler." + ); + } else if (scenario.handler === "none") { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Permission could not be granted automatically." + ); + } + if (scenario.scope === "subagent") { + expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); + expect( + subagentIds.has(shellResult?.agentId ?? ""), + JSON.stringify(scenario) + ).toBe(true); + for (const agentId of permissionAgentIds) { + expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe( + true + ); } - permissionAgentIds.push(event.agentId); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }).decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - const status = (event.data as { status: string }).status; - recoveryStatuses.push(status); - sequence.push(`recovery:${status}`); - } else if (event.type === "tool.execution_complete") { - toolResults.push({ - agentId: event.agentId, - toolCallId: event.data.toolCallId, - success: event.data.success, - error: event.data.error, - result: event.data.result, - }); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); - } else if (event.type === "subagent.started") { - subagentIds.add(event.agentId); } - }); - - const marker = contractMarker(scenario.recommendation); - const prompt = - scenario.scope === "root" - ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` - : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - await session.send({ prompt }); - await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } - if (scenario.handler === "none") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); - expect(recommendations, JSON.stringify(scenario)).toEqual([]); - } else if (scenario.recommendation === "approve") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); - } else if (scenario.recommendation === "requireApproval") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual(["requireApproval"]); - } else { - expect(permissionCallbacks, JSON.stringify(scenario)).toBeGreaterThanOrEqual(1); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(permissionCallbacks).fill("excluded") - ); - } - if (scenario.handler === "none") { - expect(decisionSources, JSON.stringify(scenario)).toEqual([]); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); - } else { - expect(decisionSources, JSON.stringify(scenario)).toContain( - scenario.recommendation === "excluded" - ? "unattended_fallback" - : "assisted_approval" - ); - } - if (scenario.recommendation === "excluded") { - expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); - } else if (scenario.handler === "registered") { - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - } - const hostIndex = sequence.indexOf("host"); - const firstRecoveryIndex = sequence.findIndex((entry) => - entry.startsWith("recovery:") - ); - if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { expect( - hostIndex, - JSON.stringify({ scenario, sequence }) - ).toBeGreaterThanOrEqual(0); - expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( - firstRecoveryIndex - ); - } - - const shellResult = toolResults.find( - (result) => - result.toolCallId === - contractShellCallId(scenario.scope, scenario.recommendation) - ); - expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); - expect(shellResult?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation === "approve" && scenario.handler === "registered" - ); - if ( - scenario.recommendation === "requireApproval" && - scenario.handler === "registered" - ) { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Ask the registered human permission handler." - ); - } else if (scenario.handler === "none") { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Permission could not be granted automatically." + existsSync( + join( + workDir, + contractFixtureName(scenario.scope, scenario.recommendation) + ) + ), + JSON.stringify(scenario) + ).toBe( + scenario.recommendation === "approve" && scenario.handler === "registered" ); - } - if (scenario.scope === "subagent") { - expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); expect( - subagentIds.has(shellResult?.agentId ?? ""), + (await session.rpc.permissions.pendingRequests()).items, JSON.stringify(scenario) - ).toBe(true); - for (const agentId of permissionAgentIds) { - expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe(true); + ).toEqual([]); + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation !== "excluded" && scenario.handler === "registered" + ); + if (scenario.handler === "none") { + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + outcome: "continue", + reason: "Autopilot is still recovering from a required permission.", + }); + } + } finally { + if (session) { + enterPhase(`${label}:cleanup`); + await session.abort(); + await session.disconnect(); } } + } - expect( - existsSync( - join(workDir, contractFixtureName(scenario.scope, scenario.recommendation)) - ), - JSON.stringify(scenario) - ).toBe(scenario.recommendation === "approve" && scenario.handler === "registered"); - expect( - (await session.rpc.permissions.pendingRequests()).items, - JSON.stringify(scenario) - ).toEqual([]); - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation !== "excluded" && scenario.handler === "registered" - ); - if (scenario.handler === "none") { - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - outcome: "continue", - reason: "Autopilot is still recovering from a required permission.", - }); - } - } finally { - if (session) { - await session.abort(); - await session.disconnect(); - } + enterPhase("final-assertions"); + expect(providerFailures).toEqual([]); + if (contract === "handler") { + expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + HUMAN_REVIEW_OUTPUT, + ]); } + console.info( + "Assisted contract completed:", + JSON.stringify({ + contract, + elapsedMs: Date.now() - started, + cleanupMs: completedPhases + .filter(({ phase }) => /cleanup|disconnect-before-resume/.test(phase)) + .reduce((total, { elapsedMs }) => total + elapsedMs, 0), + agentCalls, + judgeCalls: judgeOutputs.length, + providerFailures: providerFailures.length, + }) + ); } - - expect(providerFailures).toEqual([]); - expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - HUMAN_REVIEW_OUTPUT, - ]); - }); + ); }); async function configureAssistedAutopilot(session: CopilotSession, workDir: string): Promise { diff --git a/nodejs/test/rust-codegen.test.ts b/nodejs/test/rust-codegen.test.ts index 53c471b104..71c31fed5f 100644 --- a/nodejs/test/rust-codegen.test.ts +++ b/nodejs/test/rust-codegen.test.ts @@ -1,5 +1,6 @@ import type { ApiSchema } from "../../scripts/codegen/utils.ts"; import { + getApiSchemaPath, normalizeSchemaBrandCasing, postProcessSchema, propagateInternalVisibility, @@ -1015,18 +1016,13 @@ describe("Rust x-legacy-parameters", () => { ).toThrow(/Rust string enum Kind is requested for different values/); }); - it("keeps every const discriminator of the committed API schema distinct in Rust", () => { + it("keeps every const discriminator of the selected API schema distinct in Rust", async () => { // Mirror the generator's own schema preparation so emission order matches. const schema = propagateInternalVisibility( postProcessSchema( stripBooleanLiterals( normalizeSchemaBrandCasing( - JSON.parse( - readFileSync( - new URL("../../../../generated/api.schema.json", import.meta.url), - "utf8" - ) - ) as ApiSchema + JSON.parse(readFileSync(await getApiSchemaPath(), "utf8")) as ApiSchema ) ) as JSONSchema7 ) diff --git a/scripts/ci/device-policy-bootstrap.mjs b/scripts/ci/device-policy-bootstrap.mjs new file mode 100644 index 0000000000..48e30268e8 --- /dev/null +++ b/scripts/ci/device-policy-bootstrap.mjs @@ -0,0 +1,85 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { POLICY_FILE, verifyIdentity, verifyIsolation } from "./device-policy-fixture.mjs"; + +function start() { + const plan = JSON.parse(process.env.COPILOT_CI_DEVICE_POLICY_PLAN); + if (process.platform !== "linux" || typeof process.execve !== "function") + throw new Error("Unsupported device-policy container runtime"); + if (process.argv[2] === "--runtime") { + const policy = fs.lstatSync(POLICY_FILE); + verifyIdentity(plan, { + uid: process.getuid(), + gid: process.getgid(), + groups: process.getgroups(), + policy: { + isFile: policy.isFile(), + isSymbolicLink: policy.isSymbolicLink(), + uid: policy.uid, + gid: policy.gid, + mode: policy.mode, + }, + }); + process.execve(plan.executable, plan.argv, plan.env); + throw new Error("Runtime exec unexpectedly returned"); + } + verifyIsolation({ + originalNamespace: plan.originalNamespace, + namespace: fs.readlinkSync("/proc/self/ns/mnt"), + mountinfo: fs.readFileSync("/proc/self/mountinfo", "utf8"), + uid: process.getuid(), + gid: process.getgid(), + }); + if ( + !Number.isSafeInteger(plan.uid) || + plan.uid <= 0 || + !Number.isSafeInteger(plan.gid) || + plan.gid < 0 || + !plan.groups.every((group) => Number.isSafeInteger(group) && group >= 0) + ) + throw new Error("Invalid runner identity"); + const directory = path.dirname(POLICY_FILE); + if (fs.existsSync(directory)) throw new Error("The container already has a device-policy directory"); + fs.mkdirSync(directory, { mode: 0o755 }); + fs.writeFileSync(POLICY_FILE, plan.policy, { mode: 0o644, flag: "wx" }); + fs.chmodSync(POLICY_FILE, 0o644); + + const passwd = fs.readFileSync("/etc/passwd", "utf8"); + if (!passwd.split("\n").some((line) => Number(line.split(":")[2]) === plan.uid)) { + if (/[:\r\n]/.test(plan.home)) throw new Error("Invalid runner home"); + fs.appendFileSync("/etc/passwd", `copilot-sdk-ci:x:${plan.uid}:${plan.gid}::${plan.home}:/bin/sh\n`); + } + const group = fs.readFileSync("/etc/group", "utf8"); + if (!group.split("\n").some((line) => Number(line.split(":")[2]) === plan.gid)) { + fs.appendFileSync("/etc/group", `copilot-sdk-ci:x:${plan.gid}:\n`); + } + if (!fs.existsSync(plan.home)) { + fs.mkdirSync(plan.home, { recursive: true, mode: 0o700 }); + } + fs.chownSync(plan.home, plan.uid, plan.gid); + const args = [ + "/usr/bin/setpriv", + `--reuid=${plan.uid}`, + `--regid=${plan.gid}`, + ...(plan.groups.length ? [`--groups=${plan.groups.join(",")}`] : ["--clear-groups"]), + plan.nodeExecutable, + fileURLToPath(import.meta.url), + "--runtime", + ]; + process.execve("/usr/bin/setpriv", args, process.env); + throw new Error("Identity transition unexpectedly returned"); +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + try { + start(); + } catch (error) { + console.error(error.message); + process.exitCode = 1; + } +} diff --git a/scripts/ci/device-policy-fixture.mjs b/scripts/ci/device-policy-fixture.mjs new file mode 100644 index 0000000000..7c65a71354 --- /dev/null +++ b/scripts/ci/device-policy-fixture.mjs @@ -0,0 +1,468 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import { spawn, spawnSync } from "node:child_process"; +import { randomUUID } from "node:crypto"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const SDK_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); +export const IMAGE = "node:22-bookworm"; +export const POLICY_FILE = "/etc/github-copilot/managed-settings.json"; +export const OWNER_LABEL = "com.github.copilot-sdk.device-policy"; +export const CLIENT_LABEL = `${OWNER_LABEL}.client`; +export const REQUIRED_CASES = [ + { + file: "managed_plugin_progress.e2e.test.ts", + title: "emits presentation-neutral completion after installing required plugins", + }, + { + file: "rpc_server.e2e.test.ts", + title: "should round trip sessionless managed settings", + }, +]; + +const escapeRegex = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +export const REQUIRED_PATTERN = REQUIRED_CASES.map(({ title }) => escapeRegex(title)).join("|"); +export const PORTABLE_PATTERN = `^(?!.*(?:${REQUIRED_PATTERN})).*$`; + +export function portableTestPattern(source) { + if (source === "checkout") return ""; + if (source !== "published") throw new Error("Unknown runtime artifact source"); + return PORTABLE_PATTERN; +} + +export function runtimeInvocation(kind, runtime, nodeExecutable, args, env) { + if (!["native", "legacy"].includes(kind)) throw new Error("Unknown device-policy entry point"); + return { + executable: kind === "native" ? runtime : nodeExecutable, + argv: kind === "native" ? [runtime, ...args] : [nodeExecutable, runtime, ...args], + env: { ...env }, + }; +} + +export function verifyRequiredResults(report) { + for (const expected of REQUIRED_CASES) { + const matches = (report.testResults ?? []) + .filter((suite) => suite.name.replaceAll("\\", "/").endsWith(`/${expected.file}`)) + .flatMap((suite) => suite.assertionResults ?? []) + .filter((test) => test.title === expected.title); + if (matches.length !== 1 || matches[0].status !== "passed") { + throw new Error(`Required device-policy control did not pass: ${expected.title}`); + } + } +} + +function inside(parent, child) { + const relative = path.posix.relative(parent, child); + return relative === "" || (relative !== ".." && !relative.startsWith("../") && !path.posix.isAbsolute(relative)); +} + +function safeBind(directory) { + if (!path.posix.isAbsolute(directory) || /[:,\r\n]/.test(directory)) { + throw new Error("Device-policy binds require unambiguous absolute Linux paths"); + } + if ( + ["/", "/etc", "/proc", "/sys", "/dev"].some((root) => + root === "/" ? directory === root : inside(root, directory), + ) + ) { + throw new Error("A device-policy container cannot bind host system or policy paths"); + } + return directory; +} + +export function fixtureDirectories({ cwd, env, temporaryDirectory }) { + const candidates = [ + cwd, + ...["COPILOT_HOME", "GH_CONFIG_DIR", "XDG_CONFIG_HOME", "XDG_STATE_HOME"] + .map((name) => env[name]) + .filter(Boolean), + ]; + const roots = new Set(); + for (const candidate of candidates) { + const relative = path.posix.relative(temporaryDirectory, candidate); + const first = relative.split("/")[0]; + if (!inside(temporaryDirectory, candidate) || !/^copilot-test-(work|home|config)-[^/]+$/.test(first)) { + throw new Error("Writable device-policy binds must belong to an SDK E2E fixture"); + } + roots.add(safeBind(path.posix.join(temporaryDirectory, first))); + } + return [...roots]; +} + +export function dockerArguments(plan, cidfile) { + const args = [ + "run", + "--rm", + "--init", + "--interactive", + "--network", + "host", + "--cidfile", + cidfile, + "--label", + `${OWNER_LABEL}=${plan.owner}`, + "--label", + `${CLIENT_LABEL}=${plan.client}`, + "--workdir", + plan.cwd, + "--env", + "COPILOT_CI_DEVICE_POLICY_PLAN", + ]; + const mounts = new Map(); + for (const directory of plan.readonly) mounts.set(safeBind(directory), "ro"); + for (const directory of plan.writable) { + safeBind(directory); + if (mounts.has(directory)) throw new Error("A device-policy bind cannot be both read-only and writable"); + mounts.set(directory, "rw"); + } + for (const [directory, mode] of mounts) args.push("--volume", `${directory}:${directory}:${mode}`); + args.push(IMAGE, "node", path.posix.join(plan.sdkRoot, "scripts/ci/device-policy-bootstrap.mjs")); + return args; +} + +export function verifyIsolation({ originalNamespace, namespace, mountinfo, uid, gid }) { + if (uid !== 0 || gid !== 0 || namespace === originalNamespace || !namespace || !originalNamespace) { + throw new Error("Device policy must be provisioned inside a separate root container"); + } + const mounts = mountinfo + .trim() + .split("\n") + .map((line) => { + const fields = line.split(" "); + const separator = fields.indexOf("-"); + return { target: fields[4], type: fields[separator + 1], propagation: fields.slice(6, separator) }; + }); + const policyMount = mounts + .filter(({ target }) => inside(target, POLICY_FILE)) + .sort((a, b) => b.target.length - a.target.length)[0]; + if ( + !policyMount || + policyMount.target !== "/" || + policyMount.type !== "overlay" || + policyMount.propagation.some((field) => /^(shared|master|propagate_from):/.test(field)) + ) { + throw new Error("Device policy requires a private container root, not a host bind"); + } +} + +export function verifyIdentity(plan, { uid, gid, groups, policy }) { + if ( + uid !== plan.uid || + gid !== plan.gid || + JSON.stringify([...groups].sort((a, b) => a - b)) !== JSON.stringify([...plan.groups].sort((a, b) => a - b)) + ) { + throw new Error("The device-policy runtime must retain the original runner identity"); + } + if ( + !policy.isFile || + policy.isSymbolicLink || + policy.uid !== 0 || + policy.gid !== 0 || + (policy.mode & 0o777) !== 0o644 + ) { + throw new Error("The device policy must be a regular root-owned mode-0644 file"); + } +} + +function command(command, args, options = {}) { + const result = spawnSync(command, args, { encoding: "utf8", ...options }); + if (result.error) throw result.error; + if (result.status !== 0) throw new Error(`${command} failed (${result.status}): ${result.stderr ?? ""}`); + return result.stdout; +} + +export function containerId(value) { + const id = value.trim(); + if (!/^[a-f0-9]{64}$/.test(id)) throw new Error("Invalid owned device-policy container ID"); + return id; +} + +export function verifyOwnership(label, owner) { + if (!owner || label.trim() !== owner) throw new Error("Refusing to mutate a container owned by another job"); +} + +export function pendingLaunches(entries) { + for (const entry of entries) { + if (!/^[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}\.(pending|cid)$/.test(entry)) + throw new Error("Unknown device-policy cleanup registry entry"); + const pending = entry.replace(/\.cid$/, ".pending"); + if (!entries.includes(pending)) throw new Error("Device-policy container identity has no launch receipt"); + } + return entries.filter((entry) => entry.endsWith(".pending")); +} + +function inspectOwned(id, owner) { + const inspected = spawnSync("docker", ["inspect", "--format", `{{index .Config.Labels "${OWNER_LABEL}"}}`, id], { + encoding: "utf8", + }); + if (inspected.status !== 0 && /No such (object|container)/i.test(inspected.stderr)) return false; + if (inspected.error || inspected.status !== 0) + throw inspected.error ?? new Error(`Cannot inspect device-policy container: ${inspected.stderr}`); + verifyOwnership(inspected.stdout, owner); + return true; +} + +function removeOwned(id, owner) { + if (!inspectOwned(id, owner)) return; + const removed = spawnSync("docker", ["rm", "--force", id], { encoding: "utf8" }); + if (removed.status !== 0 && /No such (object|container)/i.test(removed.stderr)) return; + if (removed.error || removed.status !== 0) + throw removed.error ?? new Error(`Cannot remove device-policy container: ${removed.stderr}`); +} + +function stopOwned(id, owner, signal) { + if (!inspectOwned(id, owner)) return; + const stopped = spawnSync("docker", ["stop", "--signal", signal, "--timeout", "5", id], { encoding: "utf8" }); + if (stopped.status !== 0 && /No such (object|container)/i.test(stopped.stderr)) return; + if (stopped.error || stopped.status !== 0) + throw stopped.error ?? new Error(`Cannot stop device-policy container: ${stopped.stderr}`); +} + +function ownedRegistry() { + const registry = process.env.COPILOT_CI_DEVICE_POLICY_REGISTRY; + if (!registry || !process.env.RUNNER_TEMP) throw new Error("Missing owned device-policy container registry"); + const info = fs.lstatSync(registry); + if ( + !inside(process.env.RUNNER_TEMP, registry) || + !path.basename(registry).startsWith("sdk-device-policy-") || + fs.realpathSync(registry) !== registry || + !info.isDirectory() || + info.isSymbolicLink() || + info.uid !== process.getuid() + ) { + throw new Error("Device-policy cidfiles must belong to this runner's temporary registry"); + } + return registry; +} + +export async function launchRuntime(kind) { + if (process.platform !== "linux" || typeof process.execve !== "function") { + throw new Error("The device-policy launcher requires Linux and Node.js 22.15 or newer"); + } + const runtime = + process.env[ + kind === "native" ? "COPILOT_CI_DEVICE_POLICY_NATIVE_PATH" : "COPILOT_CI_DEVICE_POLICY_LEGACY_PATH" + ]; + if (!runtime) throw new Error("Missing original device-policy runtime path"); + const invocation = runtimeInvocation(kind, runtime, process.execPath, process.argv.slice(2), process.env); + const originalEnv = invocation.env; + const fixture = process.env.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; + if (!fixture) { + process.execve(invocation.executable, invocation.argv, originalEnv); + throw new Error("Original runtime exec unexpectedly returned"); + } + + if (!inside(process.cwd(), fixture) || fs.realpathSync(fixture) !== fixture) + throw new Error("Device policy must be a regular file inside the test workspace"); + const descriptor = fs.openSync(fixture, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW); + let policy; + try { + const fixtureInfo = fs.fstatSync(descriptor); + if (!fixtureInfo.isFile() || fixtureInfo.uid !== process.getuid()) + throw new Error("Device policy must be a regular file owned by the runner"); + policy = fs.readFileSync(descriptor, "utf8"); + } finally { + fs.closeSync(descriptor); + } + const parsed = JSON.parse(policy); + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) + throw new Error("Device policy must be a JSON object"); + // The launcher consumes the test-only path; the runtime must discover the production file. + delete originalEnv.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; + const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; + if (!owner) throw new Error("Missing device-policy job ownership"); + const registry = ownedRegistry(); + const writable = fixtureDirectories({ cwd: process.cwd(), env: originalEnv, temporaryDirectory: os.tmpdir() }); + for (const directory of writable) { + const info = fs.lstatSync(directory); + if (fs.realpathSync(directory) !== directory || !info.isDirectory() || info.uid !== process.getuid()) + throw new Error("A writable fixture bind must be a real directory owned by the runner"); + } + const readonly = [ + SDK_ROOT, + path.dirname(process.env.COPILOT_CI_DEVICE_POLICY_LEGACY_PATH), + path.dirname(process.execPath), + ]; + for (const name of [ + "NODE_EXTRA_CA_CERTS", + "SSL_CERT_FILE", + "REQUESTS_CA_BUNDLE", + "CURL_CA_BUNDLE", + "GIT_SSL_CAINFO", + ]) { + const certificate = originalEnv[name]; + if (certificate) { + if (!inside(os.tmpdir(), certificate) || !fs.statSync(certificate).isFile()) + throw new Error("Replay certificates must belong to the test temporary directory"); + readonly.push(certificate); + } + } + const plan = { + owner, + client: randomUUID(), + sdkRoot: SDK_ROOT, + readonly: [...new Set(readonly)], + writable, + policy, + ...invocation, + cwd: process.cwd(), + nodeExecutable: process.execPath, + uid: process.getuid(), + gid: process.getgid(), + groups: process.getgroups(), + home: os.homedir(), + originalNamespace: fs.readlinkSync("/proc/self/ns/mnt"), + }; + const cidfile = path.join(registry, `${plan.client}.cid`); + const pending = path.join(registry, `${plan.client}.pending`); + fs.writeFileSync(pending, "", { flag: "wx", mode: 0o600 }); + const child = spawn("docker", dockerArguments(plan, cidfile), { + stdio: "inherit", + env: { ...originalEnv, COPILOT_CI_DEVICE_POLICY_PLAN: JSON.stringify(plan) }, + }); + let stopping; + const stop = (signal) => { + if (stopping) return; + stopping = signal; + try { + if (fs.existsSync(cidfile)) stopOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner, signal); + else child.kill(signal); + } catch (error) { + console.error(error.message); + process.exitCode = 1; + child.kill("SIGTERM"); + } + }; + const onTerm = () => stop("SIGTERM"); + const onInt = () => stop("SIGINT"); + process.once("SIGTERM", onTerm); + process.once("SIGINT", onInt); + let result; + try { + result = await new Promise((resolve, reject) => { + child.once("error", reject); + child.once("close", (code, signal) => resolve({ code, signal })); + }); + } finally { + process.removeListener("SIGTERM", onTerm); + process.removeListener("SIGINT", onInt); + // A cancelled Docker attach can close before its cidfile is written. + const ids = command("docker", [ + "ps", + "--all", + "--quiet", + "--no-trunc", + "--filter", + `label=${OWNER_LABEL}=${owner}`, + "--filter", + `label=${CLIENT_LABEL}=${plan.client}`, + ]).trim(); + for (const id of ids ? ids.split("\n") : []) removeOwned(containerId(id), owner); + if (fs.existsSync(cidfile)) { + removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); + fs.unlinkSync(cidfile); + fs.unlinkSync(pending); + } else { + throw new Error("Docker launch has no container identity; cleanup cannot prove settlement"); + } + } + if (process.exitCode) return; + if (stopping || result.signal) process.kill(process.pid, stopping ?? result.signal); + else process.exitCode = result.code ?? 1; +} + +function prepare() { + if ( + process.platform !== "linux" || + !process.env.GITHUB_ENV || + !process.env.GITHUB_OUTPUT || + !process.env.RUNNER_TEMP + ) { + throw new Error("Device-policy containers can only be prepared in Linux CI"); + } + const native = fs.realpathSync(process.env.COPILOT_CLI_PATH); + const legacy = fs.realpathSync(process.env.COPILOT_LEGACY_CLI_PATH); + const packageRoot = path.dirname(legacy); + if ( + native !== path.join(packageRoot, "prebuilds/linux-x64/copilot-runtime") || + path.basename(legacy) !== "app.js" + ) { + throw new Error("Device-policy fixtures require the staged GNU package's two original entry points"); + } + const expected = JSON.parse(fs.readFileSync(path.join(SDK_ROOT, "nodejs/package.json"), "utf8")).copilotCliVersion; + const actual = JSON.parse(fs.readFileSync(path.join(packageRoot, "package.json"), "utf8")).version; + if (expected !== actual) throw new Error("Device-policy package differs from the pinned CLI version"); + if (fs.existsSync(POLICY_FILE)) throw new Error("The CI host already has a device policy"); + command("docker", ["pull", IMAGE], { stdio: "inherit" }); + const registry = fs.mkdtempSync(path.join(process.env.RUNNER_TEMP, "sdk-device-policy-")); + const owner = randomUUID(); + fs.appendFileSync( + process.env.GITHUB_ENV, + `COPILOT_CI_DEVICE_POLICY_NATIVE_PATH=${native}\nCOPILOT_CI_DEVICE_POLICY_LEGACY_PATH=${legacy}\nCOPILOT_CI_DEVICE_POLICY_OWNER=${owner}\nCOPILOT_CI_DEVICE_POLICY_REGISTRY=${registry}\n`, + ); + fs.appendFileSync( + process.env.GITHUB_OUTPUT, + `native-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-native.js")}\nlegacy-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-legacy.js")}\n`, + ); +} + +function cleanup() { + const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; + if (!owner) throw new Error("Missing device-policy job ownership"); + const ids = command("docker", [ + "ps", + "--all", + "--quiet", + "--no-trunc", + "--filter", + `label=${OWNER_LABEL}=${owner}`, + ]).trim(); + const errors = []; + for (const id of ids ? ids.split("\n") : []) { + try { + removeOwned(containerId(id), owner); + } catch (error) { + errors.push(error.message); + } + } + const registry = ownedRegistry(); + for (const entry of pendingLaunches(fs.readdirSync(registry))) { + const pending = path.join(registry, entry); + const cidfile = path.join(registry, entry.replace(/\.pending$/, ".cid")); + try { + if (!fs.existsSync(cidfile)) throw new Error("Unsettled Docker launch has no container identity"); + removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); + fs.unlinkSync(cidfile); + fs.unlinkSync(pending); + } catch (error) { + errors.push(error.message); + } + } + if (errors.length) throw new Error(`Device-policy cleanup did not settle: ${errors.join("; ")}`); + if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + try { + const [action, argument] = process.argv.slice(2); + if (action === "prepare") prepare(); + else if (action === "cleanup") cleanup(); + else if (action === "portable-pattern") console.log(portableTestPattern(argument)); + else if (action === "required-pattern") console.log(REQUIRED_PATTERN); + else if (action === "verify" && argument) { + verifyRequiredResults(JSON.parse(fs.readFileSync(argument, "utf8"))); + if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); + } else + throw new Error( + "Usage: device-policy-fixture.mjs ", + ); + } catch (error) { + console.error(error.message); + process.exitCode = 1; + } +} diff --git a/scripts/ci/device-policy-fixture.test.mjs b/scripts/ci/device-policy-fixture.test.mjs new file mode 100644 index 0000000000..8f33a412c3 --- /dev/null +++ b/scripts/ci/device-policy-fixture.test.mjs @@ -0,0 +1,207 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import assert from "node:assert/strict"; +import test from "node:test"; +import "./device-policy-bootstrap.mjs"; +import { + CLIENT_LABEL, + containerId, + dockerArguments, + fixtureDirectories, + OWNER_LABEL, + pendingLaunches, + POLICY_FILE, + portableTestPattern, + runtimeInvocation, + PORTABLE_PATTERN, + REQUIRED_CASES, + REQUIRED_PATTERN, + verifyIdentity, + verifyIsolation, + verifyOwnership, + verifyRequiredResults, +} from "./device-policy-fixture.mjs"; + +const report = () => ({ + testResults: REQUIRED_CASES.map(({ file, title }) => ({ + name: `/workspace/nodejs/test/e2e/${file}`, + assertionResults: [{ title, status: "passed" }], + })), +}); + +test("requires both unchanged strict controls to pass", () => { + verifyRequiredResults(report()); + for (const index of [0, 1]) { + for (const status of ["pending", "skipped", "failed", "todo"]) { + const result = report(); + result.testResults[index].assertionResults[0].status = status; + assert.throws(() => verifyRequiredResults(result), /did not pass/); + } + const missing = report(); + missing.testResults.splice(index, 1); + assert.throws(() => verifyRequiredResults(missing), /did not pass/); + const duplicate = report(); + duplicate.testResults.push(duplicate.testResults[index]); + assert.throws(() => verifyRequiredResults(duplicate), /did not pass/); + } + assert.throws(() => verifyRequiredResults({}), /did not pass/); +}); + +test("assigns only the two device-fixture cases to the mandatory Linux gate", () => { + assert.equal(portableTestPattern("checkout"), ""); + assert.equal(portableTestPattern("published"), PORTABLE_PATTERN); + assert.throws(() => portableTestPattern(""), /Unknown/); + const portable = new RegExp(PORTABLE_PATTERN); + const required = new RegExp(REQUIRED_PATTERN); + for (const { title } of REQUIRED_CASES) { + assert.equal(portable.test(`Suite ${title}`), false); + assert.equal(required.test(title), true); + } + for (const title of [ + "should clear the managed settings cache", + "should expose the managed settings schema", + "should list server sessions", + ]) { + assert.equal(portable.test(title), true); + assert.equal(required.test(title), false); + } +}); + +test("preserves both runtime entry points' original argv and environment", () => { + const args = ["--stdio", "--argument-with-spaces=one two", "--literal=;$()"]; + const env = { TOKEN: "private-value", COPILOT_HOME: "/tmp/copilot-test-home-one" }; + const native = runtimeInvocation("native", "/package/copilot-runtime", "/node/bin/node", args, env); + assert.equal(native.executable, "/package/copilot-runtime"); + assert.deepEqual(native.argv, ["/package/copilot-runtime", ...args]); + const legacy = runtimeInvocation("legacy", "/package/app.js", "/node/bin/node", args, env); + assert.equal(legacy.executable, "/node/bin/node"); + assert.deepEqual(legacy.argv, ["/node/bin/node", "/package/app.js", ...args]); + assert.deepEqual(native.env, env); + assert.notEqual(native.env, env); + assert.throws(() => runtimeInvocation("unknown", "", "", [], {}), /entry point/); +}); + +test("binds only explicitly owned E2E fixture directories for writing", () => { + const roots = fixtureDirectories({ + cwd: "/tmp/copilot-test-work-one", + env: { + COPILOT_HOME: "/tmp/copilot-test-home-one", + GH_CONFIG_DIR: "/tmp/copilot-test-config-one", + XDG_CONFIG_HOME: "/tmp/copilot-test-config-one", + XDG_STATE_HOME: "/tmp/copilot-test-work-one/local-home", + }, + temporaryDirectory: "/tmp", + }); + assert.deepEqual(roots, [ + "/tmp/copilot-test-work-one", + "/tmp/copilot-test-home-one", + "/tmp/copilot-test-config-one", + ]); + for (const cwd of ["/etc", "/home/runner", "/tmp", "/tmp/../../etc", "/tmp/not-a-fixture"]) { + assert.throws(() => fixtureDirectories({ cwd, env: {}, temporaryDirectory: "/tmp" }), /fixture/); + } +}); + +test("keeps policy and secret environment contents out of Docker arguments", () => { + const plan = { + owner: "job-one", + client: "client-one", + sdkRoot: "/workspace", + cwd: "/tmp/copilot-test-work-one", + readonly: ["/workspace", "/artifacts/package"], + writable: ["/tmp/copilot-test-work-one"], + env: { SECRET: "must-not-appear" }, + policy: '{"model":"secret-model"}', + }; + const args = dockerArguments(plan, "/registry/client-one.cid"); + assert.deepEqual(args.slice(0, 7), ["run", "--rm", "--init", "--interactive", "--network", "host", "--cidfile"]); + assert.equal(args.includes(`${OWNER_LABEL}=job-one`), true); + assert.equal(args.includes(`${CLIENT_LABEL}=client-one`), true); + assert.equal(args.includes("COPILOT_CI_DEVICE_POLICY_PLAN"), true); + assert.equal(args.includes("/artifacts/package:/artifacts/package:ro"), true); + assert.equal(args.includes("/tmp/copilot-test-work-one:/tmp/copilot-test-work-one:rw"), true); + assert.equal(args.join(" ").includes("must-not-appear"), false); + assert.equal(args.join(" ").includes("secret-model"), false); + const other = dockerArguments({ ...plan, client: "client-two" }, "/registry/client-two.cid"); + assert.notDeepEqual(args, other); + for (const directory of ["/", "/etc", "/etc/github-copilot", "/proc", "/sys", "/dev", "/path:ambiguous"]) { + assert.throws(() => dockerArguments({ ...plan, readonly: [directory] }, "/registry/client.cid")); + } + assert.throws(() => dockerArguments({ ...plan, readonly: plan.writable }, "/registry/client.cid"), /both/); +}); + +test("requires exact job ownership and full container IDs before cleanup", () => { + verifyOwnership("job-one\n", "job-one"); + for (const [label, owner] of [ + ["job-two", "job-one"], + ["", ""], + ["", "job-one"], + ]) { + assert.throws(() => verifyOwnership(label, owner), /another job/); + } + assert.equal(containerId(`${"a".repeat(64)}\n`), "a".repeat(64)); + for (const id of ["", "a".repeat(12), `${"a".repeat(64)} extra`, "G".repeat(64)]) { + assert.throws(() => containerId(id), /container ID/); + } +}); + +test("exposes interrupted or unknown launch receipts instead of claiming cleanup", () => { + const client = "12345678-1234-1234-1234-123456789abc"; + assert.deepEqual(pendingLaunches([]), []); + assert.deepEqual(pendingLaunches([`${client}.pending`, `${client}.cid`]), [`${client}.pending`]); + assert.deepEqual(pendingLaunches([`${client}.pending`]), [`${client}.pending`]); + assert.throws(() => pendingLaunches([`${client}.cid`]), /no launch receipt/); + assert.throws(() => pendingLaunches(["unknown"]), /Unknown/); +}); + +const isolated = () => ({ + originalNamespace: "mnt:[100]", + namespace: "mnt:[200]", + uid: 0, + gid: 0, + mountinfo: "1 0 0:1 / / rw - overlay overlay rw\n2 1 0:2 / /workspace ro - ext4 disk ro", +}); + +test("requires private container backing before any policy mutation", () => { + verifyIsolation(isolated()); + for (const change of [ + { namespace: "mnt:[100]" }, + { uid: 1001 }, + { gid: 1001 }, + { originalNamespace: "" }, + { mountinfo: "1 0 0:1 / / rw shared:1 - overlay overlay rw" }, + { mountinfo: "1 0 0:1 / / rw - ext4 disk rw" }, + { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc rw - ext4 host rw` }, + { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc/github-copilot rw - ext4 host rw` }, + ]) + assert.throws(() => verifyIsolation({ ...isolated(), ...change })); +}); + +test("retains runner uid, gid and groups with an ordinary root-owned device file", () => { + const plan = { uid: 1001, gid: 1001, groups: [1001, 118] }; + const runtime = { + uid: 1001, + gid: 1001, + groups: [118, 1001], + policy: { isFile: true, isSymbolicLink: false, uid: 0, gid: 0, mode: 0o100644 }, + }; + verifyIdentity(plan, runtime); + for (const change of [{ uid: 0 }, { gid: 0 }, { groups: [] }]) { + assert.throws(() => verifyIdentity(plan, { ...runtime, ...change }), /identity/); + } + for (const change of [ + { uid: 1001 }, + { gid: 1001 }, + { mode: 0o100666 }, + { isSymbolicLink: true }, + { isFile: false }, + ]) { + assert.throws( + () => verifyIdentity(plan, { ...runtime, policy: { ...runtime.policy, ...change } }), + /root-owned/, + ); + } + assert.equal(POLICY_FILE, "/etc/github-copilot/managed-settings.json"); +}); diff --git a/scripts/ci/device-policy-legacy.js b/scripts/ci/device-policy-legacy.js new file mode 100644 index 0000000000..806647cf11 --- /dev/null +++ b/scripts/ci/device-policy-legacy.js @@ -0,0 +1,10 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import("./device-policy-fixture.mjs") + .then(({ launchRuntime }) => launchRuntime("legacy")) + .catch((error) => { + console.error(error.message); + process.exitCode = 1; + }); diff --git a/scripts/ci/device-policy-native.js b/scripts/ci/device-policy-native.js new file mode 100644 index 0000000000..d269a07ea6 --- /dev/null +++ b/scripts/ci/device-policy-native.js @@ -0,0 +1,10 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (c) Microsoft Corporation. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import("./device-policy-fixture.mjs") + .then(({ launchRuntime }) => launchRuntime("native")) + .catch((error) => { + console.error(error.message); + process.exitCode = 1; + }); diff --git a/scripts/ci/run-dotnet-tests.sh b/scripts/ci/run-dotnet-tests.sh index ea0312c4a6..14d9a97208 100755 --- a/scripts/ci/run-dotnet-tests.sh +++ b/scripts/ci/run-dotnet-tests.sh @@ -11,6 +11,7 @@ Usage: run-dotnet-tests.sh [--help] Runs the full .NET SDK test project. Environment variables: DOTNET_TEST_FILTER Optional dotnet test filter (e.g. for a backend or transport). DOTNET_TEST_RUNTIME Optional runtime identifier passed to dotnet test. + DOTNET_TEST_FRAMEWORK Optional target framework; unset runs all project targets. DOTNET_TEST_RESULTS_DIRECTORY Results directory (default: TestResults). EOF } @@ -40,6 +41,7 @@ args=( ) filter="${DOTNET_TEST_FILTER:-}" runtime="${DOTNET_TEST_RUNTIME:-}" +framework="${DOTNET_TEST_FRAMEWORK:-}" if [[ -n "$filter" ]]; then args+=(--filter "$filter") @@ -47,5 +49,8 @@ fi if [[ -n "$runtime" ]]; then args+=(--runtime "$runtime") fi +if [[ -n "$framework" ]]; then + args+=(--framework "$framework") +fi dotnet test "${args[@]}" From c83855abfdd269d58d51d5c6cf2bb2b7bb3f2305 Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 20:18:06 -0700 Subject: [PATCH 14/15] test: expose queued-shell failure boundary without changing scheduling Record only phase, counts and monotonic timing from existing observations. Preserve original RPCs, waits, assertions and cleanup. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../e2e/rpc_tasks_and_handlers.e2e.test.ts | 52 +++++++++++++++++-- 1 file changed, 48 insertions(+), 4 deletions(-) diff --git a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts index e6e6a18c8f..b1336ddf27 100644 --- a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts +++ b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts @@ -2,7 +2,7 @@ * Copyright (c) Microsoft Corporation. All rights reserved. *--------------------------------------------------------------------------------------------*/ -import { describe, expect, it } from "vitest"; +import { describe, expect, it, onTestFailed } from "vitest"; import { z } from "zod"; import { approveAll, CopilotRequestHandler } from "../../src/index.js"; import type { SessionEvent, CopilotRequestContext, CopilotSession } from "../../src/index.js"; @@ -390,25 +390,69 @@ describe("Session tasks RPC and pending handlers", async () => { { timeout: 120_000 }, async () => { await withRunningAttachedShell(async (session, shellId, eventTypes, replies) => { + const started = performance.now(); + let phase = "enqueue"; + let queuePolls = 0; + let pendingCount = 0; + let steeringCount = 0; + let inFlightSteeringCount: number | undefined; + let queuedMatch = false; + onTestFailed(() => { + console.error( + "Attached shell queue failure boundary:", + JSON.stringify({ + phase, + elapsedMs: Math.round(performance.now() - started), + queuePolls, + pendingCount, + steeringCount, + inFlightSteeringCount, + queuedMatch, + replyCount: replies.length, + hasQueuedReply: replies.some((reply) => reply.includes("QUEUED_DONE")), + sessionIdleCount: eventTypes.filter((type) => type === "session.idle") + .length, + assistantIdleCount: eventTypes.filter( + (type) => type === "assistant.idle" + ).length, + backgroundTaskChangeCount: eventTypes.filter( + (type) => type === "session.background_tasks_changed" + ).length, + }) + ); + }); // The running attached shell holds idle, so an enqueued message waits behind it. await session.send({ prompt: "Reply with exactly QUEUED_DONE.", mode: "enqueue" }); + phase = "pending-queue"; await waitForCondition( - async () => - (await session.rpc.queue.pendingItems()).items.some((item) => + async () => { + const pending = await session.rpc.queue.pendingItems(); + const { items } = pending; + queuePolls++; + pendingCount = items.length; + steeringCount = pending.steeringMessages.length; + inFlightSteeringCount = pending.inFlightSteeringCount; + queuedMatch = items.some((item) => item.displayText.includes("QUEUED_DONE") - ), + ); + return queuedMatch; + }, { timeoutMessage: "The enqueued message was not parked behind the shell" } ); + phase = "assert-parked"; expect(replies.some((reply) => reply.includes("QUEUED_DONE"))).toBe(false); expect(eventTypes).not.toContain("session.idle"); + phase = "cancel-shell"; expect((await session.rpc.tasks.cancel({ id: shellId })).cancelled).toBe(true); + phase = "queued-reply"; await waitForCondition( () => replies.some((reply) => reply.includes("QUEUED_DONE")), { timeoutMessage: `The queued message never ran after cancelling shell ${shellId}`, } ); + phase = "session-idle"; await waitForCondition(() => eventTypes.includes("session.idle"), { timeoutMessage: "session.idle never followed the queued message", }); From 2c4bbe596284b690b3cf65f1ce72962be6a06d2e Mon Sep 17 00:00:00 2001 From: Chuck Lantz Date: Sat, 10 Oct 2026 20:44:36 -0700 Subject: [PATCH 15/15] chore: remove unrelated SDK CI expansion Restore the accepted canonical test and workflow setup. Keep the Rust-only resume capability and its regression tests unchanged while runtime root-cause investigation continues separately. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/actions/run-alpine-tests/action.yml | 1 - .github/workflows/sdk-dotnet.yml | 18 +- .github/workflows/sdk-go.yml | 7 +- .github/workflows/sdk-java.yml | 7 +- .github/workflows/sdk-nodejs.yml | 65 +- .github/workflows/sdk-platform.yml | 9 - .github/workflows/sdk-python.yml | 7 +- .github/workflows/sdk-rust.yml | 7 +- .github/workflows/sdk.yml | 5 - CONTRIBUTING.md | 37 - .../assisted_autopilot_permission.e2e.test.ts | 1922 ++++++++--------- .../e2e/rpc_tasks_and_handlers.e2e.test.ts | 52 +- nodejs/test/rust-codegen.test.ts | 10 +- scripts/ci/device-policy-bootstrap.mjs | 85 - scripts/ci/device-policy-fixture.mjs | 468 ---- scripts/ci/device-policy-fixture.test.mjs | 207 -- scripts/ci/device-policy-legacy.js | 10 - scripts/ci/device-policy-native.js | 10 - scripts/ci/run-dotnet-tests.sh | 5 - 19 files changed, 935 insertions(+), 1997 deletions(-) delete mode 100644 scripts/ci/device-policy-bootstrap.mjs delete mode 100644 scripts/ci/device-policy-fixture.mjs delete mode 100644 scripts/ci/device-policy-fixture.test.mjs delete mode 100644 scripts/ci/device-policy-legacy.js delete mode 100644 scripts/ci/device-policy-native.js diff --git a/.github/actions/run-alpine-tests/action.yml b/.github/actions/run-alpine-tests/action.yml index 7808ca3f67..a9842c11be 100644 --- a/.github/actions/run-alpine-tests/action.yml +++ b/.github/actions/run-alpine-tests/action.yml @@ -54,7 +54,6 @@ runs: --env COPILOT_SDK_ROOT="$ALPINE_TEST_SDK_ROOT" \ --env COPILOT_HMAC_KEY \ --env COPILOT_SDK_E2E_BACKEND \ - --env COPILOT_CI_RUNTIME_SOURCE \ --env BUNDLED_CLI_CACHE_DIR \ --env CARGO_TERM_COLOR \ --env RUST_BACKTRACE \ diff --git a/.github/workflows/sdk-dotnet.yml b/.github/workflows/sdk-dotnet.yml index 7b2e426e81..14a4c3265b 100644 --- a/.github/workflows/sdk-dotnet.yml +++ b/.github/workflows/sdk-dotnet.yml @@ -19,9 +19,6 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read @@ -102,9 +99,9 @@ jobs: retention-days: 7 dotnet-darwin-arm64: - name: ".NET (${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }}, default, CAPI)" + name: ".NET (macos-26-xlarge-agent-runtime, default, CAPI)" if: inputs.platform == 'all' || inputs.platform == 'darwin-arm64' - runs-on: ${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge-agent-runtime' }} + runs-on: macos-26-xlarge-agent-runtime timeout-minutes: 30 defaults: run: @@ -150,18 +147,14 @@ jobs: - if: failure() uses: actions/upload-artifact@v7 with: - name: dotnet-test-diagnostics-${{ inputs.runtime-source == 'published' && 'macos-26' || 'macos-26-xlarge' }}-default-capi-${{ github.run_attempt }} + name: dotnet-test-diagnostics-macos-26-xlarge-default-capi-${{ github.run_attempt }} path: ${{ inputs.sdk-home }}/dotnet/TestResults/ if-no-files-found: warn retention-days: 7 dotnet-win32-x64: - name: ".NET (windows-latest, ${{ matrix.framework == 'all' && 'default' || matrix.framework }}, CAPI)" + name: ".NET (windows-latest, default, CAPI)" if: inputs.platform == 'all' || inputs.platform == 'win32-x64' - strategy: - fail-fast: false - matrix: - framework: ${{ fromJSON(inputs.runtime-source == 'published' && '["net8.0","net472"]' || '["all"]') }} runs-on: windows-latest timeout-minutes: 30 defaults: @@ -171,7 +164,6 @@ jobs: env: COPILOT_SDK_E2E_BACKEND: capi DOTNET_TEST_RUNTIME: win-x64 - DOTNET_TEST_FRAMEWORK: ${{ matrix.framework != 'all' && matrix.framework || '' }} DOTNET_TEST_RESULTS_DIRECTORY: TestResults/subprocess steps: - uses: actions/checkout@v7 @@ -211,7 +203,7 @@ jobs: - if: failure() uses: actions/upload-artifact@v7 with: - name: dotnet-test-diagnostics-windows-latest-${{ matrix.framework == 'all' && 'default' || matrix.framework }}-capi-${{ github.run_attempt }} + name: dotnet-test-diagnostics-windows-latest-default-capi-${{ github.run_attempt }} path: ${{ inputs.sdk-home }}/dotnet/TestResults/ if-no-files-found: warn retention-days: 7 diff --git a/.github/workflows/sdk-go.yml b/.github/workflows/sdk-go.yml index 25b63039a9..b5593f26b1 100644 --- a/.github/workflows/sdk-go.yml +++ b/.github/workflows/sdk-go.yml @@ -19,22 +19,19 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: go: - name: "Go (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" + name: "Go (${{ matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-java.yml b/.github/workflows/sdk-java.yml index 3d022571e3..011fef0898 100644 --- a/.github/workflows/sdk-java.yml +++ b/.github/workflows/sdk-java.yml @@ -18,16 +18,13 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: java: - name: "Java (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, JDK ${{ matrix.test-jdk }})" + name: "Java (${{ matrix.os }}, JDK ${{ matrix.test-jdk }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false @@ -39,7 +36,7 @@ jobs: test-jdk: "17" - os: windows-latest test-jdk: "17" - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk-nodejs.yml b/.github/workflows/sdk-nodejs.yml index f94e00bfe4..db14685c13 100644 --- a/.github/workflows/sdk-nodejs.yml +++ b/.github/workflows/sdk-nodejs.yml @@ -7,7 +7,6 @@ env: HUSKY: 0 POWERSHELL_UPDATECHECK: Off SDK_HOME: ${{ inputs.sdk-home }} - COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} SETUP_NODE_TIMEOUT_MINUTES: 10 on: @@ -20,9 +19,6 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string outputs: capi-result: description: "Standard-platform CAPI job result, independent of BYOK jobs" @@ -34,13 +30,13 @@ permissions: jobs: nodejs: - name: "Node.js (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }}, CAPI)" + name: "Node.js (${{ matrix.os }}, CAPI)" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash @@ -93,51 +89,16 @@ jobs: working-directory: ${{ inputs.sdk-home }}/scripts/corrections - if: runner.os == 'Windows' run: pwsh.exe -Command "Write-Host 'PowerShell ready'" - - name: Test device-policy CI plans and required-result guard - if: runner.os == 'Linux' - working-directory: . - run: node --test "$SDK_HOME/scripts/ci/device-policy-fixture.test.mjs" - - name: Prepare isolated production device-policy fixtures - id: device-policy - if: inputs.runtime-source == 'published' && runner.os == 'Linux' - working-directory: . - run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" prepare - name: Test out-of-process id: subprocess env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} run: | - test_args=() - pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - case "$pattern" in - "") ;; - *) - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - test_args+=(--testNamePattern "$pattern") - ;; - esac if [ "$GITHUB_EVENT_NAME" = "merge_group" ] || [ "$GITHUB_EVENT_NAME" = "pull_request" ] || [ "$GITHUB_EVENT_NAME" = "push" ]; then - npm test -- "${test_args[@]}" --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" + npm test -- --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-nodejs-results.json" else - npm test -- "${test_args[@]}" + npm test fi - - name: Test required production device-policy controls - if: ${{ !cancelled() && inputs.runtime-source == 'published' && runner.os == 'Linux' && (steps.subprocess.outcome == 'success' || steps.subprocess.outcome == 'failure') }} - env: - COPILOT_CLI_PATH: ${{ steps.device-policy.outputs.native-launcher }} - COPILOT_LEGACY_CLI_PATH: ${{ steps.device-policy.outputs.legacy-launcher }} - COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - run: | - status=0 - npm test -- test/e2e/managed_plugin_progress.e2e.test.ts test/e2e/rpc_server.e2e.test.ts \ - --testNamePattern "$(node ../scripts/ci/device-policy-fixture.mjs required-pattern)" \ - --reporter=default --reporter=json --outputFile="$RUNNER_TEMP/sdk-device-policy-results.json" || status=$? - node ../scripts/ci/device-policy-fixture.mjs verify "$RUNNER_TEMP/sdk-device-policy-results.json" - exit "$status" - - name: Clean up owned device-policy containers - if: always() && steps.device-policy.outcome == 'success' - working-directory: . - run: node "$SDK_HOME/scripts/ci/device-policy-fixture.mjs" cleanup - name: Upload Flake Finder Node.js test results if: always() && (github.event_name == 'merge_group' || github.event_name == 'pull_request' || github.event_name == 'push') continue-on-error: true @@ -201,14 +162,7 @@ jobs: - name: Test out-of-process E2Es env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - run: | - test_args=() - pattern=$(node ../scripts/ci/device-policy-fixture.mjs portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - if [[ -n "$pattern" ]]; then - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - test_args+=(--testNamePattern "$pattern") - fi - npm test -- test/e2e "${test_args[@]}" + run: npm test -- test/e2e nodejs-musl-x64: name: "Node.js (Alpine x64, CAPI)" @@ -228,7 +182,6 @@ jobs: - uses: ./.github/actions/run-alpine-tests env: COPILOT_HMAC_KEY: ${{ secrets.COPILOT_DEVELOPER_CLI_INTEGRATION_HMAC_KEY }} - COPILOT_CI_RUNTIME_SOURCE: ${{ inputs.runtime-source }} with: image: node:22-alpine sdk-root: /workspace/${{ inputs.sdk-home }} @@ -253,12 +206,6 @@ jobs: test -x "$COPILOT_CLI_PATH" test -f "$(dirname "$COPILOT_CLI_PATH")/runtime.node" status=0 - pattern=$(node "$COPILOT_SDK_ROOT/scripts/ci/device-policy-fixture.mjs" portable-pattern "$COPILOT_CI_RUNTIME_SOURCE") - if [ -n "$pattern" ]; then - echo "Device-policy fixture controls are assigned to the mandatory Linux CAPI step" - npm test -- --testNamePattern "$pattern" || status=$? - else - npm test || status=$? - fi + npm test || status=$? COPILOT_SDK_DEFAULT_CONNECTION=inprocess npm test -- test/e2e/inprocess_ffi.e2e.test.ts test/e2e/auth_host.e2e.test.ts || status=$? exit "$status" diff --git a/.github/workflows/sdk-platform.yml b/.github/workflows/sdk-platform.yml index 85937f783c..f1ed4e6ed4 100644 --- a/.github/workflows/sdk-platform.yml +++ b/.github/workflows/sdk-platform.yml @@ -11,9 +11,6 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string outputs: nodejs-result: description: "Node.js SDK result, independent of other languages" @@ -32,7 +29,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit python: @@ -41,7 +37,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit go: @@ -50,7 +45,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit dotnet: @@ -59,7 +53,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit rust: @@ -68,7 +61,6 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit java: @@ -77,5 +69,4 @@ jobs: with: platform: ${{ inputs.platform }} sdk-home: ${{ inputs.sdk-home }} - runtime-source: ${{ inputs.runtime-source }} secrets: inherit diff --git a/.github/workflows/sdk-python.yml b/.github/workflows/sdk-python.yml index e81bb63975..df9cbab1a9 100644 --- a/.github/workflows/sdk-python.yml +++ b/.github/workflows/sdk-python.yml @@ -19,22 +19,19 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: python: - name: "Python (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }})" + name: "Python (${{ matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} timeout-minutes: 20 defaults: run: diff --git a/.github/workflows/sdk-rust.yml b/.github/workflows/sdk-rust.yml index 12506d5cad..04500c3489 100644 --- a/.github/workflows/sdk-rust.yml +++ b/.github/workflows/sdk-rust.yml @@ -19,23 +19,20 @@ on: sdk-home: required: true type: string - runtime-source: - required: true - type: string permissions: contents: read jobs: rust: - name: "Rust (${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }})" + name: "Rust (${{ matrix.os }})" if: contains(fromJSON('["all","linux-x64","darwin-arm64","win32-x64"]'), inputs.platform) timeout-minutes: 60 strategy: fail-fast: false matrix: os: ${{ fromJSON(inputs.platform == 'linux-x64' && '["ubuntu-latest"]' || inputs.platform == 'darwin-arm64' && '["macos-26-xlarge-agent-runtime"]' || inputs.platform == 'win32-x64' && '["windows-latest"]' || '["ubuntu-latest","macos-26-xlarge-agent-runtime","windows-latest"]') }} - runs-on: ${{ inputs.runtime-source == 'published' && matrix.os == 'macos-26-xlarge-agent-runtime' && 'macos-26' || matrix.os }} + runs-on: ${{ matrix.os }} defaults: run: shell: bash diff --git a/.github/workflows/sdk.yml b/.github/workflows/sdk.yml index 3c70b00d33..602d02b027 100644 --- a/.github/workflows/sdk.yml +++ b/.github/workflows/sdk.yml @@ -446,7 +446,6 @@ jobs: with: platform: linux-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-x64: @@ -457,7 +456,6 @@ jobs: with: platform: linuxmusl-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-linuxmusl-arm64: @@ -468,7 +466,6 @@ jobs: with: platform: linuxmusl-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-darwin-arm64: @@ -479,7 +476,6 @@ jobs: with: platform: darwin-arm64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit sdk-win32-x64: @@ -491,7 +487,6 @@ jobs: with: platform: win32-x64 sdk-home: ${{ needs.detect-layout.outputs.sdk-home }} - runtime-source: ${{ needs.detect-layout.outputs.runtime-source }} secrets: inherit # Only Linux CAPI gates this rollup; the full SDK aggregate still reports all coverage. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ed83ef2a5a..74519107b4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -363,49 +363,12 @@ failure fails the job. Java uses JDK 25 on all four platforms, plus a Linux/glibc JDK 17 compatibility job using precompiled classes. Merge groups retain the reduced Linux TypeScript CAPI subprocess coverage. -Standalone published-runtime macOS jobs use the standard `macos-26` ARM64 -GitHub-hosted runner. Source-runtime jobs retain their configured runtime -runners, including the larger Rust and .NET profiles. Test selection and -timeouts are unchanged; failures on the standard runner still fail coverage. - -Standalone Windows .NET runs `net8.0` and `net472` in separate mandatory jobs, -each with full subprocess coverage, the existing in-process smoke step, and -the existing 30-minute job limit. This increases total allowed runner time, -not one aggregate deadline. Source-runtime Windows jobs retain both targets -in the original single job. - -The Assisted permission contract runs as three complete scenario groups: -lifecycle, authorization, and handler. All assertions and sequential awaited -cleanup remain; each group uses the existing test timeout. This increases -total permitted suite time rather than reducing runtime teardown latency. - The `sdk-typescript` required rollup checks only the Linux CAPI job, including its build, packaging, and applicable static checks. Other platforms, BYOK backends, and languages keep their existing scheduling and failure reporting; they do not gate this rollup. The full `SDK` aggregate still requires all scheduled coverage to succeed. -Standalone CI assigns the two managed-device fixture controls to a separate -mandatory Linux CAPI step. Each fixture-bearing CLI child runs the unchanged, -pinned published runtime in its own disposable container, with the fixture -installed at the [documented production device-policy location](https://docs.github.com/en/copilot/how-tos/administer-copilot/manage-for-enterprise/use-managed-settings/deploy-managed-settings). -The launcher consumes the temporary-file test hint, verifies isolation and -root file ownership, then restores the runner's identity before launching the -runtime. It never writes policy to the host. A result guard requires both -original plugin-lifecycle and sessionless device/model tests to pass; missing -or skipped controls fail the job. Other published-runtime profiles explicitly -delegate only those two cases to this step. Source-runtime profiles retain -their existing full test selection and launch setup. - -The CI-only container setup requires Linux, Docker, and Node.js 22.15 or newer. -Do not provision a machine-wide policy file on a developer workstation to -run these fixtures. Launcher plan, selection, and result-guard unit controls -run without Docker, a CLI, or dependency installation: - -```bash -node --test scripts/ci/device-policy-fixture.test.mjs -``` - The three BYOK backend sweeps run in separate Linux TypeScript jobs, alongside the normal CAPI job; they do not repeat unit tests, packaging, or static checks. After preparing the runtime as described above, run a sweep from the diff --git a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts index a06e01ebe8..5577d03761 100644 --- a/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts +++ b/nodejs/test/e2e/assisted_autopilot_permission.e2e.test.ts @@ -8,7 +8,7 @@ import { mkdir, realpath, rm, writeFile } from "node:fs/promises"; import { createServer } from "node:http"; import { join } from "node:path"; import { text } from "node:stream/consumers"; -import { describe, expect, it, onTestFailed, onTestFinished } from "vitest"; +import { describe, expect, it, onTestFinished } from "vitest"; import { createAttributedPermissionResult, type CopilotSession, @@ -45,314 +45,157 @@ function contractFixtureName(scope: ContractScope, recommendation: ContractRecom describe("Assisted permission handling in Autopilot", async () => { const { copilotClient: client, workDir } = await createSdkTestContext({ useStdio: true }); - it.each(["lifecycle", "authorization", "handler"] as const)( - "resolves Assisted recommendations before Autopilot permission recovery (%s contract)", - async (contract) => { - let agentCalls = 0; - const judgeOutputs: string[] = []; - const authorizationJudgeRequests: string[] = []; - const providerFailures: Error[] = []; - const started = Date.now(); - let phase = "setup"; - let phaseStarted = started; - const completedPhases: Array<{ phase: string; elapsedMs: number }> = []; - const enterPhase = (next: string) => { - const now = Date.now(); - completedPhases.push({ phase, elapsedMs: now - phaseStarted }); - phase = next; - phaseStarted = now; - }; - onTestFailed(() => { - console.error( - "Assisted contract failure boundary:", - JSON.stringify({ - phase, - phaseElapsedMs: Date.now() - phaseStarted, - elapsedMs: Date.now() - started, - completedPhases, - agentCalls, - judgeCalls: judgeOutputs.length, - providerFailures: providerFailures.length, - }) - ); - }); - const outsideDir = `${workDir}-outside`; - await mkdir(outsideDir, { recursive: true }); - await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); - const modelServer = createServer((request, response) => { - void (async () => { - const body = JSON.parse(await text(request)) as { - model: string; - stream?: boolean; - messages: Array<{ role?: string; content?: unknown }>; - }; - const isJudge = body.model === "gpt-6-luna"; - const messages = JSON.stringify(body.messages); - const latestUserMessage = [...body.messages] - .reverse() - .find((entry) => entry.role === "user"); - const latestUserText = JSON.stringify(latestUserMessage?.content); - let message: - | { role: "assistant"; content: string } - | { - role: "assistant"; - content: string; - tool_calls: Array<{ - id: string; - type: "function"; - function: { name: string; arguments: string }; - }>; - }; - if (isJudge) { - const earlierRestrictionIndex = messages.lastIndexOf( - EARLIER_RESTRICTION_MARKER - ); - const laterAuthorizationIndex = messages.lastIndexOf( - LATER_AUTHORIZATION_MARKER - ); - const authorizationRoute = laterAuthorizationIndex !== -1; - if (authorizationRoute) { - authorizationJudgeRequests.push(messages); - } - const contractRequiresApproval = - messages.includes(contractFixtureName("root", "requireApproval")) || - messages.includes(contractFixtureName("subagent", "requireApproval")); - const output = authorizationRoute - ? earlierRestrictionIndex !== -1 && - earlierRestrictionIndex < laterAuthorizationIndex - ? JUDGE_OUTPUT - : HUMAN_REVIEW_OUTPUT - : messages.includes("assisted-human-ask") || contractRequiresApproval - ? HUMAN_REVIEW_OUTPUT - : JUDGE_OUTPUT; - judgeOutputs.push(output); - message = { role: "assistant", content: output }; - } else { - agentCalls++; - const toolResultCount = body.messages.filter( - (entry) => entry.role === "tool" - ).length; - const contractChildRoute = latestUserText.includes( - "ASSISTED_CONTRACT_CHILD_" - ); - const contractSubagentRoute = latestUserText.includes( - "ASSISTED_CONTRACT_SUBAGENT_" - ); - const contractRootRoute = - latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); - const authorizationRestrictionRoute = latestUserText.includes( - EARLIER_RESTRICTION_MARKER - ); - const authorizationAllowRoute = latestUserText.includes( - LATER_AUTHORIZATION_MARKER - ); - if (authorizationRestrictionRoute) { - message = { role: "assistant", content: "restriction-recorded" }; - } else if (authorizationAllowRoute) { - message = - toolResultCount >= 2 - ? { role: "assistant", content: "authorization-updated" } - : toolResultCount === 1 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-task-complete", - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: "authorization-updated", - }), - }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: "assisted-authorization-shell", - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: - process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` - : `mkdir ${AUTHORIZATION_FIXTURE}`, - description: - "Create the fixture authorized by the latest instruction", - }), - }, - }, - ], - }; - } else if ( - contractChildRoute || - contractSubagentRoute || - contractRootRoute - ) { - const recommendation: ContractRecommendation = latestUserText.includes( - "_REQUIRE" - ) - ? "requireApproval" - : latestUserText.includes("_EXCLUDED") - ? "excluded" - : "approve"; - const scope: ContractScope = - contractChildRoute || contractSubagentRoute ? "subagent" : "root"; - const finalText = `contract-${scope}-${ - recommendation === "approve" ? "approved" : "blocked" - }`; - if (contractSubagentRoute) { - message = - toolResultCount === 0 - ? { - role: "assistant", - content: "", - tool_calls: [ - { - id: `assisted-contract-task-${recommendation}`, - type: "function", - function: { - name: "task", - arguments: JSON.stringify({ - name: "assisted-contract", - description: - "Exercise an Assisted permission request", - prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( - recommendation - )}: Run the requested shell command once and report whether it completed.`, - agent_type: "task", - mode: "sync", - }), - }, - }, - ], - } - : { - role: "assistant", - content: "", - tool_calls: [ - { - id: `assisted-contract-task-complete-${recommendation}`, - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ - summary: finalText, - }), - }, - }, - ], - }; - } else if (contractChildRoute && toolResultCount > 0) { - message = { role: "assistant", content: finalText }; - } else if ( - (contractChildRoute || contractRootRoute) && - (toolResultCount === 0 || - (recommendation === "excluded" && toolResultCount === 1)) - ) { - const command = - recommendation === "excluded" - ? process.platform === "win32" - ? "Get-Content approval-probe.txt | & $runner" - : "printf payload | $runner" - : process.platform === "win32" - ? `New-Item -ItemType Directory -Path ${contractFixtureName( - scope, - recommendation - )}` - : `mkdir ${contractFixtureName(scope, recommendation)}`; - message = { - role: "assistant", - content: "", - tool_calls: [ - { - id: - toolResultCount === 0 - ? contractShellCallId(scope, recommendation) - : `${contractShellCallId( - scope, - recommendation - )}-retry`, - type: "function", - function: { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command, - description: - "Exercise Assisted permission routing", - }), - }, - }, - ], - }; - } else { - message = { - role: "assistant", - content: "", - tool_calls: [ - { - id: `assisted-contract-task-complete-${recommendation}`, - type: "function", - function: { - name: "task_complete", - arguments: JSON.stringify({ summary: finalText }), - }, - }, - ], - }; - } - } else { - const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); - const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); - const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); - const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); - const primeRoute = - primeIndex > Math.max(shellIndex, pathIndex, humanIndex); - const humanRoute = - humanIndex > Math.max(primeIndex, shellIndex, pathIndex); - const shellRoute = - shellIndex > Math.max(primeIndex, pathIndex, humanIndex); - const resumedShellRoute = - messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; - const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); - const humanLifecycle = - messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || - messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") - ? "resume" - : "create"; - const finalText = humanRoute - ? humanApproved - ? "human-approved" - : "human-denied" - : shellRoute - ? "shell-approved" - : "approval-probe.txt"; - message = primeRoute - ? { role: "assistant", content: "prime-ready" } - : toolResultCount >= 2 + it("resolves Assisted recommendations before Autopilot permission recovery", async () => { + let agentCalls = 0; + const judgeOutputs: string[] = []; + const authorizationJudgeRequests: string[] = []; + const providerFailures: Error[] = []; + const outsideDir = `${workDir}-outside`; + await mkdir(outsideDir, { recursive: true }); + await writeFile(join(outsideDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + onTestFinished(() => rm(outsideDir, { recursive: true, force: true })); + const modelServer = createServer((request, response) => { + void (async () => { + const body = JSON.parse(await text(request)) as { + model: string; + stream?: boolean; + messages: Array<{ role?: string; content?: unknown }>; + }; + const isJudge = body.model === "gpt-6-luna"; + const messages = JSON.stringify(body.messages); + const latestUserMessage = [...body.messages] + .reverse() + .find((entry) => entry.role === "user"); + const latestUserText = JSON.stringify(latestUserMessage?.content); + let message: + | { role: "assistant"; content: string } + | { + role: "assistant"; + content: string; + tool_calls: Array<{ + id: string; + type: "function"; + function: { name: string; arguments: string }; + }>; + }; + if (isJudge) { + const earlierRestrictionIndex = messages.lastIndexOf( + EARLIER_RESTRICTION_MARKER + ); + const laterAuthorizationIndex = messages.lastIndexOf( + LATER_AUTHORIZATION_MARKER + ); + const authorizationRoute = laterAuthorizationIndex !== -1; + if (authorizationRoute) { + authorizationJudgeRequests.push(messages); + } + const contractRequiresApproval = + messages.includes(contractFixtureName("root", "requireApproval")) || + messages.includes(contractFixtureName("subagent", "requireApproval")); + const output = authorizationRoute + ? earlierRestrictionIndex !== -1 && + earlierRestrictionIndex < laterAuthorizationIndex + ? JUDGE_OUTPUT + : HUMAN_REVIEW_OUTPUT + : messages.includes("assisted-human-ask") || contractRequiresApproval + ? HUMAN_REVIEW_OUTPUT + : JUDGE_OUTPUT; + judgeOutputs.push(output); + message = { role: "assistant", content: output }; + } else { + agentCalls++; + const toolResultCount = body.messages.filter( + (entry) => entry.role === "tool" + ).length; + const contractChildRoute = latestUserText.includes("ASSISTED_CONTRACT_CHILD_"); + const contractSubagentRoute = latestUserText.includes( + "ASSISTED_CONTRACT_SUBAGENT_" + ); + const contractRootRoute = latestUserText.includes("ASSISTED_CONTRACT_ROOT_"); + const authorizationRestrictionRoute = latestUserText.includes( + EARLIER_RESTRICTION_MARKER + ); + const authorizationAllowRoute = latestUserText.includes( + LATER_AUTHORIZATION_MARKER + ); + if (authorizationRestrictionRoute) { + message = { role: "assistant", content: "restriction-recorded" }; + } else if (authorizationAllowRoute) { + message = + toolResultCount >= 2 + ? { role: "assistant", content: "authorization-updated" } + : toolResultCount === 1 ? { role: "assistant", - content: finalText, + content: "", + tool_calls: [ + { + id: "assisted-authorization-task-complete", + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: "authorization-updated", + }), + }, + }, + ], } - : toolResultCount === 1 + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-authorization-shell", + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: + process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${AUTHORIZATION_FIXTURE}` + : `mkdir ${AUTHORIZATION_FIXTURE}`, + description: + "Create the fixture authorized by the latest instruction", + }), + }, + }, + ], + }; + } else if (contractChildRoute || contractSubagentRoute || contractRootRoute) { + const recommendation: ContractRecommendation = latestUserText.includes( + "_REQUIRE" + ) + ? "requireApproval" + : latestUserText.includes("_EXCLUDED") + ? "excluded" + : "approve"; + const scope: ContractScope = + contractChildRoute || contractSubagentRoute ? "subagent" : "root"; + const finalText = `contract-${scope}-${ + recommendation === "approve" ? "approved" : "blocked" + }`; + if (contractSubagentRoute) { + message = + toolResultCount === 0 ? { role: "assistant", content: "", tool_calls: [ { - id: "assisted-autopilot-task-complete", + id: `assisted-contract-task-${recommendation}`, type: "function", function: { - name: "task_complete", + name: "task", arguments: JSON.stringify({ - summary: finalText, + name: "assisted-contract", + description: + "Exercise an Assisted permission request", + prompt: `ASSISTED_CONTRACT_CHILD_${contractMarker( + recommendation + )}: Run the requested shell command once and report whether it completed.`, + agent_type: "task", + mode: "sync", }), }, }, @@ -363,788 +206,833 @@ describe("Assisted permission handling in Autopilot", async () => { content: "", tool_calls: [ { - id: shellRoute - ? "assisted-autopilot-shell" - : humanRoute - ? "assisted-autopilot-human" - : "assisted-autopilot-glob", + id: `assisted-contract-task-complete-${recommendation}`, type: "function", - function: - shellRoute || humanRoute - ? { - name: SHELL_TOOL, - arguments: JSON.stringify({ - command: humanRoute - ? process.platform === - "win32" - ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` - : process.platform === - "win32" - ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` - : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, - description: humanRoute - ? "Create the human-reviewed SDK permission fixture" - : "Create the authorized SDK permission fixture", - }), - } - : { - name: "glob", - arguments: JSON.stringify({ - pattern: "*.txt", - path: outsideDir, - }), - }, + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: finalText, + }), + }, }, ], }; - } - } - - const choice = { - index: 0, - message, - finish_reason: "tool_calls" in message ? "tool_calls" : "stop", - }; - const completion = { - id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, - object: "chat.completion", - created: 1, - model: body.model, - choices: [choice], - usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, - }; - if (body.stream) { - response.writeHead(200, { "content-type": "text/event-stream" }); - response.end( - `data: ${JSON.stringify({ - ...completion, - object: "chat.completion.chunk", - choices: [ + } else if (contractChildRoute && toolResultCount > 0) { + message = { role: "assistant", content: finalText }; + } else if ( + (contractChildRoute || contractRootRoute) && + (toolResultCount === 0 || + (recommendation === "excluded" && toolResultCount === 1)) + ) { + const command = + recommendation === "excluded" + ? process.platform === "win32" + ? "Get-Content approval-probe.txt | & $runner" + : "printf payload | $runner" + : process.platform === "win32" + ? `New-Item -ItemType Directory -Path ${contractFixtureName( + scope, + recommendation + )}` + : `mkdir ${contractFixtureName(scope, recommendation)}`; + message = { + role: "assistant", + content: "", + tool_calls: [ { - index: 0, - delta: { - ...message, - ...("tool_calls" in message - ? { - tool_calls: message.tool_calls.map( - (call, index) => ({ index, ...call }) - ), - } - : {}), + id: + toolResultCount === 0 + ? contractShellCallId(scope, recommendation) + : `${contractShellCallId( + scope, + recommendation + )}-retry`, + type: "function", + function: { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command, + description: "Exercise Assisted permission routing", + }), }, - finish_reason: choice.finish_reason, }, ], - })}\n\ndata: [DONE]\n\n` - ); + }; + } else { + message = { + role: "assistant", + content: "", + tool_calls: [ + { + id: `assisted-contract-task-complete-${recommendation}`, + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ summary: finalText }), + }, + }, + ], + }; + } } else { - response.writeHead(200, { "content-type": "application/json" }); - response.end(JSON.stringify(completion)); + const primeIndex = messages.lastIndexOf("ASSISTED_SESSION_PRIME"); + const shellIndex = messages.lastIndexOf("ASSISTED_SHELL_"); + const pathIndex = messages.lastIndexOf("ASSISTED_PATH_ROUTE"); + const humanIndex = messages.lastIndexOf("ASSISTED_HUMAN_"); + const primeRoute = primeIndex > Math.max(shellIndex, pathIndex, humanIndex); + const humanRoute = humanIndex > Math.max(primeIndex, shellIndex, pathIndex); + const shellRoute = shellIndex > Math.max(primeIndex, pathIndex, humanIndex); + const resumedShellRoute = + messages.lastIndexOf("ASSISTED_SHELL_RESUME_ROUTE") === shellIndex; + const humanApproved = messages.includes("ASSISTED_HUMAN_APPROVE_"); + const humanLifecycle = + messages.includes("ASSISTED_HUMAN_APPROVE_RESUME_ROUTE") || + messages.includes("ASSISTED_HUMAN_DENY_RESUME_ROUTE") + ? "resume" + : "create"; + const finalText = humanRoute + ? humanApproved + ? "human-approved" + : "human-denied" + : shellRoute + ? "shell-approved" + : "approval-probe.txt"; + message = primeRoute + ? { role: "assistant", content: "prime-ready" } + : toolResultCount >= 2 + ? { + role: "assistant", + content: finalText, + } + : toolResultCount === 1 + ? { + role: "assistant", + content: "", + tool_calls: [ + { + id: "assisted-autopilot-task-complete", + type: "function", + function: { + name: "task_complete", + arguments: JSON.stringify({ + summary: finalText, + }), + }, + }, + ], + } + : { + role: "assistant", + content: "", + tool_calls: [ + { + id: shellRoute + ? "assisted-autopilot-shell" + : humanRoute + ? "assisted-autopilot-human" + : "assisted-autopilot-glob", + type: "function", + function: + shellRoute || humanRoute + ? { + name: SHELL_TOOL, + arguments: JSON.stringify({ + command: humanRoute + ? process.platform === "win32" + ? `New-Item -ItemType Directory -Path assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : `mkdir assisted-human-ask-${humanApproved ? "approve" : "deny"}-${humanLifecycle}` + : process.platform === "win32" + ? `New-Item -ItemType Directory -Path assisted-shell-${resumedShellRoute ? "resume" : "create"}` + : `mkdir assisted-shell-${resumedShellRoute ? "resume" : "create"}`, + description: humanRoute + ? "Create the human-reviewed SDK permission fixture" + : "Create the authorized SDK permission fixture", + }), + } + : { + name: "glob", + arguments: JSON.stringify({ + pattern: "*.txt", + path: outsideDir, + }), + }, + }, + ], + }; } - })().catch((error: unknown) => { - providerFailures.push( - error instanceof Error ? error : new Error(String(error)) + } + + const choice = { + index: 0, + message, + finish_reason: "tool_calls" in message ? "tool_calls" : "stop", + }; + const completion = { + id: `assisted-autopilot-${judgeOutputs.length + agentCalls}`, + object: "chat.completion", + created: 1, + model: body.model, + choices: [choice], + usage: { prompt_tokens: 100, completion_tokens: 20, total_tokens: 120 }, + }; + if (body.stream) { + response.writeHead(200, { "content-type": "text/event-stream" }); + response.end( + `data: ${JSON.stringify({ + ...completion, + object: "chat.completion.chunk", + choices: [ + { + index: 0, + delta: { + ...message, + ...("tool_calls" in message + ? { + tool_calls: message.tool_calls.map( + (call, index) => ({ index, ...call }) + ), + } + : {}), + }, + finish_reason: choice.finish_reason, + }, + ], + })}\n\ndata: [DONE]\n\n` ); - response.writeHead(500).end(); - }); - }); - await new Promise((resolve, reject) => { - modelServer.once("error", reject); - modelServer.listen(0, "127.0.0.1", resolve); - }); - onTestFinished(async () => { - modelServer.closeAllConnections(); - if (modelServer.listening) { - await new Promise((resolve, reject) => { - modelServer.close((error) => (error ? reject(error) : resolve())); - }); + } else { + response.writeHead(200, { "content-type": "application/json" }); + response.end(JSON.stringify(completion)); } + })().catch((error: unknown) => { + providerFailures.push(error instanceof Error ? error : new Error(String(error))); + response.writeHead(500).end(); }); - const address = modelServer.address(); - if (!address || typeof address === "string") { - throw new Error("Missing local model server address"); + }); + await new Promise((resolve, reject) => { + modelServer.once("error", reject); + modelServer.listen(0, "127.0.0.1", resolve); + }); + onTestFinished(async () => { + modelServer.closeAllConnections(); + if (modelServer.listening) { + await new Promise((resolve, reject) => { + modelServer.close((error) => (error ? reject(error) : resolve())); + }); } + }); + const address = modelServer.address(); + if (!address || typeof address === "string") { + throw new Error("Missing local model server address"); + } - execFileSync("git", ["init", "--quiet"], { cwd: workDir }); - await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); + execFileSync("git", ["init", "--quiet"], { cwd: workDir }); + await writeFile(join(workDir, "approval-probe.txt"), "ASSISTED_READ_PROBE\n"); - const providers: NamedProviderConfig[] = [ - { - name: "local", - type: "openai", - baseUrl: `http://127.0.0.1:${address.port}/v1`, - apiKey: "synthetic-test-token", - wireApi: "completions", + const providers: NamedProviderConfig[] = [ + { + name: "local", + type: "openai", + baseUrl: `http://127.0.0.1:${address.port}/v1`, + apiKey: "synthetic-test-token", + wireApi: "completions", + }, + ]; + const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ + id, + provider: "local", + modelId: "gpt-4o", + wireModel: id, + })); + const scenarios = [ + { route: "shell", lifecycle: "create", decision: "judge" }, + { route: "path", lifecycle: "create", decision: "judge" }, + { route: "shell", lifecycle: "resume", decision: "judge" }, + { route: "path", lifecycle: "resume", decision: "judge" }, + { route: "human", lifecycle: "create", decision: "approve" }, + { route: "human", lifecycle: "create", decision: "deny" }, + { route: "human", lifecycle: "resume", decision: "approve" }, + { route: "human", lifecycle: "resume", decision: "deny" }, + ] as const; + for (const scenario of scenarios) { + let permissionCallbacks = 0; + const expectedRecommendation = + scenario.decision === "judge" ? "approve" : "requireApproval"; + const recommendationByToolCallId = new Map(); + const recommendationWaiters = new Map< + string, + (recommendation: string | undefined) => void + >(); + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; + const decisionSources: string[] = []; + const taskOutcomes: Array<{ + success?: boolean; + summary?: string; + outcome?: string; + reason?: string; + }> = []; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "glob", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, }, - ]; - const models: ProviderModelConfig[] = ["gpt-5.6-sol", "gpt-6-luna"].map((id) => ({ - id, - provider: "local", - modelId: "gpt-4o", - wireModel: id, - })); - const scenarios = [ - { route: "shell", lifecycle: "create", decision: "judge" }, - { route: "path", lifecycle: "create", decision: "judge" }, - { route: "shell", lifecycle: "resume", decision: "judge" }, - { route: "path", lifecycle: "resume", decision: "judge" }, - { route: "human", lifecycle: "create", decision: "approve" }, - { route: "human", lifecycle: "create", decision: "deny" }, - { route: "human", lifecycle: "resume", decision: "approve" }, - { route: "human", lifecycle: "resume", decision: "deny" }, - ] as const; - for (const scenario of contract === "lifecycle" ? scenarios : []) { - const label = `initial:${scenario.route}:${scenario.lifecycle}:${scenario.decision}`; - let permissionCallbacks = 0; - const expectedRecommendation = - scenario.decision === "judge" ? "approve" : "requireApproval"; - const recommendationByToolCallId = new Map(); - const recommendationWaiters = new Map< - string, - (recommendation: string | undefined) => void - >(); - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const toolResults: Array<{ success?: boolean; result?: { content?: string } }> = []; - const decisionSources: string[] = []; - const taskOutcomes: Array<{ - success?: boolean; - summary?: string; - outcome?: string; - reason?: string; - }> = []; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "glob", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - onPermissionRequest: async (request: PermissionRequest) => { - permissionCallbacks++; - if (!request.toolCallId) { - throw new Error( - "Expected the permission request to identify its tool call" - ); - } - const recommendation = recommendationByToolCallId.has(request.toolCallId) - ? recommendationByToolCallId.get(request.toolCallId) - : await new Promise((resolve) => { - recommendationWaiters.set(request.toolCallId!, resolve); - }); - recommendationByToolCallId.delete(request.toolCallId); - if (recommendation !== expectedRecommendation) { - throw new Error( - `Expected Assisted recommendation ${expectedRecommendation}, got ${String( - recommendation - )}` - ); - } - if (scenario.decision === "judge") { - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - } + onPermissionRequest: async (request: PermissionRequest) => { + permissionCallbacks++; + if (!request.toolCallId) { + throw new Error( + "Expected the permission request to identify its tool call" + ); + } + const recommendation = recommendationByToolCallId.has(request.toolCallId) + ? recommendationByToolCallId.get(request.toolCallId) + : await new Promise((resolve) => { + recommendationWaiters.set(request.toolCallId!, resolve); + }); + recommendationByToolCallId.delete(request.toolCallId); + if (recommendation !== expectedRecommendation) { + throw new Error( + `Expected Assisted recommendation ${expectedRecommendation}, got ${String( + recommendation + )}` + ); + } + if (scenario.decision === "judge") { return createAttributedPermissionResult( - { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, + { kind: "approve-once" }, { - outcome: "prompted_user", - source: "human_response", + outcome: "auto_approved", + source: "assisted_approval", surface: "sdk", - responseCapability: "interactive", + responseCapability: "headless", } ); - }, - } as const; - let session: CopilotSession | undefined; - try { - enterPhase(`${label}:create`); - session = await client.createSession(sessionConfig); - if (scenario.lifecycle === "resume") { - const sessionId = session.sessionId; - enterPhase(`${label}:prime`); - await session.sendAndWait({ - prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", - }); - enterPhase(`${label}:configure-before-resume`); - await configureAssistedAutopilot(session, workDir); - await expectAssistedAutopilotConfigured(session, workDir); - enterPhase(`${label}:disconnect-before-resume`); - await session.disconnect(); - session = undefined; - enterPhase(`${label}:resume`); - session = await client.resumeSession(sessionId, sessionConfig); - enterPhase(`${label}:verify-resume`); - await expectAssistedAutopilotConfigured(session, workDir); - } else { - enterPhase(`${label}:configure`); - await configureAssistedAutopilot(session, workDir); } - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; - } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - const toolCallId = ( - event.data as { - permissionRequest?: { toolCallId?: string }; - } - ).permissionRequest?.toolCallId; - if (toolCallId) { - const waiter = recommendationWaiters.get(toolCallId); - if (waiter) { - recommendationWaiters.delete(toolCallId); - waiter(recommendation); - } else { - recommendationByToolCallId.set(toolCallId, recommendation); - } - } - if (recommendation) recommendations.push(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }) - .decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - recoveryStatuses.push((event.data as { status: string }).status); - } else if (event.type === "tool.execution_complete") { - toolResults.push(event.data); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); + return createAttributedPermissionResult( + { kind: scenario.decision === "approve" ? "approve-once" : "reject" }, + { + outcome: "prompted_user", + source: "human_response", + surface: "sdk", + responseCapability: "interactive", } - }); - - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - enterPhase(`${label}:send`); - await session.send({ - prompt: - scenario.route === "shell" - ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` - : scenario.route === "path" - ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` - : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, - }); - enterPhase(`${label}:task-complete`); - await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } - - enterPhase(`${label}:assert`); - const expectedCompletions = - scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill(expectedRecommendation) ); - expect(decisionSources, JSON.stringify(scenario)).toEqual( - Array(expectedCompletions).fill( - scenario.decision === "judge" ? "assisted_approval" : "human_response" - ) - ); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); - if (scenario.decision === "deny") { - expect( - toolResults.some((result) => result.success === false), - JSON.stringify(scenario) - ).toBe(true); - expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); - } else { - expect( - toolResults.every((result) => result.success === true), - JSON.stringify(scenario) - ).toBe(true); - } - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - success: true, + }, + } as const; + let session: CopilotSession | undefined; + try { + session = await client.createSession(sessionConfig); + if (scenario.lifecycle === "resume") { + const sessionId = session.sessionId; + await session.sendAndWait({ + prompt: "ASSISTED_SESSION_PRIME: Reply with prime-ready without tools.", }); - expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( - scenario.route === "shell" - ? "shell-approved" - : scenario.route === "path" - ? "approval-probe.txt" - : `human-${scenario.decision === "approve" ? "approved" : "denied"}` - ); - if (scenario.route === "human") { - expect( - existsSync( - join( - workDir, - `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` - ) - ), - JSON.stringify(scenario) - ).toBe(scenario.decision === "approve"); - } - } finally { - if (session) { - enterPhase(`${label}:cleanup`); - await session.abort(); - await session.disconnect(); - } + await configureAssistedAutopilot(session, workDir); + await expectAssistedAutopilotConfigured(session, workDir); + await session.disconnect(); + session = undefined; + session = await client.resumeSession(sessionId, sessionConfig); + await expectAssistedAutopilotConfigured(session, workDir); + } else { + await configureAssistedAutopilot(session, workDir); } - } - - if (contract === "lifecycle") { - expect(providerFailures).toEqual([]); - expect(judgeOutputs).toEqual([ - ...Array(4).fill(JUDGE_OUTPUT), - ...Array(4).fill(HUMAN_REVIEW_OUTPUT), - ]); - expect(agentCalls).toBe(scenarios.length * 2 + 4); - } - - if (contract === "authorization") { - const authorizationJudgeStart = judgeOutputs.length; - let authorizationPermissionCallbacks = 0; - const authorizationRecommendations: string[] = []; - const authorizationDecisionSources: string[] = []; - const authorizationRecoveryStatuses: string[] = []; - const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = - []; - const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = - []; - let resolveAuthorizationRecommendation: ( - recommendation: string | undefined - ) => void; - const authorizationRecommendation = new Promise((resolve) => { - resolveAuthorizationRecommendation = resolve; - }); - let authorizationSession: CopilotSession | undefined; - try { - enterPhase("authorization:create"); - authorizationSession = await client.createSession({ - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - onPermissionRequest: async (request) => { - authorizationPermissionCallbacks++; - if (request.toolCallId !== "assisted-authorization-shell") { - throw new Error( - `Expected authorization shell permission, got ${String( - request.toolCallId - )}` - ); + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; } - const recommendation = await authorizationRecommendation; - if (recommendation !== "approve") { - throw new Error( - `Expected later authorization to reach the judge as approve, got ${String( - recommendation - )}` - ); + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + const toolCallId = ( + event.data as { + permissionRequest?: { toolCallId?: string }; } - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - }, - }); - enterPhase("authorization:prime"); - await authorizationSession.sendAndWait({ - prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, - }); - enterPhase("authorization:configure"); - await configureAssistedAutopilot(authorizationSession, workDir); - authorizationSession.on((event) => { - if (event.type === "permission.requested") { - const recommendation = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - }; - } - ).promptRequest?.assistedApproval?.recommendation; - if (recommendation) { - authorizationRecommendations.push(recommendation); + ).permissionRequest?.toolCallId; + if (toolCallId) { + const waiter = recommendationWaiters.get(toolCallId); + if (waiter) { + recommendationWaiters.delete(toolCallId); + waiter(recommendation); + } else { + recommendationByToolCallId.set(toolCallId, recommendation); } - resolveAuthorizationRecommendation(recommendation); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }) - .decisionSource; - if (source) authorizationDecisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - authorizationRecoveryStatuses.push( - (event.data as { status: string }).status - ); - } else if (event.type === "tool.execution_complete") { - authorizationToolResults.push({ - toolCallId: event.data.toolCallId, - success: event.data.success, - }); - } else if (event.type === "session.task_complete") { - authorizationTaskOutcomes.push(event.data); } - }); + if (recommendation) recommendations.push(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }).decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + recoveryStatuses.push((event.data as { status: string }).status); + } else if (event.type === "tool.execution_complete") { + toolResults.push(event.data); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); + } + }); - const taskComplete = getNextEventOfType( - authorizationSession, - "session.task_complete" - ); - enterPhase("authorization:send"); - await authorizationSession.send({ - prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + await session.send({ + prompt: + scenario.route === "shell" + ? `ASSISTED_SHELL_${scenario.lifecycle.toUpperCase()}_ROUTE: Create the authorized fixture directory once and report shell-approved.` + : scenario.route === "path" + ? `ASSISTED_PATH_ROUTE: Use only glob to find *.txt in ${outsideDir} and report the matching filename.` + : `ASSISTED_HUMAN_${scenario.decision.toUpperCase()}_${scenario.lifecycle.toUpperCase()}_ROUTE: Try to create the human-reviewed fixture directory once and report human-${scenario.decision === "approve" ? "approved" : "denied"}.`, }); - enterPhase("authorization:task-complete"); await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } - enterPhase("authorization:assert"); - expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); - expect(authorizationJudgeRequests).toHaveLength(1); - const authorizationRequest = authorizationJudgeRequests[0] ?? ""; + const expectedCompletions = + scenario.route === "path" && scenario.lifecycle === "create" ? 2 : 1; + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(expectedCompletions); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill(expectedRecommendation) + ); + expect(decisionSources, JSON.stringify(scenario)).toEqual( + Array(expectedCompletions).fill( + scenario.decision === "judge" ? "assisted_approval" : "human_response" + ) + ); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + expect(toolResults, JSON.stringify(scenario)).toHaveLength(2); + if (scenario.decision === "deny") { expect( - authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) - ).toBeGreaterThanOrEqual(0); + toolResults.some((result) => result.success === false), + JSON.stringify(scenario) + ).toBe(true); + expect(toolResults.at(-1)?.success, JSON.stringify(scenario)).toBe(true); + } else { expect( - authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER) - ).toBeGreaterThan(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)); - expect(authorizationPermissionCallbacks).toBe(1); - expect(authorizationRecommendations).toEqual(["approve"]); - expect(authorizationDecisionSources).toEqual(["assisted_approval"]); - expect(authorizationRecoveryStatuses).toEqual([]); - expect(authorizationToolResults).toContainEqual({ - toolCallId: "assisted-authorization-shell", - success: true, - }); - expect(authorizationTaskOutcomes).toEqual([ - expect.objectContaining({ - success: true, - summary: "authorization-updated", - }), - ]); - expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); - } finally { - if (authorizationSession) { - enterPhase("authorization:cleanup"); - await authorizationSession.abort(); - await authorizationSession.disconnect(); - } + toolResults.every((result) => result.success === true), + JSON.stringify(scenario) + ).toBe(true); + } + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + success: true, + }); + expect(taskOutcomes[0]?.summary, JSON.stringify(scenario)).toContain( + scenario.route === "shell" + ? "shell-approved" + : scenario.route === "path" + ? "approval-probe.txt" + : `human-${scenario.decision === "approve" ? "approved" : "denied"}` + ); + if (scenario.route === "human") { + expect( + existsSync( + join( + workDir, + `assisted-human-ask-${scenario.decision}-${scenario.lifecycle}` + ) + ), + JSON.stringify(scenario) + ).toBe(scenario.decision === "approve"); + } + } finally { + if (session) { + await session.abort(); + await session.disconnect(); } } + } - const contractScenarios = [ - { scope: "root", recommendation: "approve", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "excluded", handler: "registered" }, - { scope: "subagent", recommendation: "approve", handler: "registered" }, - { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, - { scope: "root", recommendation: "requireApproval", handler: "none" }, - ] as const satisfies readonly { - scope: ContractScope; - recommendation: ContractRecommendation; - handler: "registered" | "none"; - }[]; - const contractJudgeStart = judgeOutputs.length; - for (const scenario of contract === "handler" ? contractScenarios : []) { - const label = `contract:${scenario.scope}:${scenario.recommendation}:${scenario.handler}`; - let permissionCallbacks = 0; - const recommendations: string[] = []; - const recoveryStatuses: string[] = []; - const decisionSources: string[] = []; - const sequence: string[] = []; - const subagentIds = new Set(); - const permissionAgentIds: Array = []; - const toolResults: Array<{ - agentId?: string; - toolCallId?: string; - success?: boolean; - error?: { message?: string }; - result?: unknown; - }> = []; - const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; - const onPermissionRequest = () => { - permissionCallbacks++; - sequence.push("host"); - if (scenario.recommendation === "approve") { - return createAttributedPermissionResult( - { kind: "approve-once" }, - { - outcome: "auto_approved", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } - ); - } - if (scenario.recommendation === "requireApproval") { - return createAttributedPermissionResult( - { - kind: "reject", - feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), - }, - { - outcome: "autopilot_denied", - source: "assisted_approval", - surface: "sdk", - responseCapability: "headless", - } + expect(providerFailures).toEqual([]); + expect(judgeOutputs).toEqual([ + ...Array(4).fill(JUDGE_OUTPUT), + ...Array(4).fill(HUMAN_REVIEW_OUTPUT), + ]); + expect(agentCalls).toBe(scenarios.length * 2 + 4); + + const authorizationJudgeStart = judgeOutputs.length; + let authorizationPermissionCallbacks = 0; + const authorizationRecommendations: string[] = []; + const authorizationDecisionSources: string[] = []; + const authorizationRecoveryStatuses: string[] = []; + const authorizationToolResults: Array<{ toolCallId?: string; success?: boolean }> = []; + const authorizationTaskOutcomes: Array<{ success?: boolean; summary?: string }> = []; + let resolveAuthorizationRecommendation: (recommendation: string | undefined) => void; + const authorizationRecommendation = new Promise((resolve) => { + resolveAuthorizationRecommendation = resolve; + }); + let authorizationSession: CopilotSession | undefined; + try { + authorizationSession = await client.createSession({ + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + onPermissionRequest: async (request) => { + authorizationPermissionCallbacks++; + if (request.toolCallId !== "assisted-authorization-shell") { + throw new Error( + `Expected authorization shell permission, got ${String( + request.toolCallId + )}` ); } - if (scenario.recommendation === "excluded") { - return createAttributedPermissionResult( - { kind: "user-not-available" }, - { - outcome: "autopilot_denied", - source: "unattended_fallback", - surface: "sdk", - responseCapability: "headless", - } + const recommendation = await authorizationRecommendation; + if (recommendation !== "approve") { + throw new Error( + `Expected later authorization to reach the judge as approve, got ${String( + recommendation + )}` ); } - throw new Error("Unexpected Assisted permission recommendation"); - }; - const sessionConfig = { - model: "local/gpt-5.6-sol", - providers, - models, - availableTools: [SHELL_TOOL, "task", "task_complete"], - streaming: false, - skipCustomInstructions: true, - featureFlags: { - AUTO_APPROVAL: true, - ASSISTED_PERMISSIONS_V2: true, - }, - ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), - } as const; - let session: CopilotSession | undefined; - try { - enterPhase(`${label}:create`); - session = await client.createSession(sessionConfig); - enterPhase(`${label}:configure`); - await configureAssistedAutopilot(session, workDir); - session.on((event) => { - if (event.type === "permission.requested") { - const promptRequest = ( - event.data as { - promptRequest?: { - assistedApproval?: { recommendation?: string }; - autoApproval?: { recommendation?: string }; - }; - } - ).promptRequest; - const recommendation = - promptRequest?.assistedApproval?.recommendation ?? - promptRequest?.autoApproval?.recommendation; - if (recommendation) { - recommendations.push(recommendation); - sequence.push(`permission:${recommendation}`); - } - permissionAgentIds.push(event.agentId); - } else if (event.type === "permission.completed") { - const source = (event.data as { decisionSource?: string }) - .decisionSource; - if (source) decisionSources.push(source); - } else if (event.type === "session.permission_recovery") { - const status = (event.data as { status: string }).status; - recoveryStatuses.push(status); - sequence.push(`recovery:${status}`); - } else if (event.type === "tool.execution_complete") { - toolResults.push({ - agentId: event.agentId, - toolCallId: event.data.toolCallId, - success: event.data.success, - error: event.data.error, - result: event.data.result, - }); - } else if (event.type === "session.task_complete") { - taskOutcomes.push(event.data); - } else if (event.type === "subagent.started") { - subagentIds.add(event.agentId); + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } + ); + }, + }); + await authorizationSession.sendAndWait({ + prompt: `${EARLIER_RESTRICTION_MARKER}: Do not create the authorization fixture. Reply restriction-recorded without tools.`, + }); + await configureAssistedAutopilot(authorizationSession, workDir); + authorizationSession.on((event) => { + if (event.type === "permission.requested") { + const recommendation = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + }; } + ).promptRequest?.assistedApproval?.recommendation; + if (recommendation) { + authorizationRecommendations.push(recommendation); + } + resolveAuthorizationRecommendation(recommendation); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }).decisionSource; + if (source) authorizationDecisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + authorizationRecoveryStatuses.push((event.data as { status: string }).status); + } else if (event.type === "tool.execution_complete") { + authorizationToolResults.push({ + toolCallId: event.data.toolCallId, + success: event.data.success, }); + } else if (event.type === "session.task_complete") { + authorizationTaskOutcomes.push(event.data); + } + }); - const marker = contractMarker(scenario.recommendation); - const prompt = - scenario.scope === "root" - ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` - : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; - try { - const taskComplete = getNextEventOfType(session, "session.task_complete"); - enterPhase(`${label}:send`); - await session.send({ prompt }); - enterPhase(`${label}:task-complete`); - await taskComplete; - } catch (error) { - throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); - } + const taskComplete = getNextEventOfType(authorizationSession, "session.task_complete"); + await authorizationSession.send({ + prompt: `${LATER_AUTHORIZATION_MARKER}: The earlier restriction is superseded. Create the authorization fixture now and report authorization-updated.`, + }); + await taskComplete; - enterPhase(`${label}:assert`); - if (scenario.handler === "none") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); - expect(recommendations, JSON.stringify(scenario)).toEqual([]); - } else if (scenario.recommendation === "approve") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); - } else if (scenario.recommendation === "requireApproval") { - expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); - expect(recommendations, JSON.stringify(scenario)).toEqual([ - "requireApproval", - ]); - } else { - expect( - permissionCallbacks, - JSON.stringify(scenario) - ).toBeGreaterThanOrEqual(1); - expect(recommendations, JSON.stringify(scenario)).toEqual( - Array(permissionCallbacks).fill("excluded") - ); - } - if (scenario.handler === "none") { - expect(decisionSources, JSON.stringify(scenario)).toEqual([]); - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); - } else { - expect(decisionSources, JSON.stringify(scenario)).toContain( - scenario.recommendation === "excluded" - ? "unattended_fallback" - : "assisted_approval" - ); - } - if (scenario.recommendation === "excluded") { - expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); - } else if (scenario.handler === "registered") { - expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); - } - const hostIndex = sequence.indexOf("host"); - const firstRecoveryIndex = sequence.findIndex((entry) => - entry.startsWith("recovery:") - ); - if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { - expect( - hostIndex, - JSON.stringify({ scenario, sequence }) - ).toBeGreaterThanOrEqual(0); - expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( - firstRecoveryIndex - ); - } + expect(judgeOutputs.slice(authorizationJudgeStart)).toEqual([JUDGE_OUTPUT]); + expect(authorizationJudgeRequests).toHaveLength(1); + const authorizationRequest = authorizationJudgeRequests[0] ?? ""; + expect(authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER)).toBeGreaterThanOrEqual( + 0 + ); + expect(authorizationRequest.indexOf(LATER_AUTHORIZATION_MARKER)).toBeGreaterThan( + authorizationRequest.indexOf(EARLIER_RESTRICTION_MARKER) + ); + expect(authorizationPermissionCallbacks).toBe(1); + expect(authorizationRecommendations).toEqual(["approve"]); + expect(authorizationDecisionSources).toEqual(["assisted_approval"]); + expect(authorizationRecoveryStatuses).toEqual([]); + expect(authorizationToolResults).toContainEqual({ + toolCallId: "assisted-authorization-shell", + success: true, + }); + expect(authorizationTaskOutcomes).toEqual([ + expect.objectContaining({ + success: true, + summary: "authorization-updated", + }), + ]); + expect(existsSync(join(workDir, AUTHORIZATION_FIXTURE))).toBe(true); + } finally { + if (authorizationSession) { + await authorizationSession.abort(); + await authorizationSession.disconnect(); + } + } - const shellResult = toolResults.find( - (result) => - result.toolCallId === - contractShellCallId(scenario.scope, scenario.recommendation) + const contractScenarios = [ + { scope: "root", recommendation: "approve", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "excluded", handler: "registered" }, + { scope: "subagent", recommendation: "approve", handler: "registered" }, + { scope: "subagent", recommendation: "requireApproval", handler: "registered" }, + { scope: "root", recommendation: "requireApproval", handler: "none" }, + ] as const satisfies readonly { + scope: ContractScope; + recommendation: ContractRecommendation; + handler: "registered" | "none"; + }[]; + const contractJudgeStart = judgeOutputs.length; + for (const scenario of contractScenarios) { + let permissionCallbacks = 0; + const recommendations: string[] = []; + const recoveryStatuses: string[] = []; + const decisionSources: string[] = []; + const sequence: string[] = []; + const subagentIds = new Set(); + const permissionAgentIds: Array = []; + const toolResults: Array<{ + agentId?: string; + toolCallId?: string; + success?: boolean; + error?: { message?: string }; + result?: unknown; + }> = []; + const taskOutcomes: Array<{ success?: boolean; summary?: string }> = []; + const onPermissionRequest = () => { + permissionCallbacks++; + sequence.push("host"); + if (scenario.recommendation === "approve") { + return createAttributedPermissionResult( + { kind: "approve-once" }, + { + outcome: "auto_approved", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); - expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); - expect(shellResult?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation === "approve" && scenario.handler === "registered" + } + if (scenario.recommendation === "requireApproval") { + return createAttributedPermissionResult( + { + kind: "reject", + feedback: HUMAN_REVIEW_OUTPUT.replace(/^DENY:\s*/, ""), + }, + { + outcome: "autopilot_denied", + source: "assisted_approval", + surface: "sdk", + responseCapability: "headless", + } ); - if ( - scenario.recommendation === "requireApproval" && - scenario.handler === "registered" - ) { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Ask the registered human permission handler." - ); - } else if (scenario.handler === "none") { - expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( - "Permission could not be granted automatically." - ); - } - if (scenario.scope === "subagent") { - expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); - expect( - subagentIds.has(shellResult?.agentId ?? ""), - JSON.stringify(scenario) - ).toBe(true); - for (const agentId of permissionAgentIds) { - expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe( - true - ); + } + if (scenario.recommendation === "excluded") { + return createAttributedPermissionResult( + { kind: "user-not-available" }, + { + outcome: "autopilot_denied", + source: "unattended_fallback", + surface: "sdk", + responseCapability: "headless", } + ); + } + throw new Error("Unexpected Assisted permission recommendation"); + }; + const sessionConfig = { + model: "local/gpt-5.6-sol", + providers, + models, + availableTools: [SHELL_TOOL, "task", "task_complete"], + streaming: false, + skipCustomInstructions: true, + featureFlags: { + AUTO_APPROVAL: true, + ASSISTED_PERMISSIONS_V2: true, + }, + ...(scenario.handler === "registered" ? { onPermissionRequest } : {}), + } as const; + let session: CopilotSession | undefined; + try { + session = await client.createSession(sessionConfig); + await configureAssistedAutopilot(session, workDir); + session.on((event) => { + if (event.type === "permission.requested") { + const promptRequest = ( + event.data as { + promptRequest?: { + assistedApproval?: { recommendation?: string }; + autoApproval?: { recommendation?: string }; + }; + } + ).promptRequest; + const recommendation = + promptRequest?.assistedApproval?.recommendation ?? + promptRequest?.autoApproval?.recommendation; + if (recommendation) { + recommendations.push(recommendation); + sequence.push(`permission:${recommendation}`); + } + permissionAgentIds.push(event.agentId); + } else if (event.type === "permission.completed") { + const source = (event.data as { decisionSource?: string }).decisionSource; + if (source) decisionSources.push(source); + } else if (event.type === "session.permission_recovery") { + const status = (event.data as { status: string }).status; + recoveryStatuses.push(status); + sequence.push(`recovery:${status}`); + } else if (event.type === "tool.execution_complete") { + toolResults.push({ + agentId: event.agentId, + toolCallId: event.data.toolCallId, + success: event.data.success, + error: event.data.error, + result: event.data.result, + }); + } else if (event.type === "session.task_complete") { + taskOutcomes.push(event.data); + } else if (event.type === "subagent.started") { + subagentIds.add(event.agentId); } + }); + + const marker = contractMarker(scenario.recommendation); + const prompt = + scenario.scope === "root" + ? `ASSISTED_CONTRACT_ROOT_${marker}: Run the requested shell command once and report the outcome.` + : `ASSISTED_CONTRACT_SUBAGENT_${marker}: Use the task tool once so a task subagent runs the requested shell command.`; + try { + const taskComplete = getNextEventOfType(session, "session.task_complete"); + await session.send({ prompt }); + await taskComplete; + } catch (error) { + throw new Error(`${JSON.stringify(scenario)} failed`, { cause: error }); + } + if (scenario.handler === "none") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(0); + expect(recommendations, JSON.stringify(scenario)).toEqual([]); + } else if (scenario.recommendation === "approve") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual(["approve"]); + } else if (scenario.recommendation === "requireApproval") { + expect(permissionCallbacks, JSON.stringify(scenario)).toBe(1); + expect(recommendations, JSON.stringify(scenario)).toEqual(["requireApproval"]); + } else { + expect(permissionCallbacks, JSON.stringify(scenario)).toBeGreaterThanOrEqual(1); + expect(recommendations, JSON.stringify(scenario)).toEqual( + Array(permissionCallbacks).fill("excluded") + ); + } + if (scenario.handler === "none") { + expect(decisionSources, JSON.stringify(scenario)).toEqual([]); + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual(["recovering"]); + } else { + expect(decisionSources, JSON.stringify(scenario)).toContain( + scenario.recommendation === "excluded" + ? "unattended_fallback" + : "assisted_approval" + ); + } + if (scenario.recommendation === "excluded") { + expect(recoveryStatuses.at(-1), JSON.stringify(scenario)).toBe("blocked"); + } else if (scenario.handler === "registered") { + expect(recoveryStatuses, JSON.stringify(scenario)).toEqual([]); + } + const hostIndex = sequence.indexOf("host"); + const firstRecoveryIndex = sequence.findIndex((entry) => + entry.startsWith("recovery:") + ); + if (firstRecoveryIndex !== -1 && scenario.handler === "registered") { expect( - existsSync( - join( - workDir, - contractFixtureName(scenario.scope, scenario.recommendation) - ) - ), - JSON.stringify(scenario) - ).toBe( - scenario.recommendation === "approve" && scenario.handler === "registered" + hostIndex, + JSON.stringify({ scenario, sequence }) + ).toBeGreaterThanOrEqual(0); + expect(hostIndex, JSON.stringify({ scenario, sequence })).toBeLessThan( + firstRecoveryIndex + ); + } + + const shellResult = toolResults.find( + (result) => + result.toolCallId === + contractShellCallId(scenario.scope, scenario.recommendation) + ); + expect(shellResult, JSON.stringify({ scenario, toolResults })).toBeDefined(); + expect(shellResult?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation === "approve" && scenario.handler === "registered" + ); + if ( + scenario.recommendation === "requireApproval" && + scenario.handler === "registered" + ) { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Ask the registered human permission handler." + ); + } else if (scenario.handler === "none") { + expect(shellResult?.error?.message, JSON.stringify(scenario)).toContain( + "Permission could not be granted automatically." ); + } + if (scenario.scope === "subagent") { + expect(subagentIds.size, JSON.stringify(scenario)).toBe(1); expect( - (await session.rpc.permissions.pendingRequests()).items, + subagentIds.has(shellResult?.agentId ?? ""), JSON.stringify(scenario) - ).toEqual([]); - expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); - expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( - scenario.recommendation !== "excluded" && scenario.handler === "registered" - ); - if (scenario.handler === "none") { - expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ - outcome: "continue", - reason: "Autopilot is still recovering from a required permission.", - }); - } - } finally { - if (session) { - enterPhase(`${label}:cleanup`); - await session.abort(); - await session.disconnect(); + ).toBe(true); + for (const agentId of permissionAgentIds) { + expect(subagentIds.has(agentId ?? ""), JSON.stringify(scenario)).toBe(true); } } - } - enterPhase("final-assertions"); - expect(providerFailures).toEqual([]); - if (contract === "handler") { - expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - JUDGE_OUTPUT, - HUMAN_REVIEW_OUTPUT, - HUMAN_REVIEW_OUTPUT, - ]); + expect( + existsSync( + join(workDir, contractFixtureName(scenario.scope, scenario.recommendation)) + ), + JSON.stringify(scenario) + ).toBe(scenario.recommendation === "approve" && scenario.handler === "registered"); + expect( + (await session.rpc.permissions.pendingRequests()).items, + JSON.stringify(scenario) + ).toEqual([]); + expect(taskOutcomes, JSON.stringify(scenario)).toHaveLength(1); + expect(taskOutcomes[0]?.success, JSON.stringify(scenario)).toBe( + scenario.recommendation !== "excluded" && scenario.handler === "registered" + ); + if (scenario.handler === "none") { + expect(taskOutcomes[0], JSON.stringify(scenario)).toMatchObject({ + outcome: "continue", + reason: "Autopilot is still recovering from a required permission.", + }); + } + } finally { + if (session) { + await session.abort(); + await session.disconnect(); + } } - console.info( - "Assisted contract completed:", - JSON.stringify({ - contract, - elapsedMs: Date.now() - started, - cleanupMs: completedPhases - .filter(({ phase }) => /cleanup|disconnect-before-resume/.test(phase)) - .reduce((total, { elapsedMs }) => total + elapsedMs, 0), - agentCalls, - judgeCalls: judgeOutputs.length, - providerFailures: providerFailures.length, - }) - ); } - ); + + expect(providerFailures).toEqual([]); + expect(judgeOutputs.slice(contractJudgeStart)).toEqual([ + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + JUDGE_OUTPUT, + HUMAN_REVIEW_OUTPUT, + HUMAN_REVIEW_OUTPUT, + ]); + }); }); async function configureAssistedAutopilot(session: CopilotSession, workDir: string): Promise { diff --git a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts index b1336ddf27..e6e6a18c8f 100644 --- a/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts +++ b/nodejs/test/e2e/rpc_tasks_and_handlers.e2e.test.ts @@ -2,7 +2,7 @@ * Copyright (c) Microsoft Corporation. All rights reserved. *--------------------------------------------------------------------------------------------*/ -import { describe, expect, it, onTestFailed } from "vitest"; +import { describe, expect, it } from "vitest"; import { z } from "zod"; import { approveAll, CopilotRequestHandler } from "../../src/index.js"; import type { SessionEvent, CopilotRequestContext, CopilotSession } from "../../src/index.js"; @@ -390,69 +390,25 @@ describe("Session tasks RPC and pending handlers", async () => { { timeout: 120_000 }, async () => { await withRunningAttachedShell(async (session, shellId, eventTypes, replies) => { - const started = performance.now(); - let phase = "enqueue"; - let queuePolls = 0; - let pendingCount = 0; - let steeringCount = 0; - let inFlightSteeringCount: number | undefined; - let queuedMatch = false; - onTestFailed(() => { - console.error( - "Attached shell queue failure boundary:", - JSON.stringify({ - phase, - elapsedMs: Math.round(performance.now() - started), - queuePolls, - pendingCount, - steeringCount, - inFlightSteeringCount, - queuedMatch, - replyCount: replies.length, - hasQueuedReply: replies.some((reply) => reply.includes("QUEUED_DONE")), - sessionIdleCount: eventTypes.filter((type) => type === "session.idle") - .length, - assistantIdleCount: eventTypes.filter( - (type) => type === "assistant.idle" - ).length, - backgroundTaskChangeCount: eventTypes.filter( - (type) => type === "session.background_tasks_changed" - ).length, - }) - ); - }); // The running attached shell holds idle, so an enqueued message waits behind it. await session.send({ prompt: "Reply with exactly QUEUED_DONE.", mode: "enqueue" }); - phase = "pending-queue"; await waitForCondition( - async () => { - const pending = await session.rpc.queue.pendingItems(); - const { items } = pending; - queuePolls++; - pendingCount = items.length; - steeringCount = pending.steeringMessages.length; - inFlightSteeringCount = pending.inFlightSteeringCount; - queuedMatch = items.some((item) => + async () => + (await session.rpc.queue.pendingItems()).items.some((item) => item.displayText.includes("QUEUED_DONE") - ); - return queuedMatch; - }, + ), { timeoutMessage: "The enqueued message was not parked behind the shell" } ); - phase = "assert-parked"; expect(replies.some((reply) => reply.includes("QUEUED_DONE"))).toBe(false); expect(eventTypes).not.toContain("session.idle"); - phase = "cancel-shell"; expect((await session.rpc.tasks.cancel({ id: shellId })).cancelled).toBe(true); - phase = "queued-reply"; await waitForCondition( () => replies.some((reply) => reply.includes("QUEUED_DONE")), { timeoutMessage: `The queued message never ran after cancelling shell ${shellId}`, } ); - phase = "session-idle"; await waitForCondition(() => eventTypes.includes("session.idle"), { timeoutMessage: "session.idle never followed the queued message", }); diff --git a/nodejs/test/rust-codegen.test.ts b/nodejs/test/rust-codegen.test.ts index 71c31fed5f..53c471b104 100644 --- a/nodejs/test/rust-codegen.test.ts +++ b/nodejs/test/rust-codegen.test.ts @@ -1,6 +1,5 @@ import type { ApiSchema } from "../../scripts/codegen/utils.ts"; import { - getApiSchemaPath, normalizeSchemaBrandCasing, postProcessSchema, propagateInternalVisibility, @@ -1016,13 +1015,18 @@ describe("Rust x-legacy-parameters", () => { ).toThrow(/Rust string enum Kind is requested for different values/); }); - it("keeps every const discriminator of the selected API schema distinct in Rust", async () => { + it("keeps every const discriminator of the committed API schema distinct in Rust", () => { // Mirror the generator's own schema preparation so emission order matches. const schema = propagateInternalVisibility( postProcessSchema( stripBooleanLiterals( normalizeSchemaBrandCasing( - JSON.parse(readFileSync(await getApiSchemaPath(), "utf8")) as ApiSchema + JSON.parse( + readFileSync( + new URL("../../../../generated/api.schema.json", import.meta.url), + "utf8" + ) + ) as ApiSchema ) ) as JSONSchema7 ) diff --git a/scripts/ci/device-policy-bootstrap.mjs b/scripts/ci/device-policy-bootstrap.mjs deleted file mode 100644 index 48e30268e8..0000000000 --- a/scripts/ci/device-policy-bootstrap.mjs +++ /dev/null @@ -1,85 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import fs from "node:fs"; -import path from "node:path"; -import { fileURLToPath } from "node:url"; -import { POLICY_FILE, verifyIdentity, verifyIsolation } from "./device-policy-fixture.mjs"; - -function start() { - const plan = JSON.parse(process.env.COPILOT_CI_DEVICE_POLICY_PLAN); - if (process.platform !== "linux" || typeof process.execve !== "function") - throw new Error("Unsupported device-policy container runtime"); - if (process.argv[2] === "--runtime") { - const policy = fs.lstatSync(POLICY_FILE); - verifyIdentity(plan, { - uid: process.getuid(), - gid: process.getgid(), - groups: process.getgroups(), - policy: { - isFile: policy.isFile(), - isSymbolicLink: policy.isSymbolicLink(), - uid: policy.uid, - gid: policy.gid, - mode: policy.mode, - }, - }); - process.execve(plan.executable, plan.argv, plan.env); - throw new Error("Runtime exec unexpectedly returned"); - } - verifyIsolation({ - originalNamespace: plan.originalNamespace, - namespace: fs.readlinkSync("/proc/self/ns/mnt"), - mountinfo: fs.readFileSync("/proc/self/mountinfo", "utf8"), - uid: process.getuid(), - gid: process.getgid(), - }); - if ( - !Number.isSafeInteger(plan.uid) || - plan.uid <= 0 || - !Number.isSafeInteger(plan.gid) || - plan.gid < 0 || - !plan.groups.every((group) => Number.isSafeInteger(group) && group >= 0) - ) - throw new Error("Invalid runner identity"); - const directory = path.dirname(POLICY_FILE); - if (fs.existsSync(directory)) throw new Error("The container already has a device-policy directory"); - fs.mkdirSync(directory, { mode: 0o755 }); - fs.writeFileSync(POLICY_FILE, plan.policy, { mode: 0o644, flag: "wx" }); - fs.chmodSync(POLICY_FILE, 0o644); - - const passwd = fs.readFileSync("/etc/passwd", "utf8"); - if (!passwd.split("\n").some((line) => Number(line.split(":")[2]) === plan.uid)) { - if (/[:\r\n]/.test(plan.home)) throw new Error("Invalid runner home"); - fs.appendFileSync("/etc/passwd", `copilot-sdk-ci:x:${plan.uid}:${plan.gid}::${plan.home}:/bin/sh\n`); - } - const group = fs.readFileSync("/etc/group", "utf8"); - if (!group.split("\n").some((line) => Number(line.split(":")[2]) === plan.gid)) { - fs.appendFileSync("/etc/group", `copilot-sdk-ci:x:${plan.gid}:\n`); - } - if (!fs.existsSync(plan.home)) { - fs.mkdirSync(plan.home, { recursive: true, mode: 0o700 }); - } - fs.chownSync(plan.home, plan.uid, plan.gid); - const args = [ - "/usr/bin/setpriv", - `--reuid=${plan.uid}`, - `--regid=${plan.gid}`, - ...(plan.groups.length ? [`--groups=${plan.groups.join(",")}`] : ["--clear-groups"]), - plan.nodeExecutable, - fileURLToPath(import.meta.url), - "--runtime", - ]; - process.execve("/usr/bin/setpriv", args, process.env); - throw new Error("Identity transition unexpectedly returned"); -} - -if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { - try { - start(); - } catch (error) { - console.error(error.message); - process.exitCode = 1; - } -} diff --git a/scripts/ci/device-policy-fixture.mjs b/scripts/ci/device-policy-fixture.mjs deleted file mode 100644 index 7c65a71354..0000000000 --- a/scripts/ci/device-policy-fixture.mjs +++ /dev/null @@ -1,468 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import { spawn, spawnSync } from "node:child_process"; -import { randomUUID } from "node:crypto"; -import fs from "node:fs"; -import os from "node:os"; -import path from "node:path"; -import { fileURLToPath } from "node:url"; - -const SDK_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); -export const IMAGE = "node:22-bookworm"; -export const POLICY_FILE = "/etc/github-copilot/managed-settings.json"; -export const OWNER_LABEL = "com.github.copilot-sdk.device-policy"; -export const CLIENT_LABEL = `${OWNER_LABEL}.client`; -export const REQUIRED_CASES = [ - { - file: "managed_plugin_progress.e2e.test.ts", - title: "emits presentation-neutral completion after installing required plugins", - }, - { - file: "rpc_server.e2e.test.ts", - title: "should round trip sessionless managed settings", - }, -]; - -const escapeRegex = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); -export const REQUIRED_PATTERN = REQUIRED_CASES.map(({ title }) => escapeRegex(title)).join("|"); -export const PORTABLE_PATTERN = `^(?!.*(?:${REQUIRED_PATTERN})).*$`; - -export function portableTestPattern(source) { - if (source === "checkout") return ""; - if (source !== "published") throw new Error("Unknown runtime artifact source"); - return PORTABLE_PATTERN; -} - -export function runtimeInvocation(kind, runtime, nodeExecutable, args, env) { - if (!["native", "legacy"].includes(kind)) throw new Error("Unknown device-policy entry point"); - return { - executable: kind === "native" ? runtime : nodeExecutable, - argv: kind === "native" ? [runtime, ...args] : [nodeExecutable, runtime, ...args], - env: { ...env }, - }; -} - -export function verifyRequiredResults(report) { - for (const expected of REQUIRED_CASES) { - const matches = (report.testResults ?? []) - .filter((suite) => suite.name.replaceAll("\\", "/").endsWith(`/${expected.file}`)) - .flatMap((suite) => suite.assertionResults ?? []) - .filter((test) => test.title === expected.title); - if (matches.length !== 1 || matches[0].status !== "passed") { - throw new Error(`Required device-policy control did not pass: ${expected.title}`); - } - } -} - -function inside(parent, child) { - const relative = path.posix.relative(parent, child); - return relative === "" || (relative !== ".." && !relative.startsWith("../") && !path.posix.isAbsolute(relative)); -} - -function safeBind(directory) { - if (!path.posix.isAbsolute(directory) || /[:,\r\n]/.test(directory)) { - throw new Error("Device-policy binds require unambiguous absolute Linux paths"); - } - if ( - ["/", "/etc", "/proc", "/sys", "/dev"].some((root) => - root === "/" ? directory === root : inside(root, directory), - ) - ) { - throw new Error("A device-policy container cannot bind host system or policy paths"); - } - return directory; -} - -export function fixtureDirectories({ cwd, env, temporaryDirectory }) { - const candidates = [ - cwd, - ...["COPILOT_HOME", "GH_CONFIG_DIR", "XDG_CONFIG_HOME", "XDG_STATE_HOME"] - .map((name) => env[name]) - .filter(Boolean), - ]; - const roots = new Set(); - for (const candidate of candidates) { - const relative = path.posix.relative(temporaryDirectory, candidate); - const first = relative.split("/")[0]; - if (!inside(temporaryDirectory, candidate) || !/^copilot-test-(work|home|config)-[^/]+$/.test(first)) { - throw new Error("Writable device-policy binds must belong to an SDK E2E fixture"); - } - roots.add(safeBind(path.posix.join(temporaryDirectory, first))); - } - return [...roots]; -} - -export function dockerArguments(plan, cidfile) { - const args = [ - "run", - "--rm", - "--init", - "--interactive", - "--network", - "host", - "--cidfile", - cidfile, - "--label", - `${OWNER_LABEL}=${plan.owner}`, - "--label", - `${CLIENT_LABEL}=${plan.client}`, - "--workdir", - plan.cwd, - "--env", - "COPILOT_CI_DEVICE_POLICY_PLAN", - ]; - const mounts = new Map(); - for (const directory of plan.readonly) mounts.set(safeBind(directory), "ro"); - for (const directory of plan.writable) { - safeBind(directory); - if (mounts.has(directory)) throw new Error("A device-policy bind cannot be both read-only and writable"); - mounts.set(directory, "rw"); - } - for (const [directory, mode] of mounts) args.push("--volume", `${directory}:${directory}:${mode}`); - args.push(IMAGE, "node", path.posix.join(plan.sdkRoot, "scripts/ci/device-policy-bootstrap.mjs")); - return args; -} - -export function verifyIsolation({ originalNamespace, namespace, mountinfo, uid, gid }) { - if (uid !== 0 || gid !== 0 || namespace === originalNamespace || !namespace || !originalNamespace) { - throw new Error("Device policy must be provisioned inside a separate root container"); - } - const mounts = mountinfo - .trim() - .split("\n") - .map((line) => { - const fields = line.split(" "); - const separator = fields.indexOf("-"); - return { target: fields[4], type: fields[separator + 1], propagation: fields.slice(6, separator) }; - }); - const policyMount = mounts - .filter(({ target }) => inside(target, POLICY_FILE)) - .sort((a, b) => b.target.length - a.target.length)[0]; - if ( - !policyMount || - policyMount.target !== "/" || - policyMount.type !== "overlay" || - policyMount.propagation.some((field) => /^(shared|master|propagate_from):/.test(field)) - ) { - throw new Error("Device policy requires a private container root, not a host bind"); - } -} - -export function verifyIdentity(plan, { uid, gid, groups, policy }) { - if ( - uid !== plan.uid || - gid !== plan.gid || - JSON.stringify([...groups].sort((a, b) => a - b)) !== JSON.stringify([...plan.groups].sort((a, b) => a - b)) - ) { - throw new Error("The device-policy runtime must retain the original runner identity"); - } - if ( - !policy.isFile || - policy.isSymbolicLink || - policy.uid !== 0 || - policy.gid !== 0 || - (policy.mode & 0o777) !== 0o644 - ) { - throw new Error("The device policy must be a regular root-owned mode-0644 file"); - } -} - -function command(command, args, options = {}) { - const result = spawnSync(command, args, { encoding: "utf8", ...options }); - if (result.error) throw result.error; - if (result.status !== 0) throw new Error(`${command} failed (${result.status}): ${result.stderr ?? ""}`); - return result.stdout; -} - -export function containerId(value) { - const id = value.trim(); - if (!/^[a-f0-9]{64}$/.test(id)) throw new Error("Invalid owned device-policy container ID"); - return id; -} - -export function verifyOwnership(label, owner) { - if (!owner || label.trim() !== owner) throw new Error("Refusing to mutate a container owned by another job"); -} - -export function pendingLaunches(entries) { - for (const entry of entries) { - if (!/^[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}\.(pending|cid)$/.test(entry)) - throw new Error("Unknown device-policy cleanup registry entry"); - const pending = entry.replace(/\.cid$/, ".pending"); - if (!entries.includes(pending)) throw new Error("Device-policy container identity has no launch receipt"); - } - return entries.filter((entry) => entry.endsWith(".pending")); -} - -function inspectOwned(id, owner) { - const inspected = spawnSync("docker", ["inspect", "--format", `{{index .Config.Labels "${OWNER_LABEL}"}}`, id], { - encoding: "utf8", - }); - if (inspected.status !== 0 && /No such (object|container)/i.test(inspected.stderr)) return false; - if (inspected.error || inspected.status !== 0) - throw inspected.error ?? new Error(`Cannot inspect device-policy container: ${inspected.stderr}`); - verifyOwnership(inspected.stdout, owner); - return true; -} - -function removeOwned(id, owner) { - if (!inspectOwned(id, owner)) return; - const removed = spawnSync("docker", ["rm", "--force", id], { encoding: "utf8" }); - if (removed.status !== 0 && /No such (object|container)/i.test(removed.stderr)) return; - if (removed.error || removed.status !== 0) - throw removed.error ?? new Error(`Cannot remove device-policy container: ${removed.stderr}`); -} - -function stopOwned(id, owner, signal) { - if (!inspectOwned(id, owner)) return; - const stopped = spawnSync("docker", ["stop", "--signal", signal, "--timeout", "5", id], { encoding: "utf8" }); - if (stopped.status !== 0 && /No such (object|container)/i.test(stopped.stderr)) return; - if (stopped.error || stopped.status !== 0) - throw stopped.error ?? new Error(`Cannot stop device-policy container: ${stopped.stderr}`); -} - -function ownedRegistry() { - const registry = process.env.COPILOT_CI_DEVICE_POLICY_REGISTRY; - if (!registry || !process.env.RUNNER_TEMP) throw new Error("Missing owned device-policy container registry"); - const info = fs.lstatSync(registry); - if ( - !inside(process.env.RUNNER_TEMP, registry) || - !path.basename(registry).startsWith("sdk-device-policy-") || - fs.realpathSync(registry) !== registry || - !info.isDirectory() || - info.isSymbolicLink() || - info.uid !== process.getuid() - ) { - throw new Error("Device-policy cidfiles must belong to this runner's temporary registry"); - } - return registry; -} - -export async function launchRuntime(kind) { - if (process.platform !== "linux" || typeof process.execve !== "function") { - throw new Error("The device-policy launcher requires Linux and Node.js 22.15 or newer"); - } - const runtime = - process.env[ - kind === "native" ? "COPILOT_CI_DEVICE_POLICY_NATIVE_PATH" : "COPILOT_CI_DEVICE_POLICY_LEGACY_PATH" - ]; - if (!runtime) throw new Error("Missing original device-policy runtime path"); - const invocation = runtimeInvocation(kind, runtime, process.execPath, process.argv.slice(2), process.env); - const originalEnv = invocation.env; - const fixture = process.env.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; - if (!fixture) { - process.execve(invocation.executable, invocation.argv, originalEnv); - throw new Error("Original runtime exec unexpectedly returned"); - } - - if (!inside(process.cwd(), fixture) || fs.realpathSync(fixture) !== fixture) - throw new Error("Device policy must be a regular file inside the test workspace"); - const descriptor = fs.openSync(fixture, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW); - let policy; - try { - const fixtureInfo = fs.fstatSync(descriptor); - if (!fixtureInfo.isFile() || fixtureInfo.uid !== process.getuid()) - throw new Error("Device policy must be a regular file owned by the runner"); - policy = fs.readFileSync(descriptor, "utf8"); - } finally { - fs.closeSync(descriptor); - } - const parsed = JSON.parse(policy); - if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) - throw new Error("Device policy must be a JSON object"); - // The launcher consumes the test-only path; the runtime must discover the production file. - delete originalEnv.COPILOT_TEST_MANAGED_SETTINGS_FILE_PATH; - const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; - if (!owner) throw new Error("Missing device-policy job ownership"); - const registry = ownedRegistry(); - const writable = fixtureDirectories({ cwd: process.cwd(), env: originalEnv, temporaryDirectory: os.tmpdir() }); - for (const directory of writable) { - const info = fs.lstatSync(directory); - if (fs.realpathSync(directory) !== directory || !info.isDirectory() || info.uid !== process.getuid()) - throw new Error("A writable fixture bind must be a real directory owned by the runner"); - } - const readonly = [ - SDK_ROOT, - path.dirname(process.env.COPILOT_CI_DEVICE_POLICY_LEGACY_PATH), - path.dirname(process.execPath), - ]; - for (const name of [ - "NODE_EXTRA_CA_CERTS", - "SSL_CERT_FILE", - "REQUESTS_CA_BUNDLE", - "CURL_CA_BUNDLE", - "GIT_SSL_CAINFO", - ]) { - const certificate = originalEnv[name]; - if (certificate) { - if (!inside(os.tmpdir(), certificate) || !fs.statSync(certificate).isFile()) - throw new Error("Replay certificates must belong to the test temporary directory"); - readonly.push(certificate); - } - } - const plan = { - owner, - client: randomUUID(), - sdkRoot: SDK_ROOT, - readonly: [...new Set(readonly)], - writable, - policy, - ...invocation, - cwd: process.cwd(), - nodeExecutable: process.execPath, - uid: process.getuid(), - gid: process.getgid(), - groups: process.getgroups(), - home: os.homedir(), - originalNamespace: fs.readlinkSync("/proc/self/ns/mnt"), - }; - const cidfile = path.join(registry, `${plan.client}.cid`); - const pending = path.join(registry, `${plan.client}.pending`); - fs.writeFileSync(pending, "", { flag: "wx", mode: 0o600 }); - const child = spawn("docker", dockerArguments(plan, cidfile), { - stdio: "inherit", - env: { ...originalEnv, COPILOT_CI_DEVICE_POLICY_PLAN: JSON.stringify(plan) }, - }); - let stopping; - const stop = (signal) => { - if (stopping) return; - stopping = signal; - try { - if (fs.existsSync(cidfile)) stopOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner, signal); - else child.kill(signal); - } catch (error) { - console.error(error.message); - process.exitCode = 1; - child.kill("SIGTERM"); - } - }; - const onTerm = () => stop("SIGTERM"); - const onInt = () => stop("SIGINT"); - process.once("SIGTERM", onTerm); - process.once("SIGINT", onInt); - let result; - try { - result = await new Promise((resolve, reject) => { - child.once("error", reject); - child.once("close", (code, signal) => resolve({ code, signal })); - }); - } finally { - process.removeListener("SIGTERM", onTerm); - process.removeListener("SIGINT", onInt); - // A cancelled Docker attach can close before its cidfile is written. - const ids = command("docker", [ - "ps", - "--all", - "--quiet", - "--no-trunc", - "--filter", - `label=${OWNER_LABEL}=${owner}`, - "--filter", - `label=${CLIENT_LABEL}=${plan.client}`, - ]).trim(); - for (const id of ids ? ids.split("\n") : []) removeOwned(containerId(id), owner); - if (fs.existsSync(cidfile)) { - removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); - fs.unlinkSync(cidfile); - fs.unlinkSync(pending); - } else { - throw new Error("Docker launch has no container identity; cleanup cannot prove settlement"); - } - } - if (process.exitCode) return; - if (stopping || result.signal) process.kill(process.pid, stopping ?? result.signal); - else process.exitCode = result.code ?? 1; -} - -function prepare() { - if ( - process.platform !== "linux" || - !process.env.GITHUB_ENV || - !process.env.GITHUB_OUTPUT || - !process.env.RUNNER_TEMP - ) { - throw new Error("Device-policy containers can only be prepared in Linux CI"); - } - const native = fs.realpathSync(process.env.COPILOT_CLI_PATH); - const legacy = fs.realpathSync(process.env.COPILOT_LEGACY_CLI_PATH); - const packageRoot = path.dirname(legacy); - if ( - native !== path.join(packageRoot, "prebuilds/linux-x64/copilot-runtime") || - path.basename(legacy) !== "app.js" - ) { - throw new Error("Device-policy fixtures require the staged GNU package's two original entry points"); - } - const expected = JSON.parse(fs.readFileSync(path.join(SDK_ROOT, "nodejs/package.json"), "utf8")).copilotCliVersion; - const actual = JSON.parse(fs.readFileSync(path.join(packageRoot, "package.json"), "utf8")).version; - if (expected !== actual) throw new Error("Device-policy package differs from the pinned CLI version"); - if (fs.existsSync(POLICY_FILE)) throw new Error("The CI host already has a device policy"); - command("docker", ["pull", IMAGE], { stdio: "inherit" }); - const registry = fs.mkdtempSync(path.join(process.env.RUNNER_TEMP, "sdk-device-policy-")); - const owner = randomUUID(); - fs.appendFileSync( - process.env.GITHUB_ENV, - `COPILOT_CI_DEVICE_POLICY_NATIVE_PATH=${native}\nCOPILOT_CI_DEVICE_POLICY_LEGACY_PATH=${legacy}\nCOPILOT_CI_DEVICE_POLICY_OWNER=${owner}\nCOPILOT_CI_DEVICE_POLICY_REGISTRY=${registry}\n`, - ); - fs.appendFileSync( - process.env.GITHUB_OUTPUT, - `native-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-native.js")}\nlegacy-launcher=${path.join(SDK_ROOT, "scripts/ci/device-policy-legacy.js")}\n`, - ); -} - -function cleanup() { - const owner = process.env.COPILOT_CI_DEVICE_POLICY_OWNER; - if (!owner) throw new Error("Missing device-policy job ownership"); - const ids = command("docker", [ - "ps", - "--all", - "--quiet", - "--no-trunc", - "--filter", - `label=${OWNER_LABEL}=${owner}`, - ]).trim(); - const errors = []; - for (const id of ids ? ids.split("\n") : []) { - try { - removeOwned(containerId(id), owner); - } catch (error) { - errors.push(error.message); - } - } - const registry = ownedRegistry(); - for (const entry of pendingLaunches(fs.readdirSync(registry))) { - const pending = path.join(registry, entry); - const cidfile = path.join(registry, entry.replace(/\.pending$/, ".cid")); - try { - if (!fs.existsSync(cidfile)) throw new Error("Unsettled Docker launch has no container identity"); - removeOwned(containerId(fs.readFileSync(cidfile, "utf8")), owner); - fs.unlinkSync(cidfile); - fs.unlinkSync(pending); - } catch (error) { - errors.push(error.message); - } - } - if (errors.length) throw new Error(`Device-policy cleanup did not settle: ${errors.join("; ")}`); - if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); -} - -if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { - try { - const [action, argument] = process.argv.slice(2); - if (action === "prepare") prepare(); - else if (action === "cleanup") cleanup(); - else if (action === "portable-pattern") console.log(portableTestPattern(argument)); - else if (action === "required-pattern") console.log(REQUIRED_PATTERN); - else if (action === "verify" && argument) { - verifyRequiredResults(JSON.parse(fs.readFileSync(argument, "utf8"))); - if (fs.existsSync(POLICY_FILE)) throw new Error("The device fixture modified the CI host"); - } else - throw new Error( - "Usage: device-policy-fixture.mjs ", - ); - } catch (error) { - console.error(error.message); - process.exitCode = 1; - } -} diff --git a/scripts/ci/device-policy-fixture.test.mjs b/scripts/ci/device-policy-fixture.test.mjs deleted file mode 100644 index 8f33a412c3..0000000000 --- a/scripts/ci/device-policy-fixture.test.mjs +++ /dev/null @@ -1,207 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import assert from "node:assert/strict"; -import test from "node:test"; -import "./device-policy-bootstrap.mjs"; -import { - CLIENT_LABEL, - containerId, - dockerArguments, - fixtureDirectories, - OWNER_LABEL, - pendingLaunches, - POLICY_FILE, - portableTestPattern, - runtimeInvocation, - PORTABLE_PATTERN, - REQUIRED_CASES, - REQUIRED_PATTERN, - verifyIdentity, - verifyIsolation, - verifyOwnership, - verifyRequiredResults, -} from "./device-policy-fixture.mjs"; - -const report = () => ({ - testResults: REQUIRED_CASES.map(({ file, title }) => ({ - name: `/workspace/nodejs/test/e2e/${file}`, - assertionResults: [{ title, status: "passed" }], - })), -}); - -test("requires both unchanged strict controls to pass", () => { - verifyRequiredResults(report()); - for (const index of [0, 1]) { - for (const status of ["pending", "skipped", "failed", "todo"]) { - const result = report(); - result.testResults[index].assertionResults[0].status = status; - assert.throws(() => verifyRequiredResults(result), /did not pass/); - } - const missing = report(); - missing.testResults.splice(index, 1); - assert.throws(() => verifyRequiredResults(missing), /did not pass/); - const duplicate = report(); - duplicate.testResults.push(duplicate.testResults[index]); - assert.throws(() => verifyRequiredResults(duplicate), /did not pass/); - } - assert.throws(() => verifyRequiredResults({}), /did not pass/); -}); - -test("assigns only the two device-fixture cases to the mandatory Linux gate", () => { - assert.equal(portableTestPattern("checkout"), ""); - assert.equal(portableTestPattern("published"), PORTABLE_PATTERN); - assert.throws(() => portableTestPattern(""), /Unknown/); - const portable = new RegExp(PORTABLE_PATTERN); - const required = new RegExp(REQUIRED_PATTERN); - for (const { title } of REQUIRED_CASES) { - assert.equal(portable.test(`Suite ${title}`), false); - assert.equal(required.test(title), true); - } - for (const title of [ - "should clear the managed settings cache", - "should expose the managed settings schema", - "should list server sessions", - ]) { - assert.equal(portable.test(title), true); - assert.equal(required.test(title), false); - } -}); - -test("preserves both runtime entry points' original argv and environment", () => { - const args = ["--stdio", "--argument-with-spaces=one two", "--literal=;$()"]; - const env = { TOKEN: "private-value", COPILOT_HOME: "/tmp/copilot-test-home-one" }; - const native = runtimeInvocation("native", "/package/copilot-runtime", "/node/bin/node", args, env); - assert.equal(native.executable, "/package/copilot-runtime"); - assert.deepEqual(native.argv, ["/package/copilot-runtime", ...args]); - const legacy = runtimeInvocation("legacy", "/package/app.js", "/node/bin/node", args, env); - assert.equal(legacy.executable, "/node/bin/node"); - assert.deepEqual(legacy.argv, ["/node/bin/node", "/package/app.js", ...args]); - assert.deepEqual(native.env, env); - assert.notEqual(native.env, env); - assert.throws(() => runtimeInvocation("unknown", "", "", [], {}), /entry point/); -}); - -test("binds only explicitly owned E2E fixture directories for writing", () => { - const roots = fixtureDirectories({ - cwd: "/tmp/copilot-test-work-one", - env: { - COPILOT_HOME: "/tmp/copilot-test-home-one", - GH_CONFIG_DIR: "/tmp/copilot-test-config-one", - XDG_CONFIG_HOME: "/tmp/copilot-test-config-one", - XDG_STATE_HOME: "/tmp/copilot-test-work-one/local-home", - }, - temporaryDirectory: "/tmp", - }); - assert.deepEqual(roots, [ - "/tmp/copilot-test-work-one", - "/tmp/copilot-test-home-one", - "/tmp/copilot-test-config-one", - ]); - for (const cwd of ["/etc", "/home/runner", "/tmp", "/tmp/../../etc", "/tmp/not-a-fixture"]) { - assert.throws(() => fixtureDirectories({ cwd, env: {}, temporaryDirectory: "/tmp" }), /fixture/); - } -}); - -test("keeps policy and secret environment contents out of Docker arguments", () => { - const plan = { - owner: "job-one", - client: "client-one", - sdkRoot: "/workspace", - cwd: "/tmp/copilot-test-work-one", - readonly: ["/workspace", "/artifacts/package"], - writable: ["/tmp/copilot-test-work-one"], - env: { SECRET: "must-not-appear" }, - policy: '{"model":"secret-model"}', - }; - const args = dockerArguments(plan, "/registry/client-one.cid"); - assert.deepEqual(args.slice(0, 7), ["run", "--rm", "--init", "--interactive", "--network", "host", "--cidfile"]); - assert.equal(args.includes(`${OWNER_LABEL}=job-one`), true); - assert.equal(args.includes(`${CLIENT_LABEL}=client-one`), true); - assert.equal(args.includes("COPILOT_CI_DEVICE_POLICY_PLAN"), true); - assert.equal(args.includes("/artifacts/package:/artifacts/package:ro"), true); - assert.equal(args.includes("/tmp/copilot-test-work-one:/tmp/copilot-test-work-one:rw"), true); - assert.equal(args.join(" ").includes("must-not-appear"), false); - assert.equal(args.join(" ").includes("secret-model"), false); - const other = dockerArguments({ ...plan, client: "client-two" }, "/registry/client-two.cid"); - assert.notDeepEqual(args, other); - for (const directory of ["/", "/etc", "/etc/github-copilot", "/proc", "/sys", "/dev", "/path:ambiguous"]) { - assert.throws(() => dockerArguments({ ...plan, readonly: [directory] }, "/registry/client.cid")); - } - assert.throws(() => dockerArguments({ ...plan, readonly: plan.writable }, "/registry/client.cid"), /both/); -}); - -test("requires exact job ownership and full container IDs before cleanup", () => { - verifyOwnership("job-one\n", "job-one"); - for (const [label, owner] of [ - ["job-two", "job-one"], - ["", ""], - ["", "job-one"], - ]) { - assert.throws(() => verifyOwnership(label, owner), /another job/); - } - assert.equal(containerId(`${"a".repeat(64)}\n`), "a".repeat(64)); - for (const id of ["", "a".repeat(12), `${"a".repeat(64)} extra`, "G".repeat(64)]) { - assert.throws(() => containerId(id), /container ID/); - } -}); - -test("exposes interrupted or unknown launch receipts instead of claiming cleanup", () => { - const client = "12345678-1234-1234-1234-123456789abc"; - assert.deepEqual(pendingLaunches([]), []); - assert.deepEqual(pendingLaunches([`${client}.pending`, `${client}.cid`]), [`${client}.pending`]); - assert.deepEqual(pendingLaunches([`${client}.pending`]), [`${client}.pending`]); - assert.throws(() => pendingLaunches([`${client}.cid`]), /no launch receipt/); - assert.throws(() => pendingLaunches(["unknown"]), /Unknown/); -}); - -const isolated = () => ({ - originalNamespace: "mnt:[100]", - namespace: "mnt:[200]", - uid: 0, - gid: 0, - mountinfo: "1 0 0:1 / / rw - overlay overlay rw\n2 1 0:2 / /workspace ro - ext4 disk ro", -}); - -test("requires private container backing before any policy mutation", () => { - verifyIsolation(isolated()); - for (const change of [ - { namespace: "mnt:[100]" }, - { uid: 1001 }, - { gid: 1001 }, - { originalNamespace: "" }, - { mountinfo: "1 0 0:1 / / rw shared:1 - overlay overlay rw" }, - { mountinfo: "1 0 0:1 / / rw - ext4 disk rw" }, - { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc rw - ext4 host rw` }, - { mountinfo: `${isolated().mountinfo}\n3 1 0:3 / /etc/github-copilot rw - ext4 host rw` }, - ]) - assert.throws(() => verifyIsolation({ ...isolated(), ...change })); -}); - -test("retains runner uid, gid and groups with an ordinary root-owned device file", () => { - const plan = { uid: 1001, gid: 1001, groups: [1001, 118] }; - const runtime = { - uid: 1001, - gid: 1001, - groups: [118, 1001], - policy: { isFile: true, isSymbolicLink: false, uid: 0, gid: 0, mode: 0o100644 }, - }; - verifyIdentity(plan, runtime); - for (const change of [{ uid: 0 }, { gid: 0 }, { groups: [] }]) { - assert.throws(() => verifyIdentity(plan, { ...runtime, ...change }), /identity/); - } - for (const change of [ - { uid: 1001 }, - { gid: 1001 }, - { mode: 0o100666 }, - { isSymbolicLink: true }, - { isFile: false }, - ]) { - assert.throws( - () => verifyIdentity(plan, { ...runtime, policy: { ...runtime.policy, ...change } }), - /root-owned/, - ); - } - assert.equal(POLICY_FILE, "/etc/github-copilot/managed-settings.json"); -}); diff --git a/scripts/ci/device-policy-legacy.js b/scripts/ci/device-policy-legacy.js deleted file mode 100644 index 806647cf11..0000000000 --- a/scripts/ci/device-policy-legacy.js +++ /dev/null @@ -1,10 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import("./device-policy-fixture.mjs") - .then(({ launchRuntime }) => launchRuntime("legacy")) - .catch((error) => { - console.error(error.message); - process.exitCode = 1; - }); diff --git a/scripts/ci/device-policy-native.js b/scripts/ci/device-policy-native.js deleted file mode 100644 index d269a07ea6..0000000000 --- a/scripts/ci/device-policy-native.js +++ /dev/null @@ -1,10 +0,0 @@ -/*--------------------------------------------------------------------------------------------- - * Copyright (c) Microsoft Corporation. All rights reserved. - *--------------------------------------------------------------------------------------------*/ - -import("./device-policy-fixture.mjs") - .then(({ launchRuntime }) => launchRuntime("native")) - .catch((error) => { - console.error(error.message); - process.exitCode = 1; - }); diff --git a/scripts/ci/run-dotnet-tests.sh b/scripts/ci/run-dotnet-tests.sh index 14d9a97208..ea0312c4a6 100755 --- a/scripts/ci/run-dotnet-tests.sh +++ b/scripts/ci/run-dotnet-tests.sh @@ -11,7 +11,6 @@ Usage: run-dotnet-tests.sh [--help] Runs the full .NET SDK test project. Environment variables: DOTNET_TEST_FILTER Optional dotnet test filter (e.g. for a backend or transport). DOTNET_TEST_RUNTIME Optional runtime identifier passed to dotnet test. - DOTNET_TEST_FRAMEWORK Optional target framework; unset runs all project targets. DOTNET_TEST_RESULTS_DIRECTORY Results directory (default: TestResults). EOF } @@ -41,7 +40,6 @@ args=( ) filter="${DOTNET_TEST_FILTER:-}" runtime="${DOTNET_TEST_RUNTIME:-}" -framework="${DOTNET_TEST_FRAMEWORK:-}" if [[ -n "$filter" ]]; then args+=(--filter "$filter") @@ -49,8 +47,5 @@ fi if [[ -n "$runtime" ]]; then args+=(--runtime "$runtime") fi -if [[ -n "$framework" ]]; then - args+=(--framework "$framework") -fi dotnet test "${args[@]}"