diff --git a/.github/workflows/daily-rig-decomposition-bench.lock.yml b/.github/workflows/daily-rig-decomposition-bench.lock.yml index f5e3720..7af6b5a 100644 --- a/.github/workflows/daily-rig-decomposition-bench.lock.yml +++ b/.github/workflows/daily-rig-decomposition-bench.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"a6d6f9950a2cc8a66a481573fb54ad6b2b25079742bdc06be67ef92361b6ef0b","body_hash":"80e2309abed14b28d3c8752fd4b40858abdd4aaacfe1826fc3afbcacd434b3ee","compiler_version":"v0.91.4","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.92","copilot-sdk":"1.0.16"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"a6d6f9950a2cc8a66a481573fb54ad6b2b25079742bdc06be67ef92361b6ef0b","body_hash":"6304c8370967c6d3c1cbd51b93a744bbfdf6cb4f40c5635701fd6ccebe7ceec6","compiler_version":"v0.91.4","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.92","copilot-sdk":"1.0.16"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_DEFAULT_OTLP_ENDPOINT","GH_AW_DEFAULT_OTLP_HEADERS","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"a63fe074b43ff5f66bcf1af0dc77f7bfc85b9d93","version":"v0.91.4"}],"skills":["githubnext/rig/skills/rig@0cf43bec9424276137fc101983033039d4e0e592"],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.31","digest":"sha256:c4ab1d48d533cc7daaa5f2e1193d1a644888242fb8813d49c6de46d129224b76","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.31@sha256:c4ab1d48d533cc7daaa5f2e1193d1a644888242fb8813d49c6de46d129224b76"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.31","digest":"sha256:a5a37489635109334a5e2b5cb2eaba32e8c8f1a89d8dea6009baa962a8904c4a","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.31@sha256:a5a37489635109334a5e2b5cb2eaba32e8c8f1a89d8dea6009baa962a8904c4a"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.31","digest":"sha256:90a46d2e6e910ace2c09c2dcf6b56dc374ed513d9da642a4e9fcfcd6e9248a9d","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.31@sha256:90a46d2e6e910ace2c09c2dcf6b56dc374ed513d9da642a4e9fcfcd6e9248a9d"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.29","digest":"sha256:ec08867ac8a4823e01efb2de2ba85a313199bb7ae666ef70ae58effc162a9bf3","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.29@sha256:ec08867ac8a4823e01efb2de2ba85a313199bb7ae666ef70ae58effc162a9bf3"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:a8082161d7dceda14b68f32eb39d0eaa96b825d07f5895b096afab9d9e0c7748","pinned_image":"ghcr.io/github/gh-aw-node@sha256:a8082161d7dceda14b68f32eb39d0eaa96b825d07f5895b096afab9d9e0c7748"},{"image":"ghcr.io/github/github-mcp-server:v1.12.2","digest":"sha256:508a0857ec762b1ab1cece29193345b501fab1dd9d1228a7b617062954cecac6","pinned_image":"ghcr.io/github/github-mcp-server:v1.12.2@sha256:508a0857ec762b1ab1cece29193345b501fab1dd9d1228a7b617062954cecac6"}],"mcp_servers":[{"name":"github","tools":["get_commit","get_file_contents","get_latest_release","get_me","get_pull_request","get_pull_request_comments","get_pull_request_diff","get_pull_request_files","get_pull_request_review_comments","get_pull_request_reviews","get_pull_request_status","get_release_by_tag","get_tag","issue_read","list_branches","list_commits","list_issue_types","list_issues","list_pull_requests","list_releases","list_starred_repositories","list_tags","pull_request_read","search_code","search_issues","search_pull_requests","search_repositories"]},{"name":"safeoutputs","tools":["create_issue","missing_data","missing_tool","noop","report_incomplete"]}],"threat_detection":{"mode":"enabled"}} # This file was automatically generated by gh-aw (v0.91.4). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/daily-rig-decomposition-bench.md b/.github/workflows/daily-rig-decomposition-bench.md index 23f792f..7ffefa3 100644 --- a/.github/workflows/daily-rig-decomposition-bench.md +++ b/.github/workflows/daily-rig-decomposition-bench.md @@ -140,16 +140,21 @@ timeout. Treat them as non-negotiable: solution. Reserve object schemas for structured metadata such as the task picker, and do not make any solver re-emit a large solution inside a wrapper object. - Verify it by piping the source over stdin to the installed rig CLI, run from the skill - directory so Node's package self-reference resolves the bare `"rig"` import: - `--typecheck` first, and only on success `--server` to execute it. Give the writer up to - 2 attempts total; on a failure, pass it back its own previous source and the exact - captured error, repeat the skeleton, and ask it to fix precisely what the error names, - preserving what already worked — do not invent unrelated API edits of your own. Stop - early if there is not enough time budget left for another attempt. Record each attempt's - typecheck and execute pass/fail together with the exact captured output, and parse the - final solution out of the successful run's JSON stdout. Record how long the whole - decomposition phase takes. + Verify non-empty source by piping it over stdin to the installed rig CLI, run from the + skill directory so Node's package self-reference resolves the bare `"rig"` import: + `--typecheck` first, and only on success `--server` to execute it. Treat a `null`, + missing, or whitespace-only writer result as a source-generation failure: never pass it + to the fixer or invoke the CLI with it. Give the writer up to 2 attempts total. If the + first result is non-empty, pass that source and the exact captured typecheck/execute error + to the fixer, repeat the skeleton, and ask it to fix precisely what the error names while + preserving what already worked — do not invent unrelated API edits of your own. If the + first result is empty, use the second attempt to ask the fixer to generate the complete + program from the original task and skeleton, explicitly noting that there is no source to + fix. Stop early if there is not enough time budget left for another attempt. Record each + attempt's source-generation result and typecheck/execute status (including "skipped" when + there is no valid source) with the exact captured output, and parse the final solution + out of the successful run's JSON stdout. Record how long the whole decomposition phase + takes. 4. **Grade both.** A `large` agent limited to a single turn scores each solution 0-10 on how completely and correctly it satisfies the success criteria, picks a winner of @@ -182,9 +187,10 @@ Emit one `create-issue` safe output with: - The chosen **task** (title, domain, one-paragraph description, success criteria list). - A **timing comparison** table: single-call duration vs. decomposed duration (ms), and the decomposed program's final pass/fail status. - - For each decomposition attempt, in order: the attempt number, typecheck pass/fail, and - execute pass/fail, with the exact captured error text (if any) in a collapsible - `
` block, and a one-line note on what was fixed in the next attempt (if any). + - For each decomposition attempt, in order: the attempt number, source-generation result, + typecheck, and execute status (including "skipped" when there is no valid source), with + the exact captured error text (if any) in a collapsible `
` block, and a one-line + note on what was fixed in the next attempt (if any). - The **grading** results: both scores, the winner, and the grader's rationale, verbatim. - The complete, verbatim **single-call solution** in a collapsible `
` block. - The complete, verbatim **decomposed rig program source** in a ```ts fence, and the