diff --git a/.changeset/patch-report-awf-ai-credits.md b/.changeset/patch-report-awf-ai-credits.md new file mode 100644 index 00000000000..b380111b1ff --- /dev/null +++ b/.changeset/patch-report-awf-ai-credits.md @@ -0,0 +1,5 @@ +--- +"gh-aw": patch +--- + +Fixed AI Credits usage reports to preserve AWF-reported per-request and cumulative totals, including explicit cache-token semantics when available, while retaining legacy repricing for older records. diff --git a/.github/workflows/pr-code-quality-reviewer.lock.yml b/.github/workflows/pr-code-quality-reviewer.lock.yml index 61bcfea1fde..85ab61b646e 100644 --- a/.github/workflows/pr-code-quality-reviewer.lock.yml +++ b/.github/workflows/pr-code-quality-reviewer.lock.yml @@ -1,5 +1,5 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c729c0941817ab38ba55591368d4df3b75d04434663a5b0e4add6a4e72c488f5","body_hash":"6f65228daeb1726800c2a892e59cc430623f2c159d03cf73b21eae97622b61c8","strict":true,"agent_id":"pi","agent_model":"openai/gpt-5.4","engine_versions":{"pi":"0.84.3"}} -# gh-aw-manifest: {"version":1,"secrets":["CODEX_API_KEY","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN","OPENAI_API_KEY"],"actions":[{"repo":"actions/cache","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10","digest":"sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10@sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10","digest":"sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10@sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.10","digest":"sha256:a61070cb7f21840c5f2ec74d55b49adf0652d0348ce059015aaaca33a8cb6b45","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.10@sha256:a61070cb7f21840c5f2ec74d55b49adf0652d0348ce059015aaaca33a8cb6b45"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10","digest":"sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10@sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.13","digest":"sha256:ec4008521c610e1113ed557ecec0ff64a2c2111e4cfa817bab54d9b7da24c7cc","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.13@sha256:ec4008521c610e1113ed557ecec0ff64a2c2111e4cfa817bab54d9b7da24c7cc"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e","pinned_image":"ghcr.io/github/gh-aw-node@sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e"},{"image":"ghcr.io/github/github-mcp-server:v1.11.0","digest":"sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699","pinned_image":"ghcr.io/github/github-mcp-server:v1.11.0@sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699"}],"has_pull_request":true,"mcp_servers":[{"name":"safeoutputs","tools":["create_check_run","create_pull_request_review_comment","missing_data","missing_tool","noop","submit_pull_request_review"]}]} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"60bc588a3c6afeea8f9da8df62fa2c5776277781953a802b77459bd54d59ecff","body_hash":"6f65228daeb1726800c2a892e59cc430623f2c159d03cf73b21eae97622b61c8","strict":true,"agent_id":"copilot","agent_model":"copilot/gpt-5.4","engine_versions":{"copilot":"1.0.80"}} +# gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10","digest":"sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10@sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10","digest":"sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10@sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.10","digest":"sha256:a61070cb7f21840c5f2ec74d55b49adf0652d0348ce059015aaaca33a8cb6b45","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.10@sha256:a61070cb7f21840c5f2ec74d55b49adf0652d0348ce059015aaaca33a8cb6b45"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10","digest":"sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10@sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.13","digest":"sha256:ec4008521c610e1113ed557ecec0ff64a2c2111e4cfa817bab54d9b7da24c7cc","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.13@sha256:ec4008521c610e1113ed557ecec0ff64a2c2111e4cfa817bab54d9b7da24c7cc"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e","pinned_image":"ghcr.io/github/gh-aw-node@sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e"},{"image":"ghcr.io/github/github-mcp-server:v1.11.0","digest":"sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699","pinned_image":"ghcr.io/github/github-mcp-server:v1.11.0@sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699"}],"has_pull_request":true,"mcp_servers":[{"name":"safeoutputs","tools":["create_check_run","create_pull_request_review_comment","missing_data","missing_tool","noop","submit_pull_request_review"]}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # # ___ _ _ @@ -34,7 +34,6 @@ # - shared/pr-review-base.md # # Secrets used: -# - CODEX_API_KEY # - GH_AW_GITHUB_MCP_SERVER_TOKEN # - GH_AW_GITHUB_TOKEN # - GH_AW_OTEL_GRAFANA_AUTHORIZATION @@ -42,7 +41,6 @@ # - GH_AW_OTEL_SENTRY_AUTHORIZATION # - GH_AW_OTEL_SENTRY_ENDPOINT # - GITHUB_TOKEN -# - OPENAI_API_KEY # # Custom actions used: # - actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 @@ -93,7 +91,7 @@ run-name: "PR Code Quality Reviewer" env: OTEL_EXPORTER_OTLP_ENDPOINT: ${{ secrets.GH_AW_OTEL_SENTRY_ENDPOINT }} OTEL_SERVICE_NAME: gh-aw.pr-code-quality-reviewer - OTEL_RESOURCE_ATTRIBUTES: 'gh-aw.workflow.name=PR%20Code%20Quality%20Reviewer,gh-aw.repository=${{ github.repository }},gh-aw.run.id=${{ github.run_id }},github.run_id=${{ github.run_id }},gh-aw.engine.id=pi' + OTEL_RESOURCE_ATTRIBUTES: 'gh-aw.workflow.name=PR%20Code%20Quality%20Reviewer,gh-aw.repository=${{ github.repository }},gh-aw.run.id=${{ github.run_id }},github.run_id=${{ github.run_id }},gh-aw.engine.id=copilot' OTEL_EXPORTER_OTLP_HEADERS: x-sentry-auth=${{ secrets.GH_AW_OTEL_SENTRY_AUTHORIZATION }} GH_AW_OTLP_ALL_HEADERS: x-sentry-auth=${{ secrets.GH_AW_OTEL_SENTRY_AUTHORIZATION }},Authorization=${{ secrets.GH_AW_OTEL_GRAFANA_AUTHORIZATION }} GH_AW_OTLP_ENDPOINTS: '[{"url":"${{ secrets.GH_AW_OTEL_SENTRY_ENDPOINT }}","headers":"x-sentry-auth=${{ secrets.GH_AW_OTEL_SENTRY_AUTHORIZATION }}"},{"url":"${{ secrets.GH_AW_OTEL_GRAFANA_ENDPOINT }}","headers":"Authorization=${{ secrets.GH_AW_OTEL_GRAFANA_AUTHORIZATION }}"}]' @@ -129,7 +127,6 @@ jobs: lockdown_check_failed: ${{ steps.generate_aw_info.outputs.lockdown_check_failed == 'true' }} model: ${{ steps.generate_aw_info.outputs.model }} oauth_token_check_failed: ${{ steps.check-oauth-tokens.outputs.oauth_token_check_failed == 'true' }} - secret_verification_result: ${{ steps.validate-secret.outputs.verification_result }} setup-parent-span-id: ${{ steps.setup.outputs.parent-span-id || steps.setup.outputs.span-id }} setup-span-id: ${{ steps.setup.outputs.span-id }} setup-trace-id: ${{ steps.setup.outputs.trace-id }} @@ -158,19 +155,19 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Mask OTLP telemetry headers run: bash "${RUNNER_TEMP}/gh-aw/actions/mask_otlp_headers.sh" - name: Generate agentic run info id: generate_aw_info env: - GH_AW_INFO_ENGINE_ID: "pi" - GH_AW_INFO_ENGINE_NAME: "Pi" - GH_AW_INFO_MODEL: "openai/gpt-5.4" - GH_AW_INFO_VERSION: "0.84.3" - GH_AW_INFO_AGENT_VERSION: "0.84.3" + GH_AW_INFO_ENGINE_ID: "copilot" + GH_AW_INFO_ENGINE_NAME: "GitHub Copilot CLI" + GH_AW_INFO_MODEL: "copilot/gpt-5.4" + GH_AW_INFO_VERSION: "1.0.80" + GH_AW_INFO_AGENT_VERSION: "1.0.80" GH_AW_INFO_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_INFO_EXPERIMENTAL: "false" GH_AW_INFO_SUPPORTS_TOOLS_ALLOWLIST: "true" @@ -256,12 +253,6 @@ jobs: setupGlobals(core, github, context, exec, io, getOctokit); const { main } = require(path.join(actionsDir, 'add_reaction.cjs')); await main(); - - name: Validate CODEX_API_KEY or OPENAI_API_KEY secret - id: validate-secret - run: bash "${RUNNER_TEMP}/gh-aw/actions/validate_multi_secret.sh" CODEX_API_KEY OPENAI_API_KEY Pi https://github.github.com/gh-aw/reference/engines/#pi - env: - CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - name: Check for OAuth tokens id: check-oauth-tokens run: bash "${RUNNER_TEMP}/gh-aw/actions/check_oauth_tokens.sh" @@ -284,8 +275,8 @@ jobs: fetch-depth: 1 - name: Save agent config folders for base branch restoration env: - GH_AW_AGENT_FOLDERS: ".agents .github .pi" - GH_AW_AGENT_FILES: "AGENTS.md PI.md" + GH_AW_AGENT_FOLDERS: ".agents .github" + GH_AW_AGENT_FILES: "AGENTS.md" run: | bash "${RUNNER_TEMP}/gh-aw/actions/save_base_github_folders.sh" - name: Check workflow lock file @@ -371,7 +362,7 @@ jobs: uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: GH_AW_PROMPT: ${{ runner.temp }}/gh-aw/aw-prompts/prompt.txt - GH_AW_ENGINE_ID: "pi" + GH_AW_ENGINE_ID: "copilot" GH_AW_GITHUB_ACTOR: ${{ github.actor }} GH_AW_EXPR_799BE623: ${{ github.event.issue.number || github.event.pull_request.number }} GH_AW_GITHUB_REPOSITORY: ${{ github.repository }} @@ -456,8 +447,8 @@ jobs: /tmp/gh-aw/aw-prompts/prompt-import-tree.json /tmp/gh-aw/github_rate_limits.jsonl /tmp/gh-aw/base - /tmp/gh-aw/.pi/agents - /tmp/gh-aw/.pi/skills + /tmp/gh-aw/.github/agents + /tmp/gh-aw/.github/skills if-no-files-found: ignore retention-days: 1 @@ -487,18 +478,28 @@ jobs: GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} GH_AW_WORKFLOW_ID_SANITIZED: prcodequalityreviewer outputs: + agentic_engine_timeout: ${{ steps.detect-agent-errors.outputs.agentic_engine_timeout || 'false' }} ai_credits_rate_limit_error: ${{ steps.parse-mcp-gateway.outputs.ai_credits_rate_limit_error || 'false' }} aic: ${{ steps.parse-mcp-gateway.outputs.aic }} ambient_context: ${{ steps.parse-mcp-gateway.outputs.ambient_context }} checkout_pr_success: ${{ steps.checkout-pr.outputs.checkout_pr_success || 'true' }} effective_tokens: ${{ steps.parse-mcp-gateway.outputs.effective_tokens }} has_patch: ${{ steps.collect_output.outputs.has_patch }} + http_400_response_error: ${{ steps.detect-agent-errors.outputs.http_400_response_error || 'false' }} + inference_access_error: ${{ steps.detect-agent-errors.outputs.inference_access_error || 'false' }} + invocation_cap_exceeded: ${{ steps.detect-agent-errors.outputs.invocation_cap_exceeded || 'false' }} + max_cache_misses_exceeded: ${{ steps.detect-agent-errors.outputs.max_cache_misses_exceeded || 'false' }} + mcp_policy_error: ${{ steps.detect-agent-errors.outputs.mcp_policy_error || 'false' }} + missing_model_pricing_error: ${{ steps.detect-agent-errors.outputs.missing_model_pricing_error || 'false' }} + missing_model_pricing_model_name: ${{ steps.detect-agent-errors.outputs.missing_model_pricing_model_name || '' }} model: ${{ needs.activation.outputs.model }} + model_not_supported_error: ${{ steps.detect-agent-errors.outputs.model_not_supported_error || 'false' }} output: ${{ steps.collect_output.outputs.output }} output_types: ${{ steps.collect_output.outputs.output_types }} setup-parent-span-id: ${{ steps.setup.outputs.parent-span-id || steps.setup.outputs.span-id }} setup-span-id: ${{ steps.setup.outputs.span-id }} setup-trace-id: ${{ steps.setup.outputs.trace-id }} + shell_expansion_guard_rejected: ${{ steps.detect-agent-errors.outputs.shell_expansion_guard_rejected || 'false' }} unknown_model_ai_credits: ${{ steps.parse-mcp-gateway.outputs.unknown_model_ai_credits || 'false' }} steps: - name: Checkout actions folder @@ -520,9 +521,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Set runtime paths id: set-runtime-paths run: | @@ -598,15 +599,13 @@ jobs: setupGlobals(core, github, context, exec, io, getOctokit); const { main } = require(path.join(actionsDir, 'checkout_pr_branch.cjs')); await main(); - - name: Setup Node.js - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 - with: - node-version: '24' - package-manager-cache: false + - name: Install GitHub Copilot CLI + run: bash "${RUNNER_TEMP}/gh-aw/actions/install_copilot_cli.sh" + env: + GH_HOST: github.com + GH_AW_COMPILED_VERSION: dev - name: Install AWF binary run: bash "${RUNNER_TEMP}/gh-aw/actions/install_awf_binary.sh" v0.28.10 --rootless - - name: Install Pi CLI - run: npm install --ignore-scripts -g @earendil-works/pi-coding-agent@0.84.3 - name: Determine automatic lockdown mode for GitHub MCP Server id: determine-automatic-lockdown uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 (source v9) @@ -631,17 +630,17 @@ jobs: - name: Restore agent config folders from base branch if: steps.checkout-pr.outcome == 'success' env: - GH_AW_AGENT_FOLDERS: ".agents .github .pi" - GH_AW_AGENT_FILES: "AGENTS.md PI.md" + GH_AW_AGENT_FOLDERS: ".agents .github" + GH_AW_AGENT_FILES: "AGENTS.md" run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_base_github_folders.sh" - name: Restore inline sub-agents from activation artifact env: - GH_AW_SUB_AGENT_DIR: ".pi/agents" + GH_AW_SUB_AGENT_DIR: ".github/agents" GH_AW_SUB_AGENT_EXT: ".agent.md" run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_inline_sub_agents.sh" - name: Restore inline skills from activation artifact env: - GH_AW_SKILL_DIR: ".pi/skills" + GH_AW_SKILL_DIR: ".github/skills" run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_inline_skills.sh" - env: EXPR_GITHUB_REPOSITORY: ${{ github.repository }} @@ -956,18 +955,20 @@ jobs: export GH_AW_PR_HEAD_REPO="${GH_AW_PR_HEAD_REPO:-}" export DEBUG="*" - export GH_AW_ENGINE="pi" + export GH_AW_ENGINE="copilot" export GH_AW_MCP_CLI_SERVERS='["safeoutputs"]' MCP_GATEWAY_UID=$(id -u 2>/dev/null || echo '0') MCP_GATEWAY_GID=$(id -g 2>/dev/null || echo '0') source "${RUNNER_TEMP}/gh-aw/actions/resolve_docker_socket_gid.sh" export MCP_GATEWAY_DOCKER_COMMAND='docker run -i --rm --network bridge -p 127.0.0.1:'"${MCP_GATEWAY_PORT}"':'"${MCP_GATEWAY_PORT}"' --name awmg-mcpg --add-host host.docker.internal:host-gateway --user '"${MCP_GATEWAY_UID}"':'"${MCP_GATEWAY_GID}"' --group-add '"${DOCKER_SOCK_GID}"' -v '"${DOCKER_SOCK_PATH}"':/var/run/docker.sock -e MCP_GATEWAY_PORT -e MCP_GATEWAY_DOMAIN -e MCP_GATEWAY_API_KEY -e MCP_GATEWAY_PAYLOAD_DIR -e MCP_GATEWAY_PAYLOAD_SIZE_THRESHOLD -e DOCKER_HOST=unix:///var/run/docker.sock -e DEBUG -e MCP_GATEWAY_LOG_DIR -e GH_AW_MCP_LOG_DIR -e GH_AW_SAFE_OUTPUTS -e GH_AW_SAFE_OUTPUTS_CONFIG_PATH -e GH_AW_SAFE_OUTPUTS_TOOLS_PATH -e GH_AW_PR_HEAD_BASE_BRANCH -e GH_AW_PR_HEAD_BASE_SHA -e GH_AW_PR_HEAD_BASE_REPO -e GH_AW_PR_HEAD_BASE_PR_NUMBER -e GH_AW_PR_HEAD_BASE_REF -e GH_AW_PR_HEAD_REPO -e GH_AW_POLICY_ALLOW_CREATE_PULL_REQUEST -e GH_AW_ASSETS_BRANCH -e GH_AW_ASSETS_MAX_SIZE_KB -e GH_AW_ASSETS_ALLOWED_EXTS -e DEFAULT_BRANCH -e GITHUB_MCP_SERVER_TOKEN -e GITHUB_MCP_GUARD_MIN_INTEGRITY -e GITHUB_MCP_GUARD_REPOS -e GH_AW_SINK_VISIBILITY -e GITHUB_REPOSITORY -e GITHUB_SERVER_URL -e GITHUB_SHA -e GITHUB_WORKSPACE -e GITHUB_TOKEN -e GITHUB_RUN_ID -e GITHUB_RUN_NUMBER -e GITHUB_RUN_ATTEMPT -e GITHUB_JOB -e GITHUB_ACTION -e GITHUB_EVENT_NAME -e GITHUB_EVENT_PATH -e GITHUB_ACTOR -e GITHUB_ACTOR_ID -e GITHUB_TRIGGERING_ACTOR -e GITHUB_WORKFLOW -e GITHUB_WORKFLOW_REF -e GITHUB_WORKFLOW_SHA -e GITHUB_REF -e GITHUB_REF_NAME -e GITHUB_REF_TYPE -e GITHUB_HEAD_REF -e GITHUB_BASE_REF -e RUNNER_TEMP -e RUNNER_TOOL_CACHE -e MCP_GATEWAY_ALLOWED_MOUNT_ROOTS -e GITHUB_AW_OTEL_TRACE_ID -e GITHUB_AW_OTEL_PARENT_SPAN_ID -e OTEL_EXPORTER_OTLP_HEADERS -v /tmp/gh-aw/mcp-payloads:/tmp/gh-aw/mcp-payloads:rw -v /opt:/opt:ro -v /tmp:/tmp:rw -v '"${GITHUB_WORKSPACE}"':'"${GITHUB_WORKSPACE}"':rw -v '"${RUNNER_TEMP}"'/gh-aw/safeoutputs:'"${RUNNER_TEMP}"'/gh-aw/safeoutputs:rw ghcr.io/github/gh-aw-mcpg:v0.4.13' + mkdir -p "$HOME/.copilot" GH_AW_NODE=$(which node 2>/dev/null || command -v node 2>/dev/null || echo node) - cat << GH_AW_MCP_CONFIG_59232f301f388d66_EOF | "$GH_AW_NODE" "${RUNNER_TEMP}/gh-aw/actions/start_mcp_gateway.cjs" + cat << GH_AW_MCP_CONFIG_905a02f8a48112ff_EOF | "$GH_AW_NODE" "${RUNNER_TEMP}/gh-aw/actions/start_mcp_gateway.cjs" { "mcpServers": { "safeoutputs": { + "type": "stdio", "container": "ghcr.io/github/gh-aw-node", "mounts": ["\${GITHUB_WORKSPACE}:\${GITHUB_WORKSPACE}:rw", "${RUNNER_TEMP}/gh-aw/safeoutputs:${RUNNER_TEMP}/gh-aw/safeoutputs:rw", "/tmp/gh-aw:/tmp/gh-aw:rw"], "args": ["-w", "\${GITHUB_WORKSPACE}"], @@ -1021,7 +1022,7 @@ jobs: } } } - GH_AW_MCP_CONFIG_59232f301f388d66_EOF + GH_AW_MCP_CONFIG_905a02f8a48112ff_EOF - name: Mount MCP servers as CLIs id: mount-mcp-clis continue-on-error: true @@ -1060,18 +1061,36 @@ jobs: CLI_PROXY_IMAGE: 'ghcr.io/github/gh-aw-mcpg:v0.4.13' run: | bash "${RUNNER_TEMP}/gh-aw/actions/start_cli_proxy.sh" - - name: Execute Pi CLI + - name: Execute GitHub Copilot CLI id: agentic_execution + # Copilot CLI tool arguments (sorted): timeout-minutes: 15 run: | set -o pipefail - trap 'gh_aw_exit_code=$?; mkdir -p /tmp/gh-aw >/dev/null 2>&1 || true; printf "%s" "$gh_aw_exit_code" > /tmp/gh-aw/agent_execution_exit_code.txt || true; if [ "$gh_aw_exit_code" -ne 0 ]; then echo "::error::Agent execution exited with code $gh_aw_exit_code"; fi' EXIT printf '%s' "$(date +%s%3N)" > /tmp/gh-aw/agent_cli_start_ms.txt + trap 'gh_aw_exit_code=$?; mkdir -p /tmp/gh-aw >/dev/null 2>&1 || true; printf "%s" "$gh_aw_exit_code" > /tmp/gh-aw/agent_execution_exit_code.txt || true; rm -f "$HOME/.copilot/settings.json"; if [ "$gh_aw_exit_code" -ne 0 ]; then echo "::error::Agent execution exited with code $gh_aw_exit_code"; fi' EXIT + mkdir -p "$HOME/.copilot" + printf '%s' '{"builtInAgents":{"rubberDuck":false}}' > "$HOME/.copilot/settings.json" + export XDG_CONFIG_HOME="$HOME" + export GH_AW_MCP_CONFIG="$HOME/.copilot/mcp-config.json" + GH_AW_COPILOT_SRC="$(command -v copilot 2>/dev/null || true)" + if [ -z "$GH_AW_COPILOT_SRC" ] || [ ! -x "$GH_AW_COPILOT_SRC" ]; then + echo "GitHub Copilot CLI executable not found on PATH after installation" >&2 + exit 127 + fi + GH_AW_COPILOT_BIN="${RUNNER_TEMP}/gh-aw/bin/copilot" + mkdir -p "${RUNNER_TEMP}/gh-aw/bin" + if [ "$GH_AW_COPILOT_SRC" != "$GH_AW_COPILOT_BIN" ]; then + cp "$GH_AW_COPILOT_SRC" "$GH_AW_COPILOT_BIN" + fi + chmod 755 "$GH_AW_COPILOT_BIN" + touch /tmp/gh-aw/agent-step-summary.md GH_AW_NODE_BIN=$(command -v node 2>/dev/null || true) export GH_AW_NODE_BIN + export COPILOT_API_KEY="$COPILOT_DUMMY_BYOK" (umask 177 && touch /tmp/gh-aw/agent-stdio.log) - GH_AW_MAX_AI_CREDITS="${{ vars.GH_AW_DEFAULT_MAX_AI_CREDITS || '1000' }}" + GH_AW_MAX_AI_CREDITS="${GH_AW_MAX_AI_CREDITS:-1000}" printf '%s\n' "{\"\$schema\":\"https://github.com/github/gh-aw-firewall/releases/download/v0.28.10/awf-config.schema.json\",\"network\":{\"allowDomains\":[\"*.grafana.net\",\"*.sentry.io\",\"api.snapcraft.io\",\"archive.ubuntu.com\",\"azure.archive.ubuntu.com\",\"crl.geotrust.com\",\"crl.globalsign.com\",\"crl.identrust.com\",\"crl.sectigo.com\",\"crl.thawte.com\",\"crl.usertrust.com\",\"crl.verisign.com\",\"crl3.digicert.com\",\"crl4.digicert.com\",\"crls.ssl.com\",\"go.dev\",\"golang.org\",\"goproxy.io\",\"json-schema.org\",\"json.schemastore.org\",\"keyserver.ubuntu.com\",\"ocsp.digicert.com\",\"ocsp.geotrust.com\",\"ocsp.globalsign.com\",\"ocsp.identrust.com\",\"ocsp.sectigo.com\",\"ocsp.ssl.com\",\"ocsp.thawte.com\",\"ocsp.usertrust.com\",\"ocsp.verisign.com\",\"packagecloud.io\",\"packages.cloud.google.com\",\"packages.microsoft.com\",\"pkg.go.dev\",\"ppa.launchpad.net\",\"proxy.golang.org\",\"s.symcb.com\",\"s.symcd.com\",\"security.ubuntu.com\",\"storage.googleapis.com\",\"sum.golang.org\",\"ts-crl.ws.symantec.com\",\"ts-ocsp.ws.symantec.com\",\"www.googleapis.com\"],\"isolation\":true,\"topologyAttach\":[\"awmg-mcpg\",\"awmg-cli-proxy\"]},\"apiProxy\":{\"enabled\":true,\"enableTokenSteering\":true,\"maxRuns\":500,\"maxAiCredits\":${GH_AW_MAX_AI_CREDITS},\"maxCacheMisses\":5,\"models\":{\"agent\":[\"sonnet-6x\",\"gpt-5.4\",\"gpt-5.5\",\"gpt-5.6\",\"gpt-5.3\",\"gemini-pro\",\"any\"],\"antigravity\":[\"copilot/antigravity*\",\"google/antigravity*\",\"gemini/antigravity*\"],\"any\":[\"copilot/*\",\"anthropic/*\",\"openai/*\",\"google/*\",\"gemini/*\"],\"auto\":[\"copilot/auto\",\"large\"],\"claude\":[\"agent\"],\"codex\":[\"agent\"],\"coding\":[\"copilot/gpt-5*codex*\",\"openai/gpt-5*codex*\",\"gpt-5-codex\",\"kimi\"],\"computer-use\":[\"copilot/*computer-use*\",\"google/*computer-use*\",\"gemini/*computer-use*\",\"openai/*computer-use*\"],\"copilot\":[\"agent\"],\"deep-research\":[\"copilot/deep-research*\",\"copilot/o3-deep-research*\",\"copilot/o4-mini-deep-research*\",\"google/deep-research*\",\"gemini/deep-research*\",\"openai/o3-deep-research*\",\"openai/o4-mini-deep-research*\"],\"detection\":[\"small\"],\"evals\":[\"small\"],\"fable\":[\"copilot/*fable*\",\"anthropic/*fable*\"],\"gemini\":[\"agent\"],\"gemini-3-flash\":[\"copilot/gemini-3*flash*\",\"google/gemini-3*flash*\",\"gemini/gemini-3*flash*\"],\"gemini-3-pro\":[\"copilot/gemini-3*pro*\",\"google/gemini-3*pro*\",\"google/nano-banana*\",\"gemini/gemini-3*pro*\"],\"gemini-3.1-flash\":[\"copilot/gemini-3.1*flash*\",\"google/gemini-3.1*flash*\",\"gemini/gemini-3.1*flash*\"],\"gemini-3.1-pro\":[\"copilot/gemini-3.1*pro*\",\"google/gemini-3.1*pro*\",\"gemini/gemini-3.1*pro*\"],\"gemini-3.5-flash\":[\"copilot/gemini-3.5*flash*\",\"google/gemini-3.5*flash*\",\"gemini/gemini-3.5*flash*\"],\"gemini-3.6-flash\":[\"copilot/gemini-3.6*flash*\",\"google/gemini-3.6*flash*\",\"gemini/gemini-3.6*flash*\"],\"gemini-3.7-flash\":[\"copilot/gemini-3.7*flash*\",\"google/gemini-3.7*flash*\",\"gemini/gemini-3.7*flash*\"],\"gemini-flash\":[\"copilot/gemini-*flash*\",\"google/gemini-*flash*\",\"gemini/gemini-*flash*\"],\"gemini-flash-lite\":[\"copilot/gemini-*flash*lite*\",\"google/gemini-*flash*lite*\",\"gemini/gemini-*flash*lite*\"],\"gemini-omni\":[\"copilot/gemini-omni*\",\"google/gemini-omni*\",\"gemini/gemini-omni*\"],\"gemini-pro\":[\"copilot/gemini-*pro*\",\"google/gemini-*pro*\",\"gemini/gemini-*pro*\"],\"gemma\":[\"copilot/gemma*\",\"google/gemma*\",\"gemini/gemma*\"],\"gpt-5\":[\"copilot/gpt-5*\",\"openai/gpt-5*\"],\"gpt-5-codex\":[\"copilot/gpt-5*codex*\",\"openai/gpt-5*codex*\"],\"gpt-5-mini\":[\"copilot/gpt-5*mini*\",\"openai/gpt-5*mini*\"],\"gpt-5-nano\":[\"copilot/gpt-5*nano*\",\"openai/gpt-5*nano*\"],\"gpt-5-pro\":[\"copilot/gpt-5*pro*\",\"openai/gpt-5*pro*\"],\"gpt-5.1\":[\"copilot/gpt-5.1*\",\"openai/gpt-5.1*\"],\"gpt-5.2\":[\"copilot/gpt-5.2*\",\"openai/gpt-5.2*\"],\"gpt-5.3\":[\"copilot/gpt-5.3*\",\"openai/gpt-5.3*\"],\"gpt-5.4\":[\"copilot/gpt-5.4*\",\"openai/gpt-5.4*\"],\"gpt-5.5\":[\"copilot/gpt-5.5*\",\"openai/gpt-5.5*\"],\"gpt-5.6\":[\"copilot/gpt-5.6*\",\"openai/gpt-5.6*\"],\"grok\":[\"copilot/*grok*\",\"openai/*grok*\"],\"haiku\":[\"copilot/*haiku*\",\"anthropic/*haiku*\"],\"image-generation\":[\"copilot/gpt-image*\",\"openai/gpt-image*\",\"openai/chatgpt-image*\",\"copilot/gemini-*image*\",\"google/gemini-*image*\",\"gemini/gemini-*image*\",\"google/imagen*\"],\"kimi\":[\"copilot/kimi*\",\"openai/kimi*\"],\"kiwi\":[\"copilot/kiwi*\",\"openai/kiwi*\"],\"large\":[\"sonnet\",\"gpt-5-pro\",\"gpt-5\",\"gemini-pro\"],\"lyria\":[\"google/lyria*\",\"gemini/lyria*\",\"copilot/lyria*\"],\"mai-code\":[\"copilot/MAI-Code*\",\"copilot/mai-code*\",\"openai/MAI-Code*\"],\"mai-code-1-flash-picker\":[\"copilot/MAI-Code-1-Flash-picker*\",\"copilot/mai-code-1-flash-picker*\",\"openai/MAI-Code-1-Flash-picker*\"],\"mini\":[\"haiku\",\"gpt-5-mini\",\"gpt-5-nano\",\"gemini-flash-lite\"],\"nano-banana\":[\"copilot/nano-banana*\",\"google/nano-banana*\",\"gemini/nano-banana*\"],\"opus\":[\"copilot/*opus*\",\"anthropic/*opus*\"],\"opusplan\":[\"opus?effort=high\"],\"raptor-mini\":[\"copilot/raptor*\",\"openai/raptor*\"],\"reasoning\":[\"copilot/o1*\",\"copilot/o3*\",\"copilot/o4*\",\"openai/o1*\",\"openai/o3*\",\"openai/o4*\"],\"robotics\":[\"copilot/*robotics*\",\"google/*robotics*\",\"gemini/*robotics*\"],\"small\":[\"mini\"],\"small-agent\":[\"haiku\",\"gpt-5-mini\",\"gemini-flash\"],\"sonnet\":[\"copilot/*sonnet*\",\"anthropic/*sonnet*\"],\"sonnet-6x\":[\"copilot/*sonnet-4.5*\",\"copilot/*sonnet-4.6*\",\"copilot/*sonnet-5*\",\"copilot/*sonnet-4-5-*\",\"anthropic/*sonnet-4-5-*\",\"copilot/*sonnet-4-6*\",\"anthropic/*sonnet-4-6*\",\"anthropic/*sonnet-5*\"],\"summarization\":[\"haiku\",\"gpt-5-mini\",\"gemini-flash-lite\",\"mini\"],\"veo\":[\"google/veo*\",\"gemini/veo*\"],\"vision\":[\"copilot/gemini-*image*\",\"google/gemini-*image*\",\"gemini/gemini-*image*\",\"copilot/gemini-*flash*\",\"google/gemini-*flash*\",\"gemini/gemini-*flash*\"]}},\"container\":{\"imageTag\":\"0.28.10,squid=sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6,agent=sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e,api-proxy=sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64,cli-proxy=sha256:a61070cb7f21840c5f2ec74d55b49adf0652d0348ce059015aaaca33a8cb6b45\"},\"logging\":{\"proxyLogsDir\":\"/tmp/gh-aw/sandbox/firewall/logs\",\"auditDir\":\"/tmp/gh-aw/sandbox/firewall/audit\"}}" > "${RUNNER_TEMP}/gh-aw/awf-config.json" cp "${RUNNER_TEMP}/gh-aw/awf-config.json" /tmp/gh-aw/awf-config.json export GH_AW_MODELS_JSON_PATH="/tmp/gh-aw/models.json" @@ -1090,46 +1109,74 @@ jobs: fi fi # shellcheck disable=SC1003,SC2016,SC2086 - GH_AW_AWF_ENGINE_NAME=pi \ - GH_AW_AWF_HARNESS_MARKER='[pi-harness]' \ + GH_AW_AWF_ENGINE_NAME=copilot \ + GH_AW_AWF_HARNESS_MARKER='[copilot-harness]' \ GH_AW_AWF_LOG_FILE=/tmp/gh-aw/agent-stdio.log \ - GH_AW_AWF_ATTEMPT_LOG_NAME=pi \ + GH_AW_AWF_ATTEMPT_LOG_NAME=copilot \ bash "${RUNNER_TEMP}/gh-aw/actions/run_awf_with_startup_retries.sh" -- \ - awf --config "${RUNNER_TEMP}/gh-aw/awf-config.json" --container-workdir "${GITHUB_WORKSPACE}" --mount "${RUNNER_TEMP}/gh-aw:${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw:/host${RUNNER_TEMP}/gh-aw:ro" ${GH_AW_TOOL_CACHE_MOUNT:+--mount "$GH_AW_TOOL_CACHE_MOUNT"} ${GH_AW_DOCKER_HOST:+--docker-host "$GH_AW_DOCKER_HOST"} --env-all --exclude-env ACTIONS_ID_TOKEN_REQUEST_TOKEN --exclude-env ACTIONS_ID_TOKEN_REQUEST_URL --exclude-env CODEX_API_KEY --exclude-env GH_TOKEN --exclude-env GITHUB_MCP_SERVER_TOKEN --exclude-env MCP_GATEWAY_API_KEY --exclude-env OPENAI_API_KEY --mount /tmp/gh-aw:/tmp/gh-aw:rw --log-level info --skip-pull --difc-proxy-host awmg-cli-proxy:18443 --difc-proxy-ca-cert /tmp/gh-aw/difc-proxy-tls/ca.crt \ - -- /bin/bash -c 'set +o histexpand; GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" ${RUNNER_TEMP}/gh-aw/actions/shell_harness.cjs pi '\''export PATH="${RUNNER_TEMP}/gh-aw/mcp-cli/bin:$PATH" && : "${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}"; GH_AW_TOOL_CACHE="$RUNNER_TOOL_CACHE"; export PATH="$(find "$GH_AW_TOOL_CACHE" -maxdepth 5 -type d -name bin 2>/dev/null | tr '\''\'\'''\''\n'\''\'\'''\'' '\''\'\'''\'':'\''\'\'''\'')$PATH"; [ -n "$GOROOT" ] && export PATH="$GOROOT/bin:$PATH" || true; [ -n "$ERLANG_HOME" ] && export PATH="$ERLANG_HOME/bin:$PATH" || true && cd "${GITHUB_WORKSPACE}" && export GH_AW_PI_MODEL_ID=gpt-5.4 GH_AW_PI_GATEWAY_SECRET_ENV=CODEX_API_KEY GH_AW_PI_GATEWAY_FALLBACK_PORT=10000 GH_AW_LLM_PROVIDER=openai && ( GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" "${RUNNER_TEMP}/gh-aw/actions/pi_models_json.cjs" ) && cat /tmp/gh-aw/aw-prompts/prompt.txt | pi --print --mode json --no-session --model aw-gateway/gpt-5.4 --extension "${RUNNER_TEMP}/gh-aw/actions/pi_provider.cjs" --extension "${RUNNER_TEMP}/gh-aw/actions/pi_steering_extension.cjs" 2>&1 | tee /tmp/gh-aw/pi-streaming.jsonl'\''' + awf --config "${RUNNER_TEMP}/gh-aw/awf-config.json" --container-workdir "${GITHUB_WORKSPACE}" --mount "${RUNNER_TEMP}/gh-aw:${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw:/host${RUNNER_TEMP}/gh-aw:ro" ${GH_AW_TOOL_CACHE_MOUNT:+--mount "$GH_AW_TOOL_CACHE_MOUNT"} ${GH_AW_DOCKER_HOST:+--docker-host "$GH_AW_DOCKER_HOST"} --env-all --exclude-env ACTIONS_ID_TOKEN_REQUEST_TOKEN --exclude-env ACTIONS_ID_TOKEN_REQUEST_URL --exclude-env COPILOT_GITHUB_TOKEN --exclude-env GH_TOKEN --exclude-env GITHUB_MCP_SERVER_TOKEN --exclude-env MCP_GATEWAY_API_KEY --mount /tmp/gh-aw:/tmp/gh-aw:rw --log-level info --skip-pull --difc-proxy-host awmg-cli-proxy:18443 --difc-proxy-ca-cert /tmp/gh-aw/difc-proxy-tls/ca.crt \ + -- /bin/bash -c 'set +o histexpand; export PATH="${RUNNER_TEMP}/gh-aw/mcp-cli/bin:$PATH" && : "${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}"; GH_AW_TOOL_CACHE="$RUNNER_TOOL_CACHE"; export PATH="$(find "$GH_AW_TOOL_CACHE" -maxdepth 5 -type d -name bin 2>/dev/null | tr '\''\n'\'' '\'':'\'')$PATH"; [ -n "$GOROOT" ] && export PATH="$GOROOT/bin:$PATH" || true; [ -n "$ERLANG_HOME" ] && export PATH="$ERLANG_HOME/bin:$PATH" || true && GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" "${RUNNER_TEMP}/gh-aw/actions/copilot_harness.cjs" "${RUNNER_TEMP}/gh-aw/bin/copilot" --add-dir /tmp/gh-aw/ --log-level all --log-dir /tmp/gh-aw/sandbox/agent/logs/ --disable-builtin-mcps --no-ask-user --allow-all-tools --allow-all-paths --add-dir "${GITHUB_WORKSPACE}" --prompt-file /tmp/gh-aw/aw-prompts/prompt.txt' env: AWF_REFLECT_ENABLED: 1 - CODEX_API_KEY: ${{ secrets.CODEX_API_KEY || secrets.OPENAI_API_KEY }} + COPILOT_AGENT_RUNNER_TYPE: STANDALONE + COPILOT_DUMMY_BYOK: dummy-byok-key-for-offline-mode + COPILOT_GITHUB_TOKEN: ${{ github.token }} + COPILOT_MODEL: copilot/gpt-5.4 + GH_AW_LLM_PROVIDER: github + GH_AW_MAX_AI_CREDITS: ${{ vars.GH_AW_DEFAULT_MAX_AI_CREDITS || '1000' }} GH_AW_MAX_TURNS: ${{ vars.GH_AW_DEFAULT_MAX_TURNS || '' }} GH_AW_PHASE: agent - GH_AW_PI_MODEL: openai/gpt-5.4 GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} GH_AW_TIMEOUT_MINUTES: 15 GH_AW_VERSION: dev GH_TOKEN: ${{ secrets.GH_AW_GITHUB_TOKEN || github.token }} + GITHUB_API_URL: ${{ github.api_url }} GITHUB_AW: true + GITHUB_COPILOT_INTEGRATION_ID: agentic-workflows + GITHUB_HEAD_REF: ${{ github.head_ref }} + GITHUB_MCP_SERVER_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN || secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + GITHUB_REF_NAME: ${{ github.ref_name }} + GITHUB_SERVER_URL: ${{ github.server_url }} GITHUB_STEP_SUMMARY: /tmp/gh-aw/agent-step-summary.md GITHUB_WORKSPACE: ${{ github.workspace }} GIT_AUTHOR_EMAIL: github-actions[bot]@users.noreply.github.com GIT_AUTHOR_NAME: github-actions[bot] GIT_COMMITTER_EMAIL: github-actions[bot]@users.noreply.github.com GIT_COMMITTER_NAME: github-actions[bot] - OPENAI_API_KEY: ${{ secrets.CODEX_API_KEY || secrets.OPENAI_API_KEY }} - PI_CODING_AGENT_DIR: /tmp/gh-aw/pi-agent-dir - PI_OFFLINE: 1 RUNNER_TEMP: ${{ runner.temp }} + S2STOKENS: true TRACEPARENT: ${{ env.GITHUB_AW_OTEL_TRACE_ID != '' && env.GITHUB_AW_OTEL_PARENT_SPAN_ID != '' && format('00-{0}-{1}-01', env.GITHUB_AW_OTEL_TRACE_ID, env.GITHUB_AW_OTEL_PARENT_SPAN_ID) || '' }} - name: Stop CLI Proxy if: always() continue-on-error: true run: bash "${RUNNER_TEMP}/gh-aw/actions/stop_cli_proxy.sh" + - name: Detect agent errors + if: always() + id: detect-agent-errors + continue-on-error: true + env: + GH_AW_AGENTIC_EXECUTION_OUTCOME: ${{ steps.agentic_execution.outcome }} + GH_AW_ENGINE_STEP_TIMEOUT_MINUTES: 15 + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const path = require('path'); + const actionsDir = path.join(process.env.RUNNER_TEMP, 'gh-aw', 'actions'); + const { setupGlobals } = require(path.join(actionsDir, 'setup_globals.cjs')); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require(path.join(actionsDir, 'detect_agent_errors.cjs')); + await main(); - name: Configure Git credentials env: GITHUB_REPOSITORY: ${{ github.repository }} GITHUB_SERVER_URL: ${{ github.server_url }} GITHUB_TOKEN: ${{ github.token }} run: bash "${RUNNER_TEMP}/gh-aw/actions/configure_git_credentials.sh" + - name: Copy Copilot session state files to logs + if: always() + continue-on-error: true + run: bash "${RUNNER_TEMP}/gh-aw/actions/copy_copilot_session_state.sh" - name: Stop MCP Gateway if: always() continue-on-error: true @@ -1151,12 +1198,10 @@ jobs: const { main } = require(path.join(actionsDir, 'redact_secrets.cjs')); await main(); env: - GH_AW_SECRET_NAMES: 'CODEX_API_KEY,GH_AW_GITHUB_MCP_SERVER_TOKEN,GH_AW_GITHUB_TOKEN,GITHUB_TOKEN,OPENAI_API_KEY' - SECRET_CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} + GH_AW_SECRET_NAMES: 'GH_AW_GITHUB_MCP_SERVER_TOKEN,GH_AW_GITHUB_TOKEN,GITHUB_TOKEN' SECRET_GH_AW_GITHUB_MCP_SERVER_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN }} SECRET_GH_AW_GITHUB_TOKEN: ${{ secrets.GH_AW_GITHUB_TOKEN }} SECRET_GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - SECRET_OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - name: Append agent step summary if: always() run: bash "${RUNNER_TEMP}/gh-aw/actions/append_agent_step_summary.sh" @@ -1189,7 +1234,7 @@ jobs: if: always() uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: - GH_AW_AGENT_OUTPUT: /tmp/gh-aw/pi-streaming.jsonl + GH_AW_AGENT_OUTPUT: /tmp/gh-aw/sandbox/agent/logs/ GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} with: script: | @@ -1197,7 +1242,7 @@ jobs: const actionsDir = path.join(process.env.RUNNER_TEMP, 'gh-aw', 'actions'); const { setupGlobals } = require(path.join(actionsDir, 'setup_globals.cjs')); setupGlobals(core, github, context, exec, io, getOctokit); - const { main } = require(path.join(actionsDir, 'parse_pi_log.cjs')); + const { main } = require(path.join(actionsDir, 'parse_copilot_log.cjs')); await main(); - name: Parse MCP Gateway logs for step summary if: always() @@ -1278,7 +1323,7 @@ jobs: name: agent path: | /tmp/gh-aw/aw-prompts/prompt.txt - /tmp/gh-aw/pi-streaming.jsonl + /tmp/gh-aw/sandbox/agent/logs/ /tmp/gh-aw/redacted-urls.log /tmp/gh-aw/mcp-logs/ /tmp/gh-aw/proxy-logs/ @@ -1312,7 +1357,7 @@ jobs: if: > always() && (needs.agent.result != 'skipped' || needs.activation.outputs.lockdown_check_failed == 'true' || needs.activation.outputs.oauth_token_check_failed == 'true' || needs.activation.outputs.stale_lock_file_failed == 'true' || - needs.activation.outputs.secret_verification_result == 'failed' || needs.activation.outputs.daily_ai_credits_exceeded == 'true') + needs.activation.outputs.daily_ai_credits_exceeded == 'true') runs-on: ubuntu-slim permissions: actions: read @@ -1350,9 +1395,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Download agent output artifact id: download-agent-output continue-on-error: true @@ -1550,8 +1595,7 @@ jobs: GH_AW_AGENT_CONCLUSION: ${{ needs.agent.result }} GH_AW_WORKFLOW_ID: "pr-code-quality-reviewer" GH_AW_ACTION_FAILURE_ISSUE_EXPIRES_HOURS: "12" - GH_AW_ENGINE_ID: "pi" - GH_AW_SECRET_VERIFICATION_RESULT: ${{ needs.activation.outputs.secret_verification_result }} + GH_AW_ENGINE_ID: "copilot" GH_AW_CHECKOUT_PR_SUCCESS: ${{ needs.agent.outputs.checkout_pr_success }} GH_AW_EFFECTIVE_TOKENS: ${{ needs.agent.outputs.effective_tokens || '' }} GH_AW_AI_CREDITS_RATE_LIMIT_ERROR: ${{ needs.agent.outputs.ai_credits_rate_limit_error || 'false' }} @@ -1560,6 +1604,16 @@ jobs: GH_AW_THREAT_DETECTION_AIC: ${{ needs.detection.outputs.aic }} GH_AW_EVALS_AIC: ${{ needs.evals.outputs.aic }} GH_AW_MAX_AI_CREDITS: ${{ vars.GH_AW_DEFAULT_MAX_AI_CREDITS || '1000' }} + GH_AW_INFERENCE_ACCESS_ERROR: ${{ needs.agent.outputs.inference_access_error }} + GH_AW_MCP_POLICY_ERROR: ${{ needs.agent.outputs.mcp_policy_error }} + GH_AW_AGENTIC_ENGINE_TIMEOUT: ${{ needs.agent.outputs.agentic_engine_timeout }} + GH_AW_MODEL_NOT_SUPPORTED_ERROR: ${{ needs.agent.outputs.model_not_supported_error }} + GH_AW_HTTP_400_RESPONSE_ERROR: ${{ needs.agent.outputs.http_400_response_error }} + GH_AW_MAX_CACHE_MISSES_EXCEEDED: ${{ needs.agent.outputs.max_cache_misses_exceeded }} + GH_AW_MISSING_MODEL_PRICING_ERROR: ${{ needs.agent.outputs.missing_model_pricing_error }} + GH_AW_MISSING_MODEL_PRICING_MODEL_NAME: ${{ needs.agent.outputs.missing_model_pricing_model_name }} + GH_AW_SHELL_EXPANSION_GUARD_REJECTED: ${{ needs.agent.outputs.shell_expansion_guard_rejected }} + GH_AW_ENGINE_API_HOSTS: "api.enterprise.githubcopilot.com,api.githubcopilot.com,api.business.githubcopilot.com,api.individual.githubcopilot.com" GH_AW_LOCKDOWN_CHECK_FAILED: ${{ needs.activation.outputs.lockdown_check_failed }} GH_AW_OAUTH_TOKEN_CHECK_FAILED: ${{ needs.activation.outputs.oauth_token_check_failed }} GH_AW_STALE_LOCK_FILE_FAILED: ${{ needs.activation.outputs.stale_lock_file_failed }} @@ -1661,9 +1715,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Download activation artifact continue-on-error: true uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 @@ -1767,7 +1821,7 @@ jobs: COPILOT_AGENT_RUNNER_TYPE: STANDALONE COPILOT_DUMMY_BYOK: dummy-byok-key-for-offline-mode COPILOT_GITHUB_TOKEN: ${{ github.token }} - COPILOT_MODEL: gpt-5.4 + COPILOT_MODEL: copilot/gpt-5.4 GH_AW_HARNESS_MAX_RETRIES: 0 GH_AW_LLM_PROVIDER: github GH_AW_MAX_AI_CREDITS: ${{ vars.GH_AW_DEFAULT_DETECTION_MAX_AI_CREDITS || '400' }} @@ -1922,9 +1976,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Download agent output artifact id: download-agent-output continue-on-error: true @@ -1960,7 +2014,7 @@ jobs: uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: GH_AW_EVALS_QUESTIONS: '[{"id":"review_posted","question":"Did the agent post a code review comment on the pull request?"},{"id":"findings_scoped","question":"Does the agent output show that the review findings are limited to changes in the pull request diff rather than unrelated code?"}]' - GH_AW_EVALS_MODEL: "openai/gpt-5.4" + GH_AW_EVALS_MODEL: "copilot/gpt-5.4" GH_AW_EVALS_PHASE: setup with: script: | @@ -1979,24 +2033,44 @@ jobs: with: node-version: '24' package-manager-cache: false + - name: Install GitHub Copilot CLI + run: bash "${RUNNER_TEMP}/gh-aw/actions/install_copilot_cli.sh" + env: + GH_HOST: github.com + GH_AW_COMPILED_VERSION: dev - name: Install AWF binary run: bash "${RUNNER_TEMP}/gh-aw/actions/install_awf_binary.sh" v0.28.10 --rootless - - name: Install Pi CLI - run: npm install --ignore-scripts -g @earendil-works/pi-coding-agent@0.84.3 - - name: Execute Pi CLI + - name: Execute GitHub Copilot CLI if: always() continue-on-error: true id: evals_agentic_execution + # Copilot CLI tool arguments (sorted): timeout-minutes: 15 run: | set -o pipefail - trap 'gh_aw_exit_code=$?; mkdir -p /tmp/gh-aw >/dev/null 2>&1 || true; printf "%s" "$gh_aw_exit_code" > /tmp/gh-aw/agent_execution_exit_code.txt || true; if [ "$gh_aw_exit_code" -ne 0 ]; then echo "::error::Agent execution exited with code $gh_aw_exit_code"; fi' EXIT printf '%s' "$(date +%s%3N)" > /tmp/gh-aw/agent_cli_start_ms.txt + trap 'gh_aw_exit_code=$?; mkdir -p /tmp/gh-aw >/dev/null 2>&1 || true; printf "%s" "$gh_aw_exit_code" > /tmp/gh-aw/agent_execution_exit_code.txt || true; rm -f "$HOME/.copilot/settings.json"; if [ "$gh_aw_exit_code" -ne 0 ]; then echo "::error::Agent execution exited with code $gh_aw_exit_code"; fi' EXIT + mkdir -p "$HOME/.copilot" + printf '%s' '{"builtInAgents":{"rubberDuck":false}}' > "$HOME/.copilot/settings.json" + export XDG_CONFIG_HOME="$HOME" + GH_AW_COPILOT_SRC="$(command -v copilot 2>/dev/null || true)" + if [ -z "$GH_AW_COPILOT_SRC" ] || [ ! -x "$GH_AW_COPILOT_SRC" ]; then + echo "GitHub Copilot CLI executable not found on PATH after installation" >&2 + exit 127 + fi + GH_AW_COPILOT_BIN="${RUNNER_TEMP}/gh-aw/bin/copilot" + mkdir -p "${RUNNER_TEMP}/gh-aw/bin" + if [ "$GH_AW_COPILOT_SRC" != "$GH_AW_COPILOT_BIN" ]; then + cp "$GH_AW_COPILOT_SRC" "$GH_AW_COPILOT_BIN" + fi + chmod 755 "$GH_AW_COPILOT_BIN" + touch /tmp/gh-aw/agent-step-summary.md GH_AW_NODE_BIN=$(command -v node 2>/dev/null || true) export GH_AW_NODE_BIN + export COPILOT_API_KEY="$COPILOT_DUMMY_BYOK" (umask 177 && touch /tmp/gh-aw/evals/evals.log) - GH_AW_MAX_AI_CREDITS="${{ vars.GH_AW_DEFAULT_EVALS_MAX_AI_CREDITS || '400' }}" + GH_AW_MAX_AI_CREDITS="${GH_AW_MAX_AI_CREDITS:-400}" printf '%s\n' "{\"\$schema\":\"https://github.com/github/gh-aw-firewall/releases/download/v0.28.10/awf-config.schema.json\",\"apiProxy\":{\"enabled\":true,\"enableTokenSteering\":true,\"maxRuns\":500,\"maxAiCredits\":${GH_AW_MAX_AI_CREDITS},\"maxCacheMisses\":5,\"models\":{\"agent\":[\"sonnet-6x\",\"gpt-5.4\",\"gpt-5.5\",\"gpt-5.6\",\"gpt-5.3\",\"gemini-pro\",\"any\"],\"antigravity\":[\"copilot/antigravity*\",\"google/antigravity*\",\"gemini/antigravity*\"],\"any\":[\"copilot/*\",\"anthropic/*\",\"openai/*\",\"google/*\",\"gemini/*\"],\"auto\":[\"copilot/auto\",\"large\"],\"claude\":[\"agent\"],\"codex\":[\"agent\"],\"coding\":[\"copilot/gpt-5*codex*\",\"openai/gpt-5*codex*\",\"gpt-5-codex\",\"kimi\"],\"computer-use\":[\"copilot/*computer-use*\",\"google/*computer-use*\",\"gemini/*computer-use*\",\"openai/*computer-use*\"],\"copilot\":[\"agent\"],\"deep-research\":[\"copilot/deep-research*\",\"copilot/o3-deep-research*\",\"copilot/o4-mini-deep-research*\",\"google/deep-research*\",\"gemini/deep-research*\",\"openai/o3-deep-research*\",\"openai/o4-mini-deep-research*\"],\"detection\":[\"small\"],\"evals\":[\"small\"],\"fable\":[\"copilot/*fable*\",\"anthropic/*fable*\"],\"gemini\":[\"agent\"],\"gemini-3-flash\":[\"copilot/gemini-3*flash*\",\"google/gemini-3*flash*\",\"gemini/gemini-3*flash*\"],\"gemini-3-pro\":[\"copilot/gemini-3*pro*\",\"google/gemini-3*pro*\",\"google/nano-banana*\",\"gemini/gemini-3*pro*\"],\"gemini-3.1-flash\":[\"copilot/gemini-3.1*flash*\",\"google/gemini-3.1*flash*\",\"gemini/gemini-3.1*flash*\"],\"gemini-3.1-pro\":[\"copilot/gemini-3.1*pro*\",\"google/gemini-3.1*pro*\",\"gemini/gemini-3.1*pro*\"],\"gemini-3.5-flash\":[\"copilot/gemini-3.5*flash*\",\"google/gemini-3.5*flash*\",\"gemini/gemini-3.5*flash*\"],\"gemini-3.6-flash\":[\"copilot/gemini-3.6*flash*\",\"google/gemini-3.6*flash*\",\"gemini/gemini-3.6*flash*\"],\"gemini-3.7-flash\":[\"copilot/gemini-3.7*flash*\",\"google/gemini-3.7*flash*\",\"gemini/gemini-3.7*flash*\"],\"gemini-flash\":[\"copilot/gemini-*flash*\",\"google/gemini-*flash*\",\"gemini/gemini-*flash*\"],\"gemini-flash-lite\":[\"copilot/gemini-*flash*lite*\",\"google/gemini-*flash*lite*\",\"gemini/gemini-*flash*lite*\"],\"gemini-omni\":[\"copilot/gemini-omni*\",\"google/gemini-omni*\",\"gemini/gemini-omni*\"],\"gemini-pro\":[\"copilot/gemini-*pro*\",\"google/gemini-*pro*\",\"gemini/gemini-*pro*\"],\"gemma\":[\"copilot/gemma*\",\"google/gemma*\",\"gemini/gemma*\"],\"gpt-5\":[\"copilot/gpt-5*\",\"openai/gpt-5*\"],\"gpt-5-codex\":[\"copilot/gpt-5*codex*\",\"openai/gpt-5*codex*\"],\"gpt-5-mini\":[\"copilot/gpt-5*mini*\",\"openai/gpt-5*mini*\"],\"gpt-5-nano\":[\"copilot/gpt-5*nano*\",\"openai/gpt-5*nano*\"],\"gpt-5-pro\":[\"copilot/gpt-5*pro*\",\"openai/gpt-5*pro*\"],\"gpt-5.1\":[\"copilot/gpt-5.1*\",\"openai/gpt-5.1*\"],\"gpt-5.2\":[\"copilot/gpt-5.2*\",\"openai/gpt-5.2*\"],\"gpt-5.3\":[\"copilot/gpt-5.3*\",\"openai/gpt-5.3*\"],\"gpt-5.4\":[\"copilot/gpt-5.4*\",\"openai/gpt-5.4*\"],\"gpt-5.5\":[\"copilot/gpt-5.5*\",\"openai/gpt-5.5*\"],\"gpt-5.6\":[\"copilot/gpt-5.6*\",\"openai/gpt-5.6*\"],\"grok\":[\"copilot/*grok*\",\"openai/*grok*\"],\"haiku\":[\"copilot/*haiku*\",\"anthropic/*haiku*\"],\"image-generation\":[\"copilot/gpt-image*\",\"openai/gpt-image*\",\"openai/chatgpt-image*\",\"copilot/gemini-*image*\",\"google/gemini-*image*\",\"gemini/gemini-*image*\",\"google/imagen*\"],\"kimi\":[\"copilot/kimi*\",\"openai/kimi*\"],\"kiwi\":[\"copilot/kiwi*\",\"openai/kiwi*\"],\"large\":[\"sonnet\",\"gpt-5-pro\",\"gpt-5\",\"gemini-pro\"],\"lyria\":[\"google/lyria*\",\"gemini/lyria*\",\"copilot/lyria*\"],\"mai-code\":[\"copilot/MAI-Code*\",\"copilot/mai-code*\",\"openai/MAI-Code*\"],\"mai-code-1-flash-picker\":[\"copilot/MAI-Code-1-Flash-picker*\",\"copilot/mai-code-1-flash-picker*\",\"openai/MAI-Code-1-Flash-picker*\"],\"mini\":[\"haiku\",\"gpt-5-mini\",\"gpt-5-nano\",\"gemini-flash-lite\"],\"nano-banana\":[\"copilot/nano-banana*\",\"google/nano-banana*\",\"gemini/nano-banana*\"],\"opus\":[\"copilot/*opus*\",\"anthropic/*opus*\"],\"opusplan\":[\"opus?effort=high\"],\"raptor-mini\":[\"copilot/raptor*\",\"openai/raptor*\"],\"reasoning\":[\"copilot/o1*\",\"copilot/o3*\",\"copilot/o4*\",\"openai/o1*\",\"openai/o3*\",\"openai/o4*\"],\"robotics\":[\"copilot/*robotics*\",\"google/*robotics*\",\"gemini/*robotics*\"],\"small\":[\"mini\"],\"small-agent\":[\"haiku\",\"gpt-5-mini\",\"gemini-flash\"],\"sonnet\":[\"copilot/*sonnet*\",\"anthropic/*sonnet*\"],\"sonnet-6x\":[\"copilot/*sonnet-4.5*\",\"copilot/*sonnet-4.6*\",\"copilot/*sonnet-5*\",\"copilot/*sonnet-4-5-*\",\"anthropic/*sonnet-4-5-*\",\"copilot/*sonnet-4-6*\",\"anthropic/*sonnet-4-6*\",\"anthropic/*sonnet-5*\"],\"summarization\":[\"haiku\",\"gpt-5-mini\",\"gemini-flash-lite\",\"mini\"],\"veo\":[\"google/veo*\",\"gemini/veo*\"],\"vision\":[\"copilot/gemini-*image*\",\"google/gemini-*image*\",\"gemini/gemini-*image*\",\"copilot/gemini-*flash*\",\"google/gemini-*flash*\",\"gemini/gemini-*flash*\"]}},\"container\":{\"imageTag\":\"0.28.10,squid=sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6,agent=sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e,api-proxy=sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64,cli-proxy=sha256:a61070cb7f21840c5f2ec74d55b49adf0652d0348ce059015aaaca33a8cb6b45\"},\"logging\":{\"proxyLogsDir\":\"/tmp/gh-aw/sandbox/firewall/logs\",\"auditDir\":\"/tmp/gh-aw/sandbox/firewall/audit\"}}" > "${RUNNER_TEMP}/gh-aw/awf-config.json" cp "${RUNNER_TEMP}/gh-aw/awf-config.json" /tmp/gh-aw/awf-config.json export GH_AW_MODELS_JSON_PATH="/tmp/gh-aw/models.json" @@ -2015,33 +2089,40 @@ jobs: fi fi # shellcheck disable=SC1003,SC2016,SC2086 - GH_AW_AWF_ENGINE_NAME=pi \ - GH_AW_AWF_HARNESS_MARKER='[pi-harness]' \ + GH_AW_AWF_ENGINE_NAME=copilot \ + GH_AW_AWF_HARNESS_MARKER='[copilot-harness]' \ GH_AW_AWF_LOG_FILE=/tmp/gh-aw/evals/evals.log \ - GH_AW_AWF_ATTEMPT_LOG_NAME=pi \ + GH_AW_AWF_ATTEMPT_LOG_NAME=copilot \ bash "${RUNNER_TEMP}/gh-aw/actions/run_awf_with_startup_retries.sh" -- \ - awf --config "${RUNNER_TEMP}/gh-aw/awf-config.json" --container-workdir "${GITHUB_WORKSPACE}" --mount "${RUNNER_TEMP}/gh-aw:${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw:/host${RUNNER_TEMP}/gh-aw:ro" ${GH_AW_TOOL_CACHE_MOUNT:+--mount "$GH_AW_TOOL_CACHE_MOUNT"} ${GH_AW_DOCKER_HOST:+--docker-host "$GH_AW_DOCKER_HOST"} --env-all --exclude-env ACTIONS_ID_TOKEN_REQUEST_TOKEN --exclude-env ACTIONS_ID_TOKEN_REQUEST_URL --exclude-env CODEX_API_KEY --exclude-env OPENAI_API_KEY --mount /tmp/gh-aw:/tmp/gh-aw:rw --log-level info --skip-pull \ - -- /bin/bash -c 'set +o histexpand; GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" ${RUNNER_TEMP}/gh-aw/actions/shell_harness.cjs pi '\'': "${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}"; GH_AW_TOOL_CACHE="$RUNNER_TOOL_CACHE"; export PATH="$(find "$GH_AW_TOOL_CACHE" -maxdepth 5 -type d -name bin 2>/dev/null | tr '\''\'\'''\''\n'\''\'\'''\'' '\''\'\'''\'':'\''\'\'''\'')$PATH"; [ -n "$GOROOT" ] && export PATH="$GOROOT/bin:$PATH" || true; [ -n "$ERLANG_HOME" ] && export PATH="$ERLANG_HOME/bin:$PATH" || true && cd "${GITHUB_WORKSPACE}" && export GH_AW_PI_MODEL_ID=gpt-5.4 GH_AW_PI_GATEWAY_SECRET_ENV=CODEX_API_KEY GH_AW_PI_GATEWAY_FALLBACK_PORT=10000 GH_AW_LLM_PROVIDER=openai && ( GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" "${RUNNER_TEMP}/gh-aw/actions/pi_models_json.cjs" ) && cat /tmp/gh-aw/aw-prompts/prompt.txt | pi --print --mode json --no-session --model aw-gateway/gpt-5.4 --extension "${RUNNER_TEMP}/gh-aw/actions/pi_provider.cjs" --extension "${RUNNER_TEMP}/gh-aw/actions/pi_steering_extension.cjs" 2>&1 | tee /tmp/gh-aw/pi-streaming.jsonl'\''' + awf --config "${RUNNER_TEMP}/gh-aw/awf-config.json" --container-workdir "${GITHUB_WORKSPACE}" --mount "${RUNNER_TEMP}/gh-aw:${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw:/host${RUNNER_TEMP}/gh-aw:ro" ${GH_AW_TOOL_CACHE_MOUNT:+--mount "$GH_AW_TOOL_CACHE_MOUNT"} ${GH_AW_DOCKER_HOST:+--docker-host "$GH_AW_DOCKER_HOST"} --env-all --exclude-env ACTIONS_ID_TOKEN_REQUEST_TOKEN --exclude-env ACTIONS_ID_TOKEN_REQUEST_URL --exclude-env COPILOT_GITHUB_TOKEN --mount /tmp/gh-aw:/tmp/gh-aw:rw --log-level info --skip-pull \ + -- /bin/bash -c 'set +o histexpand; : "${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}"; GH_AW_TOOL_CACHE="$RUNNER_TOOL_CACHE"; export PATH="$(find "$GH_AW_TOOL_CACHE" -maxdepth 5 -type d -name bin 2>/dev/null | tr '\''\n'\'' '\'':'\'')$PATH"; [ -n "$GOROOT" ] && export PATH="$GOROOT/bin:$PATH" || true; [ -n "$ERLANG_HOME" ] && export PATH="$ERLANG_HOME/bin:$PATH" || true && GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" "${RUNNER_TEMP}/gh-aw/actions/copilot_harness.cjs" "${RUNNER_TEMP}/gh-aw/bin/copilot" --add-dir /tmp/gh-aw/ --log-level all --log-dir /tmp/gh-aw/sandbox/agent/logs/ --disable-builtin-mcps --no-ask-user --allow-all-tools --add-dir "${GITHUB_WORKSPACE}" --prompt-file /tmp/gh-aw/aw-prompts/prompt.txt' env: AWF_REFLECT_ENABLED: 1 - CODEX_API_KEY: ${{ secrets.CODEX_API_KEY || secrets.OPENAI_API_KEY }} + COPILOT_AGENT_RUNNER_TYPE: STANDALONE + COPILOT_DUMMY_BYOK: dummy-byok-key-for-offline-mode + COPILOT_GITHUB_TOKEN: ${{ github.token }} + COPILOT_MODEL: copilot/gpt-5.4 + GH_AW_LLM_PROVIDER: github + GH_AW_MAX_AI_CREDITS: ${{ vars.GH_AW_DEFAULT_EVALS_MAX_AI_CREDITS || '400' }} GH_AW_MAX_TURNS: ${{ vars.GH_AW_DEFAULT_MAX_TURNS || '' }} GH_AW_PHASE: evals - GH_AW_PI_MODEL: openai/gpt-5.4 GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt GH_AW_TIMEOUT_MINUTES: 15 GH_AW_VERSION: dev + GITHUB_API_URL: ${{ github.api_url }} GITHUB_AW: true + GITHUB_COPILOT_INTEGRATION_ID: agentic-workflows + GITHUB_HEAD_REF: ${{ github.head_ref }} + GITHUB_REF_NAME: ${{ github.ref_name }} + GITHUB_SERVER_URL: ${{ github.server_url }} GITHUB_STEP_SUMMARY: /tmp/gh-aw/agent-step-summary.md GITHUB_WORKSPACE: ${{ github.workspace }} GIT_AUTHOR_EMAIL: github-actions[bot]@users.noreply.github.com GIT_AUTHOR_NAME: github-actions[bot] GIT_COMMITTER_EMAIL: github-actions[bot]@users.noreply.github.com GIT_COMMITTER_NAME: github-actions[bot] - OPENAI_API_KEY: ${{ secrets.CODEX_API_KEY || secrets.OPENAI_API_KEY }} - PI_CODING_AGENT_DIR: /tmp/gh-aw/pi-agent-dir - PI_OFFLINE: 1 RUNNER_TEMP: ${{ runner.temp }} + S2STOKENS: true TRACEPARENT: ${{ env.GITHUB_AW_OTEL_TRACE_ID != '' && env.GITHUB_AW_OTEL_PARENT_SPAN_ID != '' && format('00-{0}-{1}-01', env.GITHUB_AW_OTEL_TRACE_ID, env.GITHUB_AW_OTEL_PARENT_SPAN_ID) || '' }} - name: Parse MCP Gateway logs for step summary if: always() @@ -2061,7 +2142,7 @@ jobs: uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: GH_AW_EVALS_QUESTIONS: '[{"id":"review_posted","question":"Did the agent post a code review comment on the pull request?"},{"id":"findings_scoped","question":"Does the agent output show that the review findings are limited to changes in the pull request diff rather than unrelated code?"}]' - GH_AW_EVALS_MODEL: "openai/gpt-5.4" + GH_AW_EVALS_MODEL: "copilot/gpt-5.4" GH_AW_EVALS_PHASE: parse GITHUB_RUN_ID: ${{ github.run_id }} with: @@ -2148,9 +2229,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Check command position id: check_command_position uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 @@ -2210,9 +2291,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Checkout repository uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: @@ -2283,8 +2364,8 @@ jobs: GH_AW_DETECTION_CONCLUSION: ${{ needs.detection.outputs.detection_conclusion }} GH_AW_DETECTION_REASON: ${{ needs.detection.outputs.detection_reason }} GH_AW_EFFECTIVE_TOKENS: ${{ needs.agent.outputs.effective_tokens }} - GH_AW_ENGINE_ID: "pi" - GH_AW_ENGINE_MODEL: "openai/gpt-5.4" + GH_AW_ENGINE_ID: "copilot" + GH_AW_ENGINE_MODEL: "copilot/gpt-5.4" GH_AW_HEAD_SHA: ${{ github.event.pull_request.head.sha }} GH_AW_PROJECT_UTC: "-08:00" GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} @@ -2329,9 +2410,9 @@ jobs: env: GH_AW_SETUP_WORKFLOW_NAME: "PR Code Quality Reviewer" GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/pr-code-quality-reviewer.lock.yml@${{ github.ref }} - GH_AW_INFO_VERSION: "0.84.3" + GH_AW_INFO_VERSION: "1.0.80" GH_AW_INFO_AWF_VERSION: "v0.28.10" - GH_AW_INFO_ENGINE_ID: "pi" + GH_AW_INFO_ENGINE_ID: "copilot" - name: Mask OTLP telemetry headers run: bash "${RUNNER_TEMP}/gh-aw/actions/mask_otlp_headers.sh" - name: Download agent output artifact diff --git a/.github/workflows/pr-code-quality-reviewer.md b/.github/workflows/pr-code-quality-reviewer.md index d1e6d4551a9..107fb20b0c0 100644 --- a/.github/workflows/pr-code-quality-reviewer.md +++ b/.github/workflows/pr-code-quality-reviewer.md @@ -16,9 +16,8 @@ on: name: review events: [pull_request_comment, pull_request_review_comment] engine: - id: pi - model-provider: openai -model: openai/gpt-5.4 + id: copilot +model: copilot/gpt-5.4 permissions: contents: read issues: read diff --git a/actions/setup/js/fixtures/awf-v0.28.7-aic-token-usage.jsonl b/actions/setup/js/fixtures/awf-v0.28.7-aic-token-usage.jsonl new file mode 100644 index 00000000000..6c9b2fdc41d --- /dev/null +++ b/actions/setup/js/fixtures/awf-v0.28.7-aic-token-usage.jsonl @@ -0,0 +1,5 @@ +{"_schema":"token-usage/v0.28.7","timestamp":"2026-08-28T08:59:54.002Z","event":"token_usage","request_id":"5c6141a6-7331-400f-94c8-d38defdcbc7e","provider":"copilot","model":"gpt-4o-mini-2024-07-18","path":"/chat/completions","status":200,"streaming":true,"input_tokens":19288,"output_tokens":35,"cache_read_tokens":0,"cache_write_tokens":0,"duration_ms":2242,"response_bytes":8642,"x_initiator":"user","ai_credits_this_response":0.29142,"ai_credits_total":0.29142,"ai_credits_pricing_source":"models.dev","ai_credits_pricing_tier":"default","ai_credits_accounting_policy":"concrete_model","ai_credits_fallback_pricing_used":false} +{"_schema":"token-usage/v0.28.7","timestamp":"2026-08-28T08:59:56.563Z","event":"token_usage","request_id":"cc30e0e7-0491-484e-aaa9-6b020a2707f4","provider":"copilot","model":"gpt-4o-mini-2024-07-18","path":"/chat/completions","status":200,"streaming":true,"input_tokens":19380,"output_tokens":35,"cache_read_tokens":0,"cache_write_tokens":0,"duration_ms":2469,"response_bytes":8642,"x_initiator":"agent","ai_credits_this_response":0.2928,"ai_credits_total":0.58422,"ai_credits_pricing_source":"models.dev","ai_credits_pricing_tier":"default","ai_credits_accounting_policy":"concrete_model","ai_credits_fallback_pricing_used":false} +{"_schema":"token-usage/v0.28.7","timestamp":"2026-08-28T08:59:58.490Z","event":"token_usage","request_id":"bf3f5fcd-173c-4da9-90d9-680dbc836352","provider":"copilot","model":"gpt-4o-mini-2024-07-18","path":"/chat/completions","status":200,"streaming":true,"input_tokens":272,"output_tokens":35,"cache_read_tokens":19200,"cache_write_tokens":0,"duration_ms":1799,"response_bytes":8648,"x_initiator":"agent","ai_credits_this_response":0.15018,"ai_credits_total":0.7344,"ai_credits_pricing_source":"models.dev","ai_credits_pricing_tier":"default","ai_credits_accounting_policy":"concrete_model","ai_credits_fallback_pricing_used":false} +{"_schema":"token-usage/v0.28.7","timestamp":"2026-08-28T09:00:00.426Z","event":"token_usage","request_id":"776b23fd-a45c-46d4-9e74-95d5f0707402","provider":"copilot","model":"gpt-4o-mini-2024-07-18","path":"/chat/completions","status":200,"streaming":true,"input_tokens":236,"output_tokens":35,"cache_read_tokens":19328,"cache_write_tokens":0,"duration_ms":1861,"response_bytes":8648,"x_initiator":"agent","ai_credits_this_response":0.1506,"ai_credits_total":0.885,"ai_credits_pricing_source":"models.dev","ai_credits_pricing_tier":"default","ai_credits_accounting_policy":"concrete_model","ai_credits_fallback_pricing_used":false} +{"_schema":"token-usage/v0.28.7","timestamp":"2026-08-28T09:00:02.132Z","event":"token_usage","request_id":"59db8b91-cfc2-4cb9-9315-dbcb5e7bcfe7","provider":"copilot","model":"gpt-4o-mini-2024-07-18","path":"/chat/completions","status":200,"streaming":true,"input_tokens":200,"output_tokens":35,"cache_read_tokens":19456,"cache_write_tokens":0,"duration_ms":1552,"response_bytes":8648,"x_initiator":"agent","ai_credits_this_response":0.15102,"ai_credits_total":1.03602,"ai_credits_pricing_source":"models.dev","ai_credits_pricing_tier":"default","ai_credits_accounting_policy":"concrete_model","ai_credits_fallback_pricing_used":false} diff --git a/actions/setup/js/model_costs.cjs b/actions/setup/js/model_costs.cjs index b086fdb58a5..46730644cb5 100644 --- a/actions/setup/js/model_costs.cjs +++ b/actions/setup/js/model_costs.cjs @@ -190,9 +190,10 @@ function usdToAIC(usd) { * @param {number} params.cacheReadTokens * @param {number} params.cacheWriteTokens * @param {number} [params.reasoningTokens] + * @param {boolean} [params.inputTokensIncludeCache] * @returns {number} */ -function computeInferenceCostUSD({ provider, model, inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens, reasoningTokens = 0 }) { +function computeInferenceCostUSD({ provider, model, inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens, reasoningTokens = 0, inputTokensIncludeCache }) { const pricing = findModelPricing(provider, model); if (!pricing) return 0; @@ -201,7 +202,12 @@ function computeInferenceCostUSD({ provider, model, inputTokens, outputTokens, c const cacheRead = cacheReadTokens || 0; const cacheWrite = cacheWriteTokens || 0; const reasoning = reasoningTokens || 0; - const effectiveInput = providerIncludesCacheReadsInInput(provider) ? Math.max(input - cacheRead, 0) : input; + let effectiveInput = input; + if (inputTokensIncludeCache === true) { + effectiveInput = Math.max(input - cacheRead - cacheWrite, 0); + } else if (typeof inputTokensIncludeCache !== "boolean" && providerIncludesCacheReadsInInput(provider)) { + effectiveInput = Math.max(input - cacheRead, 0); + } const promptPrice = pricing.input || 0; const completionPrice = pricing.output || 0; @@ -221,6 +227,7 @@ function computeInferenceCostUSD({ provider, model, inputTokens, outputTokens, c * @param {number} params.cacheReadTokens * @param {number} params.cacheWriteTokens * @param {number} [params.reasoningTokens] + * @param {boolean} [params.inputTokensIncludeCache] * @returns {number} */ function computeInferenceAIC(params) { diff --git a/actions/setup/js/model_costs.test.cjs b/actions/setup/js/model_costs.test.cjs index dc813bb4bb9..d64ce1e943d 100644 --- a/actions/setup/js/model_costs.test.cjs +++ b/actions/setup/js/model_costs.test.cjs @@ -207,6 +207,64 @@ describe("model_costs.cjs", () => { expect(aic).toBeCloseTo(0.336, 6); }); + it("uses explicit inclusive cache semantics when present", async () => { + writeModelsFixture({ + "github-copilot": { + models: { + "gpt-4o-mini": { + cost: { + input: "0.00000015", + output: "0.0000006", + cache_read: "0.000000075", + }, + }, + }, + }, + }); + + const { computeInferenceAIC } = await import("./model_costs.cjs"); + const aic = computeInferenceAIC({ + provider: "copilot", + model: "gpt-4o-mini", + inputTokens: 1000, + outputTokens: 100, + cacheReadTokens: 400, + cacheWriteTokens: 100, + inputTokensIncludeCache: true, + }); + + expect(aic).toBeCloseTo(0.018, 6); + }); + + it("uses explicit additive cache semantics when present", async () => { + writeModelsFixture({ + "github-copilot": { + models: { + "gpt-4o-mini": { + cost: { + input: "0.00000015", + output: "0.0000006", + cache_read: "0.000000075", + }, + }, + }, + }, + }); + + const { computeInferenceAIC } = await import("./model_costs.cjs"); + const aic = computeInferenceAIC({ + provider: "copilot", + model: "gpt-4o-mini", + inputTokens: 1000, + outputTokens: 100, + cacheReadTokens: 400, + cacheWriteTokens: 100, + inputTokensIncludeCache: false, + }); + + expect(aic).toBeCloseTo(0.0255, 6); + }); + it("falls back to bundled models.json when GH_AW_MODELS_JSON_PATH points to a non-existent file", async () => { // Simulate the detection/evals job scenario: GH_AW_MODELS_JSON_PATH is set to // /tmp/gh-aw/models.json but that file was never downloaded from the activation artifact. diff --git a/actions/setup/js/parse_mcp_gateway_log.cjs b/actions/setup/js/parse_mcp_gateway_log.cjs index 7927c11cf63..10b99103418 100644 --- a/actions/setup/js/parse_mcp_gateway_log.cjs +++ b/actions/setup/js/parse_mcp_gateway_log.cjs @@ -67,14 +67,59 @@ function formatDurationMs(ms) { return `${minutes}m${secs}s`; } +/** + * AWF-reported AIC fields are numeric JSON fields. Invalid or missing values + * are identified so the caller can report that legacy pricing was used. + * + * @param {unknown} value + * @returns {number | null} + */ +function parseNonNegativeFiniteNumber(value) { + return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null; +} + +/** + * Preserve AWF's reported six-decimal precision while retaining the + * historical three-decimal export for recomputed legacy records. + * + * @param {number} value + * @param {"awf_reported"|"recomputed"} source + * @returns {string} + */ +function formatAICForOutput(value, source) { + if (!Number.isFinite(value) || value < 0) return ""; + if (source !== "awf_reported") return value.toFixed(3); + const rounded = Number(value.toFixed(6)); + return String(rounded); +} + +/** + * Keep the human-readable table compact for large totals while preserving + * exact AWF precision for the normal per-run range. + * + * @param {number} value + * @param {"awf_reported"|"recomputed"} source + * @returns {string} + */ +function formatAICForTable(value, source) { + return source === "awf_reported" && value < 1000 ? formatAICForOutput(value, source) : formatAIC(value); +} + /** * Parses token-usage.jsonl content and returns an aggregated summary. + * + * token-usage.jsonl is agent-visible runtime telemetry. This parser uses its + * AWF-computed AIC fields only for diagnostics and public reporting; budget + * aborts, retries, authentication, and safe outputs use separate paths. + * * @param {string} jsonlContent - The token-usage.jsonl file content - * @returns {{totalInputTokens: number, totalOutputTokens: number, totalCacheReadTokens: number, totalCacheWriteTokens: number, totalRequests: number, totalDurationMs: number, totalAIC: number, ambientContextTokens: number|undefined, byModel: Object, entries: Array} | null} + * @returns {{totalInputTokens: number, totalOutputTokens: number, totalCacheReadTokens: number, totalCacheWriteTokens: number, totalRequests: number, totalDurationMs: number, totalAIC: number, aiCreditsSource: "awf_reported"|"recomputed", aiCreditsWarnings: string[], ambientContextTokens: number|undefined, byModel: Record, entries: Array} | null} * ambientContextTokens records first-request context size as: * input_tokens + ((cache_read_tokens + cache_write_tokens) / 10). */ function parseTokenUsageJsonl(jsonlContent) { + const seenRequestIds = new Set(); + let duplicateRecordCount = 0; const summary = { totalInputTokens: 0, totalOutputTokens: 0, @@ -83,9 +128,15 @@ function parseTokenUsageJsonl(jsonlContent) { totalRequests: 0, totalDurationMs: 0, totalAIC: 0, + /** @type {"awf_reported"|"recomputed"} */ + aiCreditsSource: "recomputed", + /** @type {string[]} */ + aiCreditsWarnings: [], + /** @type {number | undefined} */ ambientContextTokens: undefined, - byModel: {}, - /** @type {{ model: string, provider: string, inputTokens: number, outputTokens: number, cacheReadTokens: number, cacheWriteTokens: number, reasoningTokens: number, durationMs: number, deltaAIC: number }[]} */ + /** @type {Record} */ + byModel: Object.create(null), + /** @type {{ model: string, provider: string, inputTokens: number, outputTokens: number, cacheReadTokens: number, cacheWriteTokens: number, reasoningTokens: number, durationMs: number, timestampMs: number|null, originalIndex: number, inputTokensIncludeCache: boolean|undefined, hasInputTokensIncludeCacheField: boolean, reportedDeltaAIC: number|null, reportedTotalAIC: number|null, hasReportedDeltaField: boolean, hasReportedTotalField: boolean, deltaAIC: number, runningAIC: number }[]} */ entries: [], }; @@ -96,6 +147,14 @@ function parseTokenUsageJsonl(jsonlContent) { try { const entry = JSON.parse(trimmed); if (!entry || typeof entry !== "object") continue; + const requestId = typeof entry.request_id === "string" ? entry.request_id.trim() : ""; + const eventName = typeof entry.event === "string" && entry.event ? entry.event : "token_usage"; + const dedupeKey = requestId ? `${eventName}:${requestId}` : ""; + if (dedupeKey && seenRequestIds.has(dedupeKey)) { + duplicateRecordCount++; + continue; + } + if (dedupeKey) seenRequestIds.add(dedupeKey); const inputTokens = entry.input_tokens || 0; const outputTokens = entry.output_tokens || 0; @@ -103,6 +162,13 @@ function parseTokenUsageJsonl(jsonlContent) { const cacheWriteTokens = entry.cache_write_tokens || 0; const reasoningTokens = entry.reasoning_tokens || 0; const durationMs = entry.duration_ms || 0; + const parsedTimestamp = typeof entry.timestamp === "string" ? Date.parse(entry.timestamp) : Number.NaN; + const hasInputTokensIncludeCacheField = Object.prototype.hasOwnProperty.call(entry, "input_tokens_include_cache") && entry.input_tokens_include_cache !== null; + const inputTokensIncludeCache = typeof entry.input_tokens_include_cache === "boolean" ? entry.input_tokens_include_cache : undefined; + const hasReportedDelta = Object.prototype.hasOwnProperty.call(entry, "ai_credits_this_response"); + const hasReportedTotal = Object.prototype.hasOwnProperty.call(entry, "ai_credits_total"); + const reportedDeltaAIC = parseNonNegativeFiniteNumber(entry.ai_credits_this_response); + const reportedTotalAIC = parseNonNegativeFiniteNumber(entry.ai_credits_total); summary.totalInputTokens += inputTokens; summary.totalOutputTokens += outputTokens; @@ -136,41 +202,130 @@ function parseTokenUsageJsonl(jsonlContent) { m.requests++; m.durationMs += durationMs; - summary.entries.push({ model, provider: m.provider, inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens, reasoningTokens, durationMs, deltaAIC: 0 }); + summary.entries.push({ + model, + provider: m.provider, + inputTokens, + outputTokens, + cacheReadTokens, + cacheWriteTokens, + reasoningTokens, + durationMs, + timestampMs: Number.isFinite(parsedTimestamp) ? parsedTimestamp : null, + originalIndex: summary.entries.length, + inputTokensIncludeCache, + hasInputTokensIncludeCacheField, + reportedDeltaAIC, + reportedTotalAIC, + hasReportedDeltaField: hasReportedDelta, + hasReportedTotalField: hasReportedTotal, + deltaAIC: 0, + runningAIC: 0, + }); } catch { // Malformed line — ignored. } } if (summary.totalRequests === 0) return null; + if (duplicateRecordCount > 0) { + summary.aiCreditsWarnings.push(`${duplicateRecordCount} duplicate token usage record(s) were ignored by event and request_id.`); + } + + const hasReportedAIC = summary.entries.some(entry => entry.reportedDeltaAIC !== null || entry.reportedTotalAIC !== null); + const hasAnyReportedAICFields = summary.entries.some(entry => entry.hasReportedDeltaField || entry.hasReportedTotalField); + const hasExplicitCacheSemantics = summary.entries.some(entry => typeof entry.inputTokensIncludeCache === "boolean"); + let invalidCacheSemanticsCount = 0; + + if (!hasAnyReportedAICFields && !hasExplicitCacheSemantics) { + invalidCacheSemanticsCount = summary.entries.filter(entry => entry.hasInputTokensIncludeCacheField && typeof entry.inputTokensIncludeCache !== "boolean").length; + // Preserve the legacy aggregation contract exactly for records emitted before + // AWF added reported AIC and explicit cache-semantics fields. + let totalAIC = 0; + for (const [model, usage] of Object.entries(summary.byModel)) { + const aic = computeInferenceAIC({ + provider: usage.provider || "", + model, + inputTokens: usage.inputTokens, + outputTokens: usage.outputTokens, + cacheReadTokens: usage.cacheReadTokens, + cacheWriteTokens: usage.cacheWriteTokens, + reasoningTokens: usage.reasoningTokens || 0, + }); + usage.aic = aic; + totalAIC += aic; + } + summary.totalAIC = totalAIC; - let totalAIC = 0; - for (const [model, usage] of Object.entries(summary.byModel)) { - const aic = computeInferenceAIC({ - provider: usage.provider || "", - model, - inputTokens: usage.inputTokens, - outputTokens: usage.outputTokens, - cacheReadTokens: usage.cacheReadTokens, - cacheWriteTokens: usage.cacheWriteTokens, - reasoningTokens: usage.reasoningTokens || 0, + for (const entry of summary.entries) { + entry.deltaAIC = computeInferenceAIC({ + provider: entry.provider || "", + model: entry.model, + inputTokens: entry.inputTokens, + outputTokens: entry.outputTokens, + cacheReadTokens: entry.cacheReadTokens, + cacheWriteTokens: entry.cacheWriteTokens, + reasoningTokens: entry.reasoningTokens || 0, + }); + } + } else { + summary.entries.sort((left, right) => { + if (left.timestampMs !== null && right.timestampMs !== null) { + return left.timestampMs - right.timestampMs || left.originalIndex - right.originalIndex; + } + if (left.timestampMs !== null) return -1; + if (right.timestampMs !== null) return 1; + return left.originalIndex - right.originalIndex; }); - usage.aic = aic; - totalAIC += aic; + const firstEntry = summary.entries[0]; + if (firstEntry) { + summary.ambientContextTokens = firstEntry.inputTokens + (firstEntry.cacheReadTokens + firstEntry.cacheWriteTokens) / 10; + } + let runningAIC = 0; + let fallbackRecordCount = 0; + for (const usage of Object.values(summary.byModel)) { + usage.aic = 0; + } + + for (let index = 0; index < summary.entries.length; index++) { + const entry = summary.entries[index]; + const reportedFieldsMissingOrInvalid = hasAnyReportedAICFields && (!entry.hasReportedDeltaField || entry.reportedDeltaAIC === null || !entry.hasReportedTotalField || entry.reportedTotalAIC === null); + if (reportedFieldsMissingOrInvalid) fallbackRecordCount++; + + if (entry.reportedDeltaAIC !== null) { + entry.deltaAIC = entry.reportedDeltaAIC; + } else { + if (entry.hasInputTokensIncludeCacheField && typeof entry.inputTokensIncludeCache !== "boolean") { + invalidCacheSemanticsCount++; + } + entry.deltaAIC = computeInferenceAIC({ + provider: entry.provider || "", + model: entry.model, + inputTokens: entry.inputTokens, + outputTokens: entry.outputTokens, + cacheReadTokens: entry.cacheReadTokens, + cacheWriteTokens: entry.cacheWriteTokens, + reasoningTokens: entry.reasoningTokens || 0, + inputTokensIncludeCache: entry.inputTokensIncludeCache, + }); + } + summary.byModel[entry.model].aic += entry.deltaAIC; + runningAIC = entry.reportedTotalAIC ?? runningAIC + entry.deltaAIC; + entry.runningAIC = runningAIC; + } + + summary.totalAIC = runningAIC; + summary.aiCreditsSource = hasReportedAIC ? "awf_reported" : "recomputed"; + if (fallbackRecordCount > 0) { + summary.aiCreditsWarnings.push(`${fallbackRecordCount} token usage record(s) had missing or invalid AWF-reported AI Credits fields; fallback accounting was used for the missing values.`); + } + const summedDeltaAIC = Object.values(summary.byModel).reduce((total, usage) => total + (usage.aic || 0), 0); + if (summary.aiCreditsSource === "awf_reported" && Math.abs(summedDeltaAIC - summary.totalAIC) > 1e-6 * Math.max(1, Math.abs(summary.totalAIC))) { + summary.aiCreditsWarnings.push("The AWF-reported cumulative AI Credits total differs from the sum of per-request credits; the cumulative total was preserved for reporting."); + } } - summary.totalAIC = totalAIC; - - // Compute per-request AI credits. - for (const entry of summary.entries) { - entry.deltaAIC = computeInferenceAIC({ - provider: entry.provider || "", - model: entry.model, - inputTokens: entry.inputTokens, - outputTokens: entry.outputTokens, - cacheReadTokens: entry.cacheReadTokens, - cacheWriteTokens: entry.cacheWriteTokens, - reasoningTokens: entry.reasoningTokens || 0, - }); + if (invalidCacheSemanticsCount > 0) { + summary.aiCreditsWarnings.push(`${invalidCacheSemanticsCount} token usage record(s) had invalid input_tokens_include_cache values; legacy provider cache semantics were used.`); } return summary; @@ -180,7 +335,7 @@ function parseTokenUsageJsonl(jsonlContent) { * Generates a markdown summary section for token usage data. * Renders one row per request in chronological order with per-request AI credits, * a running AI credits total, followed by an aggregate totals row and legend. - * @param {{totalInputTokens: number, totalOutputTokens: number, totalCacheReadTokens: number, totalCacheWriteTokens: number, totalRequests: number, totalDurationMs: number, totalAIC: number, byModel: Object, entries: Array} | null} summary + * @param {ReturnType} summary * @returns {string} Markdown section, or empty string if no data */ function generateTokenUsageSummary(summary) { @@ -192,22 +347,33 @@ function generateTokenUsageSummary(summary) { const entries = summary.entries || []; let compoundedAIC = 0; + const formatSummaryAIC = value => formatAICForTable(value, summary.aiCreditsSource); for (let i = 0; i < entries.length; i++) { const entry = entries[i]; const deltaAIC = entry.deltaAIC || 0; compoundedAIC += deltaAIC; + const runningAIC = summary.aiCreditsSource === "awf_reported" ? entry.runningAIC : compoundedAIC; lines.push( - `| ${i + 1} | ${formatModelEmojiAlias(entry.model) || entry.model} | ${entry.inputTokens.toLocaleString()} | ${entry.outputTokens.toLocaleString()} | ${entry.cacheReadTokens.toLocaleString()} | ${entry.cacheWriteTokens.toLocaleString()} | ${formatAIC(deltaAIC)} | ${formatAIC(compoundedAIC)} | ${formatDurationMs(entry.durationMs)} |` + `| ${i + 1} | ${formatModelEmojiAlias(entry.model) || entry.model} | ${entry.inputTokens.toLocaleString()} | ${entry.outputTokens.toLocaleString()} | ${entry.cacheReadTokens.toLocaleString()} | ${entry.cacheWriteTokens.toLocaleString()} | ${formatSummaryAIC(deltaAIC)} | ${formatSummaryAIC(runningAIC)} | ${formatDurationMs(entry.durationMs)} |` ); } - const totalAIC = formatAIC(summary.totalAIC || 0); + const totalAIC = formatSummaryAIC(summary.totalAIC || 0); lines.push( `| **Total** | | **${summary.totalInputTokens.toLocaleString()}** | **${summary.totalOutputTokens.toLocaleString()}** | **${summary.totalCacheReadTokens.toLocaleString()}** | **${summary.totalCacheWriteTokens.toLocaleString()}** | | **${totalAIC}** | **${formatDurationMs(summary.totalDurationMs)}** |` ); lines.push(""); - lines.push("Legend: `Alias` shows the model shorthand used in the table. `ΔAI Credits` is the per-request cost, and `AI Credits` is the running total computed with the current AI credits pricing model."); + const accountingDescription = + summary.aiCreditsSource === "awf_reported" + ? summary.aiCreditsWarnings.length > 0 + ? "mirrored from AWF fields where available, with warned fallback accounting" + : "mirrored from AWF fields for reporting" + : "recomputed with the current AI credits pricing model for legacy records"; + lines.push(`Legend: \`Alias\` shows the model shorthand used in the table. \`ΔAI Credits\` is the per-request cost, and \`AI Credits\` is the running total ${accountingDescription}.`); + for (const warning of summary.aiCreditsWarnings) { + lines.push(`Warning: ${warning}`); + } lines.push(""); return lines.join("\n") + "\n"; @@ -233,11 +399,14 @@ async function writeStepSummaryWithTokenUsage(coreObj) { if (content?.trim()) { coreObj.info(`Found token-usage.jsonl (${content.length} bytes)`); const parsedSummary = parseTokenUsageJsonl(content); - if (parsedSummary && parsedSummary.totalAIC > 0) { - const roundedAIC = parsedSummary.totalAIC.toFixed(3); - coreObj.exportVariable("GH_AW_AIC", roundedAIC); - coreObj.setOutput("aic", roundedAIC); - coreObj.info(`AI Credits: ${roundedAIC}`); + for (const warning of parsedSummary?.aiCreditsWarnings || []) { + coreObj.warning?.(`[ai-credits] ${warning}`); + } + if (parsedSummary && (parsedSummary.aiCreditsSource === "awf_reported" || parsedSummary.totalAIC > 0)) { + const aic = formatAICForOutput(parsedSummary.totalAIC, parsedSummary.aiCreditsSource); + coreObj.exportVariable("GH_AW_AIC", aic); + coreObj.setOutput("aic", aic); + coreObj.info(`AI Credits: ${aic}`); } if (parsedSummary && typeof parsedSummary.ambientContextTokens === "number" && parsedSummary.ambientContextTokens > 0) { const roundedAmbientContext = String(Math.round(parsedSummary.ambientContextTokens)); @@ -1158,6 +1327,8 @@ if (typeof module !== "undefined" && module.exports) { parseTokenUsageJsonl, generateTokenUsageSummary, formatDurationMs, + formatAICForOutput, + writeStepSummaryWithTokenUsage, hasAICreditsRateLimitError, hasUnknownModelAICreditsError, setUnknownModelAICreditsOutput, diff --git a/actions/setup/js/parse_mcp_gateway_log.test.cjs b/actions/setup/js/parse_mcp_gateway_log.test.cjs index 183862e9ff3..ac5064067cd 100644 --- a/actions/setup/js/parse_mcp_gateway_log.test.cjs +++ b/actions/setup/js/parse_mcp_gateway_log.test.cjs @@ -1,6 +1,8 @@ // @ts-check /// +const fs = require("fs"); +const path = require("path"); const { generateGatewayLogSummary, generatePlainTextGatewaySummary, @@ -19,6 +21,7 @@ const { parseTokenUsageJsonl, generateTokenUsageSummary, formatDurationMs, + writeStepSummaryWithTokenUsage, } = require("./parse_mcp_gateway_log.cjs"); describe("parse_mcp_gateway_log", () => { @@ -1839,6 +1842,395 @@ Some content here.`; expect(parseTokenUsageJsonl(" \n ")).toBeNull(); }); + test("prefers the exact AWF-reported total from run 33157406852", () => { + const content = fs.readFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8"); + const summary = parseTokenUsageJsonl(content); + + expect(summary).not.toBeNull(); + expect(summary.totalRequests).toBe(5); + expect(summary.totalInputTokens).toBe(39376); + expect(summary.totalCacheReadTokens).toBe(57984); + expect(summary.totalOutputTokens).toBe(175); + expect(summary.aiCreditsSource).toBe("awf_reported"); + expect(summary.entries.map(entry => entry.deltaAIC)).toEqual([0.29142, 0.2928, 0.15018, 0.1506, 0.15102]); + expect(summary.totalAIC).toBe(1.03602); + expect(generateTokenUsageSummary(summary)).toContain("1.03602"); + }); + + test("retains legacy recomputation when AWF-reported fields are absent", () => { + const content = fs + .readFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8") + .trim() + .split("\n") + .map(line => { + const record = JSON.parse(line); + delete record.ai_credits_this_response; + delete record.ai_credits_total; + return JSON.stringify(record); + }) + .join("\n"); + const summary = parseTokenUsageJsonl(content); + + expect(summary.aiCreditsSource).toBe("recomputed"); + expect(summary.totalAIC).toBeCloseTo(0.44538, 6); + }); + + test("falls back to legacy pricing and warns for malformed AWF-reported AIC fields", () => { + const content = JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 19288, + output_tokens: 35, + cache_read_tokens: 0, + cache_write_tokens: 0, + ai_credits_this_response: "0.29142", + ai_credits_total: -1, + }); + const summary = parseTokenUsageJsonl(content); + + expect(summary.aiCreditsSource).toBe("recomputed"); + expect(summary.entries[0].deltaAIC).toBeCloseTo(0.29142, 6); + expect(summary.totalAIC).toBeCloseTo(0.29142, 6); + expect(summary.aiCreditsWarnings).toEqual([expect.stringContaining("1 token usage record(s)")]); + expect(generateTokenUsageSummary(summary)).toContain("Warning:"); + }); + + test("treats null AWF-reported AIC fields as malformed rather than zero", () => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 19288, + output_tokens: 35, + cache_read_tokens: 0, + cache_write_tokens: 0, + ai_credits_this_response: null, + ai_credits_total: null, + }) + ); + + expect(summary.aiCreditsSource).toBe("recomputed"); + expect(summary.totalAIC).toBeCloseTo(0.29142, 6); + expect(summary.aiCreditsWarnings).toEqual([expect.stringContaining("fallback accounting")]); + }); + + test("extends the last valid AWF-reported total with fallback pricing when a later total is malformed", () => { + const records = fs + .readFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8") + .trim() + .split("\n") + .map(line => JSON.parse(line)); + records.at(-1).ai_credits_total = "1.03602"; + const summary = parseTokenUsageJsonl(records.map(record => JSON.stringify(record)).join("\n")); + + expect(summary.aiCreditsSource).toBe("awf_reported"); + expect(summary.entries.at(-1).deltaAIC).toBe(0.15102); + expect(summary.totalAIC).toBe(1.03602); + expect(summary.aiCreditsWarnings).toEqual([expect.stringContaining("1 token usage record(s)")]); + }); + + test("exports the exact AWF-reported total to the main job output", async () => { + const tokenUsagePath = "/tmp/gh-aw/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl"; + const content = fs.readFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8"); + const existsSpy = vi.spyOn(fs, "existsSync").mockImplementation(filePath => filePath === tokenUsagePath); + const readSpy = vi.spyOn(fs, "readFileSync").mockImplementation((filePath, encoding) => { + if (filePath === tokenUsagePath) return content; + return ""; + }); + const coreObj = { + debug: vi.fn(), + info: vi.fn(), + exportVariable: vi.fn(), + setOutput: vi.fn(), + summary: { + addRaw: vi.fn(), + write: vi.fn().mockResolvedValue(undefined), + }, + }; + + try { + await writeStepSummaryWithTokenUsage(coreObj); + } finally { + existsSpy.mockRestore(); + readSpy.mockRestore(); + } + + expect(coreObj.exportVariable).toHaveBeenCalledWith("GH_AW_AIC", "1.03602"); + expect(coreObj.setOutput).toHaveBeenCalledWith("aic", "1.03602"); + expect(coreObj.info).toHaveBeenCalledWith("AI Credits: 1.03602"); + }); + + test.each([ + [true, 0.018], + [false, 0.0255], + ])("propagates explicit input_tokens_include_cache=%s through legacy repricing", (inputTokensIncludeCache, expectedAIC) => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 1000, + output_tokens: 100, + cache_read_tokens: 400, + cache_write_tokens: 100, + input_tokens_include_cache: inputTokensIncludeCache, + }) + ); + + expect(summary.entries[0].inputTokensIncludeCache).toBe(inputTokensIncludeCache); + expect(summary.entries[0].deltaAIC).toBeCloseTo(expectedAIC, 6); + expect(summary.totalAIC).toBeCloseTo(expectedAIC, 6); + }); + + test("distinguishes zero AWF-reported credits from absent fields", () => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 1000, + output_tokens: 100, + cache_read_tokens: 0, + cache_write_tokens: 0, + ai_credits_this_response: 0, + ai_credits_total: 0, + }) + ); + + expect(summary.aiCreditsSource).toBe("awf_reported"); + expect(summary.entries[0].deltaAIC).toBe(0); + expect(summary.totalAIC).toBe(0); + expect(summary.aiCreditsWarnings).toEqual([]); + expect(generateTokenUsageSummary(summary)).toContain("| **Total** | | **1,000** | **100** | **0** | **0** | | **0** |"); + }); + + test("exports AWF-reported zero to the main job output", async () => { + const tokenUsagePath = "/tmp/gh-aw/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl"; + const content = JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 1000, + output_tokens: 100, + ai_credits_this_response: 0, + ai_credits_total: 0, + }); + const existsSpy = vi.spyOn(fs, "existsSync").mockImplementation(filePath => filePath === tokenUsagePath); + const readSpy = vi.spyOn(fs, "readFileSync").mockImplementation(filePath => { + if (filePath === tokenUsagePath) return content; + return ""; + }); + const coreObj = { + debug: vi.fn(), + info: vi.fn(), + exportVariable: vi.fn(), + setOutput: vi.fn(), + summary: { + addRaw: vi.fn(), + write: vi.fn().mockResolvedValue(undefined), + }, + }; + + try { + await writeStepSummaryWithTokenUsage(coreObj); + } finally { + existsSpy.mockRestore(); + readSpy.mockRestore(); + } + + expect(coreObj.exportVariable).toHaveBeenCalledWith("GH_AW_AIC", "0"); + expect(coreObj.setOutput).toHaveBeenCalledWith("aic", "0"); + expect(coreObj.info).toHaveBeenCalledWith("AI Credits: 0"); + }); + + test("aggregates AWF-reported credits by model without changing the run total", () => { + const content = [ + { + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 10, + output_tokens: 1, + ai_credits_this_response: 0.2, + ai_credits_total: 0.2, + }, + { + model: "claude-sonnet-4-6", + provider: "copilot", + input_tokens: 20, + output_tokens: 2, + ai_credits_this_response: 0.8, + ai_credits_total: 1, + }, + ] + .map(record => JSON.stringify(record)) + .join("\n"); + + const summary = parseTokenUsageJsonl(content); + + expect(summary.byModel["gpt-4o-mini"].aic).toBe(0.2); + expect(summary.byModel["claude-sonnet-4-6"].aic).toBe(0.8); + expect(summary.totalAIC).toBe(1); + }); + + test("keeps model names such as __proto__ as data without polluting object prototypes", () => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + model: "__proto__", + provider: "copilot", + input_tokens: 10, + output_tokens: 1, + ai_credits_this_response: 0.1, + ai_credits_total: 0.1, + }) + ); + + expect(Object.getPrototypeOf(summary.byModel)).toBeNull(); + expect(summary.byModel["__proto__"].aic).toBe(0.1); + expect(Object.prototype).not.toHaveProperty("aic"); + }); + + test("deduplicates repeated request IDs before aggregating reported credits", () => { + const record = { + request_id: "same-request", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 10, + output_tokens: 1, + ai_credits_this_response: 0.2, + ai_credits_total: 0.2, + }; + const summary = parseTokenUsageJsonl(`${JSON.stringify(record)}\n${JSON.stringify(record)}\n`); + + expect(summary.totalRequests).toBe(1); + expect(summary.totalInputTokens).toBe(10); + expect(summary.byModel["gpt-4o-mini"].aic).toBe(0.2); + expect(summary.totalAIC).toBe(0.2); + expect(summary.aiCreditsWarnings).toEqual([expect.stringContaining("1 duplicate token usage record(s)")]); + }); + + test("warns and uses legacy provider semantics for invalid input_tokens_include_cache", () => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 1000, + output_tokens: 100, + cache_read_tokens: 400, + cache_write_tokens: 100, + input_tokens_include_cache: "invalid", + }) + ); + + expect(summary.totalAIC).toBeCloseTo(0.0195, 6); + expect(summary.aiCreditsWarnings).toEqual([expect.stringContaining("invalid input_tokens_include_cache")]); + }); + + test("does not warn for invalid input_tokens_include_cache when AWF delta is valid", () => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + provider: "copilot", + model: "gpt-4o-mini-2024-07-18", + input_tokens: 1000, + output_tokens: 100, + cache_read_tokens: 400, + cache_write_tokens: 100, + input_tokens_include_cache: "invalid", + ai_credits_this_response: 0.123, + ai_credits_total: 0.123, + }) + ); + + expect(summary.totalAIC).toBe(0.123); + expect(summary.aiCreditsWarnings).toEqual([]); + }); + + test("uses the chronologically last valid AWF-reported total", () => { + const content = [ + { + timestamp: "2026-08-28T09:00:02Z", + request_id: "third", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 300, + output_tokens: 1, + ai_credits_this_response: 1, + ai_credits_total: 3, + }, + { + timestamp: "2026-08-28T09:00:00Z", + request_id: "first", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 100, + output_tokens: 1, + ai_credits_this_response: 1, + ai_credits_total: 1, + }, + { + timestamp: "2026-08-28T09:00:01Z", + request_id: "second", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 200, + output_tokens: 1, + ai_credits_this_response: 1, + ai_credits_total: 2, + }, + ] + .map(record => JSON.stringify(record)) + .join("\n"); + + const summary = parseTokenUsageJsonl(content); + + expect(summary.entries.map(entry => entry.runningAIC)).toEqual([1, 2, 3]); + expect(summary.totalAIC).toBe(3); + expect(summary.ambientContextTokens).toBe(100); + expect(summary.aiCreditsWarnings).toEqual([]); + }); + + test("warns when AWF cumulative and per-request reported credits diverge", () => { + const content = [ + { + request_id: "one", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 1, + output_tokens: 1, + ai_credits_this_response: 0.2, + ai_credits_total: 0.2, + }, + { + request_id: "two", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 1, + output_tokens: 1, + ai_credits_this_response: 0.8, + ai_credits_total: 0.9, + }, + ] + .map(record => JSON.stringify(record)) + .join("\n"); + + const summary = parseTokenUsageJsonl(content); + + expect(summary.totalAIC).toBe(0.9); + expect(summary.aiCreditsWarnings).toEqual([expect.stringContaining("differs from the sum")]); + }); + + test("keeps large AWF-reported totals compact in the human-readable table", () => { + const summary = parseTokenUsageJsonl( + JSON.stringify({ + request_id: "large", + model: "gpt-4o-mini", + provider: "copilot", + input_tokens: 1, + output_tokens: 1, + ai_credits_this_response: 1234.56789, + ai_credits_total: 1234.56789, + }) + ); + + expect(generateTokenUsageSummary(summary)).toContain("| 1.2K | 1.2K |"); + }); + test("parses a single entry and aggregates totals", () => { const content = JSON.stringify({ timestamp: "2026-04-01T17:56:38.042Z", diff --git a/actions/setup/js/parse_token_usage.cjs b/actions/setup/js/parse_token_usage.cjs index c571610a0ef..e7daa3d3276 100644 --- a/actions/setup/js/parse_token_usage.cjs +++ b/actions/setup/js/parse_token_usage.cjs @@ -4,7 +4,7 @@ const fs = require("fs"); const { getErrorMessage } = require("./error_helpers.cjs"); const { ERR_PARSE } = require("./error_codes.cjs"); -const { parseTokenUsageJsonl, generateTokenUsageSummary } = require("./parse_mcp_gateway_log.cjs"); +const { parseTokenUsageJsonl, generateTokenUsageSummary, formatAICForOutput } = require("./parse_mcp_gateway_log.cjs"); const { calculateWorkingSetFromJSONL } = require("./working_set_metrics.cjs"); /** @@ -50,12 +50,24 @@ function getReadableTokenUsagePaths(paths) { * @returns {string} */ function extractRequestId(line) { - const match = line.match(/"request_id"\s*:\s*"((?:\\.|[^"\\])*)"/); - return match ? match[1] : ""; + const requestMatch = line.match(/"request_id"\s*:\s*"((?:\\.|[^"\\])*)"/); + return requestMatch ? requestMatch[1] : ""; } /** - * Reads token usage files and deduplicates overlapping lines by request_id. + * Extracts a cross-file dedupe key with lightweight matching (no full JSON parse). + * @param {string} line + * @returns {string} + */ +function extractTokenUsageDedupeKey(line) { + const requestId = extractRequestId(line); + if (!requestId) return ""; + const eventMatch = line.match(/"event"\s*:\s*"((?:\\.|[^"\\])*)"/); + return `${eventMatch ? eventMatch[1] : "token_usage"}:${requestId}`; +} + +/** + * Reads token usage files and deduplicates overlapping lines by event and request_id. * Falls back to raw line dedupe when request_id is absent. * @param {string[]} paths * @returns {string} @@ -76,8 +88,7 @@ function readDedupedTokenUsage(paths) { for (const line of fileContent.split("\n")) { const trimmed = line.trim(); if (!trimmed) continue; - const requestId = extractRequestId(trimmed); - const dedupeKey = requestId ? `request_id:${requestId}` : trimmed; + const dedupeKey = extractTokenUsageDedupeKey(trimmed) || trimmed; if (uniqueLineKeys.has(dedupeKey)) continue; uniqueLineKeys.add(dedupeKey); dedupedLines.push(trimmed); @@ -205,6 +216,9 @@ async function main() { core.info("Token usage file contained no valid entries"); return; } + for (const warning of summary.aiCreditsWarnings) { + core.warning(`[ai-credits] ${warning}`); + } const markdown = generateTokenUsageSummary(summary); const workingSet = calculateWorkingSetFromJSONL(content).workingSet; if (markdown.length > 0) { @@ -234,7 +248,7 @@ async function main() { cache_read_tokens: summary.totalCacheReadTokens, cache_write_tokens: summary.totalCacheWriteTokens, ambient_context: Math.round(summary.ambientContextTokens || 0), - ai_credits: Number((summary.totalAIC || 0).toFixed(3)), + ai_credits: summary.aiCreditsSource === "awf_reported" ? Number(summary.totalAIC.toFixed(6)) : Number((summary.totalAIC || 0).toFixed(3)), ...(primaryModel ? { primary_model: primaryModel } : {}), }; fs.writeFileSync(AGENT_USAGE_PATH, JSON.stringify(agentUsage) + "\n"); @@ -244,8 +258,8 @@ async function main() { core.setOutput("primary_model", primaryModel); core.info(`Primary model: ${primaryModel}`); } - if (summary.totalAIC > 0) { - const aic = summary.totalAIC.toFixed(3); + if (summary.aiCreditsSource === "awf_reported" || summary.totalAIC > 0) { + const aic = formatAICForOutput(summary.totalAIC, summary.aiCreditsSource); core.exportVariable("GH_AW_AIC", aic); core.setOutput("aic", aic); core.info(`AI Credits: ${aic}`); @@ -267,6 +281,7 @@ if (typeof module !== "undefined" && module.exports) { main, getReadableTokenUsagePaths, extractRequestId, + extractTokenUsageDedupeKey, readDedupedTokenUsage, getSummaryTitle, buildStepSummarySection, diff --git a/actions/setup/js/parse_token_usage.test.cjs b/actions/setup/js/parse_token_usage.test.cjs index 42e0a61f0fc..c20fa977cc2 100644 --- a/actions/setup/js/parse_token_usage.test.cjs +++ b/actions/setup/js/parse_token_usage.test.cjs @@ -9,6 +9,7 @@ const { main, getReadableTokenUsagePaths, extractRequestId, + extractTokenUsageDedupeKey, readDedupedTokenUsage, getSummaryTitle, buildStepSummarySection, @@ -206,8 +207,19 @@ describe("parse_token_usage", () => { expect(mockCore.info).toHaveBeenCalledWith(expect.stringContaining("Alias")); }); - test("uses custom summary title when configured", async () => { + test("keeps threat-detection title and rows with AWF-reported credits", async () => { process.env.GH_AW_TOKEN_USAGE_SUMMARY_TITLE = "Threat Detection Token Usage"; + const reportedEntry = JSON.stringify({ + model: "gpt-4o-mini-2024-07-18", + provider: "copilot", + input_tokens: 19288, + output_tokens: 35, + cache_read_tokens: 0, + cache_write_tokens: 0, + duration_ms: 2242, + ai_credits_this_response: 0.29142, + ai_credits_total: 0.29142, + }); fs.existsSync = vi.fn(p => { if (p === TOKEN_USAGE_PATH) return true; @@ -215,12 +227,12 @@ describe("parse_token_usage", () => { return originalExistsSync(p); }); fs.statSync = vi.fn(p => { - if (p === TOKEN_USAGE_PATH) return { size: singleEntry.length }; + if (p === TOKEN_USAGE_PATH) return { size: reportedEntry.length }; if (p === TOKEN_USAGE_AUDIT_PATH) return { size: 0 }; return originalStatSync(p); }); fs.readFileSync = vi.fn((p, enc) => { - if (p === TOKEN_USAGE_PATH) return singleEntry; + if (p === TOKEN_USAGE_PATH) return reportedEntry; if (p === TOKEN_USAGE_AUDIT_PATH) return ""; return originalReadFileSync(p, enc); }); @@ -228,6 +240,8 @@ describe("parse_token_usage", () => { await main(); expect(mockCore.summary.addRaw).toHaveBeenCalledWith(expect.stringContaining("Threat Detection Token Usage"), true); + expect(mockCore.summary.addRaw).toHaveBeenCalledWith(expect.stringContaining("gpt40mini"), true); + expect(mockCore.summary.addRaw).toHaveBeenCalledWith(expect.stringContaining("0.29142"), true); }); test("appends token usage section to GITHUB_STEP_SUMMARY when configured", async () => { @@ -307,6 +321,118 @@ describe("parse_token_usage", () => { expect(mockCore.setOutput).toHaveBeenCalledWith("primary_model", "claude-sonnet-4-6"); }); + test("writes the exact AWF-reported AIC total without repricing", async () => { + const agentUsageFile = path.join(tmpDir, "agent_usage.json"); + const fixtureContent = originalReadFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8"); + + fs.existsSync = vi.fn(p => { + if (p === TOKEN_USAGE_PATH) return true; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return false; + return originalExistsSync(p); + }); + fs.statSync = vi.fn(p => { + if (p === TOKEN_USAGE_PATH) return { size: fixtureContent.length }; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return { size: 0 }; + return originalStatSync(p); + }); + fs.readFileSync = vi.fn((p, enc) => { + if (p === TOKEN_USAGE_PATH) return fixtureContent; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return ""; + return originalReadFileSync(p, enc); + }); + fs.writeFileSync = vi.fn((p, data) => { + if (p === AGENT_USAGE_PATH) { + originalWriteFileSync(agentUsageFile, data); + } else { + originalWriteFileSync(p, data); + } + }); + + await main(); + + const agentUsage = JSON.parse(originalReadFileSync(agentUsageFile, "utf8")); + expect(agentUsage.ai_credits).toBe(1.03602); + expect(mockCore.exportVariable).toHaveBeenCalledWith("GH_AW_AIC", "1.03602"); + expect(mockCore.setOutput).toHaveBeenCalledWith("aic", "1.03602"); + expect(mockCore.info).toHaveBeenCalledWith(expect.stringContaining("1.03602")); + }); + + test("exports AWF-reported zero AIC instead of treating it as missing", async () => { + const agentUsageFile = path.join(tmpDir, "agent_usage.json"); + const zeroEntry = JSON.stringify({ + model: "gpt-4o-mini-2024-07-18", + provider: "copilot", + input_tokens: 1000, + output_tokens: 100, + ai_credits_this_response: 0, + ai_credits_total: 0, + }); + + fs.existsSync = vi.fn(p => { + if (p === TOKEN_USAGE_PATH) return true; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return false; + return originalExistsSync(p); + }); + fs.statSync = vi.fn(p => { + if (p === TOKEN_USAGE_PATH) return { size: zeroEntry.length }; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return { size: 0 }; + return originalStatSync(p); + }); + fs.readFileSync = vi.fn((p, enc) => { + if (p === TOKEN_USAGE_PATH) return zeroEntry; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return ""; + return originalReadFileSync(p, enc); + }); + fs.writeFileSync = vi.fn((p, data) => { + if (p === AGENT_USAGE_PATH) { + originalWriteFileSync(agentUsageFile, data); + } else { + originalWriteFileSync(p, data); + } + }); + + await main(); + + const agentUsage = JSON.parse(originalReadFileSync(agentUsageFile, "utf8")); + expect(agentUsage.ai_credits).toBe(0); + expect(mockCore.exportVariable).toHaveBeenCalledWith("GH_AW_AIC", "0"); + expect(mockCore.setOutput).toHaveBeenCalledWith("aic", "0"); + }); + + test("surfaces fallback accounting warnings", async () => { + const malformedEntry = JSON.stringify({ + model: "gpt-4o-mini-2024-07-18", + provider: "copilot", + input_tokens: 19288, + output_tokens: 35, + cache_read_tokens: 0, + cache_write_tokens: 0, + ai_credits_this_response: null, + ai_credits_total: null, + }); + + fs.existsSync = vi.fn(p => { + if (p === TOKEN_USAGE_PATH) return true; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return false; + return originalExistsSync(p); + }); + fs.statSync = vi.fn(p => { + if (p === TOKEN_USAGE_PATH) return { size: malformedEntry.length }; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return { size: 0 }; + return originalStatSync(p); + }); + fs.readFileSync = vi.fn((p, enc) => { + if (p === TOKEN_USAGE_PATH) return malformedEntry; + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_AWF_AUDIT_PATH) return ""; + return originalReadFileSync(p, enc); + }); + + await main(); + + expect(mockCore.warning).toHaveBeenCalledWith(expect.stringContaining("[ai-credits]")); + expect(mockCore.warning).toHaveBeenCalledWith(expect.stringContaining("fallback accounting")); + }); + test("handles multiple model entries", async () => { const agentUsageFile = path.join(tmpDir, "agent_usage.json"); @@ -456,6 +582,42 @@ describe("parse_token_usage", () => { expect(agentUsage.output_tokens).toBe(305); }); + test("deduplicates mirrored AWF token usage files before writing agent_usage", async () => { + const agentUsageFile = path.join(tmpDir, "agent_usage.json"); + const fixtureContent = originalReadFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8"); + + fs.existsSync = vi.fn(p => { + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_PATH) return true; + if (p === TOKEN_USAGE_AWF_AUDIT_PATH) return false; + return originalExistsSync(p); + }); + fs.statSync = vi.fn(p => { + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_PATH) return { size: fixtureContent.length }; + if (p === TOKEN_USAGE_AWF_AUDIT_PATH) return { size: 0 }; + return originalStatSync(p); + }); + fs.readFileSync = vi.fn((p, enc) => { + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_PATH) return fixtureContent; + if (p === TOKEN_USAGE_AWF_AUDIT_PATH) return ""; + return originalReadFileSync(p, enc); + }); + fs.writeFileSync = vi.fn((p, data) => { + if (p === AGENT_USAGE_PATH) { + originalWriteFileSync(agentUsageFile, data); + } else { + originalWriteFileSync(p, data); + } + }); + + await main(); + + const agentUsage = JSON.parse(originalReadFileSync(agentUsageFile, "utf8")); + expect(agentUsage.input_tokens).toBe(39376); + expect(agentUsage.output_tokens).toBe(175); + expect(agentUsage.ai_credits).toBe(1.03602); + expect(mockCore.exportVariable).toHaveBeenCalledWith("GH_AW_AIC", "1.03602"); + }); + test("calls setFailed when an error is thrown", async () => { fs.existsSync = vi.fn(p => { if (p === TOKEN_USAGE_PATH) return true; @@ -506,6 +668,13 @@ describe("parse_token_usage", () => { expect(extractRequestId('{"model":"m"}')).toBe(""); }); + test("extractTokenUsageDedupeKey includes event and request_id", () => { + expect(extractTokenUsageDedupeKey('{"event":"token_usage","request_id":"req-123","model":"m"}')).toBe("token_usage:req-123"); + expect(extractTokenUsageDedupeKey('{"event":"other","request_id":"req-123","model":"m"}')).toBe("other:req-123"); + expect(extractTokenUsageDedupeKey('{"request_id":"req-123","model":"m"}')).toBe("token_usage:req-123"); + expect(extractTokenUsageDedupeKey('{"model":"m"}')).toBe(""); + }); + test("getReadableTokenUsagePaths skips failing stat path and keeps valid path", () => { fs.existsSync = vi.fn(p => p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_PATH); fs.statSync = vi.fn(p => { @@ -535,6 +704,37 @@ describe("parse_token_usage", () => { expect(deduped.match(/"request_id":"req-1"/g)).toHaveLength(1); }); + test("readDedupedTokenUsage keeps different events with the same request_id", () => { + const fileA = '{"event":"token_usage","request_id":"req-1","model":"m1","input_tokens":1}'; + const fileB = '{"event":"token_steering","request_id":"req-1","model":"m1","input_tokens":2}'; + + fs.readFileSync = vi.fn(p => { + if (p === TOKEN_USAGE_AUDIT_PATH) return fileA; + if (p === TOKEN_USAGE_PATH) return fileB; + return originalReadFileSync(p, "utf8"); + }); + + const deduped = readDedupedTokenUsage([TOKEN_USAGE_AUDIT_PATH, TOKEN_USAGE_PATH]); + expect(deduped).toContain('"event":"token_usage"'); + expect(deduped).toContain('"event":"token_steering"'); + expect(deduped.match(/"request_id":"req-1"/g)).toHaveLength(2); + }); + + test("deduplicates mirrored AWF records before aggregating reported credits", () => { + const fixture = originalReadFileSync(path.join(__dirname, "fixtures", "awf-v0.28.7-aic-token-usage.jsonl"), "utf8"); + fs.readFileSync = vi.fn(p => { + if (p === TOKEN_USAGE_AUDIT_PATH || p === TOKEN_USAGE_PATH) return fixture; + return originalReadFileSync(p, "utf8"); + }); + + const deduped = readDedupedTokenUsage([TOKEN_USAGE_AUDIT_PATH, TOKEN_USAGE_PATH]); + const { parseTokenUsageJsonl } = require("./parse_mcp_gateway_log.cjs"); + const summary = parseTokenUsageJsonl(deduped); + + expect(summary.totalRequests).toBe(5); + expect(summary.totalAIC).toBe(1.03602); + }); + test("getSummaryTitle returns trimmed env title", () => { process.env.GH_AW_TOKEN_USAGE_SUMMARY_TITLE = " Threat Detection Token Usage "; expect(getSummaryTitle()).toBe("Threat Detection Token Usage"); diff --git a/docs/adr/56975-use-authoritative-awf-ai-credit-totals.md b/docs/adr/56975-use-authoritative-awf-ai-credit-totals.md new file mode 100644 index 00000000000..954316400e4 --- /dev/null +++ b/docs/adr/56975-use-authoritative-awf-ai-credit-totals.md @@ -0,0 +1,50 @@ +# ADR-56975: Use Authoritative AWF AI-Credit Totals + +**Date**: 2026-08-29 +**Status**: Draft +**Deciders**: pelikhan, adr-writer agent + +--- + +### Context + +This pull request changes how `gh-aw` reports AI Credits usage from AWF token-usage telemetry across both JavaScript and Go reporting paths. The PR body documents a concrete defect: `gh-aw` recomputed AI Credit totals from raw token counts, double-subtracted cached tokens for AWF records that already reported cache information separately, and produced totals roughly half of the true value. The diff introduces shared handling for AWF-reported `ai_credits_this_response` and `ai_credits_total` fields, explicit `input_tokens_include_cache` semantics, deduplication by `event:request_id`, chronological processing, fallback warnings, and matching tests and fixtures. Because this changes the accounting contract used for summaries, console output, audits, forecasts, and generated artifacts, the decision should be recorded explicitly. + +### Decision + +We will treat AWF-reported AI Credit fields as the authoritative source for user-facing AI Credits reporting whenever valid values are present, and we will fall back to legacy token-based recomputation only for older or malformed records. We will also make cache-token accounting explicit via `input_tokens_include_cache` semantics instead of inferring cache inclusion purely from provider behavior. We chose this approach because the PR evidence shows that recomputing modern AWF records in `gh-aw` can misprice runs, while preserving fallback logic maintains compatibility for historical telemetry. + +### Alternatives Considered + +#### Alternative 1: Continue Recomputing All AI Credits From Token Counts in gh-aw + +Keep `gh-aw` as the sole calculator of AI Credit totals and ignore AWF-reported per-request and cumulative fields. + +This was considered because it preserves a single local accounting path and avoids trusting producer-supplied totals. It was not chosen because the PR body and fixture demonstrate that the local recomputation was already wrong for modern AWF records with separate cache semantics, producing materially incorrect totals in public reporting. + +#### Alternative 2: Use AWF-Reported Totals Only, Without Legacy Fallback or Explicit Cache Semantics + +Switch entirely to AWF-provided fields and drop token-based repricing behavior for records that do not include the new fields. + +This was considered because it simplifies the reporting path for current telemetry. It was not chosen because the diff explicitly supports older records and malformed fields, and removing fallback behavior would break compatibility with historical logs and make partial or transitional datasets harder to interpret. + +### Consequences + +#### Positive +- AI Credits reported in step summaries, `agent_usage.json`, audit output, and forecasts will match AWF’s authoritative totals for modern records. +- Explicit `input_tokens_include_cache` handling removes ambiguity around cache-token accounting and prevents the double-subtraction bug described in the PR. +- Shared resolution logic across JavaScript and Go surfaces reduces cross-surface drift and keeps diagnostics consistent. + +#### Negative +- Reporting now depends on the correctness and presence of AWF-emitted accounting fields for modern records, increasing coupling to AWF telemetry semantics. +- The parser and formatting logic become more complex because they must support mixed generations of records, malformed-field fallbacks, warnings, deduplication, and chronological reconstruction. +- Preserving AWF cumulative totals even when they diverge from summed per-request values may expose inconsistencies that require additional operator explanation. + +#### Neutral +- Budget enforcement, retries, authentication, and failure classification remain unchanged; the PR limits the decision to diagnostics and public reporting paths. +- Historical records without AWF AI Credit fields continue to use legacy repricing, so the repository will temporarily support two accounting modes. +- The ADR documents an accounting and telemetry interpretation rule rather than a broader runtime architecture change. + +--- + +*ADR created by [adr-writer agent]. Review and finalize before changing status from Draft to Accepted.* diff --git a/docs/src/content/docs/reference/artifacts.md b/docs/src/content/docs/reference/artifacts.md index 1d62c1a42dd..8f80d925dbf 100644 --- a/docs/src/content/docs/reference/artifacts.md +++ b/docs/src/content/docs/reference/artifacts.md @@ -148,7 +148,7 @@ The unified `agent` artifact contains agent job outputs: - Agent execution logs - Safe output data (`agent_output.json`) - GitHub API rate limit logs (`github_rate_limits.jsonl`) -- Token usage summary (`agent_usage.json`) — aggregated totals only; per-request data is in `firewall-audit-logs` +- Token usage summary (`agent_usage.json`) — aggregated totals only; per-request data is in `firewall-audit-logs`. When AWF records include valid `ai_credits_this_response` and `ai_credits_total` values, the summary preserves those reported values instead of repricing the tokens. - `otel.jsonl` — OTLP span mirror written by gh-aw's JavaScript span exporters when `observability.otlp` is configured For OTLP configuration, runtime environment variables, and span semantics, see the [OpenTelemetry guide](/gh-aw/reference/open-telemetry/). @@ -221,6 +221,8 @@ Working-Set Rebuild Factor measures cumulative context reconstruction relative t ### Accessing usage data +Token-usage files are diagnostic data produced in the agent runtime. Their mirrored AIC fields support usage reporting and analysis, but are not sufficient evidence to classify a provider failure as a trusted budget-enforcement event. + ```bash # Download only the usage artifact gh aw logs --artifacts usage diff --git a/docs/src/content/docs/specs/ai-credits-specification.md b/docs/src/content/docs/specs/ai-credits-specification.md index 7a46aef939b..d6bdc12eeb7 100644 --- a/docs/src/content/docs/specs/ai-credits-specification.md +++ b/docs/src/content/docs/specs/ai-credits-specification.md @@ -146,16 +146,16 @@ If a model entry omits optional price fields, implementations MUST apply the fol ### 3.5 Provider-Specific Input Handling -For providers that include cache-read tokens in total input tokens, implementations MUST subtract `cache_read_tokens` from `input_tokens` before applying input price and MUST NOT double-charge cache-read usage. +When an invocation record provides the optional, forward-compatible `input_tokens_include_cache` field, implementations MUST use it instead of inferring cache semantics from the provider. If it is `true`, cache-read and cache-write tokens are subsets of `input_tokens` and MUST be subtracted before applying the full input price. If it is `false`, the cache fields are additive and MUST NOT be subtracted. -The following providers are known to bundle cache-read tokens in the reported input total and MUST have §3.5 applied: +Legacy records without `input_tokens_include_cache` retain these provider defaults: | Provider (normalized) | Notes | |-----------------------|-------| | `anthropic` | Direct Anthropic API | | `openai` | Direct OpenAI API | | `azure-openai` / `azure_openai` | Azure-hosted OpenAI | -| `github-copilot` (and aliases `github`, `copilot`, `github_models`) | GitHub Copilot proxy — proxies both OpenAI and Anthropic models, which bundle cache-read tokens in input | +| `github-copilot` (and aliases `github`, `copilot`, `github_models`) | Treat cache-read tokens as included to preserve legacy repricing behavior when the optional field is absent. | ### 3.6 Aggregation @@ -333,6 +333,12 @@ These references SHOULD be treated as the external billing-alignment sources for A conforming implementation MUST expose AIC in runtime reporting outputs where cost metrics are emitted. +When AWF token-usage records provide finite, non-negative `ai_credits_this_response` values, reporting MUST prefer those per-request values. The chronologically last valid `ai_credits_total` is the reported run total. If later records omit or contain invalid reported values, reporting MUST surface a warning and extend the last valid total with available per-request values or legacy pricing until another valid cumulative total appears. Implementations MUST recompute AIC from token counts and catalog pricing for legacy records that omit these fields. + +> [!NOTE] +> Token-usage files are diagnostic artifacts visible to the agent runtime. Their mirrored AIC fields improve usage summaries, but are not sufficient evidence to classify a provider failure as budget enforcement or to change retry, authentication, or safe-output behavior. +> This reporting contract does not alter per-run or daily guardrail enforcement inputs; those require a separate first-party accounting contract. + Implementations SHOULD provide: - Per-run AIC values. diff --git a/pkg/cli/audit_render_output_test.go b/pkg/cli/audit_render_output_test.go index fc5662a4f28..64807039a08 100644 --- a/pkg/cli/audit_render_output_test.go +++ b/pkg/cli/audit_render_output_test.go @@ -21,6 +21,18 @@ func TestRenderAuditOutputConsole(t *testing.T) { assert.Contains(t, output, "Test Workflow") } +func TestRenderConsoleTokenUsageWarnings(t *testing.T) { + output := testutil.CaptureStderr(t, func() { + renderConsoleTokenUsage(&TokenUsageSummary{ + TotalRequests: 1, + Warnings: []string{"fallback accounting was used"}, + }) + }) + + assert.Contains(t, output, "token_usage_warnings:") + assert.Contains(t, output, "fallback accounting was used") +} + func TestRenderAuditCompletion(t *testing.T) { outputDir := t.TempDir() diff --git a/pkg/cli/audit_report_render.go b/pkg/cli/audit_report_render.go index dfc54f1ba41..90bfcd17449 100644 --- a/pkg/cli/audit_report_render.go +++ b/pkg/cli/audit_report_render.go @@ -164,6 +164,12 @@ func renderConsoleTokenUsage(tokenUsage *TokenUsageSummary) { tokenUsage.TotalRequests, console.FormatNumber(tokenUsage.TotalSteeringEvents), ) + if len(tokenUsage.Warnings) > 0 { + fmt.Fprintln(os.Stderr, " token_usage_warnings:") + for _, warning := range tokenUsage.Warnings { + fmt.Fprintf(os.Stderr, " %s\n", warning) + } + } } func renderConsoleGitHubAPIUsage(rateLimit *GitHubRateLimitUsage) { diff --git a/pkg/cli/model_costs_cache_semantics.go b/pkg/cli/model_costs_cache_semantics.go new file mode 100644 index 00000000000..4ef69620b56 --- /dev/null +++ b/pkg/cli/model_costs_cache_semantics.go @@ -0,0 +1,39 @@ +package cli + +func computeModelInferenceAICWithCacheSemantics(provider, model string, inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens, reasoningTokens int, inputTokensIncludeCache *bool) float64 { + if inputTokensIncludeCache == nil { + return computeModelInferenceAIC(provider, model, inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens, reasoningTokens) + } + + pricing, ok := findModelPricing(provider, model) + if !ok { + return 0 + } + + effectiveInput := inputTokens + if *inputTokensIncludeCache { + effectiveInput = max(inputTokens-cacheReadTokens-cacheWriteTokens, 0) + } + + promptPrice := pricing["input"] + completionPrice := pricing["output"] + cacheReadPrice := pricing["cache_read"] + if cacheReadPrice == 0 { + cacheReadPrice = promptPrice + } + cacheWritePrice := pricing["cache_write"] + if cacheWritePrice == 0 { + cacheWritePrice = promptPrice + } + reasoningPrice := pricing["reasoning"] + if reasoningPrice == 0 { + reasoningPrice = completionPrice + } + + costUSD := float64(effectiveInput)*promptPrice + + float64(outputTokens)*completionPrice + + float64(cacheReadTokens)*cacheReadPrice + + float64(cacheWriteTokens)*cacheWritePrice + + float64(reasoningTokens)*reasoningPrice + return usdToAIC(costUSD) +} diff --git a/pkg/cli/model_costs_test.go b/pkg/cli/model_costs_test.go index 2f07c8ac835..12cddccf40b 100644 --- a/pkg/cli/model_costs_test.go +++ b/pkg/cli/model_costs_test.go @@ -153,3 +153,15 @@ func TestComputeModelInferenceAICGitHubCopilotNoCacheRead(t *testing.T) { assert.InDelta(t, aicViaAnthropic, aicViaGitHubCopilot, 1e-9, "zero cache reads must not alter the charged input token count") } + +func TestComputeModelInferenceAICExplicitCacheSemantics(t *testing.T) { + t.Parallel() + + inclusive := true + additive := false + inclusiveAIC := computeModelInferenceAICWithCacheSemantics("copilot", "claude-sonnet-4.6", 1000, 100, 400, 100, 0, &inclusive) + additiveAIC := computeModelInferenceAICWithCacheSemantics("copilot", "claude-sonnet-4.6", 1000, 100, 400, 100, 0, &additive) + + assert.InDelta(t, 0.3495, inclusiveAIC, 1e-9) + assert.InDelta(t, 0.4995, additiveAIC, 1e-9) +} diff --git a/pkg/cli/token_usage_agent_file.go b/pkg/cli/token_usage_agent_file.go new file mode 100644 index 00000000000..f8b3d4a2d8e --- /dev/null +++ b/pkg/cli/token_usage_agent_file.go @@ -0,0 +1,104 @@ +package cli + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" + "strings" +) + +func parseAgentUsageFile(filePath string) (*TokenUsageSummary, error) { + data, err := os.ReadFile(filepath.Clean(filePath)) + if err != nil { + return nil, fmt.Errorf("failed to read agent usage file: %w", err) + } + var entry agentUsageEntry + if err := json.Unmarshal(data, &entry); err != nil { + return nil, fmt.Errorf("failed to parse agent usage file: %w", err) + } + summary := buildAgentUsageSummary(entry) + tokenUsageLog.Printf("Parsed agent usage file: input=%d, output=%d, cache_read=%d, cache_write=%d", + summary.TotalInputTokens, summary.TotalOutputTokens, summary.TotalCacheReadTokens, summary.TotalCacheWriteTokens) + return summary, nil +} + +func resolveAgentUsageModel(entry agentUsageEntry) string { + model := strings.TrimSpace(entry.PrimaryModel) + if model == "" { + model = strings.TrimSpace(entry.Model) + } + if model == "" { + return "unknown" + } + return model +} + +func buildAgentUsageSummary(entry agentUsageEntry) *TokenUsageSummary { + model := resolveAgentUsageModel(entry) + provider := strings.TrimSpace(entry.Provider) + summary := &TokenUsageSummary{ + TotalInputTokens: entry.InputTokens, + TotalOutputTokens: entry.OutputTokens, + TotalCacheReadTokens: entry.CacheReadTokens, + TotalCacheWriteTokens: entry.CacheWriteTokens, + ByModel: make(map[string]*ModelTokenUsage), + } + hasRawTokenData := entry.InputTokens > 0 || + entry.OutputTokens > 0 || + entry.CacheReadTokens > 0 || + entry.CacheWriteTokens > 0 || + entry.ReasoningTokens > 0 + if hasRawTokenData { + summary.TotalRequests = 1 + summary.ByModel[model] = buildAgentModelTokenUsage(entry, provider) + } + ambientInputTokens := entry.InputTokens + if entry.AmbientContextTokens != nil { + ambientInputTokens = *entry.AmbientContextTokens + } + summary.AmbientContext = &AmbientContextMetrics{ + InputTokens: ambientInputTokens, + CachedTokens: entry.CacheReadTokens, + } + populateAgentUsageAIC(summary, entry, model, provider, hasRawTokenData) + return summary +} + +func buildAgentModelTokenUsage(entry agentUsageEntry, provider string) *ModelTokenUsage { + return &ModelTokenUsage{ + Provider: provider, + TokenCoreMetrics: TokenCoreMetrics{ + InputTokens: entry.InputTokens, + OutputTokens: entry.OutputTokens, + CacheReadTokens: entry.CacheReadTokens, + CacheWriteTokens: entry.CacheWriteTokens, + ReasoningTokens: entry.ReasoningTokens, + }, + Requests: 1, + } +} + +func populateAgentUsageAIC(summary *TokenUsageSummary, entry agentUsageEntry, model, provider string, hasRawTokenData bool) { + aic, present, valid := parseOptionalNonNegativeFloat(entry.AICredits) + if !present || !valid { + if hasRawTokenData { + populateAIC(summary) + summary.AICFound = summary.TotalAIC > 0 + } + return + } + summary.TotalAIC = aic + summary.AICFound = true + if summary.ByModel[model] == nil { + summary.ByModel[model] = &ModelTokenUsage{} + } + usage := summary.ByModel[model] + usage.Provider = provider + usage.InputTokens = entry.InputTokens + usage.OutputTokens = entry.OutputTokens + usage.CacheReadTokens = entry.CacheReadTokens + usage.CacheWriteTokens = entry.CacheWriteTokens + usage.ReasoningTokens = entry.ReasoningTokens + usage.AIC = aic +} diff --git a/pkg/cli/token_usage_aic.go b/pkg/cli/token_usage_aic.go new file mode 100644 index 00000000000..fb1db26c3f9 --- /dev/null +++ b/pkg/cli/token_usage_aic.go @@ -0,0 +1,161 @@ +package cli + +import ( + "bytes" + "encoding/json" + "fmt" + "math" + "slices" +) + +type tokenUsageAICFieldState struct { + hasReportedFields bool + hasValidReportedFields bool + hasExplicitCacheSemantics bool +} + +func parseOptionalNonNegativeFloat(raw json.RawMessage) (value float64, present, valid bool) { + if len(raw) == 0 { + return 0, false, false + } + if bytes.Equal(bytes.TrimSpace(raw), []byte("null")) { + return 0, true, false + } + if err := json.Unmarshal(raw, &value); err != nil || math.IsNaN(value) || math.IsInf(value, 0) || value < 0 { + return 0, true, false + } + return value, true, true +} + +func parseOptionalBool(raw json.RawMessage) (*bool, bool, bool) { + if len(raw) == 0 || bytes.Equal(bytes.TrimSpace(raw), []byte("null")) { + return nil, false, false + } + var value bool + if err := json.Unmarshal(raw, &value); err != nil { + return nil, true, true + } + return &value, true, false +} + +func inspectTokenUsageAICFields(entries []TokenUsageEntry) tokenUsageAICFieldState { + state := tokenUsageAICFieldState{} + for _, entry := range entries { + if len(entry.AICreditsThisResponse) > 0 || len(entry.AICreditsTotal) > 0 { + state.hasReportedFields = true + } + if _, _, deltaValid := parseOptionalNonNegativeFloat(entry.AICreditsThisResponse); deltaValid { + state.hasValidReportedFields = true + } + if _, _, totalValid := parseOptionalNonNegativeFloat(entry.AICreditsTotal); totalValid { + state.hasValidReportedFields = true + } + if value, present, _ := parseOptionalBool(entry.InputTokensIncludeCache); present { + if value != nil { + state.hasExplicitCacheSemantics = true + } + } + } + return state +} + +func orderTokenUsageEntriesForAIC(entries []TokenUsageEntry) []TokenUsageEntry { + ordered := slices.Clone(entries) + slices.SortStableFunc(ordered, func(left, right TokenUsageEntry) int { + leftTimestamp, leftValid := parseTokenUsageTimestamp(left.Timestamp) + rightTimestamp, rightValid := parseTokenUsageTimestamp(right.Timestamp) + if leftValid && rightValid { + return leftTimestamp.Compare(rightTimestamp) + } + if leftValid { + return -1 + } + if rightValid { + return 1 + } + return 0 + }) + return ordered +} + +func applyTokenUsageAICEntries(summary *TokenUsageSummary, entries []TokenUsageEntry, hasReportedFields bool) (runningAIC float64, fallbackRecordCount int, invalidCacheSemanticsCount int) { + for _, entry := range orderTokenUsageEntriesForAIC(entries) { + model := entry.Model + if model == "" { + model = "unknown" + } + reportedDelta, deltaPresent, deltaValid := parseOptionalNonNegativeFloat(entry.AICreditsThisResponse) + reportedTotal, totalPresent, totalValid := parseOptionalNonNegativeFloat(entry.AICreditsTotal) + inputTokensIncludeCache, _, invalidCacheSemantics := parseOptionalBool(entry.InputTokensIncludeCache) + if hasReportedFields && (!deltaPresent || !deltaValid || !totalPresent || !totalValid) { + fallbackRecordCount++ + } + deltaAIC := reportedDelta + if !deltaValid { + if invalidCacheSemantics { + invalidCacheSemanticsCount++ + } + deltaAIC = computeModelInferenceAICWithCacheSemantics(entry.Provider, model, entry.InputTokens, entry.OutputTokens, entry.CacheReadTokens, entry.CacheWriteTokens, entry.ReasoningTokens, inputTokensIncludeCache) + } + if usage := summary.ByModel[model]; usage != nil { + usage.AIC += deltaAIC + } + if totalValid { + runningAIC = reportedTotal + } else { + runningAIC += deltaAIC + } + } + return runningAIC, fallbackRecordCount, invalidCacheSemanticsCount +} + +func appendTokenUsageAICWarnings(summary *TokenUsageSummary, state tokenUsageAICFieldState, fallbackRecordCount int, invalidCacheSemanticsCount int) { + if invalidCacheSemanticsCount > 0 { + addTokenUsageWarning(summary, fmt.Sprintf("%d token usage record(s) had invalid input_tokens_include_cache values; legacy provider cache semantics were used.", invalidCacheSemanticsCount)) + } + if fallbackRecordCount > 0 { + addTokenUsageWarning(summary, fmt.Sprintf("%d token usage record(s) had missing or invalid AWF-reported AI Credits fields; fallback accounting was used for the missing values.", fallbackRecordCount)) + } + summedDeltaAIC := 0.0 + for _, usage := range summary.ByModel { + if usage != nil { + summedDeltaAIC += usage.AIC + } + } + if state.hasReportedFields && math.Abs(summedDeltaAIC-summary.TotalAIC) > 1e-6*max(1, math.Abs(summary.TotalAIC)) { + addTokenUsageWarning(summary, "The AWF-reported cumulative AI Credits total differs from the sum of per-request credits; the cumulative total was preserved for reporting.") + } +} + +// populateAICFromTokenUsageEntries uses AWF-computed fields only for usage +// reporting. Budget enforcement and failure classification use separate paths. +func populateAICFromTokenUsageEntries(summary *TokenUsageSummary, entries []TokenUsageEntry) { + if summary == nil { + return + } + state := inspectTokenUsageAICFields(entries) + if !state.hasReportedFields && !state.hasExplicitCacheSemantics { + populateAIC(summary) + invalidCacheSemanticsCount := 0 + for _, entry := range entries { + if _, _, invalid := parseOptionalBool(entry.InputTokensIncludeCache); invalid { + invalidCacheSemanticsCount++ + } + } + summary.AICFound = summary.TotalAIC > 0 + appendTokenUsageAICWarnings(summary, state, 0, invalidCacheSemanticsCount) + return + } + for _, usage := range summary.ByModel { + if usage != nil { + usage.AIC = 0 + } + } + var fallbackRecordCount int + var invalidCacheSemanticsCount int + summary.TotalAIC, fallbackRecordCount, invalidCacheSemanticsCount = applyTokenUsageAICEntries(summary, entries, state.hasReportedFields) + // Any valid reported AWF credit field, including an explicit zero, means AIC + // data was found and callers must not fall back to repricing other artifacts. + summary.AICFound = state.hasValidReportedFields || summary.TotalAIC > 0 + appendTokenUsageAICWarnings(summary, state, fallbackRecordCount, invalidCacheSemanticsCount) +} diff --git a/pkg/cli/token_usage_aic_files.go b/pkg/cli/token_usage_aic_files.go new file mode 100644 index 00000000000..f9a1b27732b --- /dev/null +++ b/pkg/cli/token_usage_aic_files.go @@ -0,0 +1,137 @@ +package cli + +import ( + "bufio" + "encoding/json" + "fmt" + "maps" + "os" + "path/filepath" + "slices" + "strings" +) + +func sumAICFromUsageJSONLFiles(filePaths []string) (float64, bool, error) { + totalAIC, found, _, err := sumAICFromUsageJSONLFilesWithWarnings(filePaths) + return totalAIC, found, err +} + +func sumAICFromUsageJSONLFilesWithWarnings(filePaths []string) (float64, bool, []string, error) { + var totalAIC float64 + found := false + warnings := make([]string, 0) + awfEntries := make([]TokenUsageEntry, 0) + awfDuplicateRecordCount := 0 + seenAWFRequestIDs := make(map[string]struct{}) + for _, filePath := range filePaths { + if isKnownAWFTokenUsageJSONLFile(filePath) { + entries, duplicateRecordCount, err := scanKnownAWFTokenUsageJSONLFile(filePath, seenAWFRequestIDs) + if err != nil { + return 0, false, nil, err + } + awfEntries = append(awfEntries, entries...) + awfDuplicateRecordCount += duplicateRecordCount + continue + } + + candidateSeenRequestIDs := maps.Clone(seenAWFRequestIDs) + entries, duplicateRecordCount, awfSchemaRecordFound, err := scanTokenUsageEntriesWithSeen(filePath, candidateSeenRequestIDs) + if err != nil { + return 0, false, nil, err + } + if awfSchemaRecordFound { + // Once a file matches the AWF token-usage schema, parse the file as one + // AWF stream. Older token-usage records may lack _schema/event but still + // belong to the same stream and should share request deduplication. + seenAWFRequestIDs = candidateSeenRequestIDs + awfEntries = append(awfEntries, entries...) + awfDuplicateRecordCount += duplicateRecordCount + continue + } + + fileAIC, fileFound, err := processLegacyUsageJSONLFile(filePath) + if err != nil { + return 0, false, nil, err + } + totalAIC += fileAIC + found = found || fileFound + } + + if len(awfEntries) > 0 { + summary := buildTokenUsageSummary(awfEntries, awfDuplicateRecordCount) + if summary != nil { + totalAIC += summary.TotalAIC + found = found || summary.AICFound + for _, warning := range summary.Warnings { + if !slices.Contains(warnings, warning) { + warnings = append(warnings, warning) + } + } + } + } + return totalAIC, found, warnings, nil +} + +func isKnownAWFTokenUsageJSONLFile(filePath string) bool { + return strings.EqualFold(filepath.Base(filePath), "token_usage.jsonl") || + strings.EqualFold(filepath.Base(filePath), "token-usage.jsonl") +} + +func scanKnownAWFTokenUsageJSONLFile(filePath string, seenRequestIDs map[string]struct{}) ([]TokenUsageEntry, int, error) { + entries, duplicateRecordCount, _, err := scanTokenUsageEntriesWithSeen(filePath, seenRequestIDs) + if err != nil { + return nil, 0, err + } + return entries, duplicateRecordCount, nil +} + +func processLegacyUsageJSONLFile(filePath string) (total float64, found bool, err error) { + file, err := os.Open(filepath.Clean(filePath)) + if err != nil { + return 0, false, fmt.Errorf("failed to open usage JSONL file %s: %w", filePath, err) + } + defer func() { + if closeErr := file.Close(); closeErr != nil && err == nil { + err = fmt.Errorf("failed to close usage JSONL file %s: %w", filePath, closeErr) + } + }() + + scanner := bufio.NewScanner(file) + scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) + for scanner.Scan() { + line := strings.TrimSpace(scanner.Text()) + if line == "" || !strings.HasPrefix(line, "{") { + continue + } + recordAIC, recordFound := parseLegacyUsageJSONLAIC(line) + total += recordAIC + found = found || recordFound + } + if scanErr := scanner.Err(); scanErr != nil { + return 0, false, fmt.Errorf("error reading usage JSONL file %s: %w", filePath, scanErr) + } + return total, found, nil +} + +func parseLegacyUsageJSONLAIC(line string) (float64, bool) { + var parsed map[string]any + if err := json.Unmarshal([]byte(line), &parsed); err != nil { + return 0, false + } + usage := extractUsageRecord(parsed["usage"]) + for _, keys := range [][]string{{"ai_credits", "aiCredits"}, {"aic"}} { + if value := usageNumericValue(parsed, usage, keys...); value > 0 { + return value, true + } + } + computedAIC := computeModelInferenceAIC( + usageStringValue(parsed, usage, "provider"), + usageStringValue(parsed, usage, "model"), + int(usageNumericValue(parsed, usage, "input_tokens", "inputTokens")), + int(usageNumericValue(parsed, usage, "output_tokens", "outputTokens")), + int(usageNumericValue(parsed, usage, "cache_read_tokens", "cacheReadTokens")), + int(usageNumericValue(parsed, usage, "cache_write_tokens", "cacheWriteTokens")), + int(usageNumericValue(parsed, usage, "reasoning_tokens", "reasoningTokens")), + ) + return computedAIC, computedAIC > 0 +} diff --git a/pkg/cli/token_usage_analyze.go b/pkg/cli/token_usage_analyze.go index 82f0af3eaf3..506211ee243 100644 --- a/pkg/cli/token_usage_analyze.go +++ b/pkg/cli/token_usage_analyze.go @@ -62,12 +62,15 @@ func analyzeTokenUsageAICOnly(runDir string, verbose bool) (*TokenUsageSummary, usageJSONLFiles := findUsageJSONLFiles(runDir) if len(usageJSONLFiles) > 0 { console.LogVerbose(verbose, " Found usage JSONL files: "+strings.Join(usageJSONLFiles, ", ")) - totalAIC, found, err := sumAICFromUsageJSONLFiles(usageJSONLFiles) + totalAIC, found, warnings, err := sumAICFromUsageJSONLFilesWithWarnings(usageJSONLFiles) if err != nil { return nil, err } if found { - return &TokenUsageSummary{TotalAIC: totalAIC}, nil + for _, warning := range warnings { + tokenUsageLog.Printf("AIC-only analysis warning: %s", warning) + } + return &TokenUsageSummary{TotalAIC: totalAIC, Warnings: warnings}, nil } } @@ -78,22 +81,17 @@ func analyzeTokenUsageAICOnly(runDir string, verbose bool) (*TokenUsageSummary, console.LogVerbose(verbose, fmt.Sprintf(" Found token usage file: %s (%d bytes)", filepath.Base(filePath), fileInfo.Size())) } - entries, err := scanTokenUsageEntries(filePath) + summary, err := parseTokenUsageFile(filePath) if err != nil { return nil, err } - if len(entries) == 0 { + if summary == nil || !summary.AICFound { goto fallback } - totalAIC := 0.0 - for _, entry := range entries { - model := entry.Model - if model == "" { - model = "unknown" - } - totalAIC += computeModelInferenceAIC(entry.Provider, model, entry.InputTokens, entry.OutputTokens, entry.CacheReadTokens, entry.CacheWriteTokens, entry.ReasoningTokens) + for _, warning := range summary.Warnings { + tokenUsageLog.Printf("AIC-only analysis warning: %s", warning) } - return &TokenUsageSummary{TotalAIC: totalAIC}, nil + return &TokenUsageSummary{TotalAIC: summary.TotalAIC, Warnings: summary.Warnings}, nil } fallback: diff --git a/pkg/cli/token_usage_parse.go b/pkg/cli/token_usage_parse.go index 4d11eb4a34c..75c542d70ca 100644 --- a/pkg/cli/token_usage_parse.go +++ b/pkg/cli/token_usage_parse.go @@ -6,7 +6,6 @@ import ( "fmt" "math" "os" - "path/filepath" "slices" "strings" "time" @@ -16,17 +15,20 @@ import ( func parseTokenUsageFile(filePath string) (*TokenUsageSummary, error) { tokenUsageLog.Printf("Parsing token usage file: %s", filePath) - summary := &TokenUsageSummary{ - ByModel: make(map[string]*ModelTokenUsage), - } - - entries, err := scanTokenUsageEntries(filePath) + entries, duplicateRecordCount, err := scanTokenUsageEntries(filePath) if err != nil { return nil, err } + return buildTokenUsageSummary(entries, duplicateRecordCount), nil +} + +func buildTokenUsageSummary(entries []TokenUsageEntry, duplicateRecordCount int) *TokenUsageSummary { + summary := &TokenUsageSummary{ + ByModel: make(map[string]*ModelTokenUsage), + } if len(entries) == 0 { tokenUsageLog.Print("No token usage entries found") - return nil, nil + return nil } for _, entry := range entries { @@ -64,16 +66,24 @@ func parseTokenUsageFile(filePath string) (*TokenUsageSummary, error) { len(entries), summary.TotalInputTokens, summary.TotalOutputTokens, summary.TotalCacheReadTokens, summary.TotalCacheWriteTokens, summary.TotalRequests) - populateAIC(summary) + populateAICFromTokenUsageEntries(summary, entries) + if duplicateRecordCount > 0 { + addTokenUsageWarning(summary, fmt.Sprintf("%d duplicate token usage record(s) were ignored by event and request_id.", duplicateRecordCount)) + } summary.AmbientContext = extractAmbientContextMetrics(entries) - return summary, nil + return summary } -func scanTokenUsageEntries(filePath string) ([]TokenUsageEntry, error) { +func scanTokenUsageEntries(filePath string) ([]TokenUsageEntry, int, error) { + entries, duplicateRecordCount, _, err := scanTokenUsageEntriesWithSeen(filePath, nil) + return entries, duplicateRecordCount, err +} + +func scanTokenUsageEntriesWithSeen(filePath string, seenRequestIDs map[string]struct{}) ([]TokenUsageEntry, int, bool, error) { file, err := os.Open(filePath) if err != nil { - return nil, fmt.Errorf("failed to open token usage file: %w", err) + return nil, 0, false, fmt.Errorf("failed to open token usage file: %w", err) } defer file.Close() @@ -81,7 +91,12 @@ func scanTokenUsageEntries(filePath string) ([]TokenUsageEntry, error) { scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) entries := make([]TokenUsageEntry, 0) + if seenRequestIDs == nil { + seenRequestIDs = make(map[string]struct{}) + } + duplicateRecordCount := 0 lineNum := 0 + awfSchemaRecordFound := false for scanner.Scan() { lineNum++ line := strings.TrimSpace(scanner.Text()) @@ -94,98 +109,31 @@ func scanTokenUsageEntries(filePath string) ([]TokenUsageEntry, error) { tokenUsageLog.Printf("Skipping invalid JSON at line %d: %v", lineNum, err) continue } + if strings.HasPrefix(entry.Schema, "token-usage/") || entry.Event == "token_usage" { + awfSchemaRecordFound = true + } + if entry.RequestID != "" { + // AWF defines request_id as unique per API request. Include the event + // discriminator so additive future record types cannot collide. + eventName := entry.Event + if eventName == "" { + eventName = "token_usage" + } + dedupeKey := eventName + ":" + entry.RequestID + if _, exists := seenRequestIDs[dedupeKey]; exists { + tokenUsageLog.Printf("Skipping duplicate request_id at line %d: %s", lineNum, entry.RequestID) + duplicateRecordCount++ + continue + } + seenRequestIDs[dedupeKey] = struct{}{} + } entries = append(entries, entry) } if err := scanner.Err(); err != nil { - return nil, fmt.Errorf("error reading token usage file: %w", err) + return nil, duplicateRecordCount, awfSchemaRecordFound, fmt.Errorf("error reading token usage file: %w", err) } - return entries, nil -} - -func parseAgentUsageFile(filePath string) (*TokenUsageSummary, error) { - cleanPath := filepath.Clean(filePath) - data, err := os.ReadFile(cleanPath) - if err != nil { - return nil, fmt.Errorf("failed to read agent usage file: %w", err) - } - - var entry agentUsageEntry - if err := json.Unmarshal(data, &entry); err != nil { - return nil, fmt.Errorf("failed to parse agent usage file: %w", err) - } - - // Prefer primary_model when set; fall back to model; default to "unknown". - model := strings.TrimSpace(entry.PrimaryModel) - if model == "" { - model = strings.TrimSpace(entry.Model) - } - if model == "" { - model = "unknown" - } - // Prefer provider from entry; primary_model entries may omit it. - provider := strings.TrimSpace(entry.Provider) - - summary := &TokenUsageSummary{ - TotalInputTokens: entry.InputTokens, - TotalOutputTokens: entry.OutputTokens, - TotalCacheReadTokens: entry.CacheReadTokens, - TotalCacheWriteTokens: entry.CacheWriteTokens, - ByModel: make(map[string]*ModelTokenUsage), - } - - hasRawTokenData := summary.TotalInputTokens > 0 || - summary.TotalOutputTokens > 0 || - summary.TotalCacheReadTokens > 0 || - summary.TotalCacheWriteTokens > 0 || - entry.ReasoningTokens > 0 - hasTokenData := hasRawTokenData - if hasTokenData { - summary.TotalRequests = 1 - summary.ByModel[model] = &ModelTokenUsage{ - Provider: provider, - TokenCoreMetrics: TokenCoreMetrics{ - InputTokens: entry.InputTokens, - OutputTokens: entry.OutputTokens, - CacheReadTokens: entry.CacheReadTokens, - CacheWriteTokens: entry.CacheWriteTokens, - ReasoningTokens: entry.ReasoningTokens, - }, - Requests: 1, - } - } - - ambientInputTokens := entry.InputTokens - if entry.AmbientContextTokens != nil { - ambientInputTokens = *entry.AmbientContextTokens - } - summary.AmbientContext = &AmbientContextMetrics{ - InputTokens: ambientInputTokens, - CachedTokens: entry.CacheReadTokens, - } - - if entry.AICredits > 0 { - // Use the pre-computed AI Credits value written by parse_token_usage.cjs. - // This is more accurate than recomputing from raw token counts because it - // was computed at the time the run completed with full per-request pricing. - summary.TotalAIC = entry.AICredits - if summary.ByModel[model] == nil { - summary.ByModel[model] = &ModelTokenUsage{} - } - summary.ByModel[model].Provider = provider - summary.ByModel[model].InputTokens = entry.InputTokens - summary.ByModel[model].OutputTokens = entry.OutputTokens - summary.ByModel[model].CacheReadTokens = entry.CacheReadTokens - summary.ByModel[model].CacheWriteTokens = entry.CacheWriteTokens - summary.ByModel[model].ReasoningTokens = entry.ReasoningTokens - summary.ByModel[model].AIC = entry.AICredits - } else if hasRawTokenData { - populateAIC(summary) - } - - tokenUsageLog.Printf("Parsed agent usage file: input=%d, output=%d, cache_read=%d, cache_write=%d", - summary.TotalInputTokens, summary.TotalOutputTokens, summary.TotalCacheReadTokens, summary.TotalCacheWriteTokens) - return summary, nil + return entries, duplicateRecordCount, awfSchemaRecordFound, nil } func extractUsageRecord(value any) map[string]any { @@ -242,84 +190,6 @@ func isFinite(value float64) bool { return !math.IsNaN(value) && !math.IsInf(value, 0) } -func sumAICFromUsageJSONLFiles(filePaths []string) (float64, bool, error) { - var totalAIC float64 - found := false - - for _, filePath := range filePaths { - fileAIC, fileFound, err := processOneUsageJSONLFile(filePath) - if err != nil { - return 0, false, err - } - totalAIC += fileAIC - if fileFound { - found = true - } - } - - return totalAIC, found, nil -} - -// processOneUsageJSONLFile reads a single usage JSONL file and returns the total AIC -// accumulated from its records. The file is deferred-closed immediately after open. -func processOneUsageJSONLFile(filePath string) (total float64, found bool, err error) { - file, err := os.Open(filepath.Clean(filePath)) - if err != nil { - return 0, false, fmt.Errorf("failed to open usage JSONL file %s: %w", filePath, err) - } - defer func() { - if closeErr := file.Close(); closeErr != nil && err == nil { - err = fmt.Errorf("failed to close usage JSONL file %s: %w", filePath, closeErr) - } - }() - - scanner := bufio.NewScanner(file) - scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) - for scanner.Scan() { - line := strings.TrimSpace(scanner.Text()) - if line == "" || !strings.HasPrefix(line, "{") { - continue - } - - var parsed map[string]any - if jsonErr := json.Unmarshal([]byte(line), &parsed); jsonErr != nil { - continue - } - - usage := extractUsageRecord(parsed["usage"]) - explicitAICredits := usageNumericValue(parsed, usage, "ai_credits", "aiCredits") - if explicitAICredits > 0 { - total += explicitAICredits - found = true - continue - } - explicitAIC := usageNumericValue(parsed, usage, "aic") - if explicitAIC > 0 { - total += explicitAIC - found = true - continue - } - - computedAIC := computeModelInferenceAIC( - usageStringValue(parsed, usage, "provider"), - usageStringValue(parsed, usage, "model"), - int(usageNumericValue(parsed, usage, "input_tokens", "inputTokens")), - int(usageNumericValue(parsed, usage, "output_tokens", "outputTokens")), - int(usageNumericValue(parsed, usage, "cache_read_tokens", "cacheReadTokens")), - int(usageNumericValue(parsed, usage, "cache_write_tokens", "cacheWriteTokens")), - int(usageNumericValue(parsed, usage, "reasoning_tokens", "reasoningTokens")), - ) - if computedAIC > 0 { - total += computedAIC - found = true - } - } - if scanErr := scanner.Err(); scanErr != nil { - return 0, false, fmt.Errorf("error reading usage JSONL file %s: %w", filePath, scanErr) - } - return total, found, nil -} - func extractAmbientContextMetrics(entries []TokenUsageEntry) *AmbientContextMetrics { if len(entries) == 0 { return nil diff --git a/pkg/cli/token_usage_test.go b/pkg/cli/token_usage_test.go index 815da9e4b10..e30585c9c04 100644 --- a/pkg/cli/token_usage_test.go +++ b/pkg/cli/token_usage_test.go @@ -4,6 +4,7 @@ package cli import ( "encoding/json" + "fmt" "math" "os" "path/filepath" @@ -73,6 +74,236 @@ func TestParseTokenUsageFile(t *testing.T) { assert.InDelta(t, 0.0, summary.CacheEfficiency, 0.001, "cache efficiency is not computed from raw token counts") }) + t.Run("prefers exact AWF-reported credits", func(t *testing.T) { + fixturePath := filepath.Join("..", "..", "actions", "setup", "js", "fixtures", "awf-v0.28.7-aic-token-usage.jsonl") + + summary, err := parseTokenUsageFile(fixturePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.Equal(t, 5, summary.TotalRequests) + assert.InDelta(t, 1.03602, summary.TotalAIC, 1e-9) + require.Contains(t, summary.ByModel, "gpt-4o-mini-2024-07-18") + assert.InDelta(t, 1.03602, summary.ByModel["gpt-4o-mini-2024-07-18"].AIC, 1e-9) + assert.Empty(t, summary.Warnings) + }) + + t.Run("retains legacy repricing when AWF-reported fields are absent", func(t *testing.T) { + fixture, err := os.ReadFile(filepath.Join("..", "..", "actions", "setup", "js", "fixtures", "awf-v0.28.7-aic-token-usage.jsonl")) + require.NoError(t, err) + legacyLines := make([]string, 0, 5) + for line := range strings.SplitSeq(strings.TrimSpace(string(fixture)), "\n") { + var record map[string]any + require.NoError(t, json.Unmarshal([]byte(line), &record)) + delete(record, "ai_credits_this_response") + delete(record, "ai_credits_total") + encoded, marshalErr := json.Marshal(record) + require.NoError(t, marshalErr) + legacyLines = append(legacyLines, string(encoded)) + } + filePath := filepath.Join(testutil.TempDir(t, "token-usage-legacy-aic"), "token-usage.jsonl") + require.NoError(t, os.WriteFile(filePath, []byte(strings.Join(legacyLines, "\n")+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + assert.InDelta(t, 0.44538, summary.TotalAIC, 1e-9) + }) + + t.Run("falls back and warns for malformed AWF-reported credits", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-malformed-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":19288,"output_tokens":35,"cache_read_tokens":0,"cache_write_tokens":0,"ai_credits_this_response":"0.29142","ai_credits_total":-1}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.29142, summary.TotalAIC, 1e-9) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "fallback accounting") + }) + + t.Run("distinguishes zero AWF-reported credits from absent fields", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-zero-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":19288,"output_tokens":35,"cache_read_tokens":0,"cache_write_tokens":0,"ai_credits_this_response":0,"ai_credits_total":0}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.Zero(t, summary.TotalAIC) + assert.Zero(t, summary.ByModel["gpt-4o-mini-2024-07-18"].AIC) + assert.Empty(t, summary.Warnings) + }) + + t.Run("treats null AWF fields as malformed rather than zero", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-null-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":19288,"output_tokens":35,"cache_read_tokens":0,"cache_write_tokens":0,"ai_credits_this_response":null,"ai_credits_total":null}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.29142, summary.TotalAIC, 1e-9) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "fallback accounting") + }) + + t.Run("treats null cache semantics as absent", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-null-cache-semantics") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":1000,"output_tokens":100,"cache_read_tokens":400,"cache_write_tokens":100,"input_tokens_include_cache":null}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.0195, summary.TotalAIC, 1e-9) + assert.Empty(t, summary.Warnings) + }) + + t.Run("uses legacy provider semantics and warns for invalid cache semantics", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-invalid-cache-semantics") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":1000,"output_tokens":100,"cache_read_tokens":400,"cache_write_tokens":100,"input_tokens_include_cache":"invalid"}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.0195, summary.TotalAIC, 1e-9) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "invalid input_tokens_include_cache") + }) + + t.Run("does not warn for invalid cache semantics when reported credits are valid", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-valid-aic-invalid-cache-semantics") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":1000,"output_tokens":100,"cache_read_tokens":400,"cache_write_tokens":100,"input_tokens_include_cache":"invalid","ai_credits_this_response":0.123,"ai_credits_total":0.123}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.123, summary.TotalAIC, 1e-9) + assert.Empty(t, summary.Warnings) + }) + + for _, testCase := range []struct { + name string + inputIncludesCache bool + expectedAIC float64 + }{ + {name: "inclusive cache fields", inputIncludesCache: true, expectedAIC: 0.018}, + {name: "additive cache fields", inputIncludesCache: false, expectedAIC: 0.0255}, + } { + t.Run(testCase.name, func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-explicit-cache-semantics") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := fmt.Sprintf( + `{"provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":1000,"output_tokens":100,"cache_read_tokens":400,"cache_write_tokens":100,"input_tokens_include_cache":%t}`, + testCase.inputIncludesCache, + ) + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, testCase.expectedAIC, summary.TotalAIC, 1e-9) + assert.Empty(t, summary.Warnings) + }) + } + + t.Run("continues from the last valid reported total after malformed data", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-aic-continuation") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"request_id":"one","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":10,"output_tokens":1,"ai_credits_this_response":0.2,"ai_credits_total":0.2} +{"request_id":"two","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":10,"output_tokens":1,"ai_credits_this_response":0.3,"ai_credits_total":"invalid"}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.5, summary.TotalAIC, 1e-9) + assert.InDelta(t, 0.5, summary.ByModel["gpt-4o-mini-2024-07-18"].AIC, 1e-9) + require.Len(t, summary.Warnings, 1) + }) + + t.Run("aggregates reported credits by model while preserving the run total", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-multi-model-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"request_id":"one","provider":"copilot","model":"gpt-4o-mini","input_tokens":10,"output_tokens":1,"ai_credits_this_response":0.2,"ai_credits_total":0.2} +{"request_id":"two","provider":"copilot","model":"claude-sonnet-4-6","input_tokens":20,"output_tokens":2,"ai_credits_this_response":0.8,"ai_credits_total":1}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.2, summary.ByModel["gpt-4o-mini"].AIC, 1e-9) + assert.InDelta(t, 0.8, summary.ByModel["claude-sonnet-4-6"].AIC, 1e-9) + assert.InDelta(t, 1, summary.TotalAIC, 1e-9) + }) + + t.Run("uses the chronologically last valid reported total", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-chronological-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"timestamp":"2026-08-28T09:00:02Z","request_id":"third","provider":"copilot","model":"gpt-4o-mini","input_tokens":1,"output_tokens":1,"ai_credits_this_response":1,"ai_credits_total":3} +{"timestamp":"2026-08-28T09:00:00Z","request_id":"first","provider":"copilot","model":"gpt-4o-mini","input_tokens":1,"output_tokens":1,"ai_credits_this_response":1,"ai_credits_total":1} +{"timestamp":"2026-08-28T09:00:01Z","request_id":"second","provider":"copilot","model":"gpt-4o-mini","input_tokens":1,"output_tokens":1,"ai_credits_this_response":1,"ai_credits_total":2}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 3, summary.TotalAIC, 1e-9) + assert.Empty(t, summary.Warnings) + }) + + t.Run("warns when cumulative and per-request reported credits diverge", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-divergent-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + content := `{"request_id":"one","provider":"copilot","model":"gpt-4o-mini","input_tokens":1,"output_tokens":1,"ai_credits_this_response":0.2,"ai_credits_total":0.2} +{"request_id":"two","provider":"copilot","model":"gpt-4o-mini","input_tokens":1,"output_tokens":1,"ai_credits_this_response":0.8,"ai_credits_total":0.9}` + require.NoError(t, os.WriteFile(filePath, []byte(content+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.InDelta(t, 0.9, summary.TotalAIC, 1e-9) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "differs from the sum") + }) + + t.Run("deduplicates mirrored requests before aggregating AWF-reported credits", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "token-usage-duplicate-aic") + filePath := filepath.Join(tmpDir, "token-usage.jsonl") + record := `{"request_id":"same-request","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":10,"output_tokens":1,"ai_credits_this_response":0.2,"ai_credits_total":0.2}` + require.NoError(t, os.WriteFile(filePath, []byte(record+"\n"+record+"\n"), 0o644)) + + summary, err := parseTokenUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + + assert.Equal(t, 1, summary.TotalRequests) + assert.InDelta(t, 0.2, summary.TotalAIC, 1e-9) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "duplicate token usage") + }) + t.Run("extracts ambient context from first chronological invocation", func(t *testing.T) { tmpDir := testutil.TempDir(t, "token-usage") filePath := filepath.Join(tmpDir, "token-usage.jsonl") @@ -274,6 +505,96 @@ func TestAnalyzeTokenUsageAICOnly(t *testing.T) { require.NotNil(t, summary) assert.InDelta(t, 94.653, summary.TotalAIC, 1e-6) }) + + t.Run("falls back to agent_usage.json when token usage has no priced AIC", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "analyze-token-usage-aic-only-unpriced") + usageDir := filepath.Join(tmpDir, "usage") + agentSubDir := filepath.Join(usageDir, "agent") + require.NoError(t, os.MkdirAll(agentSubDir, 0o755)) + require.NoError(t, os.WriteFile( + filepath.Join(agentSubDir, "token_usage.jsonl"), + []byte(`{"event":"token_usage","provider":"unknown","model":"unpriced","input_tokens":10,"output_tokens":5}`+"\n"), + 0o644, + )) + require.NoError(t, os.WriteFile( + filepath.Join(usageDir, "agent_usage.json"), + []byte(`{"ai_credits":2.5,"primary_model":"unpriced"}`), + 0o644, + )) + + summary, err := analyzeTokenUsageAICOnly(tmpDir, false) + require.NoError(t, err) + require.NotNil(t, summary) + assert.InDelta(t, 2.5, summary.TotalAIC, 1e-9) + }) + + t.Run("preserves AWF-reported totals from token usage artifacts", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "analyze-token-usage-aic-only-awf") + usageDir := filepath.Join(tmpDir, "usage", "agent") + require.NoError(t, os.MkdirAll(usageDir, 0o755)) + fixture, err := os.ReadFile(filepath.Join("..", "..", "actions", "setup", "js", "fixtures", "awf-v0.28.7-aic-token-usage.jsonl")) + require.NoError(t, err) + require.NoError(t, os.WriteFile(filepath.Join(usageDir, "token_usage.jsonl"), fixture, 0o644)) + + summary, err := analyzeTokenUsageAICOnly(tmpDir, false) + require.NoError(t, err) + require.NotNil(t, summary) + assert.InDelta(t, 1.03602, summary.TotalAIC, 1e-9) + }) + + t.Run("propagates token usage fallback warnings", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "analyze-token-usage-aic-only-warnings") + usageDir := filepath.Join(tmpDir, "usage", "agent") + require.NoError(t, os.MkdirAll(usageDir, 0o755)) + content := `{"_schema":"token-usage/v0.28.7","event":"token_usage","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":19288,"output_tokens":35,"ai_credits_this_response":null,"ai_credits_total":null}` + require.NoError(t, os.WriteFile(filepath.Join(usageDir, "token_usage.jsonl"), []byte(content+"\n"), 0o644)) + + summary, err := analyzeTokenUsageAICOnly(tmpDir, false) + require.NoError(t, err) + require.NotNil(t, summary) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "fallback accounting") + }) + + t.Run("preserves zero AWF-reported totals instead of falling back to agent usage", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "analyze-token-usage-aic-only-zero-awf") + usageDir := filepath.Join(tmpDir, "usage", "agent") + require.NoError(t, os.MkdirAll(usageDir, 0o755)) + require.NoError(t, os.WriteFile( + filepath.Join(usageDir, "token_usage.jsonl"), + []byte(`{"_schema":"token-usage/v0.28.7","event":"token_usage","request_id":"zero","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":19288,"output_tokens":35,"ai_credits_this_response":0,"ai_credits_total":0}`+"\n"), + 0o644, + )) + require.NoError(t, os.WriteFile( + filepath.Join(tmpDir, "usage", "agent_usage.json"), + []byte(`{"input_tokens":19288,"output_tokens":35,"ai_credits":2.5,"primary_model":"gpt-4o-mini-2024-07-18"}`), + 0o644, + )) + + summary, err := analyzeTokenUsageAICOnly(tmpDir, false) + require.NoError(t, err) + require.NotNil(t, summary) + assert.Zero(t, summary.TotalAIC) + }) + + t.Run("deduplicates mirrored AWF artifacts while summing distinct legacy usage", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "analyze-token-usage-aic-only-mirrored-awf-and-legacy") + agentDir := filepath.Join(tmpDir, "usage", "agent") + detectionDir := filepath.Join(tmpDir, "usage", "detection") + require.NoError(t, os.MkdirAll(agentDir, 0o755)) + require.NoError(t, os.MkdirAll(detectionDir, 0o755)) + awfRecord := `{"_schema":"token-usage/v0.28.7","event":"token_usage","request_id":"shared-awf","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":10,"output_tokens":1,"ai_credits_this_response":0.2,"ai_credits_total":0.2}` + require.NoError(t, os.WriteFile(filepath.Join(agentDir, "token_usage.jsonl"), []byte(awfRecord+"\n"), 0o644)) + require.NoError(t, os.WriteFile(filepath.Join(detectionDir, "mirrored-token-usage.jsonl"), []byte(awfRecord+"\n"), 0o644)) + require.NoError(t, os.WriteFile(filepath.Join(tmpDir, "usage", "detection_usage.jsonl"), []byte(`{"ai_credits":0.75}`+"\n"), 0o644)) + + summary, err := analyzeTokenUsageAICOnly(tmpDir, false) + require.NoError(t, err) + require.NotNil(t, summary) + assert.InDelta(t, 0.95, summary.TotalAIC, 1e-9) + require.Len(t, summary.Warnings, 1) + assert.Contains(t, summary.Warnings[0], "duplicate token usage") + }) } func TestExtractUsageRecord(t *testing.T) { @@ -326,6 +647,49 @@ func TestSumAICFromUsageJSONLFiles(t *testing.T) { assert.True(t, found) assert.Greater(t, total, 1.25) }) + + t.Run("detects AWF token usage records by schema regardless of filename", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "sum-usage-jsonl-awf-schema") + filePath := filepath.Join(tmpDir, "renamed.jsonl") + fixture, err := os.ReadFile(filepath.Join("..", "..", "actions", "setup", "js", "fixtures", "awf-v0.28.7-aic-token-usage.jsonl")) + require.NoError(t, err) + require.NoError(t, os.WriteFile(filePath, fixture, 0o644)) + + total, found, err := sumAICFromUsageJSONLFiles([]string{filePath}) + require.NoError(t, err) + assert.True(t, found) + assert.InDelta(t, 1.03602, total, 1e-9) + }) + + t.Run("deduplicates mirrored AWF records across files before aggregating", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "sum-usage-jsonl-awf-cross-file-dedupe") + fileOne := filepath.Join(tmpDir, "token_usage.jsonl") + fileTwo := filepath.Join(tmpDir, "mirrored.jsonl") + record := `{"_schema":"token-usage/v0.28.7","event":"token_usage","request_id":"same-request","provider":"copilot","model":"gpt-4o-mini-2024-07-18","input_tokens":10,"output_tokens":1,"ai_credits_this_response":0.2,"ai_credits_total":0.2}` + require.NoError(t, os.WriteFile(fileOne, []byte(record+"\n"), 0o644)) + require.NoError(t, os.WriteFile(fileTwo, []byte(record+"\n"), 0o644)) + + total, found, err := sumAICFromUsageJSONLFiles([]string{fileOne, fileTwo}) + require.NoError(t, err) + assert.True(t, found) + assert.InDelta(t, 0.2, total, 1e-9) + }) + + t.Run("preserves zero ai_credits from agent_usage_json", func(t *testing.T) { + tmpDir := testutil.TempDir(t, "agent-usage-zero-aic") + filePath := filepath.Join(tmpDir, "agent_usage.json") + require.NoError(t, os.WriteFile( + filePath, + []byte(`{"input_tokens":19288,"output_tokens":35,"ai_credits":0,"primary_model":"gpt-4o-mini-2024-07-18"}`), + 0o644, + )) + + summary, err := parseAgentUsageFile(filePath) + require.NoError(t, err) + require.NotNil(t, summary) + assert.True(t, summary.AICFound) + assert.Zero(t, summary.TotalAIC) + }) } func TestTokenUsageSummaryMethods(t *testing.T) { diff --git a/pkg/cli/token_usage_types.go b/pkg/cli/token_usage_types.go index 8a278f5bfdc..3ead72c705a 100644 --- a/pkg/cli/token_usage_types.go +++ b/pkg/cli/token_usage_types.go @@ -1,6 +1,8 @@ package cli import ( + "encoding/json" + "github.com/github/gh-aw/pkg/logger" ) @@ -22,6 +24,7 @@ type TokenCoreMetrics struct { type TokenUsageEntry struct { Schema string `json:"_schema,omitempty"` // Self-describing record type, e.g. "token-usage/v0.26.0" Timestamp string `json:"timestamp"` + Event string `json:"event"` RequestID string `json:"request_id"` Provider string `json:"provider"` Model string `json:"model"` @@ -29,8 +32,11 @@ type TokenUsageEntry struct { Status int `json:"status"` Streaming bool `json:"streaming"` TokenCoreMetrics - DurationMs int `json:"duration_ms"` - ResponseBytes int `json:"response_bytes"` + DurationMs int `json:"duration_ms"` + ResponseBytes int `json:"response_bytes"` + AICreditsThisResponse json.RawMessage `json:"ai_credits_this_response,omitempty"` + AICreditsTotal json.RawMessage `json:"ai_credits_total,omitempty"` + InputTokensIncludeCache json.RawMessage `json:"input_tokens_include_cache,omitempty"` } // AmbientContextMetrics captures token footprint for the first LLM invocation. @@ -53,6 +59,7 @@ type TokenUsageSummary struct { CacheEfficiency float64 `json:"cache_efficiency"` TotalEffectiveTokens int `json:"total_effective_tokens,omitempty"` TotalAIC float64 `json:"total_aic,omitempty"` + AICFound bool `json:"-"` AmbientContext *AmbientContextMetrics `json:"ambient_context,omitempty"` ByModel map[string]*ModelTokenUsage `json:"by_model"` SubagentModelRequests []SubagentModelRequest `json:"subagent_model_requests,omitempty"` @@ -116,8 +123,8 @@ type agentUsageEntry struct { // AmbientContextTokens is the first-request ambient input token count emitted by parse_token_usage.cjs. AmbientContextTokens *int `json:"ambient_context"` // AICredits is the pre-computed total AI Credits value written by parse_token_usage.cjs. - // When present and positive it is used directly so we don't need per-model pricing. - AICredits float64 `json:"ai_credits"` + // When present and valid it is used directly so we don't need per-model pricing. + AICredits json.RawMessage `json:"ai_credits"` } // proxyEventsEntry is a JSONL record from api-proxy-logs/events.jsonl.