Spec Sync #1916
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Spec Sync | |
| on: | |
| schedule: | |
| - cron: '0 * * * *' # hourly; cron covers correctness, dispatch cuts latency | |
| workflow_dispatch: {} | |
| # Serialize runs so a cron tick and a manual dispatch can't race on the sync branch. | |
| concurrency: | |
| group: spec-sync | |
| cancel-in-progress: false | |
| permissions: | |
| contents: read # all writes go through SPEC_SYNC_TOKEN, never GITHUB_TOKEN | |
| env: | |
| # jq program for the "tool-call summary" step that follows each claude-code-action step. It prints | |
| # ONLY fixed metadata: the tool name and the NAMES of the arguments it was given — never argument | |
| # values. A prompt-injected agent that has Read its own environment could hex- or base64-encode a | |
| # credential into a file path, and neither a character allowlist nor GitHub's literal secret | |
| # masking would catch that; the same goes for assistant text, tool results and file contents, so | |
| # none of those are printed either. What the agent actually changed is listed from a trusted | |
| # source instead: the commit step prints `git diff --cached --stat`. Held here, in the trusted | |
| # workflow file, so the summary can run right after the AI step without executing anything the | |
| # agent could have edited. | |
| TOOL_CALL_SUMMARY_JQ: >- | |
| (if type == "array" then .[] else . end) | |
| | select(.type == "assistant") | .message.content[]? | |
| | select(.type == "tool_use") | |
| | "[tool] " + (.name | tostring | if test("^[A-Za-z]{1,32}$") then . else "<redacted>" end) | |
| + " " + ((.input // {}) | keys | map(if test("^[A-Za-z_]{1,32}$") then . else "<redacted>" end) | join(",")) | |
| jobs: | |
| spec-sync: | |
| # Guard so the scheduled/dispatch job only runs on the canonical repo — forks lack the | |
| # secrets and would just produce noisy failing runs. | |
| if: github.repository == 'landing-ai/ade-python' | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 60 # headroom for the AI wiring step (up to --max-turns 250) plus rye sync | |
| env: | |
| V1_SPEC_URL: https://api.va.staging.landing.ai/v1/ade/openapi.json # staging drives the loop | |
| SYNC_BRANCH: spec-sync/v1 # fixed branch: reruns update one PR in place, never a pile of them | |
| SPEC_LABEL: V1 # used only in Slack/PR-comment copy to disambiguate the two jobs | |
| steps: | |
| - uses: actions/checkout@v6 | |
| with: | |
| # SPEC_SYNC_TOKEN is a fine-grained PAT (Contents: RW, Pull requests: RW) scoped to this | |
| # repo. It must NOT be the default GITHUB_TOKEN: pushes/PRs authored by GITHUB_TOKEN do | |
| # not trigger the gate workflows (anti-recursion). | |
| token: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| fetch-depth: 0 # tags needed for surface-lock baseline | |
| - name: Install Rye | |
| run: | | |
| curl -sSf https://rye.astral.sh/get | bash | |
| echo "$HOME/.rye/shims" >> "$GITHUB_PATH" | |
| env: | |
| RYE_VERSION: '0.44.0' | |
| RYE_INSTALL_OPTION: '--yes' | |
| - name: Install dependencies | |
| run: rye sync --all-features | |
| - name: Detect drift | |
| id: drift | |
| run: | | |
| set +e | |
| ./scripts/spec-sync/check-drift.sh "$V1_SPEC_URL" specs/v1-ade.json | |
| echo "code=$?" >> "$GITHUB_OUTPUT" | |
| - name: No drift | |
| if: steps.drift.outputs.code == '0' | |
| run: echo "specs in sync; nothing to do." | |
| # An unavailable spec source (check-drift exit 20) is the EXPECTED case on staging, not an | |
| # incident: staging auto-reclaims and must be booked to come back — an unbooked cluster serves a | |
| # 404 (or the host stops answering entirely). Treat ONLY this as a no-op — log and end the run | |
| # cleanly (no drift, no PR, no Slack); the next hourly run picks up any drift once staging is | |
| # booked. Every other failure (a reachable source returning 401/403/5xx, an empty/invalid spec, | |
| # or a script error) falls through to "Fail on spec error" below and alerts. | |
| - name: Spec source unavailable — skip | |
| if: steps.drift.outputs.code == '20' | |
| run: echo "spec source unavailable (exit 20); staging is likely unbooked — skipping this run." | |
| # A reachable-but-invalid spec (empty/whitespace body, malformed JSON) or a script error is a | |
| # real problem — fail loudly so the catch-all alert fires. Distinct from the exit-20 skip above. | |
| - name: Fail on spec error | |
| if: steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20' | |
| run: | | |
| echo "spec fetch/normalize failed (exit ${{ steps.drift.outputs.code }}); staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored." | |
| exit 1 | |
| # If a sync PR is already open, do nothing: it's awaiting human review, and re-running | |
| # would duplicate gate runs and Claude API spend every hour until it merges. | |
| - name: Check for an open sync PR | |
| id: existing | |
| if: steps.drift.outputs.code == '10' | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: | | |
| if gh pr list --state open --head "$SYNC_BRANCH" --json number --jq '.[0].number' | grep -q .; then | |
| echo "open=true" >> "$GITHUB_OUTPUT" | |
| else | |
| echo "open=false" >> "$GITHUB_OUTPUT" | |
| fi | |
| # A sync PR is already open. Every hourly tick still reports drift (check-drift compares against | |
| # main, not the PR branch), so instead of a bare skip: detect drift that appeared BEYOND the | |
| # open PR and, if any, annotate the PR + emit one Slack ping per distinct new spec (deduped). | |
| - name: New drift behind the open PR | |
| id: behind | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true' | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: ./scripts/spec-sync/notify-new-drift.sh specs/v1-ade.json "$SYNC_BRANCH" "$SPEC_LABEL" | |
| - name: Slack — new drift behind the open PR | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true' && steps.behind.outputs.notify == 'true' | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: warning | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: new drift behind the open PR' | |
| text: 'New ${{ env.SPEC_LABEL }} spec drift appeared beyond the <${{ steps.behind.outputs.pr_url }}|open sync PR> (live-spec `${{ steps.behind.outputs.hash }}`). Merge or close it and the next run opens a fresh one covering the rest.' | |
| thread_ts: ${{ steps.behind.outputs.thread_ts }} | |
| # ---- Phase 1: mechanical (deterministic) ---- | |
| - name: Mechanical commit + push | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| run: | | |
| git config user.name "spec-sync[bot]" | |
| git config user.email "spec-sync@users.noreply.github.com" | |
| git checkout -B "$SYNC_BRANCH" | |
| ./scripts/spec-sync/gen-models.sh specs/v1-ade.json specs/_generated/v1_models.py | |
| git add specs/v1-ade.json specs/_generated/v1_models.py | |
| git commit -m "chore(spec-sync): update V1 spec snapshot + regenerated reference models" | |
| git push --force origin "$SYNC_BRANCH" | |
| # Open the PR now — before the AI step — so a run that dies in phase 2 leaves a visible PR | |
| # (with just the mechanical commit) instead of an orphan branch. | |
| - name: Open sync PR | |
| id: open_pr | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: | | |
| url="$(gh pr create --base main --head "$SYNC_BRANCH" \ | |
| --title "spec-sync: track V1 spec drift" \ | |
| --body $'Automated spec-sync PR.\n\n- **Commit 1 (mechanical):** normalized spec snapshot + regenerated reference models.\n- **Commit 2 (AI):** resources/methods/tests/docs wired from the spec diff (added after this PR opened).\n\nGates (surface-lock, contract tests, lint/test/typecheck) must pass. **Human review required before merge.**')" | |
| echo "url=$url" >> "$GITHUB_OUTPUT" | |
| # Post the Slack thread ROOT now — right after the PR exists, before AI wiring — so every later | |
| # event (AI outcome, new drift, gates, merge) replies under it. Persist its ts on the PR body | |
| # so other runs/workflows can find the thread. (Editing the body doesn't re-trigger pr-gates, | |
| # which only fires on opened/synchronize/reopened.) | |
| - name: Slack — PR opened (thread root) | |
| id: root | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: info | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: new spec drift → PR opened' | |
| text: '<${{ steps.open_pr.outputs.url }}|Open the PR> — mechanical spec snapshot committed; wiring the SDK next. Human review required before merge.' | |
| - name: Save thread root on the PR | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != '' | |
| continue-on-error: true # bookkeeping only; thread-ts.sh fails closed, and a hiccup must not red the run | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: ./scripts/spec-sync/thread-ts.sh set "${{ steps.open_pr.outputs.url }}" "${{ steps.root.outputs.ts }}" | |
| # ---- Phase 2: AI wiring — edits ONLY. The agent holds no git/push capability, so a | |
| # prompt-injected spec description cannot route around review to push to a branch. ---- | |
| - name: AI wiring (edits only) | |
| id: ai_wiring | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| uses: anthropics/claude-code-action@v1 | |
| with: | |
| anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} | |
| github_token: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| prompt: | | |
| The previous commit updated specs/v1-ade.json and regenerated reference models in | |
| specs/_generated/v1_models.py. Wire the SDK to match the spec diff. | |
| First inspect the mechanical commit diff (`git diff HEAD~1..HEAD`) and read | |
| src/landingai_ade/resources/parse_jobs.py — mirror its structure exactly for any | |
| new resource (sync + async classes, the raw/streaming response wrappers, param | |
| TypedDicts under src/landingai_ade/types/, response models, and registration in | |
| src/landingai_ade/resources/__init__.py and the client). Add tests under | |
| tests/api_resources/ and update api.md and the README examples. Then run | |
| `./scripts/format` and `./scripts/lint` and fix anything they report. | |
| Rules: PURELY ADDITIVE — never modify or remove an existing public signature (the | |
| surface-lock CI job will fail the PR). `git diff` (read-only, to inspect the change) | |
| is fine; do NOT stage, commit, push, or run any other git command — a later workflow | |
| step commits and pushes your edits. | |
| # Pin the model: the action otherwise inherits Claude Code's default, which floats | |
| # across CLI releases and makes spec-sync output non-reproducible. `[1m]` = 1M context. | |
| # claude-opus-5 is the pin in BOTH SDK repos (ade-python and ade-typescript) — keep them | |
| # equal, or the two SDKs quietly diverge on the same spec change. | |
| claude_args: | | |
| --model "claude-opus-5[1m]" | |
| --max-turns 250 | |
| --allowedTools "Edit,Write,Read,Glob,Grep,Bash(rye *),Bash(./scripts/*),Bash(git diff:*)" | |
| # Print the agent's tool calls — tool name and argument NAMES only, never values — to the step | |
| # log, so the shape of what the agent did can be audited later (the commit step lists what it | |
| # changed, from the trusted checkout): | |
| # the action logs only the init and result summaries, and #153 had to be root-caused without | |
| # any trace. Deliberately NO argument values, assistant text, tool results or commands: this agent | |
| # holds credential-bearing inputs and processes an untrusted spec, and a prompt-injected agent | |
| # could carry a transformed or chunked credential past GitHub's literal secret masking in | |
| # free-form narration. Runs straight off the action's own output file with the jq program held | |
| # in this workflow's top-level env, so it executes nothing the agent could have edited. | |
| # Best-effort. | |
| - name: AI wiring tool-call summary | |
| if: always() && steps.ai_wiring.outputs.execution_file != '' | |
| continue-on-error: true | |
| env: | |
| TRANSCRIPT: ${{ steps.ai_wiring.outputs.execution_file }} | |
| run: | | |
| echo "::group::AI wiring tool-call summary" | |
| set -o pipefail | |
| # Prefix every line: nothing the agent could influence may start at column 0, where a leading | |
| # `::` would be parsed as a workflow command (e.g. ::stop-commands::) even inside a group. | |
| jq -r "$TOOL_CALL_SUMMARY_JQ" "$TRANSCRIPT" | tr -d '\r' | sed 's/^/ /' || echo " tool-call summary unavailable (unexpected file shape)" | |
| echo "::endgroup::" | |
| # Deterministic: commit whatever the AI edited and push to the sync branch. This guarantees | |
| # the AI changes actually reach the PR (the action's automation mode does not push), and it | |
| # is the only step with push capability. | |
| - name: Commit + push AI wiring | |
| id: ai_push | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| run: | | |
| git add -A | |
| # Trusted list of what the AI pass changed (the tool-call summary above logs no values). | |
| git diff --cached --stat | |
| if git diff --cached --quiet; then | |
| echo "AI step produced no changes." | |
| echo "summary=Mechanical-only PR (AI wiring added no changes). Human review required before merge." >> "$GITHUB_OUTPUT" | |
| else | |
| git commit -m "feat(spec-sync): wire SDK to spec diff (AI)" | |
| git push origin HEAD:"$SYNC_BRANCH" | |
| echo "summary=Mechanical snapshot + AI wiring committed. Gates must pass; human review required before merge." >> "$GITHUB_OUTPUT" | |
| fi | |
| # Rewrite the PR description: keep the static process/safety preamble from "Open sync PR" and | |
| # APPEND an LLM "## What changed" summary of the actual PR diff. Logic is shared by both jobs | |
| # in scripts/spec-sync/summarize-pr.sh. Best-effort — the static body stands if it degrades; | |
| # the script only edits the PR body via SPEC_SYNC_TOKEN, so an injected spec description in | |
| # the diff can't escalate past text. | |
| - name: Summarize the PR diff into the description | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| continue-on-error: true # a summary is nice-to-have; never red the run over it | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} | |
| SUMMARY_TITLE_PREFIX: 'spec-sync(v1)' # LLM generates only the title summary after this prefix | |
| run: ./scripts/spec-sync/summarize-pr.sh "${{ steps.open_pr.outputs.url }}" | |
| # The AI-wiring outcome, as a reply under the PR-opened thread root (steps.root). | |
| - name: Slack — AI wiring result | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != '' | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: info | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: AI wiring' | |
| text: ${{ steps.ai_push.outputs.summary }} | |
| thread_ts: ${{ steps.root.outputs.ts }} | |
| # Catch-all failure alert for this job. An unavailable spec source (exit 20) is skipped earlier | |
| # and never reaches here; what this calls out is a reachable source that errored — an HTTP error | |
| # status / empty / invalid spec, or a script error (exit not 0/10/20) — and an AI-wiring crash | |
| # (PR left mechanical-only), with a generic fallback. | |
| - name: Slack — spec-sync failed | |
| if: failure() | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: failure | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: run failed' | |
| text: >- | |
| ${{ (steps.drift.outputs.code != '' && steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20') | |
| && format('Spec fetch/normalize failed (exit {0}) — staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored (an unbooked cluster 404s → exit 20, skipped, not alerted). Investigate.', steps.drift.outputs.code) | |
| || (steps.ai_wiring.outcome == 'failure' | |
| && 'AI wiring step failed. Any open sync PR has only the mechanical commit and needs manual wiring.' | |
| || 'spec-sync run failed. See the workflow run for details.') }} | |
| · <${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}|View run> | |
| thread_ts: ${{ steps.root.outputs.ts }} | |
| # V2 loop. Same machinery as the V1 job above, pointed at the V2 spec on the AIDE gateway | |
| # and a separate sync branch, with a V2-tailored wiring prompt. Kept as a distinct job (not a | |
| # matrix) so the proven V1 loop is untouched and the two prompts can diverge freely. | |
| # | |
| # HOSTS (do not conflate): the V2 *spec* is published at aide.[env]/openapi.json; the V2 *API* | |
| # (what the SDK and contract tests call) is api.ade.[env]. We fetch drift from aide; the SDK | |
| # never talks to aide (its paths are staff-SSO-gated). | |
| spec-sync-v2: | |
| if: github.repository == 'landing-ai/ade-python' | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 60 # headroom for the AI wiring step (up to --max-turns 250) plus rye sync | |
| env: | |
| V2_SPEC_URL: https://aide.staging.landing.ai/openapi.json # staging drives the loop | |
| SYNC_BRANCH: spec-sync/v2 # separate branch so V1 and V2 sync PRs never collide | |
| SPEC_LABEL: V2 # used only in Slack/PR-comment copy to disambiguate the two jobs | |
| steps: | |
| - uses: actions/checkout@v6 | |
| with: | |
| # SPEC_SYNC_TOKEN (fine-grained PAT), NOT GITHUB_TOKEN: pushes/PRs authored by | |
| # GITHUB_TOKEN do not trigger the gate workflows (anti-recursion). | |
| token: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| fetch-depth: 0 # tags needed for surface-lock baseline | |
| - name: Install Rye | |
| run: | | |
| curl -sSf https://rye.astral.sh/get | bash | |
| echo "$HOME/.rye/shims" >> "$GITHUB_PATH" | |
| env: | |
| RYE_VERSION: '0.44.0' | |
| RYE_INSTALL_OPTION: '--yes' | |
| - name: Install dependencies | |
| run: rye sync --all-features | |
| - name: Detect drift | |
| id: drift | |
| run: | | |
| set +e | |
| ./scripts/spec-sync/check-drift.sh "$V2_SPEC_URL" specs/v2-aide.json | |
| echo "code=$?" >> "$GITHUB_OUTPUT" | |
| - name: No drift | |
| if: steps.drift.outputs.code == '0' | |
| run: echo "specs in sync; nothing to do." | |
| # An unavailable spec source (check-drift exit 20) is the EXPECTED case on staging, not an | |
| # incident: staging auto-reclaims and must be booked to come back — an unbooked cluster serves a | |
| # 404 (or the host stops answering entirely). Treat ONLY this as a no-op — log and end the run | |
| # cleanly (no drift, no PR, no Slack); the next hourly run picks up any drift once staging is | |
| # booked. Every other failure (a reachable source returning 401/403/5xx, an empty/invalid spec, | |
| # or a script error) falls through to "Fail on spec error" below and alerts. | |
| - name: Spec source unavailable — skip | |
| if: steps.drift.outputs.code == '20' | |
| run: echo "spec source unavailable (exit 20); staging is likely unbooked — skipping this run." | |
| # A reachable-but-invalid spec (empty/whitespace body, malformed JSON) or a script error is a | |
| # real problem — fail loudly so the catch-all alert fires. Distinct from the exit-20 skip above. | |
| - name: Fail on spec error | |
| if: steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20' | |
| run: | | |
| echo "spec fetch/normalize failed (exit ${{ steps.drift.outputs.code }}); staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored." | |
| exit 1 | |
| - name: Check for an open sync PR | |
| id: existing | |
| if: steps.drift.outputs.code == '10' | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: | | |
| # Fail CLOSED: capture the query result separately so a transient gh/API failure aborts | |
| # the job (default bash is `set -eo pipefail`, but a failure inside an `if` condition is | |
| # exempt from set -e — which would silently become open=false and force-push over an | |
| # existing human-reviewed PR's branch). | |
| count="$(gh pr list --state open --head "$SYNC_BRANCH" --json number --jq 'length')" | |
| if [ "$count" -gt 0 ]; then | |
| echo "open=true" >> "$GITHUB_OUTPUT" | |
| else | |
| echo "open=false" >> "$GITHUB_OUTPUT" | |
| fi | |
| # A sync PR is already open. Every hourly tick still reports drift (check-drift compares against | |
| # main, not the PR branch), so instead of a bare skip: detect drift that appeared BEYOND the | |
| # open PR and, if any, annotate the PR + emit one Slack ping per distinct new spec (deduped). | |
| - name: New drift behind the open PR | |
| id: behind | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true' | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: ./scripts/spec-sync/notify-new-drift.sh specs/v2-aide.json "$SYNC_BRANCH" "$SPEC_LABEL" | |
| - name: Slack — new drift behind the open PR | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true' && steps.behind.outputs.notify == 'true' | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: warning | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: new drift behind the open PR' | |
| text: 'New ${{ env.SPEC_LABEL }} spec drift appeared beyond the <${{ steps.behind.outputs.pr_url }}|open sync PR> (live-spec `${{ steps.behind.outputs.hash }}`). Merge or close it and the next run opens a fresh one covering the rest.' | |
| thread_ts: ${{ steps.behind.outputs.thread_ts }} | |
| # ---- Phase 1: mechanical (deterministic) ---- | |
| - name: Mechanical commit + push | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| run: | | |
| git config user.name "spec-sync[bot]" | |
| git config user.email "spec-sync@users.noreply.github.com" | |
| git checkout -B "$SYNC_BRANCH" | |
| ./scripts/spec-sync/gen-models.sh specs/v2-aide.json specs/_generated/v2_models.py | |
| git add specs/v2-aide.json specs/_generated/v2_models.py | |
| git commit -m "chore(spec-sync): update V2 spec snapshot + regenerated reference models" | |
| git push --force origin "$SYNC_BRANCH" | |
| - name: Open sync PR | |
| id: open_pr | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: | | |
| url="$(gh pr create --base main --head "$SYNC_BRANCH" \ | |
| --title "spec-sync: track V2 spec drift" \ | |
| --body $'Automated V2 spec-sync PR (`client.v2`).\n\n- **Commit 1 (mechanical):** normalized V2 spec snapshot + regenerated reference models.\n- **Commit 2 (AI, only if the spec diff needs SDK changes):** client.v2 resources/methods/tests/docs wired from the diff, added after this PR opened. Workflow-only drift is excluded and an AI no-op is skipped, so some drifts produce a **mechanical-only PR with no second commit**.\n\nGates (surface-lock, V2 contract tests, lint/test/typecheck) must pass. When present, the AI commit is a **draft a human finishes** (the V2 ergonomic layer — unified Job, dual-host, schema coercion — is not in the spec). **Human review required before merge.**')" | |
| echo "url=$url" >> "$GITHUB_OUTPUT" | |
| # Post the Slack thread ROOT now — right after the PR exists, before AI wiring — so every later | |
| # event (AI outcome, new drift, gates, merge) replies under it. Persist its ts on the PR body | |
| # so other runs/workflows can find the thread. (Editing the body doesn't re-trigger pr-gates, | |
| # which only fires on opened/synchronize/reopened.) | |
| - name: Slack — PR opened (thread root) | |
| id: root | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: info | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: new spec drift → PR opened' | |
| text: '<${{ steps.open_pr.outputs.url }}|Open the PR> — mechanical spec snapshot committed; wiring the SDK next. Human review required before merge.' | |
| - name: Save thread root on the PR | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != '' | |
| continue-on-error: true # bookkeeping only; thread-ts.sh fails closed, and a hiccup must not red the run | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| run: ./scripts/spec-sync/thread-ts.sh set "${{ steps.open_pr.outputs.url }}" "${{ steps.root.outputs.ts }}" | |
| # ---- Everything the AI pass may see or reach is decided HERE, in a trusted step -------- | |
| # | |
| # 1. Precompute the mechanical diff. The agent needs it, but must not run `git diff` to get | |
| # it: git honours `diff.external` / textconv drivers from .git/config, so an agent that | |
| # can write .git/config turns a "read-only" git diff into arbitrary execution INSIDE its | |
| # own step — before any allowlist or hardening runs. Handing it a plain file keeps the | |
| # input and removes the primitive. (Untracked and not in the `git clean` exclusions, so | |
| # the allowlist step deletes it before anything is committed.) | |
| # | |
| # 2. Take the push credential away BEFORE the agent runs, not after. actions/checkout | |
| # persists it in .git/config; nothing between here and the explicit push needs it, so | |
| # even a successful in-step escape has no token to reach. Snapshot the now-clean config | |
| # so each post-AI restore can put it back wholesale, rather than unsetting known-bad keys | |
| # one at a time — an agent can add `filter.*.clean`, `diff.external`, aliases, and any | |
| # enumeration of those will be incomplete. | |
| # | |
| # 3. Drop .git/hooks and .git/info. Neither is a worktree path, so neither `git checkout` | |
| # nor `git clean` restores them: a planted pre-commit hook runs on our own `git commit`, | |
| # and one line in .git/info/exclude makes `git clean -fd` skip the agent's own file — | |
| # walking it straight past the product-code allowlist. | |
| - name: Precompute the spec diff and harden the checkout | |
| id: harden | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| run: | | |
| git diff HEAD~1..HEAD > .spec-sync-diff.txt | |
| git config --local --unset-all 'http.https://github.com/.extraheader' || true | |
| git config --local --unset-all core.hooksPath || true | |
| rm -rf .git/hooks .git/info | |
| # The snapshot is a step OUTPUT, not a file. Every path on disk — $RUNNER_TEMP included — | |
| # is writable by AI-authored code: nothing here executes src/ any more, but ruff/pyright | |
| # run as binaries out of the gitignored .venv, which the allowlist cannot clean, so a | |
| # replaced binary could overwrite a file snapshot between it being taken and restored and | |
| # thereby reinstall an attacker-controlled core.hooksPath before the credentialed push. | |
| # Step outputs are captured by the runner when this step ends and injected as expressions | |
| # afterwards, so nothing running later in the job can alter this value. | |
| echo "gitconfig=$(base64 -w0 < .git/config)" >> "$GITHUB_OUTPUT" | |
| # ---- Phase 2: AI wiring — edits ONLY (no git/push; a later step commits). ---- | |
| - name: AI wiring (edits only) | |
| id: ai_wiring | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| uses: anthropics/claude-code-action@v1 | |
| with: | |
| anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} | |
| # NOT SPEC_SYNC_TOKEN. claude-code-action exports the token it is given into the agent's | |
| # environment (and configures `origin` with it), so handing it the write PAT would put a | |
| # push-capable credential inside the step that processes the untrusted spec — where the | |
| # agent's own Read tool could recover it and Write it into an allowlisted file that the | |
| # trusted push then publishes. `github.token` is read-only here (see `permissions:` at | |
| # the top of this file); SPEC_SYNC_TOKEN is reserved for the trusted push step. | |
| github_token: ${{ github.token }} | |
| prompt: | | |
| The previous commit updated specs/v2-aide.json and regenerated reference models in | |
| specs/_generated/v2_models.py. Wire the client.v2 SDK to match the spec diff. This is | |
| the V2 surface (client.v2.*), NOT V1 — do not touch V1 resources. | |
| SCOPE — exactly which routes back client.v2: the `/v2/*` paths in this spec, and | |
| nothing else. This spec is the AIDE gateway's and ALSO carries `/v1/*` routes (for | |
| example /v1/ade/parse, /v1/classify, /v1/split): that is the V1-compatibility surface, | |
| tracked by the separate V1 spec-sync job against its own spec, and it is NOT part of | |
| client.v2. Do not wire a `/v1/*` route into client.v2, and never "translate" one into a | |
| `/v2/...` URL. Every path you write into a resource must appear verbatim under `paths` | |
| in specs/v2-aide.json — a trusted step (scripts/spec-sync/check-v2-paths.sh) verifies | |
| this after you finish and rejects any path the spec does not have. A diff whose only new | |
| ROUTES are `/v1/*` is not "nothing to do": still wire every in-scope change listed | |
| below (a new request field on /v2/parse, say) and leave the `/v1/*` routes alone. | |
| SCOPE, one level down — exactly which FIELDS back a /v2 request: the TOP-LEVEL | |
| properties of that operation's `requestBody` schema in this spec, and nothing else. A | |
| field the spec declares inside a nested object is not a top-level field — the | |
| encrypted-PDF password is declared at `options.password` on /v2/parse and | |
| /v2/parse/jobs and NOWHERE else, so it is sent there and only there. Two traps, both | |
| of which have bitten this repo: (a) the SAME spec declares a top-level `password` on | |
| /v1/ade/parse and /v1/ade/parse/jobs — same name, different level, different route | |
| family — and lifting a `/v1/*` top-level field onto a /v2 operation is the field-level | |
| version of translating a /v1 route into a /v2 URL; (b) inventing a top-level | |
| convenience parameter that folds one wire field into another. Exactly three exist | |
| (`password` on parse, `strict` and `grounding` on extract); they are cross-SDK decisions | |
| shared with ade-typescript and written down in CONTRIBUTING.md, and changing one is a | |
| maintainer's call — do NOT add, remove or re-target an alias. Wire the spec's own | |
| shape, and if a spec change looks like it needs a new one, leave that field unwired | |
| and say so in your summary. | |
| OUT OF SCOPE — do NOT implement: /v2/workflow, /v2/workflow/jobs, and | |
| /v2/workflow/jobs/{job_id}. That surface is intentionally deferred; skip it entirely | |
| even though it appears in the spec and reference models. If the spec diff touches only | |
| workflow and/or `/v1/*` routes, make no changes. | |
| First read the ENTIRE mechanical commit diff, which a trusted step has already written | |
| to `.spec-sync-diff.txt` in the repo root (you have no shell — use Read, do not try to | |
| run git; the file runs to thousands of lines, so page through it with offset/limit until | |
| you have seen the final hunk) — the operation/path hunks AND changed entries under | |
| `components.schemas` — to find every change backing client.v2. A response-field | |
| addition often changes ONLY a component schema (e.g. /v2/parse references | |
| #/components/schemas/ParseResponse) and its generated model, not the operation hunk, so | |
| for every changed schema trace its `$ref` consumers to the non-workflow `/v2/*` | |
| operations that use it. Cover: (a) newly added `/v2/*` routes; (b) request-field, | |
| response-field or schema changes on existing `/v2/*` operations, INCLUDING | |
| component-only changes. A new optional request field on /v2/parse is exactly case (b) | |
| and must be wired even when the same diff is dominated by unrelated `/v1/*` additions. | |
| Distinguish REQUEST fields — wire as a new optional keyword parameter on the existing | |
| method (see the additive-changes rule below) — from RESPONSE-model fields — add the | |
| field to the corresponding response model under src/landingai_ade/types/v2/. For a | |
| genuinely new route, read src/landingai_ade/resources/v2/extract.py and | |
| src/landingai_ade/resources/v2/parse.py and mirror their structure: | |
| - Add a resource module under src/landingai_ade/resources/v2/ with a sync run() | |
| (multipart or JSON per the spec; route a 504 through | |
| lib/v2_errors.raise_if_sync_timeout, passing THIS endpoint's own async route as | |
| `jobs_resource` — e.g. `build_schema_jobs` — so the 504 remediation names the right | |
| resource; never let it fall back to a parse/extract-specific message, and if the | |
| helper doesn't yet take a per-endpoint `jobs_resource`, extend it and update every | |
| existing call site and helper test to pass its own matching resource rather than | |
| reusing another endpoint's wording) and, for an async route, a *JobsResource with | |
| create / get / wait / list. Decide multipart-vs-JSON from the SPEC, not from a | |
| generated class name: an operation accepts file uploads iff its | |
| `requestBody.content` has a `multipart/form-data` variant with a property (or array | |
| item) of `format: binary`. For such a field, accept file inputs (FileTypes: a str for | |
| inline content, or a Path/bytes/file object for uploads). Pick the content type from | |
| what the operation DECLARES, not a fixed JSON default: send `multipart/form-data` when | |
| any value is a file OR the operation declares only multipart (no `application/json` | |
| variant — e.g. `/v2/parse`, where even an all-string `document_url` call must be | |
| multipart); send JSON only when the operation declares an `application/json` variant | |
| AND every value is a string. Identify the file-capable | |
| request model by that `bytes`/binary field type — the generated `*PostRequest` / | |
| `*PostRequest1` variants are ordered by content-type and the suffix can flip if the | |
| spec reorders them, so NEVER key off the numeric suffix. | |
| - Register it on the V2Resource/AsyncV2Resource container in | |
| src/landingai_ade/resources/v2/v2.py exactly as parse/extract are wired (lazy | |
| cached_property + an explicit-signature top-level delegator). Do NOT re-export the | |
| sub-resource from src/landingai_ade/resources/v2/__init__.py — that file exports | |
| only V2Resource/AsyncV2Resource, matching the reference files. | |
| - Normalize any async job into the unified Job (src/landingai_ade/types/v2/job.py) by | |
| adding a normalize_* function in src/landingai_ade/resources/v2/_normalize.py, | |
| mirroring normalize_parse_job / normalize_extract_job (tolerant status via | |
| _status(); map the envelope's result / error / timestamps). | |
| - Put response models under src/landingai_ade/types/v2/. If a run accepts a JSON | |
| Schema, accept a pydantic model / dict / JSON string and coerce via | |
| lib/schema_utils.coerce_schema_to_dict. | |
| - Build request URLs with V2ResourceMixin._v2_url — dual-host routing is automatic; | |
| never hardcode a host. | |
| Optionality comes from the schema, never from the prose: a response field is Optional | |
| iff it is absent from that schema's `required` array. A field can be described as | |
| "always present for model X" and still be optional — read `required`, not the | |
| `description`. Model an optional field as `Optional[T] = None` and treat "the key is | |
| missing" and "the key is null" as the same state (`if x is not None`), because the | |
| gateway omits optional fields rather than sending an explicit null. | |
| Add a live check to tests/contract/test_v2_smoke.py (marked `contract`) and unit tests | |
| under tests/api_resources/. The contract file hits LIVE staging, so it must assert only | |
| what staging is guaranteed to return: NEVER pin `model=` there (a model family the spec | |
| documents may not be servable in whatever way the staging cluster happens to be | |
| provisioned — the word-granularity family, currently `dpt-3-verity` and formerly | |
| `dpt-3-fast`, needs a GPU-backed booking, and against one without it the | |
| request hangs until the gate times out; that is an environment property, not an SDK | |
| contract, so no live test should depend on it), and for an optional field assert | |
| absent-or-valid ("if present, it is in range"), never that it is populated. Assertions | |
| on a specific value belong in the mocked unit tests under tests/api_resources/, where | |
| you control the response body — put them there instead, and reuse the module's existing | |
| `staging_client` fixture rather than constructing your own client. | |
| Update api.md, and create or update docs/v2-testing.md (the V2 QA guide) with the new | |
| surface. Write clean, import-sorted code matching the style of the files you mirror — a | |
| later fixed step runs `./scripts/format` and then `./scripts/lint`, and you get one | |
| chance to fix whatever lint reports before the PR opens. | |
| You may only edit product code: src/, tests/, docs/, and api.md. Do NOT edit scripts/, | |
| .github/, specs/, or any dotfile. | |
| The repo type-checks with pyright in STRICT mode, which you cannot run yourself. Two | |
| rules that keep it green: annotate every parameter and local with its concrete type | |
| (importing the real model from landingai_ade.types.v2) — never `object` or `Any` when | |
| you then access attributes on the value, which yields `reportUnknown*` errors; and do | |
| not add `# type: ignore` to silence an annotation you could have written correctly. | |
| Before you finish, self-check coverage: list every operation hunk and every changed | |
| component schema in `.spec-sync-diff.txt`, mark each IN SCOPE (`/v2/*`, non-workflow) or | |
| OUT OF SCOPE (`/v1/*`, `/v2/workflow*`), confirm every in-scope item maps to an edit you | |
| made, and confirm every URL path you wrote exists under `paths` in specs/v2-aide.json. | |
| Close any gap before you stop. | |
| Additive-changes rule: BACKWARD-COMPATIBLE ADDITIONS ONLY. You MAY add a new optional | |
| keyword parameter (with a default) to an existing parse/extract method to wire an | |
| additive request field — surface-lock permits that. You must NOT remove or rename a | |
| public symbol or parameter, change a type, or make an existing parameter required; | |
| surface-lock will fail the PR on any of those. You have NO shell and no git access at | |
| all; the diff you need is already in `.spec-sync-diff.txt`. A later workflow step | |
| commits and pushes your edits. | |
| # Pin the model: the action otherwise inherits Claude Code's default, which floats | |
| # across CLI releases and makes spec-sync output non-reproducible. `[1m]` = 1M context — | |
| # the mechanical diff alone can run to ~100K tokens. claude-opus-5 is the pin in BOTH SDK | |
| # repos; keep them equal. (On the drift behind #153 the then-pinned claude-opus-4-8 | |
| # rewrote new /v1 routes as non-existent /v2 ones and skipped the in-scope field, while | |
| # ade-typescript's unpinned run — opus-5 by default — wired exactly the right thing, twice.) | |
| # | |
| # NO Bash tool. `Bash(git diff:*)` was allowed here so the agent could inspect the | |
| # mechanical diff, but `git diff` is NOT inert: an agent that writes `.git/config` can | |
| # point `diff.external` (or a textconv driver) at a script it also wrote, and the | |
| # permitted `git diff` then EXECUTES it — inside the AI step, before any allowlist or | |
| # hardening applies. A trusted step precomputes the diff into .spec-sync-diff.txt | |
| # instead, so the agent keeps the input and loses the execution primitive. | |
| claude_args: | | |
| --model "claude-opus-5[1m]" | |
| --max-turns 250 | |
| --allowedTools "Edit,Write,Read,Glob,Grep" | |
| # Print the agent's tool calls — tool name and argument NAMES only, never values — to the step | |
| # log, so the shape of what the agent did can be audited later (the commit step lists what it | |
| # changed, from the trusted checkout): | |
| # the action logs only the init and result summaries, and #153 had to be root-caused without | |
| # any trace. Deliberately NO argument values, assistant text, tool results or commands: the spec this | |
| # agent reads is untrusted input, and a prompt-injected agent that has Read its own environment | |
| # could carry a transformed or chunked credential past GitHub's literal secret masking in | |
| # free-form narration. Runs BEFORE anything AI-authored can execute, straight off the action's | |
| # own output file, with the jq program held in this workflow's top-level env (not in a repo | |
| # script the agent could have edited). Best-effort. | |
| - name: AI wiring tool-call summary | |
| if: always() && steps.ai_wiring.outputs.execution_file != '' | |
| continue-on-error: true | |
| env: | |
| TRANSCRIPT: ${{ steps.ai_wiring.outputs.execution_file }} | |
| run: | | |
| echo "::group::AI wiring tool-call summary" | |
| set -o pipefail | |
| # Prefix every line: nothing the agent could influence may start at column 0, where a leading | |
| # `::` would be parsed as a workflow command (e.g. ::stop-commands::) even inside a group. | |
| jq -r "$TOOL_CALL_SUMMARY_JQ" "$TRANSCRIPT" | tr -d '\r' | sed 's/^/ /' || echo " tool-call summary unavailable (unexpected file shape)" | |
| echo "::endgroup::" | |
| # ---- Harden before running ANY trusted command over the AI's edits -------------------- | |
| # | |
| # The AI step has no shell, so it cannot execute anything ITSELF. But the allowlist revert | |
| # below is not a complete boundary, because `git checkout` only restores TRACKED files and | |
| # `git clean -fd` (no -x) only deletes untracked, non-ignored ones. Two repo-wide Write | |
| # targets survive both — and the trusted steps that follow EXECUTE both: | |
| # * .venv/** — gitignored, so never cleaned; `rye run check:ruff` / `typecheck` / `format` | |
| # all run binaries out of it. | |
| # * .git/** — not a worktree path at all; `.git/hooks/pre-commit` would run on our own | |
| # `git commit`, and `core.hooksPath` can redirect hooks anywhere. | |
| # Either one turns "the AI wrote a file" into "code runs in a trusted step". | |
| # | |
| # So do two things before any of those steps: drop the push credential actions/checkout | |
| # persisted into .git/config (nothing executed downstream can reach a token that is not | |
| # there — the AI-wiring push below re-supplies it explicitly for that one command), and | |
| # neutralize the hook paths so our own git commands cannot run anything. | |
| # | |
| # This narrows the blast radius; it does not make executing AI-authored code safe. Full | |
| # isolation — run the AI pass in a credential-free checkout and copy only the validated | |
| # allowlist diff back — is the real fix and is deliberately NOT attempted here. | |
| - name: Restore trusted git metadata after the AI pass | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| env: | |
| TRUSTED_GITCONFIG: ${{ steps.harden.outputs.gitconfig }} | |
| run: | | |
| printf '%s' "$TRUSTED_GITCONFIG" | base64 -d > .git/config | |
| rm -rf .git/hooks .git/info | |
| # Its Write tool is repo-wide, so before running the trusted | |
| # formatter — and before the `git add -A` below — we enforce the product-code ALLOWLIST: | |
| # revert every tracked edit and delete every untracked file OUTSIDE src/ tests/ docs/ api.md. | |
| # This stops an injected edit to e.g. pyproject.toml (which defines what `rye run format` | |
| # runs) from executing here, and stops any out-of-scope edit from reaching the commit. | |
| # (`git clean` has no -x, so gitignored paths like .venv are untouched.) | |
| - name: Restrict AI edits to product code, then format | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| run: | | |
| git checkout -- . ':(exclude)src' ':(exclude)tests' ':(exclude)docs' ':(exclude)api.md' | |
| git clean -fd -e /src -e /tests -e /docs | |
| # A .gitattributes INSIDE the allowlist survives the two lines above and can bind content | |
| # filters / diff drivers to paths. The config half of that attack is already gone (the | |
| # restore step above replaces .git/config wholesale), but remove the file half too. This | |
| # repo tracks no .gitattributes anywhere, so deleting any that appeared is always safe. | |
| find src tests docs -name .gitattributes -delete 2>/dev/null || true | |
| ./scripts/format | |
| # --------------------------------------------------------------------------------------- | |
| # Verify the AI wiring, then give it ONE chance to fix itself. | |
| # | |
| # The wiring step has no shell, so nothing has type-checked its output: the step above only | |
| # *formats*. Historically that meant every AI mistake pyright can see (annotating a param | |
| # `object` and reaching for attributes on it, `# type: ignore` over a fixable annotation) | |
| # landed as a red `lint` job on a PR that otherwise looked review-ready. So run the real lint | |
| # HERE, in a trusted step, and hand its output back to a second shell-less Claude pass. | |
| # | |
| # This closes the loop WITHOUT granting the AI a shell: nothing here EXECUTES the code it just | |
| # wrote, and its edits are still funnelled through the product-code allowlist below before they | |
| # can reach a commit. The lint log it reads is ruff/pyright/mypy output quoting the AI's own | |
| # just-written code — no new trust boundary is crossed by feeding it back. | |
| # | |
| # That "nothing executes it" property is why this runs a STATIC SUBSET and not `./scripts/lint`: | |
| # that script — and the `rye run lint` chain it calls — ends in `check:importable`, i.e. | |
| # `python -c 'import landingai_ade'`, which would import the AI's freshly written `src/` inside | |
| # a trusted step that still holds the checkout's SPEC_SYNC_TOKEN. A top-level statement in any | |
| # imported module would then run with push credentials, defeating the whole shell-less design. | |
| # ruff, pyright, and mypy only ever READ the source. The import check is not lost — it still | |
| # runs on the PR in CI's own `lint` job, which has no push credentials. | |
| # | |
| # `specs/` is excluded from pyright (see [tool.pyright] in pyproject.toml), so anything lint | |
| # reports here lives in the AI's own edits under src/ tests/ docs/ — i.e. within its reach. | |
| - name: Lint the AI wiring | |
| id: ai_lint | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| # A lint failure here is the normal case this loop exists to handle, not a run failure. | |
| continue-on-error: true | |
| run: | | |
| set -o pipefail | |
| # Run all three even if one fails, so the repair pass sees every error at once. | |
| status=0 | |
| rye run check:ruff 2>&1 | tee .spec-sync-lint.log || status=1 | |
| rye run typecheck 2>&1 | tee -a .spec-sync-lint.log || status=1 | |
| # Both directions: a wired path the spec lacks (#153 invented /v2/classify and /v2/split) | |
| # and a /v2 spec route nothing sends. grep + jq over source only — as safe here as ruff. | |
| ./scripts/spec-sync/check-v2-paths.sh specs/v2-aide.json src/landingai_ade/resources/v2 2>&1 | tee -a .spec-sync-lint.log || status=1 | |
| exit $status | |
| - name: AI lint repair (edits only) | |
| id: ai_lint_fix | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure' | |
| uses: anthropics/claude-code-action@v1 | |
| with: | |
| anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} | |
| # NOT SPEC_SYNC_TOKEN. claude-code-action exports the token it is given into the agent's | |
| # environment (and configures `origin` with it), so handing it the write PAT would put a | |
| # push-capable credential inside the step that processes the untrusted spec — where the | |
| # agent's own Read tool could recover it and Write it into an allowlisted file that the | |
| # trusted push then publishes. `github.token` is read-only here (see `permissions:` at | |
| # the top of this file); SPEC_SYNC_TOKEN is reserved for the trusted push step. | |
| github_token: ${{ github.token }} | |
| prompt: | | |
| The trusted lint step failed on the client.v2 wiring you just wrote. Read | |
| .spec-sync-lint.log in the repo root — it is the captured output of that run (ruff | |
| check, then pyright in STRICT mode and mypy, then scripts/spec-sync/check-v2-paths.sh, | |
| which cross-checks the URL paths the V2 resources send against specs/v2-aide.json) — | |
| and fix every error it reports. | |
| Fix the CAUSE, not the symptom. Specifically, do NOT silence anything: no | |
| `# type: ignore`, no `cast(Any, ...)`, no widening an annotation to `object`/`Any`, no | |
| deleting the assertion or test that tripped the error. A `reportUnknown*` error means a | |
| value has no concrete type — annotate it with the real model imported from | |
| landingai_ade.types.v2 (a recursive helper over parse elements takes | |
| `List[V2ParseElement]`, not `object`). Keep the behavior the wiring was written to have. | |
| A `check-v2-paths` error means one of two things. (a) A resource sends a path that is | |
| not under `paths` in specs/v2-aide.json: remove that method/resource together with its | |
| registration in src/landingai_ade/resources/v2/v2.py, its types, tests and docs — never | |
| invent or "translate" a path (a /v1/... route is NOT a /v2/... route; only `/v2/*` | |
| backs client.v2). (b) A `/v2/*` route in the spec has no resource sending it: wire it, | |
| mirroring src/landingai_ade/resources/v2/parse.py and extract.py. Paths under | |
| /v2/workflow* are out of scope for every AI pass, this one included: never add, | |
| remove or edit a workflow resource, and if a `check-v2-paths` error concerns a | |
| /v2/workflow* path, leave it for a maintainer. | |
| You may only edit src/, tests/, docs/, and api.md — the same allowlist as before. Do | |
| NOT edit scripts/, .github/, specs/, pyproject.toml, or any dotfile (including | |
| .spec-sync-lint.log itself: it is an input, and it is deleted before anything is | |
| committed). You have NO shell and no git access at all. | |
| # NO Bash tool — see the wiring step above for why `Bash(git diff:*)` was removed. | |
| # Same model pin as the wiring step, for the same reasons. | |
| claude_args: | | |
| --model "claude-opus-5[1m]" | |
| --max-turns 60 | |
| --allowedTools "Edit,Write,Read,Glob,Grep" | |
| # Same tool-call summary as after the wiring step — see there for why it is the log, and why | |
| # it runs before anything AI-authored executes. | |
| - name: AI lint repair tool-call summary | |
| if: always() && steps.ai_lint_fix.outputs.execution_file != '' | |
| continue-on-error: true | |
| env: | |
| TRANSCRIPT: ${{ steps.ai_lint_fix.outputs.execution_file }} | |
| run: | | |
| echo "::group::AI lint repair tool-call summary" | |
| set -o pipefail | |
| # Prefix every line: nothing the agent could influence may start at column 0, where a leading | |
| # `::` would be parsed as a workflow command (e.g. ::stop-commands::) even inside a group. | |
| jq -r "$TOOL_CALL_SUMMARY_JQ" "$TRANSCRIPT" | tr -d '\r' | sed 's/^/ /' || echo " tool-call summary unavailable (unexpected file shape)" | |
| echo "::endgroup::" | |
| # The repair pass is exactly as untrusted as the wiring pass, so repeat the hardening too: | |
| # it could have written .venv/** or .git/hooks/** just as the wiring pass could, and the | |
| # format/re-lint/commit steps below execute both. (The credential unset is already done and | |
| # idempotent; the hooks removal matters because a repair pass could recreate them.) | |
| - name: Restore trusted git metadata after the repair pass | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure' | |
| env: | |
| TRUSTED_GITCONFIG: ${{ steps.harden.outputs.gitconfig }} | |
| run: | | |
| printf '%s' "$TRUSTED_GITCONFIG" | base64 -d > .git/config | |
| rm -rf .git/hooks .git/info | |
| # Re-apply the allowlist before formatting or committing. This also deletes | |
| # .spec-sync-lint.log (untracked, not in the `git clean` exclusions), so the log never | |
| # reaches the commit. | |
| - name: Re-restrict AI edits to product code, then format | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure' | |
| run: | | |
| git checkout -- . ':(exclude)src' ':(exclude)tests' ':(exclude)docs' ':(exclude)api.md' | |
| git clean -fd -e /src -e /tests -e /docs | |
| # A .gitattributes INSIDE the allowlist survives the two lines above and can bind content | |
| # filters / diff drivers to paths. The config half of that attack is already gone (the | |
| # restore step above replaces .git/config wholesale), but remove the file half too. This | |
| # repo tracks no .gitattributes anywhere, so deleting any that appeared is always safe. | |
| find src tests docs -name .gitattributes -delete 2>/dev/null || true | |
| ./scripts/format | |
| # Record whether the repair actually worked, so the Slack reply and the PR say so honestly | |
| # instead of implying the wiring is clean when CI is about to go red. | |
| - name: Re-lint after repair | |
| id: ai_lint_final | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure' | |
| continue-on-error: true | |
| # Same static subset as the first pass — see there for why `check:importable` is excluded. | |
| run: | | |
| status=0 | |
| rye run check:ruff || status=1 | |
| rye run typecheck || status=1 | |
| ./scripts/spec-sync/check-v2-paths.sh specs/v2-aide.json src/landingai_ade/resources/v2 || status=1 | |
| exit $status | |
| # LAST line of defence, and it must come after the last step that executes anything. | |
| # `./scripts/format` and the linters all run binaries out of the gitignored .venv, which the | |
| # allowlist cannot clean — so any of them could have written outside the allowlist, or | |
| # recreated git config/hooks, AFTER the earlier restore ran. Without this, `git add -A` below | |
| # would stage those post-allowlist edits and the trusted push would publish them. | |
| # | |
| # Runs unconditionally (not just on the repair path) because the plain path executes the | |
| # formatter and linters too. Nothing after this point executes AI-authored content: the only | |
| # remaining steps are `git add`/`git commit` and the push itself. Re-running the allowlist is | |
| # safe and idempotent — src/ tests/ docs/ api.md are excluded, so every legitimate AI edit and | |
| # every formatter result is preserved. | |
| - name: Final restore + allowlist before staging | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| env: | |
| TRUSTED_GITCONFIG: ${{ steps.harden.outputs.gitconfig }} | |
| run: | | |
| printf '%s' "$TRUSTED_GITCONFIG" | base64 -d > .git/config | |
| rm -rf .git/hooks .git/info | |
| git reset -q | |
| git checkout -- . ':(exclude)src' ':(exclude)tests' ':(exclude)docs' ':(exclude)api.md' | |
| git clean -fd -e /src -e /tests -e /docs | |
| find src tests docs -name .gitattributes -delete 2>/dev/null || true | |
| # Commit and push are SEPARATE steps, and the token appears only in the push step's env. | |
| # `git add` runs content filters (`filter.<driver>.clean`), which an AI pass can install via | |
| # .git/config plus a .gitattributes inside the allowlist — the restore step above removes the | |
| # config half, and keeping the token out of this step's environment means even a filter that | |
| # somehow survived has nothing to read. By the time the push step runs, staging and committing | |
| # are already done and no AI-authored content is executed again. | |
| - name: Commit AI wiring | |
| id: ai_commit | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| env: | |
| LINT_FIRST: ${{ steps.ai_lint.outcome }} | |
| LINT_FINAL: ${{ steps.ai_lint_final.outcome }} # 'skipped' when the first lint passed | |
| run: | | |
| if [ "$LINT_FIRST" = 'success' ]; then | |
| lint_note='Lint clean.' | |
| elif [ "$LINT_FINAL" = 'success' ]; then | |
| lint_note='Lint failed on the first pass; the AI repair pass fixed it.' | |
| else | |
| lint_note='LINT STILL FAILS after the AI repair pass — the CI lint job will be red and needs a manual fix.' | |
| fi | |
| # Unconditionally, BEFORE `git add -A`: these are untracked and not gitignored, and the | |
| # re-restrict step that would have cleaned them only runs on the repair path. On the | |
| # common clean-lint path they would otherwise be committed — and on a no-op wiring run | |
| # they would be the ONLY staged change, turning "the AI changed nothing" into a bogus | |
| # wiring commit. | |
| rm -f .spec-sync-lint.log .spec-sync-diff.txt | |
| git add -A | |
| # Trusted list of what the AI passes changed (the tool-call summaries above log no values). | |
| git diff --cached --stat | |
| if git diff --cached --quiet; then | |
| echo "AI step produced no changes." | |
| echo "changed=false" >> "$GITHUB_OUTPUT" | |
| echo "summary=Mechanical-only PR (AI wiring added no changes — e.g. workflow-only drift). Human review required before merge." >> "$GITHUB_OUTPUT" | |
| else | |
| # --no-verify belts-and-braces on top of removing .git/hooks above. | |
| git commit --no-verify -m "feat(spec-sync): wire client.v2 to spec diff (AI)" | |
| echo "changed=true" >> "$GITHUB_OUTPUT" | |
| echo "summary=Mechanical snapshot + AI wiring committed (a draft a human finishes). ${lint_note} Gates must pass; human review required before merge." >> "$GITHUB_OUTPUT" | |
| fi | |
| # The only step that sees the token, and it runs exactly one command over an already-built | |
| # commit. The credential was stripped from .git/config before the AI ever ran, so it is | |
| # supplied inline here rather than via the checkout's persisted config. GitHub masks it. | |
| - name: Push AI wiring | |
| id: ai_push | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_commit.outputs.changed == 'true' | |
| env: | |
| GH_PUSH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| # -c overrides any on-disk config, so even a tampered .git/config cannot point a pre-push | |
| # hook at a surviving script in src/ and have it run in the one step that holds the token. | |
| run: git -c core.hooksPath=/dev/null push "https://x-access-token:${GH_PUSH_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" HEAD:"$SYNC_BRANCH" | |
| # Rewrite the PR description: keep the static process/safety preamble from "Open sync PR" and | |
| # APPEND an LLM "## What changed" summary of the actual PR diff. Logic is shared by both jobs | |
| # in scripts/spec-sync/summarize-pr.sh. Best-effort — the static body stands if it degrades; | |
| # the script only edits the PR body via SPEC_SYNC_TOKEN, so an injected spec description in | |
| # the diff can't escalate past text. | |
| - name: Summarize the PR diff into the description | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' | |
| continue-on-error: true # a summary is nice-to-have; never red the run over it | |
| env: | |
| GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }} | |
| ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} | |
| SUMMARY_TITLE_PREFIX: 'spec-sync(v2)' # LLM generates only the title summary after this prefix | |
| # /v2/workflow* is intentionally NOT wired (see the AI-wiring prompt above), so keep it out | |
| # of "What changed" — else a workflow-only drift would advertise SDK endpoints the PR never | |
| # implemented. | |
| SUMMARY_SCOPE_NOTE: 'Scope: this SDK intentionally does NOT implement /v2/workflow, /v2/workflow/jobs, or /v2/workflow/jobs/{job_id}; that surface is deferred and is not wired into the client. Do not describe any /v2/workflow* route as a client/SDK change. If the diff touches only workflow, say the spec snapshot was updated but no client surface changed.' | |
| run: ./scripts/spec-sync/summarize-pr.sh "${{ steps.open_pr.outputs.url }}" | |
| # The AI-wiring outcome, as a reply under the PR-opened thread root (steps.root). | |
| - name: Slack — AI wiring result | |
| if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != '' | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: info | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: AI wiring' | |
| text: ${{ steps.ai_commit.outputs.summary }} | |
| thread_ts: ${{ steps.root.outputs.ts }} | |
| # Catch-all failure alert for this job. An unavailable spec source (exit 20) is skipped earlier | |
| # and never reaches here; what this calls out is a reachable source that errored — an HTTP error | |
| # status / empty / invalid spec, or a script error (exit not 0/10/20) — and an AI-wiring crash | |
| # (PR left mechanical-only), with a generic fallback. | |
| - name: Slack — spec-sync failed | |
| if: failure() | |
| uses: ./.github/actions/slack-notify | |
| with: | |
| bot_token: ${{ secrets.SLACK_BOT_TOKEN }} | |
| webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }} | |
| status: failure | |
| title: 'spec-sync ${{ env.SPEC_LABEL }}: run failed' | |
| text: >- | |
| ${{ (steps.drift.outputs.code != '' && steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20') | |
| && format('Spec fetch/normalize failed (exit {0}) — staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored (an unbooked cluster 404s → exit 20, skipped, not alerted). Investigate.', steps.drift.outputs.code) | |
| || (steps.ai_wiring.outcome == 'failure' | |
| && 'AI wiring step failed. Any open sync PR has only the mechanical commit and needs manual wiring.' | |
| || 'spec-sync run failed. See the workflow run for details.') }} | |
| · <${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}|View run> | |
| thread_ts: ${{ steps.root.outputs.ts }} |