Skip to content

Spec Sync

Spec Sync #1930

Workflow file for this run

name: Spec Sync
on:
schedule:
- cron: '0 * * * *' # hourly; cron covers correctness, dispatch cuts latency
workflow_dispatch: {}
# Serialize runs so a cron tick and a manual dispatch can't race on the sync branch.
concurrency:
group: spec-sync
cancel-in-progress: false
permissions:
contents: read # all writes go through SPEC_SYNC_TOKEN, never GITHUB_TOKEN
env:
# jq program for the "tool-call summary" step that follows each claude-code-action step. It prints
# ONLY fixed metadata: the tool name and the NAMES of the arguments it was given — never argument
# values. A prompt-injected agent that has Read its own environment could hex- or base64-encode a
# credential into a file path, and neither a character allowlist nor GitHub's literal secret
# masking would catch that; the same goes for assistant text, tool results and file contents, so
# none of those are printed either. What the agent actually changed is listed from a trusted
# source instead: the commit step prints `git diff --cached --stat`. Held here, in the trusted
# workflow file, so the summary can run right after the AI step without executing anything the
# agent could have edited.
TOOL_CALL_SUMMARY_JQ: >-
(if type == "array" then .[] else . end)
| select(.type == "assistant") | .message.content[]?
| select(.type == "tool_use")
| "[tool] " + (.name | tostring | if test("^[A-Za-z]{1,32}$") then . else "<redacted>" end)
+ " " + ((.input // {}) | keys | map(if test("^[A-Za-z_]{1,32}$") then . else "<redacted>" end) | join(","))
jobs:
spec-sync:
# Guard so the scheduled/dispatch job only runs on the canonical repo — forks lack the
# secrets and would just produce noisy failing runs.
if: github.repository == 'landing-ai/ade-python'
runs-on: ubuntu-latest
timeout-minutes: 60 # headroom for the AI wiring step (up to --max-turns 250) plus rye sync
env:
V1_SPEC_URL: https://api.va.staging.landing.ai/v1/ade/openapi.json # staging drives the loop
SYNC_BRANCH: spec-sync/v1 # fixed branch: reruns update one PR in place, never a pile of them
SPEC_LABEL: V1 # used only in Slack/PR-comment copy to disambiguate the two jobs
steps:
- uses: actions/checkout@v6
with:
# SPEC_SYNC_TOKEN is a fine-grained PAT (Contents: RW, Pull requests: RW) scoped to this
# repo. It must NOT be the default GITHUB_TOKEN: pushes/PRs authored by GITHUB_TOKEN do
# not trigger the gate workflows (anti-recursion).
token: ${{ secrets.SPEC_SYNC_TOKEN }}
fetch-depth: 0 # tags needed for surface-lock baseline
- name: Install Rye
run: |
curl -sSf https://rye.astral.sh/get | bash
echo "$HOME/.rye/shims" >> "$GITHUB_PATH"
env:
RYE_VERSION: '0.44.0'
RYE_INSTALL_OPTION: '--yes'
- name: Install dependencies
run: rye sync --all-features
- name: Detect drift
id: drift
run: |
set +e
./scripts/spec-sync/check-drift.sh "$V1_SPEC_URL" specs/v1-ade.json
echo "code=$?" >> "$GITHUB_OUTPUT"
- name: No drift
if: steps.drift.outputs.code == '0'
run: echo "specs in sync; nothing to do."
# An unavailable spec source (check-drift exit 20) is the EXPECTED case on staging, not an
# incident: staging auto-reclaims and must be booked to come back — an unbooked cluster serves a
# 404 (or the host stops answering entirely). Treat ONLY this as a no-op — log and end the run
# cleanly (no drift, no PR, no Slack); the next hourly run picks up any drift once staging is
# booked. Every other failure (a reachable source returning 401/403/5xx, an empty/invalid spec,
# or a script error) falls through to "Fail on spec error" below and alerts.
- name: Spec source unavailable — skip
if: steps.drift.outputs.code == '20'
run: echo "spec source unavailable (exit 20); staging is likely unbooked — skipping this run."
# A reachable-but-invalid spec (empty/whitespace body, malformed JSON) or a script error is a
# real problem — fail loudly so the catch-all alert fires. Distinct from the exit-20 skip above.
- name: Fail on spec error
if: steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20'
run: |
echo "spec fetch/normalize failed (exit ${{ steps.drift.outputs.code }}); staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored."
exit 1
# If a sync PR is already open, do nothing: it's awaiting human review, and re-running
# would duplicate gate runs and Claude API spend every hour until it merges.
- name: Check for an open sync PR
id: existing
if: steps.drift.outputs.code == '10'
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: |
if gh pr list --state open --head "$SYNC_BRANCH" --json number --jq '.[0].number' | grep -q .; then
echo "open=true" >> "$GITHUB_OUTPUT"
else
echo "open=false" >> "$GITHUB_OUTPUT"
fi
# A sync PR is already open. Every hourly tick still reports drift (check-drift compares against
# main, not the PR branch), so instead of a bare skip: detect drift that appeared BEYOND the
# open PR and, if any, annotate the PR + emit one Slack ping per distinct new spec (deduped).
- name: New drift behind the open PR
id: behind
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true'
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: ./scripts/spec-sync/notify-new-drift.sh specs/v1-ade.json "$SYNC_BRANCH" "$SPEC_LABEL"
- name: Slack — new drift behind the open PR
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true' && steps.behind.outputs.notify == 'true'
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: warning
title: 'spec-sync ${{ env.SPEC_LABEL }}: new drift behind the open PR'
text: 'New ${{ env.SPEC_LABEL }} spec drift appeared beyond the <${{ steps.behind.outputs.pr_url }}|open sync PR> (live-spec `${{ steps.behind.outputs.hash }}`). Merge or close it and the next run opens a fresh one covering the rest.'
thread_ts: ${{ steps.behind.outputs.thread_ts }}
# ---- Phase 1: mechanical (deterministic) ----
- name: Mechanical commit + push
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
run: |
git config user.name "spec-sync[bot]"
git config user.email "spec-sync@users.noreply.github.com"
git checkout -B "$SYNC_BRANCH"
./scripts/spec-sync/gen-models.sh specs/v1-ade.json specs/_generated/v1_models.py
git add specs/v1-ade.json specs/_generated/v1_models.py
git commit -m "chore(spec-sync): update V1 spec snapshot + regenerated reference models"
git push --force origin "$SYNC_BRANCH"
# Open the PR now — before the AI step — so a run that dies in phase 2 leaves a visible PR
# (with just the mechanical commit) instead of an orphan branch.
- name: Open sync PR
id: open_pr
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: |
url="$(gh pr create --base main --head "$SYNC_BRANCH" \
--title "spec-sync: track V1 spec drift" \
--body $'Automated spec-sync PR.\n\n- **Commit 1 (mechanical):** normalized spec snapshot + regenerated reference models.\n- **Commit 2 (AI):** resources/methods/tests/docs wired from the spec diff (added after this PR opened).\n\nGates (surface-lock, contract tests, lint/test/typecheck) must pass. **Human review required before merge.**')"
echo "url=$url" >> "$GITHUB_OUTPUT"
# Post the Slack thread ROOT now — right after the PR exists, before AI wiring — so every later
# event (AI outcome, new drift, gates, merge) replies under it. Persist its ts on the PR body
# so other runs/workflows can find the thread. (Editing the body doesn't re-trigger pr-gates,
# which only fires on opened/synchronize/reopened.)
- name: Slack — PR opened (thread root)
id: root
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: info
title: 'spec-sync ${{ env.SPEC_LABEL }}: new spec drift → PR opened'
text: '<${{ steps.open_pr.outputs.url }}|Open the PR> — mechanical spec snapshot committed; wiring the SDK next. Human review required before merge.'
- name: Save thread root on the PR
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != ''
continue-on-error: true # bookkeeping only; thread-ts.sh fails closed, and a hiccup must not red the run
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: ./scripts/spec-sync/thread-ts.sh set "${{ steps.open_pr.outputs.url }}" "${{ steps.root.outputs.ts }}"
# ---- Phase 2: AI wiring — edits ONLY. The agent holds no git/push capability, so a
# prompt-injected spec description cannot route around review to push to a branch. ----
- name: AI wiring (edits only)
id: ai_wiring
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
uses: anthropics/claude-code-action@v1
with:
anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }}
github_token: ${{ secrets.SPEC_SYNC_TOKEN }}
prompt: |
The previous commit updated specs/v1-ade.json and regenerated reference models in
specs/_generated/v1_models.py. Wire the SDK to match the spec diff.
First inspect the mechanical commit diff (`git diff HEAD~1..HEAD`) and read
src/landingai_ade/resources/parse_jobs.py — mirror its structure exactly for any
new resource (sync + async classes, the raw/streaming response wrappers, param
TypedDicts under src/landingai_ade/types/, response models, and registration in
src/landingai_ade/resources/__init__.py and the client). Add tests under
tests/api_resources/ and update api.md and the README examples. Then run
`./scripts/format` and `./scripts/lint` and fix anything they report.
Rules: PURELY ADDITIVE — never modify or remove an existing public signature (the
surface-lock CI job will fail the PR). `git diff` (read-only, to inspect the change)
is fine; do NOT stage, commit, push, or run any other git command — a later workflow
step commits and pushes your edits.
# Pin the model: the action otherwise inherits Claude Code's default, which floats
# across CLI releases and makes spec-sync output non-reproducible. `[1m]` = 1M context.
# claude-opus-5 is the pin in BOTH SDK repos (ade-python and ade-typescript) — keep them
# equal, or the two SDKs quietly diverge on the same spec change.
claude_args: |
--model "claude-opus-5[1m]"
--max-turns 250
--allowedTools "Edit,Write,Read,Glob,Grep,Bash(rye *),Bash(./scripts/*),Bash(git diff:*)"
# Print the agent's tool calls — tool name and argument NAMES only, never values — to the step
# log, so the shape of what the agent did can be audited later (the commit step lists what it
# changed, from the trusted checkout):
# the action logs only the init and result summaries, and #153 had to be root-caused without
# any trace. Deliberately NO argument values, assistant text, tool results or commands: this agent
# holds credential-bearing inputs and processes an untrusted spec, and a prompt-injected agent
# could carry a transformed or chunked credential past GitHub's literal secret masking in
# free-form narration. Runs straight off the action's own output file with the jq program held
# in this workflow's top-level env, so it executes nothing the agent could have edited.
# Best-effort.
- name: AI wiring tool-call summary
if: always() && steps.ai_wiring.outputs.execution_file != ''
continue-on-error: true
env:
TRANSCRIPT: ${{ steps.ai_wiring.outputs.execution_file }}
run: |
echo "::group::AI wiring tool-call summary"
set -o pipefail
# Prefix every line: nothing the agent could influence may start at column 0, where a leading
# `::` would be parsed as a workflow command (e.g. ::stop-commands::) even inside a group.
jq -r "$TOOL_CALL_SUMMARY_JQ" "$TRANSCRIPT" | tr -d '\r' | sed 's/^/ /' || echo " tool-call summary unavailable (unexpected file shape)"
echo "::endgroup::"
# Deterministic: commit whatever the AI edited and push to the sync branch. This guarantees
# the AI changes actually reach the PR (the action's automation mode does not push), and it
# is the only step with push capability.
- name: Commit + push AI wiring
id: ai_push
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
run: |
git add -A
# Trusted list of what the AI pass changed (the tool-call summary above logs no values).
git diff --cached --stat
if git diff --cached --quiet; then
echo "AI step produced no changes."
echo "summary=Mechanical-only PR (AI wiring added no changes). Human review required before merge." >> "$GITHUB_OUTPUT"
else
git commit -m "feat(spec-sync): wire SDK to spec diff (AI)"
git push origin HEAD:"$SYNC_BRANCH"
echo "summary=Mechanical snapshot + AI wiring committed. Gates must pass; human review required before merge." >> "$GITHUB_OUTPUT"
fi
# Rewrite the PR description: keep the static process/safety preamble from "Open sync PR" and
# APPEND an LLM "## What changed" summary of the actual PR diff. Logic is shared by both jobs
# in scripts/spec-sync/summarize-pr.sh. Best-effort — the static body stands if it degrades;
# the script only edits the PR body via SPEC_SYNC_TOKEN, so an injected spec description in
# the diff can't escalate past text.
- name: Summarize the PR diff into the description
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
continue-on-error: true # a summary is nice-to-have; never red the run over it
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
SUMMARY_TITLE_PREFIX: 'spec-sync(v1)' # LLM generates only the title summary after this prefix
run: ./scripts/spec-sync/summarize-pr.sh "${{ steps.open_pr.outputs.url }}"
# The AI-wiring outcome, as a reply under the PR-opened thread root (steps.root).
- name: Slack — AI wiring result
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != ''
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: info
title: 'spec-sync ${{ env.SPEC_LABEL }}: AI wiring'
text: ${{ steps.ai_push.outputs.summary }}
thread_ts: ${{ steps.root.outputs.ts }}
# Catch-all failure alert for this job. An unavailable spec source (exit 20) is skipped earlier
# and never reaches here; what this calls out is a reachable source that errored — an HTTP error
# status / empty / invalid spec, or a script error (exit not 0/10/20) — and an AI-wiring crash
# (PR left mechanical-only), with a generic fallback.
- name: Slack — spec-sync failed
if: failure()
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: failure
title: 'spec-sync ${{ env.SPEC_LABEL }}: run failed'
text: >-
${{ (steps.drift.outputs.code != '' && steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20')
&& format('Spec fetch/normalize failed (exit {0}) — staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored (an unbooked cluster 404s → exit 20, skipped, not alerted). Investigate.', steps.drift.outputs.code)
|| (steps.ai_wiring.outcome == 'failure'
&& 'AI wiring step failed. Any open sync PR has only the mechanical commit and needs manual wiring.'
|| 'spec-sync run failed. See the workflow run for details.') }}
· <${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}|View run>
thread_ts: ${{ steps.root.outputs.ts }}
# V2 loop. Same machinery as the V1 job above, pointed at the V2 spec on the AIDE gateway
# and a separate sync branch, with a V2-tailored wiring prompt. Kept as a distinct job (not a
# matrix) so the proven V1 loop is untouched and the two prompts can diverge freely.
#
# HOSTS (do not conflate): the V2 *spec* is published at aide.[env]/openapi.json; the V2 *API*
# (what the SDK and contract tests call) is api.ade.[env]. We fetch drift from aide; the SDK
# never talks to aide (its paths are staff-SSO-gated).
spec-sync-v2:
if: github.repository == 'landing-ai/ade-python'
runs-on: ubuntu-latest
timeout-minutes: 60 # headroom for the AI wiring step (up to --max-turns 250) plus rye sync
env:
V2_SPEC_URL: https://aide.staging.landing.ai/openapi.json # staging drives the loop
SYNC_BRANCH: spec-sync/v2 # separate branch so V1 and V2 sync PRs never collide
SPEC_LABEL: V2 # used only in Slack/PR-comment copy to disambiguate the two jobs
steps:
- uses: actions/checkout@v6
with:
# SPEC_SYNC_TOKEN (fine-grained PAT), NOT GITHUB_TOKEN: pushes/PRs authored by
# GITHUB_TOKEN do not trigger the gate workflows (anti-recursion).
token: ${{ secrets.SPEC_SYNC_TOKEN }}
fetch-depth: 0 # tags needed for surface-lock baseline
- name: Install Rye
run: |
curl -sSf https://rye.astral.sh/get | bash
echo "$HOME/.rye/shims" >> "$GITHUB_PATH"
env:
RYE_VERSION: '0.44.0'
RYE_INSTALL_OPTION: '--yes'
- name: Install dependencies
run: rye sync --all-features
- name: Detect drift
id: drift
run: |
set +e
./scripts/spec-sync/check-drift.sh "$V2_SPEC_URL" specs/v2-aide.json
echo "code=$?" >> "$GITHUB_OUTPUT"
- name: No drift
if: steps.drift.outputs.code == '0'
run: echo "specs in sync; nothing to do."
# An unavailable spec source (check-drift exit 20) is the EXPECTED case on staging, not an
# incident: staging auto-reclaims and must be booked to come back — an unbooked cluster serves a
# 404 (or the host stops answering entirely). Treat ONLY this as a no-op — log and end the run
# cleanly (no drift, no PR, no Slack); the next hourly run picks up any drift once staging is
# booked. Every other failure (a reachable source returning 401/403/5xx, an empty/invalid spec,
# or a script error) falls through to "Fail on spec error" below and alerts.
- name: Spec source unavailable — skip
if: steps.drift.outputs.code == '20'
run: echo "spec source unavailable (exit 20); staging is likely unbooked — skipping this run."
# A reachable-but-invalid spec (empty/whitespace body, malformed JSON) or a script error is a
# real problem — fail loudly so the catch-all alert fires. Distinct from the exit-20 skip above.
- name: Fail on spec error
if: steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20'
run: |
echo "spec fetch/normalize failed (exit ${{ steps.drift.outputs.code }}); staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored."
exit 1
- name: Check for an open sync PR
id: existing
if: steps.drift.outputs.code == '10'
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: |
# Fail CLOSED: capture the query result separately so a transient gh/API failure aborts
# the job (default bash is `set -eo pipefail`, but a failure inside an `if` condition is
# exempt from set -e — which would silently become open=false and force-push over an
# existing human-reviewed PR's branch).
count="$(gh pr list --state open --head "$SYNC_BRANCH" --json number --jq 'length')"
if [ "$count" -gt 0 ]; then
echo "open=true" >> "$GITHUB_OUTPUT"
else
echo "open=false" >> "$GITHUB_OUTPUT"
fi
# A sync PR is already open. Every hourly tick still reports drift (check-drift compares against
# main, not the PR branch), so instead of a bare skip: detect drift that appeared BEYOND the
# open PR and, if any, annotate the PR + emit one Slack ping per distinct new spec (deduped).
- name: New drift behind the open PR
id: behind
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true'
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: ./scripts/spec-sync/notify-new-drift.sh specs/v2-aide.json "$SYNC_BRANCH" "$SPEC_LABEL"
- name: Slack — new drift behind the open PR
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open == 'true' && steps.behind.outputs.notify == 'true'
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: warning
title: 'spec-sync ${{ env.SPEC_LABEL }}: new drift behind the open PR'
text: 'New ${{ env.SPEC_LABEL }} spec drift appeared beyond the <${{ steps.behind.outputs.pr_url }}|open sync PR> (live-spec `${{ steps.behind.outputs.hash }}`). Merge or close it and the next run opens a fresh one covering the rest.'
thread_ts: ${{ steps.behind.outputs.thread_ts }}
# ---- Phase 1: mechanical (deterministic) ----
- name: Mechanical commit + push
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
run: |
git config user.name "spec-sync[bot]"
git config user.email "spec-sync@users.noreply.github.com"
git checkout -B "$SYNC_BRANCH"
./scripts/spec-sync/gen-models.sh specs/v2-aide.json specs/_generated/v2_models.py
git add specs/v2-aide.json specs/_generated/v2_models.py
git commit -m "chore(spec-sync): update V2 spec snapshot + regenerated reference models"
git push --force origin "$SYNC_BRANCH"
- name: Open sync PR
id: open_pr
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: |
url="$(gh pr create --base main --head "$SYNC_BRANCH" \
--title "spec-sync: track V2 spec drift" \
--body $'Automated V2 spec-sync PR (`client.v2`).\n\n- **Commit 1 (mechanical):** normalized V2 spec snapshot + regenerated reference models.\n- **Commit 2 (AI, only if the spec diff needs SDK changes):** client.v2 resources/methods/tests/docs wired from the diff, added after this PR opened. Workflow-only drift is excluded and an AI no-op is skipped, so some drifts produce a **mechanical-only PR with no second commit**.\n\nGates (surface-lock, V2 contract tests, lint/test/typecheck) must pass. When present, the AI commit is a **draft a human finishes** (the V2 ergonomic layer — unified Job, dual-host, schema coercion — is not in the spec). **Human review required before merge.**')"
echo "url=$url" >> "$GITHUB_OUTPUT"
# Post the Slack thread ROOT now — right after the PR exists, before AI wiring — so every later
# event (AI outcome, new drift, gates, merge) replies under it. Persist its ts on the PR body
# so other runs/workflows can find the thread. (Editing the body doesn't re-trigger pr-gates,
# which only fires on opened/synchronize/reopened.)
- name: Slack — PR opened (thread root)
id: root
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: info
title: 'spec-sync ${{ env.SPEC_LABEL }}: new spec drift → PR opened'
text: '<${{ steps.open_pr.outputs.url }}|Open the PR> — mechanical spec snapshot committed; wiring the SDK next. Human review required before merge.'
- name: Save thread root on the PR
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != ''
continue-on-error: true # bookkeeping only; thread-ts.sh fails closed, and a hiccup must not red the run
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
run: ./scripts/spec-sync/thread-ts.sh set "${{ steps.open_pr.outputs.url }}" "${{ steps.root.outputs.ts }}"
# ---- Everything the AI pass may see or reach is decided HERE, in a trusted step --------
#
# 1. Precompute the mechanical diff. The agent needs it, but must not run `git diff` to get
# it: git honours `diff.external` / textconv drivers from .git/config, so an agent that
# can write .git/config turns a "read-only" git diff into arbitrary execution INSIDE its
# own step — before any allowlist or hardening runs. Handing it a plain file keeps the
# input and removes the primitive. (Untracked and not in the `git clean` exclusions, so
# the allowlist step deletes it before anything is committed.)
#
# 2. Take the push credential away BEFORE the agent runs, not after. actions/checkout
# persists it in .git/config; nothing between here and the explicit push needs it, so
# even a successful in-step escape has no token to reach. Snapshot the now-clean config
# so each post-AI restore can put it back wholesale, rather than unsetting known-bad keys
# one at a time — an agent can add `filter.*.clean`, `diff.external`, aliases, and any
# enumeration of those will be incomplete.
#
# 3. Drop .git/hooks and .git/info. Neither is a worktree path, so neither `git checkout`
# nor `git clean` restores them: a planted pre-commit hook runs on our own `git commit`,
# and one line in .git/info/exclude makes `git clean -fd` skip the agent's own file —
# walking it straight past the product-code allowlist.
- name: Precompute the spec diff and harden the checkout
id: harden
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
run: |
git diff HEAD~1..HEAD > .spec-sync-diff.txt
git config --local --unset-all 'http.https://github.com/.extraheader' || true
git config --local --unset-all core.hooksPath || true
rm -rf .git/hooks .git/info
# The snapshot is a step OUTPUT, not a file. Every path on disk — $RUNNER_TEMP included —
# is writable by AI-authored code: nothing here executes src/ any more, but ruff/pyright
# run as binaries out of the gitignored .venv, which the allowlist cannot clean, so a
# replaced binary could overwrite a file snapshot between it being taken and restored and
# thereby reinstall an attacker-controlled core.hooksPath before the credentialed push.
# Step outputs are captured by the runner when this step ends and injected as expressions
# afterwards, so nothing running later in the job can alter this value.
echo "gitconfig=$(base64 -w0 < .git/config)" >> "$GITHUB_OUTPUT"
# ---- Phase 2: AI wiring — edits ONLY (no git/push; a later step commits). ----
- name: AI wiring (edits only)
id: ai_wiring
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
uses: anthropics/claude-code-action@v1
with:
anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }}
# NOT SPEC_SYNC_TOKEN. claude-code-action exports the token it is given into the agent's
# environment (and configures `origin` with it), so handing it the write PAT would put a
# push-capable credential inside the step that processes the untrusted spec — where the
# agent's own Read tool could recover it and Write it into an allowlisted file that the
# trusted push then publishes. `github.token` is read-only here (see `permissions:` at
# the top of this file); SPEC_SYNC_TOKEN is reserved for the trusted push step.
github_token: ${{ github.token }}
prompt: |
The previous commit updated specs/v2-aide.json and regenerated reference models in
specs/_generated/v2_models.py. Wire the client.v2 SDK to match the spec diff. This is
the V2 surface (client.v2.*), NOT V1 — do not touch V1 resources.
SCOPE — exactly which routes back client.v2: the `/v2/*` paths in this spec, and
nothing else. This spec is the AIDE gateway's and ALSO carries `/v1/*` routes (for
example /v1/ade/parse, /v1/classify, /v1/split): that is the V1-compatibility surface,
tracked by the separate V1 spec-sync job against its own spec, and it is NOT part of
client.v2. Do not wire a `/v1/*` route into client.v2, and never "translate" one into a
`/v2/...` URL. Every path you write into a resource must appear verbatim under `paths`
in specs/v2-aide.json — a trusted step (scripts/spec-sync/check-v2-paths.sh) verifies
this after you finish and rejects any path the spec does not have. A diff whose only new
ROUTES are `/v1/*` is not "nothing to do": still wire every in-scope change listed
below (a new request field on /v2/parse, say) and leave the `/v1/*` routes alone.
SCOPE, one level down — exactly which FIELDS back a /v2 request: the TOP-LEVEL
properties of that operation's `requestBody` schema in this spec, and nothing else. A
field the spec declares inside a nested object is not a top-level field — the
encrypted-PDF password is declared at `options.password` on /v2/parse and
/v2/parse/jobs and NOWHERE else, so it is sent there and only there. Two traps, both
of which have bitten this repo: (a) the SAME spec declares a top-level `password` on
/v1/ade/parse and /v1/ade/parse/jobs — same name, different level, different route
family — and lifting a `/v1/*` top-level field onto a /v2 operation is the field-level
version of translating a /v1 route into a /v2 URL; (b) inventing a top-level
convenience parameter that folds one wire field into another. Exactly three exist
(`password` on parse, `strict` and `grounding` on extract); they are cross-SDK decisions
shared with ade-typescript and written down in CONTRIBUTING.md, and changing one is a
maintainer's call — do NOT add, remove or re-target an alias. Wire the spec's own
shape, and if a spec change looks like it needs a new one, leave that field unwired
and say so in your summary.
OUT OF SCOPE — do NOT implement: /v2/workflow, /v2/workflow/jobs, and
/v2/workflow/jobs/{job_id}. That surface is intentionally deferred; skip it entirely
even though it appears in the spec and reference models. If the spec diff touches only
workflow and/or `/v1/*` routes, make no changes.
First read the ENTIRE mechanical commit diff, which a trusted step has already written
to `.spec-sync-diff.txt` in the repo root (you have no shell — use Read, do not try to
run git; the file runs to thousands of lines, so page through it with offset/limit until
you have seen the final hunk) — the operation/path hunks AND changed entries under
`components.schemas` — to find every change backing client.v2. A response-field
addition often changes ONLY a component schema (e.g. /v2/parse references
#/components/schemas/ParseResponse) and its generated model, not the operation hunk, so
for every changed schema trace its `$ref` consumers to the non-workflow `/v2/*`
operations that use it. Cover: (a) newly added `/v2/*` routes; (b) request-field,
response-field or schema changes on existing `/v2/*` operations, INCLUDING
component-only changes. A new optional request field on /v2/parse is exactly case (b)
and must be wired even when the same diff is dominated by unrelated `/v1/*` additions.
Distinguish REQUEST fields — wire as a new optional keyword parameter on the existing
method (see the additive-changes rule below) — from RESPONSE-model fields — add the
field to the corresponding response model under src/landingai_ade/types/v2/. For a
genuinely new route, read src/landingai_ade/resources/v2/extract.py and
src/landingai_ade/resources/v2/parse.py and mirror their structure:
- Add a resource module under src/landingai_ade/resources/v2/ with a sync run()
(multipart or JSON per the spec; route a 504 through
lib/v2_errors.raise_if_sync_timeout, passing THIS endpoint's own async route as
`jobs_resource` — e.g. `build_schema_jobs` — so the 504 remediation names the right
resource; never let it fall back to a parse/extract-specific message, and if the
helper doesn't yet take a per-endpoint `jobs_resource`, extend it and update every
existing call site and helper test to pass its own matching resource rather than
reusing another endpoint's wording) and, for an async route, a *JobsResource with
create / get / wait / list. Decide multipart-vs-JSON from the SPEC, not from a
generated class name: an operation accepts file uploads iff its
`requestBody.content` has a `multipart/form-data` variant with a property (or array
item) of `format: binary`. For such a field, accept file inputs (FileTypes: a str for
inline content, or a Path/bytes/file object for uploads). Pick the content type from
what the operation DECLARES, not a fixed JSON default: send `multipart/form-data` when
any value is a file OR the operation declares only multipart (no `application/json`
variant — e.g. `/v2/parse`, where even an all-string `document_url` call must be
multipart); send JSON only when the operation declares an `application/json` variant
AND every value is a string. Identify the file-capable
request model by that `bytes`/binary field type — the generated `*PostRequest` /
`*PostRequest1` variants are ordered by content-type and the suffix can flip if the
spec reorders them, so NEVER key off the numeric suffix.
- Register it on the V2Resource/AsyncV2Resource container in
src/landingai_ade/resources/v2/v2.py exactly as parse/extract are wired (lazy
cached_property + an explicit-signature top-level delegator). Do NOT re-export the
sub-resource from src/landingai_ade/resources/v2/__init__.py — that file exports
only V2Resource/AsyncV2Resource, matching the reference files.
- Normalize any async job into the unified Job (src/landingai_ade/types/v2/job.py) by
adding a normalize_* function in src/landingai_ade/resources/v2/_normalize.py,
mirroring normalize_parse_job / normalize_extract_job (tolerant status via
_status(); map the envelope's result / error / timestamps).
- Put response models under src/landingai_ade/types/v2/. If a run accepts a JSON
Schema, accept a pydantic model / dict / JSON string and coerce via
lib/schema_utils.coerce_schema_to_dict.
- Build request URLs with V2ResourceMixin._v2_url — dual-host routing is automatic;
never hardcode a host.
Optionality comes from the schema, never from the prose: a response field is Optional
iff it is absent from that schema's `required` array. A field can be described as
"always present for model X" and still be optional — read `required`, not the
`description`. Model an optional field as `Optional[T] = None` and treat "the key is
missing" and "the key is null" as the same state (`if x is not None`), because the
gateway omits optional fields rather than sending an explicit null.
Add a live check to tests/contract/test_v2_smoke.py (marked `contract`) and unit tests
under tests/api_resources/. The contract file hits LIVE staging, so it must assert only
what staging is guaranteed to return: NEVER pin `model=` there (a model family the spec
documents may not be servable in whatever way the staging cluster happens to be
provisioned — the word-granularity family, currently `dpt-3-verity` and formerly
`dpt-3-fast`, needs a GPU-backed booking, and against one without it the
request hangs until the gate times out; that is an environment property, not an SDK
contract, so no live test should depend on it), and for an optional field assert
absent-or-valid ("if present, it is in range"), never that it is populated. Assertions
on a specific value belong in the mocked unit tests under tests/api_resources/, where
you control the response body — put them there instead, and reuse the module's existing
`staging_client` fixture rather than constructing your own client.
Update api.md, and create or update docs/v2-testing.md (the V2 QA guide) with the new
surface. Write clean, import-sorted code matching the style of the files you mirror — a
later fixed step runs `./scripts/format` and then `./scripts/lint`, and you get one
chance to fix whatever lint reports before the PR opens.
You may only edit product code: src/, tests/, docs/, and api.md. Do NOT edit scripts/,
.github/, specs/, or any dotfile.
The repo type-checks with pyright in STRICT mode, which you cannot run yourself. Two
rules that keep it green: annotate every parameter and local with its concrete type
(importing the real model from landingai_ade.types.v2) — never `object` or `Any` when
you then access attributes on the value, which yields `reportUnknown*` errors; and do
not add `# type: ignore` to silence an annotation you could have written correctly.
Before you finish, self-check coverage: list every operation hunk and every changed
component schema in `.spec-sync-diff.txt`, mark each IN SCOPE (`/v2/*`, non-workflow) or
OUT OF SCOPE (`/v1/*`, `/v2/workflow*`), confirm every in-scope item maps to an edit you
made, and confirm every URL path you wrote exists under `paths` in specs/v2-aide.json.
Close any gap before you stop.
Additive-changes rule: BACKWARD-COMPATIBLE ADDITIONS ONLY. You MAY add a new optional
keyword parameter (with a default) to an existing parse/extract method to wire an
additive request field — surface-lock permits that. You must NOT remove or rename a
public symbol or parameter, change a type, or make an existing parameter required;
surface-lock will fail the PR on any of those. You have NO shell and no git access at
all; the diff you need is already in `.spec-sync-diff.txt`. A later workflow step
commits and pushes your edits.
# Pin the model: the action otherwise inherits Claude Code's default, which floats
# across CLI releases and makes spec-sync output non-reproducible. `[1m]` = 1M context —
# the mechanical diff alone can run to ~100K tokens. claude-opus-5 is the pin in BOTH SDK
# repos; keep them equal. (On the drift behind #153 the then-pinned claude-opus-4-8
# rewrote new /v1 routes as non-existent /v2 ones and skipped the in-scope field, while
# ade-typescript's unpinned run — opus-5 by default — wired exactly the right thing, twice.)
#
# NO Bash tool. `Bash(git diff:*)` was allowed here so the agent could inspect the
# mechanical diff, but `git diff` is NOT inert: an agent that writes `.git/config` can
# point `diff.external` (or a textconv driver) at a script it also wrote, and the
# permitted `git diff` then EXECUTES it — inside the AI step, before any allowlist or
# hardening applies. A trusted step precomputes the diff into .spec-sync-diff.txt
# instead, so the agent keeps the input and loses the execution primitive.
claude_args: |
--model "claude-opus-5[1m]"
--max-turns 250
--allowedTools "Edit,Write,Read,Glob,Grep"
# Print the agent's tool calls — tool name and argument NAMES only, never values — to the step
# log, so the shape of what the agent did can be audited later (the commit step lists what it
# changed, from the trusted checkout):
# the action logs only the init and result summaries, and #153 had to be root-caused without
# any trace. Deliberately NO argument values, assistant text, tool results or commands: the spec this
# agent reads is untrusted input, and a prompt-injected agent that has Read its own environment
# could carry a transformed or chunked credential past GitHub's literal secret masking in
# free-form narration. Runs BEFORE anything AI-authored can execute, straight off the action's
# own output file, with the jq program held in this workflow's top-level env (not in a repo
# script the agent could have edited). Best-effort.
- name: AI wiring tool-call summary
if: always() && steps.ai_wiring.outputs.execution_file != ''
continue-on-error: true
env:
TRANSCRIPT: ${{ steps.ai_wiring.outputs.execution_file }}
run: |
echo "::group::AI wiring tool-call summary"
set -o pipefail
# Prefix every line: nothing the agent could influence may start at column 0, where a leading
# `::` would be parsed as a workflow command (e.g. ::stop-commands::) even inside a group.
jq -r "$TOOL_CALL_SUMMARY_JQ" "$TRANSCRIPT" | tr -d '\r' | sed 's/^/ /' || echo " tool-call summary unavailable (unexpected file shape)"
echo "::endgroup::"
# ---- Harden before running ANY trusted command over the AI's edits --------------------
#
# The AI step has no shell, so it cannot execute anything ITSELF. But the allowlist revert
# below is not a complete boundary, because `git checkout` only restores TRACKED files and
# `git clean -fd` (no -x) only deletes untracked, non-ignored ones. Two repo-wide Write
# targets survive both — and the trusted steps that follow EXECUTE both:
# * .venv/** — gitignored, so never cleaned; `rye run check:ruff` / `typecheck` / `format`
# all run binaries out of it.
# * .git/** — not a worktree path at all; `.git/hooks/pre-commit` would run on our own
# `git commit`, and `core.hooksPath` can redirect hooks anywhere.
# Either one turns "the AI wrote a file" into "code runs in a trusted step".
#
# So do two things before any of those steps: drop the push credential actions/checkout
# persisted into .git/config (nothing executed downstream can reach a token that is not
# there — the AI-wiring push below re-supplies it explicitly for that one command), and
# neutralize the hook paths so our own git commands cannot run anything.
#
# This narrows the blast radius; it does not make executing AI-authored code safe. Full
# isolation — run the AI pass in a credential-free checkout and copy only the validated
# allowlist diff back — is the real fix and is deliberately NOT attempted here.
- name: Restore trusted git metadata after the AI pass
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
env:
TRUSTED_GITCONFIG: ${{ steps.harden.outputs.gitconfig }}
run: |
printf '%s' "$TRUSTED_GITCONFIG" | base64 -d > .git/config
rm -rf .git/hooks .git/info
# Its Write tool is repo-wide, so before running the trusted
# formatter — and before the `git add -A` below — we enforce the product-code ALLOWLIST:
# revert every tracked edit and delete every untracked file OUTSIDE src/ tests/ docs/ api.md.
# This stops an injected edit to e.g. pyproject.toml (which defines what `rye run format`
# runs) from executing here, and stops any out-of-scope edit from reaching the commit.
# (`git clean` has no -x, so gitignored paths like .venv are untouched.)
- name: Restrict AI edits to product code, then format
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
run: |
git checkout -- . ':(exclude)src' ':(exclude)tests' ':(exclude)docs' ':(exclude)api.md'
git clean -fd -e /src -e /tests -e /docs
# A .gitattributes INSIDE the allowlist survives the two lines above and can bind content
# filters / diff drivers to paths. The config half of that attack is already gone (the
# restore step above replaces .git/config wholesale), but remove the file half too. This
# repo tracks no .gitattributes anywhere, so deleting any that appeared is always safe.
find src tests docs -name .gitattributes -delete 2>/dev/null || true
./scripts/format
# ---------------------------------------------------------------------------------------
# Verify the AI wiring, then give it ONE chance to fix itself.
#
# The wiring step has no shell, so nothing has type-checked its output: the step above only
# *formats*. Historically that meant every AI mistake pyright can see (annotating a param
# `object` and reaching for attributes on it, `# type: ignore` over a fixable annotation)
# landed as a red `lint` job on a PR that otherwise looked review-ready. So run the real lint
# HERE, in a trusted step, and hand its output back to a second shell-less Claude pass.
#
# This closes the loop WITHOUT granting the AI a shell: nothing here EXECUTES the code it just
# wrote, and its edits are still funnelled through the product-code allowlist below before they
# can reach a commit. The lint log it reads is ruff/pyright/mypy output quoting the AI's own
# just-written code — no new trust boundary is crossed by feeding it back.
#
# That "nothing executes it" property is why this runs a STATIC SUBSET and not `./scripts/lint`:
# that script — and the `rye run lint` chain it calls — ends in `check:importable`, i.e.
# `python -c 'import landingai_ade'`, which would import the AI's freshly written `src/` inside
# a trusted step that still holds the checkout's SPEC_SYNC_TOKEN. A top-level statement in any
# imported module would then run with push credentials, defeating the whole shell-less design.
# ruff, pyright, and mypy only ever READ the source. The import check is not lost — it still
# runs on the PR in CI's own `lint` job, which has no push credentials.
#
# `specs/` is excluded from pyright (see [tool.pyright] in pyproject.toml), so anything lint
# reports here lives in the AI's own edits under src/ tests/ docs/ — i.e. within its reach.
- name: Lint the AI wiring
id: ai_lint
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
# A lint failure here is the normal case this loop exists to handle, not a run failure.
continue-on-error: true
run: |
set -o pipefail
# Run all three even if one fails, so the repair pass sees every error at once.
status=0
rye run check:ruff 2>&1 | tee .spec-sync-lint.log || status=1
rye run typecheck 2>&1 | tee -a .spec-sync-lint.log || status=1
# Both directions: a wired path the spec lacks (#153 invented /v2/classify and /v2/split)
# and a /v2 spec route nothing sends. grep + jq over source only — as safe here as ruff.
./scripts/spec-sync/check-v2-paths.sh specs/v2-aide.json src/landingai_ade/resources/v2 2>&1 | tee -a .spec-sync-lint.log || status=1
exit $status
- name: AI lint repair (edits only)
id: ai_lint_fix
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure'
uses: anthropics/claude-code-action@v1
with:
anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }}
# NOT SPEC_SYNC_TOKEN. claude-code-action exports the token it is given into the agent's
# environment (and configures `origin` with it), so handing it the write PAT would put a
# push-capable credential inside the step that processes the untrusted spec — where the
# agent's own Read tool could recover it and Write it into an allowlisted file that the
# trusted push then publishes. `github.token` is read-only here (see `permissions:` at
# the top of this file); SPEC_SYNC_TOKEN is reserved for the trusted push step.
github_token: ${{ github.token }}
prompt: |
The trusted lint step failed on the client.v2 wiring you just wrote. Read
.spec-sync-lint.log in the repo root — it is the captured output of that run (ruff
check, then pyright in STRICT mode and mypy, then scripts/spec-sync/check-v2-paths.sh,
which cross-checks the URL paths the V2 resources send against specs/v2-aide.json) —
and fix every error it reports.
Fix the CAUSE, not the symptom. Specifically, do NOT silence anything: no
`# type: ignore`, no `cast(Any, ...)`, no widening an annotation to `object`/`Any`, no
deleting the assertion or test that tripped the error. A `reportUnknown*` error means a
value has no concrete type — annotate it with the real model imported from
landingai_ade.types.v2 (a recursive helper over parse elements takes
`List[V2ParseElement]`, not `object`). Keep the behavior the wiring was written to have.
A `check-v2-paths` error means one of two things. (a) A resource sends a path that is
not under `paths` in specs/v2-aide.json: remove that method/resource together with its
registration in src/landingai_ade/resources/v2/v2.py, its types, tests and docs — never
invent or "translate" a path (a /v1/... route is NOT a /v2/... route; only `/v2/*`
backs client.v2). (b) A `/v2/*` route in the spec has no resource sending it: wire it,
mirroring src/landingai_ade/resources/v2/parse.py and extract.py. Paths under
/v2/workflow* are out of scope for every AI pass, this one included: never add,
remove or edit a workflow resource, and if a `check-v2-paths` error concerns a
/v2/workflow* path, leave it for a maintainer.
You may only edit src/, tests/, docs/, and api.md — the same allowlist as before. Do
NOT edit scripts/, .github/, specs/, pyproject.toml, or any dotfile (including
.spec-sync-lint.log itself: it is an input, and it is deleted before anything is
committed). You have NO shell and no git access at all.
# NO Bash tool — see the wiring step above for why `Bash(git diff:*)` was removed.
# Same model pin as the wiring step, for the same reasons.
claude_args: |
--model "claude-opus-5[1m]"
--max-turns 60
--allowedTools "Edit,Write,Read,Glob,Grep"
# Same tool-call summary as after the wiring step — see there for why it is the log, and why
# it runs before anything AI-authored executes.
- name: AI lint repair tool-call summary
if: always() && steps.ai_lint_fix.outputs.execution_file != ''
continue-on-error: true
env:
TRANSCRIPT: ${{ steps.ai_lint_fix.outputs.execution_file }}
run: |
echo "::group::AI lint repair tool-call summary"
set -o pipefail
# Prefix every line: nothing the agent could influence may start at column 0, where a leading
# `::` would be parsed as a workflow command (e.g. ::stop-commands::) even inside a group.
jq -r "$TOOL_CALL_SUMMARY_JQ" "$TRANSCRIPT" | tr -d '\r' | sed 's/^/ /' || echo " tool-call summary unavailable (unexpected file shape)"
echo "::endgroup::"
# The repair pass is exactly as untrusted as the wiring pass, so repeat the hardening too:
# it could have written .venv/** or .git/hooks/** just as the wiring pass could, and the
# format/re-lint/commit steps below execute both. (The credential unset is already done and
# idempotent; the hooks removal matters because a repair pass could recreate them.)
- name: Restore trusted git metadata after the repair pass
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure'
env:
TRUSTED_GITCONFIG: ${{ steps.harden.outputs.gitconfig }}
run: |
printf '%s' "$TRUSTED_GITCONFIG" | base64 -d > .git/config
rm -rf .git/hooks .git/info
# Re-apply the allowlist before formatting or committing. This also deletes
# .spec-sync-lint.log (untracked, not in the `git clean` exclusions), so the log never
# reaches the commit.
- name: Re-restrict AI edits to product code, then format
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure'
run: |
git checkout -- . ':(exclude)src' ':(exclude)tests' ':(exclude)docs' ':(exclude)api.md'
git clean -fd -e /src -e /tests -e /docs
# A .gitattributes INSIDE the allowlist survives the two lines above and can bind content
# filters / diff drivers to paths. The config half of that attack is already gone (the
# restore step above replaces .git/config wholesale), but remove the file half too. This
# repo tracks no .gitattributes anywhere, so deleting any that appeared is always safe.
find src tests docs -name .gitattributes -delete 2>/dev/null || true
./scripts/format
# Record whether the repair actually worked, so the Slack reply and the PR say so honestly
# instead of implying the wiring is clean when CI is about to go red.
- name: Re-lint after repair
id: ai_lint_final
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_lint.outcome == 'failure'
continue-on-error: true
# Same static subset as the first pass — see there for why `check:importable` is excluded.
run: |
status=0
rye run check:ruff || status=1
rye run typecheck || status=1
./scripts/spec-sync/check-v2-paths.sh specs/v2-aide.json src/landingai_ade/resources/v2 || status=1
exit $status
# LAST line of defence, and it must come after the last step that executes anything.
# `./scripts/format` and the linters all run binaries out of the gitignored .venv, which the
# allowlist cannot clean — so any of them could have written outside the allowlist, or
# recreated git config/hooks, AFTER the earlier restore ran. Without this, `git add -A` below
# would stage those post-allowlist edits and the trusted push would publish them.
#
# Runs unconditionally (not just on the repair path) because the plain path executes the
# formatter and linters too. Nothing after this point executes AI-authored content: the only
# remaining steps are `git add`/`git commit` and the push itself. Re-running the allowlist is
# safe and idempotent — src/ tests/ docs/ api.md are excluded, so every legitimate AI edit and
# every formatter result is preserved.
- name: Final restore + allowlist before staging
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
env:
TRUSTED_GITCONFIG: ${{ steps.harden.outputs.gitconfig }}
run: |
printf '%s' "$TRUSTED_GITCONFIG" | base64 -d > .git/config
rm -rf .git/hooks .git/info
git reset -q
git checkout -- . ':(exclude)src' ':(exclude)tests' ':(exclude)docs' ':(exclude)api.md'
git clean -fd -e /src -e /tests -e /docs
find src tests docs -name .gitattributes -delete 2>/dev/null || true
# Commit and push are SEPARATE steps, and the token appears only in the push step's env.
# `git add` runs content filters (`filter.<driver>.clean`), which an AI pass can install via
# .git/config plus a .gitattributes inside the allowlist — the restore step above removes the
# config half, and keeping the token out of this step's environment means even a filter that
# somehow survived has nothing to read. By the time the push step runs, staging and committing
# are already done and no AI-authored content is executed again.
- name: Commit AI wiring
id: ai_commit
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
env:
LINT_FIRST: ${{ steps.ai_lint.outcome }}
LINT_FINAL: ${{ steps.ai_lint_final.outcome }} # 'skipped' when the first lint passed
run: |
if [ "$LINT_FIRST" = 'success' ]; then
lint_note='Lint clean.'
elif [ "$LINT_FINAL" = 'success' ]; then
lint_note='Lint failed on the first pass; the AI repair pass fixed it.'
else
lint_note='LINT STILL FAILS after the AI repair pass — the CI lint job will be red and needs a manual fix.'
fi
# Unconditionally, BEFORE `git add -A`: these are untracked and not gitignored, and the
# re-restrict step that would have cleaned them only runs on the repair path. On the
# common clean-lint path they would otherwise be committed — and on a no-op wiring run
# they would be the ONLY staged change, turning "the AI changed nothing" into a bogus
# wiring commit.
rm -f .spec-sync-lint.log .spec-sync-diff.txt
git add -A
# Trusted list of what the AI passes changed (the tool-call summaries above log no values).
git diff --cached --stat
if git diff --cached --quiet; then
echo "AI step produced no changes."
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "summary=Mechanical-only PR (AI wiring added no changes — e.g. workflow-only drift). Human review required before merge." >> "$GITHUB_OUTPUT"
else
# --no-verify belts-and-braces on top of removing .git/hooks above.
git commit --no-verify -m "feat(spec-sync): wire client.v2 to spec diff (AI)"
echo "changed=true" >> "$GITHUB_OUTPUT"
echo "summary=Mechanical snapshot + AI wiring committed (a draft a human finishes). ${lint_note} Gates must pass; human review required before merge." >> "$GITHUB_OUTPUT"
fi
# The only step that sees the token, and it runs exactly one command over an already-built
# commit. The credential was stripped from .git/config before the AI ever ran, so it is
# supplied inline here rather than via the checkout's persisted config. GitHub masks it.
- name: Push AI wiring
id: ai_push
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.ai_commit.outputs.changed == 'true'
env:
GH_PUSH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
# -c overrides any on-disk config, so even a tampered .git/config cannot point a pre-push
# hook at a surviving script in src/ and have it run in the one step that holds the token.
run: git -c core.hooksPath=/dev/null push "https://x-access-token:${GH_PUSH_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" HEAD:"$SYNC_BRANCH"
# Rewrite the PR description: keep the static process/safety preamble from "Open sync PR" and
# APPEND an LLM "## What changed" summary of the actual PR diff. Logic is shared by both jobs
# in scripts/spec-sync/summarize-pr.sh. Best-effort — the static body stands if it degrades;
# the script only edits the PR body via SPEC_SYNC_TOKEN, so an injected spec description in
# the diff can't escalate past text.
- name: Summarize the PR diff into the description
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true'
continue-on-error: true # a summary is nice-to-have; never red the run over it
env:
GH_TOKEN: ${{ secrets.SPEC_SYNC_TOKEN }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
SUMMARY_TITLE_PREFIX: 'spec-sync(v2)' # LLM generates only the title summary after this prefix
# /v2/workflow* is intentionally NOT wired (see the AI-wiring prompt above), so keep it out
# of "What changed" — else a workflow-only drift would advertise SDK endpoints the PR never
# implemented.
SUMMARY_SCOPE_NOTE: 'Scope: this SDK intentionally does NOT implement /v2/workflow, /v2/workflow/jobs, or /v2/workflow/jobs/{job_id}; that surface is deferred and is not wired into the client. Do not describe any /v2/workflow* route as a client/SDK change. If the diff touches only workflow, say the spec snapshot was updated but no client surface changed.'
run: ./scripts/spec-sync/summarize-pr.sh "${{ steps.open_pr.outputs.url }}"
# The AI-wiring outcome, as a reply under the PR-opened thread root (steps.root).
- name: Slack — AI wiring result
if: steps.drift.outputs.code == '10' && steps.existing.outputs.open != 'true' && steps.root.outputs.ts != ''
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: info
title: 'spec-sync ${{ env.SPEC_LABEL }}: AI wiring'
text: ${{ steps.ai_commit.outputs.summary }}
thread_ts: ${{ steps.root.outputs.ts }}
# Catch-all failure alert for this job. An unavailable spec source (exit 20) is skipped earlier
# and never reaches here; what this calls out is a reachable source that errored — an HTTP error
# status / empty / invalid spec, or a script error (exit not 0/10/20) — and an AI-wiring crash
# (PR left mechanical-only), with a generic fallback.
- name: Slack — spec-sync failed
if: failure()
uses: ./.github/actions/slack-notify
with:
bot_token: ${{ secrets.SLACK_BOT_TOKEN }}
webhook: ${{ secrets.SLACK_SPEC_SYNC_WEBHOOK }}
status: failure
title: 'spec-sync ${{ env.SPEC_LABEL }}: run failed'
text: >-
${{ (steps.drift.outputs.code != '' && steps.drift.outputs.code != '0' && steps.drift.outputs.code != '10' && steps.drift.outputs.code != '20')
&& format('Spec fetch/normalize failed (exit {0}) — staging was reachable but returned an error status (401/403/5xx), an empty/invalid spec, or a script errored (an unbooked cluster 404s → exit 20, skipped, not alerted). Investigate.', steps.drift.outputs.code)
|| (steps.ai_wiring.outcome == 'failure'
&& 'AI wiring step failed. Any open sync PR has only the mechanical commit and needs manual wiring.'
|| 'spec-sync run failed. See the workflow run for details.') }}
· <${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}|View run>
thread_ts: ${{ steps.root.outputs.ts }}