From b5ec54fc9ba17415775e60be9a3fd2daa06c530b Mon Sep 17 00:00:00 2001 From: Sipke Schoorstra Date: Sat, 5 Sep 2026 10:14:39 -0700 Subject: [PATCH] Verify NuGet recovery without rebuilding release artifacts (#8024) * release: support validated NuGet artifact recovery * Address recovery validation review findings --- .agents/skills/elsa-release/SKILL.md | 1 + .../elsa-release/references/elsa-profile.json | 1 + .../skills/elsa-release/references/runbook.md | 134 ++++++- .../elsa-release/scripts/release_train.py | 326 +++++++++++++++++- .../elsa-release/tests/test_release_train.py | 219 ++++++++++++ 5 files changed, 678 insertions(+), 3 deletions(-) diff --git a/.agents/skills/elsa-release/SKILL.md b/.agents/skills/elsa-release/SKILL.md index 4cc5117b0..8d579bf44 100644 --- a/.agents/skills/elsa-release/SKILL.md +++ b/.agents/skills/elsa-release/SKILL.md @@ -32,6 +32,7 @@ Use the checkpoint helper in the runbook. It reports the next phase from GitHub - A green upload job is insufficient: verify the expected package inventory, versions, source commits, dependencies, and actual feed content. Verify Studio npm artifacts and `latest`/`next` according to release kind. - Bind exact source commits and reviewed manifests/notes before publication. Preserve existing tags; matching releases are reusable, conflicting releases require investigation. +- If the original run has exactly one configured NuGet publishing failure, use only the runbook's validated `record-recovery` receipt plus the downloaded machine-readable recovery evidence from the reviewed workflow source. Preserve the failed run and tag; reject any other failed job, retag, rebuild, artifact mismatch, forged evidence, or blanket skip, and then perform normal package verification against the recovery run. - Resume by inspecting live GitHub state and recorded package/post evidence. Do not recreate tags or repost after an uncertain result. Retain run and message IDs; wait on concrete jobs with bounded polling and backoff. - Assess advisory findings during preflight. Record severity, relevant usage, and disposition. Existing warnings are not automatically blockers or automatically accepted forever. Never insert an unvalidated dependency upgrade into a release to suppress a warning; a new material unresolved risk requires a concrete scope decision. - After package verification, invoke [Elsa Release Announcements](../elsa-release-announcements/SKILL.md). Default to publishing now on Discord, LinkedIn, and X. Draft-only is an explicit override, not completion of a request to announce. diff --git a/.agents/skills/elsa-release/references/elsa-profile.json b/.agents/skills/elsa-release/references/elsa-profile.json index 7e0e41cd7..b36764379 100644 --- a/.agents/skills/elsa-release/references/elsa-profile.json +++ b/.agents/skills/elsa-release/references/elsa-profile.json @@ -145,6 +145,7 @@ "expected_package_ids": [ "Elsa.Templates" ], + "recovery_artifact": "elsa-template-recovery-evidence", "required_jobs": [ "Build packages", "Publish to feedz.io", diff --git a/.agents/skills/elsa-release/references/runbook.md b/.agents/skills/elsa-release/references/runbook.md index 8847159e3..ef136abc8 100644 --- a/.agents/skills/elsa-release/references/runbook.md +++ b/.agents/skills/elsa-release/references/runbook.md @@ -22,6 +22,23 @@ Check: - The profile intentionally publishes named stable, RC, and preview GitHub releases to NuGet and Feedz. Automated branch previews are separate, Feedz-only builds. Studio named prereleases use npm `next`; stable uses `latest`. If the source workflow disagrees with this policy, resolve the mismatch before release; do not weaken verification to match missing output. - Identify advisories and build/test prerequisites early. Assess actual usage and document disposition; a known warning from a prior release is evidence, not permanent acceptance. Keep unrelated upgrades out of the release. New material unresolved risks or failures require an explicit resolution before irreversible publication. +Before creating a named release, perform a trusted NuGet credential preflight for +each repository whose workflow publishes to NuGet. Check only secret metadata; +never print or copy a secret value: + +```bash +for repository in elsa-workflows/elsa-core elsa-workflows/elsa-studio elsa-workflows/elsa-extensions elsa-workflows/elsa-templates; do + gh secret list --repo "$repository" --json name --jq '.[].name' \ + | grep -Fx NUGET_API_KEY >/dev/null \ + || { echo "Missing NUGET_API_KEY metadata in $repository" >&2; exit 1; } +done +``` + +Run the equivalent check for separately configured publishers. Secret presence +does not prove that its value is valid, but this catches an avoidable missing +credential before an immutable release is created. Never substitute a local key +or put a secret in workflow-dispatch input. + Use a persistent directory outside the repositories, for example `~/.codex/releases/elsa/3.9.0`. It holds source bindings, notes, artifact manifests/downloads, reports, and post receipts. Keep one release owner; the checkpoint uses an OS lock for concurrent updates. Do not run multiple publishing agents for the same version. Initialize the plan (local state only): @@ -115,7 +132,122 @@ python3 /scripts/release.py --repo-path \ Matching existing tags/releases are reused. A different SHA, version/kind, or draft status fails. Do not force-push or recreate a tag. Re-run `status`; retain the returned release run ID. It must be a release-event run at the exact tag and SHA, with all configured publishing jobs successful. -Download every artifact configured by that source workflow into a dedicated `/-artifacts/` directory using `gh run download --repo --name --dir `. Download NuGet and Studio's two npm archives in separate subdirectories of that artifact root, and download Templates' `elsa-template-packages` artifact separately. Verify: +If the original release run has exactly one infrastructure failure in its +configured `Publish to nuget.org` job, preserve that failed run, its immutable +tag, and its artifact. Do not retag, recreate the release, or rerun the package +build. Correct the workflow's explicit recovery path and dispatch it against the +reviewed workflow source SHA. The recovery may run from an infrastructure repair +commit rather than the release tag; record that SHA and verify it against the +live run. It must consume the original configured artifact and run only the +NuGet publication; a successful `Build packages` job is a rebuild, not recovery. +The recovery workflow must upload a machine-readable `recovery-receipt.json` in +the configured `elsa-template-recovery-evidence` artifact. Download the artifact +ZIP and record its live artifact metadata along with the operator receipt. The +operator receipt includes: + +```json +{ + "repository": "elsa-workflows/elsa-templates", + "version": "3.8.0", + "tag": "3.8.0", + "source_commit": "", + "original_release_run": { + "id": 33977531328, + "failed_jobs": ["Publish to nuget.org"] + }, + "artifact": { + "id": 123456, + "name": "elsa-template-packages", + "run_id": 33977531328, + "digest": "sha256:", + "size_in_bytes": 12345 + }, + "recovery_run": { + "id": 33977531329, + "event": "workflow_dispatch", + "publish_job": "Publish to nuget.org", + "artifact_id": 123456, + "artifact_digest": "sha256:", + "workflow_sha": "" + }, + "evidence": { + "id": 123457, + "name": "elsa-template-recovery-evidence", + "run_id": 33977531329, + "digest": "sha256:", + "size_in_bytes": 456 + }, + "target": { + "registry": "nuget.org", + "package_ids": ["Elsa.Templates"], + "version": "3.8.0" + } +} +``` + +The downloaded `recovery-receipt.json` must contain the same version, recovery +run ID and workflow SHA, the original release run ID/source commit, the original +artifact ID, name, run ID, digest and size, and the exact NuGet target/package IDs. This +machine-readable evidence is produced by the reviewed workflow and is checked +against GitHub's live artifact metadata; a self-authored operator receipt is not +evidence by itself. + +Its contents are shaped like this (the workflow writes the live values): + +```json +{ + "schema": 1, + "repository": "elsa-workflows/elsa-templates", + "version": "3.8.0", + "original_source_commit": "", + "recovery_run_id": 33977531329, + "recovery_workflow_sha": "", + "original_release_run_id": 33977531328, + "original_artifact": { + "id": 123456, + "name": "elsa-template-packages", + "run_id": 33977531328, + "digest": "sha256:", + "size_in_bytes": 12345 + }, + "target": { + "registry": "nuget.org", + "package_ids": ["Elsa.Templates"], + "version": "3.8.0" + } +} +``` + +Validate and register it against the checkpoint: + +```bash +python3 /scripts/release_train.py --state /state.json record-recovery \ + --repo templates \ + --receipt /templates-nuget-recovery.json \ + --evidence-archive /recovery-evidence.zip +``` + +The command checks the live tag/source, original release run and jobs, exact +artifact ID/digest/size, recovery workflow SHA and dispatch event, the successful +NuGet job, evidence artifact metadata, archive hash and contents, and the exact +target package/version. It accepts only the sole configured NuGet failure with every +other required job successful; it rejects a different failed job, retag, rebuild, +artifact mismatch, forged/missing evidence, unreviewed workflow source, or target +mismatch. +The checkpoint retains both run IDs and the original failure history, then +requires normal package verification against the recovery run. A failed or +ambiguous recovery remains `repair-pipeline`. + +Download every artifact configured by that source workflow into a dedicated `/-artifacts/` directory using `gh run download --repo --name --dir `. Download NuGet and Studio's two npm archives in separate subdirectories of that artifact root, and download Templates' `elsa-template-packages` artifact separately. For a NuGet recovery, download the recovery artifact bytes directly and preserve the ZIP because its hash is the provenance check: + +```bash +gh api "repos/elsa-workflows/elsa-templates/actions/artifacts//zip" \ + > /recovery-evidence.zip +``` + +Pass that ZIP to `record-recovery` before package verification. The helper checks +the ZIP's SHA-256 against GitHub's artifact digest and parses the single +`recovery-receipt.json` member. Verify: ```bash python3 /scripts/release_train.py --state /state.json verify \ diff --git a/.agents/skills/elsa-release/scripts/release_train.py b/.agents/skills/elsa-release/scripts/release_train.py index e7fdf141f..8a33535a9 100644 --- a/.agents/skills/elsa-release/scripts/release_train.py +++ b/.agents/skills/elsa-release/scripts/release_train.py @@ -13,6 +13,7 @@ import re import subprocess import sys import tempfile +import zipfile from datetime import datetime, timezone from urllib.parse import urlparse @@ -22,6 +23,7 @@ HERE = Path(__file__).resolve().parent DEFAULT_PROFILE = HERE.parent / 'references' / 'elsa-profile.json' SITE_TARGETS = ('website', 'documentation') VALID_SITE_STATUSES = {'completed', 'published', 'verified'} +RECOVERY_REGISTRY = 'nuget.org' def command(args, cwd=None): @@ -349,6 +351,275 @@ def published_release(state, name): raise ValueError(f'{name}: tag does not resolve to a commit') +def positive_int(value, field): + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError(f'Recovery receipt requires positive integer {field}') + return value + + +def recovery_job_names(cfg): + names = [ + name + for name in cfg.get('required_jobs', []) + if isinstance(name, str) and name.lower().rsplit(' ', 1)[-1] == RECOVERY_REGISTRY + ] + if len(names) != 1: + raise ValueError('Recovery requires exactly one configured nuget.org publishing job') + return names[0] + + +def recovery_artifact_name(cfg): + name = cfg.get('recovery_artifact') + if not isinstance(name, str) or not name: + raise ValueError('Recovery requires a configured machine-readable evidence artifact') + return name + + +def commit_sha(value, field): + if not isinstance(value, str) or not re.fullmatch(r'[0-9a-f]{40}', value): + raise ValueError(f'Recovery receipt requires a lowercase 40-character SHA for {field}') + return value + + +def workflow_jobs(cfg, run_id): + pages = gh('api', '--paginate', '--slurp', f"repos/{cfg['github']}/actions/runs/{run_id}/jobs?filter=latest&per_page=100") + return [job for page in pages for job in page.get('jobs', [])] + + +def workflow_artifacts(cfg, run_id): + pages = gh('api', '--paginate', '--slurp', f"repos/{cfg['github']}/actions/runs/{run_id}/artifacts?per_page=100") + return [artifact for page in pages for artifact in page.get('artifacts', [])] + + +def read_recovery_evidence_archive(path, expected_digest=None): + path = Path(path) + if not path.is_file(): + raise ValueError('Recovery evidence archive does not exist') + actual_digest = 'sha256:' + digest(path) + if expected_digest is not None and actual_digest != expected_digest: + raise ValueError('Downloaded recovery evidence archive hash differs from its receipt') + try: + with zipfile.ZipFile(path) as archive: + names = [name for name in archive.namelist() if name.rstrip('/') == 'recovery-receipt.json'] + if len(names) != 1: + raise ValueError('Recovery evidence archive must contain exactly one recovery-receipt.json') + info = archive.getinfo(names[0]) + if info.file_size <= 0 or info.file_size > 1024 * 1024: + raise ValueError('Recovery evidence receipt has an invalid size') + if sum(item.file_size for item in archive.infolist()) > 4 * 1024 * 1024: + raise ValueError('Recovery evidence archive is too large') + evidence = json.loads(archive.read(names[0])) + except (OSError, zipfile.BadZipFile, KeyError, json.JSONDecodeError) as error: + raise ValueError(f'Cannot read recovery evidence archive: {error}') from error + if not isinstance(evidence, dict): + raise ValueError('Recovery evidence receipt must be a JSON object') + return evidence + + +def validate_recovery_evidence(state, name, receipt, evidence, original_id, artifact_id, artifact_digest, artifact_size, recovery_id, recovery_sha): + cfg = config(state, name) + if not isinstance(evidence, dict) or evidence.get('schema') != 1: + raise ValueError('Recovery evidence must be a version 1 JSON object') + if evidence.get('repository') != cfg['github'] or evidence.get('version') != state['version']: + raise ValueError('Recovery evidence repository/version does not match the release') + if evidence.get('recovery_run_id') != recovery_id or evidence.get('recovery_workflow_sha') != recovery_sha: + raise ValueError('Recovery evidence does not identify the live recovery run/workflow source') + if evidence.get('original_release_run_id') != original_id or evidence.get('original_source_commit') != entry(state, name)['binding']['commit']: + raise ValueError('Recovery evidence is bound to a different original release run') + original_artifact = evidence.get('original_artifact') + if not isinstance(original_artifact, dict): + raise ValueError('Recovery evidence requires original_artifact metadata') + if ( + original_artifact.get('id') != artifact_id + or original_artifact.get('name') != cfg['artifact'] + or original_artifact.get('run_id') != original_id + or original_artifact.get('digest') != artifact_digest + or original_artifact.get('size_in_bytes') != artifact_size + ): + raise ValueError('Recovery evidence does not bind the original artifact payload') + target = evidence.get('target') + if not isinstance(target, dict) or target.get('registry') != RECOVERY_REGISTRY or target.get('version') != state['version']: + raise ValueError('Recovery evidence target must be the exact NuGet release/version') + package_ids = target.get('package_ids') + if not isinstance(package_ids, list) or any(not isinstance(package_id, str) for package_id in package_ids): + raise ValueError('Recovery evidence target requires package_ids') + expected_ids = config(state, name).get('expected_package_ids') + if expected_ids is None: + expected_ids = [package['id'] for package in read(entry(state, name)['binding']['manifest']).get('nuget', [])] + if sorted(package_ids, key=str.lower) != sorted(expected_ids, key=str.lower): + raise ValueError('Recovery evidence package inventory does not match the release profile') + + +def validate_recovery_receipt(state, name, receipt, expected_original_run_id=None, evidence=None): + """Validate a no-rebuild NuGet recovery against the immutable failed release run.""" + + item = entry(state, name) + cfg = config(state, name) + if not cfg.get('artifact'): + raise ValueError(f'{name}: recovery requires a configured source workflow artifact') + binding = item.get('binding') + if not isinstance(binding, dict) or not binding.get('commit'): + raise ValueError('Recovery requires a frozen source binding') + if not isinstance(receipt, dict): + raise ValueError('Recovery receipt must be a JSON object') + if receipt.get('repository') != cfg['github']: + raise ValueError('Recovery receipt repository does not match the release profile') + if receipt.get('version') != state['version'] or receipt.get('tag') != state['version']: + raise ValueError('Recovery receipt version/tag does not match the release') + if receipt.get('source_commit') != binding['commit']: + raise ValueError('Recovery receipt source commit does not match the frozen binding') + + target = receipt.get('target') + if not isinstance(target, dict) or target.get('registry') != RECOVERY_REGISTRY or target.get('version') != state['version']: + raise ValueError('Recovery receipt target must be the exact NuGet release/version') + raw_package_ids = target.get('package_ids') if isinstance(target, dict) else None + if not isinstance(raw_package_ids, list) or any(not isinstance(package_id, str) for package_id in raw_package_ids): + raise ValueError('Recovery receipt target requires package_ids') + package_ids = sorted(raw_package_ids, key=str.lower) + expected_ids = cfg.get('expected_package_ids') + if expected_ids is None: + expected_ids = [package['id'] for package in read(binding['manifest']).get('nuget', [])] + if package_ids != sorted(expected_ids, key=str.lower): + raise ValueError('Recovery receipt package inventory does not match the release profile') + + original = receipt.get('original_release_run') + artifact = receipt.get('artifact') + recovery = receipt.get('recovery_run') + if not isinstance(original, dict) or not isinstance(artifact, dict) or not isinstance(recovery, dict): + raise ValueError('Recovery receipt requires original_release_run, artifact, and recovery_run objects') + + original_id = positive_int(original.get('id'), 'original_release_run.id') + if expected_original_run_id is not None and original_id != expected_original_run_id: + raise ValueError('Recovery receipt is bound to a different original release run') + recovery_id = positive_int(recovery.get('id'), 'recovery_run.id') + if original_id == recovery_id: + raise ValueError('Recovery run must be distinct from the original release run') + artifact_id = positive_int(artifact.get('id'), 'artifact.id') + if artifact.get('name') != cfg['artifact'] or artifact.get('run_id') != original_id: + raise ValueError('Recovery artifact is not the configured artifact from the original run') + if not isinstance(artifact.get('digest'), str) or not artifact['digest'].startswith('sha256:'): + raise ValueError('Recovery artifact requires its immutable SHA-256 digest') + positive_int(artifact.get('size_in_bytes'), 'artifact.size_in_bytes') + if recovery.get('event') != 'workflow_dispatch': + raise ValueError('Recovery run must be a workflow_dispatch run') + if recovery.get('publish_job') != recovery_job_names(cfg): + raise ValueError('Recovery run must identify the configured NuGet publishing job') + recovery_sha = commit_sha(recovery.get('workflow_sha'), 'recovery_run.workflow_sha') + if recovery.get('artifact_id') != artifact_id or recovery.get('artifact_digest') != artifact['digest']: + raise ValueError('Recovery run is not bound to the original artifact payload') + evidence_metadata = receipt.get('evidence') + if not isinstance(evidence_metadata, dict): + raise ValueError('Recovery receipt requires evidence artifact metadata') + evidence_id = positive_int(evidence_metadata.get('id'), 'evidence.id') + if evidence_metadata.get('name') != recovery_artifact_name(cfg) or evidence_metadata.get('run_id') != recovery_id: + raise ValueError('Recovery evidence artifact is not from the recovery run') + if not isinstance(evidence_metadata.get('digest'), str) or not evidence_metadata['digest'].startswith('sha256:'): + raise ValueError('Recovery evidence artifact requires its immutable SHA-256 digest') + evidence_size = positive_int(evidence_metadata.get('size_in_bytes'), 'evidence.size_in_bytes') + if evidence is None: + raise ValueError('Recovery requires the downloaded machine-readable workflow evidence') + + release = published_release(state, name) + if release is None or release['commit'] != binding['commit']: + raise ValueError('Recovery requires the existing immutable release tag at the bound source commit') + + original_remote = gh('api', f"repos/{cfg['github']}/actions/runs/{original_id}") + if original_remote.get('id') != original_id or original_remote.get('event') != 'release': + raise ValueError('Original recovery run is not the release-event run') + if original_remote.get('status') != 'completed' or original_remote.get('conclusion') != 'failure': + raise ValueError('Original recovery run must preserve its completed failed history') + if original_remote.get('head_sha') != binding['commit'] or original_remote.get('head_branch') != state['version']: + raise ValueError('Original recovery run source does not match the immutable release tag') + expected_workflow = f".github/workflows/{cfg['workflow']}" + if original_remote.get('path') and original_remote['path'].lstrip('/') != expected_workflow: + raise ValueError('Original recovery run used a different workflow') + + target_job = recovery_job_names(cfg) + original_jobs = {job.get('name'): job for job in workflow_jobs(cfg, original_id)} + required_jobs = set(cfg.get('required_jobs', [])) + if not required_jobs <= original_jobs.keys(): + raise ValueError('Original recovery run is missing configured required jobs') + failed_jobs = original.get('failed_jobs') + if failed_jobs != [target_job] or original_jobs[target_job].get('conclusion') != 'failure': + raise ValueError('Recovery is allowed only for the failed NuGet publishing job') + if any(original_jobs[job].get('conclusion') != 'success' for job in required_jobs if job != target_job): + raise ValueError('Recovery cannot bypass a failed or skipped build/feed job') + + matching_artifacts = [artifact for artifact in workflow_artifacts(cfg, original_id) if artifact.get('id') == artifact_id] + if len(matching_artifacts) != 1: + raise ValueError('Recovery receipt must identify exactly one original workflow artifact') + remote_artifact = matching_artifacts[0] + if remote_artifact.get('name') != cfg['artifact'] or remote_artifact.get('expired') is True: + raise ValueError('Original recovery artifact is missing, expired, or has the wrong name') + if remote_artifact.get('workflow_run', {}).get('id') != original_id: + raise ValueError('Recovery artifact does not belong to the original release run') + if remote_artifact.get('digest') != artifact['digest'] or remote_artifact.get('size_in_bytes') != artifact['size_in_bytes']: + raise ValueError('Recovery receipt artifact payload differs from GitHub evidence') + + recovery_remote = gh('api', f"repos/{cfg['github']}/actions/runs/{recovery_id}") + if recovery_remote.get('id') != recovery_id or recovery_remote.get('event') != 'workflow_dispatch': + raise ValueError('Recovery run is not a workflow_dispatch run') + if recovery_remote.get('status') != 'completed' or recovery_remote.get('conclusion') != 'success': + raise ValueError('Recovery run did not complete successfully') + if recovery_remote.get('head_sha') != recovery_sha: + raise ValueError('Recovery run source commit does not match the reviewed recovery workflow SHA') + if recovery_remote.get('path') and recovery_remote['path'].lstrip('/') != expected_workflow: + raise ValueError('Recovery run used a different workflow') + recovery_jobs = {job.get('name'): job for job in workflow_jobs(cfg, recovery_id)} + if recovery_jobs.get(target_job, {}).get('conclusion') != 'success': + raise ValueError('Recovery NuGet publishing job did not succeed') + if recovery_jobs.get('Build packages', {}).get('conclusion') == 'success': + raise ValueError('Recovery run rebuilt packages instead of publishing the original artifact') + matching_evidence = [artifact for artifact in workflow_artifacts(cfg, recovery_id) if artifact.get('id') == evidence_id] + if len(matching_evidence) != 1: + raise ValueError('Recovery receipt must identify exactly one workflow evidence artifact') + remote_evidence = matching_evidence[0] + if remote_evidence.get('name') != evidence_metadata['name'] or remote_evidence.get('expired') is True: + raise ValueError('Recovery evidence artifact is missing, expired, or has the wrong name') + if remote_evidence.get('workflow_run', {}).get('id') != recovery_id: + raise ValueError('Recovery evidence artifact does not belong to the recovery run') + if remote_evidence.get('digest') != evidence_metadata['digest'] or remote_evidence.get('size_in_bytes') != evidence_size: + raise ValueError('Recovery evidence artifact payload differs from GitHub evidence') + validate_recovery_evidence( + state, + name, + receipt, + evidence, + original_id, + artifact_id, + artifact['digest'], + artifact['size_in_bytes'], + recovery_id, + recovery_sha, + ) + return receipt + + +def recovery_receipt_valid(state, name, record, expected_original_run_id=None): + evidence_record = record.get('evidence') if isinstance(record, dict) else None + if ( + not isinstance(record, dict) + or not isinstance(record.get('receipt'), str) + or not isinstance(record.get('sha256'), str) + or not isinstance(evidence_record, dict) + or not isinstance(evidence_record.get('archive'), str) + or not isinstance(evidence_record.get('sha256'), str) + ): + return False + try: + if not Path(record['receipt']).is_file() or digest(record['receipt']) != record['sha256']: + return False + if not Path(evidence_record['archive']).is_file() or digest(evidence_record['archive']) != evidence_record['sha256']: + return False + receipt = read(record['receipt']) + evidence_digest = receipt.get('evidence', {}).get('digest') + evidence = read_recovery_evidence_archive(evidence_record['archive'], evidence_digest) + validate_recovery_receipt(state, name, receipt, expected_original_run_id, evidence) + except (OSError, ValueError, KeyError, TypeError): + return False + return True + + def inspect_release(state, name): item = entry(state, name) cfg = config(state, name) @@ -371,6 +642,16 @@ def inspect_release(state, name): if run['status'] != 'completed': return result if run['conclusion'] != 'success': + recovery = item.get('recovery') + if recovery_receipt_valid(state, name, recovery, run['id']): + receipt = read(recovery['receipt']) + result.update( + original_run_id=run['id'], + recovery_run_id=receipt['recovery_run']['id'], + recovery_receipt=recovery['receipt'], + ) + result['phase'] = 'verified' if receipt_valid(item) else 'verify-packages' + return result return dict(result, phase='repair-pipeline', conclusion=run['conclusion']) jobs = gh('api', '--paginate', '--slurp', f"repos/{cfg['github']}/actions/runs/{run['id']}/jobs?filter=latest&per_page=100") complete = {j['name'] for page in jobs for j in page['jobs'] if j['conclusion'] == 'success'} @@ -390,7 +671,11 @@ def receipt_valid(item): if any(digest(binding[k]) != binding[k + '_sha256'] for k in ('manifest', 'notes')): return False report = read(receipt['report']) - return digest(receipt['report']) == receipt['report_sha256'] and report.get('verified') is True and report.get('source_commit') == binding['commit'] and report.get('version') == read(binding['manifest'])['version'] + valid = digest(receipt['report']) == receipt['report_sha256'] and report.get('verified') is True and report.get('source_commit') == binding['commit'] and report.get('version') == read(binding['manifest'])['version'] + recovery = item.get('recovery') + if recovery: + valid = valid and receipt.get('run_id') == recovery.get('recovery_run_id') + return valid except (OSError, ValueError, KeyError): return False @@ -592,7 +877,13 @@ def verify(state, args): data = read(report) if not data.get('verified') or data.get('source_commit') != binding['commit'] or data.get('version') != state['version']: raise ValueError('Verifier returned inconsistent evidence') - item['verification'] = {'commit': binding['commit'], 'report': str(report), 'report_sha256': digest(report), 'verified_at': datetime.now(timezone.utc).isoformat(), 'run_id': observed['run_id']} + item['verification'] = { + 'commit': binding['commit'], + 'report': str(report), + 'report_sha256': digest(report), + 'verified_at': datetime.now(timezone.utc).isoformat(), + 'run_id': observed.get('recovery_run_id', observed['run_id']), + } return {'phase': 'verified', 'report': str(report)} @@ -681,6 +972,33 @@ def record_site(state, args): return value +def record_recovery(state, args): + item = entry(state, args.repo) + if not item['publish']: + raise ValueError('Cannot record recovery for a verification-only repository') + require_upstreams(state, args.repo) + evidence_archive = getattr(args, 'evidence_archive', None) + if evidence_archive is None: + raise ValueError('Recovery requires the downloaded machine-readable workflow evidence archive') + evidence_archive = evidence_archive.resolve() + receipt = read(args.receipt) + expected_digest = receipt.get('evidence', {}).get('digest') + evidence = read_recovery_evidence_archive(evidence_archive, expected_digest) + validate_recovery_receipt(state, args.repo, receipt, evidence=evidence) + value = { + 'receipt': str(args.receipt.resolve()), + 'sha256': digest(args.receipt), + 'original_run_id': receipt['original_release_run']['id'], + 'recovery_run_id': receipt['recovery_run']['id'], + 'evidence': {'archive': str(evidence_archive), 'sha256': digest(evidence_archive)}, + } + prior = item.get('recovery') + if prior and prior != value: + raise ValueError('A different recovery receipt is already recorded; preserve the original failure history') + item['recovery'] = value + return value + + def record_announcement(state, args): if status(state)['next'] not in ('announcements', 'complete'): raise ValueError('All selected repositories, post-release sites, and upstream packages must be verified before announcements') @@ -745,6 +1063,10 @@ def main(): p.add_argument('--target', required=True, choices=SITE_TARGETS) p.add_argument('--receipt', required=True, type=Path) p.add_argument('--replace', '--replace-site-receipt', dest='replace', action='store_true', help='Replace a stale or tampered site receipt after reviewing new live evidence') + p = sub.add_parser('record-recovery') + p.add_argument('--repo', required=True) + p.add_argument('--receipt', required=True, type=Path) + p.add_argument('--evidence-archive', required=True, type=Path, help='Downloaded ZIP for the recovery evidence artifact') args = parser.parse_args() args.state = args.state.expanduser().resolve() try: diff --git a/.agents/skills/elsa-release/tests/test_release_train.py b/.agents/skills/elsa-release/tests/test_release_train.py index 629c2394c..01b2bac15 100644 --- a/.agents/skills/elsa-release/tests/test_release_train.py +++ b/.agents/skills/elsa-release/tests/test_release_train.py @@ -1,10 +1,13 @@ import copy +import hashlib +import json from pathlib import Path import sys import tempfile from types import SimpleNamespace import unittest from unittest.mock import patch +import zipfile sys.path.insert(0,str(Path(__file__).resolve().parents[1]/'scripts')) import release_train as train @@ -80,6 +83,222 @@ class TrainTests(unittest.TestCase): return [{'jobs':[{'name':n,'conclusion':'success'} for n in cfg['required_jobs']]}] self.fail(f'Unexpected GitHub call {args}') + def prepare_template_recovery(self): + self.bind_fixture('templates') + self.state['repositories'] = {'templates': self.state['repositories']['templates']} + self.state['repositories']['templates'].pop('verification', None) + + def template_recovery_receipt(self): + artifact_digest = 'sha256:' + 'd' * 64 + evidence_digest = 'sha256:' + 'f' * 64 + return { + 'repository': 'elsa-workflows/elsa-templates', + 'version': '3.9.0', + 'tag': '3.9.0', + 'source_commit': 'a' * 40, + 'original_release_run': { + 'id': 33977531328, + 'failed_jobs': ['Publish to nuget.org'], + }, + 'artifact': { + 'id': 777, + 'name': 'elsa-template-packages', + 'run_id': 33977531328, + 'digest': artifact_digest, + 'size_in_bytes': 1234, + }, + 'recovery_run': { + 'id': 33977531329, + 'event': 'workflow_dispatch', + 'publish_job': 'Publish to nuget.org', + 'artifact_id': 777, + 'artifact_digest': artifact_digest, + 'workflow_sha': 'e' * 40, + }, + 'evidence': { + 'id': 778, + 'name': 'elsa-template-recovery-evidence', + 'run_id': 33977531329, + 'digest': evidence_digest, + 'size_in_bytes': 567, + }, + 'target': { + 'registry': 'nuget.org', + 'package_ids': ['Elsa.Templates'], + 'version': '3.9.0', + }, + } + + def template_recovery_evidence(self): + return { + 'schema': 1, + 'repository': 'elsa-workflows/elsa-templates', + 'version': '3.9.0', + 'recovery_run_id': 33977531329, + 'recovery_workflow_sha': 'e' * 40, + 'original_release_run_id': 33977531328, + 'original_source_commit': 'a' * 40, + 'original_artifact': { + 'id': 777, + 'name': 'elsa-template-packages', + 'run_id': 33977531328, + 'digest': 'sha256:' + 'd' * 64, + 'size_in_bytes': 1234, + }, + 'target': { + 'registry': 'nuget.org', + 'package_ids': ['Elsa.Templates'], + 'version': '3.9.0', + }, + } + + def template_recovery_files(self): + evidence = self.template_recovery_evidence() + archive = self.root / 'recovery-evidence.zip' + payload = json.dumps(evidence, separators=(',', ':'), sort_keys=True).encode() + with zipfile.ZipFile(archive, 'w', compression=zipfile.ZIP_DEFLATED) as output: + output.writestr('recovery-receipt.json', payload) + archive_digest = 'sha256:' + hashlib.sha256(archive.read_bytes()).hexdigest() + self.recovery_evidence_digest = archive_digest + self.recovery_evidence_size = archive.stat().st_size + receipt = self.template_recovery_receipt() + receipt['evidence']['digest'] = archive_digest + receipt['evidence']['size_in_bytes'] = self.recovery_evidence_size + return receipt, archive + + def recovery_github(self, *args): + url = args[-1] + if '/releases?' in url: + return [[{'tag_name': '3.9.0', 'draft': False, 'prerelease': False, 'html_url': 'https://github.com/release'}]] + if '/git/ref/' in url: + return {'object': {'type': 'tag', 'sha': 'tag-object'}} + if '/git/tags/' in url: + return {'object': {'type': 'commit', 'sha': 'a' * 40}} + if '/workflows/' in url: + return [{'workflow_runs': [{ + 'id': 33977531328, + 'head_sha': 'a' * 40, + 'head_branch': '3.9.0', + 'event': 'release', + 'run_number': 42, + 'run_attempt': 1, + 'status': 'completed', + 'conclusion': 'failure', + 'html_url': 'https://github.com/original-run', + }]}] + if url.endswith('/actions/runs/33977531328'): + return { + 'id': 33977531328, + 'event': 'release', + 'status': 'completed', + 'conclusion': 'failure', + 'head_sha': 'a' * 40, + 'head_branch': '3.9.0', + } + if url.endswith('/actions/runs/33977531329'): + return { + 'id': 33977531329, + 'event': 'workflow_dispatch', + 'status': 'completed', + 'conclusion': 'success', + 'head_sha': 'e' * 40, + } + if '/actions/runs/33977531328/jobs?' in url: + return [{'jobs': [ + {'name': 'Build packages', 'conclusion': 'success'}, + {'name': 'Publish to feedz.io', 'conclusion': 'success'}, + {'name': 'Publish to nuget.org', 'conclusion': 'failure'}, + ]}] + if '/actions/runs/33977531329/jobs?' in url: + return [{'jobs': [ + {'name': 'Build packages', 'conclusion': 'skipped'}, + {'name': 'Publish to feedz.io', 'conclusion': 'skipped'}, + {'name': 'Publish to nuget.org', 'conclusion': 'success'}, + ]}] + if '/actions/runs/33977531328/artifacts?' in url: + return [{'artifacts': [{ + 'id': 777, + 'name': 'elsa-template-packages', + 'size_in_bytes': 1234, + 'digest': 'sha256:' + 'd' * 64, + 'expired': False, + 'workflow_run': {'id': 33977531328}, + }]}] + if '/actions/runs/33977531329/artifacts?' in url: + return [{'artifacts': [{ + 'id': 778, + 'name': 'elsa-template-recovery-evidence', + 'size_in_bytes': getattr(self, 'recovery_evidence_size', 567), + 'digest': getattr(self, 'recovery_evidence_digest', 'sha256:' + 'f' * 64), + 'expired': False, + 'workflow_run': {'id': 33977531329}, + }]}] + raise AssertionError(f'Unhandled GitHub URL in recovery_github: {url}') + + def test_nuget_recovery_binds_failed_release_and_original_artifact(self): + self.prepare_template_recovery() + receipt = self.root / 'recovery.json' + receipt_value, evidence_archive = self.template_recovery_files() + train.save(receipt, receipt_value) + with patch.object(train, 'gh', side_effect=self.recovery_github): + value = train.record_recovery(self.state, SimpleNamespace(repo='templates', receipt=receipt, evidence_archive=evidence_archive)) + observed = train.inspect_release(self.state, 'templates') + self.assertEqual(33977531328, value['original_run_id']) + self.assertEqual(33977531329, value['recovery_run_id']) + self.assertEqual('verify-packages', observed['phase']) + self.assertEqual(33977531329, observed['recovery_run_id']) + + def test_nuget_recovery_rejects_non_nuget_original_failure(self): + self.prepare_template_recovery() + receipt = self.template_recovery_receipt() + receipt['original_release_run']['failed_jobs'] = ['Build packages'] + evidence = self.template_recovery_evidence() + with patch.object(train, 'gh', side_effect=self.recovery_github): + with self.assertRaisesRegex(ValueError, 'failed NuGet publishing job'): + train.validate_recovery_receipt(self.state, 'templates', receipt, evidence=evidence) + + def test_nuget_recovery_rejects_rebuild_in_recovery_run(self): + self.prepare_template_recovery() + receipt = self.template_recovery_receipt() + evidence = self.template_recovery_evidence() + + def rebuilt(*args): + value = self.recovery_github(*args) + if '/actions/runs/33977531329/jobs?' in args[-1]: + value[0]['jobs'][0]['conclusion'] = 'success' + return value + + with patch.object(train, 'gh', side_effect=rebuilt): + with self.assertRaisesRegex(ValueError, 'rebuilt packages'): + train.validate_recovery_receipt(self.state, 'templates', receipt, evidence=evidence) + + def test_nuget_recovery_rejects_unreviewed_workflow_sha(self): + self.prepare_template_recovery() + receipt = self.template_recovery_receipt() + receipt['recovery_run']['workflow_sha'] = 'a' * 40 + with patch.object(train, 'gh', side_effect=self.recovery_github): + with self.assertRaisesRegex(ValueError, 'reviewed recovery workflow SHA'): + train.validate_recovery_receipt(self.state, 'templates', receipt, evidence=self.template_recovery_evidence()) + + def test_nuget_recovery_rejects_forged_machine_evidence_linkage(self): + self.prepare_template_recovery() + receipt = self.template_recovery_receipt() + evidence = self.template_recovery_evidence() + evidence['original_artifact']['digest'] = 'sha256:' + '0' * 64 + with patch.object(train, 'gh', side_effect=self.recovery_github): + with self.assertRaisesRegex(ValueError, 'does not bind the original artifact payload'): + train.validate_recovery_receipt(self.state, 'templates', receipt, evidence=evidence) + + def test_nuget_recovery_rejects_tampered_evidence_archive(self): + self.prepare_template_recovery() + receipt_value, evidence_archive = self.template_recovery_files() + receipt = self.root / 'recovery.json' + train.save(receipt, receipt_value) + evidence_archive.write_bytes(evidence_archive.read_bytes() + b'tampered') + with patch.object(train, 'gh', side_effect=self.recovery_github): + with self.assertRaisesRegex(ValueError, 'archive hash differs'): + train.record_recovery(self.state, SimpleNamespace(repo='templates', receipt=receipt, evidence_archive=evidence_archive)) + def test_unnumbered_rc_and_preview_need_a_resolved_version(self): for value in ['3.9.0-rc','3.9.0-preview']: with self.assertRaisesRegex(ValueError,'explicit unused'):