diff --git a/.github/aw/actions-lock.json b/.github/aw/actions-lock.json index d3810c8d7bac..dceefb04c0f8 100644 --- a/.github/aw/actions-lock.json +++ b/.github/aw/actions-lock.json @@ -19,6 +19,11 @@ "repo": "github/gh-aw-actions/setup", "version": "v0.80.9", "sha": "8c7d04ebf1ece56cd381446125da3e0f6896294a" + }, + "github/gh-aw-actions/setup@v0.87.1": { + "repo": "github/gh-aw-actions/setup", + "version": "v0.87.1", + "sha": "423b3dc04bbf1b1797194a4a75aa5cf5d0d4f5b3" } } } diff --git a/.github/workflows/mgmt-sdk-pr-review.lock.yml b/.github/workflows/mgmt-sdk-pr-review.lock.yml index ba8508522424..5953ca3aaa58 100644 --- a/.github/workflows/mgmt-sdk-pr-review.lock.yml +++ b/.github/workflows/mgmt-sdk-pr-review.lock.yml @@ -1,5 +1,5 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"aba9fa1bf476fa7c5496bf00373379973fb782e686f1be2d3a2dec9a6b0e4e81","body_hash":"363e4430ab13289db056ee0b74e2eeacd4185e1fd3ee6215df9a47a8b917ef83","compiler_version":"v0.87.1","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} -# gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"423b3dc04bbf1b1797194a4a75aa5cf5d0d4f5b3","version":"423b3dc04bbf1b1797194a4a75aa5cf5d0d4f5b3"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1","digest":"sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1@sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1","digest":"sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1@sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1","digest":"sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1@sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}],"has_pull_request_target":true} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"78c0c1579b4699c55588fe5cd13705d8f841fdaba9bd0dd447d483e127efbda3","body_hash":"afe0b46190335d915134036fbebc2df5e17e778f2d5d5497cb6e4f74974e41d4","compiler_version":"v0.87.1","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} +# gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"423b3dc04bbf1b1797194a4a75aa5cf5d0d4f5b3","version":"v0.87.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1","digest":"sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1@sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1","digest":"sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1@sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1","digest":"sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1@sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}],"has_pull_request_target":true} # This file was automatically generated by gh-aw (v0.87.1). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # # ___ _ _ @@ -39,7 +39,7 @@ # - actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 (source v9) # - actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 # - actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 -# - github/gh-aw-actions/setup@423b3dc04bbf1b1797194a4a75aa5cf5d0d4f5b3 +# - github/gh-aw-actions/setup@423b3dc04bbf1b1797194a4a75aa5cf5d0d4f5b3 # v0.87.1 # # Container images used: # - ghcr.io/github/gh-aw-firewall/agent:0.28.1@sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d @@ -428,8 +428,9 @@ jobs: GH_REPOSITORY: ${{ github.repository }} GH_TOKEN: ${{ github.token }} PR_NUMBER: ${{ github.event.pull_request.number }} + TRUSTED_BASE_SHA: ${{ github.event.pull_request.base.sha }} name: Collect management SDK review context - run: "python - <<'PY'\nimport base64\nimport binascii\nimport json\nimport os\nimport re\nimport urllib.error\nimport urllib.parse\nimport urllib.request\n\n\nAPI_ROOT = os.environ.get(\"GH_API_ROOT\", \"https://api.github.com\")\nREPOSITORY = os.environ[\"GH_REPOSITORY\"]\nPR_NUMBER = int(os.environ[\"PR_NUMBER\"])\nTOKEN = os.environ[\"GH_TOKEN\"]\nPACKAGE_PATTERN = re.compile(r\"^(sdk/[^/]+/azure-mgmt-[^/]+)(?:/|$)\")\n\n\nclass GitHubApiError(RuntimeError):\n pass\n\n\ndef api_get(path):\n request = urllib.request.Request(\n f\"{API_ROOT}{path}\",\n headers={\n \"Accept\": \"application/vnd.github+json\",\n \"Authorization\": f\"Bearer {TOKEN}\",\n \"User-Agent\": \"azure-sdk-python-mgmt-review\",\n \"X-GitHub-Api-Version\": \"2022-11-28\",\n },\n )\n try:\n with urllib.request.urlopen(request) as response:\n return json.load(response)\n except urllib.error.HTTPError as error:\n detail = error.read().decode(\"utf-8\", errors=\"replace\")\n raise GitHubApiError(f\"GitHub API request failed ({error.code}) for {path}: {detail}\") from error\n except urllib.error.URLError as error:\n raise GitHubApiError(f\"GitHub API request failed for {path}: {error.reason}\") from error\n\n\ndef paged_get(path, max_items=None):\n items = []\n page = 1\n while True:\n separator = \"&\" if \"?\" in path else \"?\"\n batch = api_get(f\"{path}{separator}per_page=100&page={page}\")\n if not isinstance(batch, list):\n raise GitHubApiError(f\"GitHub API returned a non-list response for {path}\")\n items.extend(batch)\n if max_items is not None and len(items) >= max_items:\n return items[:max_items]\n if len(batch) < 100:\n return items\n page += 1\n\n\ndef read_repository_file(path, revision):\n encoded_path = urllib.parse.quote(path, safe=\"/\")\n encoded_ref = urllib.parse.quote(revision, safe=\"\")\n payload = api_get(\n f\"/repos/{REPOSITORY}/contents/{encoded_path}?ref={encoded_ref}\"\n )\n try:\n if payload.get(\"encoding\") != \"base64\":\n raise ValueError(\"content was not base64 encoded\")\n return base64.b64decode(payload[\"content\"]).decode(\"utf-8\")\n except (binascii.Error, KeyError, TypeError, ValueError, UnicodeDecodeError) as error:\n raise GitHubApiError(f\"Could not read {path} at {revision}: {error}\") from error\n\n\ndef extract_management_review_rules(instructions):\n lines = instructions.splitlines()\n heading = \"## MGMT SDK Code Review Rules\"\n try:\n start = lines.index(heading)\n except ValueError as error:\n raise GitHubApiError(f\"{heading} was not found in .github/copilot-instructions.md\") from error\n end = next(\n (index for index in range(start + 1, len(lines)) if lines[index].startswith(\"## \")),\n len(lines),\n )\n return \"\\n\".join(lines[start:end]).strip()\n\n\ndef read_api_version(package_path, revision):\n metadata_path = f\"{package_path}/_metadata.json\"\n encoded_path = urllib.parse.quote(metadata_path, safe=\"/\")\n encoded_ref = urllib.parse.quote(revision, safe=\"\")\n try:\n payload = api_get(\n f\"/repos/{REPOSITORY}/contents/{encoded_path}?ref={encoded_ref}\"\n )\n except GitHubApiError as error:\n return None, str(error)\n\n try:\n if payload.get(\"encoding\") != \"base64\":\n raise ValueError(\"content was not base64 encoded\")\n content = base64.b64decode(payload[\"content\"]).decode(\"utf-8\")\n api_version = json.loads(content)[\"apiVersion\"]\n if not isinstance(api_version, str) or not api_version:\n raise ValueError(\"apiVersion was missing or was not a non-empty string\")\n return api_version, None\n except (binascii.Error, KeyError, TypeError, ValueError, UnicodeDecodeError) as error:\n return None, f\"Could not read apiVersion from {metadata_path} at {revision}: {error}\"\n\n\nrepository = api_get(f\"/repos/{REPOSITORY}\")\ndefault_branch = repository.get(\"default_branch\")\nif not isinstance(default_branch, str) or not default_branch:\n raise GitHubApiError(\"Repository metadata did not contain a default branch\")\ninstructions = read_repository_file(\".github/copilot-instructions.md\", default_branch)\nmanagement_review_rules = extract_management_review_rules(instructions)\n\npull_request = api_get(f\"/repos/{REPOSITORY}/pulls/{PR_NUMBER}\")\nexpected_changed_files = pull_request.get(\"changed_files\")\nif not isinstance(expected_changed_files, int) or expected_changed_files < 0:\n raise GitHubApiError(\"Pull request metadata did not contain a valid changed_files count\")\nlatest_revision = pull_request.get(\"head\", {}).get(\"sha\")\nif not isinstance(latest_revision, str) or not latest_revision:\n raise GitHubApiError(\"Pull request metadata did not contain a valid head SHA\")\n\nchanged_files = paged_get(\n f\"/repos/{REPOSITORY}/pulls/{PR_NUMBER}/files\",\n max_items=3000,\n)\nreturned_changed_files = len(changed_files)\npackage_discovery_complete = returned_changed_files == expected_changed_files\npackage_discovery_error = None\nif not package_discovery_complete:\n package_discovery_error = (\n \"Management package discovery is incomplete: pull request metadata reports \"\n f\"{expected_changed_files} changed files, but the GitHub API returned \"\n f\"{returned_changed_files}. GitHub limits pull request file responses to 3,000 files.\"\n )\n\npackage_paths = sorted(\n {\n match.group(1)\n for item in changed_files\n for field in (\"filename\", \"previous_filename\")\n for path in [item.get(field)]\n if isinstance(path, str)\n for match in [PACKAGE_PATTERN.match(path)]\n if match\n }\n)\n\ncommits = paged_get(\n f\"/repos/{REPOSITORY}/pulls/{PR_NUMBER}/commits\",\n max_items=250,\n)\ncommit_shas = [item.get(\"sha\") for item in commits if isinstance(item.get(\"sha\"), str)]\nif not commit_shas:\n raise GitHubApiError(\"Pull request metadata returned an empty commit list\")\n\nfirst_revision = commit_shas[0]\ndrift_results = []\nfor package_path in package_paths:\n first_api_version, first_error = read_api_version(package_path, first_revision)\n latest_api_version, latest_error = read_api_version(package_path, latest_revision)\n errors = [error for error in (first_error, latest_error) if error]\n if errors:\n status = \"unverified\"\n elif first_api_version == latest_api_version:\n status = \"unchanged\"\n else:\n status = \"changed\"\n drift_results.append(\n {\n \"packagePath\": package_path,\n \"metadataPath\": f\"{package_path}/_metadata.json\",\n \"status\": status,\n \"firstRevision\": first_revision,\n \"firstApiVersion\": first_api_version,\n \"latestRevision\": latest_revision,\n \"latestApiVersion\": latest_api_version,\n \"error\": \"; \".join(errors) if errors else None,\n }\n )\n\ncontext = {\n \"repository\": REPOSITORY,\n \"pullRequestNumber\": PR_NUMBER,\n \"rulesSource\": f\".github/copilot-instructions.md@{default_branch}\",\n \"mgmtSdkCodeReviewRules\": management_review_rules,\n \"packageDiscovery\": {\n \"status\": \"complete\" if package_discovery_complete else \"unverified\",\n \"expectedChangedFiles\": expected_changed_files,\n \"returnedChangedFiles\": returned_changed_files,\n \"error\": package_discovery_error,\n },\n \"affectedPackages\": package_paths,\n \"changedFiles\": [\n {\n \"filename\": item.get(\"filename\"),\n \"previousFilename\": item.get(\"previous_filename\"),\n \"status\": item.get(\"status\"),\n \"additions\": item.get(\"additions\"),\n \"deletions\": item.get(\"deletions\"),\n }\n for item in changed_files\n ],\n \"firstRevision\": first_revision,\n \"latestRevision\": latest_revision,\n \"apiVersionDrift\": drift_results,\n}\nwith open(\"review-context.json\", \"w\", encoding=\"utf-8\") as output:\n json.dump(context, output, indent=2)\n output.write(\"\\n\")\nPY\n" + run: "python - <<'PY'\nimport base64\nimport json\nimport os\nimport pathlib\nimport re\nimport urllib.parse\nimport urllib.request\n\nrepository = os.environ[\"GH_REPOSITORY\"]\nrevision = os.environ[\"TRUSTED_BASE_SHA\"]\nif not re.fullmatch(r\"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+\", repository):\n raise SystemExit(\"Invalid repository reference\")\nif not re.fullmatch(r\"[0-9a-f]{40}\", revision):\n raise SystemExit(\"Invalid trusted base revision\")\npath = \".github/workflows/scripts/mgmt_sdk_review_context.py\"\nurl = (\n f\"https://api.github.com/repos/{repository}/contents/\"\n f\"{urllib.parse.quote(path, safe='/')}?ref={revision}\"\n)\nrequest = urllib.request.Request(\n url,\n headers={\n \"Accept\": \"application/vnd.github+json\",\n \"Authorization\": f\"Bearer {os.environ['GH_TOKEN']}\",\n \"User-Agent\": \"azure-sdk-python-mgmt-review\",\n \"X-GitHub-Api-Version\": \"2022-11-28\",\n },\n)\nwith urllib.request.urlopen(request, timeout=30) as response:\n payload = json.load(response)\nencoded_content = re.sub(r\"\\s+\", \"\", payload[\"content\"])\ncontent = base64.b64decode(encoded_content, validate=True)\nif len(content) > 128 * 1024:\n raise SystemExit(\"Trusted collector exceeded the size limit\")\nscript = pathlib.Path(\"mgmt_sdk_review_context.py\")\nscript.write_bytes(content)\nPY\npython mgmt_sdk_review_context.py\n" shell: bash - name: Install GitHub Copilot CLI diff --git a/.github/workflows/mgmt-sdk-pr-review.md b/.github/workflows/mgmt-sdk-pr-review.md index 28b2f71a9ebe..a537896b2a3f 100644 --- a/.github/workflows/mgmt-sdk-pr-review.md +++ b/.github/workflows/mgmt-sdk-pr-review.md @@ -26,226 +26,54 @@ checkout: false # Collect evidence without checking out or executing pull-request-controlled code. steps: + # Fetch only from the trusted base revision. Never execute the pull request's copy of this script. - name: Collect management SDK review context shell: bash env: GH_TOKEN: ${{ github.token }} GH_REPOSITORY: ${{ github.repository }} PR_NUMBER: ${{ github.event.pull_request.number }} + TRUSTED_BASE_SHA: ${{ github.event.pull_request.base.sha }} run: | python - <<'PY' import base64 - import binascii import json import os + import pathlib import re - import urllib.error import urllib.parse import urllib.request - - API_ROOT = os.environ.get("GH_API_ROOT", "https://api.github.com") - REPOSITORY = os.environ["GH_REPOSITORY"] - PR_NUMBER = int(os.environ["PR_NUMBER"]) - TOKEN = os.environ["GH_TOKEN"] - PACKAGE_PATTERN = re.compile(r"^(sdk/[^/]+/azure-mgmt-[^/]+)(?:/|$)") - - - class GitHubApiError(RuntimeError): - pass - - - def api_get(path): - request = urllib.request.Request( - f"{API_ROOT}{path}", - headers={ - "Accept": "application/vnd.github+json", - "Authorization": f"Bearer {TOKEN}", - "User-Agent": "azure-sdk-python-mgmt-review", - "X-GitHub-Api-Version": "2022-11-28", - }, - ) - try: - with urllib.request.urlopen(request) as response: - return json.load(response) - except urllib.error.HTTPError as error: - detail = error.read().decode("utf-8", errors="replace") - raise GitHubApiError(f"GitHub API request failed ({error.code}) for {path}: {detail}") from error - except urllib.error.URLError as error: - raise GitHubApiError(f"GitHub API request failed for {path}: {error.reason}") from error - - - def paged_get(path, max_items=None): - items = [] - page = 1 - while True: - separator = "&" if "?" in path else "?" - batch = api_get(f"{path}{separator}per_page=100&page={page}") - if not isinstance(batch, list): - raise GitHubApiError(f"GitHub API returned a non-list response for {path}") - items.extend(batch) - if max_items is not None and len(items) >= max_items: - return items[:max_items] - if len(batch) < 100: - return items - page += 1 - - - def read_repository_file(path, revision): - encoded_path = urllib.parse.quote(path, safe="/") - encoded_ref = urllib.parse.quote(revision, safe="") - payload = api_get( - f"/repos/{REPOSITORY}/contents/{encoded_path}?ref={encoded_ref}" - ) - try: - if payload.get("encoding") != "base64": - raise ValueError("content was not base64 encoded") - return base64.b64decode(payload["content"]).decode("utf-8") - except (binascii.Error, KeyError, TypeError, ValueError, UnicodeDecodeError) as error: - raise GitHubApiError(f"Could not read {path} at {revision}: {error}") from error - - - def extract_management_review_rules(instructions): - lines = instructions.splitlines() - heading = "## MGMT SDK Code Review Rules" - try: - start = lines.index(heading) - except ValueError as error: - raise GitHubApiError(f"{heading} was not found in .github/copilot-instructions.md") from error - end = next( - (index for index in range(start + 1, len(lines)) if lines[index].startswith("## ")), - len(lines), - ) - return "\n".join(lines[start:end]).strip() - - - def read_api_version(package_path, revision): - metadata_path = f"{package_path}/_metadata.json" - encoded_path = urllib.parse.quote(metadata_path, safe="/") - encoded_ref = urllib.parse.quote(revision, safe="") - try: - payload = api_get( - f"/repos/{REPOSITORY}/contents/{encoded_path}?ref={encoded_ref}" - ) - except GitHubApiError as error: - return None, str(error) - - try: - if payload.get("encoding") != "base64": - raise ValueError("content was not base64 encoded") - content = base64.b64decode(payload["content"]).decode("utf-8") - api_version = json.loads(content)["apiVersion"] - if not isinstance(api_version, str) or not api_version: - raise ValueError("apiVersion was missing or was not a non-empty string") - return api_version, None - except (binascii.Error, KeyError, TypeError, ValueError, UnicodeDecodeError) as error: - return None, f"Could not read apiVersion from {metadata_path} at {revision}: {error}" - - - repository = api_get(f"/repos/{REPOSITORY}") - default_branch = repository.get("default_branch") - if not isinstance(default_branch, str) or not default_branch: - raise GitHubApiError("Repository metadata did not contain a default branch") - instructions = read_repository_file(".github/copilot-instructions.md", default_branch) - management_review_rules = extract_management_review_rules(instructions) - - pull_request = api_get(f"/repos/{REPOSITORY}/pulls/{PR_NUMBER}") - expected_changed_files = pull_request.get("changed_files") - if not isinstance(expected_changed_files, int) or expected_changed_files < 0: - raise GitHubApiError("Pull request metadata did not contain a valid changed_files count") - latest_revision = pull_request.get("head", {}).get("sha") - if not isinstance(latest_revision, str) or not latest_revision: - raise GitHubApiError("Pull request metadata did not contain a valid head SHA") - - changed_files = paged_get( - f"/repos/{REPOSITORY}/pulls/{PR_NUMBER}/files", - max_items=3000, - ) - returned_changed_files = len(changed_files) - package_discovery_complete = returned_changed_files == expected_changed_files - package_discovery_error = None - if not package_discovery_complete: - package_discovery_error = ( - "Management package discovery is incomplete: pull request metadata reports " - f"{expected_changed_files} changed files, but the GitHub API returned " - f"{returned_changed_files}. GitHub limits pull request file responses to 3,000 files." - ) - - package_paths = sorted( - { - match.group(1) - for item in changed_files - for field in ("filename", "previous_filename") - for path in [item.get(field)] - if isinstance(path, str) - for match in [PACKAGE_PATTERN.match(path)] - if match - } - ) - - commits = paged_get( - f"/repos/{REPOSITORY}/pulls/{PR_NUMBER}/commits", - max_items=250, + repository = os.environ["GH_REPOSITORY"] + revision = os.environ["TRUSTED_BASE_SHA"] + if not re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", repository): + raise SystemExit("Invalid repository reference") + if not re.fullmatch(r"[0-9a-f]{40}", revision): + raise SystemExit("Invalid trusted base revision") + path = ".github/workflows/scripts/mgmt_sdk_review_context.py" + url = ( + f"https://api.github.com/repos/{repository}/contents/" + f"{urllib.parse.quote(path, safe='/')}?ref={revision}" ) - commit_shas = [item.get("sha") for item in commits if isinstance(item.get("sha"), str)] - if not commit_shas: - raise GitHubApiError("Pull request metadata returned an empty commit list") - - first_revision = commit_shas[0] - drift_results = [] - for package_path in package_paths: - first_api_version, first_error = read_api_version(package_path, first_revision) - latest_api_version, latest_error = read_api_version(package_path, latest_revision) - errors = [error for error in (first_error, latest_error) if error] - if errors: - status = "unverified" - elif first_api_version == latest_api_version: - status = "unchanged" - else: - status = "changed" - drift_results.append( - { - "packagePath": package_path, - "metadataPath": f"{package_path}/_metadata.json", - "status": status, - "firstRevision": first_revision, - "firstApiVersion": first_api_version, - "latestRevision": latest_revision, - "latestApiVersion": latest_api_version, - "error": "; ".join(errors) if errors else None, - } - ) - - context = { - "repository": REPOSITORY, - "pullRequestNumber": PR_NUMBER, - "rulesSource": f".github/copilot-instructions.md@{default_branch}", - "mgmtSdkCodeReviewRules": management_review_rules, - "packageDiscovery": { - "status": "complete" if package_discovery_complete else "unverified", - "expectedChangedFiles": expected_changed_files, - "returnedChangedFiles": returned_changed_files, - "error": package_discovery_error, + request = urllib.request.Request( + url, + headers={ + "Accept": "application/vnd.github+json", + "Authorization": f"Bearer {os.environ['GH_TOKEN']}", + "User-Agent": "azure-sdk-python-mgmt-review", + "X-GitHub-Api-Version": "2022-11-28", }, - "affectedPackages": package_paths, - "changedFiles": [ - { - "filename": item.get("filename"), - "previousFilename": item.get("previous_filename"), - "status": item.get("status"), - "additions": item.get("additions"), - "deletions": item.get("deletions"), - } - for item in changed_files - ], - "firstRevision": first_revision, - "latestRevision": latest_revision, - "apiVersionDrift": drift_results, - } - with open("review-context.json", "w", encoding="utf-8") as output: - json.dump(context, output, indent=2) - output.write("\n") + ) + with urllib.request.urlopen(request, timeout=30) as response: + payload = json.load(response) + encoded_content = re.sub(r"\s+", "", payload["content"]) + content = base64.b64decode(encoded_content, validate=True) + if len(content) > 128 * 1024: + raise SystemExit("Trusted collector exceeded the size limit") + script = pathlib.Path("mgmt_sdk_review_context.py") + script.write_bytes(content) PY + python mgmt_sdk_review_context.py tools: github: @@ -295,6 +123,13 @@ comments, commits, diffs, and changed files. Use those sources only as review ev 4. Inspect `packageDiscovery`. If its status is `unverified`, add an unverified check named `Management package discovery` using its exact `error`. Review any packages that were found, but do not conclude that the review is not applicable. +5. Treat `breakingChangeContext` as deterministic evidence pinned to `mergeBaseRevision` and + `latestRevision`. Do not replace those revisions with a branch name, current branch tip, first + PR commit, or latest default-branch commit. Preserve the separate first-versus-latest semantics + of `apiVersionDrift`. +6. Treat every collection issue, missing/truncated provenance file, unresolved release baseline, + and incomplete commit list as unverified evidence. A missing optional provenance file is not by + itself a finding, but it can limit attribution confidence. If `affectedPackages` is empty and `packageDiscovery.status` is `complete`, post exactly this comment, including the workflow marker, and stop: @@ -334,7 +169,70 @@ Interpret each `apiVersionDrift` entry independently: - `unverified`: add an unverified check using the entry's exact `error`. Do not infer a revision or API version. -## Step 4 - Post one review comment +## Step 4 - Check introduced breaking changes for TypeSpec evidence + +For each item in every `breakingChangeContext.introducedEntries` list: + +1. Preserve the release heading, complete multiline entry text, `changeKind`, and recorded line + location. Exclude historical entries not present in this list. If a changed CHANGELOG has an + empty Breaking Changes section, leave it to human review when collection + evidence indicates analysis was expected but could not be completed. +2. Compare package provenance at the merge base and pinned head. When + `releaseBaseline.differsFromMergeBase` is true, use the release baseline provenance for causal + comparison. The inferred tag is evidence, + not proof of the changelog generator's exact comparison target; preserve the recorded `basis` + uncertainty internally. If a missing or ambiguous baseline prevents connecting a TypeSpec + change to the SDK entry, leave that entry to human review with a short reason. +3. Use `_metadata.json`, `tsp-location.yaml`, TypeSpec configuration, and permitted API artifacts + to identify the old and new specification sources and selected API versions. Focus on whether + a specific TypeSpec change directly explains the named SDK breaking change. +4. From each validated `specificationSources` repository and immutable revision, fetch only the + files needed to trace the named model, enum, operation, or parameter. Follow source-directory + moves, imports/shared models, client naming decorators, versioning annotations, API-version + selection, and renamed files. Bound investigation to 20 repository searches/file fetches and + 1 MiB of fetched text per package. Validate repository names and full 40-character SHAs before + fetching. Stop investigating an entry once direct evidence explains it. If access failures, + search truncation, ambiguous matches, or exhausted limits prevent a conclusion, leave it to + human review. +5. Do not investigate emitter/compiler/generator causes, dependency locks, or toolchain release + notes. Neither a specification commit change nor an emitter version bump alone proves a cause. + Absence of TypeSpec evidence does not prove that the toolchain caused the change or that the + TypeSpec was unchanged. Do not perform regeneration experiments. +6. Prefer permitted API artifacts such as `api.md` when available. Do not fetch or analyze files + excluded by the authoritative review rules merely to bypass those exclusions. Never execute, + build, import, regenerate, or check out pull-request-controlled code. + +Use exactly one outcome in the Cause column for each entry: + +- `TypeSpec/API`: direct evidence from a specific source definition, decorator, versioning + annotation, or API-version selection change explains the named SDK change. Show the relevant + old/new source or an explicit versioning annotation connecting them. A related model change + alone is not sufficient to explain an enum removal without evidence connecting the enum. +- `Human review`: no direct TypeSpec evidence was established. Write "Needs human review" and a + short entry-specific reason or question. Do not speculate about other causes or require the + author to provide dependency locks as a routine follow-up. + +Use `High` confidence only for direct evidence connecting the TypeSpec change to the SDK entry; +otherwise use `Human review` with `N/A` confidence instead of a tentative attribution. This +classification establishes a TypeSpec contribution, not that all toolchain contributions have +been ruled out. +Link only to immutable commit, tag-object, or release URLs. Do not claim candidate replacements are +proven mappings without source evidence connecting them. + +For every file-based evidence link or finding location, use a GitHub blob permalink pinned to the +full commit SHA with a verified 1-based line anchor (`#L42`) or minimal relevant range +(`#L42-L48`). Link to the exact definition, decorator, configuration value, or release-note entry +supporting the claim, not merely the file. Verify line numbers against the complete file at that +same revision; never infer them from a diff, truncated excerpt, or another revision. Link each +changelog entry using its `startLine` and `endLine` at `latestRevision`. When comparing old and new +code, anchor each link independently at its respective revision. If exact lines cannot be verified, +state that limitation alongside the immutable file link rather than inventing an anchor. Non-file +commit and release pages do not require code-line anchors. + +Attribution is explanatory. Do not create or escalate a rule-violation finding solely because a +breaking change is classified or left to human review. + +## Step 5 - Post one review comment Post exactly one comment through the `add-comment` safe output. Begin with this marker: @@ -349,7 +247,7 @@ Then provide findings ordered by severity: | Severity | Finding | Location | Evidence | Rule | Remediation | | --- | --- | --- | --- | --- | --- | -| `Blocking`, `Warning`, or `Suggestion` | Concise title | File and line when available | Observed evidence | Authoritative rule heading | Specific remediation | +| `Blocking`, `Warning`, or `Suggestion` | Concise title | Immutable file link with verified line anchor | Observed evidence | Authoritative rule heading | Specific remediation | ``` Use one finding per row. Preserve full revision and API-version values. Requirement violations @@ -361,7 +259,10 @@ findings, replace the findings table with: **Findings:** None. ``` -Follow it with: +Reserve unverified checks for required MGMT SDK review rules and API-version drift checks that +could not be completed. Do not list attribution baseline uncertainty, unresolved enum causation, +or missing toolchain dependencies here; keep any relevant handoff in the attribution row. +Follow the findings with: ```markdown ### Unverified checks @@ -377,6 +278,28 @@ If every check was verified, replace that table with: **Unverified checks:** None. ``` +Then include a distinct attribution section after unverified checks: + +```markdown +### Breaking-change attribution + +**Package: package name | Release: release heading** + +| Changelog entry | Cause | Evidence and explanation | Confidence | +| --- | --- | --- | --- | +| Full introduced or modified entry linked to its changelog lines | `TypeSpec/API` or `Human review` | Direct TypeSpec evidence with immutable line links and a concise explanation, or "Needs human review" with a short reason | `High` with rationale, or `N/A` for human review | +``` + +Group entries by package and release, with a label above each group's table; do not repeat package +or release in a table column. Use one row per introduced entry. Preserve multiline entry meaning while converting line breaks to +`
`, and escape Markdown table delimiters. If no introduced Breaking Changes entries were found +and collection completed, write `**Breaking-change attribution:** No newly added or modified +entries.` Do not merge attribution rows into the findings table. + +If collection is incomplete, identify the affected package or changelog under this attribution +section as needing human review; do not imply that all introduced entries were checked. Do not +add a separate attribution limitations table or repeat handoff reasons under unverified checks. + Finish with a brief `### Review summary` naming every affected package and the checks completed. ## Constraints diff --git a/.github/workflows/scripts/mgmt_sdk_review_context.py b/.github/workflows/scripts/mgmt_sdk_review_context.py new file mode 100644 index 000000000000..aa0069a7abd6 --- /dev/null +++ b/.github/workflows/scripts/mgmt_sdk_review_context.py @@ -0,0 +1,733 @@ +#!/usr/bin/env python3 +"""Collect immutable, non-executable evidence for the management SDK PR reviewer.""" + +import base64 +import binascii +import difflib +import json +import os +import re +import urllib.error +import urllib.parse +import urllib.request + + +API_ROOT = os.environ.get("GH_API_ROOT", "https://api.github.com") +MAX_API_RESPONSE_BYTES = 12 * 1024 * 1024 +MAX_TEXT_FILE_BYTES = 256 * 1024 +MAX_PAGES = 30 +MAX_API_REQUESTS = 500 +API_TIMEOUT_SECONDS = 30 +PACKAGE_PATTERN = re.compile(r"^(sdk/[^/]+/azure-mgmt-[^/]+)(?:/|$)") +REPOSITORY_PATTERN = re.compile(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$") +SHA_PATTERN = re.compile(r"^[0-9a-f]{40}$") +RELEASE_HEADING = re.compile(r"^##\s+(.+?)\s*$") +SECTION_HEADING = re.compile(r"^(#{1,6})\s+(.+?)\s*$") +BULLET = re.compile(r"^\s*[-*]\s+(.*)$") +VERSION_LIKE_KEY = re.compile( + r"typespec|emitter|compiler|generator|client-generator|http-client-python|autorest", + re.IGNORECASE, +) +PROVENANCE_PATHS = ( + "_metadata.json", + "tsp-location.yaml", + "api.metadata.yml", + "pyproject.toml", + "TempTypeSpecFiles/package-lock.json", +) +MAX_PACKAGE_API_REQUESTS = 4 * len(PROVENANCE_PATHS) + 2 + 2 + + +class GitHubApiError(RuntimeError): + def __init__(self, message, status=None): + super().__init__(message) + self.status = status + + +class GitHubClient: + def __init__(self, repository, token, api_root=API_ROOT): + if not REPOSITORY_PATTERN.fullmatch(repository): + raise ValueError(f"Invalid GitHub repository reference: {repository!r}") + self.repository = repository + self.token = token + self.api_root = api_root.rstrip("/") + self.request_count = 0 + + def get(self, path): + if not path.startswith("/"): + raise ValueError("GitHub API paths must be absolute") + if self.request_count >= MAX_API_REQUESTS: + raise GitHubApiError(f"GitHub API request limit ({MAX_API_REQUESTS}) was reached") + self.request_count += 1 + request = urllib.request.Request( + f"{self.api_root}{path}", + headers={ + "Accept": "application/vnd.github+json", + "Authorization": f"Bearer {self.token}", + "User-Agent": "azure-sdk-python-mgmt-review", + "X-GitHub-Api-Version": "2022-11-28", + }, + ) + try: + with urllib.request.urlopen(request, timeout=API_TIMEOUT_SECONDS) as response: + content_length = response.headers.get("Content-Length") + if content_length and int(content_length) > MAX_API_RESPONSE_BYTES: + raise GitHubApiError(f"GitHub API response exceeded the size limit for {path}") + payload = response.read(MAX_API_RESPONSE_BYTES + 1) + if len(payload) > MAX_API_RESPONSE_BYTES: + raise GitHubApiError(f"GitHub API response exceeded the size limit for {path}") + return json.loads(payload.decode("utf-8")) + except urllib.error.HTTPError as error: + detail = error.read(4096).decode("utf-8", errors="replace") + raise GitHubApiError( + f"GitHub API request failed ({error.code}) for {path}: {detail}", status=error.code + ) from error + except urllib.error.URLError as error: + raise GitHubApiError(f"GitHub API request failed for {path}: {error.reason}") from error + except (UnicodeDecodeError, json.JSONDecodeError, ValueError) as error: + raise GitHubApiError(f"GitHub API returned invalid or oversized data for {path}: {error}") from error + + def paged_get(self, path, max_items): + items = [] + for page in range(1, MAX_PAGES + 1): + separator = "&" if "?" in path else "?" + batch = self.get(f"{path}{separator}per_page=100&page={page}") + if not isinstance(batch, list): + raise GitHubApiError(f"GitHub API returned a non-list response for {path}") + items.extend(batch) + if len(items) >= max_items: + return items[:max_items], len(batch) == 100 + if len(batch) < 100: + return items, False + return items, True + + def read_file(self, path, revision): + encoded_path = urllib.parse.quote(path, safe="/") + encoded_ref = urllib.parse.quote(revision, safe="") + api_path = f"/repos/{self.repository}/contents/{encoded_path}?ref={encoded_ref}" + try: + payload = self.get(api_path) + except GitHubApiError as error: + return { + "status": "missing" if error.status == 404 else "unverified", + "path": path, + "revision": revision, + "error": str(error), + } + try: + if not isinstance(payload, dict): + raise TypeError("GitHub API response was not an object") + if payload.get("type") != "file" or payload.get("encoding") != "base64": + raise ValueError("content was unavailable as a base64 file") + declared_size = payload.get("size") + if isinstance(declared_size, int) and declared_size > MAX_TEXT_FILE_BYTES: + return { + "status": "truncated", + "path": path, + "revision": revision, + "htmlUrl": payload.get("html_url"), + "size": declared_size, + "error": f"File exceeded the {MAX_TEXT_FILE_BYTES}-byte evidence limit", + } + encoded_content = re.sub(r"\s+", "", payload["content"]) + content = base64.b64decode(encoded_content, validate=True) + if len(content) > MAX_TEXT_FILE_BYTES: + raise ValueError(f"decoded content exceeded {MAX_TEXT_FILE_BYTES} bytes") + return { + "status": "available", + "path": path, + "revision": revision, + "htmlUrl": payload.get("html_url"), + "sha": payload.get("sha"), + "content": content.decode("utf-8"), + } + except (binascii.Error, KeyError, TypeError, ValueError, UnicodeDecodeError) as error: + return { + "status": "unverified", + "path": path, + "revision": revision, + "error": f"Could not read {path} at {revision}: {error}", + } + + +def parse_breaking_changes(content): + """Parse Breaking Changes bullets while preserving release and source lines.""" + lines = content.splitlines() + entries = [] + empty_sections = [] + releases = [] + release = None + index = 0 + while index < len(lines): + release_match = RELEASE_HEADING.match(lines[index]) + if release_match: + release = release_match.group(1) + releases.append({"heading": release, "line": index + 1}) + index += 1 + continue + heading_match = SECTION_HEADING.match(lines[index]) + if not heading_match or len(heading_match.group(1)) != 3 or heading_match.group(2).lower() != "breaking changes": + index += 1 + continue + + section_line = index + 1 + index += 1 + section_entry_count = 0 + while index < len(lines): + next_heading = SECTION_HEADING.match(lines[index]) + if next_heading and len(next_heading.group(1)) <= 3: + break + bullet_match = BULLET.match(lines[index]) + if not bullet_match: + index += 1 + continue + start = index + entry_lines = [bullet_match.group(1).rstrip()] + index += 1 + while index < len(lines): + if BULLET.match(lines[index]) or SECTION_HEADING.match(lines[index]): + break + entry_lines.append(lines[index].rstrip()) + index += 1 + while entry_lines and not entry_lines[-1]: + entry_lines.pop() + text = "\n".join(entry_lines).strip() + entries.append( + { + "release": release, + "text": text, + "startLine": start + 1, + "endLine": start + max(1, len(entry_lines)), + "sectionLine": section_line, + } + ) + section_entry_count += 1 + if section_entry_count == 0: + empty_sections.append({"release": release, "sectionLine": section_line}) + return {"entries": entries, "emptySections": empty_sections, "releases": releases} + + +def release_key(entry): + heading = entry.get("release") + return heading.split()[0] if heading else None + + +def introduced_breaking_changes(old_entries, new_entries): + """Return new or modified target entries, excluding exact historical entries.""" + unmatched_old = list(old_entries) + introduced = [] + for new_entry in new_entries: + exact_index = next( + ( + index + for index, old_entry in enumerate(unmatched_old) + if release_key(old_entry) == release_key(new_entry) and old_entry["text"] == new_entry["text"] + ), + None, + ) + if exact_index is not None: + unmatched_old.pop(exact_index) + continue + + candidates = [entry for entry in unmatched_old if release_key(entry) == release_key(new_entry)] + previous = None + similarity = 0.0 + for candidate in candidates: + ratio = difflib.SequenceMatcher(None, candidate["text"], new_entry["text"]).ratio() + if ratio > similarity: + similarity = ratio + previous = candidate + result = dict(new_entry) + if previous is not None and similarity >= 0.55: + result["changeKind"] = "modified" + result["previousText"] = previous["text"] + unmatched_old.remove(previous) + else: + result["changeKind"] = "added" + result["previousText"] = None + introduced.append(result) + return introduced + + +def parse_json_evidence(file_evidence): + if file_evidence.get("status") != "available": + return None, file_evidence.get("error") + try: + return json.loads(file_evidence["content"]), None + except (json.JSONDecodeError, TypeError) as error: + return None, f"Invalid JSON in {file_evidence['path']}: {error}" + + +def version_value_kind(value): + if not isinstance(value, str): + return "other" + return "range" if re.search(r"[<>=~^*| ]", value) else "resolved" + + +def extract_lock_versions(lock_data): + versions = [] + packages = lock_data.get("packages", {}) if isinstance(lock_data, dict) else {} + if isinstance(packages, dict): + for path, details in packages.items(): + if not isinstance(details, dict): + continue + name = path.rsplit("node_modules/", 1)[-1] if "node_modules/" in path else details.get("name", path) + version = details.get("version") + if VERSION_LIKE_KEY.search(str(name)) and isinstance(version, str): + versions.append({"name": name, "version": version, "kind": "resolved"}) + return sorted(versions, key=lambda item: (item["name"], item["version"]))[:100] + + +def parse_tsp_location(content): + result = {"additionalDirectories": []} + current_list = None + for raw_line in content.splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + if line.startswith("-") and current_list: + result[current_list].append(line[1:].strip().strip("'\"")) + continue + match = re.fullmatch(r"([A-Za-z][A-Za-z0-9]*):\s*(.*?)\s*", line) + if not match: + continue + key, value = match.groups() + if key == "additionalDirectories": + current_list = key + if value and value != "[]": + result[key].append(value.strip("'\"")) + else: + current_list = None + result[key] = value.strip("'\"") + return result + + +def summarize_provenance(files): + summary = { + "files": [], + "metadata": None, + "tspLocation": None, + "resolvedDependencies": [], + "issues": [], + } + for evidence in files: + compact = {key: value for key, value in evidence.items() if key != "content"} + summary["files"].append(compact) + if evidence["path"].endswith("_metadata.json"): + metadata, error = parse_json_evidence(evidence) + if error: + summary["issues"].append(error) + elif isinstance(metadata, dict): + selected = {} + for key, value in metadata.items(): + if key in { + "apiVersion", + "apiVersions", + "commit", + "repository_url", + "typespec_src", + "typespecAdditionalOptions", + "emitterVersion", + "httpClientPythonVersion", + } or VERSION_LIKE_KEY.search(key): + selected[key] = { + "value": value, + "kind": version_value_kind(value), + } + summary["metadata"] = selected + elif evidence["path"].endswith("tsp-location.yaml") and evidence.get("status") == "available": + summary["tspLocation"] = parse_tsp_location(evidence["content"]) + elif evidence["path"].endswith("package-lock.json") and evidence.get("status") == "available": + lock_data, error = parse_json_evidence(evidence) + if error: + summary["issues"].append(error) + else: + summary["resolvedDependencies"] = extract_lock_versions(lock_data) + elif evidence.get("status") == "available": + summary["files"][-1]["content"] = evidence["content"] + if evidence.get("status") in {"unverified", "truncated"}: + summary["issues"].append(evidence.get("error")) + metadata = summary.get("metadata") or {} + tsp_location = summary.get("tspLocation") or {} + comparisons = (("commit", "commit"), ("typespec_src", "directory")) + for metadata_key, location_key in comparisons: + metadata_value = (metadata.get(metadata_key) or {}).get("value") + location_value = tsp_location.get(location_key) + if metadata_value and location_value and metadata_value != location_value: + summary["issues"].append( + f"Conflicting provenance: _metadata.json {metadata_key}={metadata_value!r}, " + f"tsp-location.yaml {location_key}={location_value!r}" + ) + repository_url = (metadata.get("repository_url") or {}).get("value") + location_repository = tsp_location.get("repo") + if repository_url and location_repository: + match = re.fullmatch( + r"https://github\.com/([A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+?)(?:\.git)?/?", repository_url + ) + metadata_repository = match.group(1) if match else repository_url + if metadata_repository.lower() != location_repository.lower(): + summary["issues"].append( + "Conflicting provenance: _metadata.json repository_url=" + f"{repository_url!r}, tsp-location.yaml repo={location_repository!r}" + ) + return summary + + +def metadata_api_version(provenance): + metadata = provenance.get("metadata") or {} + item = metadata.get("apiVersion") or {} + value = item.get("value") + return value if isinstance(value, str) and value else None + + +def collect_provenance(client, package_path, revision): + files = [client.read_file(f"{package_path}/{relative_path}", revision) for relative_path in PROVENANCE_PATHS] + return summarize_provenance(files) + + +def api_version_drift(package_path, first_revision, latest_revision, first_provenance, latest_provenance): + first_api_version = metadata_api_version(first_provenance) + latest_api_version = metadata_api_version(latest_provenance) + drift_errors = first_provenance["issues"] + latest_provenance["issues"] + for label, revision, version in ( + ("first", first_revision, first_api_version), + ("latest", latest_revision, latest_api_version), + ): + if not version: + drift_errors.append( + f"{package_path}/_metadata.json at {label} revision {revision} " + "does not contain a non-empty string apiVersion" + ) + return { + "packagePath": package_path, + "metadataPath": f"{package_path}/_metadata.json", + "status": ( + "unverified" + if not first_api_version or not latest_api_version + else "unchanged" if first_api_version == latest_api_version else "changed" + ), + "firstRevision": first_revision, + "firstApiVersion": first_api_version, + "latestRevision": latest_revision, + "latestApiVersion": latest_api_version, + "error": "; ".join(drift_errors) if (not first_api_version or not latest_api_version) else None, + } + + +def latest_release_version(parsed_changelog): + for release in parsed_changelog.get("releases", []): + version = release["heading"].split()[0] + if version != "0.0.0" and not re.search(r"\(\s*unreleased\s*\)", release["heading"], re.IGNORECASE): + return version + return None + + +def resolve_release_tag(client, package_name, version): + if not version: + return {"status": "unverified", "error": "No previous release heading was found at the merge base"} + tag = f"{package_name}_{version}" + encoded_tag = urllib.parse.quote(tag, safe="") + try: + ref = client.get(f"/repos/{client.repository}/git/ref/tags/{encoded_tag}") + if not isinstance(ref, dict): + raise GitHubApiError(f"Tag lookup for {tag} returned a non-object response") + target = ref.get("object", {}) + if target.get("type") == "tag": + tag_payload = client.get(f"/repos/{client.repository}/git/tags/{target.get('sha')}") + if not isinstance(tag_payload, dict): + raise GitHubApiError(f"Annotated tag lookup for {tag} returned a non-object response") + target = tag_payload.get("object", {}) + sha = target.get("sha") + if target.get("type") != "commit" or not isinstance(sha, str) or not SHA_PATTERN.fullmatch(sha): + raise GitHubApiError(f"Tag {tag} did not resolve to an immutable commit") + return {"status": "available", "tag": tag, "revision": sha} + except GitHubApiError as error: + return {"status": "unverified", "tag": tag, "error": str(error)} + + +def validated_source_reference(provenance): + conflicts = [issue for issue in provenance.get("issues", []) if issue.startswith("Conflicting provenance:")] + if conflicts: + return {"status": "unverified", "error": "; ".join(conflicts)} + metadata = provenance.get("metadata") or {} + repository_item = metadata.get("repository_url") or {} + commit_item = metadata.get("commit") or {} + repository_url = repository_item.get("value") + commit = commit_item.get("value") + tsp_location = provenance.get("tspLocation") or {} + repository = tsp_location.get("repo") + if repository_url: + match = re.fullmatch( + r"https://github\.com/([A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+?)(?:\.git)?/?", repository_url + ) + repository = match.group(1) if match else None + commit = commit or tsp_location.get("commit") + match = REPOSITORY_PATTERN.fullmatch(repository or "") + if not match or not isinstance(commit, str) or not SHA_PATTERN.fullmatch(commit): + return { + "status": "unverified", + "error": "Specification repository URL or commit was missing or invalid", + } + return {"status": "available", "repository": repository, "revision": commit} + + +def collect(): + repository = os.environ["GH_REPOSITORY"] + pr_number = int(os.environ["PR_NUMBER"]) + client = GitHubClient(repository, os.environ["GH_TOKEN"]) + + repository_data = client.get(f"/repos/{repository}") + default_branch = repository_data.get("default_branch") + branch_data = client.get(f"/repos/{repository}/branches/{urllib.parse.quote(default_branch, safe='')}") + rules_revision = branch_data.get("commit", {}).get("sha") + if not isinstance(rules_revision, str) or not SHA_PATTERN.fullmatch(rules_revision): + raise GitHubApiError("Default branch metadata did not contain an immutable commit SHA") + rules_file = client.read_file(".github/copilot-instructions.md", rules_revision) + if rules_file.get("status") != "available": + raise GitHubApiError(rules_file.get("error", "Could not load review rules")) + lines = rules_file["content"].splitlines() + heading = "## MGMT SDK Code Review Rules" + try: + start = lines.index(heading) + except ValueError as error: + raise GitHubApiError(f"{heading} was not found in .github/copilot-instructions.md") from error + end = next((index for index in range(start + 1, len(lines)) if lines[index].startswith("## ")), len(lines)) + + pull_request = client.get(f"/repos/{repository}/pulls/{pr_number}") + expected_changed_files = pull_request.get("changed_files") + latest_revision = pull_request.get("head", {}).get("sha") + base_revision = pull_request.get("base", {}).get("sha") + if not isinstance(expected_changed_files, int) or expected_changed_files < 0: + raise GitHubApiError("Pull request metadata did not contain a valid changed_files count") + if not all(isinstance(value, str) and SHA_PATTERN.fullmatch(value) for value in (latest_revision, base_revision)): + raise GitHubApiError("Pull request metadata did not contain valid base and head SHAs") + + compare = client.get(f"/repos/{repository}/compare/{base_revision}...{latest_revision}") + merge_base_revision = compare.get("merge_base_commit", {}).get("sha") + if not isinstance(merge_base_revision, str) or not SHA_PATTERN.fullmatch(merge_base_revision): + raise GitHubApiError("The compare API did not return a valid merge-base SHA") + + changed_files, file_list_truncated = client.paged_get( + f"/repos/{repository}/pulls/{pr_number}/files", max_items=3000 + ) + package_discovery_complete = len(changed_files) == expected_changed_files and not file_list_truncated + package_discovery_error = None + if not package_discovery_complete: + package_discovery_error = ( + "Management package discovery is incomplete: pull request metadata reports " + f"{expected_changed_files} changed files, the API returned {len(changed_files)}, " + "or the bounded pagination limit was reached." + ) + + package_paths = sorted( + { + match.group(1) + for item in changed_files + for field in ("filename", "previous_filename") + for path in (item.get(field),) + if isinstance(path, str) + for match in (PACKAGE_PATTERN.match(path),) + if match + } + ) + commits, commits_truncated = client.paged_get(f"/repos/{repository}/pulls/{pr_number}/commits", max_items=250) + commit_shas = [item.get("sha") for item in commits if isinstance(item.get("sha"), str)] + expected_commits = pull_request.get("commits") + commit_discovery_complete = ( + isinstance(expected_commits, int) + and not isinstance(expected_commits, bool) + and expected_commits > 0 + and len(commits) == expected_commits + and len(commit_shas) == len(commits) + and not commits_truncated + ) + if not commit_shas: + raise GitHubApiError("Pull request metadata returned an empty commit list") + first_revision = commit_shas[0] + + drift_results = [] + breaking_change_context = [] + for package_path in package_paths: + remaining_requests = MAX_API_REQUESTS - client.request_count + if remaining_requests < MAX_PACKAGE_API_REQUESTS: + reason = ( + f"Needs human review: evidence collection for {package_path} was skipped because only " + f"{remaining_requests} API requests remain; a package requires a budget of up to " + f"{MAX_PACKAGE_API_REQUESTS} requests (four provenance snapshots, two changelogs, " + "and two tag lookups). Completed package results are preserved." + ) + unavailable = {"status": "unverified", "error": reason} + drift_results.append( + { + "packagePath": package_path, + "metadataPath": f"{package_path}/_metadata.json", + "status": "unverified", + "firstRevision": first_revision, + "firstApiVersion": None, + "latestRevision": latest_revision, + "latestApiVersion": None, + "error": reason, + } + ) + breaking_change_context.append( + { + "packagePath": package_path, + "changelogPath": f"{package_path}/CHANGELOG.md", + "baseChangelogPath": None, + "mergeBaseRevision": merge_base_revision, + "latestRevision": latest_revision, + "status": "unverified", + "introducedEntries": [], + "emptyBreakingChangeSections": [], + "collectionIssues": [reason], + "releaseBaseline": unavailable, + "provenance": {"mergeBase": unavailable, "latest": unavailable}, + "specificationSources": {"mergeBase": unavailable, "latest": unavailable, "release": unavailable}, + } + ) + continue + first_provenance = collect_provenance(client, package_path, first_revision) + latest_provenance = collect_provenance(client, package_path, latest_revision) + drift_results.append( + api_version_drift(package_path, first_revision, latest_revision, first_provenance, latest_provenance) + ) + + head_changelog_path = f"{package_path}/CHANGELOG.md" + changelog_change = next( + (item for item in changed_files if item.get("filename") == head_changelog_path), None + ) + base_changelog_path = ( + changelog_change.get("previous_filename") + if changelog_change and changelog_change.get("status") == "renamed" + else head_changelog_path + ) + old_file = client.read_file(base_changelog_path, merge_base_revision) + new_file = client.read_file(head_changelog_path, latest_revision) + collection_issues = [] + old_parsed = {"entries": [], "emptySections": [], "releases": []} + new_parsed = {"entries": [], "emptySections": [], "releases": []} + baseline_available = old_file.get("status") == "available" or ( + old_file.get("status") == "missing" + and changelog_change is not None + and changelog_change.get("status") == "added" + ) + if old_file.get("status") == "available": + old_parsed = parse_breaking_changes(old_file["content"]) + elif not baseline_available: + collection_issues.append( + old_file.get("error") or f"Merge-base changelog was unavailable: {base_changelog_path}" + ) + if new_file.get("status") == "available": + new_parsed = parse_breaking_changes(new_file["content"]) + else: + collection_issues.append(new_file.get("error")) + + introduced = ( + introduced_breaking_changes(old_parsed["entries"], new_parsed["entries"]) + if baseline_available and new_file.get("status") == "available" + else [] + ) + previous_version = latest_release_version(old_parsed) + release_baseline = resolve_release_tag(client, package_path.rsplit("/", 1)[-1], previous_version) + if release_baseline["status"] == "available": + release_baseline["provenance"] = collect_provenance( + client, package_path, release_baseline["revision"] + ) + release_baseline["differsFromMergeBase"] = release_baseline["revision"] != merge_base_revision + release_baseline["basis"] = ( + "Inferred from the newest release heading at the merge base; the changelog generator's exact " + "comparison target is not recorded by CHANGELOG.md." + ) + else: + collection_issues.append(release_baseline.get("error")) + + merge_base_provenance = collect_provenance(client, package_path, merge_base_revision) + collection_issues.extend(merge_base_provenance["issues"]) + collection_issues.extend(latest_provenance["issues"]) + if release_baseline.get("provenance"): + collection_issues.extend(release_baseline["provenance"]["issues"]) + collection_issues = list(dict.fromkeys(issue for issue in collection_issues if issue)) + breaking_change_context.append( + { + "packagePath": package_path, + "changelogPath": head_changelog_path, + "baseChangelogPath": base_changelog_path, + "mergeBaseRevision": merge_base_revision, + "latestRevision": latest_revision, + "status": "unverified" if collection_issues else "complete", + "introducedEntries": introduced, + "emptyBreakingChangeSections": new_parsed["emptySections"], + "collectionIssues": [issue for issue in collection_issues if issue], + "releaseBaseline": release_baseline, + "provenance": { + "mergeBase": merge_base_provenance, + "latest": latest_provenance, + }, + "specificationSources": { + "mergeBase": validated_source_reference(merge_base_provenance), + "latest": validated_source_reference(latest_provenance), + "release": ( + validated_source_reference(release_baseline["provenance"]) + if release_baseline.get("provenance") + else {"status": "unverified", "error": "Release provenance was unavailable"} + ), + }, + } + ) + + context = { + "repository": repository, + "pullRequestNumber": pr_number, + "rulesSource": f".github/copilot-instructions.md@{rules_revision}", + "mgmtSdkCodeReviewRules": "\n".join(lines[start:end]).strip(), + "packageDiscovery": { + "status": "complete" if package_discovery_complete else "unverified", + "expectedChangedFiles": expected_changed_files, + "returnedChangedFiles": len(changed_files), + "error": package_discovery_error, + }, + "affectedPackages": package_paths, + "changedFiles": [ + { + "filename": item.get("filename"), + "previousFilename": item.get("previous_filename"), + "status": item.get("status"), + "additions": item.get("additions"), + "deletions": item.get("deletions"), + } + for item in changed_files + ], + "firstRevision": first_revision, + "latestRevision": latest_revision, + "mergeBaseRevision": merge_base_revision, + "commitDiscovery": { + "status": "complete" if commit_discovery_complete else "unverified", + "expectedCommits": expected_commits, + "returnedCommits": len(commits), + "error": ( + None + if commit_discovery_complete + else f"PR commit discovery is incomplete: expected {expected_commits!r}, returned {len(commits)} " + f"with {len(commit_shas)} commit SHAs; the commit endpoint is limited to 250 commits " + "and bounded pagination may be incomplete." + ), + }, + "apiVersionDrift": drift_results, + "breakingChangeContext": breaking_change_context, + "collectionLimits": { + "maxApiResponseBytes": MAX_API_RESPONSE_BYTES, + "maxTextFileBytes": MAX_TEXT_FILE_BYTES, + "maxPages": MAX_PAGES, + "maxApiRequests": MAX_API_REQUESTS, + "maxPackageApiRequests": MAX_PACKAGE_API_REQUESTS, + "apiTimeoutSeconds": API_TIMEOUT_SECONDS, + "githubApiRequests": client.request_count, + }, + } + with open("review-context.json", "w", encoding="utf-8") as output: + json.dump(context, output, indent=2) + output.write("\n") + + +if __name__ == "__main__": + collect() \ No newline at end of file diff --git a/.github/workflows/tests/test_mgmt_sdk_review_context.py b/.github/workflows/tests/test_mgmt_sdk_review_context.py new file mode 100644 index 000000000000..dac11ea79c0b --- /dev/null +++ b/.github/workflows/tests/test_mgmt_sdk_review_context.py @@ -0,0 +1,458 @@ +import base64 +import importlib.util +import io +import json +from pathlib import Path +import textwrap +import unittest +from unittest import mock +import urllib.error +import urllib.parse + + +SCRIPT = Path(__file__).parents[1] / "scripts" / "mgmt_sdk_review_context.py" +SPEC = importlib.util.spec_from_file_location("mgmt_sdk_review_context", SCRIPT) +MODULE = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(MODULE) + + +class WorkflowBootstrapTests(unittest.TestCase): + def test_setup_action_matches_compiler_version(self): + lines = (SCRIPT.parents[1] / "mgmt-sdk-pr-review.lock.yml").read_text(encoding="utf-8").splitlines() + metadata = json.loads(lines[0].removeprefix("# gh-aw-metadata: ")) + manifest = json.loads(lines[1].removeprefix("# gh-aw-manifest: ")) + setup = next(action for action in manifest["actions"] if action["repo"] == "github/gh-aw-actions/setup") + self.assertEqual(metadata["compiler_version"], setup["version"]) + uses = [line.strip() for line in lines if "uses: github/gh-aw-actions/setup@" in line] + self.assertTrue(uses) + self.assertTrue(all(line == f"uses: github/gh-aw-actions/setup@{setup['sha']} # {setup['version']}" for line in uses)) + + def test_single_trusted_collector_has_valid_python(self): + workflow = (SCRIPT.parents[1] / "mgmt-sdk-pr-review.md").read_text(encoding="utf-8") + blocks = workflow.split("python - <<'PY'")[1:] + self.assertEqual(1, len(blocks)) + source = textwrap.dedent(blocks[0].split("\n PY", 1)[0]) + compile(source, "collector-bootstrap", "exec") + self.assertIn('revision = os.environ["TRUSTED_BASE_SHA"]', source) + self.assertIn("TRUSTED_BASE_SHA: ${{ github.event.pull_request.base.sha }}", workflow) + self.assertEqual(1, workflow.count(" python mgmt_sdk_review_context.py")) + + def test_generated_bootstrap_has_valid_python(self): + workflow = (SCRIPT.parents[1] / "mgmt-sdk-pr-review.lock.yml").read_text(encoding="utf-8") + runs = [ + json.loads(line.strip().removeprefix("run: ")) + for line in workflow.splitlines() + if line.strip().startswith('run: "python - ') + ] + self.assertEqual(1, len(runs)) + source = runs[0].split("python - <<'PY'\n", 1)[1].split("\nPY\n", 1)[0] + compile(source, "generated-collector-bootstrap", "exec") + + +class BreakingChangeParserTests(unittest.TestCase): + def test_date_correction_preserves_exact_and_modified_matching(self): + old = MODULE.parse_breaking_changes( + "## 2.0.0 (2026-01-01)\n### Breaking Changes\n" + "- Deleted model `OldWidget`.\n- Method `Widgets.get` was renamed.\n" + ) + corrected = MODULE.parse_breaking_changes( + "## 2.0.0 (2026-01-02)\n### Breaking Changes\n" + "- Deleted model `OldWidget`.\n- Method `Widgets.get` was renamed.\n" + ) + self.assertEqual([], MODULE.introduced_breaking_changes(old["entries"], corrected["entries"])) + + corrected["entries"][1]["text"] = "Method `Widgets.get` was renamed to `Widgets.fetch`." + introduced = MODULE.introduced_breaking_changes(old["entries"], corrected["entries"]) + self.assertEqual(1, len(introduced)) + self.assertEqual("modified", introduced[0]["changeKind"]) + self.assertEqual("2.0.0 (2026-01-02)", introduced[0]["release"]) + self.assertEqual(old["entries"][1]["text"], introduced[0]["previousText"]) + + def test_identical_entry_in_different_version_is_added(self): + old = MODULE.parse_breaking_changes("## 1.0.0 (2026-01-01)\n### Breaking Changes\n- Deleted model.\n") + new = MODULE.parse_breaking_changes("## 2.0.0 (2026-01-02)\n### Breaking Changes\n- Deleted model.\n") + self.assertEqual("added", MODULE.introduced_breaking_changes(old["entries"], new["entries"])[0]["changeKind"]) + + def test_added_modified_multiline_and_historical_entries(self): + old = MODULE.parse_breaking_changes( + """# Release History + +## 2.0.0 (2026-01-01) +### Breaking Changes + - Method `Widgets.get` was renamed. + +## 1.0.0 (2025-01-01) +### Breaking Changes + - Historical entry. +""" + ) + new = MODULE.parse_breaking_changes( + """# Release History + +## 2.0.0 (2026-01-01) +### Breaking Changes + - Method `Widgets.get` was renamed to `Widgets.fetch`. + Use `fetch` for new calls. + - Deleted model `OldWidget`. + +## 1.0.0 (2025-01-01) +### Breaking Changes + - Historical entry. +""" + ) + + introduced = MODULE.introduced_breaking_changes(old["entries"], new["entries"]) + + self.assertEqual(["modified", "added"], [entry["changeKind"] for entry in introduced]) + self.assertIn("Use `fetch`", introduced[0]["text"]) + self.assertEqual(5, introduced[0]["startLine"]) + self.assertNotIn("Historical entry.", [entry["text"] for entry in introduced]) + + def test_empty_section_is_recorded(self): + parsed = MODULE.parse_breaking_changes( + """## 3.0.0 (2026-02-02) +### Breaking Changes + +### Features Added +- A feature +""" + ) + + self.assertEqual([], parsed["entries"]) + self.assertEqual([{"release": "3.0.0 (2026-02-02)", "sectionLine": 2}], parsed["emptySections"]) + + def test_multiple_releases_remain_separate(self): + parsed = MODULE.parse_breaking_changes( + """## 3.0.0 (2026-02-02) +### Breaking Changes +- New break +## 2.0.0 (2026-01-01) +### Breaking Changes +- Old break +""" + ) + + self.assertEqual( + ["3.0.0 (2026-02-02)", "2.0.0 (2026-01-01)"], + [entry["release"] for entry in parsed["entries"]], + ) + + def test_latest_release_uses_all_headings_and_skips_placeholder(self): + parsed = MODULE.parse_breaking_changes( + """## 0.0.0 (Unreleased) +### Features Added +- Pending +## 1.2.0b1 (2026-01-02) +### Features Added +- Released +## 1.0.0 (2025-01-01) +### Breaking Changes +- Old break +""" + ) + + self.assertEqual("1.2.0b1", MODULE.latest_release_version(parsed)) + + def test_latest_release_skips_versioned_unreleased_headings(self): + for marker in ("Unreleased", "unreleased", "UNRELEASED"): + with self.subTest(marker=marker): + parsed = MODULE.parse_breaking_changes( + f"## 1.6.1 ({marker})\n### Bugs Fixed\n- Pending fix\n" + "## 1.6.0 (2025-07-02)\n### Other Changes\n- Released\n" + ) + self.assertEqual("1.6.0", MODULE.latest_release_version(parsed)) + + def test_only_unreleased_headings_have_no_release_baseline(self): + parsed = MODULE.parse_breaking_changes( + "## 0.0.0 (Unreleased)\n## 1.6.1 (Unreleased)\n## 2.0.0b1 (unreleased)\n" + ) + self.assertIsNone(MODULE.latest_release_version(parsed)) + + +class ProvenanceTests(unittest.TestCase): + def test_ranges_are_not_reported_as_resolved_versions(self): + metadata = { + "status": "available", + "path": "pkg/_metadata.json", + "revision": "a" * 40, + "content": json.dumps( + { + "emitterVersion": "0.63.6", + "httpClientPythonVersion": "^0.37.1", + } + ), + } + lock = { + "status": "available", + "path": "pkg/TempTypeSpecFiles/package-lock.json", + "revision": "a" * 40, + "content": json.dumps( + { + "packages": { + "node_modules/@azure-tools/typespec-python": {"version": "0.63.6"}, + "node_modules/@typespec/compiler": {"version": "1.4.0"}, + } + } + ), + } + + summary = MODULE.summarize_provenance([metadata, lock]) + + self.assertEqual("range", summary["metadata"]["httpClientPythonVersion"]["kind"]) + self.assertEqual("resolved", summary["metadata"]["emitterVersion"]["kind"]) + self.assertEqual(2, len(summary["resolvedDependencies"])) + + def test_missing_and_truncated_evidence_are_explicit(self): + summary = MODULE.summarize_provenance( + [ + {"status": "missing", "path": "pkg/tsp-location.yaml", "revision": "a" * 40, "error": "404"}, + { + "status": "truncated", + "path": "pkg/TempTypeSpecFiles/package-lock.json", + "revision": "a" * 40, + "error": "too large", + }, + ] + ) + + self.assertEqual(["too large"], summary["issues"]) + self.assertEqual(["missing", "truncated"], [item["status"] for item in summary["files"]]) + + def test_source_reference_requires_github_url_and_full_sha(self): + invalid = {"metadata": {"repository_url": {"value": "https://example.com/specs"}, "commit": {"value": "main"}}} + valid = { + "metadata": { + "repository_url": {"value": "https://github.com/Azure/azure-rest-api-specs"}, + "commit": {"value": "a" * 40}, + } + } + + self.assertEqual("unverified", MODULE.validated_source_reference(invalid)["status"]) + self.assertEqual("available", MODULE.validated_source_reference(valid)["status"]) + + def test_tsp_location_supplies_release_provenance_and_reports_conflicts(self): + tsp_location = { + "status": "available", + "path": "pkg/tsp-location.yaml", + "revision": "a" * 40, + "content": ( + "directory: specification/contoso/New\n" + f"commit: {'b' * 40}\n" + "repo: Azure/azure-rest-api-specs\n" + "additionalDirectories:\n" + " - specification/common-types/resource-management\n" + ), + } + metadata = { + "status": "available", + "path": "pkg/_metadata.json", + "revision": "a" * 40, + "content": json.dumps( + { + "typespec_src": "specification/contoso/Old", + "commit": "c" * 40, + "repository_url": "https://github.com/Azure/different-specs", + } + ), + } + + release_summary = MODULE.summarize_provenance([tsp_location]) + conflicted = MODULE.summarize_provenance([metadata, tsp_location]) + + self.assertEqual("available", MODULE.validated_source_reference(release_summary)["status"]) + self.assertEqual( + ["specification/common-types/resource-management"], + release_summary["tspLocation"]["additionalDirectories"], + ) + self.assertEqual(3, len(conflicted["issues"])) + self.assertEqual("unverified", MODULE.validated_source_reference(conflicted)["status"]) + + +class CollectionTests(unittest.TestCase): + def collect_context( + self, *, old_status=200, changelog_status="modified", commit_count=1, package_count=1, annotated_tag=False + ): + packages = [f"sdk/contoso/azure-mgmt-contoso{index}" for index in range(package_count)] + changelog = "## 1.0.0 (2026-01-01)\n### Breaking Changes\n- Historical entry.\n" + + def respond(request, timeout): + parsed = urllib.parse.urlparse(request.full_url) + path = parsed.path.removeprefix("/repos/Azure/azure-sdk-for-python") + query = urllib.parse.parse_qs(parsed.query) + if not path: + data = {"default_branch": "main"} + elif path == "/branches/main": + data = {"commit": {"sha": "f" * 40}} + elif path == "/pulls/1": + data = { + "changed_files": package_count, + "commits": commit_count, + "head": {"sha": "b" * 40}, + "base": {"sha": "e" * 40}, + } + elif path.startswith("/compare/"): + data = {"merge_base_commit": {"sha": "c" * 40}} + elif path == "/pulls/1/files": + data = [{"filename": f"{package}/CHANGELOG.md", "status": changelog_status} for package in packages] + elif path == "/pulls/1/commits": + offset = (int(query["page"][0]) - 1) * 100 + data = [{"sha": "a" * 40}] * max(0, min(100, min(commit_count, 250) - offset)) + elif path.startswith("/git/ref/tags/"): + data = {"object": {"type": "tag" if annotated_tag else "commit", "sha": "d" * 40}} + elif path.startswith("/git/tags/"): + data = {"object": {"type": "commit", "sha": "d" * 40}} + elif path.startswith("/contents/"): + filename = urllib.parse.unquote(path.removeprefix("/contents/")) + if filename == ".github/copilot-instructions.md": + content = "## MGMT SDK Code Review Rules\nReview the package.\n" + elif filename.endswith("/CHANGELOG.md"): + if query["ref"][0] == "c" * 40 and old_status != 200: + raise urllib.error.HTTPError(request.full_url, old_status, "baseline unavailable", {}, None) + content = changelog + elif filename.endswith("/_metadata.json"): + content = json.dumps({"apiVersion": "2026-01-01"}) + else: + content = "{}" + data = { + "type": "file", + "encoding": "base64", + "content": base64.b64encode(content.encode()).decode(), + } + else: + raise AssertionError(f"Unexpected API call: {request.full_url}") + response = io.BytesIO(json.dumps(data).encode()) + response.headers = {} + return response + + output = mock.mock_open() + with ( + mock.patch.dict(MODULE.os.environ, {"GH_REPOSITORY": "Azure/azure-sdk-for-python", "GH_TOKEN": "test", "PR_NUMBER": "1"}), + mock.patch.object(MODULE.urllib.request, "urlopen", side_effect=respond) as requests, + mock.patch("builtins.open", output), + ): + MODULE.collect() + output.assert_called_once_with("review-context.json", "w", encoding="utf-8") + context = json.loads("".join(call.args[0] for call in output().write.call_args_list)) + self.assertEqual(requests.call_count, context["collectionLimits"]["githubApiRequests"]) + return context + + def test_unavailable_baseline_never_emits_historical_deltas(self): + for status in (403, 404, 503): + with self.subTest(status=status): + package = self.collect_context(old_status=status)["breakingChangeContext"][0] + self.assertEqual("unverified", package["status"]) + self.assertEqual([], package["introducedEntries"]) + self.assertTrue(any(str(status) in issue for issue in package["collectionIssues"])) + + def test_added_changelog_can_use_missing_baseline(self): + package = self.collect_context(old_status=404, changelog_status="added")["breakingChangeContext"][0] + self.assertEqual(1, len(package["introducedEntries"])) + self.assertEqual("added", package["introducedEntries"][0]["changeKind"]) + self.assertFalse(any("404" in issue for issue in package["collectionIssues"])) + + def test_available_baseline_excludes_unchanged_history(self): + package = self.collect_context()["breakingChangeContext"][0] + self.assertEqual("complete", package["status"]) + self.assertEqual([], package["introducedEntries"]) + + def test_commit_discovery_checks_declared_count_at_endpoint_cap(self): + for count, expected in ((249, "complete"), (250, "complete"), (251, "unverified"), (400, "unverified")): + with self.subTest(count=count): + context = self.collect_context(commit_count=count) + discovery = context["commitDiscovery"] + self.assertEqual(expected, discovery["status"]) + self.assertEqual(count, discovery["expectedCommits"]) + self.assertEqual(min(count, 250), discovery["returnedCommits"]) + self.assertEqual("a" * 40, context["firstRevision"]) + if expected == "unverified": + self.assertIn(f"expected {count}, returned 250", discovery["error"]) + else: + self.assertIsNone(discovery["error"]) + + def test_budget_preserves_completed_packages_and_hands_off_remainder(self): + with mock.patch.object(MODULE, "MAX_API_REQUESTS", 40): + context = self.collect_context(package_count=3, annotated_tag=True) + packages = context["breakingChangeContext"] + self.assertEqual("complete", context["packageDiscovery"]["status"]) + self.assertEqual(3, len(context["affectedPackages"])) + self.assertEqual(["complete", "unverified", "unverified"], [package["status"] for package in packages]) + self.assertEqual(["unchanged", "unverified", "unverified"], [drift["status"] for drift in context["apiVersionDrift"]]) + self.assertEqual(31, context["collectionLimits"]["githubApiRequests"]) + for package in packages[1:]: + self.assertEqual([], package["introducedEntries"]) + self.assertIn("Needs human review", package["collectionIssues"][0]) + self.assertIn("only 9 API requests remain", package["collectionIssues"][0]) + + def test_budget_boundary_allows_full_package_or_explicit_handoff(self): + for limit, expected in ((30, "unverified"), (31, "complete")): + with self.subTest(limit=limit), mock.patch.object(MODULE, "MAX_API_REQUESTS", limit): + context = self.collect_context(annotated_tag=True) + self.assertEqual(expected, context["breakingChangeContext"][0]["status"]) + self.assertLessEqual(context["collectionLimits"]["githubApiRequests"], limit) + + +class FailureHandlingTests(unittest.TestCase): + def test_drift_reports_missing_or_invalid_api_version(self): + for metadata in ({}, [], None, {"apiVersion": ""}, {"apiVersion": 42}): + with self.subTest(metadata=metadata): + provenance = MODULE.summarize_provenance( + [{"status": "available", "path": "pkg/_metadata.json", "content": json.dumps(metadata)}] + ) + result = MODULE.api_version_drift("pkg", "a" * 40, "b" * 40, provenance, provenance) + + self.assertEqual("unverified", result["status"]) + self.assertIn(f"first revision {'a' * 40}", result["error"]) + self.assertIn(f"latest revision {'b' * 40}", result["error"]) + self.assertIn("non-empty string apiVersion", result["error"]) + self.assertEqual([], provenance["issues"]) + + def test_drift_preserves_valid_comparisons_and_failure_details(self): + first = {"metadata": {"apiVersion": {"value": "2026-01-01"}}, "issues": []} + latest = {"metadata": {"apiVersion": {"value": "2026-02-01"}}, "issues": []} + for provenance, expected in ((first, "unchanged"), (latest, "changed")): + result = MODULE.api_version_drift("pkg", "a" * 40, "b" * 40, first, provenance) + self.assertEqual(expected, result["status"]) + self.assertIsNone(result["error"]) + + result = MODULE.api_version_drift( + "pkg", "a" * 40, "b" * 40, first, {"metadata": None, "issues": ["rate limited"]} + ) + self.assertEqual("unverified", result["status"]) + self.assertIn("rate limited", result["error"]) + self.assertNotIn("first revision", result["error"]) + + def test_invalid_repository_is_rejected(self): + with self.assertRaises(ValueError): + MODULE.GitHubClient("https://github.com/Azure/repo", "token") + + def test_read_file_records_api_failure(self): + client = MODULE.GitHubClient("Azure/azure-sdk-for-python", "token") + + def fail(_): + raise MODULE.GitHubApiError("rate limited", status=403) + + client.get = fail + evidence = client.read_file("CHANGELOG.md", "a" * 40) + + self.assertEqual("unverified", evidence["status"]) + self.assertIn("rate limited", evidence["error"]) + + def test_read_file_records_malformed_success_payload(self): + client = MODULE.GitHubClient("Azure/azure-sdk-for-python", "token") + client.get = lambda _: [] + + evidence = client.read_file("CHANGELOG.md", "a" * 40) + + self.assertEqual("unverified", evidence["status"]) + self.assertIn("not an object", evidence["error"]) + + def test_request_limit_fails_before_network_access(self): + client = MODULE.GitHubClient("Azure/azure-sdk-for-python", "token") + client.request_count = MODULE.MAX_API_REQUESTS + + with self.assertRaisesRegex(MODULE.GitHubApiError, "request limit"): + client.get("/repos/Azure/azure-sdk-for-python") + + +if __name__ == "__main__": + unittest.main() \ No newline at end of file