diff --git a/.overhaul/APPLY.sh b/.overhaul/APPLY.sh new file mode 100755 index 0000000..6ba8312 --- /dev/null +++ b/.overhaul/APPLY.sh @@ -0,0 +1,720 @@ +#!/usr/bin/env bash +# Apply the public GitHub metadata in repository-lifecycle.yml to the +# OpenAdaptAI organization: the organization description and website, and each +# listed repository's description, website, and topics. +# +# Generated by `python3 scripts/github_metadata.py render-apply` from +# repository-lifecycle.yml. Don't edit this file by hand. Edit the registry +# and render it again; scripts/check_profile.py fails when the two differ. +# +# Usage: +# APPLY.sh Dry run. Prints each command and request body. +# Calls nothing and changes nothing. +# APPLY.sh --apply Makes the changes, then reads GitHub back and +# compares every value with the registry. +# APPLY.sh --verify Reads GitHub back and compares. Changes nothing. +# +# Exit status: 0 when GitHub matches the registry. 1 when a command failed, +# a value couldn't be read, or a value differs. 2 when everything matches +# except the pinned repositories, which only an owner can set by hand. +# 64 for a usage error. +# +# Needs the GitHub CLI (gh) signed in as an owner of OpenAdaptAI, with a token +# that has the admin:org and repo scopes. To add them, run: +# gh auth refresh -s admin:org,repo +# +# Each listed repository is set exactly: an empty website clears the field, +# and the topic list replaces every current topic. +# +# GitHub has no API for organization pins. On https://github.com/OpenAdaptAI, +# choose "Customize pins" and pin these repositories in this order: +# 1. OpenAdapt +# 2. openadapt-flow +# 3. openadapt-desktop +# 4. openadapt-capture +# 5. openadapt-agent +# 6. ehr-integration-directory + +set -uo pipefail + +ORG=OpenAdaptAI +PINS=OpenAdapt,openadapt-flow,openadapt-desktop,openadapt-capture,openadapt-agent,ehr-integration-directory +REGISTERED='OpenAdapt openadapt-flow openadapt-desktop openadapt-capture openadapt-agent ehr-integration-directory openadapt-cloud .github openadapt-web openadapt-ops openadapt-blog openadapt-wright openadapt-herald openadapt-crier openadapt-consilium openadapt-telemetry openadapt-viewer openadapt-privacy openadapt-types openadapt-console openadapt-tray openadapt-ml openadapt-evals openadapt-retrieval openadapt-grounding OmniMCP SoM PydanticPrompt OpenSanitizer' + +MODE=dry-run +if [ "$#" -gt 1 ]; then + printf 'Use one option: --dry-run (the default), --apply, or --verify.\n' >&2 + exit 64 +fi +case "${1:-}" in + "" | --dry-run) MODE=dry-run ;; + --apply) MODE=apply ;; + --verify) MODE=verify ;; + -h | --help) + sed -n '2,/^$/p' "$0" + exit 0 + ;; + *) + printf 'Unknown option: %s. Use --dry-run, --apply, or --verify.\n' "$1" >&2 + exit 64 + ;; +esac + +ERR_FILE=$(mktemp "${TMPDIR:-/tmp}/openadapt-apply.XXXXXX") || exit 1 +trap 'rm -f "$ERR_FILE"' EXIT + +CHANGES=0 +FAILED=0 +CHECKED=0 +DIFFERENT=0 +PINS_PENDING=0 + +# change LABEL COMMAND...: run one write. The request body arrives on stdin. +change() { + local label=$1 + shift + local body + body=$(cat) + CHANGES=$((CHANGES + 1)) + if [ "$MODE" = dry-run ]; then + printf '\n# %s\n' "$label" + printf '%q ' "$@" + printf "<<'JSON'\n%s\nJSON\n" "$body" + return 0 + fi + if printf '%s\n' "$body" | "$@" >/dev/null 2>"$ERR_FILE"; then + printf 'changed %s\n' "$label" + else + FAILED=$((FAILED + 1)) + printf 'FAILED %s\n' "$label" >&2 + sed 's/^/ /' "$ERR_FILE" >&2 + fi +} + +# read_value LABEL COMMAND...: print one value read from GitHub, or fail. +read_value() { + local label=$1 + shift + if ! "$@" 2>"$ERR_FILE"; then + printf 'UNREAD %s\n' "$label" >&2 + sed 's/^/ /' "$ERR_FILE" >&2 + return 1 + fi +} + +# expect LABEL EXPECTED COMMAND...: compare one value on GitHub with the registry. +expect() { + local label=$1 expected=$2 actual + shift 2 + CHECKED=$((CHECKED + 1)) + if ! actual=$(read_value "$label" "$@"); then + DIFFERENT=$((DIFFERENT + 1)) + return + fi + if [ "$actual" = "$expected" ]; then + printf 'matches %s\n' "$label" + else + DIFFERENT=$((DIFFERENT + 1)) + printf 'DIFFERS %s\n GitHub: %s\n registry: %s\n' \ + "$label" "$actual" "$expected" >&2 + fi +} + +check_pins() { + local actual + CHECKED=$((CHECKED + 1)) + # The GraphQL query names $org itself, so the shell must not expand it. + # shellcheck disable=SC2016 + if ! actual=$(read_value "organization: pinned repositories" \ + gh api graphql -F org="$ORG" -f query='query($org: String!) { organization(login: $org) { pinnedItems(first: 6, types: REPOSITORY) { nodes { ... on Repository { name } } } } }' \ + --jq '[.data.organization.pinnedItems.nodes[].name] | join(",")'); then + DIFFERENT=$((DIFFERENT + 1)) + return + fi + if [ "$actual" = "$PINS" ]; then + printf 'matches organization: pinned repositories\n' + else + PINS_PENDING=1 + printf 'TO PIN organization: pinned repositories (set by hand)\n GitHub: %s\n registry: %s\n' \ + "$actual" "$PINS" >&2 + fi +} + +# Public repositories that the registry doesn't describe. A note, not a failure. +note_unregistered() { + local names name + if ! names=$(read_value "organization: public repositories" \ + gh api --paginate "orgs/$ORG/repos?type=public&per_page=100" \ + --jq '.[] | select(.archived | not) | .name'); then + return + fi + for name in $names; do + case " $REGISTERED " in + *" $name "*) ;; + *) printf 'note %s is public but has no entry in repository-lifecycle.yml\n' "$name" ;; + esac + done +} + +preflight() { + if ! command -v gh >/dev/null 2>&1; then + printf "The GitHub CLI (gh) isn't installed. Nothing was changed.\n" >&2 + exit 1 + fi + if ! gh auth status >/dev/null 2>&1; then + printf "gh isn't signed in. Run gh auth login, then try again. Nothing was changed.\n" >&2 + exit 1 + fi + if [ "$MODE" = apply ]; then + local role + role=$(gh api "user/memberships/orgs/$ORG" --jq .role 2>/dev/null) || role="" + if [ "$role" != admin ]; then + printf "This gh account isn't an owner of %s (role: %s). Nothing was changed.\n" \ + "$ORG" "${role:-none}" >&2 + exit 1 + fi + fi +} + +apply_changes() { + change "organization: description and website" \ + gh api --method PATCH "orgs/$ORG" --input - <<'JSON' +{"description": "Enters approved information into the systems your team already uses, checks that it saved, and stops to ask a person when something doesn't match.", "blog": "https://openadapt.ai/"} +JSON + change 'OpenAdapt: description and website' \ + gh api --method PATCH "repos/$ORG/OpenAdapt" --input - <<'JSON' +{"description": "OpenAdapt takes the last manual step off your team. It enters approved information into EMRs, portals, and desktop or Citrix apps, then checks that it saved.", "homepage": "https://openadapt.ai"} +JSON + change 'OpenAdapt: topics' \ + gh api --method PUT "repos/$ORG/OpenAdapt/topics" --input - <<'JSON' +{"names": ["rpa", "robotic-process-automation", "process-automation", "workflow-automation", "desktop-automation", "browser-automation", "gui-automation", "ui-automation", "citrix", "rdp", "healthcare-automation", "emr", "ehr", "computer-use", "human-in-the-loop", "audit-trail", "local-first", "automation-from-demonstration", "mcp", "python"]} +JSON + change 'openadapt-flow: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-flow" --input - <<'JSON' +{"description": "OpenAdapt's open-source engine. It turns one recorded task into a program that runs on your computer and never reports an unchecked entry as done.", "homepage": "https://openadapt.ai/platform/flow"} +JSON + change 'openadapt-flow: topics' \ + gh api --method PUT "repos/$ORG/openadapt-flow/topics" --input - <<'JSON' +{"names": ["browser-automation", "computer-use", "deterministic-replay", "gui-automation", "healthcare-automation", "rpa", "workflow-automation", "automation-from-demonstration", "citrix", "demonstration-compiler", "desktop-automation", "effect-verification", "local-first", "python", "rdp", "ui-automation", "human-in-the-loop", "emr", "audit-trail", "robotic-process-automation"]} +JSON + change 'openadapt-desktop: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-desktop" --input - <<'JSON' +{"description": "The OpenAdapt app. Record a task once, watch it run, see what it entered, and decide what happens when a run stops.", "homepage": "https://openadapt.ai/platform/desktop"} +JSON + change 'openadapt-desktop: topics' \ + gh api --method PUT "repos/$ORG/openadapt-desktop/topics" --input - <<'JSON' +{"names": ["automation-from-demonstration", "desktop-automation", "gui-automation", "local-first", "openadapt", "python", "human-in-the-loop", "rpa", "citrix"]} +JSON + change 'openadapt-capture: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-capture" --input - <<'JSON' +{"description": "Records screen, mouse, and keyboard on Windows, macOS, and Linux so OpenAdapt can turn one recorded task into an automation.", "homepage": "https://openadapt.ai/platform/capture"} +JSON + change 'openadapt-capture: topics' \ + gh api --method PUT "repos/$ORG/openadapt-capture/topics" --input - <<'JSON' +{"names": ["gui-automation", "python", "screen-capture", "input-recording", "openadapt"]} +JSON + change 'openadapt-agent: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-agent" --input - <<'JSON' +{"description": "Lets your AI agent run approved OpenAdapt automations on your computer. Each run reports how it ended, and a run that stops goes to a person.", "homepage": "https://openadapt.ai/platform/agent"} +JSON + change 'openadapt-agent: topics' \ + gh api --method PUT "repos/$ORG/openadapt-agent/topics" --input - <<'JSON' +{"names": ["agent-skills", "gui-automation", "local-first", "mcp", "openadapt", "workflow-automation", "computer-use", "human-in-the-loop"]} +JSON + change 'ehr-integration-directory: description and website' \ + gh api --method PATCH "repos/$ORG/ehr-integration-directory" --input - <<'JSON' +{"description": "Shows which EHR tasks have a public API and which still end at a screen. Each entry links its source. Published by OpenAdapt.", "homepage": "https://ehrintegrationdirectory.com"} +JSON + change 'ehr-integration-directory: topics' \ + gh api --method PUT "repos/$ORG/ehr-integration-directory/topics" --input - <<'JSON' +{"names": ["ehr", "emr", "fhir", "hl7", "smart-on-fhir", "healthcare", "healthcare-it", "interoperability", "openadapt"]} +JSON + change 'openadapt-cloud: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-cloud" --input - <<'JSON' +{"description": "Proprietary source for app.openadapt.ai, the hosted service that runs automations, tests them before they go live, and sends stopped runs to a person.", "homepage": "https://app.openadapt.ai"} +JSON + change 'openadapt-cloud: topics' \ + gh api --method PUT "repos/$ORG/openadapt-cloud/topics" --input - <<'JSON' +{"names": []} +JSON + change '.github: description and website' \ + gh api --method PATCH "repos/$ORG/.github" --input - <<'JSON' +{"description": "Support: OpenAdapt's GitHub profile, the status of every repository, and shared community files.", "homepage": "https://openadapt.ai"} +JSON + change '.github: topics' \ + gh api --method PUT "repos/$ORG/.github/topics" --input - <<'JSON' +{"names": ["openadapt"]} +JSON + change 'openadapt-web: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-web" --input - <<'JSON' +{"description": "Source for openadapt.ai, including product pages, pricing, and evidence.", "homepage": "https://openadapt.ai"} +JSON + change 'openadapt-web: topics' \ + gh api --method PUT "repos/$ORG/openadapt-web/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-ops: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-ops" --input - <<'JSON' +{"description": "Support: source for docs.openadapt.ai, plus the checks that keep the docs in step with each release.", "homepage": "https://docs.openadapt.ai"} +JSON + change 'openadapt-ops: topics' \ + gh api --method PUT "repos/$ORG/openadapt-ops/topics" --input - <<'JSON' +{"names": ["openadapt"]} +JSON + change 'openadapt-blog: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-blog" --input - <<'JSON' +{"description": "Support: source for blog.openadapt.ai, with guides, product updates, and test write-ups.", "homepage": "https://blog.openadapt.ai"} +JSON + change 'openadapt-blog: topics' \ + gh api --method PUT "repos/$ORG/openadapt-blog/topics" --input - <<'JSON' +{"names": ["openadapt"]} +JSON + change 'openadapt-wright: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-wright" --input - <<'JSON' +{"description": "Support: team tool that drafts code changes and pull requests for a person to approve. Not part of the product.", "homepage": ""} +JSON + change 'openadapt-wright: topics' \ + gh api --method PUT "repos/$ORG/openadapt-wright/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-herald: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-herald" --input - <<'JSON' +{"description": "Support: team tool that drafts release announcements from git history. Not part of the product.", "homepage": ""} +JSON + change 'openadapt-herald: topics' \ + gh api --method PUT "repos/$ORG/openadapt-herald/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-crier: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-crier" --input - <<'JSON' +{"description": "Support: team tool that sends drafted social posts to Telegram for a person to approve. Not part of the product.", "homepage": ""} +JSON + change 'openadapt-crier: topics' \ + gh api --method PUT "repos/$ORG/openadapt-crier/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-consilium: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-consilium" --input - <<'JSON' +{"description": "Support: team tool that asks several AI models one question and combines their answers. Not part of the product.", "homepage": ""} +JSON + change 'openadapt-consilium: topics' \ + gh api --method PUT "repos/$ORG/openadapt-consilium/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-telemetry: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-telemetry" --input - <<'JSON' +{"description": "Support: error reports and usage counts for OpenAdapt packages, with personal details scrubbed. Set DO_NOT_TRACK=1 to opt out.", "homepage": ""} +JSON + change 'openadapt-telemetry: topics' \ + gh api --method PUT "repos/$ORG/openadapt-telemetry/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-viewer: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-viewer" --input - <<'JSON' +{"description": "Support: turns test results and screenshots into one HTML page you can share. Not needed to automate a task.", "homepage": "https://pypi.org/project/openadapt-viewer/"} +JSON + change 'openadapt-viewer: topics' \ + gh api --method PUT "repos/$ORG/openadapt-viewer/topics" --input - <<'JSON' +{"names": ["html", "openadapt", "python", "visualization"]} +JSON + change 'openadapt-privacy: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-privacy" --input - <<'JSON' +{"description": "Experimental: detects and redacts personal and health information in recordings and screenshots before they're shared.", "homepage": "https://pypi.org/project/openadapt-privacy/"} +JSON + change 'openadapt-privacy: topics' \ + gh api --method PUT "repos/$ORG/openadapt-privacy/topics" --input - <<'JSON' +{"names": ["openadapt", "pii-detection", "privacy", "python", "scrubbing"]} +JSON + change 'openadapt-types: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-types" --input - <<'JSON' +{"description": "Experimental: shared data formats for screens and actions inside OpenAdapt. Not needed to automate a task.", "homepage": ""} +JSON + change 'openadapt-types: topics' \ + gh api --method PUT "repos/$ORG/openadapt-types/topics" --input - <<'JSON' +{"names": ["openadapt"]} +JSON + change 'openadapt-console: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-console" --input - <<'JSON' +{"description": "Experimental: an early web console prototype. Not part of the product today.", "homepage": ""} +JSON + change 'openadapt-console: topics' \ + gh api --method PUT "repos/$ORG/openadapt-console/topics" --input - <<'JSON' +{"names": []} +JSON + change 'openadapt-tray: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-tray" --input - <<'JSON' +{"description": "Experimental: a tray icon that shows what OpenAdapt is doing, with start and stop controls. Optional.", "homepage": ""} +JSON + change 'openadapt-tray: topics' \ + gh api --method PUT "repos/$ORG/openadapt-tray/topics" --input - <<'JSON' +{"names": ["openadapt"]} +JSON + change 'openadapt-ml: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-ml" --input - <<'JSON' +{"description": "Research: optional model training for computer-use agents. Not needed to record, build, or run an OpenAdapt automation.", "homepage": "https://pypi.org/project/openadapt-ml/"} +JSON + change 'openadapt-ml: topics' \ + gh api --method PUT "repos/$ORG/openadapt-ml/topics" --input - <<'JSON' +{"names": ["computer-use", "gui-automation", "machine-learning", "openadapt", "python", "vlm", "research"]} +JSON + change 'openadapt-evals: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-evals" --input - <<'JSON' +{"description": "Research: scores computer-use agents and OpenAdapt programs on public benchmarks, including Windows Agent Arena. Not needed to automate a task.", "homepage": "https://openadapt.ai/platform/evals"} +JSON + change 'openadapt-evals: topics' \ + gh api --method PUT "repos/$ORG/openadapt-evals/topics" --input - <<'JSON' +{"names": ["benchmarks", "evaluation", "gui-automation", "openadapt", "python"]} +JSON + change 'openadapt-retrieval: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-retrieval" --input - <<'JSON' +{"description": "Research: searches past recordings by image and text for model-driven agents. Not needed to automate a task.", "homepage": "https://pypi.org/project/openadapt-retrieval/"} +JSON + change 'openadapt-retrieval: topics' \ + gh api --method PUT "repos/$ORG/openadapt-retrieval/topics" --input - <<'JSON' +{"names": ["embeddings", "openadapt", "python", "retrieval", "demo-retrieval"]} +JSON + change 'openadapt-grounding: description and website' \ + gh api --method PATCH "repos/$ORG/openadapt-grounding" --input - <<'JSON' +{"description": "Research: finds buttons and fields on screen with vision models. Not needed to automate a task.", "homepage": "https://pypi.org/project/openadapt-grounding/"} +JSON + change 'openadapt-grounding: topics' \ + gh api --method PUT "repos/$ORG/openadapt-grounding/topics" --input - <<'JSON' +{"names": ["computer-vision", "openadapt", "python", "set-of-mark", "ui-detection"]} +JSON + change 'OmniMCP: description and website' \ + gh api --method PATCH "repos/$ORG/OmniMCP" --input - <<'JSON' +{"description": "Labs: an agent experiment that asks a model at every step. Not the OpenAdapt product; see openadapt-flow.", "homepage": "https://omnimcp.openadapt.ai/"} +JSON + change 'OmniMCP: topics' \ + gh api --method PUT "repos/$ORG/OmniMCP/topics" --input - <<'JSON' +{"names": ["anthropic", "aws", "computeruse", "gemini", "generative-ai", "model-context-protocol", "omniparser", "openai", "labs", "research"]} +JSON + change 'SoM: description and website' \ + gh api --method PATCH "repos/$ORG/SoM" --input - <<'JSON' +{"description": "Labs: fork of Microsoft's Set-of-Mark visual prompting research. Not the OpenAdapt product.", "homepage": ""} +JSON + change 'SoM: topics' \ + gh api --method PUT "repos/$ORG/SoM/topics" --input - <<'JSON' +{"names": []} +JSON + change 'PydanticPrompt: description and website' \ + gh api --method PATCH "repos/$ORG/PydanticPrompt" --input - <<'JSON' +{"description": "Labs: a small Python library that documents Pydantic models for LLM prompts. Not the OpenAdapt product.", "homepage": ""} +JSON + change 'PydanticPrompt: topics' \ + gh api --method PUT "repos/$ORG/PydanticPrompt/topics" --input - <<'JSON' +{"names": []} +JSON + change 'OpenSanitizer: description and website' \ + gh api --method PATCH "repos/$ORG/OpenSanitizer" --input - <<'JSON' +{"description": "Superseded by openadapt-privacy, which new projects should use. Older PII and PHI scrubbing for screen data.", "homepage": ""} +JSON + change 'OpenSanitizer: topics' \ + gh api --method PUT "repos/$ORG/OpenSanitizer/topics" --input - <<'JSON' +{"names": []} +JSON +} + +verify_github() { + expect 'organization: description' \ + 'Enters approved information into the systems your team already uses, checks that it saved, and stops to ask a person when something doesn'"'"'t match.' \ + gh api "orgs/$ORG" --jq '.description // ""' + expect 'organization: website' \ + https://openadapt.ai/ \ + gh api "orgs/$ORG" --jq '.blog // ""' + expect 'OpenAdapt: description' \ + 'OpenAdapt takes the last manual step off your team. It enters approved information into EMRs, portals, and desktop or Citrix apps, then checks that it saved.' \ + gh api "repos/$ORG/OpenAdapt" --jq '.description // ""' + expect 'OpenAdapt: website' \ + https://openadapt.ai \ + gh api "repos/$ORG/OpenAdapt" --jq '.homepage // ""' + expect 'OpenAdapt: topics' \ + audit-trail,automation-from-demonstration,browser-automation,citrix,computer-use,desktop-automation,ehr,emr,gui-automation,healthcare-automation,human-in-the-loop,local-first,mcp,process-automation,python,rdp,robotic-process-automation,rpa,ui-automation,workflow-automation \ + gh api "repos/$ORG/OpenAdapt/topics" --jq '.names | sort | join(",")' + expect 'openadapt-flow: description' \ + 'OpenAdapt'"'"'s open-source engine. It turns one recorded task into a program that runs on your computer and never reports an unchecked entry as done.' \ + gh api "repos/$ORG/openadapt-flow" --jq '.description // ""' + expect 'openadapt-flow: website' \ + https://openadapt.ai/platform/flow \ + gh api "repos/$ORG/openadapt-flow" --jq '.homepage // ""' + expect 'openadapt-flow: topics' \ + audit-trail,automation-from-demonstration,browser-automation,citrix,computer-use,demonstration-compiler,desktop-automation,deterministic-replay,effect-verification,emr,gui-automation,healthcare-automation,human-in-the-loop,local-first,python,rdp,robotic-process-automation,rpa,ui-automation,workflow-automation \ + gh api "repos/$ORG/openadapt-flow/topics" --jq '.names | sort | join(",")' + expect 'openadapt-desktop: description' \ + 'The OpenAdapt app. Record a task once, watch it run, see what it entered, and decide what happens when a run stops.' \ + gh api "repos/$ORG/openadapt-desktop" --jq '.description // ""' + expect 'openadapt-desktop: website' \ + https://openadapt.ai/platform/desktop \ + gh api "repos/$ORG/openadapt-desktop" --jq '.homepage // ""' + expect 'openadapt-desktop: topics' \ + automation-from-demonstration,citrix,desktop-automation,gui-automation,human-in-the-loop,local-first,openadapt,python,rpa \ + gh api "repos/$ORG/openadapt-desktop/topics" --jq '.names | sort | join(",")' + expect 'openadapt-capture: description' \ + 'Records screen, mouse, and keyboard on Windows, macOS, and Linux so OpenAdapt can turn one recorded task into an automation.' \ + gh api "repos/$ORG/openadapt-capture" --jq '.description // ""' + expect 'openadapt-capture: website' \ + https://openadapt.ai/platform/capture \ + gh api "repos/$ORG/openadapt-capture" --jq '.homepage // ""' + expect 'openadapt-capture: topics' \ + gui-automation,input-recording,openadapt,python,screen-capture \ + gh api "repos/$ORG/openadapt-capture/topics" --jq '.names | sort | join(",")' + expect 'openadapt-agent: description' \ + 'Lets your AI agent run approved OpenAdapt automations on your computer. Each run reports how it ended, and a run that stops goes to a person.' \ + gh api "repos/$ORG/openadapt-agent" --jq '.description // ""' + expect 'openadapt-agent: website' \ + https://openadapt.ai/platform/agent \ + gh api "repos/$ORG/openadapt-agent" --jq '.homepage // ""' + expect 'openadapt-agent: topics' \ + agent-skills,computer-use,gui-automation,human-in-the-loop,local-first,mcp,openadapt,workflow-automation \ + gh api "repos/$ORG/openadapt-agent/topics" --jq '.names | sort | join(",")' + expect 'ehr-integration-directory: description' \ + 'Shows which EHR tasks have a public API and which still end at a screen. Each entry links its source. Published by OpenAdapt.' \ + gh api "repos/$ORG/ehr-integration-directory" --jq '.description // ""' + expect 'ehr-integration-directory: website' \ + https://ehrintegrationdirectory.com \ + gh api "repos/$ORG/ehr-integration-directory" --jq '.homepage // ""' + expect 'ehr-integration-directory: topics' \ + ehr,emr,fhir,healthcare,healthcare-it,hl7,interoperability,openadapt,smart-on-fhir \ + gh api "repos/$ORG/ehr-integration-directory/topics" --jq '.names | sort | join(",")' + expect 'openadapt-cloud: description' \ + 'Proprietary source for app.openadapt.ai, the hosted service that runs automations, tests them before they go live, and sends stopped runs to a person.' \ + gh api "repos/$ORG/openadapt-cloud" --jq '.description // ""' + expect 'openadapt-cloud: website' \ + https://app.openadapt.ai \ + gh api "repos/$ORG/openadapt-cloud" --jq '.homepage // ""' + expect 'openadapt-cloud: topics' \ + '' \ + gh api "repos/$ORG/openadapt-cloud/topics" --jq '.names | sort | join(",")' + expect '.github: description' \ + 'Support: OpenAdapt'"'"'s GitHub profile, the status of every repository, and shared community files.' \ + gh api "repos/$ORG/.github" --jq '.description // ""' + expect '.github: website' \ + https://openadapt.ai \ + gh api "repos/$ORG/.github" --jq '.homepage // ""' + expect '.github: topics' \ + openadapt \ + gh api "repos/$ORG/.github/topics" --jq '.names | sort | join(",")' + expect 'openadapt-web: description' \ + 'Source for openadapt.ai, including product pages, pricing, and evidence.' \ + gh api "repos/$ORG/openadapt-web" --jq '.description // ""' + expect 'openadapt-web: website' \ + https://openadapt.ai \ + gh api "repos/$ORG/openadapt-web" --jq '.homepage // ""' + expect 'openadapt-web: topics' \ + '' \ + gh api "repos/$ORG/openadapt-web/topics" --jq '.names | sort | join(",")' + expect 'openadapt-ops: description' \ + 'Support: source for docs.openadapt.ai, plus the checks that keep the docs in step with each release.' \ + gh api "repos/$ORG/openadapt-ops" --jq '.description // ""' + expect 'openadapt-ops: website' \ + https://docs.openadapt.ai \ + gh api "repos/$ORG/openadapt-ops" --jq '.homepage // ""' + expect 'openadapt-ops: topics' \ + openadapt \ + gh api "repos/$ORG/openadapt-ops/topics" --jq '.names | sort | join(",")' + expect 'openadapt-blog: description' \ + 'Support: source for blog.openadapt.ai, with guides, product updates, and test write-ups.' \ + gh api "repos/$ORG/openadapt-blog" --jq '.description // ""' + expect 'openadapt-blog: website' \ + https://blog.openadapt.ai \ + gh api "repos/$ORG/openadapt-blog" --jq '.homepage // ""' + expect 'openadapt-blog: topics' \ + openadapt \ + gh api "repos/$ORG/openadapt-blog/topics" --jq '.names | sort | join(",")' + expect 'openadapt-wright: description' \ + 'Support: team tool that drafts code changes and pull requests for a person to approve. Not part of the product.' \ + gh api "repos/$ORG/openadapt-wright" --jq '.description // ""' + expect 'openadapt-wright: website' \ + '' \ + gh api "repos/$ORG/openadapt-wright" --jq '.homepage // ""' + expect 'openadapt-wright: topics' \ + '' \ + gh api "repos/$ORG/openadapt-wright/topics" --jq '.names | sort | join(",")' + expect 'openadapt-herald: description' \ + 'Support: team tool that drafts release announcements from git history. Not part of the product.' \ + gh api "repos/$ORG/openadapt-herald" --jq '.description // ""' + expect 'openadapt-herald: website' \ + '' \ + gh api "repos/$ORG/openadapt-herald" --jq '.homepage // ""' + expect 'openadapt-herald: topics' \ + '' \ + gh api "repos/$ORG/openadapt-herald/topics" --jq '.names | sort | join(",")' + expect 'openadapt-crier: description' \ + 'Support: team tool that sends drafted social posts to Telegram for a person to approve. Not part of the product.' \ + gh api "repos/$ORG/openadapt-crier" --jq '.description // ""' + expect 'openadapt-crier: website' \ + '' \ + gh api "repos/$ORG/openadapt-crier" --jq '.homepage // ""' + expect 'openadapt-crier: topics' \ + '' \ + gh api "repos/$ORG/openadapt-crier/topics" --jq '.names | sort | join(",")' + expect 'openadapt-consilium: description' \ + 'Support: team tool that asks several AI models one question and combines their answers. Not part of the product.' \ + gh api "repos/$ORG/openadapt-consilium" --jq '.description // ""' + expect 'openadapt-consilium: website' \ + '' \ + gh api "repos/$ORG/openadapt-consilium" --jq '.homepage // ""' + expect 'openadapt-consilium: topics' \ + '' \ + gh api "repos/$ORG/openadapt-consilium/topics" --jq '.names | sort | join(",")' + expect 'openadapt-telemetry: description' \ + 'Support: error reports and usage counts for OpenAdapt packages, with personal details scrubbed. Set DO_NOT_TRACK=1 to opt out.' \ + gh api "repos/$ORG/openadapt-telemetry" --jq '.description // ""' + expect 'openadapt-telemetry: website' \ + '' \ + gh api "repos/$ORG/openadapt-telemetry" --jq '.homepage // ""' + expect 'openadapt-telemetry: topics' \ + '' \ + gh api "repos/$ORG/openadapt-telemetry/topics" --jq '.names | sort | join(",")' + expect 'openadapt-viewer: description' \ + 'Support: turns test results and screenshots into one HTML page you can share. Not needed to automate a task.' \ + gh api "repos/$ORG/openadapt-viewer" --jq '.description // ""' + expect 'openadapt-viewer: website' \ + https://pypi.org/project/openadapt-viewer/ \ + gh api "repos/$ORG/openadapt-viewer" --jq '.homepage // ""' + expect 'openadapt-viewer: topics' \ + html,openadapt,python,visualization \ + gh api "repos/$ORG/openadapt-viewer/topics" --jq '.names | sort | join(",")' + expect 'openadapt-privacy: description' \ + 'Experimental: detects and redacts personal and health information in recordings and screenshots before they'"'"'re shared.' \ + gh api "repos/$ORG/openadapt-privacy" --jq '.description // ""' + expect 'openadapt-privacy: website' \ + https://pypi.org/project/openadapt-privacy/ \ + gh api "repos/$ORG/openadapt-privacy" --jq '.homepage // ""' + expect 'openadapt-privacy: topics' \ + openadapt,pii-detection,privacy,python,scrubbing \ + gh api "repos/$ORG/openadapt-privacy/topics" --jq '.names | sort | join(",")' + expect 'openadapt-types: description' \ + 'Experimental: shared data formats for screens and actions inside OpenAdapt. Not needed to automate a task.' \ + gh api "repos/$ORG/openadapt-types" --jq '.description // ""' + expect 'openadapt-types: website' \ + '' \ + gh api "repos/$ORG/openadapt-types" --jq '.homepage // ""' + expect 'openadapt-types: topics' \ + openadapt \ + gh api "repos/$ORG/openadapt-types/topics" --jq '.names | sort | join(",")' + expect 'openadapt-console: description' \ + 'Experimental: an early web console prototype. Not part of the product today.' \ + gh api "repos/$ORG/openadapt-console" --jq '.description // ""' + expect 'openadapt-console: website' \ + '' \ + gh api "repos/$ORG/openadapt-console" --jq '.homepage // ""' + expect 'openadapt-console: topics' \ + '' \ + gh api "repos/$ORG/openadapt-console/topics" --jq '.names | sort | join(",")' + expect 'openadapt-tray: description' \ + 'Experimental: a tray icon that shows what OpenAdapt is doing, with start and stop controls. Optional.' \ + gh api "repos/$ORG/openadapt-tray" --jq '.description // ""' + expect 'openadapt-tray: website' \ + '' \ + gh api "repos/$ORG/openadapt-tray" --jq '.homepage // ""' + expect 'openadapt-tray: topics' \ + openadapt \ + gh api "repos/$ORG/openadapt-tray/topics" --jq '.names | sort | join(",")' + expect 'openadapt-ml: description' \ + 'Research: optional model training for computer-use agents. Not needed to record, build, or run an OpenAdapt automation.' \ + gh api "repos/$ORG/openadapt-ml" --jq '.description // ""' + expect 'openadapt-ml: website' \ + https://pypi.org/project/openadapt-ml/ \ + gh api "repos/$ORG/openadapt-ml" --jq '.homepage // ""' + expect 'openadapt-ml: topics' \ + computer-use,gui-automation,machine-learning,openadapt,python,research,vlm \ + gh api "repos/$ORG/openadapt-ml/topics" --jq '.names | sort | join(",")' + expect 'openadapt-evals: description' \ + 'Research: scores computer-use agents and OpenAdapt programs on public benchmarks, including Windows Agent Arena. Not needed to automate a task.' \ + gh api "repos/$ORG/openadapt-evals" --jq '.description // ""' + expect 'openadapt-evals: website' \ + https://openadapt.ai/platform/evals \ + gh api "repos/$ORG/openadapt-evals" --jq '.homepage // ""' + expect 'openadapt-evals: topics' \ + benchmarks,evaluation,gui-automation,openadapt,python \ + gh api "repos/$ORG/openadapt-evals/topics" --jq '.names | sort | join(",")' + expect 'openadapt-retrieval: description' \ + 'Research: searches past recordings by image and text for model-driven agents. Not needed to automate a task.' \ + gh api "repos/$ORG/openadapt-retrieval" --jq '.description // ""' + expect 'openadapt-retrieval: website' \ + https://pypi.org/project/openadapt-retrieval/ \ + gh api "repos/$ORG/openadapt-retrieval" --jq '.homepage // ""' + expect 'openadapt-retrieval: topics' \ + demo-retrieval,embeddings,openadapt,python,retrieval \ + gh api "repos/$ORG/openadapt-retrieval/topics" --jq '.names | sort | join(",")' + expect 'openadapt-grounding: description' \ + 'Research: finds buttons and fields on screen with vision models. Not needed to automate a task.' \ + gh api "repos/$ORG/openadapt-grounding" --jq '.description // ""' + expect 'openadapt-grounding: website' \ + https://pypi.org/project/openadapt-grounding/ \ + gh api "repos/$ORG/openadapt-grounding" --jq '.homepage // ""' + expect 'openadapt-grounding: topics' \ + computer-vision,openadapt,python,set-of-mark,ui-detection \ + gh api "repos/$ORG/openadapt-grounding/topics" --jq '.names | sort | join(",")' + expect 'OmniMCP: description' \ + 'Labs: an agent experiment that asks a model at every step. Not the OpenAdapt product; see openadapt-flow.' \ + gh api "repos/$ORG/OmniMCP" --jq '.description // ""' + expect 'OmniMCP: website' \ + https://omnimcp.openadapt.ai/ \ + gh api "repos/$ORG/OmniMCP" --jq '.homepage // ""' + expect 'OmniMCP: topics' \ + anthropic,aws,computeruse,gemini,generative-ai,labs,model-context-protocol,omniparser,openai,research \ + gh api "repos/$ORG/OmniMCP/topics" --jq '.names | sort | join(",")' + expect 'SoM: description' \ + 'Labs: fork of Microsoft'"'"'s Set-of-Mark visual prompting research. Not the OpenAdapt product.' \ + gh api "repos/$ORG/SoM" --jq '.description // ""' + expect 'SoM: website' \ + '' \ + gh api "repos/$ORG/SoM" --jq '.homepage // ""' + expect 'SoM: topics' \ + '' \ + gh api "repos/$ORG/SoM/topics" --jq '.names | sort | join(",")' + expect 'PydanticPrompt: description' \ + 'Labs: a small Python library that documents Pydantic models for LLM prompts. Not the OpenAdapt product.' \ + gh api "repos/$ORG/PydanticPrompt" --jq '.description // ""' + expect 'PydanticPrompt: website' \ + '' \ + gh api "repos/$ORG/PydanticPrompt" --jq '.homepage // ""' + expect 'PydanticPrompt: topics' \ + '' \ + gh api "repos/$ORG/PydanticPrompt/topics" --jq '.names | sort | join(",")' + expect 'OpenSanitizer: description' \ + 'Superseded by openadapt-privacy, which new projects should use. Older PII and PHI scrubbing for screen data.' \ + gh api "repos/$ORG/OpenSanitizer" --jq '.description // ""' + expect 'OpenSanitizer: website' \ + '' \ + gh api "repos/$ORG/OpenSanitizer" --jq '.homepage // ""' + expect 'OpenSanitizer: topics' \ + '' \ + gh api "repos/$ORG/OpenSanitizer/topics" --jq '.names | sort | join(",")' + check_pins +} + +if [ "$MODE" = dry-run ]; then + apply_changes + printf '\nDry run: %s changes listed. Nothing was called or changed.\n' "$CHANGES" + printf 'Run with --apply to make them, then pin the repositories by hand.\n' + exit 0 +fi + +preflight +if [ "$MODE" = apply ]; then + apply_changes + printf '\n' +fi +verify_github +note_unregistered + +if [ "$FAILED" -gt 0 ]; then + printf '\n%s of %s changes failed. GitHub may be partly updated; the checks above show what differs.\n' \ + "$FAILED" "$CHANGES" >&2 + exit 1 +fi +if [ "$DIFFERENT" -gt 0 ]; then + printf '\n%s of %s values on GitHub differ from the registry or could not be read.\n' \ + "$DIFFERENT" "$CHECKED" >&2 + exit 1 +fi +if [ "$PINS_PENDING" -gt 0 ]; then + printf '\nEvery other value matches. Pin the repositories by hand, in this order: %s\n' \ + "$PINS" >&2 + exit 2 +fi +printf '\nGitHub matches repository-lifecycle.yml. All %s values checked.\n' "$CHECKED" +exit 0 diff --git a/README.md b/README.md index 3a809e4..f9e5f6d 100644 --- a/README.md +++ b/README.md @@ -1,33 +1,48 @@ -# OpenAdaptAI Organization Configuration +# OpenAdaptAI organization configuration -This repository owns the public organization profile, lifecycle registry, and -the intended GitHub metadata for the repositories presented as the product. It -does not change GitHub organization settings automatically. +This repository holds the public organization profile, the repository +lifecycle registry, and the GitHub settings that visitors see for each +repository. The profile in [`profile/README.md`](profile/README.md) goes live +when a change merges. The GitHub settings don't: an organization owner applies +them. The durable cross-repository launch acceptance contract is maintained in [LAUNCH_PLAN.md](LAUNCH_PLAN.md). Current execution state belongs in the workspace `STATUS.md`, not in this public repository. -## Manual GitHub Actions - -Organization owners should apply these settings after this branch is merged: - -1. Set the organization description to the `organization_description` value in - [`repository-lifecycle.yml`](repository-lifecycle.yml). -2. Apply the exact `repository_descriptions` values from that file. They keep - the Desktop, native Capture, and agent-bridge descriptions aligned with the - lifecycle registry. -3. Pin the exact `pinned_repositories` list so visitors see the product - surface: the flagship installer, the Flow engine, Desktop, Capture, Agent, - and evals. -4. The Cloud implementation repository is private and cannot be a public - organization pin. The documentation implementation is a public Support - repository and remains reachable through the docs link instead of occupying - a product pin. -5. Apply the archive queue in - [REPOSITORY_LIFECYCLE.md](REPOSITORY_LIFECYCLE.md) only after each repository - has an archive notice and any dirty local work is preserved. - -Organization descriptions, repository descriptions, and pins are GitHub -settings. Editing `profile/README.md` does not change them; an organization -owner must apply the machine-readable values after merge. +## Apply the GitHub settings after a merge + +[`repository-lifecycle.yml`](repository-lifecycle.yml) is the source of truth +for the organization description and website, the profile headline, the +pinned repositories, and each listed repository's description, website, and +topics. [`scripts/check_profile.py`](scripts/check_profile.py) checks these +values on every pull request. + +An organization owner applies them: + +1. Review the exact commands. Run `.overhaul/APPLY.sh`. This dry run prints + each `gh` command and its request body, and it doesn't call GitHub. +2. Apply them. Run `.overhaul/APPLY.sh --apply`. It needs `gh` signed in as an + owner of the organization, with the `admin:org` and `repo` scopes. It + changes nothing if the account isn't an owner. After the changes, it reads + every value back from GitHub and lists each one that differs. +3. Pin the repositories by hand, in the order that `pinned_repositories` + lists them. GitHub has no API for organization pins. +4. Confirm. Run `.overhaul/APPLY.sh --verify`. It exits 0 only when every + value on GitHub matches the registry, and 2 when only the pins are left. + +After you edit the registry, render the script again. The profile check fails +until the script matches the registry. + +```bash +python3 scripts/github_metadata.py render-apply --output .overhaul/APPLY.sh +``` + +The Cloud source repository is private, so it can't be a public pin. The +documentation source, `openadapt-ops`, is a public Support repository that the +profile links instead of pinning. Each repository's social preview image is +also set by hand, in that repository's settings. + +Apply the archive queue in [REPOSITORY_LIFECYCLE.md](REPOSITORY_LIFECYCLE.md) +only after each repository has an archive notice and any uncommitted local work +is preserved. diff --git a/REPOSITORY_LIFECYCLE.md b/REPOSITORY_LIFECYCLE.md index 0e8cc90..87c0feb 100644 --- a/REPOSITORY_LIFECYCLE.md +++ b/REPOSITORY_LIFECYCLE.md @@ -1,6 +1,6 @@ -# OpenAdapt Repository Lifecycle Registry +# OpenAdapt repository lifecycle registry -Last reviewed: 2026-09-02 +Last reviewed: 2026-10-09 This public registry separates the product from experiments and records the intended lifecycle of organization repositories. It does not authorize moving @@ -9,8 +9,12 @@ checkout state and credential-response details belong in private operations records, not in this public repository. The machine-readable source is [`repository-lifecycle.yml`](repository-lifecycle.yml). +It also holds each listed repository's public GitHub description, website, +and topics, and the organization pins. Support repositories include the +public [EHR Integration Directory](https://ehrintegrationdirectory.com), +which OpenAdapt publishes. -## Lifecycle Definitions +## Lifecycle definitions | Status | Meaning | |--------|---------| @@ -26,7 +30,7 @@ The machine-readable source is [`repository-lifecycle.yml`](repository-lifecycle | **Deprecated** | Superseded; migration fixes only, no new integrations | | **Archived** | Historical and read-only | -## Production Admission +## Production admission Production is a derived per-release state. It is not a static repository label. A person cannot create it by changing a table or repository description. The canonical @@ -83,7 +87,7 @@ Static Production membership is forbidden. Consumers derive current Production at read time from the signed admission, its revocation state, and until-revoked validity. -## Admission-Gated Targets +## Admission-gated targets The seven product targets do not have fallback lifecycle labels. A target that doesn't have a current admission is **not actively admitted**. Expiry, @@ -110,11 +114,11 @@ These are the derived states: | `agent` | **Production** | Local MCP and Agent Skills bridge for governed Flow workflows | | `docs` | **Production** | `docs.openadapt.ai` deployment sourced from `openadapt-ops` | -## Other Repository Lifecycles +## Other repository lifecycles | Group | Repositories | |-------|--------------| -| **Support** | `.github`, `openadapt-web`, `openadapt-ops`, `openadapt-wright`, `openadapt-herald`, `openadapt-crier`, `openadapt-consilium`, `openadapt-telemetry`, `openadapt-viewer`, `openadapt-blog` | +| **Support** | `.github`, `openadapt-web`, `openadapt-ops`, `openadapt-wright`, `openadapt-herald`, `openadapt-crier`, `openadapt-consilium`, `openadapt-telemetry`, `openadapt-viewer`, `openadapt-blog`, `ehr-integration-directory` | | **Experimental** | `openadapt-privacy`, `openadapt-types`, `openadapt-console`, `openadapt-tray` | | **Research** | `openadapt-ml`, `openadapt-evals`, `openadapt-retrieval`, `openadapt-grounding`, `openadapt-verifier` | | **Internal** | `openadapt-bootstrap`, `openadapt-internal`, `openadapt-yc`, `openadapt-presenter` | @@ -122,7 +126,7 @@ These are the derived states: | **Archived historical directions** | `OpenAdapter`, `OpenReflector` | | **Superseded** | `OpenSanitizer` (successor: `openadapt-privacy`) | -## Retirement Queue +## Retirement queue | Repository | Lifecycle | Public action | |------------|-----------|---------------| @@ -136,7 +140,7 @@ Experimental, Research, Labs, and Internal repositories are not deprecated by default. Moving local checkouts is a separate operational decision that must use private, current evidence. -## Archive Procedure +## Archive procedure 1. Preserve or intentionally discard every tracked and untracked local change. 2. Confirm the branch is pushed and record the final commit in private diff --git a/benchmark-claims.json b/benchmark-claims.json index a589363..4b62965 100644 --- a/benchmark-claims.json +++ b/benchmark-claims.json @@ -45,246 +45,176 @@ ], "claims": [ { - "id": "openemr-compiled-success-ratio", + "id": "effect-wrong-effect-count", "surface": "profile/README.md", - "context": "compiled replay went **19/20 at 39.2s p50 with zero model calls**", - "text": "19/20", - "renderer": "ratio", + "context": "In 72 of 90 runs, the record ended up wrong", + "text": "72 of 90", + "renderer": "count_of", "decimals": null, - "source": "openemr", - "pointers": [ - "/arms/compiled/success_count", - "/arms/compiled/n" - ], - "status": "bound", - "drift": null - }, - { - "id": "openemr-compiled-p50", - "surface": "profile/README.md", - "context": "compiled replay went **19/20 at 39.2s p50 with zero model calls**", - "text": "39.2s", - "renderer": "seconds", - "decimals": 1, - "source": "openemr", + "source": "effect-e2e", "pointers": [ - "/arms/compiled/wall_s_p50" + "/metrics/per_arm/screen/n_wrong_effect", + "/metrics/per_arm/screen/n_runs" ], "status": "bound", "drift": null }, { - "id": "openemr-agent-success-ratio", + "id": "effect-screen-silent-wrong-count", "surface": "profile/README.md", - "context": "the agent went 10/10 at 70.4s p50 at about $0.55 of model charge per run.", - "text": "10/10", - "renderer": "ratio", + "context": "| Trust the app's success banner | 54 of 72 (75.0%) |", + "text": "54 of 72", + "renderer": "count_of", "decimals": null, - "source": "openemr", + "source": "effect-e2e", "pointers": [ - "/arms/agent/success_count", - "/arms/agent/n" + "/metrics/per_arm/screen/silent_wrong_count", + "/metrics/per_arm/screen/n_wrong_effect" ], "status": "bound", "drift": null }, { - "id": "openemr-agent-p50", + "id": "effect-screen-undetected-rate", "surface": "profile/README.md", - "context": "the agent went 10/10 at 70.4s p50 at about $0.55 of model charge per run.", - "text": "70.4s", - "renderer": "seconds", + "context": "| Trust the app's success banner | 54 of 72 (75.0%) |", + "text": "75.0%", + "renderer": "percent", "decimals": 1, - "source": "openemr", - "pointers": [ - "/arms/agent/wall_s_p50" - ], - "status": "bound", - "drift": null - }, - { - "id": "openemr-agent-cost-per-run", - "surface": "profile/README.md", - "context": "the agent went 10/10 at 70.4s p50 at about $0.55 of model charge per run.", - "text": "$0.55", - "renderer": "usd", - "decimals": 2, - "source": "openemr", - "pointers": [ - "/arms/agent/cost_usd_per_run" - ], - "status": "bound", - "drift": null - }, - { - "id": "mockmed-compiled-success-ratio", - "surface": "profile/README.md", - "context": "retained rows marked 100/100 compiled runs and 20/20 agent runs successful under the 2026-07-08 OCR check.", - "text": "100/100", - "renderer": "ratio", - "decimals": null, - "source": "mockmed", + "source": "effect-e2e", "pointers": [ - "/arms/compiled/success_count", - "/arms/compiled/n" + "/metrics/per_arm/screen/undetected_wrong_rate" ], "status": "bound", "drift": null }, { - "id": "mockmed-agent-success-ratio", + "id": "effect-rest-silent-wrong-count", "surface": "profile/README.md", - "context": "retained rows marked 100/100 compiled runs and 20/20 agent runs successful under the 2026-07-08 OCR check.", - "text": "20/20", - "renderer": "ratio", + "context": "| One separate read of the record | 9 of 72 (12.5%) |", + "text": "9 of 72", + "renderer": "count_of", "decimals": null, - "source": "mockmed", - "pointers": [ - "/arms/agent/success_count", - "/arms/agent/n" - ], - "status": "bound", - "drift": null - }, - { - "id": "mockmed-compiled-p50", - "surface": "profile/README.md", - "context": "The compiled arm recorded 4.9s p50 and $0 per run in model API charges;", - "text": "4.9s", - "renderer": "seconds", - "decimals": 1, - "source": "mockmed", - "pointers": [ - "/arms/compiled/wall_s_p50" - ], - "status": "bound", - "drift": null - }, - { - "id": "mockmed-compiled-cost-per-run", - "surface": "profile/README.md", - "context": "The compiled arm recorded 4.9s p50 and $0 per run in model API charges;", - "text": "$0", - "renderer": "usd", - "decimals": 0, - "source": "mockmed", + "source": "effect-e2e", "pointers": [ - "/arms/compiled/cost_usd_per_run" + "/metrics/per_arm/effect_rest/silent_wrong_count", + "/metrics/per_arm/effect_rest/n_wrong_effect" ], "status": "bound", "drift": null }, { - "id": "mockmed-agent-p50", + "id": "effect-rest-undetected-rate", "surface": "profile/README.md", - "context": "the agent arm recorded 37.5s p50 and about $0.27 per run.", - "text": "37.5s", - "renderer": "seconds", + "context": "| One separate read of the record | 9 of 72 (12.5%) |", + "text": "12.5%", + "renderer": "percent", "decimals": 1, - "source": "mockmed", + "source": "effect-e2e", "pointers": [ - "/arms/agent/wall_s_p50" + "/metrics/per_arm/effect_rest/undetected_wrong_rate" ], "status": "bound", "drift": null }, { - "id": "mockmed-agent-cost-per-run", + "id": "effect-full-silent-wrong-count", "surface": "profile/README.md", - "context": "the agent arm recorded 37.5s p50 and about $0.27 per run.", - "text": "$0.27", - "renderer": "usd", - "decimals": 2, - "source": "mockmed", + "context": "| A read of every table in the test database | 0 of 72 |", + "text": "0 of 72", + "renderer": "count_of", + "decimals": null, + "source": "effect-e2e", "pointers": [ - "/arms/agent/cost_usd_per_run" + "/metrics/per_arm/effect_full/silent_wrong_count", + "/metrics/per_arm/effect_full/n_wrong_effect" ], "status": "bound", "drift": null }, { - "id": "effect-screen-undetected-rate", + "id": "effect-false-abort-count", "surface": "profile/README.md", - "context": "oracle silently accepted **75.0%** of the wrong effects that actually occurred (54 of 90 runs).", - "text": "75.0%", - "renderer": "percent", - "decimals": 1, + "context": "Every check also stopped 9 of 18 runs that had saved correctly", + "text": "9 of 18", + "renderer": "count_of", + "decimals": null, "source": "effect-e2e", "pointers": [ - "/metrics/per_arm/screen/undetected_wrong_rate" + "/metrics/per_arm/screen/false_abort_count", + "/metrics/per_arm/screen/n_correct_effect" ], "status": "bound", "drift": null }, { - "id": "effect-screen-silent-wrong-count", + "id": "openemr-compiled-success-ratio", "surface": "profile/README.md", - "context": "oracle silently accepted **75.0%** of the wrong effects that actually occurred (54 of 90 runs).", - "text": "54 of 90", - "renderer": "count_of", + "context": "compiled replay finished 19/20 runs at 39.2s median with no model calls.", + "text": "19/20", + "renderer": "ratio", "decimals": null, - "source": "effect-e2e", + "source": "openemr", "pointers": [ - "/metrics/per_arm/screen/silent_wrong_count", - "/metrics/per_arm/screen/n_runs" + "/arms/compiled/success_count", + "/arms/compiled/n" ], "status": "bound", "drift": null }, { - "id": "effect-rest-undetected-rate", + "id": "openemr-compiled-p50", "surface": "profile/README.md", - "context": "cut that to **12.5%** (9 of 90).", - "text": "12.5%", - "renderer": "percent", + "context": "compiled replay finished 19/20 runs at 39.2s median with no model calls.", + "text": "39.2s", + "renderer": "seconds", "decimals": 1, - "source": "effect-e2e", + "source": "openemr", "pointers": [ - "/metrics/per_arm/effect_rest/undetected_wrong_rate" + "/arms/compiled/wall_s_p50" ], "status": "bound", "drift": null }, { - "id": "effect-rest-silent-wrong-count", + "id": "openemr-agent-success-ratio", "surface": "profile/README.md", - "context": "cut that to **12.5%** (9 of 90).", - "text": "9 of 90", - "renderer": "count_of", + "context": "A computer-use agent finished 10/10 runs at 70.4s median, at about $0.55 a run at the July 2026 list price.", + "text": "10/10", + "renderer": "ratio", "decimals": null, - "source": "effect-e2e", + "source": "openemr", "pointers": [ - "/metrics/per_arm/effect_rest/silent_wrong_count", - "/metrics/per_arm/effect_rest/n_runs" + "/arms/agent/success_count", + "/arms/agent/n" ], "status": "bound", "drift": null }, { - "id": "effect-full-silent-wrong-count", + "id": "openemr-agent-p50", "surface": "profile/README.md", - "context": "A complete read path over every mutable surface reaches 0 of 90,", - "text": "0 of 90", - "renderer": "count_of", - "decimals": null, - "source": "effect-e2e", + "context": "A computer-use agent finished 10/10 runs at 70.4s median, at about $0.55 a run at the July 2026 list price.", + "text": "70.4s", + "renderer": "seconds", + "decimals": 1, + "source": "openemr", "pointers": [ - "/metrics/per_arm/effect_full/silent_wrong_count", - "/metrics/per_arm/effect_full/n_runs" + "/arms/agent/wall_s_p50" ], "status": "bound", "drift": null }, { - "id": "effect-full-undetected-rate", + "id": "openemr-agent-cost-per-run", "surface": "profile/README.md", - "context": "one out-of-band oracle — not the 0%. All nine", - "text": "0%", - "renderer": "percent", - "decimals": 0, - "source": "effect-e2e", + "context": "A computer-use agent finished 10/10 runs at 70.4s median, at about $0.55 a run at the July 2026 list price.", + "text": "$0.55", + "renderer": "usd", + "decimals": 2, + "source": "openemr", "pointers": [ - "/metrics/per_arm/effect_full/undetected_wrong_rate" + "/arms/agent/cost_usd_per_run" ], "status": "bound", "drift": null diff --git a/profile/README.md b/profile/README.md index 54d1592..affa29c 100644 --- a/profile/README.md +++ b/profile/README.md @@ -1,6 +1,101 @@ # OpenAdapt -**Automate the UI-only work your APIs can't reach.** +**OpenAdapt enters approved information into the systems your team already uses, and then it checks that the entry saved.** + +It does this last manual step for your team in EMRs, payer portals, and +desktop apps, including apps your staff reach through Citrix or remote +desktop. When something doesn't match, it stops and asks a person instead of +guessing. + +[Watch the live demo](https://app.openadapt.ai/demo) · +[Discuss your workflow](https://openadapt.ai/book) · +[Run it yourself](#run-it-yourself) · +[Visit openadapt.ai](https://openadapt.ai/) + +The live demo plays recorded runs that use synthetic data. + +## What changes for your team + +Take a faxed referral that staff enter into the EMR. + +| | Today | With OpenAdapt | +|---|---|---| +| Entering it | A staff member keys each referral by hand. Two hospital time studies put one faxed referral at about 10 to 12 minutes of staff time. | OpenAdapt enters the approved fields in the same EMR screens your staff use. | +| Knowing it saved | Someone sees "Saved" and moves on. | OpenAdapt reads the saved record back and compares it with what it entered. | +| When something's off | It turns up later, at booking or billing. | The run stops before it guesses, and a person decides what happens next. | +| Your staff's part | Type every referral. | Review the runs that OpenAdapt stops on. | + +The time studies are from +[UCSF, JAMIA Open 2020](https://pmc.ncbi.nlm.nih.gov/articles/PMC7660949/) +(its mean includes a call to the patient) and +[Calderdale and Huddersfield NHS Foundation Trust, HFMA 2024](https://www.hfma.org.uk/system/files/2024-04/Using%20digital%20technologies%20to%20process%20admin%20tasks%20Case%20Study%20v7.pdf). +Nobody has measured an "after" time with OpenAdapt yet, so this table doesn't +claim a time saving. + +OpenAdapt doesn't need an EMR integration. The record check needs a second way +to read the record, such as a report, an API, or a read-only login. To see +which EHR tasks have a public API and which still end at a screen, browse the +[EHR Integration Directory](https://ehrintegrationdirectory.com) that we +publish ([source](https://github.com/OpenAdaptAI/ehr-integration-directory)). +An EHR listed there isn't an OpenAdapt partner. + +## How a run ends + +Every run ends with one of these results, and every run leaves a report. + +| Result | What it means | +|---|---| +| Done and checked | It saved the entry and read the record back to confirm it. | +| Stopped before saving | Something didn't match, so it stopped before it changed anything and asked a person. Nothing was written. | +| Check the record | A save may have gone through. A person checks the record before anything is retried. | +| Finished, not checked | It did the steps but didn't confirm the saved result, as in a practice run. | +| Didn't finish | It couldn't complete the task. Its report says what's known about whether anything was written. | + +It runs on a computer you control, or on one we host. A routine run makes no +AI model calls. + +## Two ways to start + +Hand the workflow to us, or run the open-source engine yourself. + +### Hand off a workflow + +Tell us about a task your team repeats: +[discuss your workflow](https://openadapt.ai/book). + +1. You show us the task once, on your own system. +2. We build the automation and test it with your own cases before it goes + live. +3. We keep it working. When a screen changes, we propose a fix, and a person + on your team approves it. + +We scope [services](https://openadapt.ai/services) per workflow. If your own +AI product decides what to enter, OpenAdapt can make that final entry and +return the result to your product. See +[OpenAdapt Execute](https://openadapt.ai/execute). + +### Run it yourself + +The engine is free and open source under the MIT license. With Python 3.10, +3.11, or 3.12, run: + +```bash +python -m pip install --upgrade openadapt +openadapt quickstart --break-it +``` + +This runs a synthetic clinic task twice on your computer. The first run is +done and checked. In the second run, the test server rejects the save after +the app shows that it saved. The record check finds no saved record, so the +run ends with "Check the record" instead of reporting success. On Python 3.13 +or newer, install with `uv tool install --python 3.12 openadapt` instead. + +To record your own task, follow the +[walkthrough](https://docs.openadapt.ai/get-started/). To have us run the +computer for browser workflows, see the hosted plans on the +[pricing page](https://openadapt.ai/pricing). + +## For developers OpenAdapt compiles demonstrations into governed workflows across browser, native desktop, RDP, and Citrix. The default healthy path executes @@ -9,117 +104,104 @@ actions are identity-gated, results are checked against the workflow's evidence contract, and uncertainty halts for review instead of being reported as success. -OpenAdapt is for repeated work trapped behind browser, desktop, and -virtual-desktop interfaces: too visual or variable for brittle selectors, but -too consequential to hand to a free-form computer-use agent on every run. +Each result in [How a run ends](#how-a-run-ends) comes from the engine's +outcome. Both kinds of stop report `HALTED`, so read the transaction outcome +before you tell anyone whether something was written. -[Install OpenAdapt](https://github.com/OpenAdaptAI/OpenAdapt) · -[Watch the live demo](https://app.openadapt.ai/demo) · -[Read the docs](https://docs.openadapt.ai) · -[Visit openadapt.ai](https://openadapt.ai/) +| Result | Engine outcome | +|---|---| +| Done and checked | `VERIFIED` | +| Stopped before saving | `HALTED` with transaction outcome `HALTED_BEFORE_EFFECT` | +| Check the record | `HALTED` with transaction outcome `RECONCILIATION_REQUIRED`, or any uncertain delivery | +| Finished, not checked | `COMPLETED_UNVERIFIED` | +| Didn't finish | `FAILED` | -The installer is [`OpenAdapt`](https://github.com/OpenAdaptAI/OpenAdapt). -The canonical engine is -[`openadapt-flow`](https://github.com/OpenAdaptAI/openadapt-flow). +A caller must report a halt as a halt, never as success. -## Start Locally +## Product surfaces -```bash -python -m pip install --upgrade 'openadapt[browser]' -openadapt quickstart -``` +| Target | Role | +|---|---| +| `openadapt` | [`OpenAdapt`](https://github.com/OpenAdaptAI/OpenAdapt) installs the `openadapt` command and the engine. | +| `flow` | [`openadapt-flow`](https://github.com/OpenAdaptAI/openadapt-flow) is the compiler and governed runtime. Engine changes go here. | +| `cloud` | [`app.openadapt.ai`](https://app.openadapt.ai/) runs hosted browser workflows and manages runs on computers you control. Its source is private. | +| `desktop` | [`openadapt-desktop`](https://github.com/OpenAdaptAI/openadapt-desktop) records, tests, runs, and reviews workflows on your own computer, and handles approved fixes. | +| `capture` | [`openadapt-capture`](https://github.com/OpenAdaptAI/openadapt-capture) records screen, input, timing, and window evidence for Desktop and Flow. | +| `agent` | [`openadapt-agent`](https://github.com/OpenAdaptAI/openadapt-agent) gives agents governed Flow workflows as local MCP tools and Agent Skills. | +| `docs` | [`docs.openadapt.ai`](https://docs.openadapt.ai) is the documentation site. The Support repository [`openadapt-ops`](https://github.com/OpenAdaptAI/openadapt-ops) publishes it. | -This runs the bundled MockMed lifecycle. It records, compiles, certifies, -replays, independently verifies the synthetic effect, and writes a -human-readable `REPORT.md`. Use the -[five-minute guide](https://docs.openadapt.ai/get-started/) to add lint, drift, -repair, and deployment. +These targets form one product across browser, Windows, macOS, Linux, RDP, and +Citrix or other virtual desktops. Qualification is per workflow. A product +release admission doesn't qualify any workflow. Current signed admissions are +in the [live record](https://docs.openadapt.ai/production-lifecycle.json). ## Evidence -Published head-to-head comparisons, each graded by an external success check -that is independent of every arm: - -- **Live third-party EMR** (OpenEMR public demo, fake patients, 18-step - add-patient-note workflow): compiled replay went **19/20 at 39.2s p50 with - zero model calls**; the agent went 10/10 at 70.4s p50 at about $0.55 of model - charge per run. Compiled run 20 didn't pass. The saved-row oracle, tightened - on 2026-07-28, refuses to count a note still sitting in the unsaved entry - form, and the replayer had already halted at step 17 rather than press on. - Small sample on a shared, daily-resetting demo — not CI-reproducible. - [Methodology and caveats](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/benchmark/openemr/BENCHMARK.md). -- **Historical MockMed control** (bundled task): retained rows marked 100/100 - compiled runs and 20/20 agent runs successful under the 2026-07-08 OCR - check. The final frames were not retained, so the current verifier cannot - rescore those outcomes. Use these rows only for latency and estimated model - API charge comparison. The compiled arm recorded 4.9s p50 and $0 per run in - model API charges; the agent arm recorded 37.5s p50 and about $0.27 per run. - [Methodology and caveats](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/benchmark/BENCHMARK.md). -- **Independent effect verification** (fault-injection study, 90 runs per arm, - end to end through the real replayer into an on-disk SQLite system of record, - graded by a direct read-only database connection that bypasses the service): - a screen-only "success banner" oracle silently accepted **75.0%** of the wrong - effects that actually occurred (54 of 90 runs). Adding **one** out-of-band - verifier that reads the system of record cut that to **12.5%** (9 of 90). A - complete read path over every mutable surface reaches 0 of 90, but the - realistic deployment number is the middle rung — one out-of-band oracle — not - the 0%. All nine residual misses are a single named class: a collateral write - to a surface the oracle does not read. Every run terminates in an explicit - transaction outcome (VERIFIED, HALTED_BEFORE_EFFECT, - RECONCILIATION_REQUIRED, and others), so uncertain delivery is surfaced for - reconciliation, never reported as success. - [Methodology and caveats](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/benchmark/effect_e2e/EFFECT_E2E.md). - -The recorded zero model API calls mean no generative-model API charge for that -benchmark run. The figure excludes authoring, review, maintenance, and -infrastructure, and it is not a production-reliability or clinical-safety -claim. Read the -[limits](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/docs/LIMITS.md) -before extrapolating either result. +Both studies grade each run with a check that doesn't depend on the thing +being tested. -## Product Surfaces +### Fault test -| Target | Role | +We injected faults into saves that our own engine made to a test record +system: nine injected faults and one clean control, nine runs each. In 72 of +90 runs, the record ended up wrong: missing, partial, duplicated, on the wrong +record, or with other records deleted or added. The table shows how many of +those runs each kind of check still passed. + +| Check | Wrong records it passed | |---|---| -| `openadapt` | [`OpenAdapt`](https://github.com/OpenAdaptAI/OpenAdapt) installs the unified CLI. | -| `flow` | [`openadapt-flow`](https://github.com/OpenAdaptAI/openadapt-flow) is the canonical compiler and governed runtime. | -| `cloud` | [`app.openadapt.ai`](https://app.openadapt.ai/) provides the control plane for managed browser and customer-controlled execution. Its implementation repository is private. | -| `desktop` | [`openadapt-desktop`](https://github.com/OpenAdaptAI/openadapt-desktop) provides local recording, qualification, execution, evidence review, and governed repair. | -| `capture` | [`openadapt-capture`](https://github.com/OpenAdaptAI/openadapt-capture) records native screen, input, timing, and window-scoped evidence for Desktop and Flow. | -| `agent` | [`openadapt-agent`](https://github.com/OpenAdaptAI/openadapt-agent) exposes governed Flow workflows as local MCP tools and Agent Skills. | -| `docs` | [`docs.openadapt.ai`](https://docs.openadapt.ai) is the canonical documentation site. [`openadapt-ops`](https://github.com/OpenAdaptAI/openadapt-ops) is its **Support** publishing source. | - -[`openadapt-evals`](https://github.com/OpenAdaptAI/openadapt-evals) is a -Research repository. Runnable references live in +| Trust the app's success banner | 54 of 72 (75.0%) | +| One separate read of the record | 9 of 72 (12.5%) | +| A read of every table in the test database | 0 of 72 | + +The 9 that one separate read passed were all one kind: a write to a billing +table that read didn't cover. The zero covers only the audited test database. +Every check also stopped 9 of 18 runs that had saved correctly, because the +client never got the server's reply. This test covers a fixed list of faults. +It doesn't measure a real-world error rate. +[Method and caveats](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/benchmark/effect_e2e/EFFECT_E2E.md). + +### OpenEMR field test + +On the shared OpenEMR public demo, with fake patients and an 18-step task that +adds a patient note, compiled replay finished 19/20 runs at 39.2s median with +no model calls. A computer-use agent finished 10/10 runs at 70.4s median, at +about $0.55 a run at the July 2026 list price. The one compiled run that didn't +finish stopped at step 17, before saving, instead of reporting success. We +measured this on 2026-07-08 with a development build of Flow 0.1.0. It's a +small sample on a demo that resets daily, so CI can't reproduce it. OpenAdapt +isn't affiliated with the OpenEMR project. +[Method and caveats](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/benchmark/openemr/BENCHMARK.md). + +Neither study is a production reliability or clinical safety claim. The model +charge leaves out authoring, review, maintenance, and infrastructure. Read the +[limits](https://github.com/OpenAdaptAI/openadapt-flow/blob/main/docs/LIMITS.md) +before you extrapolate. Runnable references are in [`openadapt-flow/docs/showcase`](https://github.com/OpenAdaptAI/openadapt-flow/tree/main/docs/showcase), -with methods and evidence under +and each method is under [`benchmark`](https://github.com/OpenAdaptAI/openadapt-flow/tree/main/benchmark). -These targets form one product across browser, Windows, macOS, Linux, RDP, and -Citrix/VDI. Qualification is per workflow, not a blanket Production claim. -Current signed admissions are in the -[live record](https://docs.openadapt.ai/production-lifecycle.json). - -## Research and Labs +## Research and labs -Model training, retrieval, grounding, and general computer-use work remain +Model training, retrieval, grounding, and general computer-use work are Research: [`openadapt-ml`](https://github.com/OpenAdaptAI/openadapt-ml), -[`openadapt-retrieval`](https://github.com/OpenAdaptAI/openadapt-retrieval), and -[`openadapt-grounding`](https://github.com/OpenAdaptAI/openadapt-grounding). -They are not required for healthy deterministic replay. - -OmniMCP, SoM, and PydanticPrompt are **Labs**, not product dependencies. -Historical, Superseded, Deprecated, Archived, and Internal repositories are -classified in the public +[`openadapt-retrieval`](https://github.com/OpenAdaptAI/openadapt-retrieval), +[`openadapt-grounding`](https://github.com/OpenAdaptAI/openadapt-grounding), +and [`openadapt-evals`](https://github.com/OpenAdaptAI/openadapt-evals), which +scores agents and OpenAdapt programs on public benchmarks. You don't need any +of them to record, build, or run an automation. + +OmniMCP, SoM, and PydanticPrompt are Labs projects. The product doesn't depend +on them. The public [lifecycle registry](https://github.com/OpenAdaptAI/.github/blob/main/REPOSITORY_LIFECYCLE.md) -rather than presented as the product. +classifies every other repository as Support, Experimental, Historical, +Superseded, Deprecated, Archived, or Internal. ## Contribute -Product-engine changes belong in -[`openadapt-flow`](https://github.com/OpenAdaptAI/openadapt-flow). Packaging and -launcher changes belong in [`OpenAdapt`](https://github.com/OpenAdaptAI/OpenAdapt). -Use each repository's issues for scoped work, or visit -[`openadapt.ai`](https://openadapt.ai/) for deployment inquiries. - -Unless a repository says otherwise, OpenAdapt code is MIT licensed. +Engine changes belong in +[`openadapt-flow`](https://github.com/OpenAdaptAI/openadapt-flow). Packaging +and launcher changes belong in +[`OpenAdapt`](https://github.com/OpenAdaptAI/OpenAdapt). Use each repository's +issues for scoped work. Unless a repository says otherwise, OpenAdapt code is +MIT licensed. diff --git a/repository-lifecycle.yml b/repository-lifecycle.yml index 487446a..711083a 100644 --- a/repository-lifecycle.yml +++ b/repository-lifecycle.yml @@ -1,46 +1,328 @@ schema_version: 2 -reviewed_on: 2026-08-26 +reviewed_on: 2026-10-09 canonical_product: launcher: OpenAdapt engine: openadapt-flow +# public_metadata is the source of truth for the GitHub settings a visitor +# sees: the organization description and website, the profile headline, the +# pinned repositories, and each listed repository's description, website, and +# topics. Editing this file doesn't change GitHub. After merge, an organization +# owner runs .overhaul/APPLY.sh, which +# `python3 scripts/github_metadata.py render-apply` renders from this file. +# Each listed repository is set exactly: an empty homepage clears the website +# field, and the topic list replaces every current topic. public_metadata: organization_description: >- - Governed automation across browser, desktop, RDP, and Citrix: compile a - demonstration, verify the result, and halt on uncertainty. + Enters approved information into the systems your team already uses, + checks that it saved, and stops to ask a person when something doesn't + match. + organization_homepage: https://openadapt.ai/ + profile_headline: >- + OpenAdapt enters approved information into the systems your team already + uses, and then it checks that the entry saved. pinned_repositories: - OpenAdapt - openadapt-flow - openadapt-desktop - openadapt-capture - openadapt-agent - - openadapt-evals - repository_descriptions: - .github: >- - Organization profile and shared community files for OpenAdapt. - OpenAdapt: >- - Installer and unified CLI for OpenAdapt. The compiler and governed - runtime live in openadapt-flow. - openadapt-flow: >- - Compiler and governed runtime for demonstrated GUI workflows, with - explicit identity, policy, and result checks. - openadapt-web: >- - Source for openadapt.ai, including product pages, evidence, pricing, and - workflow qualification. - openadapt-ops: >- - Source for docs.openadapt.ai, with checks and runbooks that keep the - documentation current. - openadapt-cloud: >- - Proprietary hosted control plane for workflow qualification, deployment, - attended operation, and governed repair. - openadapt-desktop: >- - Desktop application for local workflow authoring, operation, evidence - review, and governed repair. - openadapt-capture: >- - Native desktop recorder that keeps screen, input, timing, and - window-scoped evidence synchronized. - openadapt-agent: >- - Local bridge that exposes governed OpenAdapt workflows to MCP clients and - Agent Skills. + - ehr-integration-directory + repositories: + OpenAdapt: + description: >- + OpenAdapt takes the last manual step off your team. It enters approved + information into EMRs, portals, and desktop or Citrix apps, then checks + that it saved. + homepage: https://openadapt.ai + topics: + - rpa + - robotic-process-automation + - process-automation + - workflow-automation + - desktop-automation + - browser-automation + - gui-automation + - ui-automation + - citrix + - rdp + - healthcare-automation + - emr + - ehr + - computer-use + - human-in-the-loop + - audit-trail + - local-first + - automation-from-demonstration + - mcp + - python + openadapt-flow: + description: >- + OpenAdapt's open-source engine. It turns one recorded task into a + program that runs on your computer and never reports an unchecked entry + as done. + homepage: https://openadapt.ai/platform/flow + topics: + - browser-automation + - computer-use + - deterministic-replay + - gui-automation + - healthcare-automation + - rpa + - workflow-automation + - automation-from-demonstration + - citrix + - demonstration-compiler + - desktop-automation + - effect-verification + - local-first + - python + - rdp + - ui-automation + - human-in-the-loop + - emr + - audit-trail + - robotic-process-automation + openadapt-desktop: + description: >- + The OpenAdapt app. Record a task once, watch it run, see what it + entered, and decide what happens when a run stops. + homepage: https://openadapt.ai/platform/desktop + topics: + - automation-from-demonstration + - desktop-automation + - gui-automation + - local-first + - openadapt + - python + - human-in-the-loop + - rpa + - citrix + openadapt-capture: + description: >- + Records screen, mouse, and keyboard on Windows, macOS, and Linux so + OpenAdapt can turn one recorded task into an automation. + homepage: https://openadapt.ai/platform/capture + topics: + - gui-automation + - python + - screen-capture + - input-recording + - openadapt + openadapt-agent: + description: >- + Lets your AI agent run approved OpenAdapt automations on your computer. + Each run reports how it ended, and a run that stops goes to a person. + homepage: https://openadapt.ai/platform/agent + topics: + - agent-skills + - gui-automation + - local-first + - mcp + - openadapt + - workflow-automation + - computer-use + - human-in-the-loop + ehr-integration-directory: + description: >- + Shows which EHR tasks have a public API and which still end at a + screen. Each entry links its source. Published by OpenAdapt. + homepage: https://ehrintegrationdirectory.com + topics: + - ehr + - emr + - fhir + - hl7 + - smart-on-fhir + - healthcare + - healthcare-it + - interoperability + - openadapt + openadapt-cloud: + description: >- + Proprietary source for app.openadapt.ai, the hosted service that runs + automations, tests them before they go live, and sends stopped runs to + a person. + homepage: https://app.openadapt.ai + topics: [] + .github: + description: >- + Support: OpenAdapt's GitHub profile, the status of every repository, + and shared community files. + homepage: https://openadapt.ai + topics: + - openadapt + openadapt-web: + description: >- + Source for openadapt.ai, including product pages, pricing, and + evidence. + homepage: https://openadapt.ai + topics: [] + openadapt-ops: + description: >- + Support: source for docs.openadapt.ai, plus the checks that keep the + docs in step with each release. + homepage: https://docs.openadapt.ai + topics: + - openadapt + openadapt-blog: + description: >- + Support: source for blog.openadapt.ai, with guides, product updates, + and test write-ups. + homepage: https://blog.openadapt.ai + topics: + - openadapt + openadapt-wright: + description: >- + Support: team tool that drafts code changes and pull requests for a + person to approve. Not part of the product. + homepage: "" + topics: [] + openadapt-herald: + description: >- + Support: team tool that drafts release announcements from git history. + Not part of the product. + homepage: "" + topics: [] + openadapt-crier: + description: >- + Support: team tool that sends drafted social posts to Telegram for a + person to approve. Not part of the product. + homepage: "" + topics: [] + openadapt-consilium: + description: >- + Support: team tool that asks several AI models one question and + combines their answers. Not part of the product. + homepage: "" + topics: [] + openadapt-telemetry: + description: >- + Support: error reports and usage counts for OpenAdapt packages, with + personal details scrubbed. Set DO_NOT_TRACK=1 to opt out. + homepage: "" + topics: [] + openadapt-viewer: + description: >- + Support: turns test results and screenshots into one HTML page you can + share. Not needed to automate a task. + homepage: https://pypi.org/project/openadapt-viewer/ + topics: + - html + - openadapt + - python + - visualization + openadapt-privacy: + description: >- + Experimental: detects and redacts personal and health information in + recordings and screenshots before they're shared. + homepage: https://pypi.org/project/openadapt-privacy/ + topics: + - openadapt + - pii-detection + - privacy + - python + - scrubbing + openadapt-types: + description: >- + Experimental: shared data formats for screens and actions inside + OpenAdapt. Not needed to automate a task. + homepage: "" + topics: + - openadapt + openadapt-console: + description: >- + Experimental: an early web console prototype. Not part of the product + today. + homepage: "" + topics: [] + openadapt-tray: + description: >- + Experimental: a tray icon that shows what OpenAdapt is doing, with + start and stop controls. Optional. + homepage: "" + topics: + - openadapt + openadapt-ml: + description: >- + Research: optional model training for computer-use agents. Not needed + to record, build, or run an OpenAdapt automation. + homepage: https://pypi.org/project/openadapt-ml/ + topics: + - computer-use + - gui-automation + - machine-learning + - openadapt + - python + - vlm + - research + openadapt-evals: + description: >- + Research: scores computer-use agents and OpenAdapt programs on public + benchmarks, including Windows Agent Arena. Not needed to automate a + task. + homepage: https://openadapt.ai/platform/evals + topics: + - benchmarks + - evaluation + - gui-automation + - openadapt + - python + openadapt-retrieval: + description: >- + Research: searches past recordings by image and text for model-driven + agents. Not needed to automate a task. + homepage: https://pypi.org/project/openadapt-retrieval/ + topics: + - embeddings + - openadapt + - python + - retrieval + - demo-retrieval + openadapt-grounding: + description: >- + Research: finds buttons and fields on screen with vision models. Not + needed to automate a task. + homepage: https://pypi.org/project/openadapt-grounding/ + topics: + - computer-vision + - openadapt + - python + - set-of-mark + - ui-detection + OmniMCP: + description: >- + Labs: an agent experiment that asks a model at every step. Not the + OpenAdapt product; see openadapt-flow. + homepage: https://omnimcp.openadapt.ai/ + topics: + - anthropic + - aws + - computeruse + - gemini + - generative-ai + - model-context-protocol + - omniparser + - openai + - labs + - research + SoM: + description: >- + Labs: fork of Microsoft's Set-of-Mark visual prompting research. Not + the OpenAdapt product. + homepage: "" + topics: [] + PydanticPrompt: + description: >- + Labs: a small Python library that documents Pydantic models for LLM + prompts. Not the OpenAdapt product. + homepage: "" + topics: [] + OpenSanitizer: + description: >- + Superseded by openadapt-privacy, which new projects should use. Older + PII and PHI scrubbing for screen data. + homepage: "" + topics: [] lifecycle: production: [] support: @@ -54,6 +336,7 @@ lifecycle: - openadapt-telemetry - openadapt-viewer - openadapt-blog + - ehr-integration-directory beta: [] experimental: - openadapt-privacy diff --git a/scripts/check_profile.py b/scripts/check_profile.py index 324428d..a875c8e 100755 --- a/scripts/check_profile.py +++ b/scripts/check_profile.py @@ -1,14 +1,22 @@ #!/usr/bin/env python3 -"""Check the public profile and its evidence-gated lifecycle.""" +"""Check the public profile, its GitHub metadata, and its evidence-gated lifecycle.""" from __future__ import annotations import json +import os import re import sys from pathlib import Path from urllib.parse import unquote, urlsplit +from github_metadata import ( + APPLY_SCRIPT, + MetadataError, + PublicMetadata, + load_public_metadata, + render_apply_script, +) from validate_production_lifecycle import LifecycleError, validate_files ROOT = Path(__file__).resolve().parents[1] @@ -32,17 +40,78 @@ "https://github.com/OpenAdaptAI/openadapt-agent", "https://github.com/OpenAdaptAI/openadapt-ops", "https://github.com/OpenAdaptAI/openadapt-evals", + "https://github.com/OpenAdaptAI/ehr-integration-directory", "https://github.com/OpenAdaptAI/openadapt-flow/tree/main/docs/showcase", "https://openadapt.ai/", + "https://openadapt.ai/book", + "https://openadapt.ai/services", "https://app.openadapt.ai/", + "https://app.openadapt.ai/demo", "https://docs.openadapt.ai", "https://docs.openadapt.ai/production-lifecycle.json", } -REQUIRED_PROFILE_MARKERS = { - "## Product Surfaces", - "## Research and Labs", - "Qualification is per workflow, not a blanket Production claim.", -} +# The business half of the profile comes first. Everything above +# DEVELOPER_HEADING is written for people who own a process. +DEVELOPER_HEADING = "## For developers" +PRODUCT_HEADING = "## Product surfaces" +PROFILE_HEADINGS = ( + "## What changes for your team", + "## How a run ends", + "## Two ways to start", + "### Hand off a workflow", + "### Run it yourself", + DEVELOPER_HEADING, + PRODUCT_HEADING, + "## Research and labs", +) +REQUIRED_PROFILE_MARKERS = PROFILE_HEADINGS + ( + "Qualification is per workflow. A product release admission doesn't " + "qualify any workflow.", +) +CANONICAL_QUICKSTART = ( + "python -m pip install --upgrade openadapt", + "openadapt quickstart --break-it", +) +# Phrases that misstate a claim of record or publish a retired path. The fault +# test's screen-only check passed 54 of the 72 runs that left the record wrong; +# "54 of 90" pairs that count with every run, and pairing it with 75.0% is +# wrong arithmetic. +PROHIBITED_PROFILE_PHRASES = ( + "54 of 90", + "openadapt flow demo-record --out rec", +) +# Terms that never appear where a business reader starts. The technical term +# stays available one level deeper: below DEVELOPER_HEADING, and in the +# developer repositories' own descriptions. +BUSINESS_FORBIDDEN_TERMS = ( + (re.compile(r"\bseals?\b", re.IGNORECASE), "Seal"), + (re.compile(r"\badmi(?:t|ts|tted|tting|ssion|ssions)\b", re.IGNORECASE), "admission"), + ( + re.compile(r"\bqualif(?:y|ied|ies|ying|ication|ications)\b", re.IGNORECASE), + "qualification", + ), + (re.compile(r"\bgoverned\b", re.IGNORECASE), "governed"), + (re.compile(r"\boracles?\b", re.IGNORECASE), "oracle"), + (re.compile(r"\beffect contracts?\b", re.IGNORECASE), "effect contract"), + (re.compile(r"\bsubstrates?\b", re.IGNORECASE), "substrate"), + (re.compile(r"\bfixtures?\b", re.IGNORECASE), "fixture"), + (re.compile(r"REST\+SQL", re.IGNORECASE), "REST+SQL parity"), + (re.compile(r"non-target delta", re.IGNORECASE), "non-target delta"), + (re.compile(r"\bsha256\b", re.IGNORECASE), "sha256"), + (re.compile(r"\bstandard profile\b", re.IGNORECASE), "Standard profile"), + (re.compile(r"program state console", re.IGNORECASE), "Program State Console"), + (re.compile(r"\brunners?\b", re.IGNORECASE), "runner"), + (re.compile(r"\bMCP\b"), "MCP"), + ( + re.compile(r"\b(?:VERIFIED|HALTED|COMPLETED_UNVERIFIED|FAILED)\b"), + "an all-caps run outcome", + ), + ( + re.compile(r"RECONCILIATION_REQUIRED|HALTED_BEFORE_EFFECT"), + "a transaction outcome name", + ), +) +DASHES = ("–", "—") EXPECTED_PINNED_REPOSITORIES = ( "OpenAdapt", @@ -50,7 +119,7 @@ "openadapt-desktop", "openadapt-capture", "openadapt-agent", - "openadapt-evals", + "ehr-integration-directory", ) LINK_RE = re.compile(r"!?\[[^\]]+\]\(([^\s)]+)(?:\s+[^)]*)?\)") PROFILE_TARGET_ROW_RE = re.compile(r"^\| `([a-z]+)` \|\s*(.*?)\s*\|$") @@ -74,9 +143,39 @@ "openadapt-web": "support", "openadapt-ops": "support", "openadapt-blog": "support", + "ehr-integration-directory": "support", "OpenAdapter": "archived", "OpenReflector": "archived", } +# A product target's state is derived from signed admissions, so its +# repository description never carries a static lifecycle label. A described +# repository that a visitor could mistake for the product opens with its +# lifecycle label instead. +STATIC_TARGET_LABELS = ("Beta", "Experimental", "Early access", "Exploratory") +REQUIRED_DESCRIPTION_LABELS = { + "experimental": "Experimental: ", + "research": "Research: ", + "labs": "Labs: ", + "superseded": "Superseded by ", + "deprecated": "Deprecated", + "historical": "Historical", + "archived": "Archived", +} +LEADING_LABEL_RE = re.compile( + r"^(Production|Support|Beta|Experimental|Research|Internal|Labs|Historical|" + r"Superseded|Deprecated|Archived)\b" +) +UNPINNABLE_GROUPS = { + "beta", + "experimental", + "research", + "internal", + "labs", + "historical", + "superseded", + "deprecated", + "archived", +} FORBIDDEN_PUBLIC_OPERATIONS_MARKERS = ( "/Users/", "~/", @@ -115,6 +214,119 @@ def check_link(source: Path, destination: str) -> str | None: return None +def business_term_errors(label: str, text: str) -> list[str]: + """Name every business-vocabulary rule that `text` breaks.""" + return [ + f"{label} uses {name!r}; business-facing copy uses the plain term" + for pattern, name in BUSINESS_FORBIDDEN_TERMS + if pattern.search(text) + ] + + +def reader_words(markdown: str) -> str: + """Drop link destinations and fenced code, which a reader doesn't read as prose.""" + without_code = re.sub(r"```.*?```", " ", markdown, flags=re.DOTALL) + return re.sub(r"\]\([^)]*\)", "]", without_code) + + +def parse_lifecycle_groups( + lifecycle_text: str, errors: list[str] +) -> dict[str, list[str]]: + lifecycle_section = lifecycle_text.split("\nlifecycle:\n", maxsplit=1) + if len(lifecycle_section) != 2: + errors.append("repository-lifecycle.yml is missing its lifecycle mapping") + return {} + groups: dict[str, list[str]] = {} + active_group: str | None = None + for line in lifecycle_section[1].splitlines(): + if match := LIFECYCLE_GROUP_RE.fullmatch(line): + active_group = match.group(1) + groups[active_group] = [] + elif match := LIFECYCLE_REPOSITORY_RE.fullmatch(line): + if active_group is None: + errors.append( + "repository-lifecycle.yml has a repository outside a lifecycle group" + ) + else: + groups[active_group].append(match.group(1)) + elif line and not line.startswith(" "): + break + return groups + + +def metadata_errors( + metadata: PublicMetadata, + product_repositories: set[str], + lifecycles: dict[str, str], +) -> list[str]: + """Check the public metadata against the lifecycle and the copy rules.""" + errors: list[str] = [] + + if metadata.pinned_repositories != EXPECTED_PINNED_REPOSITORIES: + errors.append( + "repository-lifecycle.yml product pins do not match the public contract: " + f"{list(metadata.pinned_repositories)}" + ) + + for label, text in ( + ("organization_description", metadata.organization_description), + ("profile_headline", metadata.profile_headline), + ): + errors += business_term_errors(f"repository-lifecycle.yml {label}", text) + + static_label_patterns = [ + (label, re.compile(rf"\b{re.escape(label)}\b", re.IGNORECASE)) + for label in STATIC_TARGET_LABELS + ] + for name, repository in metadata.repositories.items(): + group = lifecycles.get(name) + is_product = name in product_repositories + if group is None and not is_product: + errors.append( + f"repository-lifecycle.yml describes {name}, which is neither a " + "product target repository nor in a lifecycle group" + ) + + if is_product or name in metadata.pinned_repositories: + static = sorted( + label + for label, pattern in static_label_patterns + if pattern.search(repository.description) + ) + if static: + errors.append( + f"repository-lifecycle.yml {name} description carries static " + f"lifecycle labels: {static}" + ) + + required = REQUIRED_DESCRIPTION_LABELS.get(group or "") + if required and not repository.description.startswith(required): + errors.append( + f"repository-lifecycle.yml {name} is {group}, so its description " + f"must start with {required.strip()!r}" + ) + if match := LEADING_LABEL_RE.match(repository.description): + if match.group(1).lower() != group: + errors.append( + f"repository-lifecycle.yml {name} description opens with " + f"{match.group(1)!r}, but its lifecycle is {group!r}" + ) + + for name in metadata.pinned_repositories: + group = lifecycles.get(name) + if group in UNPINNABLE_GROUPS: + errors.append( + f"repository-lifecycle.yml pins {name}, which is {group}; pins are " + "for product and public Support repositories" + ) + if name in metadata.repositories: + errors += business_term_errors( + f"repository-lifecycle.yml pinned {name} description", + metadata.repositories[name].description, + ) + return errors + + def main() -> int: errors: list[str] = [] profile_text = PROFILE.read_text(encoding="utf-8") @@ -127,20 +339,32 @@ def main() -> int: errors.append("profile/README.md is missing the canonical product truth") for marker in REQUIRED_PROFILE_MARKERS: - if marker not in normalized_profile: + if " ".join(marker.split()) not in normalized_profile: errors.append(f"profile/README.md is missing required marker: {marker}") + heading_positions = [ + profile_text.find(f"\n{heading}\n") for heading in PROFILE_HEADINGS + ] + if -1 not in heading_positions and heading_positions != sorted(heading_positions): + errors.append( + "profile/README.md must keep its sections in order, with every " + f"business section before {DEVELOPER_HEADING!r}" + ) - canonical_quickstart = ( - "python -m pip install --upgrade 'openadapt[browser]'", - "openadapt quickstart", - ) - for command in canonical_quickstart: + for command in CANONICAL_QUICKSTART: if command not in profile_text: errors.append( f"profile/README.md is missing canonical quickstart command: {command}" ) - if "openadapt flow demo-record --out rec" in profile_text: - errors.append("profile/README.md publishes the superseded manual tutorial path") + for phrase in PROHIBITED_PROFILE_PHRASES: + if phrase in normalized_profile: + errors.append(f"profile/README.md publishes a prohibited phrase: {phrase!r}") + if any(dash in profile_text for dash in DASHES): + errors.append("profile/README.md uses an en or em dash") + + business_section = profile_text.split(f"\n{DEVELOPER_HEADING}\n", maxsplit=1)[0] + errors += business_term_errors( + "profile/README.md business section", reader_words(business_section) + ) profile_links = set(LINK_RE.findall(profile_text)) missing_links = sorted(REQUIRED_PROFILE_LINKS - profile_links) @@ -157,9 +381,12 @@ def main() -> int: (ROOT / "production-lifecycle-policy.json").read_text(encoding="utf-8") ) expected_profile_targets = tuple(target["id"] for target in policy["targets"]) - product_section_parts = profile_text.split("## Product Surfaces\n", maxsplit=1) + product_repositories = { + target["source_repository"].split("/", 1)[1] for target in policy["targets"] + } + product_section_parts = profile_text.split(f"{PRODUCT_HEADING}\n", maxsplit=1) if len(product_section_parts) != 2: - errors.append("profile/README.md is missing the Product Surfaces section") + errors.append("profile/README.md is missing the Product surfaces section") else: product_section = product_section_parts[1].split("\n## ", maxsplit=1)[0] profile_target_rows = [ @@ -215,52 +442,8 @@ def main() -> int: ) lifecycle_text = LIFECYCLE_DATA.read_text(encoding="utf-8") - public_metadata_text = lifecycle_text.split("lifecycle:\n", maxsplit=1)[0] - static_target_labels = sorted( - label - for label in ("Beta", "Experimental", "Early access", "Exploratory") - if re.search(rf"\b{re.escape(label)}\b", public_metadata_text, re.IGNORECASE) - ) - if static_target_labels: - errors.append( - "repository-lifecycle.yml public target metadata contains static " - f"lifecycle labels: {static_target_labels}" - ) - pinned_section = lifecycle_text.split(" pinned_repositories:\n", maxsplit=1) - if len(pinned_section) != 2: - errors.append("repository-lifecycle.yml is missing pinned_repositories") - else: - pinned: list[str] = [] - for line in pinned_section[1].splitlines(): - if match := LIFECYCLE_REPOSITORY_RE.fullmatch(line): - pinned.append(match.group(1)) - elif line and not line.startswith(" "): - break - if tuple(pinned) != EXPECTED_PINNED_REPOSITORIES: - errors.append( - "repository-lifecycle.yml product pins do not match the public contract" - ) - - lifecycle_section = lifecycle_text.split("lifecycle:\n", maxsplit=1) - if len(lifecycle_section) != 2: - errors.append("repository-lifecycle.yml is missing its lifecycle mapping") - else: - groups: dict[str, list[str]] = {} - active_group: str | None = None - for line in lifecycle_section[1].splitlines(): - if match := LIFECYCLE_GROUP_RE.fullmatch(line): - active_group = match.group(1) - groups[active_group] = [] - elif match := LIFECYCLE_REPOSITORY_RE.fullmatch(line): - if active_group is None: - errors.append( - "repository-lifecycle.yml has a repository outside a lifecycle group" - ) - else: - groups[active_group].append(match.group(1)) - elif line and not line.startswith(" "): - break - + groups = parse_lifecycle_groups(lifecycle_text, errors) + if groups: if set(groups) != EXPECTED_LIFECYCLE_GROUPS: errors.append( "repository-lifecycle.yml lifecycle groups do not match the public schema" @@ -277,20 +460,48 @@ def main() -> int: errors.append( f"repository-lifecycle.yml assigns multiple lifecycles: {duplicates}" ) - actual_lifecycles = { - repository: group - for group, group_repositories in groups.items() - for repository in group_repositories - } - for repository, expected_group in EXPECTED_CRITICAL_LIFECYCLES.items(): - if actual_lifecycles.get(repository) != expected_group: + actual_lifecycles = { + repository: group + for group, group_repositories in groups.items() + for repository in group_repositories + } + for repository, expected_group in EXPECTED_CRITICAL_LIFECYCLES.items(): + if actual_lifecycles.get(repository) != expected_group: + errors.append( + "repository-lifecycle.yml assigns " + f"{repository} to {actual_lifecycles.get(repository)!r}; " + f"expected {expected_group!r}" + ) + + metadata: PublicMetadata | None + try: + metadata = load_public_metadata(LIFECYCLE_DATA) + except MetadataError as exc: + errors.append(f"repository-lifecycle.yml public_metadata: {exc}") + metadata = None + if metadata is not None: + errors += metadata_errors(metadata, product_repositories, actual_lifecycles) + first_bold = next( + (line for line in profile_text.splitlines() if line.startswith("**")), "" + ) + if first_bold != f"**{metadata.profile_headline}**": + errors.append( + "profile/README.md must open with the profile_headline from " + "repository-lifecycle.yml as its first bold line" + ) + if APPLY_SCRIPT.exists(): + relative = APPLY_SCRIPT.relative_to(ROOT) + if APPLY_SCRIPT.read_text(encoding="utf-8") != render_apply_script(metadata): errors.append( - "repository-lifecycle.yml assigns " - f"{repository} to {actual_lifecycles.get(repository)!r}; " - f"expected {expected_group!r}" + f"{relative} does not match repository-lifecycle.yml. Run: " + f"python3 scripts/github_metadata.py render-apply --output {relative}" ) + if not os.access(APPLY_SCRIPT, os.X_OK): + errors.append(f"{relative} is not executable") public_operations_text = lifecycle_text + lifecycle_doc_text + if APPLY_SCRIPT.exists(): + public_operations_text += APPLY_SCRIPT.read_text(encoding="utf-8") leaked_markers = sorted( marker for marker in FORBIDDEN_PUBLIC_OPERATIONS_MARKERS @@ -313,7 +524,7 @@ def main() -> int: if error := check_link(source, destination): errors.append(f"{source.relative_to(ROOT)}: {error}") - if errors: + if errors or metadata is None: for error in errors: print(f"ERROR: {error}", file=sys.stderr) return 1 @@ -322,7 +533,11 @@ def main() -> int: len(LINK_RE.findall(path.read_text(encoding="utf-8"))) for path in MARKDOWN_FILES ) - print(f"Validated canonical product truth and {link_count} Markdown links.") + print( + "Validated canonical product truth, " + f"{len(metadata.repositories)} repository descriptions, " + f"{len(metadata.pinned_repositories)} pins, and {link_count} Markdown links." + ) return 0 diff --git a/scripts/github_metadata.py b/scripts/github_metadata.py new file mode 100755 index 0000000..8d5c06c --- /dev/null +++ b/scripts/github_metadata.py @@ -0,0 +1,616 @@ +#!/usr/bin/env python3 +"""Read the public GitHub metadata in repository-lifecycle.yml and render it. + +`public_metadata` in `repository-lifecycle.yml` is the source of truth for the +GitHub settings a visitor sees: the organization description and website, the +profile headline, the pinned repositories, and each listed repository's +description, website, and topics. GitHub doesn't read this file. An +organization owner applies it after merge with the script this module renders: + + python3 scripts/github_metadata.py render-apply --output .overhaul/APPLY.sh + +The parser reads only the small, strict YAML subset the registry uses, so the +profile check needs no third-party YAML package. It refuses anything outside +that subset instead of guessing what it means. +""" + +from __future__ import annotations + +import argparse +import json +import re +import shlex +import sys +from dataclasses import dataclass +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +LIFECYCLE_DATA = ROOT / "repository-lifecycle.yml" +APPLY_SCRIPT = ROOT / ".overhaul" / "APPLY.sh" +ORGANIZATION = "OpenAdaptAI" + +METADATA_KEYS = ( + "organization_description", + "organization_homepage", + "profile_headline", + "pinned_repositories", + "repositories", +) +REPOSITORY_KEYS = ("description", "homepage", "topics") + +# GitHub allows up to 20 topics per repository, each lowercase letters, digits, +# and hyphens, starting with a letter or digit, at most 50 characters. +MAX_TOPICS = 20 +TOPIC_RE = re.compile(r"^[a-z0-9][a-z0-9-]{0,49}$") +MAX_PINS = 6 +# Our own limit, well under GitHub's: a description longer than this wraps +# onto a second line on pinned cards and in search results. +MAX_DESCRIPTION_LENGTH = 160 +REPOSITORY_NAME_RE = re.compile(r"^[A-Za-z0-9._-]+$") +URL_RE = re.compile(r"^https://[A-Za-z0-9.-]+\.[A-Za-z]{2,}(?:/[A-Za-z0-9._~/-]*)?$") +PLAIN_SCALAR_RE = re.compile(r"^[A-Za-z0-9][^#'\"]*$") +KEY_LINE_RE = re.compile(r"^(?P *)(?P[A-Za-z0-9._-]+):(?: (?P.*))?$") +ITEM_LINE_RE = re.compile(r"^(?P *)- (?P.*)$") +DASHES = ("–", "—") + + +class MetadataError(Exception): + """The registry's public metadata is malformed or outside the subset.""" + + +@dataclass(frozen=True) +class RepositoryMetadata: + name: str + description: str + homepage: str + topics: tuple[str, ...] + + +@dataclass(frozen=True) +class PublicMetadata: + organization_description: str + organization_homepage: str + profile_headline: str + pinned_repositories: tuple[str, ...] + repositories: dict[str, RepositoryMetadata] + + +Value = str | list[str] | dict[str, "Value"] + + +def _block(text: str, key: str) -> list[tuple[int, str]]: + """Return the numbered lines under a top-level key, without the key.""" + lines = text.splitlines() + try: + start = lines.index(f"{key}:") + except ValueError as exc: + raise MetadataError(f"repository-lifecycle.yml has no top-level {key}:") from exc + block: list[tuple[int, str]] = [] + for number, line in enumerate(lines[start + 1 :], start=start + 2): + if line and not line.startswith(" "): + break + stripped = line.strip() + if not stripped or stripped.startswith("#"): + continue + block.append((number, line)) + return block + + +def _scalar(number: int, raw: str) -> str: + if raw == '""': + return "" + if not PLAIN_SCALAR_RE.fullmatch(raw) or ": " in raw or raw.endswith(":"): + raise MetadataError( + f"line {number}: {raw!r} is not a plain value. Use a folded " + "'>-' block for text, or '\"\"' for an empty value." + ) + return raw + + +def _parse(lines: list[tuple[int, str]], indent: int) -> dict[str, Value]: + """Parse one mapping level of the strict subset.""" + result: dict[str, Value] = {} + index = 0 + while index < len(lines): + number, line = lines[index] + match = KEY_LINE_RE.fullmatch(line) + if match is None or len(match.group("indent")) != indent: + raise MetadataError( + f"line {number}: expected a key indented {indent} spaces: {line!r}" + ) + key = match.group("key") + if key in result: + raise MetadataError(f"line {number}: duplicate key {key!r}") + rest = match.group("rest") + index += 1 + children: list[tuple[int, str]] = [] + while index < len(lines): + child_indent = len(lines[index][1]) - len(lines[index][1].lstrip(" ")) + if child_indent <= indent: + break + children.append(lines[index]) + index += 1 + + if rest is None: + if not children: + raise MetadataError(f"line {number}: {key!r} has no value") + if _lines_are_items(children): + result[key] = _items(children, indent + 2) + else: + result[key] = _parse(children, indent + 2) + elif rest == ">-": + if not children: + raise MetadataError(f"line {number}: {key!r} has an empty text block") + result[key] = " ".join(child.strip() for _, child in children) + else: + if children: + raise MetadataError( + f"line {number}: {key!r} has a value and an indented block" + ) + result[key] = [] if rest == "[]" else _scalar(number, rest) + return result + + +def _lines_are_items(lines: list[tuple[int, str]]) -> bool: + return ITEM_LINE_RE.fullmatch(lines[0][1]) is not None + + +def _items(lines: list[tuple[int, str]], indent: int) -> list[str]: + values: list[str] = [] + for number, line in lines: + match = ITEM_LINE_RE.fullmatch(line) + if match is None or len(match.group("indent")) != indent: + raise MetadataError( + f"line {number}: expected a list item indented {indent} spaces" + ) + values.append(_scalar(number, match.group("value"))) + return values + + +def parse_public_metadata_mapping(text: str) -> dict[str, Value]: + """Parse `public_metadata` into plain Python values.""" + return _parse(_block(text, "public_metadata"), 2) + + +def _text(value: Value, label: str) -> str: + if not isinstance(value, str): + raise MetadataError(f"{label} must be text") + return value + + +def _names(value: Value, label: str) -> tuple[str, ...]: + if not isinstance(value, list): + raise MetadataError(f"{label} must be a list") + return tuple(value) + + +def load_public_metadata(path: Path = LIFECYCLE_DATA) -> PublicMetadata: + """Load and structurally validate the registry's public metadata.""" + raw = parse_public_metadata_mapping(path.read_text(encoding="utf-8")) + if tuple(raw) != METADATA_KEYS: + raise MetadataError( + f"public_metadata keys are {list(raw)}; expected {list(METADATA_KEYS)} " + "in that order" + ) + repositories_raw = raw["repositories"] + if not isinstance(repositories_raw, dict) or not repositories_raw: + raise MetadataError("public_metadata.repositories must be a non-empty mapping") + repositories: dict[str, RepositoryMetadata] = {} + for name, entry in repositories_raw.items(): + if not isinstance(entry, dict) or tuple(entry) != REPOSITORY_KEYS: + raise MetadataError( + f"repository {name} must have exactly {list(REPOSITORY_KEYS)}, " + "in that order" + ) + repositories[name] = RepositoryMetadata( + name=name, + description=_text(entry["description"], f"{name}.description"), + homepage=_text(entry["homepage"], f"{name}.homepage"), + topics=_names(entry["topics"], f"{name}.topics"), + ) + metadata = PublicMetadata( + organization_description=_text( + raw["organization_description"], "organization_description" + ), + organization_homepage=_text(raw["organization_homepage"], "organization_homepage"), + profile_headline=_text(raw["profile_headline"], "profile_headline"), + pinned_repositories=_names(raw["pinned_repositories"], "pinned_repositories"), + repositories=repositories, + ) + errors = structural_errors(metadata) + if errors: + raise MetadataError("; ".join(errors)) + return metadata + + +def _text_errors(label: str, text: str, *, max_length: int | None) -> list[str]: + errors: list[str] = [] + if not text.strip(): + errors.append(f"{label} is empty") + if not text.isascii(): + errors.append(f"{label} has non-ASCII characters (curly quotes or dashes?)") + if any(dash in text for dash in DASHES) or " -- " in text: + errors.append(f"{label} uses a dash; short copy takes none") + if max_length is not None and len(text) > max_length: + errors.append(f"{label} is {len(text)} characters; the limit is {max_length}") + return errors + + +def structural_errors(metadata: PublicMetadata) -> list[str]: + """Check formats GitHub enforces, plus our own short-copy rules.""" + errors: list[str] = [] + errors += _text_errors( + "organization_description", + metadata.organization_description, + max_length=MAX_DESCRIPTION_LENGTH, + ) + errors += _text_errors("profile_headline", metadata.profile_headline, max_length=None) + if not URL_RE.fullmatch(metadata.organization_homepage): + errors.append("organization_homepage must be an https URL") + + pins = metadata.pinned_repositories + if not pins or len(pins) > MAX_PINS: + errors.append(f"pinned_repositories must list 1 to {MAX_PINS} repositories") + if len(set(pins)) != len(pins): + errors.append("pinned_repositories lists a repository twice") + for name in pins: + if name not in metadata.repositories: + errors.append(f"pinned repository {name} has no entry under repositories") + + for name, repository in metadata.repositories.items(): + if not REPOSITORY_NAME_RE.fullmatch(name): + errors.append(f"{name!r} is not a GitHub repository name") + errors += _text_errors( + f"{name}.description", + repository.description, + max_length=MAX_DESCRIPTION_LENGTH, + ) + if repository.homepage and not URL_RE.fullmatch(repository.homepage): + errors.append(f"{name}.homepage must be an https URL or \"\"") + if len(repository.topics) > MAX_TOPICS: + errors.append(f"{name} has {len(repository.topics)} topics; GitHub allows {MAX_TOPICS}") + if len(set(repository.topics)) != len(repository.topics): + errors.append(f"{name} lists a topic twice") + for topic in repository.topics: + if not TOPIC_RE.fullmatch(topic): + errors.append(f"{name} topic {topic!r} isn't a valid GitHub topic") + return errors + + +# The apply script ---------------------------------------------------------- + +PINS_QUERY = ( + "query($org: String!) { organization(login: $org) { " + "pinnedItems(first: 6, types: REPOSITORY) { nodes { ... on Repository { name } } } } }" +) + +SCRIPT_HEAD = """\ +#!/usr/bin/env bash +# Apply the public GitHub metadata in repository-lifecycle.yml to the +# {org} organization: the organization description and website, and each +# listed repository's description, website, and topics. +# +# Generated by `python3 scripts/github_metadata.py render-apply` from +# repository-lifecycle.yml. Don't edit this file by hand. Edit the registry +# and render it again; scripts/check_profile.py fails when the two differ. +# +# Usage: +# APPLY.sh Dry run. Prints each command and request body. +# Calls nothing and changes nothing. +# APPLY.sh --apply Makes the changes, then reads GitHub back and +# compares every value with the registry. +# APPLY.sh --verify Reads GitHub back and compares. Changes nothing. +# +# Exit status: 0 when GitHub matches the registry. 1 when a command failed, +# a value couldn't be read, or a value differs. 2 when everything matches +# except the pinned repositories, which only an owner can set by hand. +# 64 for a usage error. +# +# Needs the GitHub CLI (gh) signed in as an owner of {org}, with a token +# that has the admin:org and repo scopes. To add them, run: +# gh auth refresh -s admin:org,repo +# +# Each listed repository is set exactly: an empty website clears the field, +# and the topic list replaces every current topic. +# +# GitHub has no API for organization pins. On https://github.com/{org}, +# choose "Customize pins" and pin these repositories in this order: +{pin_lines} + +set -uo pipefail + +ORG={org} +PINS={pins_csv} +REGISTERED={registered} + +MODE=dry-run +if [ "$#" -gt 1 ]; then + printf 'Use one option: --dry-run (the default), --apply, or --verify.\\n' >&2 + exit 64 +fi +case "${{1:-}}" in + "" | --dry-run) MODE=dry-run ;; + --apply) MODE=apply ;; + --verify) MODE=verify ;; + -h | --help) + sed -n '2,/^$/p' "$0" + exit 0 + ;; + *) + printf 'Unknown option: %s. Use --dry-run, --apply, or --verify.\\n' "$1" >&2 + exit 64 + ;; +esac + +ERR_FILE=$(mktemp "${{TMPDIR:-/tmp}}/openadapt-apply.XXXXXX") || exit 1 +trap 'rm -f "$ERR_FILE"' EXIT + +CHANGES=0 +FAILED=0 +CHECKED=0 +DIFFERENT=0 +PINS_PENDING=0 + +# change LABEL COMMAND...: run one write. The request body arrives on stdin. +change() {{ + local label=$1 + shift + local body + body=$(cat) + CHANGES=$((CHANGES + 1)) + if [ "$MODE" = dry-run ]; then + printf '\\n# %s\\n' "$label" + printf '%q ' "$@" + printf "<<'JSON'\\n%s\\nJSON\\n" "$body" + return 0 + fi + if printf '%s\\n' "$body" | "$@" >/dev/null 2>"$ERR_FILE"; then + printf 'changed %s\\n' "$label" + else + FAILED=$((FAILED + 1)) + printf 'FAILED %s\\n' "$label" >&2 + sed 's/^/ /' "$ERR_FILE" >&2 + fi +}} + +# read_value LABEL COMMAND...: print one value read from GitHub, or fail. +read_value() {{ + local label=$1 + shift + if ! "$@" 2>"$ERR_FILE"; then + printf 'UNREAD %s\\n' "$label" >&2 + sed 's/^/ /' "$ERR_FILE" >&2 + return 1 + fi +}} + +# expect LABEL EXPECTED COMMAND...: compare one value on GitHub with the registry. +expect() {{ + local label=$1 expected=$2 actual + shift 2 + CHECKED=$((CHECKED + 1)) + if ! actual=$(read_value "$label" "$@"); then + DIFFERENT=$((DIFFERENT + 1)) + return + fi + if [ "$actual" = "$expected" ]; then + printf 'matches %s\\n' "$label" + else + DIFFERENT=$((DIFFERENT + 1)) + printf 'DIFFERS %s\\n GitHub: %s\\n registry: %s\\n' \\ + "$label" "$actual" "$expected" >&2 + fi +}} + +check_pins() {{ + local actual + CHECKED=$((CHECKED + 1)) + # The GraphQL query names $org itself, so the shell must not expand it. + # shellcheck disable=SC2016 + if ! actual=$(read_value "organization: pinned repositories" \\ + gh api graphql -F org="$ORG" -f query={pins_query} \\ + --jq '[.data.organization.pinnedItems.nodes[].name] | join(",")'); then + DIFFERENT=$((DIFFERENT + 1)) + return + fi + if [ "$actual" = "$PINS" ]; then + printf 'matches organization: pinned repositories\\n' + else + PINS_PENDING=1 + printf 'TO PIN organization: pinned repositories (set by hand)\\n GitHub: %s\\n registry: %s\\n' \\ + "$actual" "$PINS" >&2 + fi +}} + +# Public repositories that the registry doesn't describe. A note, not a failure. +note_unregistered() {{ + local names name + if ! names=$(read_value "organization: public repositories" \\ + gh api --paginate "orgs/$ORG/repos?type=public&per_page=100" \\ + --jq '.[] | select(.archived | not) | .name'); then + return + fi + for name in $names; do + case " $REGISTERED " in + *" $name "*) ;; + *) printf 'note %s is public but has no entry in repository-lifecycle.yml\\n' "$name" ;; + esac + done +}} + +preflight() {{ + if ! command -v gh >/dev/null 2>&1; then + printf "The GitHub CLI (gh) isn't installed. Nothing was changed.\\n" >&2 + exit 1 + fi + if ! gh auth status >/dev/null 2>&1; then + printf "gh isn't signed in. Run gh auth login, then try again. Nothing was changed.\\n" >&2 + exit 1 + fi + if [ "$MODE" = apply ]; then + local role + role=$(gh api "user/memberships/orgs/$ORG" --jq .role 2>/dev/null) || role="" + if [ "$role" != admin ]; then + printf "This gh account isn't an owner of %s (role: %s). Nothing was changed.\\n" \\ + "$ORG" "${{role:-none}}" >&2 + exit 1 + fi + fi +}} +""" + +SCRIPT_TAIL = """\ + +if [ "$MODE" = dry-run ]; then + apply_changes + printf '\\nDry run: %s changes listed. Nothing was called or changed.\\n' "$CHANGES" + printf 'Run with --apply to make them, then pin the repositories by hand.\\n' + exit 0 +fi + +preflight +if [ "$MODE" = apply ]; then + apply_changes + printf '\\n' +fi +verify_github +note_unregistered + +if [ "$FAILED" -gt 0 ]; then + printf '\\n%s of %s changes failed. GitHub may be partly updated; the checks above show what differs.\\n' \\ + "$FAILED" "$CHANGES" >&2 + exit 1 +fi +if [ "$DIFFERENT" -gt 0 ]; then + printf '\\n%s of %s values on GitHub differ from the registry or could not be read.\\n' \\ + "$DIFFERENT" "$CHECKED" >&2 + exit 1 +fi +if [ "$PINS_PENDING" -gt 0 ]; then + printf '\\nEvery other value matches. Pin the repositories by hand, in this order: %s\\n' \\ + "$PINS" >&2 + exit 2 +fi +printf '\\nGitHub matches repository-lifecycle.yml. All %s values checked.\\n' "$CHECKED" +exit 0 +""" + + +def _body(payload: dict[str, object]) -> str: + # One line of JSON always starts with "{", so it can't end the heredoc. + return json.dumps(payload, ensure_ascii=True) + + +def render_apply_script( + metadata: PublicMetadata, organization: str = ORGANIZATION +) -> str: + """Render the reviewed bash script that applies `metadata` to GitHub.""" + pins_csv = ",".join(metadata.pinned_repositories) + head = SCRIPT_HEAD.format( + org=organization, + pin_lines="\n".join( + f"# {index}. {name}" + for index, name in enumerate(metadata.pinned_repositories, start=1) + ), + pins_csv=shlex.quote(pins_csv), + registered=shlex.quote(" ".join(metadata.repositories)), + pins_query=shlex.quote(PINS_QUERY), + ) + + writes = ["", "apply_changes() {"] + writes += [ + ' change "organization: description and website" \\', + ' gh api --method PATCH "orgs/$ORG" --input - <<\'JSON\'', + _body( + { + "description": metadata.organization_description, + "blog": metadata.organization_homepage, + } + ), + "JSON", + ] + for name, repository in metadata.repositories.items(): + quoted = shlex.quote(f"{name}: description and website") + writes += [ + f" change {quoted} \\", + f' gh api --method PATCH "repos/$ORG/{name}" --input - <<\'JSON\'', + _body( + { + "description": repository.description, + "homepage": repository.homepage, + } + ), + "JSON", + f" change {shlex.quote(f'{name}: topics')} \\", + f' gh api --method PUT "repos/$ORG/{name}/topics" --input - <<\'JSON\'', + _body({"names": list(repository.topics)}), + "JSON", + ] + writes.append("}") + + reads = ["", "verify_github() {"] + reads += [ + " expect 'organization: description' \\", + f" {shlex.quote(metadata.organization_description)} \\", + " gh api \"orgs/$ORG\" --jq '.description // \"\"'", + " expect 'organization: website' \\", + f" {shlex.quote(metadata.organization_homepage)} \\", + " gh api \"orgs/$ORG\" --jq '.blog // \"\"'", + ] + for name, repository in metadata.repositories.items(): + reads += [ + f" expect {shlex.quote(f'{name}: description')} \\", + f" {shlex.quote(repository.description)} \\", + f" gh api \"repos/$ORG/{name}\" --jq '.description // \"\"'", + f" expect {shlex.quote(f'{name}: website')} \\", + f" {shlex.quote(repository.homepage)} \\", + f" gh api \"repos/$ORG/{name}\" --jq '.homepage // \"\"'", + f" expect {shlex.quote(f'{name}: topics')} \\", + f" {shlex.quote(','.join(sorted(repository.topics)))} \\", + f" gh api \"repos/$ORG/{name}/topics\" --jq '.names | sort | join(\",\")'", + ] + reads += [" check_pins", "}"] + + return head + "\n".join(writes) + "\n" + "\n".join(reads) + "\n" + SCRIPT_TAIL + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + commands = parser.add_subparsers(dest="command", required=True) + render = commands.add_parser( + "render-apply", help="print the script that applies the registry to GitHub" + ) + render.add_argument( + "--output", + type=Path, + help="write the script here (mode 0755) instead of printing it", + ) + commands.add_parser("check", help="validate the registry's public metadata") + arguments = parser.parse_args(argv) + + try: + metadata = load_public_metadata() + except MetadataError as exc: + print(f"ERROR: {exc}", file=sys.stderr) + return 1 + + if arguments.command == "check": + print( + f"public_metadata is valid: {len(metadata.repositories)} repositories, " + f"{len(metadata.pinned_repositories)} pins." + ) + return 0 + + script = render_apply_script(metadata) + if arguments.output is None: + sys.stdout.write(script) + return 0 + arguments.output.parent.mkdir(parents=True, exist_ok=True) + arguments.output.write_text(script, encoding="utf-8") + arguments.output.chmod(0o755) + print(f"Wrote {arguments.output}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_benchmark_claims.py b/tests/test_benchmark_claims.py index d298630..ff35367 100644 --- a/tests/test_benchmark_claims.py +++ b/tests/test_benchmark_claims.py @@ -448,7 +448,7 @@ def test_the_published_registry_is_structurally_valid(self) -> None: def test_every_published_figure_is_bound_or_recorded(self) -> None: code, out, _ = run(ROOT, "--allow-recorded-drift") self.assertEqual(code, 0) - self.assertIn("Bound 17 published figures", out) + self.assertIn("Bound 12 published figures", out) def test_the_published_surfaces_cover_both_front_pages(self) -> None: registry = claims.load_registry(ROOT / "benchmark-claims.json") diff --git a/tests/test_github_metadata.py b/tests/test_github_metadata.py new file mode 100644 index 0000000..f2fbb81 --- /dev/null +++ b/tests/test_github_metadata.py @@ -0,0 +1,517 @@ +"""Tests for the registry's public GitHub metadata and the script that applies it.""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import tempfile +import unittest +from dataclasses import replace +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +import check_profile # noqa: E402 +import github_metadata as metadata_module # noqa: E402 +from github_metadata import ( # noqa: E402 + MetadataError, + PublicMetadata, + RepositoryMetadata, + load_public_metadata, + parse_public_metadata_mapping, + render_apply_script, + structural_errors, +) + +BASH = shutil.which("bash") + + +def published() -> PublicMetadata: + return load_public_metadata(ROOT / "repository-lifecycle.yml") + + +def product_repositories() -> set[str]: + policy = json.loads( + (ROOT / "production-lifecycle-policy.json").read_text(encoding="utf-8") + ) + return {target["source_repository"].split("/", 1)[1] for target in policy["targets"]} + + +def lifecycles() -> dict[str, str]: + errors: list[str] = [] + groups = check_profile.parse_lifecycle_groups( + (ROOT / "repository-lifecycle.yml").read_text(encoding="utf-8"), errors + ) + assert not errors, errors + return {name: group for group, names in groups.items() for name in names} + + +def with_repository(metadata: PublicMetadata, name: str, **changes: object) -> PublicMetadata: + repositories = dict(metadata.repositories) + repositories[name] = replace(repositories[name], **changes) + return replace(metadata, repositories=repositories) + + +SAMPLE = """\ +schema_version: 2 +public_metadata: + organization_description: >- + Enters approved information and checks + that it saved. + organization_homepage: https://example.com/ + profile_headline: >- + One plain line. + pinned_repositories: + - alpha + repositories: + alpha: + description: >- + Alpha's description. + homepage: https://example.com/alpha + topics: + - one + - two + .github: + description: >- + Support: the profile. + homepage: "" + topics: [] +lifecycle: + support: + - .github +""" + + +class ParserTests(unittest.TestCase): + def write(self, text: str) -> Path: + directory = tempfile.mkdtemp() + self.addCleanup(shutil.rmtree, directory, True) + path = Path(directory) / "repository-lifecycle.yml" + path.write_text(text, encoding="utf-8") + return path + + def test_the_published_registry_loads(self) -> None: + metadata = published() + self.assertEqual( + metadata.pinned_repositories, check_profile.EXPECTED_PINNED_REPOSITORIES + ) + self.assertIn("ehr-integration-directory", metadata.repositories) + + def test_the_parser_agrees_with_a_full_yaml_parser(self) -> None: + try: + import yaml + except ImportError: # pragma: no cover - PyYAML is optional here + self.skipTest("PyYAML isn't installed") + text = (ROOT / "repository-lifecycle.yml").read_text(encoding="utf-8") + self.assertEqual( + parse_public_metadata_mapping(text), yaml.safe_load(text)["public_metadata"] + ) + self.assertEqual( + parse_public_metadata_mapping(SAMPLE), yaml.safe_load(SAMPLE)["public_metadata"] + ) + + def test_folded_text_empty_values_and_lists(self) -> None: + metadata = load_public_metadata(self.write(SAMPLE)) + self.assertEqual( + metadata.organization_description, + "Enters approved information and checks that it saved.", + ) + self.assertEqual(metadata.repositories["alpha"].topics, ("one", "two")) + self.assertEqual(metadata.repositories[".github"].homepage, "") + self.assertEqual(metadata.repositories[".github"].topics, ()) + + def test_a_quoted_or_ambiguous_value_is_refused(self) -> None: + for bad in ( + " homepage: 'https://example.com'", + " homepage: https://example.com # comment", + " homepage: key: value", + ): + text = SAMPLE.replace(" homepage: https://example.com/alpha", bad) + with self.subTest(bad=bad), self.assertRaises(MetadataError): + load_public_metadata(self.write(text)) + + def test_a_duplicate_key_is_refused(self) -> None: + text = SAMPLE.replace( + " homepage: https://example.com/alpha\n", + " homepage: https://example.com/alpha\n" + " homepage: https://example.com/beta\n", + ) + with self.assertRaisesRegex(MetadataError, "duplicate key"): + load_public_metadata(self.write(text)) + + def test_a_misindented_key_is_refused(self) -> None: + text = SAMPLE.replace(" homepage: \"\"", " homepage: \"\"") + with self.assertRaises(MetadataError): + load_public_metadata(self.write(text)) + + def test_an_unknown_or_missing_key_is_refused(self) -> None: + text = SAMPLE.replace(" profile_headline: >-\n One plain line.\n", "") + with self.assertRaisesRegex(MetadataError, "keys are"): + load_public_metadata(self.write(text)) + + def test_github_and_copy_rules(self) -> None: + base = load_public_metadata(self.write(SAMPLE)) + cases = { + "too many topics": with_repository( + base, "alpha", topics=tuple(f"t{index}" for index in range(21)) + ), + "invalid topic": with_repository(base, "alpha", topics=("Not_Valid",)), + "repeated topic": with_repository(base, "alpha", topics=("one", "one")), + "em dash": with_repository(base, "alpha", description="Alpha — beta."), + "curly quote": with_repository(base, "alpha", description="Alpha’s app."), + "too long": with_repository(base, "alpha", description="x" * 161), + "http website": with_repository(base, "alpha", homepage="http://example.com"), + "pin without entry": replace(base, pinned_repositories=("alpha", "missing")), + "repeated pin": replace(base, pinned_repositories=("alpha", "alpha")), + "seven pins": replace(base, pinned_repositories=("alpha",) * 7), + } + self.assertEqual(structural_errors(base), []) + for label, metadata in cases.items(): + with self.subTest(label): + self.assertTrue(structural_errors(metadata)) + + +class CheckRuleTests(unittest.TestCase): + def errors(self, metadata: PublicMetadata, groups: dict[str, str] | None = None) -> list[str]: + return check_profile.metadata_errors( + metadata, product_repositories(), lifecycles() if groups is None else groups + ) + + def test_the_published_metadata_passes(self) -> None: + self.assertEqual(self.errors(published()), []) + + def test_a_research_repository_cannot_be_pinned(self) -> None: + metadata = replace( + published(), + pinned_repositories=published().pinned_repositories[:5] + ("openadapt-evals",), + ) + errors = self.errors(metadata) + self.assertTrue(any("pins openadapt-evals, which is research" in e for e in errors)) + self.assertTrue(any("product pins do not match" in e for e in errors)) + + def test_a_non_product_description_opens_with_its_lifecycle(self) -> None: + metadata = with_repository( + published(), "openadapt-ml", description="Optional model training." + ) + self.assertTrue( + any("must start with 'Research:'" in e for e in self.errors(metadata)) + ) + + def test_a_label_must_match_the_lifecycle(self) -> None: + metadata = with_repository( + published(), "openadapt-viewer", description="Research: a viewer." + ) + self.assertTrue(any("opens with 'Research'" in e for e in self.errors(metadata))) + + def test_a_product_description_carries_no_static_label(self) -> None: + metadata = with_repository( + published(), "openadapt-desktop", description="Experimental desktop app." + ) + self.assertTrue(any("static lifecycle labels" in e for e in self.errors(metadata))) + + def test_a_described_repository_needs_a_lifecycle(self) -> None: + groups = lifecycles() + del groups["openadapt-wright"] + self.assertTrue( + any("openadapt-wright, which is neither" in e for e in self.errors(published(), groups)) + ) + + def test_pinned_and_organization_copy_use_business_words(self) -> None: + pinned = with_repository( + published(), + "openadapt-agent", + description="Exposes governed workflows over MCP.", + ) + errors = self.errors(pinned) + self.assertTrue(any("'governed'" in e for e in errors)) + self.assertTrue(any("'MCP'" in e for e in errors)) + organization = replace( + published(), organization_description="Reports VERIFIED after an oracle read." + ) + self.assertEqual( + len([e for e in self.errors(organization) if "organization_description" in e]), 2 + ) + + def test_the_profile_business_section_uses_business_words(self) -> None: + profile = (ROOT / "profile" / "README.md").read_text(encoding="utf-8") + business, developer = profile.split(f"\n{check_profile.DEVELOPER_HEADING}\n", 1) + self.assertEqual( + check_profile.business_term_errors("business", check_profile.reader_words(business)), + [], + ) + # The technical terms live one level deeper, below the developer heading. + self.assertIn("RECONCILIATION_REQUIRED", developer) + self.assertIn("VERIFIED", developer) + + def test_link_destinations_and_code_are_not_prose(self) -> None: + text = "See [the docs](https://example.com/governed).\n```\nopenadapt mcp\n```\n" + self.assertEqual( + check_profile.business_term_errors("x", check_profile.reader_words(text)), [] + ) + + def test_the_profile_never_prints_the_misstated_fault_count(self) -> None: + profile = " ".join( + (ROOT / "profile" / "README.md").read_text(encoding="utf-8").split() + ) + self.assertNotIn("54 of 90", profile) + self.assertIn("54 of 72 (75.0%)", profile) + + def test_the_whole_profile_check_passes(self) -> None: + completed = subprocess.run( + [sys.executable, str(ROOT / "scripts" / "check_profile.py")], + capture_output=True, + text=True, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + + +FAKE_GH = """\ +#!{python} +import json, os, sys + +state_path = os.environ["FAKE_GH_STATE"] +with open(os.environ["FAKE_GH_LOG"], "a") as log: + log.write(json.dumps(sys.argv[1:]) + "\\n") +with open(state_path) as handle: + state = json.load(handle) + +def save(): + with open(state_path, "w") as handle: + json.dump(state, handle) + +args = sys.argv[1:] +if args[:2] == ["auth", "status"]: + sys.exit(0 if state.get("signed_in", True) else 1) +assert args[0] == "api", args +method, jq, path, index = "GET", None, None, 1 +while index < len(args): + arg = args[index] + if arg in ("--method", "--jq", "-f", "-F", "--input"): + if arg == "--method": + method = args[index + 1] + elif arg == "--jq": + jq = args[index + 1] + index += 2 + continue + if arg != "--paginate" and path is None: + path = arg + index += 1 + +if method + " " + path in state.get("fail", []): + print("HTTP 403: refused", file=sys.stderr) + sys.exit(1) +parts = path.split("/") +blank = {{"description": "", "homepage": "", "topics": []}} +if method == "PATCH" and parts[0] == "orgs": + state["org"].update(json.load(sys.stdin)) +elif method == "PATCH" and parts[0] == "repos": + state["repos"].setdefault(parts[2], dict(blank)).update(json.load(sys.stdin)) +elif method == "PUT" and parts[-1] == "topics": + state["repos"].setdefault(parts[2], dict(blank))["topics"] = json.load(sys.stdin)["names"] +elif path.startswith("user/memberships/orgs/"): + print(state["role"]) + sys.exit(0) +elif path == "graphql": + assert jq == '[.data.organization.pinnedItems.nodes[].name] | join(",")', jq + print(",".join(state["pins"])) + sys.exit(0) +elif path.endswith("/repos?type=public&per_page=100"): + print("\\n".join(state["public"])) + sys.exit(0) +elif parts[0] == "orgs": + field = {{'.description // ""': "description", '.blog // ""': "blog"}}[jq] + print(state["org"].get(field) or "") + sys.exit(0) +elif parts[0] == "repos": + repository = state["repos"].get(parts[2]) + if repository is None: + print("HTTP 404: Not Found", file=sys.stderr) + sys.exit(1) + if parts[-1] == "topics": + assert jq == '.names | sort | join(",")', jq + print(",".join(sorted(repository["topics"]))) + else: + field = {{'.description // ""': "description", '.homepage // ""': "homepage"}}[jq] + print(repository.get(field) or "") + sys.exit(0) +else: + print("unexpected call: " + " ".join(args), file=sys.stderr) + sys.exit(3) +save() +""" + +EXAMPLE = PublicMetadata( + organization_description="Enters approved information and checks that it saved.", + organization_homepage="https://example.com/", + profile_headline="One plain line.", + pinned_repositories=("alpha", "beta"), + repositories={ + "alpha": RepositoryMetadata( + "alpha", "Alpha's description.", "https://example.com/alpha", ("two", "one") + ), + "beta": RepositoryMetadata("beta", "Support: beta.", "", ()), + }, +) + + +@unittest.skipIf(BASH is None, "bash isn't installed") +class ApplyScriptTests(unittest.TestCase): + def setUp(self) -> None: + directory = tempfile.mkdtemp() + self.addCleanup(shutil.rmtree, directory, True) + self.directory = Path(directory) + bin_directory = self.directory / "bin" + bin_directory.mkdir() + gh = bin_directory / "gh" + gh.write_text(FAKE_GH.format(python=sys.executable), encoding="utf-8") + gh.chmod(0o755) + self.state_path = self.directory / "state.json" + self.log_path = self.directory / "calls.log" + self.log_path.write_text("", encoding="utf-8") + self.script = self.directory / "APPLY.sh" + self.script.write_text( + render_apply_script(EXAMPLE, organization="ExampleOrg"), encoding="utf-8" + ) + self.script.chmod(0o755) + self.env = { + "PATH": f"{bin_directory}{os.pathsep}/usr/bin{os.pathsep}/bin", + "TMPDIR": str(self.directory), + "FAKE_GH_STATE": str(self.state_path), + "FAKE_GH_LOG": str(self.log_path), + } + self.set_state() + + def set_state(self, **changes: object) -> None: + state = { + "role": "admin", + "org": {"description": "Old.", "blog": "https://example.com/"}, + "repos": { + "alpha": {"description": "Old alpha.", "homepage": "", "topics": ["old"]}, + "beta": {"description": "", "homepage": "https://old.example", "topics": []}, + }, + "pins": ["alpha", "beta"], + "public": ["alpha", "beta", "stray-fork"], + "fail": [], + } + state.update(changes) + self.state_path.write_text(json.dumps(state), encoding="utf-8") + + def run_script(self, *arguments: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [BASH, str(self.script), *arguments], + capture_output=True, + text=True, + env=self.env, + timeout=120, + ) + + def calls(self) -> list[list[str]]: + return [ + json.loads(line) + for line in self.log_path.read_text(encoding="utf-8").splitlines() + ] + + def writes(self) -> list[list[str]]: + return [call for call in self.calls() if "--method" in call] + + def test_the_committed_script_matches_the_registry(self) -> None: + script = metadata_module.APPLY_SCRIPT + self.assertTrue(script.exists(), "render it with github_metadata.py render-apply") + self.assertEqual(script.read_text(encoding="utf-8"), render_apply_script(published())) + self.assertTrue(os.access(script, os.X_OK)) + completed = subprocess.run([BASH, "-n", str(script)], capture_output=True, text=True) + self.assertEqual(completed.returncode, 0, completed.stderr) + + def test_the_committed_script_dry_run_calls_nothing(self) -> None: + completed = subprocess.run( + [BASH, str(metadata_module.APPLY_SCRIPT)], + capture_output=True, + text=True, + env=self.env, + timeout=60, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + self.assertEqual(self.calls(), []) + for name in published().repositories: + self.assertIn(f"# {name}: description and website", completed.stdout) + self.assertIn("Nothing was called or changed.", completed.stdout) + + def test_a_dry_run_prints_exact_commands_and_bodies(self) -> None: + completed = self.run_script("--dry-run") + self.assertEqual(completed.returncode, 0, completed.stderr) + self.assertEqual(self.calls(), []) + self.assertIn( + "gh api --method PATCH repos/ExampleOrg/alpha --input - <<'JSON'\n" + '{"description": "Alpha\'s description.", "homepage": "https://example.com/alpha"}\n' + "JSON\n", + completed.stdout, + ) + self.assertIn('{"names": ["two", "one"]}', completed.stdout) + + def test_apply_writes_reads_back_and_reports_success_only_when_all_match(self) -> None: + completed = self.run_script("--apply") + self.assertEqual(completed.returncode, 0, completed.stdout + completed.stderr) + state = json.loads(self.state_path.read_text(encoding="utf-8")) + self.assertEqual(state["repos"]["alpha"]["description"], "Alpha's description.") + self.assertEqual(state["repos"]["alpha"]["topics"], ["two", "one"]) + self.assertEqual(state["repos"]["beta"]["homepage"], "") + self.assertEqual(state["org"]["description"], EXAMPLE.organization_description) + self.assertIn("GitHub matches repository-lifecycle.yml", completed.stdout) + self.assertIn("note stray-fork is public", completed.stdout) + self.assertEqual(len(self.writes()), 5) + + def test_one_failed_write_fails_the_run_and_names_the_change(self) -> None: + self.set_state(fail=["PATCH repos/ExampleOrg/alpha"]) + completed = self.run_script("--apply") + self.assertEqual(completed.returncode, 1) + self.assertIn("FAILED alpha: description and website", completed.stderr) + self.assertIn("HTTP 403: refused", completed.stderr) + self.assertIn("DIFFERS alpha: description", completed.stderr) + self.assertIn("1 of 5 changes failed", completed.stderr) + self.assertNotIn("GitHub matches", completed.stdout) + + def test_an_unreadable_value_is_not_a_match(self) -> None: + self.set_state(fail=["GET repos/ExampleOrg/beta/topics"]) + completed = self.run_script("--verify") + self.assertEqual(completed.returncode, 1) + self.assertIn("UNREAD beta: topics", completed.stderr) + self.assertNotIn("GitHub matches", completed.stdout) + + def test_pins_left_for_an_owner_exit_two(self) -> None: + self.set_state(pins=["beta", "alpha"]) + completed = self.run_script("--apply") + self.assertEqual(completed.returncode, 2, completed.stdout + completed.stderr) + self.assertIn("TO PIN", completed.stderr) + self.assertNotIn("GitHub matches", completed.stdout) + + def test_apply_refuses_an_account_that_is_not_an_owner(self) -> None: + self.set_state(role="member") + completed = self.run_script("--apply") + self.assertEqual(completed.returncode, 1) + self.assertIn("isn't an owner of ExampleOrg", completed.stderr) + self.assertEqual(self.writes(), []) + + def test_apply_refuses_when_gh_is_signed_out(self) -> None: + self.set_state(signed_in=False) + completed = self.run_script("--apply") + self.assertEqual(completed.returncode, 1) + self.assertEqual(self.writes(), []) + + def test_verify_never_writes(self) -> None: + completed = self.run_script("--verify") + self.assertEqual(completed.returncode, 1) + self.assertEqual(self.writes(), []) + self.assertIn("DIFFERS organization: description", completed.stderr) + + def test_an_unknown_option_is_a_usage_error(self) -> None: + self.assertEqual(self.run_script("--force").returncode, 64) + self.assertEqual(self.run_script("--apply", "--verify").returncode, 64) + self.assertEqual(self.calls(), []) + + +if __name__ == "__main__": + unittest.main()