From 056da1af8a856d97746f5119c15b51916e97cc14 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Thu, 1 Oct 2026 09:25:53 +0900 Subject: [PATCH] Backport #906: Run SQLAlchemy and Spark tests when shared core modules change The Test workflow's changes job now also selects the SQLAlchemy and Spark tests when a pull request changes a top-level module of pyathena or pyathena.aio, or a shared test fixture (the tests and tests/pyathena(/aio) package initializers and conftest files, tests/pyathena/tables.py, tests/pyathena/util.py, tests/resources/). Changes limited to other subpackages still skip them. The selection now uses dorny/paths-filter v4.0.3 (SHA-pinned) instead of gh api and grep; other events still run every suite. (cherry picked from commit 9b9421d8464fec020841c2ebda374caac8d6aad6) The merge commit applies cleanly; the added and removed lines match #906 exactly. 3.x has no tests/pyathena/tables.py, so that pattern matches nothing there. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/test.yaml | 80 ++++++++++++++++++++++--------------- 1 file changed, 48 insertions(+), 32 deletions(-) diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml index 4f2cf48a..d293b133 100644 --- a/.github/workflows/test.yaml +++ b/.github/workflows/test.yaml @@ -61,9 +61,10 @@ jobs: # requests run none. A ready pull request always runs the PyAthena suite; it # runs the SQLAlchemy tests (the compliance suites and the PyAthena suite's # SQLAlchemy tests) and the Spark tests only when their code, tests, - # dependencies, or this workflow change. Pull requests and the schedule test - # the newest Python version; a dispatch tests the requested versions or - # every version, and the Release workflow every version. + # dependencies, this workflow, the top-level modules of pyathena and + # pyathena.aio, or the shared test fixtures change. Pull requests and the + # schedule test the newest Python version; a dispatch tests the requested + # versions or every version, and the Release workflow every version. changes: if: >- github.event_name != 'pull_request' || @@ -73,16 +74,53 @@ jobs: permissions: pull-requests: read outputs: - sqla: ${{ steps.filter.outputs.sqla }} - spark: ${{ steps.filter.outputs.spark }} - python-versions: ${{ steps.filter.outputs.python-versions }} + sqla: ${{ github.event_name != 'pull_request' || steps.paths.outputs.sqla == 'true' }} + spark: ${{ github.event_name != 'pull_request' || steps.paths.outputs.spark == 'true' }} + python-versions: ${{ steps.versions.outputs.python-versions }} steps: - - id: filter + - id: paths + if: github.event_name == 'pull_request' + uses: dorny/paths-filter@ceb8a2b8f2d89434be7ff52d3de7ec3738c5cc9d # v4.0.3 + with: + filters: | + shared: &shared + - .github/workflows/test.yaml + - .github/workflows/test-suite.yaml + - justfile + - pyproject.toml + - uv.lock + # The SQLAlchemy and Spark packages load the top-level modules, + # directly or transitively, and their tests use the shared fixtures. + core: &core + - pyathena/*.py + - pyathena/aio/*.py + - tests/__init__.py + - tests/pyathena/__init__.py + - tests/pyathena/conftest.py + - tests/pyathena/tables.py + - tests/pyathena/util.py + - tests/pyathena/aio/__init__.py + - tests/pyathena/aio/conftest.py + - tests/resources/** + sqla: + - *shared + - *core + - setup.cfg + - pyathena/sqlalchemy/** + - pyathena/aio/sqlalchemy/** + - tests/sqlalchemy/** + - tests/pyathena/sqlalchemy/** + - tests/pyathena/aio/sqlalchemy/** + spark: + - *shared + - *core + - pyathena/spark/** + - pyathena/aio/spark/** + - tests/pyathena/spark/** + - tests/pyathena/aio/spark/** + - id: versions env: - GH_TOKEN: ${{ github.token }} EVENT_NAME: ${{ github.event_name }} - REPO: ${{ github.repository }} - PR_NUMBER: ${{ github.event.pull_request.number }} REQUESTED_VERSIONS: ${{ inputs.python-versions }} # Every supported version, oldest first; keep in sync with the # pyproject.toml classifiers. @@ -106,28 +144,6 @@ jobs: ;; esac echo "python-versions=$versions" >> "$GITHUB_OUTPUT" - if [[ "$EVENT_NAME" != "pull_request" ]]; then - { - echo "sqla=true" - echo "spark=true" - } >> "$GITHUB_OUTPUT" - exit 0 - fi - files=$(gh api "repos/$REPO/pulls/$PR_NUMBER/files" --paginate --jq '.[].filename') - printf 'Changed files:\n%s\n' "$files" - shared='^(\.github/workflows/test(-suite)?\.yaml|justfile|pyproject\.toml|uv\.lock)$' - sqla="$shared|^pyathena/(aio/)?sqlalchemy/|^tests/sqlalchemy/|^tests/pyathena/(aio/)?sqlalchemy/|^setup\.cfg$" - spark="$shared|^pyathena/(aio/)?spark/|^tests/pyathena/(aio/)?spark/" - if grep -qE "$sqla" <<< "$files"; then - echo "sqla=true" >> "$GITHUB_OUTPUT" - else - echo "sqla=false" >> "$GITHUB_OUTPUT" - fi - if grep -qE "$spark" <<< "$files"; then - echo "spark=true" >> "$GITHUB_OUTPUT" - else - echo "spark=false" >> "$GITHUB_OUTPUT" - fi # The three suites create their own schemas and tables, so they run in # parallel; each is still a separate job for "Re-run failed jobs".