From 6aa2432998b77a93fefe67205127923a01b7fa09 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 5 Oct 2026 18:52:22 +0000 Subject: [PATCH] Add arrow-odbc 10.4.2 riscv64 wheel build arrow-odbc-py is a maturin cffi-bindings crate over the arrow-odbc Rust crate, which links unixODBC's driver manager at build and run time. unixODBC-devel, postgresql-server and postgresql-odbc are all packaged for riscv64 in Rocky 10's own repodata (same as build-pyodbc.yml already relies on), so the build installs a stable Rust toolchain plus unixODBC-devel in-container and runs with one CIBW_BUILD identifier list (py3-none, gotcha 479) instead of a per-interpreter matrix. Upstream's own test suite is hardwired to a Microsoft SQL Server service unavailable on riscv64, so this stages a riscv64 smoke-test file that exercises the same read/insert/error-handling surface against a PostgreSQL server started once in CIBW_BEFORE_ALL_LINUX. --- .github/workflows/build-arrow-odbc.yml | 218 +++++++++++++++++++++++++ docs/packages/arrow-odbc.yaml | 5 + 2 files changed, 223 insertions(+) create mode 100644 .github/workflows/build-arrow-odbc.yml create mode 100644 docs/packages/arrow-odbc.yaml diff --git a/.github/workflows/build-arrow-odbc.yml b/.github/workflows/build-arrow-odbc.yml new file mode 100644 index 00000000000..61232b7c815 --- /dev/null +++ b/.github/workflows/build-arrow-odbc.yml @@ -0,0 +1,218 @@ +# SPDX-FileCopyrightText: 2026 The RISE Project +# SPDX-License-Identifier: MIT +--- +name: Build arrow-odbc wheels (riscv64) + +on: + workflow_dispatch: + inputs: + version: + description: 'Version glob to (re)build; empty builds every version of docs/packages/arrow-odbc.yaml not released yet' + required: false + default: '' + pull_request: + branches: [main] + paths: + - '.github/workflows/build-arrow-odbc.yml' + - 'docs/packages/arrow-odbc.yaml' + push: + branches: [main] + paths: + - '.github/workflows/build-arrow-odbc.yml' + - 'docs/packages/arrow-odbc.yaml' + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }} + cancel-in-progress: true + +permissions: + contents: read # to fetch code (actions/checkout) + +env: + MANYLINUX_RISCV64_IMAGE: quay.io/pypa/manylinux_2_39_riscv64 + +jobs: + setup: + uses: $/.github/workflows/_setup.yml + with: + package: arrow-odbc + version: ${{ inputs.version }} + + build_wheels: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + name: Build arrow-odbc ${{ matrix.version }} py3-none-manylinux_riscv64 + runs-on: ubuntu-24.04-riscv + timeout-minutes: 180 + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + + env: + ARROW_ODBC_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout arrow-odbc-py v${{ env.ARROW_ODBC_VERSION }} + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: pacman82/arrow-odbc-py + ref: v${{ env.ARROW_ODBC_VERSION }} + persist-credentials: false + + - name: Stage the riscv64 smoke tests + # Upstream's own tests/test_arrow_odbc.py is hardwired to a Microsoft + # SQL Server service; neither the mssql/server docker image nor the + # ODBC Driver for SQL Server publish a riscv64 build. PostgreSQL and + # its ODBC driver both package for riscv64 (same as build-pyodbc.yml + # relies on), so this exercises the same public surface against a + # PostgreSQL server started inside the build container. + run: | + cat > tests/test_riscv64_smoke.py <<'EOF' + import pyarrow as pa + import pytest + from arrow_odbc import ( + Error, + from_table_to_db, + insert_into_table, + log_to_stderr, + read_arrow_batches_from_odbc, + ) + + POSTGRES = "DRIVER=PostgreSQL;SERVER=127.0.0.1;PORT=5432;UID=postgres;DATABASE=test" + + log_to_stderr() + + + def run_statement(statement): + # read_arrow_batches_from_odbc also executes statements that + # produce no result set (upstream's own test_no_result_set + # relies on the same behavior for an INSERT); draining the + # reader runs DDL/DML like CREATE, DROP and INSERT. + reader = read_arrow_batches_from_odbc( + query=statement, batch_size=1, connection_string=POSTGRES + ) + list(reader) + + + def test_select_literal(): + reader = read_arrow_batches_from_odbc( + query="SELECT 1 AS a", connection_string=POSTGRES + ) + batches = list(reader) + assert len(batches) == 1 + assert batches[0].column("a").to_pylist() == [1] + + + def test_should_report_error_on_invalid_connection_string(): + # The driver manager reports this error identically regardless + # of which driver would have been selected, so this is portable + # from upstream's own + # test_should_report_error_on_invalid_connection_string_reading. + with pytest.raises(Error, match="Data source name not found"): + read_arrow_batches_from_odbc( + query="SELECT * FROM Table", batch_size=100, connection_string="foo" + ) + + + def test_from_table_to_db_and_read_back_roundtrip(): + table = "RiscvRoundtrip" + run_statement(f"DROP TABLE IF EXISTS {table}") + run_statement(f"CREATE TABLE {table} (a INTEGER, b TEXT)") + + source = pa.table({"a": [1, 2, 3], "b": ["x", "y", "z"]}) + from_table_to_db(source=source, target=table, connection_string=POSTGRES, chunk_size=2) + + reader = read_arrow_batches_from_odbc( + query=f"SELECT a, b FROM {table} ORDER BY a", connection_string=POSTGRES + ) + result = pa.Table.from_batches(list(reader)) + assert result.column("a").to_pylist() == [1, 2, 3] + assert result.column("b").to_pylist() == ["x", "y", "z"] + + + def test_insert_into_table_with_record_batch_reader(): + table = "RiscvInsertIntoTable" + run_statement(f"DROP TABLE IF EXISTS {table}") + run_statement(f"CREATE TABLE {table} (a INTEGER)") + + source = pa.table({"a": [10, 20, 30]}) + batch_reader = pa.RecordBatchReader.from_batches(source.schema, source.to_batches()) + insert_into_table( + reader=batch_reader, chunk_size=1000, table=table, connection_string=POSTGRES + ) + + reader = read_arrow_batches_from_odbc( + query=f"SELECT a FROM {table} ORDER BY a", connection_string=POSTGRES + ) + result = pa.Table.from_batches(list(reader)) + assert result.column("a").to_pylist() == [10, 20, 30] + EOF + + - name: Build wheels + uses: pypa/cibuildwheel@1828c10ab37f080699c7b81cea34097c684a7074 # v4.2.0 + with: + output-dir: wheelhouse/ + env: + # maturin's cffi bindings reach the Rust cdylib through cffi's ABI + # mode, so the wheel is py3-none: one build serves every + # interpreter, and the further identifiers only re-run the test + # command against it. + CIBW_BUILD: cp312-manylinux_riscv64 cp313-manylinux_riscv64 cp314-manylinux_riscv64 cp314t-manylinux_riscv64 + CIBW_MANYLINUX_RISCV64_IMAGE: ${{ env.MANYLINUX_RISCV64_IMAGE }} + # unixODBC-devel is the link-time half of the same unixODBC that + # build-pyodbc.yml installs; postgresql-server/-odbc back the + # smoke tests above and are started once here so the running + # server survives the later, build-skipped test-only legs. + CIBW_BEFORE_ALL_LINUX: >- + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && + dnf -y install unixODBC-devel postgresql-server postgresql-odbc && + install -d -o postgres -g postgres /run/postgresql /var/lib/pgsql/data && + su postgres -c "/usr/bin/initdb -A trust -D /var/lib/pgsql/data" && + su postgres -c "/usr/bin/pg_ctl -D /var/lib/pgsql/data -l /tmp/pg.log -o '-c listen_addresses=127.0.0.1' -w start" && + su postgres -c "/usr/bin/createdb test" + CIBW_ENVIRONMENT_LINUX: >- + PATH="$PATH:$HOME/.cargo/bin" + PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + CIBW_REPAIR_WHEEL_COMMAND_LINUX: >- + auditwheel repair + --exclude "libodbc.so.*" + --wheel-dir {dest_dir} + {wheel} + CIBW_TEST_REQUIRES: pytest + CIBW_TEST_SOURCES: tests + CIBW_TEST_COMMAND: python -m pytest -v tests/test_riscv64_smoke.py + + - name: Check the wheel carries the Rust cdylib and its licence + run: | + python3 - wheelhouse/*.whl <<'EOF' + import sys, zipfile + for whl in sys.argv[1:]: + names = zipfile.ZipFile(whl).namelist() + assert any("arrow_odbc" in n and n.endswith(".so") for n in names), names + assert any(n.endswith(".dist-info/licenses/LICENSE") for n in names), names + print(whl, "ok") + EOF + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: arrow-odbc-${{ env.ARROW_ODBC_VERSION }}-py3-none-manylinux_riscv64 + path: wheelhouse/*.whl + if-no-files-found: error + + publish: + name: Publish arrow-odbc ${{ matrix.version }} + needs: [setup, build_wheels] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + permissions: + contents: write + pull-requests: write + uses: $/.github/workflows/_publish-wheel.yml + secrets: + app-private-key: ${{ secrets.RISEPROJECT_APP_PRIVATE_KEY }} + with: + artifact-pattern: arrow-odbc-${{ matrix.version }}-py3-none-manylinux_riscv64 diff --git a/docs/packages/arrow-odbc.yaml b/docs/packages/arrow-odbc.yaml new file mode 100644 index 00000000000..8a5c1f1be3a --- /dev/null +++ b/docs/packages/arrow-odbc.yaml @@ -0,0 +1,5 @@ +package-name: arrow-odbc +source-code: https://github.com/pacman82/arrow-odbc-py +license: MIT +versions: +- version: 10.4.2