mirror of
https://github.com/apache/superset.git
synced 2026-09-05 23:12:01 +00:00
Stacked on feat/testcontainers-nightly-only-gating. All six extras already
existed in pyproject.toml. oceanbase and vertica run nightly_only: true
(heavy first-boot and a ~12GB RAM floor, respectively), so they don't run
per-PR; databend/risingwave/firebird/ydb run on every PR like the rest of
this suite.
- oceanbase_py pins sqlalchemy-utils<0.39, which conflicts outright with
Superset's own sqlalchemy-utils==0.42.1 pin -- kept out of the baseline
dev install (same reason as db2's ibm-db-sa) and installed on demand,
--no-deps, only for its own CI leg (it never actually imports
sqlalchemy_utils itself, so the version mismatch is inert at runtime).
- databend: connects to the local standalone image's builtin `root` user
(no password) with sslmode=disable, since Superset's default
encryption_parameters assume TLS the local image doesn't have.
- risingwave: RisingWave's storage engine checkpoints asynchronously --
a SELECT immediately after INSERT can see zero rows without an explicit
FLUSH (confirmed on a real instance). Uses the shared _pagination.py
helper's after_insert hook (originally added for CrateDB) to do that.
- firebird: sqlalchemy-firebird's driver is a pure-Python ctypes wrapper
(py3-none-any wheel, confirmed by downloading it directly) that
dynamically loads the native libfbclient from the host rather than
bundling it -- CI installs that system package on demand. Also confirms
in the test docstring that FirebirdEngineSpec's `limit_method =
LimitMethod.FETCH_MANY` (comment: "uses FIRST to limit") is stale
against the modern driver, which compiles real ROWS-based pagination.
- ydb: needed three real fixes to make a generic DockerContainer usable
at all. (1) YDB's gRPC client does endpoint discovery and reconnects to
whatever the server reports, which by default is the container's own
internal Docker hostname -- fixed by binding the same port on the host
as inside the container and advertising "localhost" as the container's
own hostname, so the discovered endpoint is actually reachable. (2) The
gRPC port opens before storage pools are fully initialized, so an early
CREATE TABLE fails; the fixture retries a real metadata.create_all()
probe rather than trusting the open port. (3) YDB rejects DDL inside an
explicit transaction ("Scheme operations cannot be executed inside
transaction") -- confirmed this only affects a raw text("CREATE
TABLE..."), not metadata.create_all()'s own DDL execution path, which
already does the right thing.
188 lines
7.8 KiB
YAML
188 lines
7.8 KiB
YAML
# db_engine_specs tests against real databases (testcontainers)
|
|
name: Testcontainers
|
|
|
|
# Spins up real Docker containers (see tests/testcontainers/ for the current
|
|
# dialect list) via testcontainers-python, which catches real dialect/driver
|
|
# regressions -- the kind mocked db_engine_specs unit tests structurally
|
|
# cannot, e.g. apache/superset#42899 (Trino emitting OFFSET before LIMIT).
|
|
# Runs on a nightly cron (catches drift from a driver's own releases, not
|
|
# just from Superset's changes) and on pull_request, scoped via `paths` to
|
|
# only PRs that actually touch this test suite or the workflow itself, so
|
|
# unrelated PRs across the repo are never affected.
|
|
#
|
|
# A matrix entry can set `nightly_only: true` to run only on the cron (or a
|
|
# manual workflow_dispatch), never on pull_request -- for a dialect whose
|
|
# image is too heavy (a multi-service cluster, a many-GB image, a slow
|
|
# licensed installer) to justify adding its wall-clock/resource cost to
|
|
# every PR that merely touches this suite. Omit the field entirely for a
|
|
# normal dialect; it isn't nightly-only by default.
|
|
permissions:
|
|
contents: read
|
|
|
|
on:
|
|
schedule:
|
|
- cron: "0 5 * * *"
|
|
workflow_dispatch: {}
|
|
pull_request:
|
|
paths:
|
|
- ".github/workflows/testcontainers.yml"
|
|
- "tests/testcontainers/**"
|
|
- "superset/db_engine_specs/**"
|
|
- "pyproject.toml"
|
|
- "requirements/development.in"
|
|
- "requirements/development.txt"
|
|
|
|
concurrency:
|
|
# Scoped by ref, not just workflow name -- otherwise every PR run and the
|
|
# nightly cron share one group, and starting the workflow on another PR
|
|
# (or the nightly firing mid-PR-run) cancels an unrelated in-progress run.
|
|
group: ${{ github.workflow }}-${{ github.ref }}
|
|
cancel-in-progress: true
|
|
|
|
jobs:
|
|
testcontainers:
|
|
runs-on: ubuntu-26.04
|
|
strategy:
|
|
fail-fast: false
|
|
matrix:
|
|
include:
|
|
# One job per dialect rather than one job for the whole suite: a
|
|
# single slow container would otherwise inflate the wall-clock
|
|
# time for every dialect, not just its own. Running in parallel
|
|
# means the suite's total time is bounded by the slowest dialect,
|
|
# not the sum of all of them. Db2's first-boot init is documented
|
|
# upstream as notably slow (a real instance bring-up, not just a
|
|
# process start) and untested locally here (no arm64 image), so
|
|
# it gets a wider timeout margin than the rest until real CI data
|
|
# says otherwise.
|
|
- dialect: cockroachdb
|
|
timeout: 10
|
|
- dialect: crate
|
|
timeout: 10
|
|
- dialect: trino
|
|
timeout: 10
|
|
- dialect: mssql
|
|
timeout: 10
|
|
- dialect: elasticsearch
|
|
timeout: 10
|
|
- dialect: oracle
|
|
timeout: 15
|
|
- dialect: db2
|
|
timeout: 25
|
|
- dialect: mariadb
|
|
timeout: 10
|
|
- dialect: timescaledb
|
|
timeout: 10
|
|
- dialect: yugabytedb
|
|
timeout: 10
|
|
- dialect: monetdb
|
|
timeout: 10
|
|
- dialect: mongodb
|
|
timeout: 10
|
|
- dialect: postgres
|
|
timeout: 10
|
|
- dialect: mysql
|
|
timeout: 10
|
|
- dialect: clickhouse
|
|
timeout: 10
|
|
# StarRocks' allin1-ubuntu image brings up both FE and BE in one
|
|
# container, which is a heavier bring-up than a single-process
|
|
# database -- wider margin until real CI data says otherwise.
|
|
- dialect: starrocks
|
|
timeout: 15
|
|
- dialect: databend
|
|
timeout: 10
|
|
- dialect: risingwave
|
|
timeout: 10
|
|
- dialect: firebird
|
|
timeout: 10
|
|
- dialect: ydb
|
|
timeout: 10
|
|
# OceanBase bootstraps a distributed-style cluster even in
|
|
# single-node MODE=MINI, and Vertica Community Edition has a
|
|
# well-documented ~12GB RAM floor to even start -- both too heavy
|
|
# for every PR's CI budget, so both run on the nightly cron /
|
|
# manual dispatch only.
|
|
- dialect: oceanbase
|
|
timeout: 20
|
|
nightly_only: true
|
|
- dialect: vertica
|
|
timeout: 20
|
|
nightly_only: true
|
|
timeout-minutes: ${{ matrix.timeout }}
|
|
env:
|
|
PYTHONPATH: ${{ github.workspace }}
|
|
SUPERSET_TESTENV: true
|
|
SUPERSET_SECRET_KEY: not-a-secret
|
|
# This job's matrix installs exactly one dialect's testcontainers
|
|
# driver for exactly this job, so treat that driver as required: a
|
|
# broken/missing import should fail the job, not silently skip to a
|
|
# misleadingly green, zero-tests-run result. See _driver.py.
|
|
SUPERSET_TESTCONTAINERS_STRICT: true
|
|
steps:
|
|
- name: Checkout
|
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
with:
|
|
persist-credentials: false
|
|
- name: Setup Python
|
|
uses: ./.github/actions/setup-backend/
|
|
with:
|
|
python-version: current
|
|
- name: Install db2 driver (ibm-db-sa)
|
|
# ibm-db (the db2 DBAPI) ships no Linux arm64 wheel, so it's kept out
|
|
# of the baseline dev install (requirements/development.in) to avoid
|
|
# breaking the multi-platform dev Docker image build. Install it here
|
|
# instead, only for this leg of the matrix.
|
|
if: matrix.dialect == 'db2'
|
|
run: uv pip install --system -e .[db2]
|
|
- name: Install oceanbase driver (oceanbase_py)
|
|
# oceanbase_py pins sqlalchemy-utils>=0.38.3,<0.39, which conflicts
|
|
# outright with Superset's own sqlalchemy-utils==0.42.1 pin -- kept
|
|
# out of the baseline dev install for the same reason as db2 above.
|
|
# --no-deps sidesteps that pin entirely: this job only needs
|
|
# oceanbase_py's dialect module importable, not its sqlalchemy-utils
|
|
# dependency satisfied, since nothing here calls into it.
|
|
if: matrix.dialect == 'oceanbase'
|
|
run: uv pip install --system --no-deps -e .[oceanbase]
|
|
- name: Install Firebird client library (libfbclient2)
|
|
# sqlalchemy-firebird's driver (firebird-driver) is a pure-Python
|
|
# ctypes wrapper (its wheel is py3-none-any) that dynamically loads
|
|
# the native Firebird client library from the host at import time
|
|
# -- it doesn't bundle that library itself, so it has to come from
|
|
# the system package manager, only for this leg of the matrix.
|
|
if: matrix.dialect == 'firebird'
|
|
run: |
|
|
sudo apt-get update
|
|
sudo apt-get install -y libfbclient2
|
|
- name: Run testcontainers db_engine_specs tests (${{ matrix.dialect }})
|
|
# A job-level `if:` can't reference `matrix` (only github/inputs/
|
|
# needs/vars are available there), so the nightly_only skip has to
|
|
# live on the step instead. A dialect without `nightly_only` set
|
|
# evaluates the left side true (unset is null, and `null != true`
|
|
# is true) and always runs; one WITH it set only runs on the cron
|
|
# or a manual dispatch, never on pull_request.
|
|
if: >-
|
|
matrix.nightly_only != true ||
|
|
github.event_name == 'schedule' ||
|
|
github.event_name == 'workflow_dispatch'
|
|
run: |
|
|
pytest --durations-min=2 -v -m testcontainers \
|
|
./tests/testcontainers/db_engine_specs/test_${{ matrix.dialect }}.py \
|
|
--junit-xml=test-results/junit-testcontainers-${{ matrix.dialect }}.xml
|
|
- name: Upload JUnit test results
|
|
if: always()
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: junit-results-testcontainers-${{ matrix.dialect }}
|
|
path: test-results/
|
|
retention-days: 7
|
|
|
|
actions-timeline:
|
|
needs: [testcontainers]
|
|
if: always()
|
|
runs-on: ubuntu-26.04
|
|
permissions:
|
|
actions: read
|
|
steps:
|
|
- uses: Kesin11/actions-timeline@57fc93f20c6da7fbc14063c6d24a2a5627c799ad # v3.2.0
|