Files
superset2/tests/testcontainers/db_engine_specs/test_risingwave.py
T
Superset Dev 05e8d24d9d feat(ci): expand testcontainers coverage to databend, risingwave, firebird, ydb, oceanbase, vertica
Stacked on feat/testcontainers-nightly-only-gating. All six extras already
existed in pyproject.toml. oceanbase and vertica run nightly_only: true
(heavy first-boot and a ~12GB RAM floor, respectively), so they don't run
per-PR; databend/risingwave/firebird/ydb run on every PR like the rest of
this suite.

- oceanbase_py pins sqlalchemy-utils<0.39, which conflicts outright with
  Superset's own sqlalchemy-utils==0.42.1 pin -- kept out of the baseline
  dev install (same reason as db2's ibm-db-sa) and installed on demand,
  --no-deps, only for its own CI leg (it never actually imports
  sqlalchemy_utils itself, so the version mismatch is inert at runtime).
- databend: connects to the local standalone image's builtin `root` user
  (no password) with sslmode=disable, since Superset's default
  encryption_parameters assume TLS the local image doesn't have.
- risingwave: RisingWave's storage engine checkpoints asynchronously --
  a SELECT immediately after INSERT can see zero rows without an explicit
  FLUSH (confirmed on a real instance). Uses the shared _pagination.py
  helper's after_insert hook (originally added for CrateDB) to do that.
- firebird: sqlalchemy-firebird's driver is a pure-Python ctypes wrapper
  (py3-none-any wheel, confirmed by downloading it directly) that
  dynamically loads the native libfbclient from the host rather than
  bundling it -- CI installs that system package on demand. Also confirms
  in the test docstring that FirebirdEngineSpec's `limit_method =
  LimitMethod.FETCH_MANY` (comment: "uses FIRST to limit") is stale
  against the modern driver, which compiles real ROWS-based pagination.
- ydb: needed three real fixes to make a generic DockerContainer usable
  at all. (1) YDB's gRPC client does endpoint discovery and reconnects to
  whatever the server reports, which by default is the container's own
  internal Docker hostname -- fixed by binding the same port on the host
  as inside the container and advertising "localhost" as the container's
  own hostname, so the discovered endpoint is actually reachable. (2) The
  gRPC port opens before storage pools are fully initialized, so an early
  CREATE TABLE fails; the fixture retries a real metadata.create_all()
  probe rather than trusting the open port. (3) YDB rejects DDL inside an
  explicit transaction ("Scheme operations cannot be executed inside
  transaction") -- confirmed this only affects a raw text("CREATE
  TABLE..."), not metadata.create_all()'s own DDL execution path, which
  already does the right thing.
2026-08-28 18:35:55 -07:00

128 lines
4.7 KiB
Python

# Licensed to the Apache Software Foundation (ASF) under one
# or more contributor license agreements. See the NOTICE file
# distributed with this work for additional information
# regarding copyright ownership. The ASF licenses this file
# to you under the Apache License, Version 2.0 (the
# "License"); you may not use this file except in compliance
# with the License. You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing,
# software distributed under the License is distributed on an
# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
# KIND, either express or implied. See the License for the
# specific language governing permissions and limitations
# under the License.
"""
Tests db_engine_specs.risingwave against a real RisingWave instance, spun
up on demand via testcontainers. Run via .github/workflows/testcontainers.yml.
RisingWave speaks the Postgres wire protocol, but doesn't run the real
Postgres server binary or its POSTGRES_PASSWORD-style bootstrap env vars,
so this can't reuse `PostgresContainer` the way TimescaleDB/YugabyteDB do
-- it needs a generic DockerContainer against the official
`risingwavelabs/risingwave` single-binary playground image instead.
`RisingWaveDbEngineSpec` extends `PostgresEngineSpec`, and
`sqlalchemy-risingwave`'s dialect is a genuine subclass of SQLAlchemy's own
Postgres dialect (via psycopg2), so DDL/pagination compile with standard
Postgres semantics -- no ClickHouse-style mandatory table option needed.
"""
import re
from collections.abc import Iterator
import pytest
from sqlalchemy import (
Column,
create_engine,
inspect,
Integer,
MetaData,
Table as SATable,
text,
)
from sqlalchemy.engine import Connection, Engine
from superset.db_engine_specs.risingwave import RisingWaveDbEngineSpec
from superset.sql.parse import Table
from superset.utils.core import GenericDataType
pytestmark = pytest.mark.testcontainers
from ._driver import require_driver # noqa: E402
require_driver("testcontainers.core.container")
require_driver("sqlalchemy_risingwave")
from testcontainers.core.container import DockerContainer # noqa: E402
from testcontainers.core.wait_strategies import LogMessageWaitStrategy # noqa: E402
from ._pagination import ( # noqa: E402
assert_paginated_query_returns_correct_rows_in_order,
)
PORT = 4566
@pytest.fixture(scope="module")
def engine() -> Iterator[Engine]:
container = DockerContainer("risingwavelabs/risingwave")
container.with_exposed_ports(PORT)
container.with_command("playground")
# The actual startup banner reads "RisingWave standalone mode is
# ready." -- confirmed against a real container's logs.
container.waiting_for(
LogMessageWaitStrategy(re.compile("RisingWave standalone mode is ready"))
)
with container:
host = container.get_container_host_ip()
port = container.get_exposed_port(PORT)
yield create_engine(f"risingwave://root@{host}:{port}/dev")
def _flush(conn: Connection) -> None:
# RisingWave's storage engine checkpoints asynchronously: without an
# explicit FLUSH, a SELECT immediately after INSERT can see zero rows
# -- confirmed against a real instance (a bare INSERT commits fine, but
# the data isn't visible to a subsequent query until flushed).
conn.execute(text("FLUSH"))
def test_paginated_query_returns_correct_rows_in_order(engine: Engine) -> None:
"""
A plain SQLAlchemy Core LIMIT/OFFSET query, compiled and executed against
a real instance. Mocked tests cannot catch a dialect compiling this
incorrectly (see apache/superset#42899, where Trino emitted OFFSET
before LIMIT) -- only real execution can.
"""
assert_paginated_query_returns_correct_rows_in_order(engine, after_insert=_flush)
def test_get_columns_maps_native_types(engine: Engine) -> None:
"""
RisingWaveDbEngineSpec.get_columns wraps a real SQLAlchemy Inspector;
this exercises that against actual server-reported column metadata
rather than a mocked Inspector.
"""
metadata = MetaData()
SATable(
"pilot_types",
metadata,
Column("id", Integer, primary_key=True),
Column("amount", Integer),
)
metadata.create_all(engine)
inspector = inspect(engine)
columns = RisingWaveDbEngineSpec.get_columns(inspector, Table("pilot_types"))
by_name = {col["column_name"]: col for col in columns}
assert set(by_name) == {"id", "amount"}
for col in by_name.values():
spec = RisingWaveDbEngineSpec.get_column_spec(str(col["type"]))
assert spec is not None
assert spec.generic_type == GenericDataType.NUMERIC
assert isinstance(spec.sqla_type, Integer)