From e5798a058a1f3341e1027af31c7d0290558ead9b Mon Sep 17 00:00:00 2001 From: he_sk Date: Sun, 27 Sep 2026 04:41:06 +0800 Subject: [PATCH] test: verify existing connections across DM8 HA takeover --- docs/test-results/2026-09-27-ha-routing.md | 52 +++++- scripts/verify_dm_open_connection_failover.py | 175 ++++++++++++++++++ 2 files changed, 221 insertions(+), 6 deletions(-) create mode 100644 scripts/verify_dm_open_connection_failover.py diff --git a/docs/test-results/2026-09-27-ha-routing.md b/docs/test-results/2026-09-27-ha-routing.md index 1e3a184..28a0b9d 100644 --- a/docs/test-results/2026-09-27-ha-routing.md +++ b/docs/test-results/2026-09-27-ha-routing.md @@ -36,8 +36,8 @@ needs permission to create and drop a table on the primary. This HA check currently runs locally: the GitHub hosted runners cannot reach the user's Orb network. The regular GitHub ARM real-database matrix still -checks single-node behavior and build compatibility. MPP routing and UKey -authentication remain unverified. +checks single-node behavior and build compatibility. MPP has since been +verified separately; UKey authentication remains unverified. ## Automatic takeover and new connections @@ -69,8 +69,8 @@ table. Both phases passed. The script requires a unique `--table` name, `DM_HA_SERVICE_PATH`; `prepare` also needs `DM_HA_STANDBY_HOST` and optional `DM_HA_STANDBY_PORT`. Pass the standby instance name printed by `prepare` as `--expected-new-primary` to `verify`. The operator triggers the fault between -the two phases. Existing open connections were not tested for transparent -reconnection; the verified behavior is a **new connection** after takeover. +the two phases. This two-phase check verified a **new connection** after +takeover; the already-open case was tested separately below. The service configuration used two reachable endpoints: @@ -95,5 +95,45 @@ pin one physical connection; `commit()` reports an error after that connection is lost, and a fresh connection confirms the row was not persisted. The repeatable check is `scripts/verify_dm_restart_transaction.py`; it also runs in a dedicated GitHub ARM CI job. Local full real-database regression with a -dedicated test user passed 193 cases. A separate primary/standby takeover with -an already-open connection has still not been verified. +dedicated test user passed 193 cases. + +## Existing connections across a real primary/standby takeover + +The follow-up used a new isolated two-node watcher group and confirmation +monitor in local Orb, with the same ARM image +`dm8:dm8_20250924_rev288894_HWarm_kylin10_64` (image ID +`sha256:eb5f243c8f4322596d0eac611c0fa2d8cd8a03a5d3f69889f3549a530354b210`). +The primary was `GRP453932_DW1`, and the standby was `GRP453932_DW2`. The +service listed both endpoints with `LOGIN_MODE=1`. The image's missing +`TIME_ZONE` substitution was repaired only inside these temporary containers. + +`scripts/verify_dm_open_connection_failover.py` kept an autocommit service +connection and an uncommitted manual-transaction service connection open on +the primary. It wrote `12345678901234567890.12345678` to a `DECIMAL(30,8)` +column and confirmed the exact value on the standby. It then inserted an +uncommitted second row and confirmed that the standby could not see it. The +script killed `dmwatcher` and `dmserver` inside the disposable primary +container, preserving the container's network identity. The monitor logged +`AUTO TAKEOVER GRP453932_DW2` and `auto takeover success` at 04:36:25–27 +local time. The standby became `PRIMARY, OPEN`. + +The old transaction's `commit()` raised `dmPython.Error`; the lost row remained +absent. The *same* already-open autocommit Python connection reached the +promoted primary without a visible query error in this run, then wrote and +read `98765432109876543210.87654321` exactly. The promoted primary also read +the original replicated value exactly. The verifier removed its table. + +After the former primary rejoined as `STANDBY, OPEN`, the same verifier ran in +the reverse direction. The monitor logged automatic takeover by +`GRP453932_DW1` at 04:39:57–58. The already-open service connection reached +that promoted node with zero visible query errors; the lost manual transaction +again raised on `commit()`, and both exact decimal checks passed. + +To repeat this check against a disposable HA group, set `DM_HA_USER`, +`DM_HA_PASSWORD`, `DM_HA_SERVICE_NAME`, `DM_HA_SERVICE_PATH` (directory +containing `dm_svc.conf`), `DM_HA_STANDBY_HOST`, optional +`DM_HA_STANDBY_PORT`, and `DM_HA_PRIMARY_CONTAINER`. The service must route +only to the primary; the standby address must connect directly. Run the script +from a Python environment with the local extension and bridge library. It +stops the named primary's database and watcher processes. This local HA check +is not yet a GitHub-hosted CI gate because the runner cannot reach Orb. diff --git a/scripts/verify_dm_open_connection_failover.py b/scripts/verify_dm_open_connection_failover.py new file mode 100644 index 0000000..144c4f4 --- /dev/null +++ b/scripts/verify_dm_open_connection_failover.py @@ -0,0 +1,175 @@ +"""Verify already-open Python connections across a real DM8 HA takeover. + +Run only against a disposable two-node watcher group with a confirmation +monitor. This script kills the primary's dmwatcher and dmserver processes. +The service name must list both endpoints with LOGIN_MODE=1. +""" + +from __future__ import annotations + +import os +import subprocess +import time +import uuid +from decimal import Decimal + +import dmPython + + +BEFORE = Decimal("12345678901234567890.12345678") +AFTER = Decimal("98765432109876543210.87654321") + + +def connect(*, autocommit: bool, standby: bool = False): + options = { + "user": os.environ["DM_HA_USER"], + "password": os.environ["DM_HA_PASSWORD"], + "server": ( + os.environ["DM_HA_STANDBY_HOST"] + if standby + else os.environ["DM_HA_SERVICE_NAME"] + ), + "autoCommit": ( + dmPython.DSQL_AUTOCOMMIT_ON if autocommit else dmPython.DSQL_AUTOCOMMIT_OFF + ), + "login_timeout": 3000, + "connection_timeout": 3, + } + if standby: + options["port"] = int(os.environ.get("DM_HA_STANDBY_PORT", "5236")) + else: + options["dmsvc_path"] = os.environ["DM_HA_SERVICE_PATH"] + return dmPython.connect(**options) + + +def instance(conn): + with conn.cursor() as cur: + cur.execute("SELECT INSTANCE_NAME, MODE$, STATUS$ FROM V$INSTANCE") + return tuple(cur.fetchone()) + + +def amount_at(conn, table: str, row_id: int): + with conn.cursor() as cur: + cur.execute( + f"SELECT CAST(AMOUNT AS VARCHAR(80)) FROM {table} WHERE ID=?", (row_id,) + ) + row = cur.fetchone() + return None if row is None else Decimal(row[0]) + + +def wait_until(check, description: str, seconds: int = 90): + deadline = time.monotonic() + seconds + last_error = None + while time.monotonic() < deadline: + try: + result = check() + if result: + return result + except dmPython.Error as exc: + last_error = exc + time.sleep(1) + raise AssertionError(f"timed out waiting for {description}: {last_error}") + + +def stop_primary(container: str): + subprocess.run( + [ + "docker", + "exec", + container, + "bash", + "-c", + "pkill -9 -x dmwatcher && pkill -9 -x dmserver", + ], + check=True, + stdout=subprocess.DEVNULL, + ) + + +def main(): + container = os.environ["DM_HA_PRIMARY_CONTAINER"] + table = "DMPY_HA_OPEN_" + uuid.uuid4().hex[:8].upper() + with ( + connect(autocommit=True) as held, + connect(autocommit=True, standby=True) as standby, + ): + old_name, old_mode, old_status = instance(held) + new_name, new_mode, new_status = instance(standby) + assert (old_mode, old_status) == ("PRIMARY", "OPEN") + assert (new_mode, new_status) == ("STANDBY", "OPEN") + assert old_name != new_name + + with held.cursor() as cur: + cur.execute( + f"CREATE TABLE {table} (ID INT PRIMARY KEY, AMOUNT DECIMAL(30,8))" + ) + cur.execute(f"INSERT INTO {table} VALUES (?, ?)", (1, BEFORE)) + + try: + wait_until( + lambda: amount_at(standby, table, 1) == BEFORE, + "exact decimal on standby", + seconds=20, + ) + tx = connect(autocommit=False) + try: + assert instance(tx)[0] == old_name + with tx.cursor() as cur: + cur.execute(f"INSERT INTO {table} VALUES (?, ?)", (2, AFTER)) + assert amount_at(standby, table, 2) is None + + stop_primary(container) + wait_until( + lambda: instance(standby) == (new_name, "PRIMARY", "OPEN"), + "standby promotion", + ) + assert amount_at(standby, table, 1) == BEFORE + assert amount_at(standby, table, 2) is None + + try: + tx.commit() + except dmPython.Error: + pass + else: + raise AssertionError( + "lost transaction reported a successful commit" + ) + assert amount_at(standby, table, 2) is None + + errors = 0 + + def held_connection_recovered(): + nonlocal errors + try: + return instance(held) == (new_name, "PRIMARY", "OPEN") + except dmPython.Error: + errors += 1 + return False + + wait_until( + held_connection_recovered, "held autocommit connection recovery" + ) + with held.cursor() as cur: + cur.execute(f"INSERT INTO {table} VALUES (?, ?)", (3, AFTER)) + assert amount_at(held, table, 3) == AFTER + assert amount_at(standby, table, 3) == AFTER + print( + f"promoted {new_name}; held connection recovered after {errors} error(s)" + ) + print("lost transaction rejected; DECIMAL(30,8) remained exact") + finally: + tx.close() + finally: + # The disposable HA database is retained for inspection if the + # takeover fails, but remove the test table when a primary exists. + try: + with connect(autocommit=True, standby=True) as cleanup: + if instance(cleanup)[1] == "PRIMARY": + with cleanup.cursor() as cur: + cur.execute(f"DROP TABLE {table}") + except dmPython.Error: + pass + + +if __name__ == "__main__": + main()