From 397b4418e8e4e947885007eb2334f4c6ecc3b186 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 09:40:02 -0400 Subject: [PATCH 01/30] Move CI provisioning from v1 workspaces to v2 clusters DEFAULT_MANAGEMENT_VERSION is v2 and the test suite already deploys clusters, but CI still drove v1 workspace groups and tore down through a hardcoded /v1/workspaces URL, emitting a DeprecationWarning on every run. resources/create_test_cluster.py now makes one create_cluster() call. wait_on_active covers ACTIVE, the endpoint and the firewall, replacing both polling loops and the bare sleep, and closing a gap: the script never waited for the firewall, which the API applies outside the state machine. The shared "Python Client Testing" group -- never deleted by CI -- is gone with the flat cluster model. Fixes a live leak: --expires was parsed and never passed to the API, so CI clusters had no expiry and a failed shutdown job leaked a billable deployment indefinitely. It now reaches expires_at=. POST /v2/clusters generates the admin password and ignores what is sent; PATCH ignores it too (docs/management-api-audit.md item 9), so create-then-PATCH is not available and secrets.CLUSTER_PASSWORD can no longer be the password of the cluster CI just made. The generated value is read off the create response and propagated as a cluster-password job output. Job outputs are not secrets, so it is ::add-mask::-ed where it is emitted and again in every consuming job -- the mask does not cross job boundaries, and omitting the re-mask would leak a live credential. The generated password is drawn from the full printable set (one observed value: {:D}TK*[F3Ll}Ups2pNv), so a percent-encoded cluster-password-url output is emitted alongside it for the SINGLESTOREDB_URL and CIBW_ENVIRONMENT call sites; the raw form reaches drop_db.py through the environment rather than being interpolated into a shell word. Verified that the encoded form round-trips through the SDK's own URL parser. Cluster names are cleaned to [a-z0-9]([a-z0-9-]*[a-z0-9])? and truncated to 32 characters -- CI passes a workflow name that can overrun the limit. Regions are matched to a Region object, since v2 regions have no ID, and the pattern is tried against both the display and provider region names. A project is required by the API and does not auto-resolve in a multi-project org, so --project was added, defaulting to the STANDARD-edition project and pinnable with the CLUSTER_PROJECT variable. drop_test_cluster.py now takes a cluster ID, matching what the create script emits; its old contract said workspace-id but slugified the argument into a name. The workflow teardown stays curl, so the shutdown job needs no install, with the URL moved to DELETE /v2/clusters/{id}. Also makes the deprecated v1 suite a nightly gate rather than a per-PR cost: code-check.yml and coverage.yml deselect management_v1, and coverage.yml gains a job that runs it. Verified the two selections partition the suite exactly, 905 + 65 of 970. Two docs gaps closed alongside: Cluster.update()'s admin_password lacked the warning create_cluster() carries, an asymmetry that invites the create-then-PATCH dead end; and build_docs.py rewrote workspace.Stage but not cluster.Stage, which the v2 shim now re-exports. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 6 +- .github/workflows/coverage.yml | 48 +++++- .github/workflows/publish.yml | 47 +++++- .github/workflows/smoke-test.yml | 47 +++++- resources/build_docs.py | 5 + resources/create_test_cluster.py | 198 +++++++++++++++---------- resources/drop_test_cluster.py | 40 +---- singlestoredb/management/v2/cluster.py | 13 +- 8 files changed, 281 insertions(+), 123 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 5ef487bc7..35c2d535d 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -137,8 +137,12 @@ jobs: - name: Run MySQL protocol tests (with management API) if: steps.check-changes.outputs.changes-detected == 'true' + # -m 'not management_v1' keeps the v2 management coverage while dropping + # the deprecated v1 suite, which coverage.yml runs nightly instead. The + # -m 'not management' steps below need no second term: they already + # exclude everything v1 deploys. run: | - pytest -v --cov=singlestoredb --pyargs singlestoredb.tests + pytest -v -m 'not management_v1' --cov=singlestoredb --pyargs singlestoredb.tests env: COVERAGE_FILE: "coverage-mysql.cov" SINGLESTOREDB_URL: "root:root@127.0.0.1:3307" diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 6c9546fa9..ef07ecbf0 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -36,8 +36,11 @@ jobs: pip install -e ".[dev]" - name: Run MySQL protocol tests + # -m 'not management_v1' keeps the v2 management coverage while dropping + # the deprecated v1 suite; the management-v1-tests job below is where + # that runs. run: | - pytest -v --cov=singlestoredb --pyargs singlestoredb.tests + pytest -v -m 'not management_v1' --cov=singlestoredb --pyargs singlestoredb.tests env: COVERAGE_FILE: "coverage-mysql.cov" SINGLESTOREDB_URL: "root:root@127.0.0.1:3307" @@ -82,3 +85,46 @@ jobs: coverage report coverage xml coverage html + + # The deprecated v1 management API. management.version defaults to v2, so this + # is a legacy gate: it runs here nightly rather than on every PR, and it is + # what gets deleted along with management/v1/. Selects both the mocked v1 + # units and the live v1 deployments, plus test_fusion's v1 WORKSPACE grammar. + management-v1-tests: + runs-on: ubuntu-latest + environment: Base + + services: + singlestore: + image: ghcr.io/singlestore-labs/singlestoredb-dev:latest + ports: + - 3307:3306 + - 8081:8080 + - 9081:9081 + env: + SINGLESTORE_LICENSE: ${{ secrets.SINGLESTORE_LICENSE }} + ROOT_PASSWORD: "root" + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v4 + with: + python-version: "3.10" + cache: "pip" + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -e ".[dev]" + + - name: Run v1 management API tests + run: | + pytest -v -m 'management_v1' --pyargs singlestoredb.tests + env: + SINGLESTOREDB_URL: "root:root@127.0.0.1:3307" + SINGLESTOREDB_PURE_PYTHON: 0 + SINGLESTORE_LICENSE: ${{ secrets.SINGLESTORE_LICENSE }} + SINGLESTOREDB_MANAGEMENT_TOKEN: ${{ secrets.CLUSTER_API_KEY }} + SINGLESTOREDB_FUSION_ENABLE_HIDDEN: "1" diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index d3a669c1e..eeeb663e1 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -50,7 +50,7 @@ jobs: - name: Initialize database id: initialize-database run: | - python resources/create_test_cluster.py --password="${{ secrets.CLUSTER_PASSWORD }}" --token="${{ secrets.CLUSTER_API_KEY }}" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" + python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="${{ vars.CLUSTER_PROJECT }}" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: PYTHONPATH: ${{ github.workspace }} @@ -58,6 +58,11 @@ jobs: cluster-id: ${{ steps.initialize-database.outputs.cluster-id }} cluster-host: ${{ steps.initialize-database.outputs.cluster-host }} cluster-database: ${{ steps.initialize-database.outputs.cluster-database }} + # POST /v2/clusters generates the admin password and reports it only on + # the create response, so it travels as a job output rather than living + # in secrets. Job outputs are not secrets: every job below re-masks it. + cluster-password: ${{ steps.initialize-database.outputs.cluster-password }} + cluster-password-url: ${{ steps.initialize-database.outputs.cluster-password-url }} build-and-test: needs: setup-database @@ -73,6 +78,17 @@ jobs: - windows-2022 steps: + # ::add-mask:: does not cross job boundaries, so the generated admin + # password arrives here unmasked and has to be registered again before + # any step can echo it into the log. + - name: Mask cluster password + env: + CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} + run: | + echo "::add-mask::$CLUSTER_PASSWORD" + echo "::add-mask::$CLUSTER_PASSWORD_URL" + - uses: actions/checkout@v3 - name: Set up Python ${{ matrix.python-version }} @@ -118,7 +134,12 @@ jobs: # points --pyargs at the workspace, so that pyproject is the inifile # here. Without the plugin pytest exits on the unknown arguments. CIBW_TEST_REQUIRES: "pytest pytest-xdist" - CIBW_ENVIRONMENT: "SINGLESTOREDB_URL='mysql://${{ secrets.CLUSTER_USER }}:${{ secrets.CLUSTER_PASSWORD }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=0'" + # cluster-password-url, not cluster-password: the generated password + # is drawn from the full printable set, and this value has to survive + # both the userinfo half of the URL and the single-quoted shell word + # cibuildwheel evaluates. Percent-encoding leaves only unreserved + # characters and %, which are inert in both. + CIBW_ENVIRONMENT: "SINGLESTOREDB_URL='mysql://${{ secrets.CLUSTER_USER }}:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=0'" PYTHONPATH: ${{ github.workspace }} # - name: Build conda @@ -228,6 +249,15 @@ jobs: runs-on: ubuntu-latest steps: + # ::add-mask:: does not cross job boundaries; see the build-and-test job. + - name: Mask cluster password + env: + CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} + run: | + echo "::add-mask::$CLUSTER_PASSWORD" + echo "::add-mask::$CLUSTER_PASSWORD_URL" + - uses: actions/checkout@v3 - name: Install dependencies @@ -238,14 +268,21 @@ jobs: - name: Drop database if: ${{ always() }} + # The password reaches the script through the environment rather than + # being interpolated into the command: the generated value can contain + # any printable character, including ones the shell would act on. run: | - python resources/drop_db.py --user "${{ secrets.CLUSTER_USER }}" --password "${{ secrets.CLUSTER_PASSWORD }}" --host "${{ needs.setup-database.outputs.cluster-host }}" --port 3306 --database "${{ needs.setup-database.outputs.cluster-database }}" + python resources/drop_db.py --user "$CLUSTER_USER" --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" env: PYTHONPATH: ${{ github.workspace }} + CLUSTER_USER: ${{ secrets.CLUSTER_USER }} + CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_HOST: ${{ needs.setup-database.outputs.cluster-host }} + CLUSTER_DATABASE: ${{ needs.setup-database.outputs.cluster-database }} - - name: Shutdown workspace + - name: Shutdown cluster if: ${{ always() }} run: | - curl -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v1/workspaces/${{ env.CLUSTER_ID }}" + curl -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v2/clusters/${{ env.CLUSTER_ID }}?force=true" env: CLUSTER_ID: ${{ needs.setup-database.outputs.cluster-id }} diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 688a2dc18..ab60501c3 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -29,7 +29,7 @@ jobs: - name: Initialize database id: initialize-database run: | - python resources/create_test_cluster.py --password="${{ secrets.CLUSTER_PASSWORD }}" --token="${{ secrets.CLUSTER_API_KEY }}" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" + python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="${{ vars.CLUSTER_PROJECT }}" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: PYTHONPATH: ${{ github.workspace }} @@ -37,6 +37,11 @@ jobs: cluster-id: ${{ steps.initialize-database.outputs.cluster-id }} cluster-host: ${{ steps.initialize-database.outputs.cluster-host }} cluster-database: ${{ steps.initialize-database.outputs.cluster-database }} + # POST /v2/clusters generates the admin password and reports it only on + # the create response, so it travels as a job output rather than living + # in secrets. Job outputs are not secrets: every job below re-masks it. + cluster-password: ${{ steps.initialize-database.outputs.cluster-password }} + cluster-password-url: ${{ steps.initialize-database.outputs.cluster-password-url }} smoke-test: @@ -100,6 +105,17 @@ jobs: buffered: 1 steps: + # ::add-mask:: does not cross job boundaries, so the generated admin + # password arrives here unmasked and has to be registered again before + # any step can echo it into the log. + - name: Mask cluster password + env: + CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} + run: | + echo "::add-mask::$CLUSTER_PASSWORD" + echo "::add-mask::$CLUSTER_PASSWORD_URL" + - uses: actions/checkout@v4 - name: Set up Python ${{ matrix.python-version }} @@ -118,7 +134,10 @@ jobs: run: pytest -v --pyargs singlestoredb.tests.test_basics env: PYTHONPATH: ${{ github.workspace }} - SINGLESTOREDB_URL: "${{ matrix.driver }}://${{ secrets.CLUSTER_USER }}:${{ secrets.CLUSTER_PASSWORD }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" + # cluster-password-url, not cluster-password: the generated password + # is drawn from the full printable set and has to be percent-encoded + # to survive the userinfo half of the URL. + SINGLESTOREDB_URL: "${{ matrix.driver }}://${{ secrets.CLUSTER_USER }}:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" - name: Run tests if: ${{ matrix.driver == 'https' }} @@ -131,7 +150,7 @@ jobs: run: pytest -v -n 0 --pyargs singlestoredb.tests.test_basics env: PYTHONPATH: ${{ github.workspace }} - SINGLESTOREDB_URL: "${{ matrix.driver }}://${{ secrets.CLUSTER_USER }}:${{ secrets.CLUSTER_PASSWORD }}@${{ needs.setup-database.outputs.cluster-host }}:443/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" + SINGLESTOREDB_URL: "${{ matrix.driver }}://${{ secrets.CLUSTER_USER }}:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:443/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" shutdown-database: @@ -140,6 +159,15 @@ jobs: runs-on: ubuntu-latest steps: + # ::add-mask:: does not cross job boundaries; see the smoke-test job. + - name: Mask cluster password + env: + CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} + run: | + echo "::add-mask::$CLUSTER_PASSWORD" + echo "::add-mask::$CLUSTER_PASSWORD_URL" + - uses: actions/checkout@v4 - name: Set up Python 3.11 @@ -156,14 +184,21 @@ jobs: - name: Drop database if: ${{ always() }} + # The password reaches the script through the environment rather than + # being interpolated into the command: the generated value can contain + # any printable character, including ones the shell would act on. run: | - python resources/drop_db.py --user "${{ secrets.CLUSTER_USER }}" --password "${{ secrets.CLUSTER_PASSWORD }}" --host "${{ needs.setup-database.outputs.cluster-host }}" --port 3306 --database "${{ needs.setup-database.outputs.cluster-database }}" + python resources/drop_db.py --user "$CLUSTER_USER" --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" env: PYTHONPATH: ${{ github.workspace }} + CLUSTER_USER: ${{ secrets.CLUSTER_USER }} + CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_HOST: ${{ needs.setup-database.outputs.cluster-host }} + CLUSTER_DATABASE: ${{ needs.setup-database.outputs.cluster-database }} - - name: Shutdown workspace + - name: Shutdown cluster if: ${{ always() }} run: | - curl -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v1/workspaces/${{ env.CLUSTER_ID }}" + curl -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v2/clusters/${{ env.CLUSTER_ID }}?force=true" env: CLUSTER_ID: ${{ needs.setup-database.outputs.cluster-id }} diff --git a/resources/build_docs.py b/resources/build_docs.py index 628f3a1fd..5e54b184c 100755 --- a/resources/build_docs.py +++ b/resources/build_docs.py @@ -384,6 +384,11 @@ def apply_content_transformations(self, content: str, links: Dict[str, str]) -> # Change workspace.Stage to workspace.stage content = re.sub(r'>workspace\.Stage\.', r'>workspace.stage.', content) + # Change cluster.Stage to cluster.stage. Stage is re-exported from the + # v2 cluster module, so it is documented under both names for as long + # as management/workspace.py is still documented. + content = re.sub(r'>cluster\.Stage\.', r'>cluster.stage.', content) + # Fix class/method links content = re.sub( r'(]+>)?(\s*]*>\s*\s*)([\w\.]+)(\s*\s*)', diff --git a/resources/create_test_cluster.py b/resources/create_test_cluster.py index 186fadfa8..8c258eb02 100755 --- a/resources/create_test_cluster.py +++ b/resources/create_test_cluster.py @@ -2,45 +2,47 @@ # type: ignore from __future__ import annotations +import json import os import random import re -import secrets import subprocess import sys -import time import uuid from optparse import OptionParser +from urllib.parse import quote import singlestoredb as s2 # Handle command-line options -usage = 'usage: %prog [options] workspace-name' +usage = 'usage: %prog [options] cluster-name' parser = OptionParser(usage=usage) parser.add_option( '-r', '--region', default='AWS::*US East 1*', - help='region pattern or ID', -) -parser.add_option( - '-p', '--password', - default=secrets.token_urlsafe(20) + '-x&$', - help='admin password', + help='region pattern to deploy into, as provider::name ' + '(AWS::*US East 1*); * is a wildcard', ) parser.add_option( '-e', '--expires', default='4h', - help='timestamp when workspace should expire (4h)', + help='when the cluster should expire, as a timestamp or a ' + 'duration such as 4h (4h)', ) parser.add_option( '-s', '--size', default='S-00', - help='size of the workspace (S-00)', + help='size of the cluster (S-00)', ) parser.add_option( '-t', '--token', - help='API key for the workspace management API', + help='API key for the management API', +) +parser.add_option( + '--project', + help='ID or name of the project to deploy into; defaults to the ' + 'organization\'s STANDARD-edition project', ) parser.add_option( '--http-port', type='int', @@ -53,7 +55,7 @@ parser.add_option( '-o', '--output', default='env', choices=['env', 'github', 'json'], - help='report workspace information in the requested format: github, env, json', + help='report cluster information in the requested format: github, env, json', ) parser.add_option( '-d', '--database', @@ -67,111 +69,153 @@ sys.exit(1) if options.init_sql and not os.path.isfile(options.init_sql): - print('ERROR: Could not locate SQL file: {options.init_sql}', file=sys.stderr) + print(f'ERROR: Could not locate SQL file: {options.init_sql}', file=sys.stderr) sys.exit(1) -# Connect to workspace. This is still the deprecated v1 workspace-group -# grammar because the v1 test suite it sets up needs workspace groups; -# it gets ported to manage_clusters() when that suite goes. Pinned to v1 -# because manage_workspaces() otherwise follows the management.version option. -wm = s2.manage_workspaces(options.token or None, version='v1') +# Pin v2 explicitly rather than following the ambient management.version +# option: this script provisions clusters, which only exist in v2. +mgr = s2.manage_clusters(options.token or None, version='v2') + + +# Find a matching region. A v2 region is identified by the +# (provider, region_name) pair rather than by an ID, so the matched Region +# object is what gets handed to create_cluster. Candidates are shuffled to +# spread deployments across whichever regions match. +# +# The pattern is tried against both the display name and the provider region +# name -- 'US East 1' and 'us-east-1' -- so it does not matter which of the two +# a given listing puts in Region.name. +pattern = options.region.replace('*', '.*') +regions = list(mgr.regions) + + +def candidates(item): + """Return the names ``item`` can be matched by, most specific first.""" + for label in (item.name, item.region_name): + if label: + yield f'{item.provider}::{label}' if '::' in options.region else label + -# Find matching region -if '::' in options.region: - pattern = options.region.replace('*', '.*') - regions = wm.regions - for item in random.sample(regions, k=len(regions)): - region_name = '{}::{}'.format(item.provider, item.name) - if re.match(pattern, region_name): - options.region = item.id - break +region = None +for item in random.sample(regions, k=len(regions)): + if any(re.match(pattern, x) for x in candidates(item)): + region = item + break -if '::' in options.region: +if region is None: print( - 'ERROR: Could not find a region mating the pattern: ' - '{options.region}', file=sys.stderr, + 'ERROR: Could not find a region matching the pattern ' + f'{options.region}; the API reports: ' + + ', '.join(sorted(f'{x.provider}::{x.name}' for x in regions)), + file=sys.stderr, ) sys.exit(1) -# Create workspace group -wg_name = 'Python Client Testing' - -wgs = [x for x in wm.workspace_groups if x.name == wg_name] -if len(wgs) > 1: - print('ERROR: There is more than one workspace group with the specified name.') - sys.exit(1) -elif len(wgs) == 1: - wg = wgs[0] +# Choose a project. projectID is required by POST /v2/clusters and only +# auto-resolves for an organization with a single project, so pick the +# STANDARD-edition one when it was not named explicitly. +if options.project: + project_id = options.project else: - wg = wm.create_workspace_group( - wg_name, - region=options.region, - admin_password=options.password, - # firewall_ranges=requests.get('https://api.github.com/meta').json()['actions'], - firewall_ranges=['0.0.0.0/0'], - allow_all_traffic=True, - ) - -# Make sure the workspace group exists before continuing -timeout = 300 -while timeout > 0 and not [x for x in wm.workspace_groups if x.name == wg_name]: - time.sleep(10) - timeout -= 10 + projects = list(mgr.projects) + standard = [x for x in projects if x.edition == 'STANDARD'] + if not standard: + print( + 'ERROR: No STANDARD-edition project in this organization; pass ' + '--project with one of: ' + + ', '.join(f'{x.name} ({x.id}, {x.edition})' for x in projects), + file=sys.stderr, + ) + sys.exit(1) + project_id = standard[0].id + + +# A cluster name must match [a-z0-9]([a-z0-9-]*[a-z0-9])? and be 1-32 +# characters, so everything outside that alphabet becomes a hyphen, runs of +# hyphens collapse, and the result is truncated with any hyphen the cut +# exposes trimmed off again. +name = re.sub(r'[^a-z0-9]+', '-', args[0].lower()).strip('-')[:32].rstrip('-') +if not name: + print(f'ERROR: Cluster name is empty after cleaning: {args[0]}', file=sys.stderr) + sys.exit(1) -ws_name = re.sub(r'^-|-$', r'', re.sub(r'-+', r'-', re.sub(r'\s+', '-', args[0].lower()))) -ws = wg.create_workspace( - ws_name, +# wait_on_active covers ACTIVE, then the endpoint, then the firewall, so the +# cluster is actually reachable by the time this returns. +cluster = mgr.create_cluster( + name, + region=region, size=options.size, + firewall_ranges=['0.0.0.0/0'], + expires_at=options.expires, + project=project_id, wait_on_active=True, + wait_timeout=1200, ) -# Make sure the endpoint exists before continuing -timeout = 300 -while timeout > 0 and not ws.endpoint: - time.sleep(10) - ws.refresh() - timeout -= 10 - -if not ws.endpoint: - print('ERROR: Endpoint was never activated.') +# The API generates the admin password and reports it only on the create +# response -- there is no route that will hand it back later, and it is None +# after any refresh(). See item 9 of docs/management-api-audit.md: the API +# accepts an adminPassword on both POST and PATCH and ignores both, which is +# why this is read back rather than set. +password = cluster.admin_password +if not password: + print( + 'ERROR: cluster was created without a readable admin password', + file=sys.stderr, + ) sys.exit(1) - -# Extra pause for server to become available -time.sleep(10) +# The generated password is drawn from the full printable set -- one observed +# value was ``{:D}TK*[F3Ll}Ups2pNv`` -- so it cannot be dropped into the +# userinfo half of a connection URL as-is. Percent-encode everything, since +# the URL parser runs unquote_plus over the password +# (singlestoredb/connection.py:287); encoding ``+`` too is what keeps that from +# turning into a space. +password_url = quote(password, safe='') database = options.database if not database: database = 'TEMP_{}'.format(uuid.uuid4()).replace('-', '_') -host = ws.endpoint +host = cluster.endpoint if ':' in host: host, port = host.split(':', 1) port = int(port) else: port = 3306 -# Print workspace information +# Print cluster information if options.output == 'env': - print(f'CLUSTER_ID={ws.id}') + print(f'CLUSTER_ID={cluster.id}') print(f'CLUSTER_HOST={host}') print(f'CLUSTER_PORT={port}') print(f'CLUSTER_DATABASE={database}') + print(f'CLUSTER_PASSWORD={password}') + print(f'CLUSTER_PASSWORD_URL={password_url}') elif options.output == 'github': + # Register both forms with the runner before anything can log them. This + # only holds within this job; each job that consumes the outputs has to + # mask them again for itself. + print(f'::add-mask::{password}') + print(f'::add-mask::{password_url}') with open(os.environ['GITHUB_OUTPUT'], 'a') as output: - print(f'cluster-id={ws.id}', file=output) + print(f'cluster-id={cluster.id}', file=output) print(f'cluster-host={host}', file=output) print(f'cluster-port={port}', file=output) print(f'cluster-database={database}', file=output) + print(f'cluster-password={password}', file=output) + print(f'cluster-password-url={password_url}', file=output) elif options.output == 'json': print('{') - print(f' "cluster-id": "{ws.id}",') + print(f' "cluster-id": "{cluster.id}",') print(f' "cluster-host": "{host}",') - print(f' "cluster-port": {port}') - print(f' "cluster-database": {database}') + print(f' "cluster-port": {port},') + print(f' "cluster-database": "{database}",') + print(f' "cluster-password": {json.dumps(password)},') + print(f' "cluster-password-url": "{password_url}"') print('}') # Initialize the database @@ -179,7 +223,7 @@ init_db = [ os.path.join(os.path.dirname(__file__), 'init_db.py'), '--host', str(host), '--port', str(port), - '--user', 'admin', '--password', options.password, + '--user', 'admin', '--password', password, '--database', database, ] diff --git a/resources/drop_test_cluster.py b/resources/drop_test_cluster.py index 16ed7539d..6a7105dd7 100755 --- a/resources/drop_test_cluster.py +++ b/resources/drop_test_cluster.py @@ -2,7 +2,6 @@ # type: ignore from __future__ import annotations -import re import sys from optparse import OptionParser @@ -10,11 +9,11 @@ # Handle command-line options -usage = 'usage: %prog [options] workspace-id' +usage = 'usage: %prog [options] cluster-id' parser = OptionParser(usage=usage) parser.add_option( '-t', '--token', - help='API key for the workspace management API', + help='API key for the management API', ) (options, args) = parser.parse_args() @@ -23,33 +22,10 @@ sys.exit(1) -# Connect to workspace. This is still the deprecated v1 workspace-group -# grammar because the v1 test suite it sets up needs workspace groups; -# it gets ported to manage_clusters() when that suite goes. Pinned to v1 -# because manage_workspaces() otherwise follows the management.version option. -wm = s2.manage_workspaces(options.token or None, version='v1') +# Pin v2 explicitly rather than following the ambient management.version +# option: clusters only exist in v2. +mgr = s2.manage_clusters(options.token or None, version='v2') -wg_name = 'Python Client Testing' - -wgs = [x for x in wm.workspace_groups if x.name == wg_name] -if len(wgs) > 1: - print('ERROR: There is more than one workspace group with the specified name.') - sys.exit(1) -elif len(wgs) == 0: - print('ERROR: There is no workspace group with the specified name.') - sys.exit(1) -wg = wgs[0] - -ws_name = re.sub(r'^-|-$', r'', re.sub(r'-+', r'-', re.sub(r'\s+', '-', args[0].lower()))) - -wss = [x for x in wg.workspaces if x.name == ws_name] -if len(wss) > 1: - print('ERROR: There is more than one workspace with the specified name.') - sys.exit(1) -elif len(wss) == 0: - print('ERROR: There is no workspace with the specified name.') - sys.exit(1) -ws = wss[0] - -# Terminate workspace -ws.terminate() +# force=True so a cluster with connections still open goes away; this only +# ever runs against clusters this repo's CI created. +mgr.get_cluster(args[0]).terminate(force=True, wait_on_terminated=True) diff --git a/singlestoredb/management/v2/cluster.py b/singlestoredb/management/v2/cluster.py index 511b7cace..9c8765cf1 100644 --- a/singlestoredb/management/v2/cluster.py +++ b/singlestoredb/management/v2/cluster.py @@ -611,7 +611,18 @@ def update( allow_all_traffic : bool, optional Allow all traffic to the cluster admin_password : str, optional - Admin password for the cluster + Admin password for the cluster. + + .. warning:: This is ignored, exactly as it is on + ``POST /v2/clusters``. ``PATCH /v2/clusters/{id}`` accepts the + field and does not honor it: a live probe found the patched value + refused with ``1045: Access denied`` while the password the + original create generated kept working. So the admin password + cannot be set after the fact either -- the only value that + authenticates is the generated one + :attr:`Cluster.admin_password` carried on the create response. + See item 9 of ``docs/management-api-audit.md``. The field is + still sent in case the API starts honoring it. expires_at : str, optional Timestamp of when the cluster will expire. Expiration time can be specified as a timestamp or a duration. From 25b8117a7ea22d9be8a4885cac17ad4e392d237e Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 09:59:10 -0400 Subject: [PATCH 02/30] Parse Go time.Time strings in management to_datetime The v2 clusters API returns expiresAt in Go's time.Time.String() format -- '2026-09-17 14:42:41.445984 +0000 UTC' -- while every other timestamp is RFC 3339. to_datetime split the string on '.' to pad the fractional seconds, produced '445984 +0000 UTC' as the fraction, and datetime_fromisoformat then returned None. Cluster.expires_at read as None on a cluster that had an expiry set, and SHOW CLUSTERS printed a blank expiry column. Normalize both shapes before parsing: pad or truncate the fraction to six digits, keep a numeric offset, and drop the trailing zone abbreviation and Go's monotonic-clock reading. An offset-aware result is converted to UTC and returned naive, matching what the RFC 3339 path already produced. Co-Authored-By: Claude Opus 5 --- singlestoredb/management/utils.py | 96 ++++++++++++++++---- singlestoredb/tests/test_management_utils.py | 72 +++++++++++++++ 2 files changed, 152 insertions(+), 16 deletions(-) diff --git a/singlestoredb/management/utils.py b/singlestoredb/management/utils.py index bfdcc8658..e00844dd3 100644 --- a/singlestoredb/management/utils.py +++ b/singlestoredb/management/utils.py @@ -408,6 +408,80 @@ def enable_http_tracing() -> None: requests_log.propagate = True +#: A Go ``time.Time`` rendered by its ``String()`` method: +#: ``2026-09-17 14:42:41.445984 +0000 UTC``. ``GET /v2/clusters/{id}`` reports +#: ``expiresAt`` in this shape while every other timestamp it returns is +#: RFC 3339, and the trailing zone name is not ISO 8601, so the whole value +#: fails to parse and the expiration silently reads as unset. The zone name and +#: the monotonic-clock reading Go appends to some values are both optional. +_GO_DATETIME_RE = re.compile( + r'^(?P\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}(?:\.\d+)?)' + r'(?:\s*(?P[+-]\d{2}:?\d{2}))?' + r'(?:\s+(?P[A-Za-z]\S*))?' + r'(?:\s+m=\S+)?$', +) + + +def _normalize_datetime(obj: str) -> str: + """ + Return ``obj`` as something :func:`converters.datetime_fromisoformat` reads. + + Handles the two shapes the management API returns -- RFC 3339 and the Go + ``time.Time.String()`` form -- by reducing both to a bare ISO 8601 + timestamp plus an optional numeric offset. Fractional seconds are padded to + microseconds, since Go trims trailing zeros. + + Parameters + ---------- + obj : str + Timestamp as reported by the API + + Returns + ------- + str + + """ + match = _GO_DATETIME_RE.match(obj.strip()) + if match is None: + # Not a shape this recognizes; hand it over untouched so the converter + # gets its usual chance to make sense of it. + return obj.replace('Z', '') + + stamp = match.group('stamp') + + # Fix datetimes with truncated zeros + if '.' in stamp: + stamp, micros = stamp.split('.', 1) + micros = micros[:6] + '0' * (6 - len(micros)) + stamp = stamp + '.' + micros + + return stamp + (match.group('offset') or '') + + +def _as_naive_utc(obj: datetime.datetime) -> datetime.datetime: + """ + Return ``obj`` as a naive UTC datetime. + + An RFC 3339 timestamp loses its ``Z`` before it is parsed, so it arrives + here naive and already meaning UTC. A value carrying a numeric offset is + shifted onto UTC and stripped, so both shapes end up on the one convention + -- otherwise two timestamps read off the same object could not be compared. + + Parameters + ---------- + obj : datetime.datetime + Parsed timestamp, with or without a timezone + + Returns + ------- + datetime.datetime + + """ + if obj.tzinfo is None: + return obj + return obj.astimezone(datetime.timezone.utc).replace(tzinfo=None) + + def to_datetime( obj: Optional[Union[str, datetime.datetime]], ) -> Optional[datetime.datetime]: @@ -418,18 +492,14 @@ def to_datetime( return obj if obj == '0001-01-01T00:00:00Z': return None - obj = obj.replace('Z', '') - # Fix datetimes with truncated zeros - if '.' in obj: - obj, micros = obj.split('.', 1) - micros = micros + '0' * (6 - len(micros)) - obj = obj + '.' + micros - out = converters.datetime_fromisoformat(obj) + out = converters.datetime_fromisoformat(_normalize_datetime(obj)) if isinstance(out, str): return None if isinstance(out, datetime.date) and not isinstance(out, datetime.datetime): return datetime.datetime(out.year, out.month, out.day) - return out + if out is None: + return None + return _as_naive_utc(out) def to_datetime_strict( @@ -442,20 +512,14 @@ def to_datetime_strict( return obj if obj == '0001-01-01T00:00:00Z': raise ValueError('not possible to convert 0001-01-01T00:00:00Z to datetime') - obj = obj.replace('Z', '') - # Fix datetimes with truncated zeros - if '.' in obj: - obj, micros = obj.split('.', 1) - micros = micros + '0' * (6 - len(micros)) - obj = obj + '.' + micros - out = converters.datetime_fromisoformat(obj) + out = converters.datetime_fromisoformat(_normalize_datetime(obj)) if not out: raise TypeError('not possible to convert None to datetime') if isinstance(out, str): raise ValueError('value cannot be str') if isinstance(out, datetime.date) and not isinstance(out, datetime.datetime): return datetime.datetime(out.year, out.month, out.day) - return out + return _as_naive_utc(out) def from_datetime( diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index b7b25f1e5..d2531bb9a 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -18,6 +18,8 @@ from singlestoredb.exceptions import ManagementError from singlestoredb.management.utils import normalize_remote_path +from singlestoredb.management.utils import to_datetime +from singlestoredb.management.utils import to_datetime_strict from singlestoredb.tests.utils import counting_file_space from singlestoredb.tests.utils import counting_stage @@ -1815,5 +1817,75 @@ def test_a_since_that_is_not_a_date_is_rejected(self): self.mod.parse_since('last tuesday') +class TestToDatetime(unittest.TestCase): + """ + ``to_datetime`` has to read both timestamp shapes the API returns. + + Most fields come back as RFC 3339, but ``GET /v2/clusters/{id}`` reports + ``expiresAt`` as a Go ``time.Time.String()`` rendering -- verified live: + ``2026-09-17 14:42:41.445984 +0000 UTC`` against a ``createdAt`` of + ``2026-09-17T13:42:41.493848Z`` on the same cluster. The trailing zone name + is not ISO 8601, and parsing it used to fail into ``None``, which reads as + "this cluster never expires". + """ + + def test_rfc_3339(self): + out = to_datetime('2026-09-17T13:42:41.493848Z') + self.assertEqual(out, datetime.datetime(2026, 9, 17, 13, 42, 41, 493848)) + + def test_go_time_string(self): + out = to_datetime('2026-09-17 14:42:41.445984 +0000 UTC') + self.assertEqual(out, datetime.datetime(2026, 9, 17, 14, 42, 41, 445984)) + + def test_go_time_string_with_truncated_fraction(self): + # Go trims trailing zeros, so the fraction is not always 6 digits. + out = to_datetime('2026-09-17 14:42:41.4 +0000 UTC') + self.assertEqual(out, datetime.datetime(2026, 9, 17, 14, 42, 41, 400000)) + + def test_go_time_string_with_monotonic_reading(self): + out = to_datetime( + '2026-09-17 14:42:41.445984 +0000 UTC m=+0.000000001', + ) + self.assertEqual(out, datetime.datetime(2026, 9, 17, 14, 42, 41, 445984)) + + def test_offset_is_applied_and_dropped(self): + # Shifted onto UTC and left naive, matching the RFC 3339 values, so two + # timestamps read off one object can be compared. + out = to_datetime('2026-09-17 09:42:41 -0500 EST') + self.assertEqual(out, datetime.datetime(2026, 9, 17, 14, 42, 41)) + self.assertIsNone(out.tzinfo) + + def test_both_shapes_subtract(self): + created = to_datetime('2026-09-17T13:42:41.493848Z') + expires = to_datetime('2026-09-17 14:42:41.445984 +0000 UTC') + self.assertAlmostEqual( + (expires - created).total_seconds(), 3600, delta=1, + ) + + def test_date_only(self): + out = to_datetime('2026-09-17') + self.assertEqual(out, datetime.datetime(2026, 9, 17)) + + def test_zero_sentinel_and_unparseable_are_none(self): + self.assertIsNone(to_datetime('0001-01-01T00:00:00Z')) + self.assertIsNone(to_datetime(None)) + self.assertIsNone(to_datetime('')) + self.assertIsNone(to_datetime('not a date')) + + def test_datetime_passes_through(self): + given = datetime.datetime(2026, 9, 17, 13, 42, 41) + self.assertIs(to_datetime(given), given) + + def test_strict_reads_the_go_shape_too(self): + out = to_datetime_strict('2026-09-17 14:42:41.445984 +0000 UTC') + self.assertEqual(out, datetime.datetime(2026, 9, 17, 14, 42, 41, 445984)) + + def test_strict_still_raises_on_nothing(self): + with self.assertRaises(TypeError): + to_datetime_strict(None) + with self.assertRaises(ValueError): + to_datetime_strict('0001-01-01T00:00:00Z') + + if __name__ == '__main__': unittest.main() From 9d3abf3bf997bd3ab85bc8d072f516988fafb798 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 10:03:37 -0400 Subject: [PATCH 03/30] Fix the CI cluster user and project in the workflows A v2 cluster has exactly one user, admin, and no route creates another, so secrets.CLUSTER_USER could only ever hold that one value. Drop the secret and name admin directly. Name the deployment project in the workflow too, rather than reading it from vars.CLUSTER_PROJECT: the target is then visible next to the create call and an unset repository variable cannot quietly change where CI deploys. create_cluster resolves a project name against GET /v2/projects and raises with the org's project list if it matches none, so a renamed project fails loudly at setup. Co-Authored-By: Claude Opus 5 --- .github/workflows/publish.yml | 11 +++++++---- .github/workflows/smoke-test.yml | 13 ++++++++----- 2 files changed, 15 insertions(+), 9 deletions(-) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index eeeb663e1..f39dc79ec 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -49,8 +49,12 @@ jobs: - name: Initialize database id: initialize-database + # A new cluster has exactly one user, admin, which is why that name is + # fixed everywhere below. The project is named here rather than read + # from a repo variable so the deployment target is visible in the + # workflow and does not depend on repository settings. run: | - python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="${{ vars.CLUSTER_PROJECT }}" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" + python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: PYTHONPATH: ${{ github.workspace }} @@ -139,7 +143,7 @@ jobs: # both the userinfo half of the URL and the single-quoted shell word # cibuildwheel evaluates. Percent-encoding leaves only unreserved # characters and %, which are inert in both. - CIBW_ENVIRONMENT: "SINGLESTOREDB_URL='mysql://${{ secrets.CLUSTER_USER }}:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=0'" + CIBW_ENVIRONMENT: "SINGLESTOREDB_URL='mysql://admin:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=0'" PYTHONPATH: ${{ github.workspace }} # - name: Build conda @@ -272,10 +276,9 @@ jobs: # being interpolated into the command: the generated value can contain # any printable character, including ones the shell would act on. run: | - python resources/drop_db.py --user "$CLUSTER_USER" --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" + python resources/drop_db.py --user admin --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" env: PYTHONPATH: ${{ github.workspace }} - CLUSTER_USER: ${{ secrets.CLUSTER_USER }} CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} CLUSTER_HOST: ${{ needs.setup-database.outputs.cluster-host }} CLUSTER_DATABASE: ${{ needs.setup-database.outputs.cluster-database }} diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index ab60501c3..50a35b5e3 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -28,8 +28,12 @@ jobs: - name: Initialize database id: initialize-database + # A new cluster has exactly one user, admin, which is why that name is + # fixed everywhere below. The project is named here rather than read + # from a repo variable so the deployment target is visible in the + # workflow and does not depend on repository settings. run: | - python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="${{ vars.CLUSTER_PROJECT }}" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" + python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: PYTHONPATH: ${{ github.workspace }} @@ -137,7 +141,7 @@ jobs: # cluster-password-url, not cluster-password: the generated password # is drawn from the full printable set and has to be percent-encoded # to survive the userinfo half of the URL. - SINGLESTOREDB_URL: "${{ matrix.driver }}://${{ secrets.CLUSTER_USER }}:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" + SINGLESTOREDB_URL: "${{ matrix.driver }}://admin:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" - name: Run tests if: ${{ matrix.driver == 'https' }} @@ -150,7 +154,7 @@ jobs: run: pytest -v -n 0 --pyargs singlestoredb.tests.test_basics env: PYTHONPATH: ${{ github.workspace }} - SINGLESTOREDB_URL: "${{ matrix.driver }}://${{ secrets.CLUSTER_USER }}:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:443/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" + SINGLESTOREDB_URL: "${{ matrix.driver }}://admin:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:443/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" shutdown-database: @@ -188,10 +192,9 @@ jobs: # being interpolated into the command: the generated value can contain # any printable character, including ones the shell would act on. run: | - python resources/drop_db.py --user "$CLUSTER_USER" --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" + python resources/drop_db.py --user admin --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" env: PYTHONPATH: ${{ github.workspace }} - CLUSTER_USER: ${{ secrets.CLUSTER_USER }} CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} CLUSTER_HOST: ${{ needs.setup-database.outputs.cluster-host }} CLUSTER_DATABASE: ${{ needs.setup-database.outputs.cluster-database }} From 6c22959dd6c1b21cc2216f39a42a4f69df973b9c Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 10:10:10 -0400 Subject: [PATCH 04/30] Add actionlint to pre-commit and fix what it reports The workflow files had no linter, so nothing checked expression syntax, needs.*.outputs.* references or the shell inside run: blocks. actionlint covers all three and ships as a pip wrapper, so the hook needs no Go toolchain or Docker. It reported 17 pre-existing problems, all fixed here so the hook lands green: - actions/checkout@v3, actions/setup-python@v4 and docker/setup-qemu-action@v2 run on node runtimes GitHub is retiring. Bumped to v4, v5 and v3, which is what the newer workflows already pin. - publish.yml named a step after ${{ matrix.python-version }} in a job whose matrix defines only os, so the name rendered with nothing after it. That job pins 3.10 as the host interpreter for cibuildwheel, so the reference was never going to resolve; dropped it. - code-check.yml left $GITHUB_OUTPUT unquoted on six redirects. Quoted. The one remaining sed is prefixing every line, which parameter expansion cannot do, so SC2001 is suppressed in place with a reason. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 16 +++++++++------- .github/workflows/coverage.yml | 4 ++-- .github/workflows/pre-commit.yml | 4 ++-- .github/workflows/publish.yml | 16 +++++++++------- .pre-commit-config.yaml | 4 ++++ 5 files changed, 26 insertions(+), 18 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 35c2d535d..0195ae88f 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -34,7 +34,7 @@ jobs: fetch-depth: 2 - name: Set up Python - uses: actions/setup-python@v4 + uses: actions/setup-python@v5 with: python-version: "3.10" cache: "pip" @@ -78,8 +78,8 @@ jobs: COMMIT_MSG=$(git log -1 --format='%s' HEAD) if [[ "$COMMIT_MSG" =~ ^Prepare\ for\ v[0-9]+\.[0-9]+\.[0-9]+\ release$ ]]; then echo "🚀 Release preparation commit detected: $COMMIT_MSG" - echo "changes-detected=true" >> $GITHUB_OUTPUT - echo "changed-directories=release" >> $GITHUB_OUTPUT + echo "changes-detected=true" >> "$GITHUB_OUTPUT" + echo "changed-directories=release" >> "$GITHUB_OUTPUT" echo "" echo "🎯 RESULT: Full test suite will run for release preparation" exit 0 @@ -98,6 +98,8 @@ jobs: if [ -n "$CHANGED_FILES" ]; then echo "✅ Changes detected in: $DIR" echo "Files changed:" + # shellcheck disable=SC2001 # prefixing every line, which + # ${var//search/replace} cannot do echo "$CHANGED_FILES" | sed 's/^/ - /' CHANGES_FOUND=true if [ -z "$CHANGED_DIRS" ]; then @@ -115,13 +117,13 @@ jobs: # Set outputs if [ "$CHANGES_FOUND" = true ]; then - echo "changes-detected=true" >> $GITHUB_OUTPUT - echo "changed-directories=$CHANGED_DIRS" >> $GITHUB_OUTPUT + echo "changes-detected=true" >> "$GITHUB_OUTPUT" + echo "changed-directories=$CHANGED_DIRS" >> "$GITHUB_OUTPUT" echo "" echo "🎯 RESULT: Changes detected in monitored directories" else - echo "changes-detected=false" >> $GITHUB_OUTPUT - echo "changed-directories=" >> $GITHUB_OUTPUT + echo "changes-detected=false" >> "$GITHUB_OUTPUT" + echo "changed-directories=" >> "$GITHUB_OUTPUT" echo "" echo "🎯 RESULT: No changes in monitored directories" fi diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index ef07ecbf0..023bdb44f 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -25,7 +25,7 @@ jobs: - uses: actions/checkout@v4 - name: Set up Python - uses: actions/setup-python@v4 + uses: actions/setup-python@v5 with: python-version: "3.10" cache: "pip" @@ -109,7 +109,7 @@ jobs: - uses: actions/checkout@v4 - name: Set up Python - uses: actions/setup-python@v4 + uses: actions/setup-python@v5 with: python-version: "3.10" cache: "pip" diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml index a9217a93f..6b3d87393 100644 --- a/.github/workflows/pre-commit.yml +++ b/.github/workflows/pre-commit.yml @@ -16,10 +16,10 @@ jobs: - "3.13" steps: - - uses: actions/checkout@v3 + - uses: actions/checkout@v4 - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v4 + uses: actions/setup-python@v5 with: python-version: ${{ matrix.python-version }} diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index f39dc79ec..fa88d195b 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -39,7 +39,7 @@ jobs: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v3 + - uses: actions/checkout@v4 - name: Install dependencies run: | @@ -93,10 +93,12 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v3 + - uses: actions/checkout@v4 - - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v4 + # This job's matrix varies only over os; cibuildwheel supplies its own + # interpreters, so 3.10 here is just the host Python that drives it. + - name: Set up Python + uses: actions/setup-python@v5 with: python-version: "3.10" cache: "pip" @@ -120,7 +122,7 @@ jobs: - name: Set up QEMU if: runner.os == 'Linux' - uses: docker/setup-qemu-action@v2 + uses: docker/setup-qemu-action@v3 with: platforms: all @@ -200,7 +202,7 @@ jobs: url: https://pypi.org/p/singlestoredb steps: - - uses: actions/checkout@v3 + - uses: actions/checkout@v4 - name: Download Linux wheels and sdist uses: actions/download-artifact@v4 @@ -262,7 +264,7 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v3 + - uses: actions/checkout@v4 - name: Install dependencies run: | diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 9d4c60017..b190dd101 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -41,3 +41,7 @@ repos: hooks: - id: mypy additional_dependencies: [types-requests] +- repo: https://github.com/Mateusz-Grzelinski/actionlint-py + rev: v1.7.7.23 + hooks: + - id: actionlint From 8828326375d0898fc34df663bf990b954bb404f5 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 10:19:06 -0400 Subject: [PATCH 05/30] Move every action pin off the Node.js 20 runtime The previous commit bumped checkout to v4 and setup-python to v5 to satisfy actionlint, but both of those majors still declare node20, so the runners kept reporting them as forced onto node24. actionlint 1.7.7 does not know about that deprecation, so it had nothing to say. Pin the lowest major of each action that declares node24, taken from action.yml at the tag rather than from the release notes: checkout v5, setup-python v6, upload-artifact v6, setup-qemu-action v4, and download-artifact v7 -- v5 and v6 of download-artifact are still node20, so it is the one that has to skip further ahead. cibuildwheel and gh-action-pypi-publish are composite actions and never had a node runtime to move. Choosing the lowest node24 major rather than the newest keeps the behavioural change to a minimum; there is no other reason to jump to checkout v7 today. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 4 ++-- .github/workflows/coverage.yml | 8 ++++---- .github/workflows/fusion-docs.yml | 4 ++-- .github/workflows/pre-commit.yml | 4 ++-- .github/workflows/publish.yml | 22 +++++++++++----------- .github/workflows/smoke-test.yml | 12 ++++++------ 6 files changed, 27 insertions(+), 27 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 0195ae88f..75114ef36 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -29,12 +29,12 @@ jobs: steps: - name: Checkout code - uses: actions/checkout@v4 + uses: actions/checkout@v5 with: fetch-depth: 2 - name: Set up Python - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: "3.10" cache: "pip" diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 023bdb44f..e2bd2a4a1 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -22,10 +22,10 @@ jobs: ROOT_PASSWORD: "root" steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: "3.10" cache: "pip" @@ -106,10 +106,10 @@ jobs: ROOT_PASSWORD: "root" steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: "3.10" cache: "pip" diff --git a/.github/workflows/fusion-docs.yml b/.github/workflows/fusion-docs.yml index 75a74ffa6..3a40c15c4 100644 --- a/.github/workflows/fusion-docs.yml +++ b/.github/workflows/fusion-docs.yml @@ -14,10 +14,10 @@ jobs: actions: write steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python 3.11 - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: 3.11 cache: "pip" diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml index 6b3d87393..10181d541 100644 --- a/.github/workflows/pre-commit.yml +++ b/.github/workflows/pre-commit.yml @@ -16,10 +16,10 @@ jobs: - "3.13" steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: ${{ matrix.python-version }} diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index fa88d195b..3a1f5b583 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -39,7 +39,7 @@ jobs: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Install dependencies run: | @@ -93,12 +93,12 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 # This job's matrix varies only over os; cibuildwheel supplies its own # interpreters, so 3.10 here is just the host Python that drives it. - name: Set up Python - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: "3.10" cache: "pip" @@ -122,7 +122,7 @@ jobs: - name: Set up QEMU if: runner.os == 'Linux' - uses: docker/setup-qemu-action@v3 + uses: docker/setup-qemu-action@v4 with: platforms: all @@ -174,14 +174,14 @@ jobs: mv ./wheelhouse/*.whl ./dist/. - name: Archive source dist and wheel - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v6 with: name: artifacts-${{ runner.os }} path: dist retention-days: 2 # - name: Archive conda -# uses: actions/upload-artifact@v4 +# uses: actions/upload-artifact@v6 # with: # name: conda-${{ matrix.os }} # path: ./conda-bld @@ -202,22 +202,22 @@ jobs: url: https://pypi.org/p/singlestoredb steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Download Linux wheels and sdist - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: artifacts-Linux path: dist - name: Download Windows wheels and sdist - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: artifacts-Windows path: dist - name: Download Mac wheels and sdist - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: artifacts-macOS path: dist @@ -264,7 +264,7 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Install dependencies run: | diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 50a35b5e3..34e0f0391 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -12,10 +12,10 @@ jobs: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python 3.11 - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: 3.11 cache: "pip" @@ -120,10 +120,10 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: ${{ matrix.python-version }} cache: "pip" @@ -172,10 +172,10 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Set up Python 3.11 - uses: actions/setup-python@v5 + uses: actions/setup-python@v6 with: python-version: 3.11 cache: "pip" From 0f16a6d6a6d1ae998f5fa25a50b58e28998d54fe Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 10:26:14 -0400 Subject: [PATCH 06/30] Pin every action at its newest major Follows the node24 move rather than stopping at the lowest major that cleared it: checkout v7, setup-python v7, upload-artifact v7, download-artifact v8, setup-qemu-action v4. One deprecation cycle instead of two. Checked the breaking changes in the majors this skips over: - setup-python dropped its default Python version, so a step that names neither python-version nor python-version-file now fails. All nine call sites name python-version, so none are affected. - download-artifact v5 changed the path layout for single downloads by ID. publish.yml downloads by name, so it is untouched. - download-artifact v8 stopped unzipping unconditionally, checking Content-Type first, and now errors on a hash mismatch. The artifacts here are ordinary zipped directory uploads, so both apply harmlessly. cibuildwheel and gh-action-pypi-publish stay where they are: both are composite actions, so neither was part of the node problem, and moving cibuildwheel two majors is a build-behaviour change that does not belong in this PR. The artifact pins are the ones with no coverage here -- publish.yml runs only on a tag or a release, so they are first exercised by the next release build. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 4 ++-- .github/workflows/coverage.yml | 8 ++++---- .github/workflows/fusion-docs.yml | 4 ++-- .github/workflows/pre-commit.yml | 4 ++-- .github/workflows/publish.yml | 20 ++++++++++---------- .github/workflows/smoke-test.yml | 12 ++++++------ 6 files changed, 26 insertions(+), 26 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 75114ef36..36b44259a 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -29,12 +29,12 @@ jobs: steps: - name: Checkout code - uses: actions/checkout@v5 + uses: actions/checkout@v7 with: fetch-depth: 2 - name: Set up Python - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: "3.10" cache: "pip" diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index e2bd2a4a1..182a4247c 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -22,10 +22,10 @@ jobs: ROOT_PASSWORD: "root" steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: "3.10" cache: "pip" @@ -106,10 +106,10 @@ jobs: ROOT_PASSWORD: "root" steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: "3.10" cache: "pip" diff --git a/.github/workflows/fusion-docs.yml b/.github/workflows/fusion-docs.yml index 3a40c15c4..bfbb35d01 100644 --- a/.github/workflows/fusion-docs.yml +++ b/.github/workflows/fusion-docs.yml @@ -14,10 +14,10 @@ jobs: actions: write steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python 3.11 - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: 3.11 cache: "pip" diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml index 10181d541..aa6145076 100644 --- a/.github/workflows/pre-commit.yml +++ b/.github/workflows/pre-commit.yml @@ -16,10 +16,10 @@ jobs: - "3.13" steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: ${{ matrix.python-version }} diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 3a1f5b583..5ee454e0f 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -39,7 +39,7 @@ jobs: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Install dependencies run: | @@ -93,12 +93,12 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 # This job's matrix varies only over os; cibuildwheel supplies its own # interpreters, so 3.10 here is just the host Python that drives it. - name: Set up Python - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: "3.10" cache: "pip" @@ -174,14 +174,14 @@ jobs: mv ./wheelhouse/*.whl ./dist/. - name: Archive source dist and wheel - uses: actions/upload-artifact@v6 + uses: actions/upload-artifact@v7 with: name: artifacts-${{ runner.os }} path: dist retention-days: 2 # - name: Archive conda -# uses: actions/upload-artifact@v6 +# uses: actions/upload-artifact@v7 # with: # name: conda-${{ matrix.os }} # path: ./conda-bld @@ -202,22 +202,22 @@ jobs: url: https://pypi.org/p/singlestoredb steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Download Linux wheels and sdist - uses: actions/download-artifact@v7 + uses: actions/download-artifact@v8 with: name: artifacts-Linux path: dist - name: Download Windows wheels and sdist - uses: actions/download-artifact@v7 + uses: actions/download-artifact@v8 with: name: artifacts-Windows path: dist - name: Download Mac wheels and sdist - uses: actions/download-artifact@v7 + uses: actions/download-artifact@v8 with: name: artifacts-macOS path: dist @@ -264,7 +264,7 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Install dependencies run: | diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 34e0f0391..4764f1458 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -12,10 +12,10 @@ jobs: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python 3.11 - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: 3.11 cache: "pip" @@ -120,10 +120,10 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: ${{ matrix.python-version }} cache: "pip" @@ -172,10 +172,10 @@ jobs: echo "::add-mask::$CLUSTER_PASSWORD" echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v5 + - uses: actions/checkout@v7 - name: Set up Python 3.11 - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: 3.11 cache: "pip" From 7a0f3ba56afc5424c8c9fb9e3228a7c82d2d0017 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 10:31:30 -0400 Subject: [PATCH 07/30] Emit the timezone offset with a colon in to_datetime Go renders the offset without a separator -- the expiresAt values come back as '+0000' -- and datetime.fromisoformat only accepts that spelling on Python 3.11 and later. On 3.9 and 3.10 it raised, the converter returned the string unchanged, and to_datetime turned that into None: exactly the silent unset expiration the Go-format handling was added to fix, just on the interpreters the previous commit did not cover. Normalize the offset to +00:00. Verified that every shape the normalizer emits parses on 3.8, 3.10 and 3.11. The new test asserts on the normalized string rather than on a parsed datetime. The six existing tests were correct and still passed on 3.11, which is how this reached CI; a string comparison fails the same way on every version. Co-Authored-By: Claude Opus 5 --- singlestoredb/management/utils.py | 9 ++++++++- singlestoredb/tests/test_management_utils.py | 21 ++++++++++++++++++++ 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/singlestoredb/management/utils.py b/singlestoredb/management/utils.py index e00844dd3..781452f13 100644 --- a/singlestoredb/management/utils.py +++ b/singlestoredb/management/utils.py @@ -455,7 +455,14 @@ def _normalize_datetime(obj: str) -> str: micros = micros[:6] + '0' * (6 - len(micros)) stamp = stamp + '.' + micros - return stamp + (match.group('offset') or '') + # Go writes the offset without a separator (+0000). Only Python 3.11 and + # later accept that spelling; 3.9 and 3.10 want +00:00, so always emit the + # colon. + offset = match.group('offset') or '' + if offset and ':' not in offset: + offset = offset[:3] + ':' + offset[3:] + + return stamp + offset def _as_naive_utc(obj: datetime.datetime) -> datetime.datetime: diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index d2531bb9a..ba4f6e745 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -17,6 +17,7 @@ from unittest.mock import patch from singlestoredb.exceptions import ManagementError +from singlestoredb.management.utils import _normalize_datetime from singlestoredb.management.utils import normalize_remote_path from singlestoredb.management.utils import to_datetime from singlestoredb.management.utils import to_datetime_strict @@ -1837,6 +1838,26 @@ def test_go_time_string(self): out = to_datetime('2026-09-17 14:42:41.445984 +0000 UTC') self.assertEqual(out, datetime.datetime(2026, 9, 17, 14, 42, 41, 445984)) + def test_offset_is_normalized_to_include_a_colon(self): + # Go writes +0000; datetime.fromisoformat only accepts that spelling on + # 3.11 and later, so the normalizer has to insert the colon itself. This + # asserts on the normalized string rather than on a parsed result + # because the parsed result is only wrong on 3.9 and 3.10, which would + # leave the failure invisible to anyone testing on a newer interpreter. + self.assertEqual( + _normalize_datetime('2026-09-17 14:42:41.445984 +0000 UTC'), + '2026-09-17 14:42:41.445984+00:00', + ) + self.assertEqual( + _normalize_datetime('2026-09-17 09:42:41 +0530 IST'), + '2026-09-17 09:42:41+05:30', + ) + # An offset that already carries a colon is left as it is. + self.assertEqual( + _normalize_datetime('2026-09-17 09:42:41 +05:30 IST'), + '2026-09-17 09:42:41+05:30', + ) + def test_go_time_string_with_truncated_fraction(self): # Go trims trailing zeros, so the fraction is not always 6 digits. out = to_datetime('2026-09-17 14:42:41.4 +0000 UTC') From 855a593cc664328803d55ee6001ea37076c70b14 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 11:10:57 -0400 Subject: [PATCH 08/30] Reset the admin password instead of passing it between jobs POST /v2/clusters generates its own admin password and ignores any that is sent, so create_test_cluster.py was reading the generated one back off the create response and reporting it as a job output. That cannot work: the value has to be masked, and the runner drops any output whose value matches a mask -- "Skip output 'cluster-password' since it may contain secret" -- so the test jobs received an empty password and failed with "1045: Access denied for user 'admin'@... (using password: NO)". Take a --password instead and hand it to the new cluster over SQL once it is active, so every job reads the credential from secrets.CLUSTER_PASSWORD and nothing crosses a job boundary. ALTER USER is the statement that works; SET PASSWORD wants a 41-digit hash and rejects a literal. The percent-encoded variant and the per-job ::add-mask:: steps both go away with the output. Co-Authored-By: Claude Opus 5 --- .github/workflows/publish.yml | 50 +++++++---------------- .github/workflows/smoke-test.yml | 50 ++++++++--------------- resources/create_test_cluster.py | 70 +++++++++++++++++++------------- 3 files changed, 72 insertions(+), 98 deletions(-) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 5ee454e0f..30a4265a7 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -53,8 +53,15 @@ jobs: # fixed everywhere below. The project is named here rather than read # from a repo variable so the deployment target is visible in the # workflow and does not depend on repository settings. + # + # POST /v2/clusters generates its own admin password and ignores any + # that is sent, so the script resets it to CLUSTER_PASSWORD over SQL + # once the cluster is up. That keeps the credential a secret the runner + # masks everywhere, instead of a job output: the runner refuses to write + # an output whose value is masked, so a generated password could not + # reach these jobs at all. run: | - python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" + python resources/create_test_cluster.py --password="${{ secrets.CLUSTER_PASSWORD }}" --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: PYTHONPATH: ${{ github.workspace }} @@ -62,11 +69,6 @@ jobs: cluster-id: ${{ steps.initialize-database.outputs.cluster-id }} cluster-host: ${{ steps.initialize-database.outputs.cluster-host }} cluster-database: ${{ steps.initialize-database.outputs.cluster-database }} - # POST /v2/clusters generates the admin password and reports it only on - # the create response, so it travels as a job output rather than living - # in secrets. Job outputs are not secrets: every job below re-masks it. - cluster-password: ${{ steps.initialize-database.outputs.cluster-password }} - cluster-password-url: ${{ steps.initialize-database.outputs.cluster-password-url }} build-and-test: needs: setup-database @@ -82,17 +84,6 @@ jobs: - windows-2022 steps: - # ::add-mask:: does not cross job boundaries, so the generated admin - # password arrives here unmasked and has to be registered again before - # any step can echo it into the log. - - name: Mask cluster password - env: - CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} - CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} - run: | - echo "::add-mask::$CLUSTER_PASSWORD" - echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v7 # This job's matrix varies only over os; cibuildwheel supplies its own @@ -140,12 +131,10 @@ jobs: # points --pyargs at the workspace, so that pyproject is the inifile # here. Without the plugin pytest exits on the unknown arguments. CIBW_TEST_REQUIRES: "pytest pytest-xdist" - # cluster-password-url, not cluster-password: the generated password - # is drawn from the full printable set, and this value has to survive - # both the userinfo half of the URL and the single-quoted shell word - # cibuildwheel evaluates. Percent-encoding leaves only unreserved - # characters and %, which are inert in both. - CIBW_ENVIRONMENT: "SINGLESTOREDB_URL='mysql://admin:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=0'" + # CLUSTER_PASSWORD has to survive both the userinfo half of the URL + # and the single-quoted shell word cibuildwheel evaluates, so keep the + # secret alphanumeric: no ':', '@', '/', '%' or quote characters. + CIBW_ENVIRONMENT: "SINGLESTOREDB_URL='mysql://admin:${{ secrets.CLUSTER_PASSWORD }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=0'" PYTHONPATH: ${{ github.workspace }} # - name: Build conda @@ -255,15 +244,6 @@ jobs: runs-on: ubuntu-latest steps: - # ::add-mask:: does not cross job boundaries; see the build-and-test job. - - name: Mask cluster password - env: - CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} - CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} - run: | - echo "::add-mask::$CLUSTER_PASSWORD" - echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v7 - name: Install dependencies @@ -275,13 +255,13 @@ jobs: - name: Drop database if: ${{ always() }} # The password reaches the script through the environment rather than - # being interpolated into the command: the generated value can contain - # any printable character, including ones the shell would act on. + # being interpolated into the command, so the shell never sees its + # characters. run: | python resources/drop_db.py --user admin --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" env: PYTHONPATH: ${{ github.workspace }} - CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_PASSWORD: ${{ secrets.CLUSTER_PASSWORD }} CLUSTER_HOST: ${{ needs.setup-database.outputs.cluster-host }} CLUSTER_DATABASE: ${{ needs.setup-database.outputs.cluster-database }} diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 4764f1458..4253a1bb4 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -32,8 +32,15 @@ jobs: # fixed everywhere below. The project is named here rather than read # from a repo variable so the deployment target is visible in the # workflow and does not depend on repository settings. + # + # POST /v2/clusters generates its own admin password and ignores any + # that is sent, so the script resets it to CLUSTER_PASSWORD over SQL + # once the cluster is up. That keeps the credential a secret the runner + # masks everywhere, instead of a job output: the runner refuses to write + # an output whose value is masked, so a generated password could not + # reach these jobs at all. run: | - python resources/create_test_cluster.py --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" + python resources/create_test_cluster.py --password="${{ secrets.CLUSTER_PASSWORD }}" --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: PYTHONPATH: ${{ github.workspace }} @@ -41,11 +48,6 @@ jobs: cluster-id: ${{ steps.initialize-database.outputs.cluster-id }} cluster-host: ${{ steps.initialize-database.outputs.cluster-host }} cluster-database: ${{ steps.initialize-database.outputs.cluster-database }} - # POST /v2/clusters generates the admin password and reports it only on - # the create response, so it travels as a job output rather than living - # in secrets. Job outputs are not secrets: every job below re-masks it. - cluster-password: ${{ steps.initialize-database.outputs.cluster-password }} - cluster-password-url: ${{ steps.initialize-database.outputs.cluster-password-url }} smoke-test: @@ -109,17 +111,6 @@ jobs: buffered: 1 steps: - # ::add-mask:: does not cross job boundaries, so the generated admin - # password arrives here unmasked and has to be registered again before - # any step can echo it into the log. - - name: Mask cluster password - env: - CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} - CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} - run: | - echo "::add-mask::$CLUSTER_PASSWORD" - echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v7 - name: Set up Python ${{ matrix.python-version }} @@ -138,10 +129,10 @@ jobs: run: pytest -v --pyargs singlestoredb.tests.test_basics env: PYTHONPATH: ${{ github.workspace }} - # cluster-password-url, not cluster-password: the generated password - # is drawn from the full printable set and has to be percent-encoded - # to survive the userinfo half of the URL. - SINGLESTOREDB_URL: "${{ matrix.driver }}://admin:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" + # CLUSTER_PASSWORD goes into the userinfo half of a URL here, so it + # has to be free of characters that would need percent-encoding -- + # ':', '@', '/', '%' and the like. Keep the secret alphanumeric. + SINGLESTOREDB_URL: "${{ matrix.driver }}://admin:${{ secrets.CLUSTER_PASSWORD }}@${{ needs.setup-database.outputs.cluster-host }}:3306/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" - name: Run tests if: ${{ matrix.driver == 'https' }} @@ -154,7 +145,7 @@ jobs: run: pytest -v -n 0 --pyargs singlestoredb.tests.test_basics env: PYTHONPATH: ${{ github.workspace }} - SINGLESTOREDB_URL: "${{ matrix.driver }}://admin:${{ needs.setup-database.outputs.cluster-password-url }}@${{ needs.setup-database.outputs.cluster-host }}:443/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" + SINGLESTOREDB_URL: "${{ matrix.driver }}://admin:${{ secrets.CLUSTER_PASSWORD }}@${{ needs.setup-database.outputs.cluster-host }}:443/${{ needs.setup-database.outputs.cluster-database }}?pure_python=${{ matrix.pure-python }}&buffered=${{ matrix.buffered }}" shutdown-database: @@ -163,15 +154,6 @@ jobs: runs-on: ubuntu-latest steps: - # ::add-mask:: does not cross job boundaries; see the smoke-test job. - - name: Mask cluster password - env: - CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} - CLUSTER_PASSWORD_URL: ${{ needs.setup-database.outputs.cluster-password-url }} - run: | - echo "::add-mask::$CLUSTER_PASSWORD" - echo "::add-mask::$CLUSTER_PASSWORD_URL" - - uses: actions/checkout@v7 - name: Set up Python 3.11 @@ -189,13 +171,13 @@ jobs: - name: Drop database if: ${{ always() }} # The password reaches the script through the environment rather than - # being interpolated into the command: the generated value can contain - # any printable character, including ones the shell would act on. + # being interpolated into the command, so the shell never sees its + # characters. run: | python resources/drop_db.py --user admin --password "$CLUSTER_PASSWORD" --host "$CLUSTER_HOST" --port 3306 --database "$CLUSTER_DATABASE" env: PYTHONPATH: ${{ github.workspace }} - CLUSTER_PASSWORD: ${{ needs.setup-database.outputs.cluster-password }} + CLUSTER_PASSWORD: ${{ secrets.CLUSTER_PASSWORD }} CLUSTER_HOST: ${{ needs.setup-database.outputs.cluster-host }} CLUSTER_DATABASE: ${{ needs.setup-database.outputs.cluster-database }} diff --git a/resources/create_test_cluster.py b/resources/create_test_cluster.py index 8c258eb02..7c0b5040d 100755 --- a/resources/create_test_cluster.py +++ b/resources/create_test_cluster.py @@ -2,7 +2,6 @@ # type: ignore from __future__ import annotations -import json import os import random import re @@ -10,7 +9,6 @@ import sys import uuid from optparse import OptionParser -from urllib.parse import quote import singlestoredb as s2 @@ -35,6 +33,12 @@ default='S-00', help='size of the cluster (S-00)', ) +parser.add_option( + '-p', '--password', + help='password to give the admin user once the cluster is up; required, ' + 'because the password the API generates cannot be handed to another ' + 'CI job (see below)', +) parser.add_option( '-t', '--token', help='API key for the management API', @@ -68,6 +72,10 @@ parser.print_help() sys.exit(1) +if not options.password: + print('ERROR: --password is required', file=sys.stderr) + sys.exit(1) + if options.init_sql and not os.path.isfile(options.init_sql): print(f'ERROR: Could not locate SQL file: {options.init_sql}', file=sys.stderr) sys.exit(1) @@ -160,26 +168,14 @@ def candidates(item): # after any refresh(). See item 9 of docs/management-api-audit.md: the API # accepts an adminPassword on both POST and PATCH and ignores both, which is # why this is read back rather than set. -password = cluster.admin_password -if not password: +generated = cluster.admin_password +if not generated: print( 'ERROR: cluster was created without a readable admin password', file=sys.stderr, ) sys.exit(1) -# The generated password is drawn from the full printable set -- one observed -# value was ``{:D}TK*[F3Ll}Ups2pNv`` -- so it cannot be dropped into the -# userinfo half of a connection URL as-is. Percent-encode everything, since -# the URL parser runs unquote_plus over the password -# (singlestoredb/connection.py:287); encoding ``+`` too is what keeps that from -# turning into a space. -password_url = quote(password, safe='') - -database = options.database -if not database: - database = 'TEMP_{}'.format(uuid.uuid4()).replace('-', '_') - host = cluster.endpoint if ':' in host: host, port = host.split(':', 1) @@ -187,35 +183,51 @@ def candidates(item): else: port = 3306 -# Print cluster information +# Trade the generated password for the caller's, because the generated one +# cannot leave this process. A caller running under GitHub Actions has to mask +# it, and the runner drops any output whose value matches a mask -- "Skip output +# 'cluster-password' since it may contain secret" -- so masking it and passing +# it to another job are mutually exclusive. The password the caller already +# holds has neither problem. +# +# ALTER USER is the statement that works: SET PASSWORD wants a pre-hashed value +# and rejects a literal with '1372: Password hash should be a 41-digit +# hexadecimal number'. Verified against a live S-00 cluster, including that the +# control plane leaves the new password alone afterwards. +password = options.password +escaped = password.replace('\\', '\\\\').replace("'", "\\'") + +with s2.connect( + host=host, port=port, user='admin', + password=generated, connect_timeout=30, +) as conn: + with conn.cursor() as cur: + cur.execute(f"ALTER USER 'admin'@'%' IDENTIFIED BY '{escaped}'") + +database = options.database +if not database: + database = 'TEMP_{}'.format(uuid.uuid4()).replace('-', '_') + +# Print cluster information. No password is reported: the caller passed it in, +# so it already knows it, and under GitHub Actions it is a secret the runner +# masks on its own. if options.output == 'env': print(f'CLUSTER_ID={cluster.id}') print(f'CLUSTER_HOST={host}') print(f'CLUSTER_PORT={port}') print(f'CLUSTER_DATABASE={database}') - print(f'CLUSTER_PASSWORD={password}') - print(f'CLUSTER_PASSWORD_URL={password_url}') elif options.output == 'github': - # Register both forms with the runner before anything can log them. This - # only holds within this job; each job that consumes the outputs has to - # mask them again for itself. - print(f'::add-mask::{password}') - print(f'::add-mask::{password_url}') with open(os.environ['GITHUB_OUTPUT'], 'a') as output: print(f'cluster-id={cluster.id}', file=output) print(f'cluster-host={host}', file=output) print(f'cluster-port={port}', file=output) print(f'cluster-database={database}', file=output) - print(f'cluster-password={password}', file=output) - print(f'cluster-password-url={password_url}', file=output) elif options.output == 'json': print('{') print(f' "cluster-id": "{cluster.id}",') print(f' "cluster-host": "{host}",') print(f' "cluster-port": {port},') - print(f' "cluster-database": "{database}",') - print(f' "cluster-password": {json.dumps(password)},') - print(f' "cluster-password-url": "{password_url}"') + print(f' "cluster-database": "{database}"') print('}') # Initialize the database From 51f6b148b9424ae9fdc1b58c43cc7ccaf14c3af7 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 17 Sep 2026 12:38:41 -0400 Subject: [PATCH 09/30] Give the change detector the ref it compares against code-check.yml checks singlestoredb/management and singlestoredb/fusion for changes and picks between a run that includes the v2 management tests and one that excludes them. It has always picked the second. The checkout was shallow, fetch-depth: 2, which creates no origin/main, so every diff against it died -- "fatal: bad revision 'origin/main'" appears twice in each run log -- and the `|| true` turned that into an empty file list, which reads as "nothing changed". The step that runs -m 'not management_v1' was unreachable and the -m 'not management' one always won, so the 37 live v2 management tests never ran on a pull request; only the nightly coverage.yml covered them. Fetch the full history so origin/main exists, which also repairs the branch-push path that probed origin/main and origin/master and fell through to HEAD~1. Drop the `|| true` as well: a git failure means the comparison did not happen, and swallowing it silently downgrades the run rather than reporting the breakage. Expect this job to get slower on any PR touching those directories -- the tests it now selects deploy real clusters. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 36b44259a..318778f29 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -31,7 +31,12 @@ jobs: - name: Checkout code uses: actions/checkout@v7 with: - fetch-depth: 2 + # Full history, because the change detector below diffs against + # origin/main. A shallow clone does not create that ref -- with + # fetch-depth: 2 every diff died on `fatal: bad revision + # 'origin/main'`, which the detector read as "nothing changed", so + # the management step never ran on a PR. + fetch-depth: 0 - name: Set up Python uses: actions/setup-python@v7 @@ -94,7 +99,10 @@ jobs: for DIR in $MONITORED_DIRS; do if [ -d "$DIR" ]; then - CHANGED_FILES=$(git diff --name-only $BASE_COMMIT HEAD -- "$DIR" || true) + # No `|| true` here: a git failure means the comparison did not + # happen, and swallowing it silently downgrades the run to the + # no-management path instead of reporting the breakage. + CHANGED_FILES=$(git diff --name-only "$BASE_COMMIT" HEAD -- "$DIR") if [ -n "$CHANGED_FILES" ]; then echo "✅ Changes detected in: $DIR" echo "Files changed:" From 46610d72eb455648aa55ed5c0c0b01b7f24b3efe Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Fri, 18 Sep 2026 09:48:26 -0400 Subject: [PATCH 10/30] Pad the fraction on RFC 3339 timestamps too A job's createdAt came back as '2026-09-18T12:39:20.43888Z'. The zone group in _GO_DATETIME_RE demanded whitespace ahead of it, so a bare Z never matched and the value fell through to the escape hatch, which strips the Z and skips the fractional-second padding. Only Python 3.11 and later read a fraction that is neither 3 nor 6 digits, so on 3.10 the converter handed back the string and to_datetime_strict raised ValueError, taking down TestJobsFusion.test_run_wait_drop_job in CI. Recognize Z as an offset so RFC 3339 goes down the same path as the Go shape and gets its fraction padded, and spell the offset out as +00:00 for the same reason the numeric ones grew a colon: nothing before 3.11 parses the short form. _as_naive_utc shifts it back off, so parsed results are unchanged. Verified on a real 3.10 that every shape the normalizer emits parses, and that the old output for this value does not. The tests assert on the normalized string, not on a parsed datetime, so they fail on 3.11 as well -- the same blind spot that let the offset bug reach CI. Co-Authored-By: Claude Opus 5 --- singlestoredb/management/utils.py | 24 +++++++++++++------- singlestoredb/tests/test_management_utils.py | 21 +++++++++++++++++ 2 files changed, 37 insertions(+), 8 deletions(-) diff --git a/singlestoredb/management/utils.py b/singlestoredb/management/utils.py index 781452f13..52aca29fb 100644 --- a/singlestoredb/management/utils.py +++ b/singlestoredb/management/utils.py @@ -414,9 +414,13 @@ def enable_http_tracing() -> None: #: RFC 3339, and the trailing zone name is not ISO 8601, so the whole value #: fails to parse and the expiration silently reads as unset. The zone name and #: the monotonic-clock reading Go appends to some values are both optional. +#: An RFC 3339 ``Z`` counts as an offset here so that shape goes down the same +#: path: its fraction needs the same padding, and until it matched, a value like +#: ``...20.43888Z`` reached the converter with five digits, which only 3.11 and +#: later parse. _GO_DATETIME_RE = re.compile( r'^(?P\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}(?:\.\d+)?)' - r'(?:\s*(?P[+-]\d{2}:?\d{2}))?' + r'(?:\s*(?P[Zz]|[+-]\d{2}:?\d{2}))?' r'(?:\s+(?P[A-Za-z]\S*))?' r'(?:\s+m=\S+)?$', ) @@ -429,7 +433,8 @@ def _normalize_datetime(obj: str) -> str: Handles the two shapes the management API returns -- RFC 3339 and the Go ``time.Time.String()`` form -- by reducing both to a bare ISO 8601 timestamp plus an optional numeric offset. Fractional seconds are padded to - microseconds, since Go trims trailing zeros. + microseconds -- Go trims trailing zeros, and ``datetime.fromisoformat`` + accepts only 3 or 6 digits before Python 3.11. Parameters ---------- @@ -457,9 +462,11 @@ def _normalize_datetime(obj: str) -> str: # Go writes the offset without a separator (+0000). Only Python 3.11 and # later accept that spelling; 3.9 and 3.10 want +00:00, so always emit the - # colon. + # colon. Z is spelled out for the same reason: nothing before 3.11 reads it. offset = match.group('offset') or '' - if offset and ':' not in offset: + if offset in ('Z', 'z'): + offset = '+00:00' + elif offset and ':' not in offset: offset = offset[:3] + ':' + offset[3:] return stamp + offset @@ -469,10 +476,11 @@ def _as_naive_utc(obj: datetime.datetime) -> datetime.datetime: """ Return ``obj`` as a naive UTC datetime. - An RFC 3339 timestamp loses its ``Z`` before it is parsed, so it arrives - here naive and already meaning UTC. A value carrying a numeric offset is - shifted onto UTC and stripped, so both shapes end up on the one convention - -- otherwise two timestamps read off the same object could not be compared. + A value carrying an offset -- which is every recognized shape, since an + RFC 3339 ``Z`` is normalized to ``+00:00`` -- is shifted onto UTC and + stripped. A value that arrives naive is already meaning UTC and is left + alone. Both end up on the one convention -- otherwise two timestamps read + off the same object could not be compared. Parameters ---------- diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index ba4f6e745..b2662c252 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -1858,6 +1858,27 @@ def test_offset_is_normalized_to_include_a_colon(self): '2026-09-17 09:42:41+05:30', ) + def test_rfc_3339_fraction_is_padded(self): + # The API trims trailing zeros here too: a job's createdAt came back as + # '2026-09-18T12:39:20.43888Z'. Only 3.11 and later read a fraction that + # is neither 3 nor 6 digits, so before Z was recognized as an offset this + # value skipped the padding and to_datetime_strict raised on 3.10. + self.assertEqual( + _normalize_datetime('2026-09-18T12:39:20.43888Z'), + '2026-09-18T12:39:20.438880+00:00', + ) + self.assertEqual( + to_datetime_strict('2026-09-18T12:39:20.43888Z'), + datetime.datetime(2026, 9, 18, 12, 39, 20, 438880), + ) + + def test_rfc_3339_nanoseconds_are_truncated(self): + # Nine digits does not fit a datetime; the extra ones are dropped. + self.assertEqual( + _normalize_datetime('2026-09-18T12:39:20.438880123Z'), + '2026-09-18T12:39:20.438880+00:00', + ) + def test_go_time_string_with_truncated_fraction(self): # Go trims trailing zeros, so the fraction is not always 6 digits. out = to_datetime('2026-09-17 14:42:41.4 +0000 UTC') From 1378da4d23fd388fe7892fd869080f69f3acd753 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Fri, 18 Sep 2026 13:57:31 -0400 Subject: [PATCH 11/30] Smoke-test on Python 3.14 The matrix stopped at 3.13 while 3.14 has been final since October 2025, so the newest interpreter the package claims to support -- requires-python is >=3.9 with no ceiling -- went untested. Crossed with the driver axis this adds two jobs, mysql and https. 3.15 is left out on purpose: it is at rc2 today with GA planned for 2026-10-01, and setup-python will not resolve a bare "3.15" until then. The condition for adding it is recorded above the matrix. Not touched: the include: block still pins macOS and Windows to 3.11, so 3.14 is covered on Linux only, and publish.yml builds one abi3 wheel from cp39, which needs no change for a new minor. Co-Authored-By: Claude Opus 5 --- .github/workflows/smoke-test.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 4253a1bb4..7e3bbf01f 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -59,12 +59,18 @@ jobs: matrix: os: - ubuntu-24.04 + # Every version from the floor in pyproject.toml (requires-python + # >=3.9) up to the newest final release. 3.15 is deliberately absent: + # as of 2026-09-18 it is at rc2 with GA planned for 2026-10-01, and + # setup-python needs allow-prereleases plus an explicit "3.15.0-rc.2" + # to install it at all. Add a bare "3.15" once it ships. python-version: - "3.9" - "3.10" - "3.11" - "3.12" - "3.13" + - "3.14" driver: - mysql - https From 4e49cfa58e7b07048907ef5ec4f17df43bde1348 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Fri, 18 Sep 2026 14:18:37 -0400 Subject: [PATCH 12/30] Report the cluster ID before anything else can fail Two findings from review. resources/create_test_cluster.py reported the cluster ID after the password reset and the SQL load, both of which can fail against a cluster that is already running and already billing. A CI teardown job would then have an empty cluster-id output and send its DELETE to /v2/clusters/, leaking the cluster. Everything the reporting block prints is known as soon as create_cluster() returns, so it now runs there. The two workflow shutdown steps also refuse to issue a DELETE with an empty ID, and --fail-with-body makes a refused one fail the step. to_datetime read Go's zero time as year 1 whenever it arrived in the Go shape -- '0001-01-01 00:00:00 +0000 UTC' -- because only the RFC 3339 spelling was compared against. That reports an expiry on a resource that does not expire. Both helpers now test the parsed value for January 1 of year 1, which covers every spelling including the offset, zone name and monotonic reading, and the check runs before the UTC shift, which can fall below MINYEAR on a year-1 value. Co-Authored-By: Claude Opus 5 --- .github/workflows/publish.yml | 10 ++- .github/workflows/smoke-test.yml | 10 ++- resources/create_test_cluster.py | 71 +++++++++++--------- singlestoredb/management/utils.py | 42 ++++++++++-- singlestoredb/tests/test_management_utils.py | 17 +++++ 5 files changed, 109 insertions(+), 41 deletions(-) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 30a4265a7..be006ab2d 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -267,7 +267,15 @@ jobs: - name: Shutdown cluster if: ${{ always() }} + # An empty ID would send the DELETE to /v2/clusters/ and leave a live + # cluster behind, so say so loudly instead: at that point the ID has to + # be recovered by hand. --fail-with-body is what makes a refused DELETE + # fail this step rather than printing the error and exiting 0. run: | - curl -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v2/clusters/${{ env.CLUSTER_ID }}?force=true" + if [ -z "$CLUSTER_ID" ]; then + echo "::error::No cluster ID from setup-database; the cluster (if any) must be terminated by hand" + exit 1 + fi + curl --fail-with-body -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v2/clusters/$CLUSTER_ID?force=true" env: CLUSTER_ID: ${{ needs.setup-database.outputs.cluster-id }} diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 7e3bbf01f..f99b820e3 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -189,7 +189,15 @@ jobs: - name: Shutdown cluster if: ${{ always() }} + # An empty ID would send the DELETE to /v2/clusters/ and leave a live + # cluster behind, so say so loudly instead: at that point the ID has to + # be recovered by hand. --fail-with-body is what makes a refused DELETE + # fail this step rather than printing the error and exiting 0. run: | - curl -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v2/clusters/${{ env.CLUSTER_ID }}?force=true" + if [ -z "$CLUSTER_ID" ]; then + echo "::error::No cluster ID from setup-database; the cluster (if any) must be terminated by hand" + exit 1 + fi + curl --fail-with-body -H "Accept: application/json" -H "Authorization: Bearer ${{ secrets.CLUSTER_API_KEY }}" -X DELETE "https://api.singlestore.com/v2/clusters/$CLUSTER_ID?force=true" env: CLUSTER_ID: ${{ needs.setup-database.outputs.cluster-id }} diff --git a/resources/create_test_cluster.py b/resources/create_test_cluster.py index 7c0b5040d..28be11c38 100755 --- a/resources/create_test_cluster.py +++ b/resources/create_test_cluster.py @@ -163,6 +163,44 @@ def candidates(item): wait_timeout=1200, ) +host = cluster.endpoint +if ':' in host: + host, port = host.split(':', 1) + port = int(port) +else: + port = 3306 + +database = options.database +if not database: + database = 'TEMP_{}'.format(uuid.uuid4()).replace('-', '_') + +# Report before touching the cluster any further. Everything below can fail +# against a cluster that already exists and is already billing, and the caller's +# only handle on it is the ID reported here -- a CI teardown job with an empty +# cluster-id output would issue its DELETE against /v2/clusters/ and leak the +# cluster it was meant to remove. +# +# No password is reported: the caller passed it in, so it already knows it, and +# under GitHub Actions it is a secret the runner masks on its own. +if options.output == 'env': + print(f'CLUSTER_ID={cluster.id}') + print(f'CLUSTER_HOST={host}') + print(f'CLUSTER_PORT={port}') + print(f'CLUSTER_DATABASE={database}') +elif options.output == 'github': + with open(os.environ['GITHUB_OUTPUT'], 'a') as output: + print(f'cluster-id={cluster.id}', file=output) + print(f'cluster-host={host}', file=output) + print(f'cluster-port={port}', file=output) + print(f'cluster-database={database}', file=output) +elif options.output == 'json': + print('{') + print(f' "cluster-id": "{cluster.id}",') + print(f' "cluster-host": "{host}",') + print(f' "cluster-port": {port},') + print(f' "cluster-database": "{database}"') + print('}') + # The API generates the admin password and reports it only on the create # response -- there is no route that will hand it back later, and it is None # after any refresh(). See item 9 of docs/management-api-audit.md: the API @@ -176,13 +214,6 @@ def candidates(item): ) sys.exit(1) -host = cluster.endpoint -if ':' in host: - host, port = host.split(':', 1) - port = int(port) -else: - port = 3306 - # Trade the generated password for the caller's, because the generated one # cannot leave this process. A caller running under GitHub Actions has to mask # it, and the runner drops any output whose value matches a mask -- "Skip output @@ -204,32 +235,6 @@ def candidates(item): with conn.cursor() as cur: cur.execute(f"ALTER USER 'admin'@'%' IDENTIFIED BY '{escaped}'") -database = options.database -if not database: - database = 'TEMP_{}'.format(uuid.uuid4()).replace('-', '_') - -# Print cluster information. No password is reported: the caller passed it in, -# so it already knows it, and under GitHub Actions it is a secret the runner -# masks on its own. -if options.output == 'env': - print(f'CLUSTER_ID={cluster.id}') - print(f'CLUSTER_HOST={host}') - print(f'CLUSTER_PORT={port}') - print(f'CLUSTER_DATABASE={database}') -elif options.output == 'github': - with open(os.environ['GITHUB_OUTPUT'], 'a') as output: - print(f'cluster-id={cluster.id}', file=output) - print(f'cluster-host={host}', file=output) - print(f'cluster-port={port}', file=output) - print(f'cluster-database={database}', file=output) -elif options.output == 'json': - print('{') - print(f' "cluster-id": "{cluster.id}",') - print(f' "cluster-host": "{host}",') - print(f' "cluster-port": {port},') - print(f' "cluster-database": "{database}"') - print('}') - # Initialize the database if options.init_sql: init_db = [ diff --git a/singlestoredb/management/utils.py b/singlestoredb/management/utils.py index 52aca29fb..ba6553de4 100644 --- a/singlestoredb/management/utils.py +++ b/singlestoredb/management/utils.py @@ -472,6 +472,31 @@ def _normalize_datetime(obj: str) -> str: return stamp + offset +def _is_go_zero_time(obj: Union[datetime.date, datetime.datetime]) -> bool: + """ + Return whether ``obj`` is Go's zero time, which means "unset". + + A Go ``time.Time`` that was never assigned renders as January 1 of year 1, + and the API returns that for a field it has no value for -- most visibly an + ``expiresAt`` on a resource that does not expire. It arrives spelled either + way the two timestamp shapes allow: ``0001-01-01T00:00:00Z`` and + ``0001-01-01 00:00:00 +0000 UTC``. Testing the parsed value rather than the + string covers both, along with any offset or monotonic reading that comes + with them. + + Parameters + ---------- + obj : datetime.date or datetime.datetime + Parsed timestamp + + Returns + ------- + bool + + """ + return (obj.year, obj.month, obj.day) == (1, 1, 1) + + def _as_naive_utc(obj: datetime.datetime) -> datetime.datetime: """ Return ``obj`` as a naive UTC datetime. @@ -505,15 +530,18 @@ def to_datetime( return None if isinstance(obj, datetime.datetime): return obj - if obj == '0001-01-01T00:00:00Z': - return None out = converters.datetime_fromisoformat(_normalize_datetime(obj)) if isinstance(out, str): return None - if isinstance(out, datetime.date) and not isinstance(out, datetime.datetime): - return datetime.datetime(out.year, out.month, out.day) if out is None: return None + # Before _as_naive_utc: shifting an aware year-1 value onto UTC can carry it + # below datetime.MINYEAR, which raises rather than returning the None this + # value means. + if _is_go_zero_time(out): + return None + if isinstance(out, datetime.date) and not isinstance(out, datetime.datetime): + return datetime.datetime(out.year, out.month, out.day) return _as_naive_utc(out) @@ -525,13 +553,15 @@ def to_datetime_strict( raise TypeError('not possible to convert None to datetime') if isinstance(obj, datetime.datetime): return obj - if obj == '0001-01-01T00:00:00Z': - raise ValueError('not possible to convert 0001-01-01T00:00:00Z to datetime') out = converters.datetime_fromisoformat(_normalize_datetime(obj)) if not out: raise TypeError('not possible to convert None to datetime') if isinstance(out, str): raise ValueError('value cannot be str') + # See to_datetime: checked here rather than after the UTC shift, which can + # raise on a year-1 value. + if _is_go_zero_time(out): + raise ValueError(f'not possible to convert {obj} to datetime') if isinstance(out, datetime.date) and not isinstance(out, datetime.datetime): return datetime.datetime(out.year, out.month, out.day) return _as_naive_utc(out) diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index b2662c252..685c352b1 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -1914,6 +1914,19 @@ def test_zero_sentinel_and_unparseable_are_none(self): self.assertIsNone(to_datetime('')) self.assertIsNone(to_datetime('not a date')) + def test_the_go_spelling_of_the_zero_sentinel_is_none_too(self): + # Go's zero time means "unset" -- an expiresAt on a resource that does + # not expire -- and arrives in whichever shape the field uses. Reading + # the Go spelling as a real timestamp reported year 1 as an expiry. + self.assertIsNone(to_datetime('0001-01-01 00:00:00 +0000 UTC')) + # Recognized from the parsed value, so the trimmings Go may add do not + # each need their own literal. + self.assertIsNone( + to_datetime('0001-01-01 00:00:00 +0000 UTC m=+0.000000001'), + ) + self.assertIsNone(to_datetime('0001-01-01 00:00:00 +0000 GMT')) + self.assertIsNone(to_datetime('0001-01-01')) + def test_datetime_passes_through(self): given = datetime.datetime(2026, 9, 17, 13, 42, 41) self.assertIs(to_datetime(given), given) @@ -1928,6 +1941,10 @@ def test_strict_still_raises_on_nothing(self): with self.assertRaises(ValueError): to_datetime_strict('0001-01-01T00:00:00Z') + def test_strict_raises_on_the_go_spelling_of_the_sentinel(self): + with self.assertRaises(ValueError): + to_datetime_strict('0001-01-01 00:00:00 +0000 UTC') + if __name__ == '__main__': unittest.main() From af46210a90f25753c5ff849406e0127df9cf5622 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 21 Sep 2026 08:35:34 -0400 Subject: [PATCH 13/30] Take the default parallelism down to two workers Three workers put more clusters in flight at once than the org wants to carry: each worker builds its own shared cluster pool, so the ceiling is the management API's tolerance for concurrent provisioning and the cluster quota, not this host's CPUs. Two is the new default. The three CI steps that pass -n 0 name the default they override in a comment, so those move with it. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 2 +- .github/workflows/coverage.yml | 2 +- .github/workflows/smoke-test.yml | 2 +- pyproject.toml | 6 ++++-- 4 files changed, 7 insertions(+), 5 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 318778f29..aca56ae5d 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -185,7 +185,7 @@ jobs: SINGLESTOREDB_FUSION_ENABLE_HIDDEN: "1" - name: Run HTTP protocol tests - # -n 0 overrides the -n 3 in pyproject.toml's addopts: the HTTP/Data API + # -n 0 overrides the -n 2 in pyproject.toml's addopts: the HTTP/Data API # run must be serial. Setup goes over SINGLESTOREDB_INIT_DB_URL (MySQL), # so load_sql takes its `SET GLOBAL HTTP_PROXY_PORT` + `RESTART PROXY` # branch (singlestoredb/tests/utils.py:227) once per worker, and a proxy diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 182a4247c..95931418d 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -61,7 +61,7 @@ jobs: SINGLESTOREDB_FUSION_ENABLE_HIDDEN: "1" - name: Run HTTP protocol tests - # -n 0 overrides the -n 3 in pyproject.toml's addopts: the HTTP/Data API + # -n 0 overrides the -n 2 in pyproject.toml's addopts: the HTTP/Data API # run must be serial. Setup goes over SINGLESTOREDB_INIT_DB_URL (MySQL), # so load_sql takes its `SET GLOBAL HTTP_PROXY_PORT` + `RESTART PROXY` # branch (singlestoredb/tests/utils.py:227) once per worker, and a proxy diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index f99b820e3..7984a8068 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -142,7 +142,7 @@ jobs: - name: Run tests if: ${{ matrix.driver == 'https' }} - # -n 0 overrides the -n 3 in pyproject.toml's addopts: the Data API is + # -n 0 overrides the -n 2 in pyproject.toml's addopts: the Data API is # not run in parallel. This job avoids the `RESTART PROXY` hazard the # code-check/coverage HTTP steps hit -- no SINGLESTOREDB_INIT_DB_URL # here, so load_sql's setup connection is itself HTTP and skips that diff --git a/pyproject.toml b/pyproject.toml index 77ff8ca60..25750c421 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -101,13 +101,15 @@ exclude = ["docs*", "resources*", "examples*", "licenses*"] # and honours the xdist_group marks below. It is here rather than in the # invocation because forgetting it costs real money. # -# 3 workers, not `auto`: the ceiling is the management API's tolerance for +# 2 workers, not `auto`: the ceiling is the management API's tolerance for # concurrent provisioning and the org's cluster quota, not this host's CPUs. +# 3 was too many in practice -- the management suite had more clusters in +# flight at once than the org wanted to carry. # # Note that xdist must be installed for pytest to start at all with these set # (`pip install -e ".[test]"`), that SINGLESTOREDB_MANAGEMENT_TRACE's terminal # summary needs -n 0, and that USE_DATA_API=1 in parallel is unverified. -addopts = ["-n", "3", "--dist", "loadgroup"] +addopts = ["-n", "2", "--dist", "loadgroup"] markers = [ "management", "management_v1: exercises the v1 management API, which v2 has replaced. Deselect with -m 'not management_v1'; the v1 endpoints only need a nightly gate now that v2 is the default.", From 69a649c1a5ac5c9c6e899f2389bf51d9ac2a73b1 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 21 Sep 2026 14:35:07 -0400 Subject: [PATCH 14/30] Record every test deployment in a durable ledger Deployment tracking lived only in the test process. Run 35631802648's test-coverage job was cancelled 19 minutes into TestClusterFusion's create_cluster(wait_on_active=True, wait_timeout=1200): the log ends at '##[error]The operation was canceled.' with no pytest summary and no sweep output, so the clusters it had in flight were left billing with nothing on disk naming them, and no workflow ran a cleanup step that could have found them. So write a JSONL record on the way in ('pending', before the API call can be interrupted), on the way to live ('live', with the id), and on the way out ('gone'), and add an `if: always()` step at the end of every job that deploys, which folds the ledger and terminates whatever is still up. The ledger is opt-in on SINGLESTOREDB_TEST_DEPLOYMENT_LOG; unset, nothing changes. Write failures are logged, never raised -- this sits on every creation path and must not be able to fail a test. Also, for the same leak from the other end: - terminate() retries a 4xx refusal for three minutes. A deployment that is still provisioning answers 400/409, and RETRY_STATUSES covers only 429/5xx, so that refusal was never retried anywhere. It also picks force by inspecting the signature rather than catching TypeError, which could issue a second, unforced DELETE. - WorkspaceGroup.terminate and Cluster.terminate send force as 'true'/ 'false'. requests renders a bool through str(), so force=True went over the wire as `force=True`. - tearDownClass terminates before dropping the database, and per object, so one failure no longer strands the rest. test_get_secret and the two setUpClass rollbacks clean up the same way. - cleanup_deployments grows --ledger, and PATTERNS matches wg-foo-*, which TestWorkspace.test_update renames a live group to and never renames back. - A worker that ends with something still tracked reports it to the controller through workeroutput, which pytest_testnodedown prints. The sweep moves to pytest_sessionfinish because xdist sends workeroutput from a hookwrapper after the yield, and pytest_unconfigure runs after that. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 19 + .github/workflows/coverage.yml | 43 ++ singlestoredb/management/v1/workspace.py | 10 +- singlestoredb/management/v2/cluster.py | 7 +- singlestoredb/tests/cleanup_deployments.py | 340 ++++++++++- singlestoredb/tests/conftest.py | 69 +++ singlestoredb/tests/test_fusion.py | 64 +- singlestoredb/tests/test_management_utils.py | 606 ++++++++++++++++++- singlestoredb/tests/test_management_v1.py | 42 +- singlestoredb/tests/utils.py | 301 ++++++++- 10 files changed, 1430 insertions(+), 71 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index aca56ae5d..7861251a7 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -16,6 +16,12 @@ jobs: contents: read actions: write + # One ledger for the whole job, so the cleanup step at the end can reap + # what any of the pytest steps created. See the matching block in + # coverage.yml. + env: + SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl + services: singlestore: image: ghcr.io/singlestore-labs/singlestoredb-dev:latest @@ -209,3 +215,16 @@ jobs: coverage report coverage xml coverage html + + # if: always() is the whole point -- this has to run when the job is + # cancelled, which is what left three clusters billing in run + # 35631802648 (see the matching step in coverage.yml). On a PR the + # management step above only runs when the change detector fires, so most + # runs reach this with an empty ledger and it reports nothing. + - name: Terminate any deployment the tests left behind + if: always() + run: | + python -m singlestoredb.tests.cleanup_deployments \ + --ledger "$SINGLESTOREDB_TEST_DEPLOYMENT_LOG" --yes + env: + SINGLESTOREDB_MANAGEMENT_TOKEN: ${{ secrets.CLUSTER_API_KEY }} diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 95931418d..8eedfae86 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -10,6 +10,13 @@ jobs: runs-on: ubuntu-latest environment: Base + # One ledger for the whole job, so the cleanup step below can reap what any + # of the pytest steps created. Per job rather than shared: the jobs here run + # concurrently, and a shared ledger would have each one terminating the + # other's clusters mid-run. + env: + SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl + services: singlestore: image: ghcr.io/singlestore-labs/singlestoredb-dev:latest @@ -86,6 +93,25 @@ jobs: coverage xml coverage html + # if: always() is the whole point -- this has to run when the job is + # cancelled, which is the case that produced the leak. Run 35631802648 + # was cancelled 19 minutes into TestClusterFusion.setUpClass's + # create_cluster(wait_on_active=True, wait_timeout=1200); the log ends at + # '##[error]The operation was canceled.' with no pytest summary and no + # sweep output, so three clusters were left billing with nothing in the + # process having recorded them. The ledger is that record. + # + # Last step in the job so it covers every pytest step above it. Placing + # it after each one instead would add nothing: a cancellation anywhere + # still runs the remaining always() steps. + - name: Terminate any deployment the tests left behind + if: always() + run: | + python -m singlestoredb.tests.cleanup_deployments \ + --ledger "$SINGLESTOREDB_TEST_DEPLOYMENT_LOG" --yes + env: + SINGLESTOREDB_MANAGEMENT_TOKEN: ${{ secrets.CLUSTER_API_KEY }} + # The deprecated v1 management API. management.version defaults to v2, so this # is a legacy gate: it runs here nightly rather than on every PR, and it is # what gets deleted along with management/v1/. Selects both the mocked v1 @@ -94,6 +120,12 @@ jobs: runs-on: ubuntu-latest environment: Base + # A ledger of its own, not shared with test-coverage: the two jobs run + # concurrently, and one sweeping the other's ledger would terminate + # clusters a live run is using. + env: + SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl + services: singlestore: image: ghcr.io/singlestore-labs/singlestoredb-dev:latest @@ -128,3 +160,14 @@ jobs: SINGLESTORE_LICENSE: ${{ secrets.SINGLESTORE_LICENSE }} SINGLESTOREDB_MANAGEMENT_TOKEN: ${{ secrets.CLUSTER_API_KEY }} SINGLESTOREDB_FUSION_ENABLE_HIDDEN: "1" + + # See the matching step in test-coverage for why this is if: always(). + # The v1 suite deploys workspace groups, which are the kind that force + # exists for. + - name: Terminate any deployment the tests left behind + if: always() + run: | + python -m singlestoredb.tests.cleanup_deployments \ + --ledger "$SINGLESTOREDB_TEST_DEPLOYMENT_LOG" --yes + env: + SINGLESTOREDB_MANAGEMENT_TOKEN: ${{ secrets.CLUSTER_API_KEY }} diff --git a/singlestoredb/management/v1/workspace.py b/singlestoredb/management/v1/workspace.py index 718b292f9..de5938dac 100644 --- a/singlestoredb/management/v1/workspace.py +++ b/singlestoredb/management/v1/workspace.py @@ -850,7 +850,15 @@ def terminate( raise ManagementError( msg='No workspace manager is associated with this object.', ) - self._manager._delete(f'workspaceGroups/{self.id}', params=dict(force=force)) + # 'true'/'false', not the bool: requests renders a bool param with + # str(), so force=True went out as force=True. Workspace.terminate + # above already builds the lowercase form by hand; this matches it. + # force is what makes a group with live workspaces in it go away, so + # the value being read is not optional. + self._manager._delete( + f'workspaceGroups/{self.id}', + params=dict(force='true' if force else 'false'), + ) if wait_on_terminated: remaining = float(wait_timeout) while True: diff --git a/singlestoredb/management/v2/cluster.py b/singlestoredb/management/v2/cluster.py index 9c8765cf1..df51765fa 100644 --- a/singlestoredb/management/v2/cluster.py +++ b/singlestoredb/management/v2/cluster.py @@ -715,7 +715,12 @@ def terminate( """ manager = self._require_manager() - manager._delete(f'clusters/{self.id}', params=dict(force=force)) + # 'true'/'false', not the bool: requests renders a bool param with + # str(), so force=True went out as force=True. + manager._delete( + f'clusters/{self.id}', + params=dict(force='true' if force else 'false'), + ) if wait_on_terminated: remaining = float(wait_timeout) while True: diff --git a/singlestoredb/tests/cleanup_deployments.py b/singlestoredb/tests/cleanup_deployments.py index 7b8d745c2..d50fea1bd 100644 --- a/singlestoredb/tests/cleanup_deployments.py +++ b/singlestoredb/tests/cleanup_deployments.py @@ -42,20 +42,31 @@ are tracked as they are created and swept per test class by ``conftest.py``, which cannot see -- or touch -- another run's deployments. -Why strays keep appearing: that tracking, the per-class sweep and this script -all live on the ``versioned-management-api`` branch and nowhere else. A run -from ``main`` has only ``tearDownClass``, so a killed run or a ``setUpClass`` -that raises leaks a workspace group permanently, and ``main`` still uses names -this script only knows through :data:`LEGACY_PATTERNS`. Until the sweep is on -the default branch, expect to run this by hand. +The exception, and the reason this is wired into CI, is ``--ledger``. A run +with ``SINGLESTOREDB_TEST_DEPLOYMENT_LOG`` set records every creation to a +JSONL file as it happens (``utils.ledger_pending``/``ledger_live``/ +``ledger_gone``), so a run that was killed outright leaves an exact list of +what it made:: + + python -m singlestoredb.tests.cleanup_deployments --ledger deployments.jsonl + +That mode replaces *both* guards above -- the name patterns and the age +filter. Neither is needed, because the ledger names the deployments rather +than guessing at them, and neither is safe: a ledger entry is minutes old by +construction, so the age filter would spare everything it lists. What keeps +such a run off other people's deployments is that it only ever touches ids and +names the ledger records, and that each CI job writes its own ledger. """ import argparse import datetime +import json +import os import re import sys import warnings from collections.abc import Container from typing import Any +from typing import Dict from typing import List from typing import Optional from typing import Tuple @@ -84,6 +95,12 @@ PATTERNS = [ # test_management_v1.py / test_management_v2.py fixtures re.compile(r'^(wg|ws|cl)-test-[A-Za-z0-9_-]+$'), + # TestWorkspace.test_update renames its live group from wg-test- to + # wg-foo- and never renames it back, so the group carries this name + # for the rest of the class. No pattern matched it, which made a group + # stranded after that test invisible to this sweep -- it would pile up + # while the tool reported nothing. + re.compile(r'^wg-foo-[A-Za-z0-9_-]+$'), re.compile(r'^starter-(ws|cl)-test-[A-Za-z0-9_-]+$'), # test_fusion.py fixtures re.compile(r'^[A-C] Fusion Testing [0-9a-f]+$'), @@ -254,7 +271,7 @@ def keep(obj: Any) -> bool: if 'cluster' in kinds or 'starter-cluster' in kinds: try: - clusters = s2.manage_clusters(version='v2') + clusters = _manager('v2') except Exception as exc: print(f'! Could not reach management API v2: {exc}', file=sys.stderr) else: @@ -274,14 +291,7 @@ def keep(obj: Any) -> bool: if 'workspace-group' in kinds or 'starter-workspace' in kinds: try: - # v1 is deprecated, and asking for it here is the point: workspace - # groups exist nowhere else, so the warning is noise on every run. - with warnings.catch_warnings(): - warnings.filterwarnings( - 'ignore', category=DeprecationWarning, - message='.*manage_workspaces.*', - ) - workspaces = s2.manage_workspaces(version='v1') + workspaces = _manager('v1') except Exception as exc: print(f'! Could not reach management API v1: {exc}', file=sys.stderr) else: @@ -304,12 +314,297 @@ def keep(obj: Any) -> bool: return found, spared, unmatched +# +# Ledger mode +# +# What this exists for: GH Actions run 35631802648, job ``test-coverage``, was +# cancelled 19 minutes into a ``create_cluster(wait_on_active=True, +# wait_timeout=1200)`` and the log ends at ``##[error]The operation was +# canceled.`` with no pytest summary and no sweep output at all. Three clusters +# were live and no in-process handler ever ran. Reading a file written as the +# clusters were created is the only way to know that from another process. +# + +#: How each ledger kind is resolved back to a live object: the management API +#: version that owns it, the point lookup for a record that has an id, and the +#: listing to search by name for a ``pending`` record that never got one. +#: +#: The kinds are the values of ``utils._KIND_BY_CLASS``; a kind this does not +#: know is reported rather than skipped, since the alternative is silently not +#: reaping it. +LEDGER_KINDS = { + 'cluster': ( + 'v2', 'get_cluster', lambda mgr: mgr.clusters, + ), + 'starter_cluster': ( + 'v2', 'get_starter_cluster', lambda mgr: mgr.starter_clusters, + ), + 'workspace_group': ( + 'v1', 'get_workspace_group', lambda mgr: mgr.workspace_groups, + ), + 'workspace': ( + # WorkspaceManager has no `workspaces` of its own, so the search goes + # group by group -- the same walk utils._CREATORS uses. + 'v1', 'get_workspace', + lambda mgr: [w for g in mgr.workspace_groups for w in g.workspaces], + ), + 'starter_workspace': ( + 'v1', 'get_starter_workspace', lambda mgr: mgr.starter_workspaces, + ), +} + + +def _manager(version: str) -> Any: + """Management API manager for ``'v1'`` or ``'v2'``.""" + if version == 'v2': + return s2.manage_clusters(version='v2') + # v1 is deprecated, and asking for it here is the point: workspace groups + # exist nowhere else, so the warning is noise on every run. + with warnings.catch_warnings(): + warnings.filterwarnings( + 'ignore', category=DeprecationWarning, + message='.*manage_workspaces.*', + ) + return s2.manage_workspaces(version='v1') + + +def fold_ledger(lines: Any) -> List[Dict[str, Any]]: + """ + Reduce ledger records to the deployments that should still be live. + + The ledger is append-only and written from several processes (one per xdist + worker), so it is a history, not a state: a deployment shows up as + ``pending``, then ``live`` once it has an id, then ``gone`` once something + terminated it. Folding keeps whatever the last event for a deployment was + not ``gone``. + + A ``pending`` is keyed by ``(kind, name)`` because that is all it has; the + matching ``live`` retires it and re-keys on the id. So the two records a + normal creation writes collapse to one entry, and a ``pending`` left + standing means the creator was interrupted before it returned -- the + cancelled-mid-``wait_on_active`` case, resolvable only by name. + + Order is creation order, since dicts preserve insertion order and a + deployment's key is first inserted when it first appears. The caller + reverses it, so a workspace goes before the group that holds it, matching + ``utils.cleanup_tracked()``. + + Malformed lines are skipped with a warning rather than aborting: this runs + as the last step of a CI job, and one truncated line -- a process killed + between the ``write`` and the ``fsync``, which the per-line fsync makes + unlikely but not impossible -- must not stop the rest from being reaped. + """ + live: Dict[Any, Dict[str, Any]] = {} + + for lineno, line in enumerate(lines, start=1): + line = line.strip() + if not line: + continue + try: + record = json.loads(line) + except ValueError as exc: + print( + f'! ledger line {lineno} is not JSON, skipping it: {exc}', + file=sys.stderr, + ) + continue + if not isinstance(record, dict): + continue + + event = record.get('event') + kind = record.get('kind') + name = record.get('name') + ident = record.get('id') + + by_name = ('name', kind, name) + by_id = ('id', kind, ident) + + if event == 'pending': + if name is not None: + live.setdefault(by_name, record) + elif event == 'live': + live.pop(by_name, None) + if ident is not None: + live[by_id] = record + elif name is not None: + # No id in the record: keep it findable by name rather than + # dropping it. Should not happen, but losing the deployment is + # the expensive direction. + live[by_name] = record + elif event == 'gone': + if ident is not None: + live.pop(by_id, None) + live.pop(by_name, None) + + return list(live.values()) + + +def read_ledger(path: str) -> List[Dict[str, Any]]: + """ + Fold the ledger at ``path``, newest first. + + A missing file is not an error: the variable can be set on a job whose + tests created nothing, and a CI cleanup step that failed in that case would + turn every such run red. + """ + if not os.path.exists(path): + print(f'No ledger at {path}; nothing this run created was recorded.') + return [] + with open(path, encoding='utf-8') as file: + records = fold_ledger(file) + # Newest first, so a workspace is terminated before its group. + records.reverse() + return records + + +def find_ledger_leftovers( + path: str, +) -> Tuple[List[Tuple[str, Any]], List[str], List[str]]: + """ + Resolve the ledger's still-live records to live deployment objects. + + Returns + ------- + (List[Tuple[str, Any]], List[str], List[str]) + The deployments to terminate, labels for the records that resolved to + nothing -- already gone, so nothing to do -- and labels for the ones + that could not be resolved *and* could still be live, which is what + makes the run exit non-zero. + + A 404 from the point lookup means the deployment is already gone, which is + the common case: the ledger records every creation, and a run that finished + normally terminated all of them. Anything else -- a transport failure, an + unknown kind -- goes in the third list, because "could not tell" and "not + there" must not read the same when the difference is a cluster billing. + """ + from singlestoredb.exceptions import ManagementError + + found: List[Tuple[str, Any]] = [] + gone: List[str] = [] + unresolved: List[str] = [] + + managers: Dict[str, Any] = {} + + def manager_for(version: str) -> Any: + if version not in managers: + managers[version] = _manager(version) + return managers[version] + + for record in read_ledger(path): + kind = record.get('kind') + name = record.get('name') + ident = record.get('id') + label = '{} {} ({})'.format( + str(kind).replace('_', ' '), name or '', ident or 'no id', + ) + + if kind not in LEDGER_KINDS: + unresolved.append(f'{label}: unknown kind {kind!r}') + continue + version, lookup_name, listing = LEDGER_KINDS[kind] + + try: + mgr = manager_for(version) + except Exception as exc: + unresolved.append( + f'{label}: could not reach management API ' + f'{version}: {exc}', + ) + continue + + obj = None + try: + if ident is not None: + obj = getattr(mgr, lookup_name)(ident) + else: + # A `pending` record: the creator never returned an id, so the + # only handle on it is the name. Matched over the listing + # exactly as utils._recover_orphan does. + for candidate in listing(mgr): + if getattr(candidate, 'name', None) == name: + obj = candidate + break + except ManagementError as exc: + if exc.errno == 404: + gone.append(label) + continue + unresolved.append(f'{label}: {exc}') + continue + except Exception as exc: + unresolved.append(f'{label}: {exc}') + continue + + if obj is None: + gone.append(label) + elif getattr(obj, 'terminated_at', None) is not None: + gone.append(f'{label} (already terminated)') + else: + found.append((label, obj)) + + return found, gone, unresolved + + +def _run_ledger_sweep(path: str, yes: bool) -> int: + """Report, and with ``yes`` terminate, everything the ledger still lists.""" + leftovers, gone, unresolved = find_ledger_leftovers(path) + + print( + f'Ledger {path}: {len(leftovers)} still live, {len(gone)} already ' + f'gone, {len(unresolved)} unresolved.\n', + ) + + if unresolved: + print( + f'{len(unresolved)} ledger record(s) could not be resolved, so ' + 'they may still be live:', + ) + for label in unresolved: + print(f' ? {label}') + print() + + if not leftovers: + # Non-zero only for the records whose state is unknown: a clean run + # whose sweep already terminated everything must not fail the job. + print('Nothing left behind by this run.') + return 1 if unresolved else 0 + + print(f'{len(leftovers)} deployment(s) left behind by this run:') + for label, _ in leftovers: + print(f' - {label}') + + if not yes: + print('\nDry run; pass --yes to terminate these.') + return 0 + + from singlestoredb.tests import utils + + failed = 0 + for label, obj in leftovers: + try: + utils.terminate(obj) + except Exception as exc: + failed += 1 + print(f'✗ {label}: {exc}') + else: + print(f'✓ terminated {label}') + + return 1 if (failed or unresolved) else 0 + + def main(argv: Optional[List[str]] = None) -> int: parser = argparse.ArgumentParser(description=__doc__.split('\n\n')[1]) parser.add_argument( '--yes', action='store_true', help='actually terminate; without this the run only reports', ) + parser.add_argument( + '--ledger', metavar='PATH', + help='sweep exactly what the run that wrote this JSONL ledger created ' + '(see SINGLESTOREDB_TEST_DEPLOYMENT_LOG). Replaces both the name ' + 'patterns and the age filter, which a ledger makes unnecessary ' + 'and which would in any case spare everything in it for being ' + 'minutes old. This is the mode CI runs as an if: always() step', + ) parser.add_argument( '--older-than', type=float, default=DEFAULT_MIN_AGE_HOURS, metavar='HOURS', @@ -357,6 +652,21 @@ def main(argv: Optional[List[str]] = None) -> int: ) args = parser.parse_args(argv) + # --ledger is a different question entirely -- "what did *this* run make?" + # rather than "what looks stranded?" -- so it does not compose with the + # name and age guards, and saying so beats silently ignoring them. + if args.ledger: + for flag, value in ( + ('--older-than', args.older_than != DEFAULT_MIN_AGE_HOURS), + ('--since', args.since is not None), + ('--any-name', args.any_name), + ('--kind', bool(args.kinds)), + ('--show-unmatched', args.show_unmatched), + ): + if value: + parser.error(f'{flag} does not apply with --ledger') + return _run_ledger_sweep(args.ledger, args.yes) + kinds = args.kinds or list(KINDS) leftovers, spared, unmatched = find_leftovers( diff --git a/singlestoredb/tests/conftest.py b/singlestoredb/tests/conftest.py index 9d426a647..c1e3d8190 100644 --- a/singlestoredb/tests/conftest.py +++ b/singlestoredb/tests/conftest.py @@ -297,6 +297,75 @@ def on_sigterm(signum: int, frame: Any) -> None: logger.debug('Not the main thread; no SIGTERM sweep installed') +#: Key the workers stash their stranded deployment labels under in +#: ``config.workeroutput``. +_STRANDED_KEY = 'singlestoredb_stranded_deployments' + + +def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: + """ + Sweep in an xdist worker, and hand what survived to the controller. + + ``addopts`` is ``-n 2`` (``pyproject.toml``), so under the default the + sweep and its ``STILL LIVE`` banner run in a worker, whose stdout the + controller discards. A leak was therefore silent even when the sweep did + run and fail -- the one case the banner exists to make loud. + + ``config.workeroutput`` is the channel xdist provides for exactly this, and + it only exists in a worker: its absence is the ``-n 0`` case, where + ``pytest_unconfigure`` prints directly to a terminal someone is reading and + nothing here is needed. + + Sweeping here rather than leaving it all to ``pytest_unconfigure`` is what + makes the labels available at all. xdist's own + ``pytest_sessionfinish`` is a hookwrapper that sends ``workeroutput`` after + yielding, so anything written to it from this hook is still included -- + but ``pytest_unconfigure`` runs after the send, so a sweep that waited + until then would have nothing left to report. The sweep is idempotent (a + successful one empties ``_tracked``), so the later call simply finds + nothing to do. + """ + workeroutput = getattr(session.config, 'workeroutput', None) + if workeroutput is None: + return + + _sweep_live_deployments() + + try: + workeroutput[_STRANDED_KEY] = _test_utils().tracked_labels() + except Exception: # pragma: no cover - shutdown path + pass + + +def pytest_testnodedown(node: Any, error: Any) -> None: + """ + Report, on the controller, what a worker could not terminate. + + Runs in the controller process, whose output the user actually sees. The + worker's own banner went to a captured stream; this is the copy that gets + read. + """ + stranded = getattr(node, 'workeroutput', {}).get(_STRANDED_KEY) or [] + if not stranded: + return + + print('\n' + '!' * 70) + print( + f'STILL LIVE on {node.gateway.id} -- these deployments could not be ' + 'terminated and are costing money:', + ) + for label in stranded: + print(f' - {label}') + print( + 'Reap them with: python -m singlestoredb.tests.cleanup_deployments ' + '--yes', + ) + print('!' * 70) + logger.error( + f'{len(stranded)} deployment(s) left live by {node.gateway.id}', + ) + + def pytest_unconfigure(config: pytest.Config) -> None: """ Pytest hook that runs after all tests complete. diff --git a/singlestoredb/tests/test_fusion.py b/singlestoredb/tests/test_fusion.py index 248259dc0..5de2b0272 100644 --- a/singlestoredb/tests/test_fusion.py +++ b/singlestoredb/tests/test_fusion.py @@ -990,10 +990,24 @@ def setUpClass(cls): @classmethod def tearDownClass(cls): - if not cls.dbexisted: - utils.drop_database(cls.dbname) + # Deployments first, and each one guarded. Dropping the database first + # -- as this used to -- meant a database error aborted the teardown + # before a single group was terminated, and an unguarded loop meant a + # failure on the first group abandoned the other two. Three workspace + # groups is the most expensive thing this file leaks. while cls.workspace_groups: - cls.workspace_groups.pop().terminate(force=True) + group = cls.workspace_groups.pop() + try: + group.terminate(force=True) + except Exception: + # Left to utils.cleanup_tracked, which retries and then reports + # it; raising here would replace the test's own failure. + pass + try: + if not cls.dbexisted: + utils.drop_database(cls.dbname) + except Exception: + pass def setUp(self): self.enabled = os.environ.get('SINGLESTOREDB_FUSION_ENABLED') @@ -1489,8 +1503,9 @@ def setUpClass(cls): @classmethod def tearDownClass(cls): - if not cls.dbexisted: - utils.drop_database(cls.dbname) + # Clusters before the database: a drop_database failure used to abort + # the teardown before anything was terminated, leaving three clusters + # to the sweep. while cls.clusters: cluster = cls.clusters.pop() try: @@ -1503,6 +1518,11 @@ def tearDownClass(cls): cluster.terminate(force=True) except Exception: pass + try: + if not cls.dbexisted: + utils.drop_database(cls.dbname) + except Exception: + pass def setUp(self): self.enabled = os.environ.get('SINGLESTOREDB_FUSION_ENABLED') @@ -1833,16 +1853,32 @@ def test_create_cluster_without_project(self): 'this test is for', ) - with self.assertRaises(Exception): - self.cur.execute( - f'create cluster "{name}" in region "{region.region_name}"', - ) + live = [] + try: + with self.assertRaises(Exception): + self.cur.execute( + f'create cluster "{name}" in region ' + f'"{region.region_name}"', + ) + finally: + # One listing, serving both purposes: the assertion that nothing + # was created, and the cleanup for when something was. The test + # only passes if this comes back empty, so the terminate below + # fires exactly when the assertion is about to fail -- which is + # also the only case where a cluster exists. Nothing else would + # remove it: the create goes through Fusion SQL rather than + # ClusterManager.create_cluster, so the tracking wrapper never sees + # it and there is no _tracked entry for the sweep to find. + live = [ + x for x in mgr.clusters + if x.name == name and x.terminated_at is None + ] + for cluster in live: + try: + utils.terminate(cluster) + except Exception: + pass - # Nothing should have been created - live = [ - x for x in mgr.clusters - if x.name == name and x.terminated_at is None - ] assert not live, live def test_create_cluster_named_project(self): diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index 685c352b1..f4c70e2e2 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -8,8 +8,10 @@ only because that is where the bugs were found. """ import datetime +import json import os import pathlib +import shutil import tempfile import unittest from types import SimpleNamespace @@ -952,8 +954,18 @@ def _restore(self): self.utils._in_flight.clear() self.utils._in_flight.extend(self.saved_in_flight) - def _deployment(self, name, terminated_at=None, state='ACTIVE'): - """A stand-in that is not a Mock, so tracking does not skip it.""" + def _deployment( + self, name, terminated_at=None, state='ACTIVE', classname=None, + ): + """ + A stand-in that is not a Mock, so tracking does not skip it. + + ``classname`` renames the class, which is how the ledger decides a + kind (``utils._KIND_BY_CLASS`` is keyed by class name). The default + ``Deployment`` is deliberately *not* a ledger kind, so the tests that + only care about tracking write no ledger records even when one is + configured. + """ class Deployment: def __init__(self): self.name = name @@ -969,6 +981,8 @@ def refresh(self): def terminate(self, force=False): self.terminated_with = force + if classname: + Deployment.__name__ = classname return Deployment() def test_mocked_deployments_are_not_tracked(self): @@ -1074,7 +1088,7 @@ def create_then_fail_waiting(recv, name, **kwargs): raise ManagementError(msg=f'Exceeded waiting time for {name}') wrapped = self.utils._tracking_wrapper( - create_then_fail_waiting, lambda recv: recv.clusters, + create_then_fail_waiting, 'cluster', lambda recv: recv.clusters, ) with self.assertRaises(ManagementError): wrapped(receiver, 'cl-test-shared-0-abc', wait_on_active=True) @@ -1099,7 +1113,7 @@ def interrupted(recv, name, **kwargs): raise KeyboardInterrupt wrapped = self.utils._tracking_wrapper( - interrupted, lambda recv: recv.clusters, + interrupted, 'cluster', lambda recv: recv.clusters, ) with self.assertRaises(KeyboardInterrupt): wrapped(receiver, 'cl-1') @@ -1121,7 +1135,7 @@ def test_a_mocked_receiver_does_not_track_what_it_returns(self): for value in (returned, 'sentinel'): wrapped = self.utils._tracking_wrapper( lambda recv, name, value=value, **kwargs: value, - lambda recv: [], + 'cluster', lambda recv: [], ) self.assertIs(wrapped(mgr, 'my-cluster'), value) @@ -1136,7 +1150,7 @@ def test_a_real_receiver_still_tracks_what_it_returns(self): returned._manager = None wrapped = self.utils._tracking_wrapper( - lambda recv, name, **kwargs: returned, lambda recv: [], + lambda recv, name, **kwargs: returned, 'cluster', lambda recv: [], ) wrapped(mgr, 'cl-1') self.assertEqual(self.utils.tracked_labels(), ["Deployment 'cl-1'"]) @@ -1147,7 +1161,9 @@ def test_a_mocked_receiver_is_not_searched_for_orphans(self): def boom(recv, name, **kwargs): raise ManagementError(msg='boom') - wrapped = self.utils._tracking_wrapper(boom, lambda recv: recv.clusters) + wrapped = self.utils._tracking_wrapper( + boom, 'cluster', lambda recv: recv.clusters, + ) with self.assertRaises(ManagementError): wrapped(MagicMock(), 'cl-1') self.assertEqual(self.utils._tracked, []) @@ -1165,7 +1181,7 @@ def boom(recv, name, **kwargs): def finder(recv): raise AssertionError('recovery called the live API') - wrapped = self.utils._tracking_wrapper(boom, finder) + wrapped = self.utils._tracking_wrapper(boom, 'cluster', finder) with self.assertRaises(ManagementError): wrapped(receiver, 'cl-1') self.assertEqual(self.utils._tracked, []) @@ -1187,7 +1203,7 @@ def create_then_wait(recv, name, **kwargs): raise AssertionError('the process would have been killed here') wrapped = self.utils._tracking_wrapper( - create_then_wait, lambda recv: recv.clusters, + create_then_wait, 'cluster', lambda recv: recv.clusters, ) with self.assertRaises(AssertionError): wrapped(receiver, 'cl-1', wait_on_active=True) @@ -1208,7 +1224,7 @@ def finder(recv): made = self._deployment('cl-1') wrapped = self.utils._tracking_wrapper( - lambda recv, name, **kwargs: made, finder, + lambda recv, name, **kwargs: made, 'cluster', finder, ) self.assertIs(wrapped(receiver, 'cl-1'), made) self.assertEqual(self.utils._in_flight, []) @@ -1219,7 +1235,9 @@ def boom(recv, name, **kwargs): receiver.clusters = [self._deployment('cl-2')] with self.assertRaises(ManagementError): - self.utils._tracking_wrapper(boom, finder)(receiver, 'cl-2') + self.utils._tracking_wrapper( + boom, 'cluster', finder, + )(receiver, 'cl-2') self.assertEqual(self.utils._in_flight, []) # The orphan was recovered once, not once per code path. self.assertEqual(len(self.utils._tracked), 2) @@ -1229,7 +1247,7 @@ def create(recv, name, **kwargs): raise AssertionError(str(self.utils._in_flight)) wrapped = self.utils._tracking_wrapper( - create, lambda recv: recv.clusters, + create, 'cluster', lambda recv: recv.clusters, ) with self.assertRaises(AssertionError) as raised: wrapped(MagicMock(), 'cl-1') @@ -1299,7 +1317,7 @@ def test_every_creation_method_is_wrapped(self): import importlib self.utils.install_deployment_tracking() - for module_name, class_name, method_name, _ in self.utils._CREATORS: + for module_name, class_name, method_name, _, _ in self.utils._CREATORS: klass = getattr(importlib.import_module(module_name), class_name) method = getattr(klass, method_name, None) self.assertIsNotNone( @@ -1316,7 +1334,9 @@ def test_every_creator_takes_name_first_and_has_a_finder(self): import importlib import inspect - for module_name, class_name, method_name, finder in \ + from singlestoredb.tests import cleanup_deployments + + for module_name, class_name, method_name, kind, finder in \ self.utils._CREATORS: klass = getattr(importlib.import_module(module_name), class_name) method = getattr(klass, method_name) @@ -1331,6 +1351,564 @@ def test_every_creator_takes_name_first_and_has_a_finder(self): 'a failed create would not be recoverable', ) self.assertTrue(callable(finder)) + # The ledger's `pending` record carries this kind, and the reaper + # resolves it through cleanup_deployments.LEDGER_KINDS. A kind + # neither side knows would make a cancelled create unreapable, + # which is the whole point of the ledger. + self.assertIn( + kind, set(self.utils._KIND_BY_CLASS.values()), + f'{class_name}.{method_name} has an unknown ledger kind', + ) + self.assertIn(kind, cleanup_deployments.LEDGER_KINDS) + + +class TestDeploymentLedger(TestDeploymentTracking): + """ + The on-disk ledger that makes a killed run's deployments reapable. + + Inherits ``TestDeploymentTracking``'s fixtures for the module globals and + the non-Mock deployment stand-in. It re-runs that class's tests with a + ledger configured, which is worth having: those tests all use the default + ``Deployment`` classname, so they also pin that a ledger being configured + changes nothing about the in-memory behaviour. + """ + + def setUp(self): + super().setUp() + self.dir = tempfile.mkdtemp() + self.addCleanup(shutil.rmtree, self.dir, True) + self.ledger = os.path.join(self.dir, 'deployments.jsonl') + patcher = patch.dict( + os.environ, {self.utils.LEDGER_ENV_VAR: self.ledger}, + ) + patcher.start() + self.addCleanup(patcher.stop) + + def records(self): + """Every record in the ledger, in the order it was written.""" + if not os.path.exists(self.ledger): + return [] + with open(self.ledger) as file: + return [json.loads(x) for x in file if x.strip()] + + def events(self): + return [(x['event'], x.get('kind'), x.get('name')) for x in + self.records()] + + # + # Writing + # + + def test_no_ledger_is_written_without_the_environment_variable(self): + """Opt-in is the whole contract: a local run must behave exactly as it + did before, with no file appearing anywhere.""" + with patch.dict(os.environ, {}, clear=False): + del os.environ[self.utils.LEDGER_ENV_VAR] + self.utils.track(self._deployment('cl-1', classname='Cluster')) + self.assertFalse(os.path.exists(self.ledger)) + + def test_a_mocked_creation_writes_nothing(self): + """The unit tests drive the creators with patched transports. Recording + those would have the reaper chasing ids that never existed, and -- worse + -- exit non-zero on every one it could not resolve.""" + wrapped = self.utils._tracking_wrapper( + lambda recv, name, **kwargs: MagicMock(), + 'cluster', lambda recv: [], + ) + wrapped(MagicMock(), 'cl-1') + self.utils.track(MagicMock()) + self.assertEqual(self.records(), []) + + def test_a_real_creation_writes_pending_then_live(self): + """In that order, and with the pending written before the creator is + even called: the window this closes is the one where the POST has landed + and nothing in the process knows an id yet.""" + made = self._deployment('cl-1', classname='Cluster') + seen = [] + + def create(recv, name, **kwargs): + # What the ledger holds *during* the wait, which is where the + # cancelled job died. + seen.extend(self.events()) + return made + + receiver = SimpleNamespace( + _get=object(), _post=object(), _delete=object(), + ) + wrapped = self.utils._tracking_wrapper( + create, 'cluster', lambda recv: [], + ) + wrapped(receiver, 'cl-1') + + self.assertEqual(seen, [('pending', 'cluster', 'cl-1')]) + self.assertEqual( + self.events(), [ + ('pending', 'cluster', 'cl-1'), + ('live', 'cluster', 'cl-1'), + ], + ) + self.assertEqual(self.records()[1]['id'], 'cl-1') + + def test_the_pending_name_comes_from_the_keyword_too(self): + receiver = SimpleNamespace( + _get=object(), _post=object(), _delete=object(), + ) + self.utils._tracking_wrapper( + lambda recv, name, **kwargs: None, 'workspace_group', + lambda recv: [], + )(receiver, name='wg-1') + self.assertEqual( + self.events(), [('pending', 'workspace_group', 'wg-1')], + ) + + def test_a_create_that_dies_mid_wait_leaves_pending_with_no_gone(self): + """The reported failure, as the ledger sees it. The creator raises and + the orphan is not in the listing yet, so nothing else is ever written -- + and that lone `pending` is what the reaper resolves by name.""" + receiver = SimpleNamespace( + _get=object(), _post=object(), _delete=object(), clusters=[], + ) + + def create_then_fail_waiting(recv, name, **kwargs): + raise ManagementError(msg=f'Exceeded waiting time for {name}') + + wrapped = self.utils._tracking_wrapper( + create_then_fail_waiting, 'cluster', lambda recv: recv.clusters, + ) + with self.assertRaises(ManagementError): + wrapped(receiver, 'a-fusion-cluster-1f2e', wait_on_active=True) + + self.assertEqual( + self.events(), + [('pending', 'cluster', 'a-fusion-cluster-1f2e')], + ) + + def test_a_recovered_orphan_is_recorded_live(self): + """``_recover_orphan`` goes through ``track()``, so the id it digs out + of the listing reaches the ledger and the reaper can use the point + lookup instead of searching by name.""" + orphan = self._deployment('cl-1', classname='Cluster') + receiver = SimpleNamespace( + _get=object(), _post=object(), _delete=object(), + clusters=[orphan], + ) + + def boom(recv, name, **kwargs): + raise ManagementError(msg='boom') + + with self.assertRaises(ManagementError): + self.utils._tracking_wrapper( + boom, 'cluster', lambda recv: recv.clusters, + )(receiver, 'cl-1') + + self.assertEqual( + self.events(), [ + ('pending', 'cluster', 'cl-1'), + ('live', 'cluster', 'cl-1'), + ], + ) + + def test_a_successful_sweep_appends_gone(self): + obj = self._deployment('cl-1', classname='Cluster') + self.utils.track(obj) + self.assertEqual(len(self.utils.cleanup_tracked()), 1) + self.assertEqual( + self.events(), [ + ('live', 'cluster', 'cl-1'), + ('gone', 'cluster', 'cl-1'), + ], + ) + + def test_a_deployment_already_gone_is_recorded_gone(self): + """A test that terminated in its own teardown: the sweep finds it gone + rather than terminating it, and the record still has to be closed or + the reaper spends a lookup on it and reports it unresolved.""" + obj = self._deployment( + 'cl-1', terminated_at='now', classname='Cluster', + ) + self.utils.track(obj) + self.assertEqual(self.utils.cleanup_tracked(), []) + self.assertEqual( + [x['event'] for x in self.records()], ['live', 'gone'], + ) + + def test_a_failed_terminate_writes_no_gone(self): + """The deployment is still live and still billing, so the reaper must + still see it.""" + obj = self._deployment('cl-1', classname='Cluster') + + def boom(force=False): + raise ManagementError(errno=500, msg='boom') + + obj.terminate = boom + self.utils.track(obj) + self.assertEqual(self.utils.cleanup_tracked(), []) + self.assertEqual([x['event'] for x in self.records()], ['live']) + + def test_untrack_records_gone_only_for_something_tracked(self): + obj = self._deployment('cl-1', classname='Cluster') + self.utils.untrack(obj) + self.assertEqual(self.records(), []) + + self.utils.track(obj) + self.utils.untrack(obj) + self.assertEqual([x['event'] for x in self.records()], ['live', 'gone']) + + def test_a_write_failure_is_logged_and_not_raised(self): + """This sits on the creation path of every management test: an + unwritable ledger must cost a warning, not a failed test run.""" + with patch.dict( + os.environ, + {self.utils.LEDGER_ENV_VAR: os.path.join(self.dir, 'no', 'such')}, + ): + with self.assertLogs(self.utils.logger, 'WARNING') as logs: + self.utils.track(self._deployment('cl-1', classname='Cluster')) + self.assertIn('deployment ledger', logs.output[0]) + + # + # Folding + # + + def fold(self, *lines): + from singlestoredb.tests import cleanup_deployments + return cleanup_deployments.fold_ledger(lines) + + def test_folding_keeps_only_what_is_not_gone(self): + kept = self.fold( + json.dumps(dict(event='pending', kind='cluster', name='cl-1')), + json.dumps( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ), + json.dumps( + dict(event='gone', kind='cluster', name='cl-1', id='id-1'), + ), + # Created and never terminated. + json.dumps(dict(event='pending', kind='cluster', name='cl-2')), + json.dumps( + dict(event='live', kind='cluster', name='cl-2', id='id-2'), + ), + # Interrupted before it returned: pending only. + json.dumps(dict(event='pending', kind='cluster', name='cl-3')), + ) + self.assertEqual( + [(x['event'], x.get('id'), x['name']) for x in kept], + [('live', 'id-2', 'cl-2'), ('pending', None, 'cl-3')], + ) + + def test_a_live_record_retires_its_pending(self): + """Otherwise the reaper resolves the same cluster twice -- once by id + and once by name -- and reports two.""" + kept = self.fold( + json.dumps(dict(event='pending', kind='cluster', name='cl-1')), + json.dumps( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ), + ) + self.assertEqual(len(kept), 1) + self.assertEqual(kept[0]['id'], 'id-1') + + def test_gone_cancels_a_pending_that_never_went_live(self): + kept = self.fold( + json.dumps(dict(event='pending', kind='cluster', name='cl-1')), + json.dumps(dict(event='gone', kind='cluster', name='cl-1')), + ) + self.assertEqual(kept, []) + + def test_the_same_name_in_two_kinds_is_two_deployments(self): + """`cl-test-abc` as a cluster and as a workspace are different things, + and a `gone` for one must not clear the other.""" + kept = self.fold( + json.dumps(dict(event='pending', kind='cluster', name='x')), + json.dumps(dict(event='pending', kind='workspace', name='x')), + json.dumps(dict(event='gone', kind='cluster', name='x')), + ) + self.assertEqual([x['kind'] for x in kept], ['workspace']) + + def test_a_malformed_line_is_skipped_rather_than_fatal(self): + """A truncated last line -- a process killed between the write and the + fsync -- must not cost the reaper every other record.""" + kept = self.fold( + json.dumps(dict(event='pending', kind='cluster', name='cl-1')), + '{"event": "pending", "kin', + '', + '[]', + ) + self.assertEqual([x['name'] for x in kept], ['cl-1']) + + def test_reading_reverses_into_newest_first(self): + """A workspace has to be terminated before the group that holds it, the + same ordering ``cleanup_tracked`` uses.""" + from singlestoredb.tests import cleanup_deployments + with open(self.ledger, 'w') as file: + for kind, name in ( + ('workspace_group', 'wg-1'), ('workspace', 'ws-1'), + ): + file.write( + json.dumps(dict(event='pending', kind=kind, name=name)) + + '\n', + ) + self.assertEqual( + [x['name'] for x in cleanup_deployments.read_ledger(self.ledger)], + ['ws-1', 'wg-1'], + ) + + # + # Resolving, with the management API stubbed out + # + + def stub_managers(self, **attrs): + """Patch the reaper's manager lookup with a namespace.""" + from singlestoredb.tests import cleanup_deployments + mgr = SimpleNamespace(**attrs) + patcher = patch.object( + cleanup_deployments, '_manager', lambda version: mgr, + ) + patcher.start() + self.addCleanup(patcher.stop) + return cleanup_deployments, mgr + + def write_ledger(self, *records): + with open(self.ledger, 'w') as file: + for record in records: + file.write(json.dumps(record) + '\n') + + def test_an_id_that_404s_is_treated_as_already_gone(self): + """The common case by far: the ledger records every creation, and a run + that ended normally terminated all of them. A clean sweep must exit 0 + and terminate nothing.""" + def get_cluster(ident): + raise ManagementError(errno=404, msg='not found') + + mod, _ = self.stub_managers(get_cluster=get_cluster) + self.write_ledger( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ) + found, gone, unresolved = mod.find_ledger_leftovers(self.ledger) + self.assertEqual(found, []) + self.assertEqual(len(gone), 1) + self.assertEqual(unresolved, []) + self.assertEqual(mod.main(['--ledger', self.ledger, '--yes']), 0) + + def test_a_live_id_is_resolved_and_terminated(self): + obj = self._deployment('cl-1', classname='Cluster') + mod, _ = self.stub_managers(get_cluster=lambda ident: obj) + self.write_ledger( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ) + self.assertEqual(mod.main(['--ledger', self.ledger, '--yes']), 0) + self.assertTrue(obj.terminated_with) + + def test_a_dry_run_terminates_nothing(self): + obj = self._deployment('cl-1', classname='Cluster') + mod, _ = self.stub_managers(get_cluster=lambda ident: obj) + self.write_ledger( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ) + self.assertEqual(mod.main(['--ledger', self.ledger]), 0) + self.assertIsNone(obj.terminated_with) + + def test_a_pending_record_is_resolved_by_name_over_the_listing(self): + """No id was ever returned, so the listing is the only handle -- the + same match ``_recover_orphan`` makes, and the case a cancelled + ``wait_on_active`` leaves.""" + wanted = self._deployment('a-fusion-cluster-1f2e', classname='Cluster') + other = self._deployment('someone-elses', classname='Cluster') + mod, _ = self.stub_managers(clusters=[other, wanted]) + self.write_ledger( + dict(event='pending', kind='cluster', name='a-fusion-cluster-1f2e'), + ) + self.assertEqual(mod.main(['--ledger', self.ledger, '--yes']), 0) + self.assertTrue(wanted.terminated_with) + self.assertIsNone(other.terminated_with) + + def test_a_pending_name_absent_from_the_listing_is_gone(self): + """The POST never landed, so there is nothing to reap and nothing to + complain about.""" + mod, _ = self.stub_managers(clusters=[]) + self.write_ledger(dict(event='pending', kind='cluster', name='cl-1')) + found, gone, unresolved = mod.find_ledger_leftovers(self.ledger) + self.assertEqual((found, len(gone), unresolved), ([], 1, [])) + + def test_an_already_terminated_deployment_is_not_terminated_again(self): + obj = self._deployment( + 'cl-1', terminated_at='now', classname='Cluster', + ) + mod, _ = self.stub_managers(get_cluster=lambda ident: obj) + self.write_ledger( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ) + self.assertEqual(mod.main(['--ledger', self.ledger, '--yes']), 0) + self.assertIsNone(obj.terminated_with) + + def test_a_lookup_failure_that_is_not_a_404_exits_non_zero(self): + """"Could not tell" and "not there" must not read the same when the + difference is a cluster billing.""" + def get_cluster(ident): + raise ManagementError(errno=500, msg='gateway sulked') + + mod, _ = self.stub_managers(get_cluster=get_cluster) + self.write_ledger( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ) + found, gone, unresolved = mod.find_ledger_leftovers(self.ledger) + self.assertEqual((found, gone), ([], [])) + self.assertEqual(len(unresolved), 1) + self.assertEqual(mod.main(['--ledger', self.ledger, '--yes']), 1) + + def test_an_unknown_kind_is_reported_rather_than_skipped(self): + mod, _ = self.stub_managers() + self.write_ledger(dict(event='live', kind='mystery', name='x', id='1')) + _, _, unresolved = mod.find_ledger_leftovers(self.ledger) + self.assertEqual(len(unresolved), 1) + self.assertIn('unknown kind', unresolved[0]) + + def test_a_missing_ledger_is_not_an_error(self): + """The variable is set for a whole job, including steps whose tests + create nothing. Failing there would turn those runs red.""" + from singlestoredb.tests import cleanup_deployments + missing = os.path.join(self.dir, 'never-written.jsonl') + self.assertEqual(cleanup_deployments.read_ledger(missing), []) + self.assertEqual( + cleanup_deployments.main(['--ledger', missing, '--yes']), 0, + ) + + def test_ledger_mode_refuses_the_guards_it_replaces(self): + """Silently ignoring --older-than would read as a safety guard that is + not there.""" + from singlestoredb.tests import cleanup_deployments + for extra in ( + ['--older-than', '0'], ['--any-name'], + ['--kind', 'cluster'], ['--show-unmatched'], + ): + with self.assertRaises(SystemExit): + cleanup_deployments.main( + ['--ledger', self.ledger] + extra, + ) + + +class TestTerminateRetry(unittest.TestCase): + """ + ``utils.terminate()``'s bounded retry for a deployment the API will not + delete yet. + + A deployment killed mid-provision is PENDING/TRANSITIONING and the DELETE + comes back 400 or 409. Nothing retried that: Manager.RETRY_STATUSES is + {429, 500, 502, 503, 504}, so the per-class sweep warned, the session-end + sweep tried once more and the cluster stayed up. + """ + + def setUp(self): + from singlestoredb.tests import utils + self.utils = utils + self.slept = [] + # A fake clock, not just a stubbed sleep: the retry budget is measured + # with time.monotonic(), so a sleep that does not advance it makes the + # deadline unreachable and the loop only ends when the stub runs out of + # refusals. That is the opposite of what the budget test asserts. + self.now = 0.0 + + def sleep(seconds): + self.slept.append(seconds) + self.now += seconds + + for name, value in ( + ('sleep', sleep), ('monotonic', lambda: self.now), + ): + patcher = patch(f'time.{name}', value) + patcher.start() + self.addCleanup(patcher.stop) + + def _refuser(self, *errnos): + """A deployment whose terminate raises these in turn, then succeeds.""" + class Deployment: + attempts = 0 + terminated_with = None + + def terminate(inner, force=False): + inner.attempts += 1 + if inner.attempts <= len(errnos): + raise ManagementError( + errno=errnos[inner.attempts - 1], + msg='still provisioning', + ) + inner.terminated_with = force + + return Deployment() + + def test_a_400_is_retried_until_it_succeeds(self): + obj = self._refuser(400, 409) + self.utils.terminate(obj) + self.assertEqual(obj.attempts, 3) + self.assertTrue(obj.terminated_with) + self.assertEqual(self.slept, [15.0, 15.0]) + + def test_a_404_is_not_retried(self): + """It is already gone; retrying would burn the whole budget waiting for + something that is not coming back.""" + obj = self._refuser(404) + with self.assertRaises(ManagementError): + self.utils.terminate(obj) + self.assertEqual(obj.attempts, 1) + self.assertEqual(self.slept, []) + + def test_a_5xx_is_not_retried_here(self): + """The transport already retried it; another round trip from this layer + is not what fixes it.""" + obj = self._refuser(503) + with self.assertRaises(ManagementError): + self.utils.terminate(obj) + self.assertEqual(obj.attempts, 1) + + def test_the_budget_is_bounded_and_the_error_is_re_raised(self): + """Raising is what keeps the deployment in ``_tracked``, so the + end-of-session sweep gets another go at it.""" + obj = self._refuser(*([409] * 100)) + with self.assertRaises(ManagementError): + self.utils.terminate(obj, timeout=45.0, interval=15.0) + self.assertEqual(obj.attempts, 3) + self.assertEqual(self.slept, [15.0, 15.0]) + + def test_a_starter_kind_is_terminated_without_force(self): + """StarterWorkspace.terminate / StarterCluster.terminate take no + arguments at all.""" + class Starter: + called = False + + def terminate(inner): + inner.called = True + + obj = Starter() + self.utils.terminate(obj) + self.assertTrue(obj.called) + + def test_force_is_passed_when_the_signature_accepts_it(self): + """``force`` is what makes a workspace group with live workspaces in it + go away, so this is not cosmetic.""" + seen = [] + + class Group: + def terminate(inner, force=False): + seen.append(force) + + self.utils.terminate(Group()) + self.assertEqual(seen, [True]) + + def test_a_type_error_from_inside_terminate_is_not_a_second_delete(self): + """The signature is inspected rather than discovered by catching + TypeError from the call. The old ``except TypeError`` also caught one + raised *inside* a terminate that did accept force, and retried without + it -- two DELETEs, the second unforced, which is exactly the shape that + leaves a workspace group behind.""" + calls = [] + + class Group: + def terminate(inner, force=False): + calls.append(force) + raise TypeError('something inside went wrong') + + with self.assertRaises(TypeError): + self.utils.terminate(Group()) + self.assertEqual(calls, [True]) class TestSharedClusterPool(unittest.TestCase): diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index 3328969a5..b0633f010 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -85,7 +85,14 @@ def setUpClass(cls): wait_on_active=True, ) except Exception: - cls.workspace_group.terminate(force=True) + # Guarded: an unguarded terminate here replaces the create failure + # with whatever the DELETE raised, which both hides the real error + # and leaves the group live with nothing having reported why. + # utils.cleanup_tracked retries it and says so. + try: + cls.workspace_group.terminate(force=True) + except Exception: + pass raise @classmethod @@ -937,18 +944,30 @@ def test_get_secret(self): except s2.ManagementError: pass - self.manager._post( + created = self.manager._post( 'secrets', json=dict( name='secret_name', value='secret_value', ), - ) - - secret = self.manager.organizations.current.get_secret('secret_name') + ).json() + + # The ID comes from the create response rather than from the lookup + # under test: binding it inside the try would leave the cleanup raising + # UnboundLocalError over whatever the lookup actually failed with. + # Without this the secret outlived every run -- it was only ever + # removed opportunistically by the sweep at the top of the *next* one. + # test_management_v2.py's twin already does it this way. + secret_id = created['secret']['secretID'] + try: + secret = self.manager.organizations.current.get_secret( + 'secret_name', + ) - assert secret.name == 'secret_name' - assert secret.value == 'secret_value' + assert secret.name == 'secret_name' + assert secret.value == 'secret_value' + finally: + self.manager._delete(f'secrets/{secret_id}') @pytest.mark.management @@ -982,7 +1001,14 @@ def setUpClass(cls): wait_on_active=True, ) except Exception: - cls.workspace_group.terminate(force=True) + # Guarded: an unguarded terminate here replaces the create failure + # with whatever the DELETE raised, which both hides the real error + # and leaves the group live with nothing having reported why. + # utils.cleanup_tracked retries it and says so. + try: + cls.workspace_group.terminate(force=True) + except Exception: + pass raise @classmethod diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index 94d754c47..1aa283d1f 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -2,6 +2,7 @@ # type: ignore """Utilities for testing.""" import glob +import json import logging import os import random @@ -336,6 +337,160 @@ def set_owner(owner: str) -> None: _owner = owner +# +# Durable deployment ledger +# +# Everything above this point is in-memory only, and that is the one leak the +# sweeps cannot cover. A cancelled CI job is the proven case: GH Actions run +# 35631802648, job ``test-coverage``, was cancelled 19 minutes into +# ``TestClusterFusion.setUpClass``'s ``create_cluster(wait_on_active=True, +# wait_timeout=1200)``. The log ends at ``##[error]The operation was +# canceled.`` with no pytest summary, no "Terminated deployments left behind +# by tests:" and no ``STILL LIVE`` banner -- the process never got to sweep, +# and GitHub's cancellation grace period is nowhere near long enough for +# pytest to unwind three nested class fixtures, list to recover three +# in-flight creations and issue three DELETEs. After the SIGKILL that follows, +# ``_tracked`` and ``_in_flight`` are gone with the process and *nothing on +# disk* records that three clusters were created. +# +# So every creation is also appended to a JSONL file, flushed and fsync'd per +# line, which ``cleanup_deployments.py --ledger`` reads afterwards from a +# separate process -- an ``if: always()`` CI step that still runs on +# cancellation. The ledger is the record; the in-memory sweeps stay exactly as +# they were and remain the fast path. +# +# Opt-in, via SINGLESTOREDB_TEST_DEPLOYMENT_LOG. With the variable unset +# nothing is written and behaviour is byte-for-byte what it was: a local run +# has a human watching it and does not need a file to reap from. +# + +#: Ledger event kinds, in the order a deployment normally produces them: +#: +#: * ``pending`` -- the creator is about to be called. Written *before* the +#: POST, from the name argument, because the whole point is the window where +#: the server has a billable deployment and this process has no id for it. +#: * ``live`` -- the creation returned (or an orphan was recovered), so there +#: is an id. +#: * ``gone`` -- it has been terminated. +#: +#: The reaper folds the file: anything whose last event is not ``gone`` is +#: still live. A ``pending`` with no matching ``live`` is the cancelled-mid- +#: wait case, and it is resolved by name rather than by id. + +#: Environment variable naming the ledger file. Read per write rather than +#: cached at import so a test can point it at a tmp_path with +#: ``mock.patch.dict(os.environ, ...)``. +LEDGER_ENV_VAR = 'SINGLESTOREDB_TEST_DEPLOYMENT_LOG' + +#: Deployment kind for each created object's class. The ledger records a kind +#: so the reaper knows which manager and which point lookup to resolve a +#: record against, instead of guessing from the name -- ``cl-test-abc`` and +#: ``ws-test-abc`` are only distinguishable by convention, and a ``--ledger`` +#: run deliberately does not consult :data:`cleanup_deployments.PATTERNS`. +#: +#: Keyed by class name rather than by the class itself to avoid importing v1 +#: and v2 management just to write a log line. +_KIND_BY_CLASS = { + 'WorkspaceGroup': 'workspace_group', + 'Workspace': 'workspace', + 'StarterWorkspace': 'starter_workspace', + 'Cluster': 'cluster', + 'StarterCluster': 'starter_cluster', +} + + +def ledger_path() -> Optional[str]: + """Path of the deployment ledger, or None if none was configured.""" + return os.environ.get(LEDGER_ENV_VAR) or None + + +def _ledger_write(**record: Any) -> None: + """ + Append one record to the deployment ledger. + + Opened, written and closed per record, with ``flush()`` and ``os.fsync()`` + before the handle goes: surviving SIGKILL is the entire purpose, and a + line still sitting in a buffer when the process dies records nothing. The + cost is one open per creation, against a creation that takes minutes. + + ``O_APPEND`` plus one ``write()`` per line is what makes this safe for the + parallel default (``-n 2``): the xdist workers are separate processes + sharing the file, and a single write of well under PIPE_BUF cannot + interleave with another's on Linux. No locking, therefore, and no partial + lines for the reaper to choke on. + + Never raises. This sits on the creation path of every management test, so + a full disk or an unwritable path must cost a warning, not a test failure. + """ + path = ledger_path() + if not path: + return + try: + # default=str so an unexpected value (a datetime, an enum) degrades to + # its repr instead of raising and losing the whole record. + line = json.dumps(record, default=str, sort_keys=True) + '\n' + with open(path, 'a', encoding='utf-8') as file: + file.write(line) + file.flush() + os.fsync(file.fileno()) + except Exception as exc: + logger.warning( + f'Could not append {record!r} to the deployment ledger at ' + f'{path!r}; a deployment this run creates may not be reaped: ' + f'{exc}', + ) + + +def _ledger_kind(obj: Any) -> Optional[str]: + """Ledger kind for a created object, or None if it is not a deployment.""" + return _KIND_BY_CLASS.get(type(obj).__name__) + + +def ledger_pending(kind: str, args: Tuple[Any, ...], kwargs: Any) -> None: + """ + Record that a deployment of this kind is about to be created. + + The name is taken the same way :func:`_recover_orphan` takes it -- keyword + first, else the first positional -- because it is the first parameter of + every creator, which ``test_management_utils.py`` pins. A record with no + usable name is skipped: there would be nothing for the reaper to resolve. + """ + name = kwargs.get('name') or (args[0] if args else None) + if not isinstance(name, str): + return + _ledger_write(event='pending', kind=kind, name=name) + + +def ledger_live(obj: Any) -> None: + """Record that a created deployment exists, now that it has an id.""" + kind = _ledger_kind(obj) + if kind is None: + return + _ledger_write( + event='live', kind=kind, + id=getattr(obj, 'id', None), + name=getattr(obj, 'name', None), + ) + + +def ledger_gone(obj: Any) -> None: + """ + Record that a deployment has been terminated. + + Carries the name as well as the id so it also cancels a ``pending`` + record: an orphan recovered by name and then swept in-process would + otherwise still be listed as live by the reaper. + """ + kind = _ledger_kind(obj) + if kind is None: + return + _ledger_write( + event='gone', kind=kind, + id=getattr(obj, 'id', None), + name=getattr(obj, 'name', None), + ) + + def _is_mocked(obj: Any) -> bool: """ Did this object come out of a mocked manager? @@ -378,6 +533,10 @@ def track(obj: Any, label: str = '') -> Any: ), obj, )) + # Here rather than in the wrapper, so an orphan that `_recover_orphan` + # digs out of a listing gets an id into the ledger too -- that path + # reaches the server only through this function. + ledger_live(obj) return obj @@ -427,24 +586,109 @@ def _recover_orphan( def untrack(obj: Any) -> None: """Forget a deployment that has been terminated.""" + found = False for i, entry in reversed(list(enumerate(_tracked))): if entry[2] is obj: _tracked.pop(i) - - -def terminate(obj: Any) -> None: + found = True + # Only for something that was actually tracked: untracking an object that + # was never registered -- a mocked one, or one already swept -- says + # nothing about whether a real deployment is gone, and a spurious ``gone`` + # would hide a live cluster from the reaper. + if found: + ledger_gone(obj) + + +#: How long :func:`terminate` keeps retrying a deployment the API will not +#: delete yet, and how long it waits between attempts. Three minutes at 15s +#: spacing: the case being covered is a deployment killed mid-provision, which +#: has to finish coming up before it can be torn down, and an S-00 cluster +#: reaching ACTIVE is ~460s at worst. Waiting the full provision out here would +#: stall the sweep between every test class, so this buys the common case -- +#: a deployment most of the way up -- and leaves the rest to the end-of-session +#: sweep and then to ``cleanup_deployments.py``. +TERMINATE_RETRY_TIMEOUT = 180.0 +TERMINATE_RETRY_INTERVAL = 15.0 + + +def _terminate_once(obj: Any) -> None: """ - Terminate a deployment, whatever kind it is. + Issue one terminate, whatever this kind's signature looks like. ``force=True`` is what makes a workspace group with live workspaces in it - go away; the starter variants take no arguments at all. + go away; the starter variants (``StarterWorkspace.terminate``, + ``StarterCluster.terminate``) take no arguments at all. + + The signature is inspected rather than discovered by catching ``TypeError`` + from the call, as this used to do. That ``except TypeError`` also caught a + ``TypeError`` raised from *inside* a terminate that did accept ``force``, + and then retried without it -- two DELETEs for one deployment, the second + of them not forced, which is the one shape that leaves a workspace group + behind. """ + import inspect + try: + params = inspect.signature(obj.terminate).parameters + except (TypeError, ValueError): # pragma: no cover - unintrospectable + # A builtin or a C-level callable. Fall back to the old behaviour. + params = {} + + if 'force' in params: obj.terminate(force=True) - except TypeError: + else: obj.terminate() +def terminate( + obj: Any, + timeout: float = TERMINATE_RETRY_TIMEOUT, + interval: float = TERMINATE_RETRY_INTERVAL, +) -> None: + """ + Terminate a deployment, whatever kind it is, retrying a 4xx refusal. + + A deployment killed mid-provision is ``PENDING``/``TRANSITIONING``, and the + API refuses to delete it in that state with a 400 or a 409. Nothing retries + that: ``Manager.RETRY_STATUSES`` is ``{429, 500, 502, 503, 504}`` + (``management/manager.py:49``), urllib3 only retries what is in that list, + and there is no wait-until-deletable helper anywhere in the SDK. So the + per-class sweep logged a warning, the session-end sweep tried exactly once + more -- usually still too early -- and the deployment was left running. + + Hence the bounded retry here. Only 4xx other than 404 is retried: + + * 404 means it is already gone, so retrying would burn the whole budget + waiting for something that will never come back. Re-raised, as before, + which is also what ``_is_gone()`` upstream normally prevents. + * 5xx and 429 are already retried inside the transport, so seeing one here + means the transport gave up; another round trip from this layer is not + what fixes it. + + Raises the last error if the budget runs out, so ``cleanup_tracked()`` + keeps the deployment tracked and the session-end sweep gets another go. + """ + import time + + deadline = time.monotonic() + timeout + while True: + try: + _terminate_once(obj) + return + except ManagementError as exc: + errno = exc.errno + if errno is None or errno == 404 or not 400 <= errno < 500: + raise + # No budget left for another attempt *plus* the wait before it. + if time.monotonic() + interval >= deadline: + raise + logger.info( + f'{obj!r} is not deletable yet ({exc}); retrying the ' + f'terminate in {interval:g}s', + ) + time.sleep(interval) + + def _creator_is_mocked(target: Any) -> bool: """ Is this creation call going through a mocked manager? @@ -473,9 +717,15 @@ def _creator_is_mocked(target: Any) -> bool: ) -#: (module, class, method, finder) tuples for the calls that bring a billable -#: deployment into existence. Wrapping them is what makes tracking automatic, -#: so a new test cannot leak a cluster by forgetting to register it. +#: (module, class, method, kind, finder) tuples for the calls that bring a +#: billable deployment into existence. Wrapping them is what makes tracking +#: automatic, so a new test cannot leak a cluster by forgetting to register it. +#: +#: ``kind`` is the ledger kind the call produces, and must be a value of +#: :data:`_KIND_BY_CLASS`: it is what lets the ``pending`` record -- written +#: before the POST, when nothing has an id yet -- say which manager the reaper +#: should search. It is stated here rather than derived from ``method_name`` +#: because ``create_workspace`` appears twice, on two different receivers. #: #: ``finder`` takes the receiver -- the manager, or the group for #: ``WorkspaceGroup.create_workspace`` -- and returns the collection to search @@ -484,34 +734,34 @@ def _creator_is_mocked(target: Any) -> bool: _CREATORS = [ ( 'singlestoredb.management.v1.workspace', 'WorkspaceManager', - 'create_workspace_group', + 'create_workspace_group', 'workspace_group', lambda recv: recv.workspace_groups, ), ( 'singlestoredb.management.v1.workspace', 'WorkspaceManager', - 'create_workspace', + 'create_workspace', 'workspace', # WorkspaceManager has no `workspaces` of its own, so the search goes # group by group. Only ever walked on the failure path. lambda recv: [w for g in recv.workspace_groups for w in g.workspaces], ), ( 'singlestoredb.management.v1.workspace', 'WorkspaceManager', - 'create_starter_workspace', + 'create_starter_workspace', 'starter_workspace', lambda recv: recv.starter_workspaces, ), ( 'singlestoredb.management.v1.workspace', 'WorkspaceGroup', - 'create_workspace', + 'create_workspace', 'workspace', lambda recv: recv.workspaces, ), ( 'singlestoredb.management.v2.cluster', 'ClusterManager', - 'create_cluster', + 'create_cluster', 'cluster', lambda recv: recv.clusters, ), ( 'singlestoredb.management.v2.cluster', 'ClusterManager', - 'create_starter_cluster', + 'create_starter_cluster', 'starter_cluster', lambda recv: recv.starter_clusters, ), ] @@ -519,7 +769,7 @@ def _creator_is_mocked(target: Any) -> bool: _tracking_installed = False -def _tracking_wrapper(func: Any, finder: Any) -> Any: +def _tracking_wrapper(func: Any, kind: str, finder: Any) -> Any: """ Wrap a creation method so its result -- or its orphan -- gets tracked. @@ -546,6 +796,14 @@ def _tracking_wrapper(func: Any, finder: Any) -> Any: unit test's stubbed ``get_cluster`` returns. Nothing a mocked creator returns names a deployment that exists, so the receiver's verdict is the authoritative one and it is the one used here. + + The ``pending`` ledger record is written here, and deliberately *before* + ``func`` is called rather than after: from the moment the creator POSTs + there is a billable deployment, and everything that could record it -- + ``track()`` on return, ``_recover_orphan()`` in the ``except``, + ``recover_in_flight()`` from a signal handler -- runs after the wait that + a cancelled CI job never survives. A ``pending`` line on disk is the only + thing that outlives a SIGKILL there. """ import functools @@ -555,6 +813,7 @@ def wrapper(receiver: Any, *args: Any, **kwargs: Any) -> Any: entry = (receiver, finder, args, kwargs) if not mocked: _in_flight.append(entry) + ledger_pending(kind, args, kwargs) try: out = func(receiver, *args, **kwargs) return out if mocked else track(out) @@ -628,12 +887,12 @@ def install_deployment_tracking() -> None: import importlib - for module_name, class_name, method_name, finder in _CREATORS: + for module_name, class_name, method_name, kind, finder in _CREATORS: try: klass = getattr(importlib.import_module(module_name), class_name) setattr( klass, method_name, - _tracking_wrapper(getattr(klass, method_name), finder), + _tracking_wrapper(getattr(klass, method_name), kind, finder), ) except AttributeError as exc: # A renamed method must not silently stop being tracked. @@ -708,6 +967,11 @@ def cleanup_tracked(owner: Optional[str] = None) -> List[str]: _, label, obj = entry if _is_gone(obj): _tracked.remove(entry) + # The server says it is gone, which is exactly what the ledger's + # ``gone`` means -- a test that terminated in its own teardown gets + # its record closed here rather than leaving the reaper to look up + # an id that 404s. + ledger_gone(obj) continue try: terminate(obj) @@ -719,6 +983,7 @@ def cleanup_tracked(owner: Optional[str] = None) -> List[str]: logger.warning(f'Could not terminate {label}: {exc}') else: _tracked.remove(entry) + ledger_gone(obj) removed.append(label) return removed From 44c8512d754b66f8232fab1bab1e2bfadc613ce2 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 21 Sep 2026 14:49:38 -0400 Subject: [PATCH 15/30] Give the reaper a budget that outlasts a provision cleanup_deployments.py was calling utils.terminate() with its default retry budget, which is 180s deliberately: that sweep runs between test classes and must not stall the suite, and its own comment says the remainder is left to this script. That handoff was empty while this script took the same default. A job cancelled early in create_cluster(wait_on_active=True) leaves a cluster that DELETE refuses with a 400/409 until it is up, and an S-00 cluster reaching ACTIVE is ~460s at worst, so the step exhausted the budget, exited, and the cluster kept billing. Nothing runs after this script, so it now has its own TERMINATE_TIMEOUT = 600.0, passed at both sweep call sites. The age/pattern sweep is not what was flagged, but --since and --older-than 0 can hand it a deployment that is still provisioning, and it has the same "nothing follows me" property. The retry interval is left at 15s; 40 attempts across 600s is fine. utils.TERMINATE_RETRY_TIMEOUT's comment no longer claims a handoff that was not happening, and a test pins the reaper's budget above the per-class one so the two cannot drift back together. Co-Authored-By: Claude Opus 5 --- singlestoredb/tests/cleanup_deployments.py | 17 ++++++++-- singlestoredb/tests/test_management_utils.py | 34 ++++++++++++++++++++ singlestoredb/tests/utils.py | 4 ++- 3 files changed, 52 insertions(+), 3 deletions(-) diff --git a/singlestoredb/tests/cleanup_deployments.py b/singlestoredb/tests/cleanup_deployments.py index d50fea1bd..71601b3f5 100644 --- a/singlestoredb/tests/cleanup_deployments.py +++ b/singlestoredb/tests/cleanup_deployments.py @@ -89,6 +89,16 @@ #: anything younger could belong to a run in progress. DEFAULT_MIN_AGE_HOURS = 6.0 +#: How long to keep retrying a deployment the API will not delete yet. This is +#: the end of the line -- nothing runs after this tool -- so it does not borrow +#: ``utils.TERMINATE_RETRY_TIMEOUT``, which is deliberately short so the sweep +#: between test classes cannot stall the suite. A cancelled job's cluster may +#: only just have been POSTed, ``DELETE`` is refused until it is up, and an +#: S-00 cluster reaching ACTIVE is ~460s at worst, so anything shorter than a +#: full provision leaves it billing. The only cost of waiting is this CI step's +#: wall clock. +TERMINATE_TIMEOUT = 600.0 + #: Names the suite generates. Anchored, because these run against a real #: organization: a pattern that matched a name someone chose by hand would #: terminate a deployment that is not ours. @@ -581,7 +591,7 @@ def _run_ledger_sweep(path: str, yes: bool) -> int: failed = 0 for label, obj in leftovers: try: - utils.terminate(obj) + utils.terminate(obj, timeout=TERMINATE_TIMEOUT) except Exception as exc: failed += 1 print(f'✗ {label}: {exc}') @@ -725,7 +735,10 @@ def main(argv: Optional[List[str]] = None) -> int: failed = 0 for label, obj in leftovers: try: - utils.terminate(obj) + # Same budget as the ledger sweep: --since or --older-than 0 can + # select a deployment that is still provisioning, and nothing runs + # after this either. + utils.terminate(obj, timeout=TERMINATE_TIMEOUT) except Exception as exc: failed += 1 print(f'✗ {label}: {exc}') diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index f4c70e2e2..99ee9594e 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -1772,6 +1772,32 @@ def test_a_missing_ledger_is_not_an_error(self): cleanup_deployments.main(['--ledger', missing, '--yes']), 0, ) + def test_the_sweep_waits_out_a_provision_rather_than_the_class_budget(self): + """The whole point of the ledger is a job cancelled inside + ``wait_on_active``, whose cluster is minutes from deletable. Borrowing + ``utils.TERMINATE_RETRY_TIMEOUT`` -- short so the per-class sweep cannot + stall the suite -- would exhaust the budget and leave it billing, and + nothing runs after this to try again.""" + from singlestoredb.tests import utils + obj = self._deployment('cl-1', classname='Cluster') + mod, _ = self.stub_managers(get_cluster=lambda ident: obj) + self.write_ledger( + dict(event='live', kind='cluster', name='cl-1', id='id-1'), + ) + + calls = [] + patcher = patch.object( + utils, 'terminate', + lambda obj, **kwargs: calls.append(kwargs), + ) + patcher.start() + self.addCleanup(patcher.stop) + + self.assertEqual(mod.main(['--ledger', self.ledger, '--yes']), 0) + self.assertEqual(len(calls), 1) + self.assertEqual(calls[0]['timeout'], mod.TERMINATE_TIMEOUT) + self.assertGreater(calls[0]['timeout'], utils.TERMINATE_RETRY_TIMEOUT) + def test_ledger_mode_refuses_the_guards_it_replaces(self): """Silently ignoring --older-than would read as a safety guard that is not there.""" @@ -2257,6 +2283,14 @@ def test_the_default_spares_anything_a_run_could_still_own(self): # Not zero: a default that swept every match would make running this # during a test run destructive. self.assertGreaterEqual(self.mod.DEFAULT_MIN_AGE_HOURS, 1) + # Nothing runs after this tool, so its terminate budget has to cover a + # full provision (~460s for an S-00 cluster reaching ACTIVE) rather than + # the per-class budget, which is short on purpose. + from singlestoredb.tests import utils + self.assertGreater( + self.mod.TERMINATE_TIMEOUT, utils.TERMINATE_RETRY_TIMEOUT, + ) + self.assertGreaterEqual(self.mod.TERMINATE_TIMEOUT, 460) names, spared = self._find([ self._cluster('cl-test-mid-run', hours=1), ]) diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index 1aa283d1f..4ea29df3f 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -606,7 +606,9 @@ def untrack(obj: Any) -> None: #: reaching ACTIVE is ~460s at worst. Waiting the full provision out here would #: stall the sweep between every test class, so this buys the common case -- #: a deployment most of the way up -- and leaves the rest to the end-of-session -#: sweep and then to ``cleanup_deployments.py``. +#: sweep and then to ``cleanup_deployments.py``, which is the end of the line +#: and waits out a full provision with a longer budget of its own +#: (``cleanup_deployments.TERMINATE_TIMEOUT``). TERMINATE_RETRY_TIMEOUT = 180.0 TERMINATE_RETRY_INTERVAL = 15.0 From 0f2d45d9abadfdae1d08d6d29cbb56f269e9be2b Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 21 Sep 2026 15:24:56 -0400 Subject: [PATCH 16/30] Let a new push cancel the run it supersedes Nothing cancelled an in-flight Code checks run when the branch moved, so two runs for the same PR provisioned at once. Each management step deploys around six clusters and holds them for the length of the suite, so the second run doubled what the org was carrying and bought no coverage: an observed run had fourteen clusters live where one run accounts for seven. Safe to cancel mid-suite because the cleanup step is `if: always()`, which does survive cancellation -- the run cancelled while writing this reaped all five of its live clusters from the ledger. Not applied to coverage.yml, which is schedule/dispatch only and whose two jobs are meant to overlap with separate ledgers. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 7861251a7..7572c1759 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -8,6 +8,19 @@ on: workflow_dispatch: +# A push to a branch supersedes the run already going for it, so cancel that +# one rather than letting both provision. The management step deploys ~6 +# clusters and holds them for the length of the suite, so two overlapping runs +# double what the org is carrying for no extra coverage. Cancellation is safe +# here because the cleanup step is `if: always()` and reaps from the ledger. +# +# Not applied to coverage.yml: it is schedule/dispatch only, and its two jobs +# are meant to run at once with separate ledgers. +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + + jobs: test-coverage: runs-on: ubuntu-latest From caed76451d1e3ce9ec09ab41bc2b8eb5bddc47bb Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 21 Sep 2026 15:25:06 -0400 Subject: [PATCH 17/30] Keep the pool's mocked units out of the real ledger TestSharedClusterPool.setUp saved and restored _pool, _pool_skip, _tracked and the owner, but not SINGLESTOREDB_TEST_DEPLOYMENT_LOG. Its stand-in manager calls the real utils.track, which ledgers, and _pool_id is the live one, so under CI these mocked units appended records naming the worker's actual pool clusters with a fabricated `id-of-...` id. The cleanup step then could not resolve those ids and exited non-zero on every run that collected this class, which buried any genuine unresolved record in noise that was never a real deployment: 2 ledger record(s) could not be resolved, so they may still be live: ? cluster cl-test-shared-1-37446865 (id-of-cl-test-shared-1-37446865): 400 Redirected to a temporary path, which is what the ledger's own suite already does. Co-Authored-By: Claude Opus 5 --- singlestoredb/tests/test_management_utils.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index 99ee9594e..d38f8b8e6 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -1947,6 +1947,21 @@ def setUp(self): from singlestoredb.tests import utils self.utils = utils + # Redirected before anything can create a cluster: the stand-in + # manager's create_cluster calls the real utils.track, which ledgers, + # and _pool_id is the live one, so under CI these mocked units used to + # append `id-of-cl-test-shared-N-` to the job's real + # ledger. The cleanup step then could not resolve those ids and exited + # non-zero on every run, burying any genuine unresolved record. + tmp = tempfile.mkdtemp() + self.addCleanup(shutil.rmtree, tmp, True) + patcher = patch.dict( + os.environ, + {utils.LEDGER_ENV_VAR: os.path.join(tmp, 'deployments.jsonl')}, + ) + patcher.start() + self.addCleanup(patcher.stop) + self.saved_pool = list(utils._pool) self.saved_skip = utils._pool_skip self.saved_tracked = list(utils._tracked) From a981afefc6f12935f4a5c3c64986768101709047 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Tue, 22 Sep 2026 09:12:36 -0400 Subject: [PATCH 18/30] Stop cancelling a run that may be mid-provision cancel-in-progress halved the clusters two overlapping runs carry, on the grounds that the ledger reaper is an if: always() step and so survives the cancellation. It does not survive it for long enough. GitHub force-terminates a cancelled job's remaining steps after a 5-minute cancellation timeout, always() included, and an S-00 cluster refuses DELETE until it reaches ACTIVE, which is ~460s at worst. A run cancelled inside its first several minutes is therefore killed while still being told 400/409, and the cluster keeps billing -- the exact leak the ledger was added to stop, reintroduced by the thing that was supposed to be safe because of it. Drop the concurrency block. A run that is not cancelled reaches the reaper with the job's full budget, which is where cleanup_deployments.TERMINATE_TIMEOUT of 600s is actually spendable. This narrows the window rather than closing it: a manually cancelled run can still be killed mid-provision, and there is no scheduled janitor, so that remainder needs cleanup_deployments.py --older-than by hand. Say so in the comments on both cleanup steps and on TERMINATE_TIMEOUT, none of which should read as though always() were a guarantee. Incident 35631802648 stays covered: it was cancelled 19 minutes in, well past ACTIVE. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 33 ++++++++++++---------- .github/workflows/coverage.yml | 8 ++++++ singlestoredb/tests/cleanup_deployments.py | 14 +++++---- 3 files changed, 35 insertions(+), 20 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 7572c1759..c18c49489 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -8,19 +8,14 @@ on: workflow_dispatch: -# A push to a branch supersedes the run already going for it, so cancel that -# one rather than letting both provision. The management step deploys ~6 -# clusters and holds them for the length of the suite, so two overlapping runs -# double what the org is carrying for no extra coverage. Cancellation is safe -# here because the cleanup step is `if: always()` and reaps from the ledger. -# -# Not applied to coverage.yml: it is schedule/dispatch only, and its two jobs -# are meant to run at once with separate ledgers. -concurrency: - group: ${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true - - +# No `concurrency` block with `cancel-in-progress`, deliberately. Cancelling the +# run a push supersedes would halve the clusters two overlapping runs carry, but +# it cannot be made safe: GitHub force-terminates a cancelled job's remaining +# steps after a 5-minute cancellation timeout, `if: always()` included, and an +# S-00 cluster refuses DELETE until it is ACTIVE (~460s). A run cancelled inside +# its first several minutes would die with a cluster it is not yet allowed to +# delete, and nothing on a schedule would come along to reap it. Letting both +# runs finish costs clusters; cancelling them costs stranded clusters. jobs: test-coverage: runs-on: ubuntu-latest @@ -229,11 +224,19 @@ jobs: coverage xml coverage html - # if: always() is the whole point -- this has to run when the job is - # cancelled, which is what left three clusters billing in run + # if: always() is the whole point -- this has to run when the job fails or + # is cancelled, which is what left three clusters billing in run # 35631802648 (see the matching step in coverage.yml). On a PR the # management step above only runs when the change detector fires, so most # runs reach this with an empty ledger and it reports nothing. + # + # always() is not a guarantee, only a best effort: a cancelled job's + # remaining steps are force-terminated after GitHub's 5-minute + # cancellation timeout, so this covers a cancel whose clusters are already + # ACTIVE -- run 35631802648 was cancelled 19 minutes in -- but not one in + # the first several minutes, where DELETE is still being refused when the + # step is killed. That remainder needs `cleanup_deployments.py + # --older-than` run by hand; nothing here is on a schedule. - name: Terminate any deployment the tests left behind if: always() run: | diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 8eedfae86..d8e101ff4 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -104,6 +104,14 @@ jobs: # Last step in the job so it covers every pytest step above it. Placing # it after each one instead would add nothing: a cancellation anywhere # still runs the remaining always() steps. + # + # What always() does not buy is unlimited time. GitHub force-terminates a + # cancelled job's remaining steps after a 5-minute cancellation timeout, + # and an S-00 cluster refuses DELETE until it is ACTIVE (~460s). The leak + # above is covered because it was cancelled 19 minutes in, well past that; + # a cancel in the first several minutes would be killed here still being + # told 400/409, and needs `cleanup_deployments.py --older-than` run by + # hand afterwards. - name: Terminate any deployment the tests left behind if: always() run: | diff --git a/singlestoredb/tests/cleanup_deployments.py b/singlestoredb/tests/cleanup_deployments.py index 71601b3f5..6b715431a 100644 --- a/singlestoredb/tests/cleanup_deployments.py +++ b/singlestoredb/tests/cleanup_deployments.py @@ -92,11 +92,15 @@ #: How long to keep retrying a deployment the API will not delete yet. This is #: the end of the line -- nothing runs after this tool -- so it does not borrow #: ``utils.TERMINATE_RETRY_TIMEOUT``, which is deliberately short so the sweep -#: between test classes cannot stall the suite. A cancelled job's cluster may -#: only just have been POSTed, ``DELETE`` is refused until it is up, and an -#: S-00 cluster reaching ACTIVE is ~460s at worst, so anything shorter than a -#: full provision leaves it billing. The only cost of waiting is this CI step's -#: wall clock. +#: between test classes cannot stall the suite. Here a deployment may still be +#: coming up, ``DELETE`` is refused until it is, and an S-00 cluster reaching +#: ACTIVE is ~460s at worst, so anything shorter than a full provision leaves it +#: billing. An upper bound on retrying, not a promise of it: the whole budget is +#: available when the job that calls this ends normally or fails, but a +#: *cancelled* job's steps are force-terminated after GitHub's 5-minute +#: cancellation timeout, so a cancel early in a provision gets killed here +#: regardless of what this says. The only cost of the larger budget is the CI +#: step's wall clock. TERMINATE_TIMEOUT = 600.0 #: Names the suite generates. Anchored, because these run against a real From 39cdb149aa56638da243bfad5b72e8a832a8af35 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Tue, 22 Sep 2026 09:14:57 -0400 Subject: [PATCH 19/30] Run the v1 management gate serially, after the v2 job The nightly coverage run put four workers on the management API at once: -n 2 in the v2 job's shared cluster pool, plus -n 2 in the v1 job deploying workspace groups of its own, with the two jobs running concurrently. That is more than the org wants to carry, and none of it buys coverage -- the parallel default was sized for the v2 suite alone. Take the v1 step to -n 0 and make its job need test-coverage, so the v1 groups are never in flight alongside the v2 pool. if: always() on the job, because this is a legacy gate rather than a downstream build: a v2 failure says nothing about the v1 endpoints, and skipping v1 for it would hide a v1 regression behind an unrelated one. This is a 01:00 cron, so the serialized wall clock costs nothing. The per-job ledgers stay per-job. Their original justification was that the jobs overlapped, which is no longer true, so say the part that still holds: separate ledgers mean a job's sweep can only reach records it wrote itself. Co-Authored-By: Claude Opus 5 --- .github/workflows/coverage.yml | 32 +++++++++++++++++++++++++------- 1 file changed, 25 insertions(+), 7 deletions(-) diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index d8e101ff4..2cbb87a08 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -11,9 +11,9 @@ jobs: environment: Base # One ledger for the whole job, so the cleanup step below can reap what any - # of the pytest steps created. Per job rather than shared: the jobs here run - # concurrently, and a shared ledger would have each one terminating the - # other's clusters mid-run. + # of the pytest steps created. Per job rather than shared: each job gets its + # own runner and workspace anyway, and keeping the ledgers separate means a + # job's sweep can only ever reach records it wrote itself. env: SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl @@ -128,9 +128,21 @@ jobs: runs-on: ubuntu-latest environment: Base - # A ledger of its own, not shared with test-coverage: the two jobs run - # concurrently, and one sweeping the other's ledger would terminate - # clusters a live run is using. + # Waits for test-coverage rather than running alongside it, so the v1 + # workspace groups are never in flight at the same time as the v2 suite's + # cluster pool -- together they put more on the org than it wants to carry. + # This is a nightly cron, so the extra wall clock costs nothing. + # + # if: always() because this is a legacy gate, not a downstream build: a v2 + # failure above says nothing about the v1 endpoints, and skipping v1 for it + # would hide a v1 regression behind an unrelated one. + needs: test-coverage + if: always() + + # A ledger of its own, not shared with test-coverage. The jobs no longer + # overlap, but they still run on separate runners with separate workspaces, + # and keeping the ledgers distinct means neither job's sweep can reach the + # other's records. env: SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl @@ -160,8 +172,14 @@ jobs: pip install -e ".[dev]" - name: Run v1 management API tests + # -n 0 overrides the -n 2 in pyproject.toml's addopts. The parallel + # default is tuned for the v2 management suite's shared cluster pool; the + # v1 classes deploy workspace groups of their own, so two workers here + # put twice that in flight, on top of whatever the v2 job is holding at + # the same time. Serial keeps this job's contribution to the org's + # cluster count to one deployment at a time. run: | - pytest -v -m 'management_v1' --pyargs singlestoredb.tests + pytest -v -n 0 -m 'management_v1' --pyargs singlestoredb.tests env: SINGLESTOREDB_URL: "root:root@127.0.0.1:3307" SINGLESTOREDB_PURE_PYTHON: 0 From 28153d76eaddbad6e41f3a5d6c8fdb48cb01baac Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Tue, 22 Sep 2026 10:33:12 -0400 Subject: [PATCH 20/30] Skip the v1 gate on a cancelled run management-v1-tests needs test-coverage with if: always(), which is true after cancellation too, so a cancelled nightly still started the job and deployed workspace groups. GitHub force-terminates a cancelled job's remaining steps after 5 minutes, so the ledger sweep would be killed with those deployments still pre-ACTIVE and refusing DELETE -- the job strands exactly what the sweep is there to reap. !cancelled() keeps the intended behaviour: the gate still runs when the v2 job fails, since a v2 failure says nothing about the v1 endpoints. Co-Authored-By: Claude Opus 5 --- .github/workflows/coverage.yml | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 2cbb87a08..aaf9d7571 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -133,11 +133,18 @@ jobs: # cluster pool -- together they put more on the org than it wants to carry. # This is a nightly cron, so the extra wall clock costs nothing. # - # if: always() because this is a legacy gate, not a downstream build: a v2 - # failure above says nothing about the v1 endpoints, and skipping v1 for it - # would hide a v1 regression behind an unrelated one. + # Runs even when test-coverage fails, because this is a legacy gate, not a + # downstream build: a v2 failure above says nothing about the v1 endpoints, + # and skipping v1 for it would hide a v1 regression behind an unrelated one. + # + # !cancelled() rather than always(), which stays true through cancellation + # too. A cancelled run must not go on to start provisioning workspace groups + # here: GitHub force-terminates a cancelled job's remaining steps after a + # 5-minute cancellation timeout, so the ledger sweep below would be killed + # while the new deployments were still pre-ACTIVE and refusing DELETE -- the + # job would strand exactly what it was added to clean up. needs: test-coverage - if: always() + if: ${{ !cancelled() }} # A ledger of its own, not shared with test-coverage. The jobs no longer # overlap, but they still run on separate runners with separate workspaces, From cb13ea0e6f27d74fe62e1831967baa68fe54df47 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Tue, 22 Sep 2026 11:00:03 -0400 Subject: [PATCH 21/30] Bound what a leaked deployment costs, and borrow the SHOW CLUSTERS fixture Three changes to what the management suites deploy. expires_at on every creation that accepts one. The sweep, the ledger and tearDownClass are all in-process, so none of them survives the runner being killed -- and GitHub force-terminates a cancelled job's remaining steps after five minutes, which is less than the ~460s an S-00 takes to become deletable. An expiry is a property of the deployment, so the control plane honours it either way. resources/create_test_cluster.py has passed one nightly since it was written; this is the same argument applied to the fixtures. It reaches v2 create_cluster and v1 create_workspace_group -- a v1 workspace has no expiry of its own and goes with its group, and the starter variants take no such argument. TestClusterFusion borrows from the shared pool instead of deploying three clusters of its own. It is four SHOW statements and mutates nothing, which is what the pool asks of a consumer. Net two fewer clusters per run: the pool grows 2 -> 3, and three per-class deployments go away. It joins the Stage group rather than the Jobs one because Stage already asks for two, so the growth costs one cluster rather than two. Its assertions now scope to shared_cluster_pattern() and count against shared_cluster_names() rather than a literal 3, so a later class asking for a bigger pool cannot break them. test_create_cluster_named_project terminated through a bare terminate(force=True) whose comment said force makes a PENDING cluster deletable. It does not -- the API refuses it with a 400 or 409 either way, which is the whole reason utils.terminate retries -- and the except swallowed the refusal, so the usual outcome was a cluster left to the sweep. Now calls utils.terminate, as its sibling test already did. The neighbouring claim that Fusion-created clusters are invisible to tracking was also wrong: the handler goes through ClusterManager.create_cluster, which is the method _CREATORS wraps. Co-Authored-By: Claude Opus 5 --- singlestoredb/tests/test_fusion.py | 136 ++++++++++++------- singlestoredb/tests/test_management_utils.py | 47 +++++++ singlestoredb/tests/test_management_v1.py | 8 ++ singlestoredb/tests/test_management_v2.py | 1 + singlestoredb/tests/utils.py | 89 ++++++++++-- 5 files changed, 220 insertions(+), 61 deletions(-) diff --git a/singlestoredb/tests/test_fusion.py b/singlestoredb/tests/test_fusion.py index 5de2b0272..f130e4673 100644 --- a/singlestoredb/tests/test_fusion.py +++ b/singlestoredb/tests/test_fusion.py @@ -969,24 +969,15 @@ def setUpClass(cls): # US-only: no test here asserts anything about these groups' regions, # and creation in some non-US regions fails with a control-plane 500. us_regions = [x for x in mgr.regions if x.name.startswith('US')] - wg = mgr.create_workspace_group( - f'A Fusion Testing {cls.id}', - region=random.choice(us_regions), - firewall_ranges=[], - ) - cls.workspace_groups.append(wg) - wg = mgr.create_workspace_group( - f'B Fusion Testing {cls.id}', - region=random.choice(us_regions), - firewall_ranges=[], - ) - cls.workspace_groups.append(wg) - wg = mgr.create_workspace_group( - f'C Fusion Testing {cls.id}', - region=random.choice(us_regions), - firewall_ranges=[], - ) - cls.workspace_groups.append(wg) + for letter in ('A', 'B', 'C'): + cls.workspace_groups.append( + mgr.create_workspace_group( + f'{letter} Fusion Testing {cls.id}', + region=random.choice(us_regions), + firewall_ranges=[], + expires_at=utils.DEPLOYMENT_EXPIRES_AT, + ), + ) @classmethod def tearDownClass(cls): @@ -1425,6 +1416,12 @@ class _ClusterFusionMixin: cluster-less ones start immediately and no class deploys more than it reads. + Only :class:`TestClusterFusionSuspendResume` still names a prefix, because + it is the only one left that both needs a cluster up front and mutates it. + :class:`TestClusterFusion` reads without mutating and so borrows from + ``utils.shared_clusters``; the lifecycle suites create their own clusters + in the test bodies, those creates being the subject under test. + Not a ``TestCase``, and named with a leading underscore: pytest collects any ``Test``-prefixed ``TestCase`` subclass it can reach, so a base that was either would run every inherited test a second time under a fixture @@ -1495,6 +1492,7 @@ def setUpClass(cls): f'{prefix}-fusion-cluster-{cls.id}', region=region, size='S-00', + expires_at=utils.DEPLOYMENT_EXPIRES_AT, project=cls.project_id, wait_on_active=True, wait_timeout=1200, @@ -1550,24 +1548,46 @@ def tearDown(self): @pytest.mark.management +@pytest.mark.xdist_group(utils.SHARED_CLUSTER_STAGE_GROUP) class TestClusterFusion(_ClusterFusionMixin, unittest.TestCase): """ - ``SHOW CLUSTERS`` against three deployed clusters. - - Three of them so the ``LIKE``/``ORDER BY``/``LIMIT`` assertions have - something to sort. Nothing here mutates a cluster, which is what makes the - fixture shareable -- ``SUSPEND``/``RESUME`` cannot share it and deploys its - own in :class:`TestClusterFusionSuspendResume`. + ``SHOW CLUSTERS`` against the shared cluster pool. + + Borrows rather than deploying: nothing here mutates a cluster -- these are + four ``SHOW`` statements -- which is the condition ``utils.shared_clusters`` + asks of a consumer. ``SUSPEND``/``RESUME`` cannot borrow and deploys its own + in :class:`TestClusterFusionSuspendResume`. + + Three of them, which is one more than the pool was built for, so the + ``LIKE``/``ORDER BY``/``LIMIT`` assertions have something to sort. Joining + the Stage group rather than the Jobs one because Stage already asks for two: + the pool grows to the largest request, so this costs that group one extra + cluster instead of three, and the Jobs group is left at one. + + Every assertion here is scoped to ``utils.shared_cluster_pattern()`` and + counted against ``utils.shared_cluster_names()``, never a literal. The pool + is shared and grows to whatever the largest request in the process turns out + to be, so a hardcoded 3 would break the day a class asks for four -- and + would break silently, as a row count, which is the failure this class had + before when its count depended on other classes' clusters leaving the list + endpoint in time. """ - fixture_prefixes = ('a', 'b', 'c') + #: Borrowed, so kept out of ``clusters``, which ``tearDownClass`` + #: terminates. A pool cluster must outlive the class that used it. + pool_clusters: List[Any] = [] + + @classmethod + def setUpClass(cls): + super().setUpClass() + cls.pool_clusters = utils.shared_clusters(3) def test_show_clusters(self): self.cur.execute('show clusters') names = [x[0] for x in self.cur.fetchall()] assert self.cur.description[0][0] == 'Name' - for prefix in ('a', 'b', 'c'): - assert f'{prefix}-fusion-cluster-{self.id}' in names, names + for cluster in type(self).pool_clusters: + assert cluster.name in names, names def test_show_clusters_columns(self): self.cur.execute('show clusters') @@ -1582,42 +1602,45 @@ def test_show_clusters_columns(self): 'TerminatedAt', ], cols + cluster = type(self).pool_clusters[0] rows = {x[0]: x for x in self.cur.fetchall()} - row = rows[f'a-fusion-cluster-{self.id}'] + row = rows[cluster.name] # Region is the provider slug; Cluster has no region object at v2. assert row[2], row assert row[5], row # ProjectName, not the ID: the column reports the name the project - # listing gives for the ID the cluster was deployed into. - project = type(self).manager.projects[type(self).project_id] - assert row[9] == project.name, row + # listing gives for the ID the cluster was deployed into. Read back + # from the cluster rather than from this class's own project_id -- + # the pool resolves its project independently, and asserting against + # the borrower's copy would be asserting the two resolutions agree. + expected = type(self).manager.get_cluster(cluster.id).project + assert row[9] == expected.name, row def test_show_clusters_like(self): - self.cur.execute(f'show clusters like "a-fusion-cluster-{self.id}"') + one = type(self).pool_clusters[0].name + self.cur.execute(f'show clusters like "{one}"') names = [x[0] for x in self.cur.fetchall()] - assert names == [f'a-fusion-cluster-{self.id}'], names + assert names == [one], names - self.cur.execute(f'show clusters like "%-fusion-cluster-{self.id}"') + self.cur.execute( + f'show clusters like "{utils.shared_cluster_pattern()}"', + ) names = [x[0] for x in self.cur.fetchall()] - assert len(names) == 3, names + assert sorted(names) == sorted(utils.shared_cluster_names()), names def test_show_clusters_order_by_and_limit(self): - self.cur.execute( - f'show clusters like "%-fusion-cluster-{self.id}" order by name', - ) + pattern = utils.shared_cluster_pattern() + + self.cur.execute(f'show clusters like "{pattern}" order by name') names = [x[0] for x in self.cur.fetchall()] assert names == sorted(names), names - self.cur.execute( - f'show clusters like "%-fusion-cluster-{self.id}" ' - 'order by name desc', - ) + self.cur.execute(f'show clusters like "{pattern}" order by name desc') names = [x[0] for x in self.cur.fetchall()] assert names == sorted(names, reverse=True), names self.cur.execute( - f'show clusters like "%-fusion-cluster-{self.id}" ' - 'order by name limit 2', + f'show clusters like "{pattern}" order by name limit 2', ) names = [x[0] for x in self.cur.fetchall()] assert len(names) == 2, names @@ -1865,10 +1888,14 @@ def test_create_cluster_without_project(self): # was created, and the cleanup for when something was. The test # only passes if this comes back empty, so the terminate below # fires exactly when the assertion is about to fail -- which is - # also the only case where a cluster exists. Nothing else would - # remove it: the create goes through Fusion SQL rather than - # ClusterManager.create_cluster, so the tracking wrapper never sees - # it and there is no _tracked entry for the sweep to find. + # also the only case where a cluster exists. + # + # Belt and braces rather than the only cleanup, contrary to what + # this used to claim: the handler reaches the API through + # ClusterManager.create_cluster (fusion/handlers/cluster.py), which + # is the method utils._CREATORS wraps, so a cluster created here is + # tracked and ledgered like any other and the sweep would find it. + # Terminating it now just means not waiting for the sweep. live = [ x for x in mgr.clusters if x.name == name and x.terminated_at is None @@ -1916,18 +1943,23 @@ def test_create_cluster_named_project(self): assert mgr.get_cluster(cluster_id).project.id == project.id finally: - # force=True: the cluster is still PENDING, having never been - # waited out, and a termination request is refused otherwise. + # utils.terminate, not a bare terminate(force=True). The cluster is + # PENDING, never having been waited out, and force does not make a + # pre-ACTIVE deployment deletable -- the API refuses it with a 400 + # or a 409 either way (see utils.terminate, which retries exactly + # that). A single forced DELETE here was therefore the likeliest + # outcome, swallowed by the except, leaving the cluster to the + # sweep; utils.terminate retries until it lands. if cluster_id is not None: try: - mgr.get_cluster(cluster_id).terminate(force=True) + utils.terminate(mgr.get_cluster(cluster_id)) except Exception: pass else: for cluster in mgr.clusters: if cluster.name == name and cluster.terminated_at is None: try: - cluster.terminate(force=True) + utils.terminate(cluster) except Exception: pass diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index d38f8b8e6..b1e784a8c 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -2126,6 +2126,53 @@ def test_an_explicit_project_does_not_need_a_standard_one(self): self.assertEqual(self.created[0][2]['project'], 'chosen-project') + def test_pool_clusters_are_given_an_expiry(self): + # The only cleanup that survives the process being killed, so it has to + # be on the POST rather than left to the sweep. + with self._patched(): + self.utils.shared_clusters(2) + + self.assertEqual( + [x[2].get('expires_at') for x in self.created], + [self.utils.DEPLOYMENT_EXPIRES_AT] * 2, + ) + + def test_the_pattern_matches_the_pool_and_is_scoped_to_this_process(self): + with self._patched(): + self.utils.shared_clusters(2) + + pattern = self.utils.shared_cluster_pattern() + prefix, _, suffix = pattern.partition('%') + + # A LIKE pattern, so assert it the way the server would read it: + # every pool name matches, and the suffix is the per-process id that + # keeps another run's pool from matching. + for name in self.utils.shared_cluster_names(): + self.assertTrue(name.startswith(prefix), (name, pattern)) + self.assertTrue(name.endswith(suffix), (name, pattern)) + + self.assertEqual(suffix, f'-{self.utils._pool_id}') + + # And another process's pool does not: same prefix, different id. + other = f'cl-test-shared-0-{"f" * 8}' + self.assertTrue(other.startswith(prefix), (other, pattern)) + self.assertFalse(other.endswith(suffix), (other, pattern)) + + def test_the_names_follow_the_pool_as_it_grows(self): + # Read at assertion time rather than cached, so a class that asks for + # more clusters later cannot leave an exact-count expectation stale. + with self._patched(): + self.utils.shared_clusters(1) + self.assertEqual(len(self.utils.shared_cluster_names()), 1) + + self.utils.shared_clusters(3) + self.assertEqual(len(self.utils.shared_cluster_names()), 3) + + self.assertEqual( + self.utils.shared_cluster_names(), + [x[0] for x in self.created], + ) + class TestClearStage(unittest.TestCase): """ diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index b0633f010..63e7b9dd0 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -35,6 +35,7 @@ from singlestoredb.management.job import TargetType from singlestoredb.management.region import Region from singlestoredb.management.utils import NamedList +from singlestoredb.tests import utils TEST_DIR = pathlib.Path(os.path.dirname(__file__)) @@ -77,9 +78,12 @@ def setUpClass(cls): region=random.choice(us_regions).id, admin_password=cls.password, firewall_ranges=['0.0.0.0/0'], + expires_at=utils.DEPLOYMENT_EXPIRES_AT, ) try: + # No expiry of its own: only the group has an expiresAt, and it + # takes its workspaces with it. See utils.DEPLOYMENT_EXPIRES_AT. cls.workspace = cls.workspace_group.create_workspace( f'ws-test-{name}-x', wait_on_active=True, @@ -382,6 +386,7 @@ def setUpClass(cls): region=random.choice(us_regions).id, admin_password=cls.password, firewall_ranges=['0.0.0.0/0'], + expires_at=utils.DEPLOYMENT_EXPIRES_AT, ) @classmethod @@ -993,9 +998,12 @@ def setUpClass(cls): region=random.choice(us_regions).id, admin_password=cls.password, firewall_ranges=['0.0.0.0/0'], + expires_at=utils.DEPLOYMENT_EXPIRES_AT, ) try: + # No expiry of its own: only the group has an expiresAt, and it + # takes its workspaces with it. See utils.DEPLOYMENT_EXPIRES_AT. cls.workspace = cls.workspace_group.create_workspace( f'ws-test-{name}-x', wait_on_active=True, diff --git a/singlestoredb/tests/test_management_v2.py b/singlestoredb/tests/test_management_v2.py index 5842bdc56..fe65e47df 100644 --- a/singlestoredb/tests/test_management_v2.py +++ b/singlestoredb/tests/test_management_v2.py @@ -1332,6 +1332,7 @@ def setUpClass(cls): region=region, size='S-00', firewall_ranges=['0.0.0.0/0'], + expires_at=utils.DEPLOYMENT_EXPIRES_AT, project=_project_id(cls.manager), wait_on_active=True, ) diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index 4ea29df3f..09f8ba3c2 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -307,6 +307,38 @@ def drop_user(name: str) -> None: # ignored -- so tracked objects do not have to be untracked by the tests that # clean up after themselves. # +# Everything above is in-process, which is the one thing it cannot fix: the +# sweep, the ledger and `tearDownClass` all die with the interpreter. A job +# killed mid-provision -- GitHub force-terminates a cancelled job's remaining +# steps after a five-minute cancellation timeout -- leaves a PENDING cluster +# that no code here will ever get another chance to delete. `expires_at` is +# the answer to that, and only that: it is a property of the deployment, so +# the control plane honours it whether or not this process is still alive. +# + +#: Expiry to request on every deployment a test creates, as the duration +#: string `POST` accepts (`resources/create_test_cluster.py` has passed one +#: nightly since it was written). The backstop under the sweep and the ledger, +#: not a replacement for either: a test still terminates what it created, and +#: nothing waits for an expiry to fire. +#: +#: Two hours, against a `wait_timeout` of 1200s and a longest test (Fusion +#: `CREATE`/`DROP`, which provisions twice in sequence) of about twenty +#: minutes. Enough headroom that an expiry can never land on a deployment a +#: test is still using, which would show up as an unrelated flake and be read +#: as an API fault. +#: +#: Applied to what accepts it, which is the deployments that cost: v2 +#: `ClusterManager.create_cluster` and v1 +#: `WorkspaceManager.create_workspace_group`. The rest take no `expires_at` and +#: need none: +#: +#: * a v1 workspace -- `expiresAt` is a property of the group, and terminating +#: the group takes its workspaces with it; +#: * the starter deployments -- `create_starter_cluster` and +#: `create_starter_workspace` have no such argument, and being shared tier +#: they are not what a leak costs. +DEPLOYMENT_EXPIRES_AT = '2h' #: (owner, label, object) for every deployment created so far and not yet #: swept. The owner is the test class that was running at creation time, so @@ -1018,10 +1050,16 @@ def tracked_labels() -> List[str]: # does ``TestWorkspaceFusion``, whose workspace groups are the subject of its # ``SHOW WORKSPACE GROUPS`` assertions and cost 40s to deploy unwaited anyway. # -# What makes the four borrowers safe is that each scopes its assertions to -# itself: every Stage path is namespaced with the class's ``cls.id``, job -# listings filter by job id rather than listing a deployment's jobs, and none -# of them asserts a row count over an org-wide listing. +# What makes the borrowers safe is that each scopes its assertions to itself: +# every Stage path is namespaced with the class's ``cls.id``, job listings +# filter by job id rather than listing a deployment's jobs, and none of them +# asserts a row count over an org-wide listing. +# +# ``TestClusterFusion`` is the one borrower that does count rows, because +# ``SHOW CLUSTERS ... LIKE`` is what it tests. It stays inside that rule by +# counting over :func:`shared_cluster_pattern` -- which matches this process's +# pool and nothing else -- against :func:`shared_cluster_names` rather than a +# literal, so growing the pool cannot break it. # # The pool is process-wide, so under ``pytest-xdist`` every worker that gets a # borrowing class builds a pool of its own. The ``xdist_group`` marks below @@ -1030,18 +1068,24 @@ def tracked_labels() -> List[str]: #: ``xdist_group`` names for the classes that borrow from the pool, so #: ``--dist loadgroup`` puts each set on one worker and each set builds one -#: pool. Two groups rather than one: a single group serialises all four classes +#: pool. Two groups rather than one: a single group serialises every borrower #: behind one pool build, and the groups run concurrently on separate workers, #: so splitting costs one extra cluster and halves that chain. #: -#: Stage wants two clusters (``TestStageFusion`` names a second one in -#: ``IN GROUP``) and jobs want one, so the split follows what they borrow: +#: The split follows what each set borrows -- three for Stage, one for Jobs: #: -#: * ``SHARED_CLUSTER_STAGE_GROUP`` -- ``TestStageFusion``, v2 ``TestStage`` +#: * ``SHARED_CLUSTER_STAGE_GROUP`` -- ``TestStageFusion`` (two; it names a +#: second in ``IN GROUP``), v2 ``TestStage`` (one), ``TestClusterFusion`` +#: (three, for its ``LIKE``/``ORDER BY``/``LIMIT`` rows) #: * ``SHARED_CLUSTER_JOBS_GROUP`` -- ``TestJobsFusion``, v2 ``TestJob`` #: +#: ``TestClusterFusion`` sits with Stage rather than Jobs deliberately: the +#: pool grows to the largest request, so putting the class that wants three +#: with the group that already wants two costs one extra cluster, where +#: putting it with Jobs would cost two and leave Stage's pool untouched. +#: #: Without ``-n``/``--dist loadgroup`` the marks do nothing: one process, one -#: pool of two, which is the serial behaviour they were added on top of. +#: pool of three, which is the serial behaviour they were added on top of. SHARED_CLUSTER_STAGE_GROUP = 'shared-cluster-stage' SHARED_CLUSTER_JOBS_GROUP = 'shared-cluster-jobs' @@ -1127,6 +1171,7 @@ def setUpClass(cls): # pool cluster stands in for those, so it has to be at # least as reachable as what it replaces. firewall_ranges=['0.0.0.0/0'], + expires_at=DEPLOYMENT_EXPIRES_AT, project=project_id, wait_on_active=True, wait_timeout=1200, @@ -1138,6 +1183,32 @@ def setUpClass(cls): return _pool[:count] +def shared_cluster_pattern() -> str: + """ + ``LIKE`` pattern matching this process's pool clusters and nothing else. + + The suffix is what scopes it: ``_pool_id`` is minted per process, so a + concurrent run's pool -- or another xdist worker's -- does not match, and + neither does any other ``cl-test-*`` deployment. + + For a suite asserting an exact row count over the pool, pair it with + :func:`shared_cluster_names` rather than a literal: the pool grows to the + largest request any class makes, so the number is not fixed at import. + """ + return f'cl-test-shared-%-{_pool_id}' + + +def shared_cluster_names() -> List[str]: + """ + Names of every cluster in the pool as it stands right now. + + Read at assertion time, not cached: a class that runs later and asks for + more clusters than this one did grows the pool, and an expectation built + from a literal count would go stale the moment that happened. + """ + return [x.name for x in _pool] + + class CountingManager: """ Stand-in for a :class:`Manager` that records every request. From df4be4b9acf955fa969f363d409aad89c2d98dc9 Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Wed, 23 Sep 2026 13:29:50 -0400 Subject: [PATCH 22/30] Retry a workspace group creation that cannot take the lock [skip ci] The v1 management API refuses a create_workspace_group with "could not acquire lock" when another creation in the organization holds it, and nothing retried that: Manager.RETRY_STATUSES covers 429 and 5xx only, so the error reached setUpClass, where one failure fails every test in the class. The last management_v1 run went that way in bulk. The lock clears on its own in seconds, so utils.create_retrying() waits and asks again -- six attempts spaced 20/40/60/60/60s with up to 5s of jitter, matched on the error message rather than on a status. Spacing is sized for lock contention, not provisioning: a group POST returns in seconds, so what is being waited out is the other holder finishing, not a deployment coming up. The jitter is there because two xdist workers -- or the v1 nightly beside a concurrent v2 job -- back off by the same amounts from the same moment, and would otherwise retry in step forever. Four minutes at worst, small enough that a genuinely stuck organization fails the class instead of idling out the job's timeout. Applied to the four v1 setUpClass creations: the three in test_management_v1.py and TestWorkspaceFusion, which deploys three groups in a row and is the likeliest source of the contention. Co-Authored-By: Claude Opus 5 --- singlestoredb/tests/test_fusion.py | 6 +- singlestoredb/tests/test_management_utils.py | 116 +++++++++++++++++++ singlestoredb/tests/test_management_v1.py | 18 ++- singlestoredb/tests/utils.py | 79 +++++++++++++ 4 files changed, 215 insertions(+), 4 deletions(-) diff --git a/singlestoredb/tests/test_fusion.py b/singlestoredb/tests/test_fusion.py index f130e4673..91cdbf070 100644 --- a/singlestoredb/tests/test_fusion.py +++ b/singlestoredb/tests/test_fusion.py @@ -970,8 +970,12 @@ def setUpClass(cls): # and creation in some non-US regions fails with a control-plane 500. us_regions = [x for x in mgr.regions if x.name.startswith('US')] for letter in ('A', 'B', 'C'): + # Retried: three creations in a row in one organization is exactly + # what comes back "could not acquire lock", and raising here fails + # every test in the class. See utils.create_retrying. cls.workspace_groups.append( - mgr.create_workspace_group( + utils.create_retrying( + mgr.create_workspace_group, f'{letter} Fusion Testing {cls.id}', region=random.choice(us_regions), firewall_ranges=[], diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index b1e784a8c..d842ca90d 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -1937,6 +1937,122 @@ def terminate(inner, force=False): self.assertEqual(calls, [True]) +class TestCreateRetryingLockConflicts(unittest.TestCase): + """ + ``utils.create_retrying()``'s retry for a creation the API will not start + because it cannot take the lock. + + A ``create_workspace_group`` that comes back "could not acquire lock" is + not retried by the transport -- Manager.RETRY_STATUSES is 429 and 5xx -- + and it happens in ``setUpClass``, so one lock conflict fails every test in + the class. A whole management_v1 run went that way. + """ + + def setUp(self): + from singlestoredb.tests import utils + self.utils = utils + self.slept = [] + + patcher = patch('time.sleep', self.slept.append) + patcher.start() + self.addCleanup(patcher.stop) + + # Jitter off, so the waits asserted below are the spacing itself. + # It is covered separately. + patcher = patch.object(utils, 'CREATE_LOCK_RETRY_JITTER', 0.0) + patcher.start() + self.addCleanup(patcher.stop) + + def _creator(self, *msgs): + """A creator that fails with these messages in turn, then succeeds.""" + calls = [] + + def create(name, **kwargs): + calls.append(name) + if len(calls) <= len(msgs): + raise ManagementError(errno=400, msg=msgs[len(calls) - 1]) + return f'deployment {name}' + + create.calls = calls + return create + + def test_a_lock_conflict_is_retried_until_it_succeeds(self): + create = self._creator( + 'could not acquire lock', 'Could not acquire lock on workspace', + ) + out = self.utils.create_retrying(create, 'wg-test-a', region='x') + self.assertEqual(out, 'deployment wg-test-a') + self.assertEqual(len(create.calls), 3) + self.assertEqual(self.slept, [20.0, 40.0]) + + def test_the_retry_reuses_the_name(self): + """The conflict is a refusal to start, so nothing was created and + there is no deployment for the second attempt to collide with.""" + create = self._creator('could not acquire lock') + self.utils.create_retrying(create, 'wg-test-a') + self.assertEqual(create.calls, ['wg-test-a', 'wg-test-a']) + + def test_another_error_is_not_retried(self): + """A rejected request is a real failure; retrying it only delays the + report by four minutes.""" + create = self._creator('region is not available') + with self.assertRaises(ManagementError): + self.utils.create_retrying(create, 'wg-test-a') + self.assertEqual(len(create.calls), 1) + self.assertEqual(self.slept, []) + + def test_the_budget_is_bounded_and_the_error_is_re_raised(self): + """A genuinely stuck organization has to fail the class rather than + idle out the job's timeout.""" + create = self._creator(*(['could not acquire lock'] * 100)) + with self.assertRaises(ManagementError): + self.utils.create_retrying(create, 'wg-test-a') + self.assertEqual( + len(create.calls), self.utils.CREATE_LOCK_RETRY_ATTEMPTS, + ) + self.assertEqual(len(self.slept), 6 - 1) + + def test_the_wait_is_capped(self): + """Growth stops at the cap: what is being waited out is another + creation finishing, not a deployment coming up.""" + create = self._creator(*(['could not acquire lock'] * 100)) + with self.assertRaises(ManagementError): + self.utils.create_retrying(create, 'wg-test-a') + self.assertLessEqual( + max(self.slept), self.utils.CREATE_LOCK_RETRY_MAX_INTERVAL, + ) + self.assertEqual(self.slept, [20.0, 40.0, 60.0, 60.0, 60.0]) + + def test_the_waits_are_jittered(self): + """Two workers that collide on the lock back off by the same amounts + from the same moment, so without jitter they retry in step forever.""" + patcher = patch.object(self.utils, 'CREATE_LOCK_RETRY_JITTER', 5.0) + patcher.start() + self.addCleanup(patcher.stop) + + create = self._creator(*(['could not acquire lock'] * 100)) + with self.assertRaises(ManagementError): + self.utils.create_retrying(create, 'wg-test-a') + self.assertNotEqual(self.slept, [20.0, 40.0, 60.0, 60.0, 60.0]) + for wait, base in zip(self.slept, [20.0, 40.0, 60.0, 60.0, 60.0]): + self.assertGreaterEqual(wait, base) + self.assertLess(wait, base + 5.0) + + def test_the_message_match_does_not_need_a_status(self): + """The status a lock conflict arrives as is not documented, and the + status alone cannot tell a conflict from a rejection.""" + calls = [] + + def create(name): + calls.append(name) + if len(calls) == 1: + raise ManagementError(msg='Could not acquire lock') + return name + + self.utils.create_retrying(create, 'wg-test-a') + self.assertEqual(len(calls), 2) + + class TestSharedClusterPool(unittest.TestCase): """ The pool in ``tests/utils.py`` that keeps the Stage and Job suites from diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index 63e7b9dd0..8da4d1dfb 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -73,7 +73,11 @@ def setUpClass(cls): name = clean_name(secrets.token_urlsafe(20)[:20]) - cls.workspace_group = cls.manager.create_workspace_group( + # Retried: another creation in the organization holding the lock makes + # this come back "could not acquire lock", and raising here fails every + # test in the class. See utils.create_retrying. + cls.workspace_group = utils.create_retrying( + cls.manager.create_workspace_group, f'wg-test-{name}', region=random.choice(us_regions).id, admin_password=cls.password, @@ -381,7 +385,11 @@ def setUpClass(cls): name = clean_name(secrets.token_urlsafe(20)[:20]) - cls.wg = cls.manager.create_workspace_group( + # Retried: another creation in the organization holding the lock makes + # this come back "could not acquire lock", and raising here fails every + # test in the class. See utils.create_retrying. + cls.wg = utils.create_retrying( + cls.manager.create_workspace_group, f'wg-test-{name}', region=random.choice(us_regions).id, admin_password=cls.password, @@ -993,7 +1001,11 @@ def setUpClass(cls): name = clean_name(secrets.token_urlsafe(20)[:20]) - cls.workspace_group = cls.manager.create_workspace_group( + # Retried: another creation in the organization holding the lock makes + # this come back "could not acquire lock", and raising here fails every + # test in the class. See utils.create_retrying. + cls.workspace_group = utils.create_retrying( + cls.manager.create_workspace_group, f'wg-test-{name}', region=random.choice(us_regions).id, admin_password=cls.password, diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index 09f8ba3c2..f15a3e7a2 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -723,6 +723,85 @@ def terminate( time.sleep(interval) +#: Error text the management API comes back with when it will not start a +#: creation because something else in the organization holds the lock it +#: needs -- "could not acquire lock". Matched on the message rather than on +#: ``errno`` because the status it arrives as is not something the API +#: documents, and because the status alone cannot distinguish a lock conflict +#: (retry and it goes through) from a rejected request (retry and it never +#: will). +LOCK_ERROR_RE = re.compile(r'acquire[^.]{0,40}lock', re.I) + +#: Retry budget for :func:`create_retrying`. Five waits at 20s growing to a +#: 60s cap -- 20, 40, 60, 60, 60, so four minutes at worst. +#: +#: Sized for a lock held by another creation rather than for provisioning: a +#: workspace group POST returns in seconds, so what is being waited out is the +#: other holder finishing, not a deployment coming up. Hence spacing much +#: shorter than ``wait_on_active``'s but longer than a transport retry's, and a +#: budget small enough that a genuinely stuck organization fails the class +#: instead of idling out the job's timeout. +CREATE_LOCK_RETRY_ATTEMPTS = 6 +CREATE_LOCK_RETRY_INTERVAL = 20.0 +CREATE_LOCK_RETRY_MAX_INTERVAL = 60.0 + +#: Random extra added to each wait. Two xdist workers -- or the v1 nightly and +#: a concurrent v2 job -- that collide on the lock otherwise retry in step +#: forever, since they back off by the same amounts from the same moment. +CREATE_LOCK_RETRY_JITTER = 5.0 + + +def create_retrying(create: Any, *args: Any, **kwargs: Any) -> Any: + """ + Call a deployment creator, retrying a lock conflict. + + For a ``setUpClass`` that has to deploy before it can test anything:: + + cls.workspace_group = utils.create_retrying( + cls.manager.create_workspace_group, f'wg-test-{name}', ..., + ) + + The v1 management API refuses a ``create_workspace_group`` with "could not + acquire lock" when another creation in the same organization holds it, and + nothing retries that: ``Manager.RETRY_STATUSES`` covers 429 and 5xx only + (``management/manager.py``), so the error reaches ``setUpClass``, and an + exception there fails every test in the class. A whole management_v1 run + went that way. The lock clears on its own in seconds, so the fix is to wait + and ask again. + + Only a lock conflict is retried -- any other ``ManagementError`` is a real + failure and is re-raised immediately, as is the last lock error if the + budget runs out. + + Safe to retry with the same name because the conflict is a refusal to + start: nothing was created, so there is no deployment to collide with and + nothing for ``_recover_orphan`` to have found. A create that got far enough + to make something and *then* failed does not come back with this message, + and would surface on the retry as a name conflict rather than being + swallowed. + """ + import time + + for attempt in range(1, CREATE_LOCK_RETRY_ATTEMPTS + 1): + try: + return create(*args, **kwargs) + except ManagementError as exc: + if not LOCK_ERROR_RE.search(str(exc)): + raise + if attempt == CREATE_LOCK_RETRY_ATTEMPTS: + raise + wait = min( + CREATE_LOCK_RETRY_INTERVAL * attempt, + CREATE_LOCK_RETRY_MAX_INTERVAL, + ) + random.uniform(0, CREATE_LOCK_RETRY_JITTER) + logger.info( + f'{getattr(create, "__name__", create)} could not take the ' + f'lock ({exc}); attempt {attempt} of ' + f'{CREATE_LOCK_RETRY_ATTEMPTS}, retrying in {wait:.1f}s', + ) + time.sleep(wait) + + def _creator_is_mocked(target: Any) -> bool: """ Is this creation call going through a mocked manager? From 763b3003c72994cb002077d5a7c833a17fac648b Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Wed, 23 Sep 2026 14:50:03 -0400 Subject: [PATCH 23/30] Retry the cluster creations that cannot take the lock too [skip ci] The same "could not acquire lock" 500 comes back from POST /clusters -- the v2 shared cluster pool lost TestClusterFusion to it an hour after the v1 job's workspace groups went the same way. Note the server says "error creating workspace" for a /clusters POST, so the match cannot key on the noun; both observed wordings are now asserted verbatim. So create_retrying() now wraps every live creation, not just the v1 workspace groups: the pool build in utils.shared_clusters (where a raise fails every class that borrows from it), _ClusterFusionMixin, v2 TestCluster and TestStarterCluster, and v1's create_workspace and create_starter_workspace. Retrying at the call site rather than by adding POST to RETRY_METHODS is the point: POST is excluded there deliberately, since a retried POST can create twice, and the transport sees only a 500. The lock message is the evidence that this particular POST created nothing, which is what makes reusing the name safe. Co-Authored-By: Claude Opus 5 --- singlestoredb/tests/test_fusion.py | 6 +- singlestoredb/tests/test_management_utils.py | 67 +++++++++++++++++++- singlestoredb/tests/test_management_v1.py | 9 ++- singlestoredb/tests/test_management_v2.py | 9 ++- singlestoredb/tests/utils.py | 45 ++++++++----- 5 files changed, 113 insertions(+), 23 deletions(-) diff --git a/singlestoredb/tests/test_fusion.py b/singlestoredb/tests/test_fusion.py index 91cdbf070..7ee519cf4 100644 --- a/singlestoredb/tests/test_fusion.py +++ b/singlestoredb/tests/test_fusion.py @@ -1491,8 +1491,12 @@ def setUpClass(cls): for prefix in cls.fixture_prefixes: region = random.choice(cls.us_regions) + # Retried: POST /clusters comes back "could not acquire lock" when + # another creation in the organization holds it, and raising here + # fails every test in the class. See utils.create_retrying. cls.clusters.append( - mgr.create_cluster( + utils.create_retrying( + mgr.create_cluster, f'{prefix}-fusion-cluster-{cls.id}', region=region, size='S-00', diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index d842ca90d..f0f51688b 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -2038,6 +2038,30 @@ def test_the_waits_are_jittered(self): self.assertGreaterEqual(wait, base) self.assertLess(wait, base + 5.0) + def test_both_observed_wordings_match(self): + """Verbatim from the two runs that failed. The v2 one says "creating + workspace" for a ``/clusters`` POST, so the match cannot key on the + noun.""" + for msg in ( + 'error creating workspace group (wg-test-8jtfylajdmax-vast7ln): ' + 'could not acquire lock within duration [0, 2026/09/23 16:45:04]', + 'error creating workspace (cl-test-shared-1-b0dad293): could not ' + 'acquire lock within duration [0, 2026-09-23T17:37:16Z]', + ): + self.assertTrue( + self.utils.LOCK_ERROR_RE.search(msg), msg, + ) + + def test_an_unrelated_500_is_not_retried(self): + """500 is also what a name collision and a bad region come back as, so + the message is the only thing that says nothing was created.""" + create = self._creator( + 'error creating workspace group (wg-test-a): already exists', + ) + with self.assertRaises(ManagementError): + self.utils.create_retrying(create, 'wg-test-a') + self.assertEqual(len(create.calls), 1) + def test_the_message_match_does_not_need_a_status(self): """The status a lock conflict arrives as is not documented, and the status alone cannot tell a conflict from a rejection.""" @@ -2095,7 +2119,10 @@ def _restore(self): self.utils._tracked[:] = self.saved_tracked self.utils.set_owner(self.saved_owner) - def _manager(self, regions=('US East 1',), projects=('STANDARD',)): + def _manager( + self, regions=('US East 1',), projects=('STANDARD',), + lock_failures=0, + ): """ A stand-in cluster manager. @@ -2138,6 +2165,8 @@ def terminate(self, force=False): region_list = [Region(x) for x in regions] project_list = [Project(x) for x in projects] + refused = {} + class Manager: regions = region_list projects = project_list @@ -2146,6 +2175,20 @@ def create_cluster(self, name, **kwargs): # The owner in force at creation time is what decides whether # the per-class sweep eats the pool. created.append((name, utils.get_owner(), kwargs)) + if refused.get(name, 0) < lock_failures: + refused[name] = refused.get(name, 0) + 1 + # Verbatim from the run that failed, so the match is + # tested against the real wording rather than a paraphrase + # of it. Note "creating workspace" for a /clusters POST. + raise ManagementError( + errno=500, + msg=( + f'error creating workspace ({name}): could not ' + f'acquire lock within duration [0, ' + f'2026-09-23T17:37:16Z, 5e340578, ' + f'2026/09/23 17:37:16]' + ), + ) return utils.track(Cluster(name, kwargs)) return Manager() @@ -2162,6 +2205,28 @@ def test_the_pool_is_built_once(self): self.assertEqual([x.id for x in first], [x.id for x in second]) self.assertEqual(len(self.created), 2) + def test_a_lock_conflict_during_the_pool_build_is_retried(self): + """POST /clusters comes back "could not acquire lock" too, and a raise + here fails every class that borrows from the pool -- which is how + TestClusterFusion went down.""" + with patch('time.sleep'), self._patched(lock_failures=1): + pool = self.utils.shared_clusters(2) + + self.assertEqual(len(pool), 2) + # Two clusters, each refused once and then created. + self.assertEqual(len(self.created), 4) + self.assertEqual(len(self.utils._tracked), 2) + + def test_a_pool_build_that_keeps_losing_the_lock_still_raises(self): + with patch('time.sleep'), self._patched(lock_failures=100): + with self.assertRaises(ManagementError): + self.utils.shared_clusters(1) + + self.assertEqual( + len(self.created), self.utils.CREATE_LOCK_RETRY_ATTEMPTS, + ) + self.assertEqual(self.utils._tracked, []) + def test_the_pool_grows_to_the_largest_request(self): with self._patched(): one = self.utils.shared_clusters(1) diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index 8da4d1dfb..8c17a268f 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -88,7 +88,8 @@ def setUpClass(cls): try: # No expiry of its own: only the group has an expiresAt, and it # takes its workspaces with it. See utils.DEPLOYMENT_EXPIRES_AT. - cls.workspace = cls.workspace_group.create_workspace( + cls.workspace = utils.create_retrying( + cls.workspace_group.create_workspace, f'ws-test-{name}-x', wait_on_active=True, ) @@ -284,7 +285,8 @@ def setUpClass(cls): if not shared_tier_region: raise ValueError('No shared tier regions found') - cls.starter_workspace = cls.manager.create_starter_workspace( + cls.starter_workspace = utils.create_retrying( + cls.manager.create_starter_workspace, f'starter-ws-test-{name}', database_name=cls.database_name, provider=shared_tier_region.provider, @@ -1016,7 +1018,8 @@ def setUpClass(cls): try: # No expiry of its own: only the group has an expiresAt, and it # takes its workspaces with it. See utils.DEPLOYMENT_EXPIRES_AT. - cls.workspace = cls.workspace_group.create_workspace( + cls.workspace = utils.create_retrying( + cls.workspace_group.create_workspace, f'ws-test-{name}-x', wait_on_active=True, ) diff --git a/singlestoredb/tests/test_management_v2.py b/singlestoredb/tests/test_management_v2.py index fe65e47df..3a11c68f6 100644 --- a/singlestoredb/tests/test_management_v2.py +++ b/singlestoredb/tests/test_management_v2.py @@ -1327,7 +1327,11 @@ def setUpClass(cls): # v2 has no workspace group: the cluster is created in one call, with # the firewall settings passed alongside the compute settings. - cls.cluster = cls.manager.create_cluster( + # Retried: POST /clusters comes back "could not acquire lock" when + # another creation in the organization holds it, and raising here fails + # every test in the class. See utils.create_retrying. + cls.cluster = utils.create_retrying( + cls.manager.create_cluster, f'cl-test-{name}', region=region, size='S-00', @@ -1523,7 +1527,8 @@ def setUpClass(cls): region = random.choice(regions) - cls.starter_cluster = cls.manager.create_starter_cluster( + cls.starter_cluster = utils.create_retrying( + cls.manager.create_starter_cluster, f'starter-cl-test-{name}', database_name=cls.database_name, region=region, diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index f15a3e7a2..77dd37c6e 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -725,19 +725,22 @@ def terminate( #: Error text the management API comes back with when it will not start a #: creation because something else in the organization holds the lock it -#: needs -- "could not acquire lock". Matched on the message rather than on -#: ``errno`` because the status it arrives as is not something the API -#: documents, and because the status alone cannot distinguish a lock conflict -#: (retry and it goes through) from a rejected request (retry and it never -#: will). +#: needs -- "could not acquire lock within duration". Seen from both +#: ``POST /workspaceGroups`` and ``POST /clusters``, so it is not a v1 quirk. +#: +#: Matched on the message rather than on ``errno``: it arrives as a 500, and +#: 500 is also what an unrelated server-side failure arrives as, so the status +#: cannot tell a lock conflict (retry and it goes through) from something that +#: will fail again identically. The wording is also the only part that says +#: nothing was created -- which is what makes retrying the same name safe. LOCK_ERROR_RE = re.compile(r'acquire[^.]{0,40}lock', re.I) #: Retry budget for :func:`create_retrying`. Five waits at 20s growing to a #: 60s cap -- 20, 40, 60, 60, 60, so four minutes at worst. #: -#: Sized for a lock held by another creation rather than for provisioning: a -#: workspace group POST returns in seconds, so what is being waited out is the -#: other holder finishing, not a deployment coming up. Hence spacing much +#: Sized for a lock held by another creation rather than for provisioning: the +#: POST that takes the lock returns in seconds, so what is being waited out is +#: the other holder finishing, not a deployment coming up. Hence spacing much #: shorter than ``wait_on_active``'s but longer than a transport retry's, and a #: budget small enough that a genuinely stuck organization fails the class #: instead of idling out the job's timeout. @@ -761,13 +764,18 @@ def create_retrying(create: Any, *args: Any, **kwargs: Any) -> Any: cls.manager.create_workspace_group, f'wg-test-{name}', ..., ) - The v1 management API refuses a ``create_workspace_group`` with "could not - acquire lock" when another creation in the same organization holds it, and - nothing retries that: ``Manager.RETRY_STATUSES`` covers 429 and 5xx only - (``management/manager.py``), so the error reaches ``setUpClass``, and an - exception there fails every test in the class. A whole management_v1 run - went that way. The lock clears on its own in seconds, so the fix is to wait - and ask again. + The management API refuses a creation with "could not acquire lock" when + another creation in the same organization holds the lock -- v1's + ``POST /workspaceGroups`` and v2's ``POST /clusters`` alike -- and nothing + retries that. ``RETRY_METHODS`` is ``{GET, HEAD, OPTIONS, PUT, DELETE}`` + (``management/manager.py``), deliberately: a retried POST can create twice. + So the 500 reaches ``setUpClass``, and an exception there fails every test + in the class. A whole ``management_v1`` run went that way, and the v2 + shared cluster pool went the same way an hour later. + + Retrying here rather than by widening ``RETRY_METHODS`` is what keeps that + guarantee: the message is the evidence that this particular POST created + nothing, which the transport, seeing only a 500, does not have. Only a lock conflict is retried -- any other ``ManagementError`` is a real failure and is re-raised immediately, as is the last lock error if the @@ -1241,8 +1249,13 @@ def setUpClass(cls): set_owner('') try: while len(_pool) < count: + # Retried: POST /clusters comes back "could not acquire lock" when + # another creation in the organization holds it, and a raise here + # fails every class that borrows from the pool. See + # create_retrying. _pool.append( - mgr.create_cluster( + create_retrying( + mgr.create_cluster, f'cl-test-shared-{len(_pool)}-{_pool_id}', region=random.choice(us_regions), size='S-00', From c24243b7f067a1aef10ae3e0cf7d6a12b4f8974b Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 24 Sep 2026 10:24:34 -0400 Subject: [PATCH 24/30] Move the lock-conflict retry into the management API [skip ci] The two previous commits retried "could not acquire lock" from the test suite, wrapping each creation call in utils.create_retrying(). That only covers the call sites someone remembered to wrap, and nothing outside the tests -- a Fusion SQL CREATE CLUSTER, or any SDK user -- got it at all. So the wait now lives on the two methods that contend for the lock: manager.retry_on_lock, worn by v1 WorkspaceManager.create_workspace_group and v2 ClusterManager.create_cluster. Fusion SQL's CREATE WORKSPACE GROUP and CREATE CLUSTER go through those same two methods, so they are covered without a second implementation, and so is every setUpClass that used to need the wrapper. Deliberately those two and nothing else, rather than a hook in Manager._doit that would have covered every route and verb: only these two take the lock, and a retry on everything is a blocking wait bolted onto calls that should just fail. Which methods wear it is asserted in test_management_utils.py rather than left to a reader to notice. Budget and matching are unchanged from create_retrying: keyed on the error message, 20/40/60/60/60s with up to 5s of jitter, now tunable through SINGLESTOREDB_MANAGEMENT_LOCK_RETRIES and SINGLESTOREDB_MANAGEMENT_LOCK_RETRY_INTERVAL, with 0 retries restoring the old fail-fast behaviour. Replaying a POST is still safe for the same reason it was there: RETRY_METHODS excludes POST because the transport sees only a 500 and cannot know whether the server acted, whereas the lock message says the creation never started. The waits go through timing.sleep, so a traced run reports them as "create_cluster organization lock" instead of losing them to unaccounted time. create_retrying() and its ten call sites are gone, and its unit tests now drive the decorator. The two TestSharedClusterPool lock tests are dropped rather than moved: their stand-in manager raises ManagementError from a fake create_cluster and never reaches the decorated method, so there was nothing left for them to assert. Co-Authored-By: Claude Opus 5 --- singlestoredb/management/manager.py | 107 ++++++ singlestoredb/management/v1/workspace.py | 2 + singlestoredb/management/v2/cluster.py | 2 + singlestoredb/tests/test_fusion.py | 12 +- singlestoredb/tests/test_management_utils.py | 379 ++++++++++--------- singlestoredb/tests/test_management_v1.py | 27 +- singlestoredb/tests/test_management_v2.py | 9 +- singlestoredb/tests/utils.py | 94 +---- 8 files changed, 320 insertions(+), 312 deletions(-) diff --git a/singlestoredb/management/manager.py b/singlestoredb/management/manager.py index 37ba35708..e2ccbe221 100644 --- a/singlestoredb/management/manager.py +++ b/singlestoredb/management/manager.py @@ -1,9 +1,14 @@ #!/usr/bin/env python """SingleStoreDB Base Manager.""" +import functools +import logging import os +import random +import re import sys import time from typing import Any +from typing import Callable from typing import Dict from typing import List from typing import Optional @@ -23,6 +28,9 @@ from .utils import get_token +logger = logging.getLogger(__name__) + + def set_organization(kwargs: Dict[str, Any]) -> None: """Set the organization ID in the dictionary.""" if kwargs.get('params', {}).get('organizationID', None): @@ -42,6 +50,10 @@ def set_organization(kwargs: Dict[str, Any]) -> None: #: cover the failure mode that actually shows up -- a keep-alive connection #: the far end closed while the client was sleeping between polls, which #: surfaces as ``RemoteDisconnected`` on the next request. +#: +#: The one exception is a creation the organization lock blocked, which +#: :func:`retry_on_lock` replays: it is identified by the error message, which +#: this policy never sees. RETRY_METHODS = frozenset(['GET', 'HEAD', 'OPTIONS', 'PUT', 'DELETE']) #: Status codes worth retrying. These are the transient ones; a 4xx other @@ -75,6 +87,101 @@ def build_retry( ) +#: "could not acquire lock within duration", the API's refusal to start a +#: creation while another one in the organization holds the lock. Seen from +#: ``POST /workspaceGroups`` and ``POST /clusters``, whose message says "error +#: creating workspace" either way, so the match cannot key on the noun. +#: +#: Matched on the message, not the status: the conflict arrives as a 500, and so +#: does a name collision. The wording is also the only part that says nothing +#: was created, which is what makes replaying the POST safe. +LOCK_ERROR_RE = re.compile(r'acquire[^.]{0,40}lock', re.I) + +#: Ceiling on the wait between lock retries, and the random extra added to each +#: one. Capped because what is being waited out is another creation's POST +#: returning, not a deployment coming up. Jittered because two clients that +#: collide back off by the same amounts from the same moment -- two xdist +#: workers, say -- and would otherwise retry in step indefinitely. +LOCK_RETRY_MAX_INTERVAL = 60.0 +LOCK_RETRY_JITTER = 5.0 + + +def lock_retry_policy() -> Tuple[int, float]: + """ + Return the (retries, interval) applied to an organization lock conflict. + + ``retries`` counts attempts *after* the first, so the defaults wait 20, 40, + 60, 60 and 60 seconds -- four minutes at worst, small enough that a stuck + organization fails rather than idling out a CI job's timeout. + + Set ``SINGLESTOREDB_MANAGEMENT_LOCK_RETRIES=0`` to raise the conflict at + once instead. + """ + return ( + int(os.environ.get('SINGLESTOREDB_MANAGEMENT_LOCK_RETRIES', '5')), + float( + os.environ.get('SINGLESTOREDB_MANAGEMENT_LOCK_RETRY_INTERVAL', '20'), + ), + ) + + +def is_lock_error(exc: BaseException) -> bool: + """Is this error the organization refusing to take the lock?""" + return bool(LOCK_ERROR_RE.search(str(exc))) + + +def lock_retry_wait(attempt: int, interval: float) -> float: + """Seconds to wait before replaying a creation that lost the lock.""" + return min(interval * attempt, LOCK_RETRY_MAX_INTERVAL) + \ + random.uniform(0, LOCK_RETRY_JITTER) + + +def retry_on_lock(func: Callable[..., Any]) -> Callable[..., Any]: + """ + Wait out an organization lock conflict on a deployment creation. + + Replaying a POST is safe here where widening :data:`RETRY_METHODS` would not + be: the transport sees only a 500 and cannot know whether the server acted, + whereas the lock message says the creation never started. A creation that + made something and *then* failed does not come back with this message, and + would surface on the replay as a name conflict rather than being swallowed. + + Worn by ``WorkspaceManager.create_workspace_group`` and + ``ClusterManager.create_cluster`` only -- the two calls that contend for the + lock. Fusion SQL's ``CREATE WORKSPACE GROUP`` and ``CREATE CLUSTER`` go + through them, so they are covered too. + + Any other ``ManagementError`` is raised at once, as is the conflict itself + once :func:`lock_retry_policy`'s budget runs out. + """ + @functools.wraps(func) + def wrapper(self: Any, *args: Any, **kwargs: Any) -> Any: + retries, interval = lock_retry_policy() + attempt = 0 + while True: + try: + return func(self, *args, **kwargs) + except ManagementError as exc: + attempt += 1 + if attempt > retries or not is_lock_error(exc): + raise + wait = lock_retry_wait(attempt, interval) + logger.info( + f'{func.__name__} could not take the organization lock ' + f'({exc}); attempt {attempt} of {retries + 1}, retrying ' + f'in {wait:.1f}s', + ) + timing.sleep(wait, f'{func.__name__} organization lock') + + # Says which methods wear this, for a test to assert against. On the + # wrapper's ``__dict__``, so ``functools.wraps`` carries it outward through + # any later decorator -- the test suite wraps these methods again to track + # what a run has deployed. + wrapper.__retry_on_lock__ = True # type: ignore[attr-defined] + + return wrapper + + def default_timeout() -> Tuple[float, float]: """ Return the (connect, read) timeout applied when a caller gives none. diff --git a/singlestoredb/management/v1/workspace.py b/singlestoredb/management/v1/workspace.py index de5938dac..f7a72aa91 100644 --- a/singlestoredb/management/v1/workspace.py +++ b/singlestoredb/management/v1/workspace.py @@ -47,6 +47,7 @@ from ...exceptions import ManagementError from ..billing import Billing as Billing from ..manager import Manager +from ..manager import retry_on_lock from ..region import Region from ..stage import StageObject as StageObject from ..utils import camel_to_snake_dict @@ -1275,6 +1276,7 @@ def shared_tier_regions(self) -> NamedList[Region]: [Region.from_dict(item, self) for item in res.json()], ) + @retry_on_lock def create_workspace_group( self, name: str, diff --git a/singlestoredb/management/v2/cluster.py b/singlestoredb/management/v2/cluster.py index df51765fa..8a38c64a2 100644 --- a/singlestoredb/management/v2/cluster.py +++ b/singlestoredb/management/v2/cluster.py @@ -24,6 +24,7 @@ from ...exceptions import ManagementError from ..billing import Billing as Billing from ..manager import Manager +from ..manager import retry_on_lock from ..organization import Organization from ..organization import Organizations as Organizations from ..region import Region @@ -1409,6 +1410,7 @@ def _resolve_project_id( ', '.join(f'{x.name} ({x.id})' for x in projects) + '.', ) + @retry_on_lock def create_cluster( self, name: str, diff --git a/singlestoredb/tests/test_fusion.py b/singlestoredb/tests/test_fusion.py index 7ee519cf4..f130e4673 100644 --- a/singlestoredb/tests/test_fusion.py +++ b/singlestoredb/tests/test_fusion.py @@ -970,12 +970,8 @@ def setUpClass(cls): # and creation in some non-US regions fails with a control-plane 500. us_regions = [x for x in mgr.regions if x.name.startswith('US')] for letter in ('A', 'B', 'C'): - # Retried: three creations in a row in one organization is exactly - # what comes back "could not acquire lock", and raising here fails - # every test in the class. See utils.create_retrying. cls.workspace_groups.append( - utils.create_retrying( - mgr.create_workspace_group, + mgr.create_workspace_group( f'{letter} Fusion Testing {cls.id}', region=random.choice(us_regions), firewall_ranges=[], @@ -1491,12 +1487,8 @@ def setUpClass(cls): for prefix in cls.fixture_prefixes: region = random.choice(cls.us_regions) - # Retried: POST /clusters comes back "could not acquire lock" when - # another creation in the organization holds it, and raising here - # fails every test in the class. See utils.create_retrying. cls.clusters.append( - utils.create_retrying( - mgr.create_cluster, + mgr.create_cluster( f'{prefix}-fusion-cluster-{cls.id}', region=region, size='S-00', diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index f0f51688b..731734c5c 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -875,6 +875,200 @@ def test_a_transport_failure_names_the_route(self): self.assertIn('clusters/abc', msg) +class TestLockRetry(unittest.TestCase): + """ + ``manager.retry_on_lock``, the wait for a creation the organization lock + blocks. + + The API refuses the creation with "could not acquire lock" while another one + in the organization holds it, and nothing else retries that: POST is out of + ``RETRY_METHODS``. In a ``setUpClass`` one conflict fails every test in the + class -- a whole ``management_v1`` run went that way, and the v2 shared + cluster pool went the same way an hour later. + """ + + #: Verbatim from the two runs that failed. The ``/clusters`` one says + #: "error creating workspace", so the match cannot key on the noun. + LOCK_MESSAGES = ( + 'error creating workspace group (wg-test-8jtfylajdmax-vast7ln): ' + 'could not acquire lock within duration [0, 2026/09/23 16:45:04]', + 'error creating workspace (cl-test-shared-1-b0dad293): could not ' + 'acquire lock within duration [0, 2026-09-23T17:37:16Z]', + ) + + #: The default budget, per lock_retry_policy. + WAITS = [20.0, 40.0, 60.0, 60.0, 60.0] + + def setUp(self): + from singlestoredb.management import manager as manager_mod + self.manager_mod = manager_mod + self.slept = [] + + # The waits go through timing.sleep, so a trace reports them as waits. + patcher = patch( + 'singlestoredb.management.timing.time.sleep', self.slept.append, + ) + patcher.start() + self.addCleanup(patcher.stop) + + # Jitter off, so the waits asserted below are the spacing itself. + patcher = patch.object(manager_mod, 'LOCK_RETRY_JITTER', 0.0) + patcher.start() + self.addCleanup(patcher.stop) + + def _creator(self, *msgs, errno=500): + """A decorated creator that fails with these messages, then succeeds.""" + calls = [] + + class Mgr: + @self.manager_mod.retry_on_lock + def create(self, name, **kwargs): + calls.append(name) + if len(calls) <= len(msgs): + raise ManagementError(errno=errno, msg=msgs[len(calls) - 1]) + return f'deployment {name}' + + mgr = Mgr() + mgr.calls = calls + return mgr + + def test_a_lock_conflict_is_retried_until_it_succeeds(self): + mgr = self._creator(*self.LOCK_MESSAGES) + self.assertEqual(mgr.create('wg-test-a', region='x'), 'deployment wg-test-a') + self.assertEqual(len(mgr.calls), 3) + self.assertEqual(self.slept, [20.0, 40.0]) + + def test_the_retry_reuses_the_name(self): + """The conflict is a refusal to start, so nothing was created and there + is no deployment for the next attempt to collide with.""" + mgr = self._creator(self.LOCK_MESSAGES[0]) + mgr.create('wg-test-a') + self.assertEqual(mgr.calls, ['wg-test-a', 'wg-test-a']) + + def test_another_error_is_not_retried(self): + """A rejected request is a real failure; retrying only delays it.""" + mgr = self._creator('region is not available') + with self.assertRaises(ManagementError): + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 1) + self.assertEqual(self.slept, []) + + def test_an_unrelated_500_is_not_retried(self): + """500 is also what a name collision comes back as, so the message is + the only thing that says nothing was created.""" + mgr = self._creator('error creating workspace group (x): already exists') + with self.assertRaises(ManagementError): + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 1) + + def test_the_budget_is_bounded_and_the_error_is_raised(self): + """A stuck organization has to fail rather than idle out the job.""" + mgr = self._creator(*([self.LOCK_MESSAGES[0]] * 100)) + with self.assertRaises(ManagementError) as cm: + mgr.create('wg-test-a') + self.assertIn('acquire lock', str(cm.exception)) + self.assertEqual(len(mgr.calls), len(self.WAITS) + 1) + self.assertEqual(len(self.slept), len(self.WAITS)) + + def test_the_wait_is_capped(self): + mgr = self._creator(*([self.LOCK_MESSAGES[0]] * 100)) + with self.assertRaises(ManagementError): + mgr.create('wg-test-a') + self.assertLessEqual( + max(self.slept), self.manager_mod.LOCK_RETRY_MAX_INTERVAL, + ) + self.assertEqual(self.slept, self.WAITS) + + def test_the_waits_are_jittered(self): + """Two clients that collide back off by the same amounts from the same + moment, so without jitter they retry in step forever.""" + patcher = patch.object(self.manager_mod, 'LOCK_RETRY_JITTER', 5.0) + patcher.start() + self.addCleanup(patcher.stop) + + mgr = self._creator(*([self.LOCK_MESSAGES[0]] * 100)) + with self.assertRaises(ManagementError): + mgr.create('wg-test-a') + self.assertNotEqual(self.slept, self.WAITS) + for wait, base in zip(self.slept, self.WAITS): + self.assertGreaterEqual(wait, base) + self.assertLess(wait, base + 5.0) + + def test_both_observed_wordings_match(self): + for msg in self.LOCK_MESSAGES: + mgr = self._creator(msg) + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 2, msg) + + def test_the_match_does_not_depend_on_the_status(self): + """The status a conflict arrives as is not documented, and it cannot + tell a conflict from a rejection either way.""" + mgr = self._creator(self.LOCK_MESSAGES[0], errno=409) + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 2) + + def test_a_success_is_not_delayed(self): + mgr = self._creator() + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 1) + self.assertEqual(self.slept, []) + + def test_the_retry_can_be_turned_off(self): + mgr = self._creator(self.LOCK_MESSAGES[0]) + with patch.dict( + os.environ, {'SINGLESTOREDB_MANAGEMENT_LOCK_RETRIES': '0'}, + ): + with self.assertRaises(ManagementError): + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 1) + self.assertEqual(self.slept, []) + + def test_the_budget_is_configurable(self): + mgr = self._creator(*([self.LOCK_MESSAGES[0]] * 100)) + with patch.dict( + os.environ, { + 'SINGLESTOREDB_MANAGEMENT_LOCK_RETRIES': '2', + 'SINGLESTOREDB_MANAGEMENT_LOCK_RETRY_INTERVAL': '3', + }, + ): + with self.assertRaises(ManagementError): + mgr.create('wg-test-a') + self.assertEqual(len(mgr.calls), 3) + self.assertEqual(self.slept, [3.0, 6.0]) + + +class TestLockRetryIsApplied(unittest.TestCase): + """Which creations wear ``retry_on_lock``: the two that take the lock.""" + + def _wears_it(self, method): + return getattr(method, '__retry_on_lock__', False) + + def test_the_two_creations_that_take_the_lock(self): + from singlestoredb.management.v1.workspace import WorkspaceManager + from singlestoredb.management.v2.cluster import ClusterManager + + self.assertTrue(self._wears_it(WorkspaceManager.create_workspace_group)) + self.assertTrue(self._wears_it(ClusterManager.create_cluster)) + + def test_and_nothing_else(self): + """Deliberately narrow: a workspace inside an existing group and the + starter deployments do not contend for this lock.""" + from singlestoredb.management.v1.workspace import WorkspaceGroup + from singlestoredb.management.v1.workspace import WorkspaceManager + from singlestoredb.management.v2.cluster import ClusterManager + + for klass, name in ( + (WorkspaceManager, 'create_workspace'), + (WorkspaceManager, 'create_starter_workspace'), + (WorkspaceGroup, 'create_workspace'), + (ClusterManager, 'create_starter_cluster'), + ): + self.assertFalse( + self._wears_it(getattr(klass, name)), + f'{klass.__name__}.{name}', + ) + + class TestWaitOnEndpoint(unittest.TestCase): """ ``Manager._wait_on_endpoint`` polls a new deployment by connecting to it. @@ -1937,146 +2131,6 @@ def terminate(inner, force=False): self.assertEqual(calls, [True]) -class TestCreateRetryingLockConflicts(unittest.TestCase): - """ - ``utils.create_retrying()``'s retry for a creation the API will not start - because it cannot take the lock. - - A ``create_workspace_group`` that comes back "could not acquire lock" is - not retried by the transport -- Manager.RETRY_STATUSES is 429 and 5xx -- - and it happens in ``setUpClass``, so one lock conflict fails every test in - the class. A whole management_v1 run went that way. - """ - - def setUp(self): - from singlestoredb.tests import utils - self.utils = utils - self.slept = [] - - patcher = patch('time.sleep', self.slept.append) - patcher.start() - self.addCleanup(patcher.stop) - - # Jitter off, so the waits asserted below are the spacing itself. - # It is covered separately. - patcher = patch.object(utils, 'CREATE_LOCK_RETRY_JITTER', 0.0) - patcher.start() - self.addCleanup(patcher.stop) - - def _creator(self, *msgs): - """A creator that fails with these messages in turn, then succeeds.""" - calls = [] - - def create(name, **kwargs): - calls.append(name) - if len(calls) <= len(msgs): - raise ManagementError(errno=400, msg=msgs[len(calls) - 1]) - return f'deployment {name}' - - create.calls = calls - return create - - def test_a_lock_conflict_is_retried_until_it_succeeds(self): - create = self._creator( - 'could not acquire lock', 'Could not acquire lock on workspace', - ) - out = self.utils.create_retrying(create, 'wg-test-a', region='x') - self.assertEqual(out, 'deployment wg-test-a') - self.assertEqual(len(create.calls), 3) - self.assertEqual(self.slept, [20.0, 40.0]) - - def test_the_retry_reuses_the_name(self): - """The conflict is a refusal to start, so nothing was created and - there is no deployment for the second attempt to collide with.""" - create = self._creator('could not acquire lock') - self.utils.create_retrying(create, 'wg-test-a') - self.assertEqual(create.calls, ['wg-test-a', 'wg-test-a']) - - def test_another_error_is_not_retried(self): - """A rejected request is a real failure; retrying it only delays the - report by four minutes.""" - create = self._creator('region is not available') - with self.assertRaises(ManagementError): - self.utils.create_retrying(create, 'wg-test-a') - self.assertEqual(len(create.calls), 1) - self.assertEqual(self.slept, []) - - def test_the_budget_is_bounded_and_the_error_is_re_raised(self): - """A genuinely stuck organization has to fail the class rather than - idle out the job's timeout.""" - create = self._creator(*(['could not acquire lock'] * 100)) - with self.assertRaises(ManagementError): - self.utils.create_retrying(create, 'wg-test-a') - self.assertEqual( - len(create.calls), self.utils.CREATE_LOCK_RETRY_ATTEMPTS, - ) - self.assertEqual(len(self.slept), 6 - 1) - - def test_the_wait_is_capped(self): - """Growth stops at the cap: what is being waited out is another - creation finishing, not a deployment coming up.""" - create = self._creator(*(['could not acquire lock'] * 100)) - with self.assertRaises(ManagementError): - self.utils.create_retrying(create, 'wg-test-a') - self.assertLessEqual( - max(self.slept), self.utils.CREATE_LOCK_RETRY_MAX_INTERVAL, - ) - self.assertEqual(self.slept, [20.0, 40.0, 60.0, 60.0, 60.0]) - - def test_the_waits_are_jittered(self): - """Two workers that collide on the lock back off by the same amounts - from the same moment, so without jitter they retry in step forever.""" - patcher = patch.object(self.utils, 'CREATE_LOCK_RETRY_JITTER', 5.0) - patcher.start() - self.addCleanup(patcher.stop) - - create = self._creator(*(['could not acquire lock'] * 100)) - with self.assertRaises(ManagementError): - self.utils.create_retrying(create, 'wg-test-a') - self.assertNotEqual(self.slept, [20.0, 40.0, 60.0, 60.0, 60.0]) - for wait, base in zip(self.slept, [20.0, 40.0, 60.0, 60.0, 60.0]): - self.assertGreaterEqual(wait, base) - self.assertLess(wait, base + 5.0) - - def test_both_observed_wordings_match(self): - """Verbatim from the two runs that failed. The v2 one says "creating - workspace" for a ``/clusters`` POST, so the match cannot key on the - noun.""" - for msg in ( - 'error creating workspace group (wg-test-8jtfylajdmax-vast7ln): ' - 'could not acquire lock within duration [0, 2026/09/23 16:45:04]', - 'error creating workspace (cl-test-shared-1-b0dad293): could not ' - 'acquire lock within duration [0, 2026-09-23T17:37:16Z]', - ): - self.assertTrue( - self.utils.LOCK_ERROR_RE.search(msg), msg, - ) - - def test_an_unrelated_500_is_not_retried(self): - """500 is also what a name collision and a bad region come back as, so - the message is the only thing that says nothing was created.""" - create = self._creator( - 'error creating workspace group (wg-test-a): already exists', - ) - with self.assertRaises(ManagementError): - self.utils.create_retrying(create, 'wg-test-a') - self.assertEqual(len(create.calls), 1) - - def test_the_message_match_does_not_need_a_status(self): - """The status a lock conflict arrives as is not documented, and the - status alone cannot tell a conflict from a rejection.""" - calls = [] - - def create(name): - calls.append(name) - if len(calls) == 1: - raise ManagementError(msg='Could not acquire lock') - return name - - self.utils.create_retrying(create, 'wg-test-a') - self.assertEqual(len(calls), 2) - - class TestSharedClusterPool(unittest.TestCase): """ The pool in ``tests/utils.py`` that keeps the Stage and Job suites from @@ -2119,10 +2173,7 @@ def _restore(self): self.utils._tracked[:] = self.saved_tracked self.utils.set_owner(self.saved_owner) - def _manager( - self, regions=('US East 1',), projects=('STANDARD',), - lock_failures=0, - ): + def _manager(self, regions=('US East 1',), projects=('STANDARD',)): """ A stand-in cluster manager. @@ -2165,8 +2216,6 @@ def terminate(self, force=False): region_list = [Region(x) for x in regions] project_list = [Project(x) for x in projects] - refused = {} - class Manager: regions = region_list projects = project_list @@ -2175,20 +2224,6 @@ def create_cluster(self, name, **kwargs): # The owner in force at creation time is what decides whether # the per-class sweep eats the pool. created.append((name, utils.get_owner(), kwargs)) - if refused.get(name, 0) < lock_failures: - refused[name] = refused.get(name, 0) + 1 - # Verbatim from the run that failed, so the match is - # tested against the real wording rather than a paraphrase - # of it. Note "creating workspace" for a /clusters POST. - raise ManagementError( - errno=500, - msg=( - f'error creating workspace ({name}): could not ' - f'acquire lock within duration [0, ' - f'2026-09-23T17:37:16Z, 5e340578, ' - f'2026/09/23 17:37:16]' - ), - ) return utils.track(Cluster(name, kwargs)) return Manager() @@ -2205,27 +2240,9 @@ def test_the_pool_is_built_once(self): self.assertEqual([x.id for x in first], [x.id for x in second]) self.assertEqual(len(self.created), 2) - def test_a_lock_conflict_during_the_pool_build_is_retried(self): - """POST /clusters comes back "could not acquire lock" too, and a raise - here fails every class that borrows from the pool -- which is how - TestClusterFusion went down.""" - with patch('time.sleep'), self._patched(lock_failures=1): - pool = self.utils.shared_clusters(2) - - self.assertEqual(len(pool), 2) - # Two clusters, each refused once and then created. - self.assertEqual(len(self.created), 4) - self.assertEqual(len(self.utils._tracked), 2) - - def test_a_pool_build_that_keeps_losing_the_lock_still_raises(self): - with patch('time.sleep'), self._patched(lock_failures=100): - with self.assertRaises(ManagementError): - self.utils.shared_clusters(1) - - self.assertEqual( - len(self.created), self.utils.CREATE_LOCK_RETRY_ATTEMPTS, - ) - self.assertEqual(self.utils._tracked, []) + # The pool build's lock conflict is waited out below this, in + # Manager._doit, which a stand-in manager does not go through: see + # TestManagerLockRetry. def test_the_pool_grows_to_the_largest_request(self): with self._patched(): diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index 8c17a268f..63e7b9dd0 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -73,11 +73,7 @@ def setUpClass(cls): name = clean_name(secrets.token_urlsafe(20)[:20]) - # Retried: another creation in the organization holding the lock makes - # this come back "could not acquire lock", and raising here fails every - # test in the class. See utils.create_retrying. - cls.workspace_group = utils.create_retrying( - cls.manager.create_workspace_group, + cls.workspace_group = cls.manager.create_workspace_group( f'wg-test-{name}', region=random.choice(us_regions).id, admin_password=cls.password, @@ -88,8 +84,7 @@ def setUpClass(cls): try: # No expiry of its own: only the group has an expiresAt, and it # takes its workspaces with it. See utils.DEPLOYMENT_EXPIRES_AT. - cls.workspace = utils.create_retrying( - cls.workspace_group.create_workspace, + cls.workspace = cls.workspace_group.create_workspace( f'ws-test-{name}-x', wait_on_active=True, ) @@ -285,8 +280,7 @@ def setUpClass(cls): if not shared_tier_region: raise ValueError('No shared tier regions found') - cls.starter_workspace = utils.create_retrying( - cls.manager.create_starter_workspace, + cls.starter_workspace = cls.manager.create_starter_workspace( f'starter-ws-test-{name}', database_name=cls.database_name, provider=shared_tier_region.provider, @@ -387,11 +381,7 @@ def setUpClass(cls): name = clean_name(secrets.token_urlsafe(20)[:20]) - # Retried: another creation in the organization holding the lock makes - # this come back "could not acquire lock", and raising here fails every - # test in the class. See utils.create_retrying. - cls.wg = utils.create_retrying( - cls.manager.create_workspace_group, + cls.wg = cls.manager.create_workspace_group( f'wg-test-{name}', region=random.choice(us_regions).id, admin_password=cls.password, @@ -1003,11 +993,7 @@ def setUpClass(cls): name = clean_name(secrets.token_urlsafe(20)[:20]) - # Retried: another creation in the organization holding the lock makes - # this come back "could not acquire lock", and raising here fails every - # test in the class. See utils.create_retrying. - cls.workspace_group = utils.create_retrying( - cls.manager.create_workspace_group, + cls.workspace_group = cls.manager.create_workspace_group( f'wg-test-{name}', region=random.choice(us_regions).id, admin_password=cls.password, @@ -1018,8 +1004,7 @@ def setUpClass(cls): try: # No expiry of its own: only the group has an expiresAt, and it # takes its workspaces with it. See utils.DEPLOYMENT_EXPIRES_AT. - cls.workspace = utils.create_retrying( - cls.workspace_group.create_workspace, + cls.workspace = cls.workspace_group.create_workspace( f'ws-test-{name}-x', wait_on_active=True, ) diff --git a/singlestoredb/tests/test_management_v2.py b/singlestoredb/tests/test_management_v2.py index 3a11c68f6..fe65e47df 100644 --- a/singlestoredb/tests/test_management_v2.py +++ b/singlestoredb/tests/test_management_v2.py @@ -1327,11 +1327,7 @@ def setUpClass(cls): # v2 has no workspace group: the cluster is created in one call, with # the firewall settings passed alongside the compute settings. - # Retried: POST /clusters comes back "could not acquire lock" when - # another creation in the organization holds it, and raising here fails - # every test in the class. See utils.create_retrying. - cls.cluster = utils.create_retrying( - cls.manager.create_cluster, + cls.cluster = cls.manager.create_cluster( f'cl-test-{name}', region=region, size='S-00', @@ -1527,8 +1523,7 @@ def setUpClass(cls): region = random.choice(regions) - cls.starter_cluster = utils.create_retrying( - cls.manager.create_starter_cluster, + cls.starter_cluster = cls.manager.create_starter_cluster( f'starter-cl-test-{name}', database_name=cls.database_name, region=region, diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index 77dd37c6e..09f8ba3c2 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -723,93 +723,6 @@ def terminate( time.sleep(interval) -#: Error text the management API comes back with when it will not start a -#: creation because something else in the organization holds the lock it -#: needs -- "could not acquire lock within duration". Seen from both -#: ``POST /workspaceGroups`` and ``POST /clusters``, so it is not a v1 quirk. -#: -#: Matched on the message rather than on ``errno``: it arrives as a 500, and -#: 500 is also what an unrelated server-side failure arrives as, so the status -#: cannot tell a lock conflict (retry and it goes through) from something that -#: will fail again identically. The wording is also the only part that says -#: nothing was created -- which is what makes retrying the same name safe. -LOCK_ERROR_RE = re.compile(r'acquire[^.]{0,40}lock', re.I) - -#: Retry budget for :func:`create_retrying`. Five waits at 20s growing to a -#: 60s cap -- 20, 40, 60, 60, 60, so four minutes at worst. -#: -#: Sized for a lock held by another creation rather than for provisioning: the -#: POST that takes the lock returns in seconds, so what is being waited out is -#: the other holder finishing, not a deployment coming up. Hence spacing much -#: shorter than ``wait_on_active``'s but longer than a transport retry's, and a -#: budget small enough that a genuinely stuck organization fails the class -#: instead of idling out the job's timeout. -CREATE_LOCK_RETRY_ATTEMPTS = 6 -CREATE_LOCK_RETRY_INTERVAL = 20.0 -CREATE_LOCK_RETRY_MAX_INTERVAL = 60.0 - -#: Random extra added to each wait. Two xdist workers -- or the v1 nightly and -#: a concurrent v2 job -- that collide on the lock otherwise retry in step -#: forever, since they back off by the same amounts from the same moment. -CREATE_LOCK_RETRY_JITTER = 5.0 - - -def create_retrying(create: Any, *args: Any, **kwargs: Any) -> Any: - """ - Call a deployment creator, retrying a lock conflict. - - For a ``setUpClass`` that has to deploy before it can test anything:: - - cls.workspace_group = utils.create_retrying( - cls.manager.create_workspace_group, f'wg-test-{name}', ..., - ) - - The management API refuses a creation with "could not acquire lock" when - another creation in the same organization holds the lock -- v1's - ``POST /workspaceGroups`` and v2's ``POST /clusters`` alike -- and nothing - retries that. ``RETRY_METHODS`` is ``{GET, HEAD, OPTIONS, PUT, DELETE}`` - (``management/manager.py``), deliberately: a retried POST can create twice. - So the 500 reaches ``setUpClass``, and an exception there fails every test - in the class. A whole ``management_v1`` run went that way, and the v2 - shared cluster pool went the same way an hour later. - - Retrying here rather than by widening ``RETRY_METHODS`` is what keeps that - guarantee: the message is the evidence that this particular POST created - nothing, which the transport, seeing only a 500, does not have. - - Only a lock conflict is retried -- any other ``ManagementError`` is a real - failure and is re-raised immediately, as is the last lock error if the - budget runs out. - - Safe to retry with the same name because the conflict is a refusal to - start: nothing was created, so there is no deployment to collide with and - nothing for ``_recover_orphan`` to have found. A create that got far enough - to make something and *then* failed does not come back with this message, - and would surface on the retry as a name conflict rather than being - swallowed. - """ - import time - - for attempt in range(1, CREATE_LOCK_RETRY_ATTEMPTS + 1): - try: - return create(*args, **kwargs) - except ManagementError as exc: - if not LOCK_ERROR_RE.search(str(exc)): - raise - if attempt == CREATE_LOCK_RETRY_ATTEMPTS: - raise - wait = min( - CREATE_LOCK_RETRY_INTERVAL * attempt, - CREATE_LOCK_RETRY_MAX_INTERVAL, - ) + random.uniform(0, CREATE_LOCK_RETRY_JITTER) - logger.info( - f'{getattr(create, "__name__", create)} could not take the ' - f'lock ({exc}); attempt {attempt} of ' - f'{CREATE_LOCK_RETRY_ATTEMPTS}, retrying in {wait:.1f}s', - ) - time.sleep(wait) - - def _creator_is_mocked(target: Any) -> bool: """ Is this creation call going through a mocked manager? @@ -1249,13 +1162,8 @@ def setUpClass(cls): set_owner('') try: while len(_pool) < count: - # Retried: POST /clusters comes back "could not acquire lock" when - # another creation in the organization holds it, and a raise here - # fails every class that borrows from the pool. See - # create_retrying. _pool.append( - create_retrying( - mgr.create_cluster, + mgr.create_cluster( f'cl-test-shared-{len(_pool)}-{_pool_id}', region=random.choice(us_regions), size='S-00', From bdcfa34dbb9908da233458b338fee043533d58cc Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Thu, 24 Sep 2026 15:32:52 -0400 Subject: [PATCH 25/30] Tighten the comments, and fix three that were wrong The prose blocks this branch added had grown past the point where anyone would read them. Condensed them across the management API, the test helpers and the workflows, keeping the load-bearing facts -- the run IDs, the 5-minute cancellation timeout, the ~460s ACTIVE floor, the audit item references -- and dropping the restatements. Three substantive fixes along the way: * the shared-pool comment in test_management_utils.py named a class that does not exist (TestLockRetry, not TestManagerLockRetry) and put the lock retry in Manager._doit, where it is a @retry_on_lock decorator on create_cluster. Both wrong since c24243b7. * _run_ledger_sweep returned 0 from its dry run even with unresolved records, while the no-leftovers path returned 1 for that same condition. An unresolved record means a deployment that may still be live, so a dry run should say so in its exit status too. * create_test_cluster.py's --password help pointed at a comment with "(see below)", which means nothing in --help output. Also dropped a dead `live = []` in test_fusion.py, whose `finally` always assigns it. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 34 +- .github/workflows/coverage.yml | 66 ++-- .github/workflows/publish.yml | 23 +- .github/workflows/smoke-test.yml | 32 +- resources/create_test_cluster.py | 53 ++- singlestoredb/management/manager.py | 67 ++-- singlestoredb/management/utils.py | 52 ++- singlestoredb/management/v1/workspace.py | 7 +- singlestoredb/management/v2/cluster.py | 18 +- singlestoredb/tests/cleanup_deployments.py | 133 +++---- singlestoredb/tests/conftest.py | 42 +-- singlestoredb/tests/test_fusion.py | 114 +++--- singlestoredb/tests/test_management_utils.py | 6 +- singlestoredb/tests/test_management_v1.py | 37 +- singlestoredb/tests/utils.py | 354 ++++++++----------- 15 files changed, 429 insertions(+), 609 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index c18c49489..bbeb266f3 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -8,14 +8,12 @@ on: workflow_dispatch: -# No `concurrency` block with `cancel-in-progress`, deliberately. Cancelling the -# run a push supersedes would halve the clusters two overlapping runs carry, but -# it cannot be made safe: GitHub force-terminates a cancelled job's remaining -# steps after a 5-minute cancellation timeout, `if: always()` included, and an -# S-00 cluster refuses DELETE until it is ACTIVE (~460s). A run cancelled inside -# its first several minutes would die with a cluster it is not yet allowed to -# delete, and nothing on a schedule would come along to reap it. Letting both -# runs finish costs clusters; cancelling them costs stranded clusters. +# No `concurrency`/`cancel-in-progress`, deliberately. GitHub force-terminates a +# cancelled job's remaining steps -- `if: always()` included -- after a 5-minute +# cancellation timeout, and an S-00 cluster refuses DELETE until it is ACTIVE +# (~460s). A run cancelled in its first few minutes would die holding a cluster +# it is not yet allowed to delete, with nothing scheduled to reap it. Letting +# both runs finish costs clusters; cancelling them costs stranded clusters. jobs: test-coverage: runs-on: ubuntu-latest @@ -46,10 +44,10 @@ jobs: uses: actions/checkout@v7 with: # Full history, because the change detector below diffs against - # origin/main. A shallow clone does not create that ref -- with + # origin/main, a ref a shallow clone does not create. With # fetch-depth: 2 every diff died on `fatal: bad revision - # 'origin/main'`, which the detector read as "nothing changed", so - # the management step never ran on a PR. + # 'origin/main'`, which the detector read as "nothing changed", so the + # management step never ran on a PR. fetch-depth: 0 - name: Set up Python @@ -228,15 +226,13 @@ jobs: # is cancelled, which is what left three clusters billing in run # 35631802648 (see the matching step in coverage.yml). On a PR the # management step above only runs when the change detector fires, so most - # runs reach this with an empty ledger and it reports nothing. + # runs reach this with an empty ledger and report nothing. # - # always() is not a guarantee, only a best effort: a cancelled job's - # remaining steps are force-terminated after GitHub's 5-minute - # cancellation timeout, so this covers a cancel whose clusters are already - # ACTIVE -- run 35631802648 was cancelled 19 minutes in -- but not one in - # the first several minutes, where DELETE is still being refused when the - # step is killed. That remainder needs `cleanup_deployments.py - # --older-than` run by hand; nothing here is on a schedule. + # Best effort, not a guarantee: a cancelled job's remaining steps are + # force-terminated after GitHub's 5-minute cancellation timeout, so this + # covers a cancel whose clusters are already ACTIVE but not one in the + # first few minutes, where DELETE is still refused. That remainder needs + # `cleanup_deployments.py --older-than` run by hand. - name: Terminate any deployment the tests left behind if: always() run: | diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index aaf9d7571..0f6e932f6 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -11,9 +11,8 @@ jobs: environment: Base # One ledger for the whole job, so the cleanup step below can reap what any - # of the pytest steps created. Per job rather than shared: each job gets its - # own runner and workspace anyway, and keeping the ledgers separate means a - # job's sweep can only ever reach records it wrote itself. + # of the pytest steps created. Per job, not shared: a job's sweep can then + # only reach records it wrote itself. env: SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl @@ -94,24 +93,21 @@ jobs: coverage html # if: always() is the whole point -- this has to run when the job is - # cancelled, which is the case that produced the leak. Run 35631802648 - # was cancelled 19 minutes into TestClusterFusion.setUpClass's - # create_cluster(wait_on_active=True, wait_timeout=1200); the log ends at + # cancelled, the case that produced the leak. Run 35631802648 was + # cancelled 19 minutes into TestClusterFusion.setUpClass's + # create_cluster(wait_on_active=True, wait_timeout=1200): the log ends at # '##[error]The operation was canceled.' with no pytest summary and no - # sweep output, so three clusters were left billing with nothing in the - # process having recorded them. The ledger is that record. + # sweep output, leaving three clusters billing with nothing in the process + # having recorded them. The ledger is that record. # - # Last step in the job so it covers every pytest step above it. Placing - # it after each one instead would add nothing: a cancellation anywhere - # still runs the remaining always() steps. + # Last step in the job, so it covers every pytest step above it. # - # What always() does not buy is unlimited time. GitHub force-terminates a - # cancelled job's remaining steps after a 5-minute cancellation timeout, - # and an S-00 cluster refuses DELETE until it is ACTIVE (~460s). The leak - # above is covered because it was cancelled 19 minutes in, well past that; - # a cancel in the first several minutes would be killed here still being - # told 400/409, and needs `cleanup_deployments.py --older-than` run by - # hand afterwards. + # Best effort, not a guarantee: a cancelled job's remaining steps are + # force-terminated after GitHub's 5-minute cancellation timeout, and an + # S-00 cluster refuses DELETE until it is ACTIVE (~460s). The leak above is + # covered, being 19 minutes in; a cancel in the first few minutes would be + # killed here still getting 400/409, and needs `cleanup_deployments.py + # --older-than` run by hand. - name: Terminate any deployment the tests left behind if: always() run: | @@ -130,26 +126,22 @@ jobs: # Waits for test-coverage rather than running alongside it, so the v1 # workspace groups are never in flight at the same time as the v2 suite's - # cluster pool -- together they put more on the org than it wants to carry. - # This is a nightly cron, so the extra wall clock costs nothing. + # cluster pool. This is a nightly cron, so the extra wall clock is free. # - # Runs even when test-coverage fails, because this is a legacy gate, not a - # downstream build: a v2 failure above says nothing about the v1 endpoints, - # and skipping v1 for it would hide a v1 regression behind an unrelated one. + # Runs even when test-coverage fails: a v2 failure above says nothing about + # the v1 endpoints, and skipping v1 for it would hide a v1 regression behind + # an unrelated one. # # !cancelled() rather than always(), which stays true through cancellation - # too. A cancelled run must not go on to start provisioning workspace groups - # here: GitHub force-terminates a cancelled job's remaining steps after a - # 5-minute cancellation timeout, so the ledger sweep below would be killed - # while the new deployments were still pre-ACTIVE and refusing DELETE -- the - # job would strand exactly what it was added to clean up. + # too. A cancelled run must not start provisioning workspace groups here: + # remaining steps are force-terminated 5 minutes into a cancel, so the sweep + # below would be killed while the new deployments were still pre-ACTIVE and + # refusing DELETE -- stranding exactly what it exists to clean up. needs: test-coverage if: ${{ !cancelled() }} - # A ledger of its own, not shared with test-coverage. The jobs no longer - # overlap, but they still run on separate runners with separate workspaces, - # and keeping the ledgers distinct means neither job's sweep can reach the - # other's records. + # A ledger of its own: separate runner, separate workspace, and neither + # job's sweep can reach the other's records. env: SINGLESTOREDB_TEST_DEPLOYMENT_LOG: ${{ github.workspace }}/deployments.jsonl @@ -179,12 +171,10 @@ jobs: pip install -e ".[dev]" - name: Run v1 management API tests - # -n 0 overrides the -n 2 in pyproject.toml's addopts. The parallel - # default is tuned for the v2 management suite's shared cluster pool; the - # v1 classes deploy workspace groups of their own, so two workers here - # put twice that in flight, on top of whatever the v2 job is holding at - # the same time. Serial keeps this job's contribution to the org's - # cluster count to one deployment at a time. + # -n 0 overrides the -n 2 in pyproject.toml's addopts, which is tuned for + # the v2 suite's shared cluster pool. The v1 classes deploy workspace + # groups of their own, so two workers would put twice that in flight. + # Serial keeps this job to one deployment at a time. run: | pytest -v -n 0 -m 'management_v1' --pyargs singlestoredb.tests env: diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index be006ab2d..459b30ec3 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -49,17 +49,14 @@ jobs: - name: Initialize database id: initialize-database - # A new cluster has exactly one user, admin, which is why that name is - # fixed everywhere below. The project is named here rather than read - # from a repo variable so the deployment target is visible in the - # workflow and does not depend on repository settings. + # A new cluster has exactly one user, admin, hence the fixed name below. + # The project is named here, not in a repo variable, so the deployment + # target is visible in the workflow. # - # POST /v2/clusters generates its own admin password and ignores any - # that is sent, so the script resets it to CLUSTER_PASSWORD over SQL - # once the cluster is up. That keeps the credential a secret the runner - # masks everywhere, instead of a job output: the runner refuses to write - # an output whose value is masked, so a generated password could not - # reach these jobs at all. + # POST /v2/clusters generates its own admin password and ignores any sent + # to it, so the script resets it to CLUSTER_PASSWORD over SQL once the + # cluster is up. A generated password could not reach the other jobs + # anyway: the runner refuses to write a masked value as a job output. run: | python resources/create_test_cluster.py --password="${{ secrets.CLUSTER_PASSWORD }}" --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: @@ -268,9 +265,9 @@ jobs: - name: Shutdown cluster if: ${{ always() }} # An empty ID would send the DELETE to /v2/clusters/ and leave a live - # cluster behind, so say so loudly instead: at that point the ID has to - # be recovered by hand. --fail-with-body is what makes a refused DELETE - # fail this step rather than printing the error and exiting 0. + # cluster behind, so fail loudly instead; the ID then has to be recovered + # by hand. --fail-with-body makes a refused DELETE fail this step rather + # than print the error and exit 0. run: | if [ -z "$CLUSTER_ID" ]; then echo "::error::No cluster ID from setup-database; the cluster (if any) must be terminated by hand" diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 7984a8068..38f63e3f3 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -28,17 +28,14 @@ jobs: - name: Initialize database id: initialize-database - # A new cluster has exactly one user, admin, which is why that name is - # fixed everywhere below. The project is named here rather than read - # from a repo variable so the deployment target is visible in the - # workflow and does not depend on repository settings. + # A new cluster has exactly one user, admin, hence the fixed name below. + # The project is named here, not in a repo variable, so the deployment + # target is visible in the workflow. # - # POST /v2/clusters generates its own admin password and ignores any - # that is sent, so the script resets it to CLUSTER_PASSWORD over SQL - # once the cluster is up. That keeps the credential a secret the runner - # masks everywhere, instead of a job output: the runner refuses to write - # an output whose value is masked, so a generated password could not - # reach these jobs at all. + # POST /v2/clusters generates its own admin password and ignores any sent + # to it, so the script resets it to CLUSTER_PASSWORD over SQL once the + # cluster is up. A generated password could not reach the other jobs + # anyway: the runner refuses to write a masked value as a job output. run: | python resources/create_test_cluster.py --password="${{ secrets.CLUSTER_PASSWORD }}" --token="${{ secrets.CLUSTER_API_KEY }}" --project="Standard Project" --init-sql singlestoredb/tests/test.sql --output=github --expires=2h "python - $GITHUB_WORKFLOW - $GITHUB_RUN_NUMBER" env: @@ -59,11 +56,10 @@ jobs: matrix: os: - ubuntu-24.04 - # Every version from the floor in pyproject.toml (requires-python - # >=3.9) up to the newest final release. 3.15 is deliberately absent: - # as of 2026-09-18 it is at rc2 with GA planned for 2026-10-01, and - # setup-python needs allow-prereleases plus an explicit "3.15.0-rc.2" - # to install it at all. Add a bare "3.15" once it ships. + # The floor in pyproject.toml (requires-python >=3.9) up to the newest + # final release. 3.15 is absent deliberately: as of 2026-09-18 it is at + # rc2 (GA 2026-10-01), and setup-python needs allow-prereleases plus an + # explicit "3.15.0-rc.2" to install it. Add a bare "3.15" once it ships. python-version: - "3.9" - "3.10" @@ -190,9 +186,9 @@ jobs: - name: Shutdown cluster if: ${{ always() }} # An empty ID would send the DELETE to /v2/clusters/ and leave a live - # cluster behind, so say so loudly instead: at that point the ID has to - # be recovered by hand. --fail-with-body is what makes a refused DELETE - # fail this step rather than printing the error and exiting 0. + # cluster behind, so fail loudly instead; the ID then has to be recovered + # by hand. --fail-with-body makes a refused DELETE fail this step rather + # than print the error and exit 0. run: | if [ -z "$CLUSTER_ID" ]; then echo "::error::No cluster ID from setup-database; the cluster (if any) must be terminated by hand" diff --git a/resources/create_test_cluster.py b/resources/create_test_cluster.py index 28be11c38..02f26ab6b 100755 --- a/resources/create_test_cluster.py +++ b/resources/create_test_cluster.py @@ -36,8 +36,8 @@ parser.add_option( '-p', '--password', help='password to give the admin user once the cluster is up; required, ' - 'because the password the API generates cannot be handed to another ' - 'CI job (see below)', + 'because the password the API generates cannot be reported to a ' + 'caller that masks it', ) parser.add_option( '-t', '--token', @@ -87,13 +87,11 @@ # Find a matching region. A v2 region is identified by the -# (provider, region_name) pair rather than by an ID, so the matched Region -# object is what gets handed to create_cluster. Candidates are shuffled to -# spread deployments across whichever regions match. -# -# The pattern is tried against both the display name and the provider region -# name -- 'US East 1' and 'us-east-1' -- so it does not matter which of the two -# a given listing puts in Region.name. +# (provider, region_name) pair rather than an ID, so the matched Region object +# itself is handed to create_cluster. Candidates are shuffled to spread +# deployments across whichever regions match, and the pattern is tried against +# both the display name and the provider region name -- 'US East 1' and +# 'us-east-1' -- since either may land in Region.name. pattern = options.region.replace('*', '.*') regions = list(mgr.regions) @@ -141,9 +139,8 @@ def candidates(item): # A cluster name must match [a-z0-9]([a-z0-9-]*[a-z0-9])? and be 1-32 -# characters, so everything outside that alphabet becomes a hyphen, runs of -# hyphens collapse, and the result is truncated with any hyphen the cut -# exposes trimmed off again. +# characters: fold everything outside that alphabet to a hyphen, truncate, and +# trim any hyphen the cut exposes. name = re.sub(r'[^a-z0-9]+', '-', args[0].lower()).strip('-')[:32].rstrip('-') if not name: print(f'ERROR: Cluster name is empty after cleaning: {args[0]}', file=sys.stderr) @@ -175,13 +172,11 @@ def candidates(item): database = 'TEMP_{}'.format(uuid.uuid4()).replace('-', '_') # Report before touching the cluster any further. Everything below can fail -# against a cluster that already exists and is already billing, and the caller's -# only handle on it is the ID reported here -- a CI teardown job with an empty -# cluster-id output would issue its DELETE against /v2/clusters/ and leak the -# cluster it was meant to remove. +# against a cluster that is already billing, and the ID reported here is the +# caller's only handle on it -- a CI teardown job with an empty cluster-id output +# would DELETE /v2/clusters/ and leak the cluster it meant to remove. # -# No password is reported: the caller passed it in, so it already knows it, and -# under GitHub Actions it is a secret the runner masks on its own. +# No password is reported: the caller passed it in, so it already has it. if options.output == 'env': print(f'CLUSTER_ID={cluster.id}') print(f'CLUSTER_HOST={host}') @@ -202,10 +197,10 @@ def candidates(item): print('}') # The API generates the admin password and reports it only on the create -# response -- there is no route that will hand it back later, and it is None -# after any refresh(). See item 9 of docs/management-api-audit.md: the API -# accepts an adminPassword on both POST and PATCH and ignores both, which is -# why this is read back rather than set. +# response: no route hands it back later, and it is None after any refresh(). +# It is read back rather than set because the API accepts an adminPassword on +# both POST and PATCH and ignores both -- item 9 of +# docs/management-api-audit.md. generated = cluster.admin_password if not generated: print( @@ -215,16 +210,12 @@ def candidates(item): sys.exit(1) # Trade the generated password for the caller's, because the generated one -# cannot leave this process. A caller running under GitHub Actions has to mask -# it, and the runner drops any output whose value matches a mask -- "Skip output -# 'cluster-password' since it may contain secret" -- so masking it and passing -# it to another job are mutually exclusive. The password the caller already -# holds has neither problem. +# cannot leave this process: a GitHub Actions runner drops any output whose value +# is masked -- "Skip output 'cluster-password' since it may contain secret" -- so +# masking it and passing it to another job are mutually exclusive. # -# ALTER USER is the statement that works: SET PASSWORD wants a pre-hashed value -# and rejects a literal with '1372: Password hash should be a 41-digit -# hexadecimal number'. Verified against a live S-00 cluster, including that the -# control plane leaves the new password alone afterwards. +# ALTER USER, not SET PASSWORD, which wants a pre-hashed value and rejects a +# literal with '1372: Password hash should be a 41-digit hexadecimal number'. password = options.password escaped = password.replace('\\', '\\\\').replace("'", "\\'") diff --git a/singlestoredb/management/manager.py b/singlestoredb/management/manager.py index e2ccbe221..c20f571de 100644 --- a/singlestoredb/management/manager.py +++ b/singlestoredb/management/manager.py @@ -43,17 +43,15 @@ def set_organization(kwargs: Dict[str, Any]) -> None: kwargs['params']['organizationID'] = org -#: Methods that may be replayed after a transport-level failure. POST is -#: absent on purpose: a dropped connection does not say whether the server -#: acted on the request, and replaying ``POST /clusters`` would deploy twice. -#: Everything the long ``wait_on_*`` loops issue is a GET, so the retries -#: cover the failure mode that actually shows up -- a keep-alive connection -#: the far end closed while the client was sleeping between polls, which -#: surfaces as ``RemoteDisconnected`` on the next request. +#: Methods that may be replayed after a transport-level failure. POST is absent +#: on purpose: a dropped connection does not say whether the server acted, and +#: replaying ``POST /clusters`` would deploy twice. Everything the long +#: ``wait_on_*`` loops issue is a GET, so this covers the failure mode that shows +#: up -- a keep-alive connection the far end closed while the client slept +#: between polls, surfacing as ``RemoteDisconnected`` on the next request. #: -#: The one exception is a creation the organization lock blocked, which -#: :func:`retry_on_lock` replays: it is identified by the error message, which -#: this policy never sees. +#: :func:`retry_on_lock` is the one POST replay, keyed on an error message this +#: policy never sees. RETRY_METHODS = frozenset(['GET', 'HEAD', 'OPTIONS', 'PUT', 'DELETE']) #: Status codes worth retrying. These are the transient ones; a 4xx other @@ -88,20 +86,20 @@ def build_retry( #: "could not acquire lock within duration", the API's refusal to start a -#: creation while another one in the organization holds the lock. Seen from -#: ``POST /workspaceGroups`` and ``POST /clusters``, whose message says "error -#: creating workspace" either way, so the match cannot key on the noun. +#: creation while another one in the organization holds the lock. Both +#: ``POST /workspaceGroups`` and ``POST /clusters`` say "error creating +#: workspace", so the match cannot key on the noun. #: -#: Matched on the message, not the status: the conflict arrives as a 500, and so -#: does a name collision. The wording is also the only part that says nothing -#: was created, which is what makes replaying the POST safe. +#: Matched on the message, not the status: a name collision is a 500 too, and +#: only the wording says nothing was created, which is what makes replaying the +#: POST safe. LOCK_ERROR_RE = re.compile(r'acquire[^.]{0,40}lock', re.I) #: Ceiling on the wait between lock retries, and the random extra added to each #: one. Capped because what is being waited out is another creation's POST #: returning, not a deployment coming up. Jittered because two clients that -#: collide back off by the same amounts from the same moment -- two xdist -#: workers, say -- and would otherwise retry in step indefinitely. +#: collided back off identically from the same moment -- two xdist workers, say +#: -- and would otherwise retry in step indefinitely. LOCK_RETRY_MAX_INTERVAL = 60.0 LOCK_RETRY_JITTER = 5.0 @@ -140,19 +138,16 @@ def retry_on_lock(func: Callable[..., Any]) -> Callable[..., Any]: """ Wait out an organization lock conflict on a deployment creation. - Replaying a POST is safe here where widening :data:`RETRY_METHODS` would not - be: the transport sees only a 500 and cannot know whether the server acted, - whereas the lock message says the creation never started. A creation that - made something and *then* failed does not come back with this message, and - would surface on the replay as a name conflict rather than being swallowed. + Replaying this POST is safe where widening :data:`RETRY_METHODS` would not + be: the lock message says the creation never started. A creation that made + something and *then* failed reports something else, and would surface on the + replay as a name conflict rather than being swallowed. Worn by ``WorkspaceManager.create_workspace_group`` and ``ClusterManager.create_cluster`` only -- the two calls that contend for the - lock. Fusion SQL's ``CREATE WORKSPACE GROUP`` and ``CREATE CLUSTER`` go - through them, so they are covered too. - - Any other ``ManagementError`` is raised at once, as is the conflict itself - once :func:`lock_retry_policy`'s budget runs out. + lock, and the ones Fusion's ``CREATE WORKSPACE GROUP``/``CREATE CLUSTER`` + go through. Any other ``ManagementError`` is raised at once, as is the + conflict itself once :func:`lock_retry_policy`'s budget runs out. """ @functools.wraps(func) def wrapper(self: Any, *args: Any, **kwargs: Any) -> Any: @@ -174,9 +169,8 @@ def wrapper(self: Any, *args: Any, **kwargs: Any) -> Any: timing.sleep(wait, f'{func.__name__} organization lock') # Says which methods wear this, for a test to assert against. On the - # wrapper's ``__dict__``, so ``functools.wraps`` carries it outward through - # any later decorator -- the test suite wraps these methods again to track - # what a run has deployed. + # wrapper's ``__dict__``, so ``functools.wraps`` carries it out through any + # later decorator -- the test suite wraps these methods again. wrapper.__retry_on_lock__ = True # type: ignore[attr-defined] return wrapper @@ -210,12 +204,11 @@ class Manager: #: Management API version if none is specified. The shared #: :data:`~singlestoredb.management._version_import.DEFAULT_VERSION`, which - #: also supplies the ``management.version`` option default, so the two - #: cannot drift. Deliberately not a reading of that option: it is read by - #: the ``manage_*`` factories at call time, and reading it here would let a - #: version-specific class declare itself to be whatever the option happened - #: to say. A class that implements one specific version pins that version - #: as a literal instead of inheriting this. + #: also supplies the ``management.version`` option default, so the two cannot + #: drift. Deliberately not a reading of that option, which the ``manage_*`` + #: factories read at call time: reading it here would let a version-specific + #: class declare itself to be whatever the option happened to say. Such a + #: class pins its version as a literal instead of inheriting this. default_version = DEFAULT_VERSION #: Base URL if none is specified. diff --git a/singlestoredb/management/utils.py b/singlestoredb/management/utils.py index ba6553de4..c596cd8e3 100644 --- a/singlestoredb/management/utils.py +++ b/singlestoredb/management/utils.py @@ -408,16 +408,15 @@ def enable_http_tracing() -> None: requests_log.propagate = True -#: A Go ``time.Time`` rendered by its ``String()`` method: -#: ``2026-09-17 14:42:41.445984 +0000 UTC``. ``GET /v2/clusters/{id}`` reports -#: ``expiresAt`` in this shape while every other timestamp it returns is -#: RFC 3339, and the trailing zone name is not ISO 8601, so the whole value -#: fails to parse and the expiration silently reads as unset. The zone name and -#: the monotonic-clock reading Go appends to some values are both optional. -#: An RFC 3339 ``Z`` counts as an offset here so that shape goes down the same -#: path: its fraction needs the same padding, and until it matched, a value like -#: ``...20.43888Z`` reached the converter with five digits, which only 3.11 and -#: later parse. +#: Both timestamp shapes the API returns: RFC 3339, and a Go ``time.Time`` +#: rendered by ``String()`` -- ``2026-09-17 14:42:41.445984 +0000 UTC``, which is +#: how ``GET /v2/clusters/{id}`` reports ``expiresAt``. The trailing zone name is +#: not ISO 8601, so that value used to fail to parse and read as unset. It and +#: Go's monotonic reading are both optional. +#: +#: ``Z`` counts as an offset so RFC 3339 gets the same fraction padding: +#: ``...20.43888Z`` otherwise reached the converter with five digits, which only +#: 3.11 and later parse. _GO_DATETIME_RE = re.compile( r'^(?P\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}(?:\.\d+)?)' r'(?:\s*(?P[Zz]|[+-]\d{2}:?\d{2}))?' @@ -430,10 +429,9 @@ def _normalize_datetime(obj: str) -> str: """ Return ``obj`` as something :func:`converters.datetime_fromisoformat` reads. - Handles the two shapes the management API returns -- RFC 3339 and the Go - ``time.Time.String()`` form -- by reducing both to a bare ISO 8601 + Reduces both shapes :data:`_GO_DATETIME_RE` matches to a bare ISO 8601 timestamp plus an optional numeric offset. Fractional seconds are padded to - microseconds -- Go trims trailing zeros, and ``datetime.fromisoformat`` + microseconds: Go trims trailing zeros, and ``datetime.fromisoformat`` accepts only 3 or 6 digits before Python 3.11. Parameters @@ -460,9 +458,8 @@ def _normalize_datetime(obj: str) -> str: micros = micros[:6] + '0' * (6 - len(micros)) stamp = stamp + '.' + micros - # Go writes the offset without a separator (+0000). Only Python 3.11 and - # later accept that spelling; 3.9 and 3.10 want +00:00, so always emit the - # colon. Z is spelled out for the same reason: nothing before 3.11 reads it. + # Go writes +0000; 3.9 and 3.10 want +00:00, so always emit the colon. Z is + # spelled out for the same reason -- nothing before 3.11 reads it. offset = match.group('offset') or '' if offset in ('Z', 'z'): offset = '+00:00' @@ -476,13 +473,11 @@ def _is_go_zero_time(obj: Union[datetime.date, datetime.datetime]) -> bool: """ Return whether ``obj`` is Go's zero time, which means "unset". - A Go ``time.Time`` that was never assigned renders as January 1 of year 1, - and the API returns that for a field it has no value for -- most visibly an - ``expiresAt`` on a resource that does not expire. It arrives spelled either - way the two timestamp shapes allow: ``0001-01-01T00:00:00Z`` and - ``0001-01-01 00:00:00 +0000 UTC``. Testing the parsed value rather than the - string covers both, along with any offset or monotonic reading that comes - with them. + An unassigned Go ``time.Time`` renders as January 1 of year 1, and the API + returns that for a field it has no value for -- most visibly an ``expiresAt`` + on a resource that does not expire. Testing the parsed value rather than the + string covers both spellings (``0001-01-01T00:00:00Z`` and + ``0001-01-01 00:00:00 +0000 UTC``) and any trimmings they carry. Parameters ---------- @@ -501,11 +496,9 @@ def _as_naive_utc(obj: datetime.datetime) -> datetime.datetime: """ Return ``obj`` as a naive UTC datetime. - A value carrying an offset -- which is every recognized shape, since an - RFC 3339 ``Z`` is normalized to ``+00:00`` -- is shifted onto UTC and - stripped. A value that arrives naive is already meaning UTC and is left - alone. Both end up on the one convention -- otherwise two timestamps read - off the same object could not be compared. + An aware value is shifted onto UTC and stripped; a naive one already means + UTC and is left alone. One convention either way, or two timestamps read off + the same object could not be compared. Parameters ---------- @@ -536,8 +529,7 @@ def to_datetime( if out is None: return None # Before _as_naive_utc: shifting an aware year-1 value onto UTC can carry it - # below datetime.MINYEAR, which raises rather than returning the None this - # value means. + # below datetime.MINYEAR, which raises instead of returning None. if _is_go_zero_time(out): return None if isinstance(out, datetime.date) and not isinstance(out, datetime.datetime): diff --git a/singlestoredb/management/v1/workspace.py b/singlestoredb/management/v1/workspace.py index f7a72aa91..7fda33e56 100644 --- a/singlestoredb/management/v1/workspace.py +++ b/singlestoredb/management/v1/workspace.py @@ -852,10 +852,9 @@ def terminate( msg='No workspace manager is associated with this object.', ) # 'true'/'false', not the bool: requests renders a bool param with - # str(), so force=True went out as force=True. Workspace.terminate - # above already builds the lowercase form by hand; this matches it. - # force is what makes a group with live workspaces in it go away, so - # the value being read is not optional. + # str(), so force=True went out as force=True. force is what makes a + # group with live workspaces in it go away, so the value has to be read. + # Workspace.terminate above spells it out by hand for the same reason. self._manager._delete( f'workspaceGroups/{self.id}', params=dict(force='true' if force else 'false'), diff --git a/singlestoredb/management/v2/cluster.py b/singlestoredb/management/v2/cluster.py index 8a38c64a2..145d21253 100644 --- a/singlestoredb/management/v2/cluster.py +++ b/singlestoredb/management/v2/cluster.py @@ -614,16 +614,14 @@ def update( admin_password : str, optional Admin password for the cluster. - .. warning:: This is ignored, exactly as it is on - ``POST /v2/clusters``. ``PATCH /v2/clusters/{id}`` accepts the - field and does not honor it: a live probe found the patched value - refused with ``1045: Access denied`` while the password the - original create generated kept working. So the admin password - cannot be set after the fact either -- the only value that - authenticates is the generated one - :attr:`Cluster.admin_password` carried on the create response. - See item 9 of ``docs/management-api-audit.md``. The field is - still sent in case the API starts honoring it. + .. warning:: Ignored, exactly as on ``POST /v2/clusters``. ``PATCH`` + accepts the field and does not honor it: a live probe found the + patched value refused with ``1045: Access denied`` while the + password the create generated kept working. The only value that + authenticates is that generated one, carried on the create + response as :attr:`Cluster.admin_password`. Still sent in case + the API starts honoring it. See item 9 of + ``docs/management-api-audit.md``. expires_at : str, optional Timestamp of when the cluster will expire. Expiration time can be specified as a timestamp or a duration. diff --git a/singlestoredb/tests/cleanup_deployments.py b/singlestoredb/tests/cleanup_deployments.py index 6b715431a..745069004 100644 --- a/singlestoredb/tests/cleanup_deployments.py +++ b/singlestoredb/tests/cleanup_deployments.py @@ -50,12 +50,11 @@ python -m singlestoredb.tests.cleanup_deployments --ledger deployments.jsonl -That mode replaces *both* guards above -- the name patterns and the age -filter. Neither is needed, because the ledger names the deployments rather -than guessing at them, and neither is safe: a ledger entry is minutes old by -construction, so the age filter would spare everything it lists. What keeps -such a run off other people's deployments is that it only ever touches ids and -names the ledger records, and that each CI job writes its own ledger. +That mode replaces *both* guards above. The ledger names deployments rather +than guessing at them, so the patterns are unnecessary; and its entries are +minutes old by construction, so the age filter would spare every one of them. +What keeps it off other people's deployments instead is that it touches only +ids and names the ledger records, and that each CI job writes its own ledger. """ import argparse import datetime @@ -89,18 +88,15 @@ #: anything younger could belong to a run in progress. DEFAULT_MIN_AGE_HOURS = 6.0 -#: How long to keep retrying a deployment the API will not delete yet. This is -#: the end of the line -- nothing runs after this tool -- so it does not borrow -#: ``utils.TERMINATE_RETRY_TIMEOUT``, which is deliberately short so the sweep -#: between test classes cannot stall the suite. Here a deployment may still be -#: coming up, ``DELETE`` is refused until it is, and an S-00 cluster reaching -#: ACTIVE is ~460s at worst, so anything shorter than a full provision leaves it -#: billing. An upper bound on retrying, not a promise of it: the whole budget is -#: available when the job that calls this ends normally or fails, but a -#: *cancelled* job's steps are force-terminated after GitHub's 5-minute -#: cancellation timeout, so a cancel early in a provision gets killed here -#: regardless of what this says. The only cost of the larger budget is the CI -#: step's wall clock. +#: How long to keep retrying a deployment the API will not delete yet. Longer +#: than ``utils.TERMINATE_RETRY_TIMEOUT``, which is short so the between-class +#: sweep cannot stall the suite: nothing runs after this tool, the deployment may +#: still be coming up, ``DELETE`` is refused until it is, and an S-00 cluster +#: reaching ACTIVE is ~460s at worst. The only cost is the CI step's wall clock. +#: +#: An upper bound, not a promise: a *cancelled* job's steps are force-terminated +#: after GitHub's 5-minute cancellation timeout, so a cancel early in a provision +#: gets killed here whatever this says. TERMINATE_TIMEOUT = 600.0 #: Names the suite generates. Anchored, because these run against a real @@ -109,11 +105,9 @@ PATTERNS = [ # test_management_v1.py / test_management_v2.py fixtures re.compile(r'^(wg|ws|cl)-test-[A-Za-z0-9_-]+$'), - # TestWorkspace.test_update renames its live group from wg-test- to - # wg-foo- and never renames it back, so the group carries this name - # for the rest of the class. No pattern matched it, which made a group - # stranded after that test invisible to this sweep -- it would pile up - # while the tool reported nothing. + # TestWorkspace.test_update renames its live group to wg-foo- and + # never renames it back, so it carries that name for the rest of the class. + # Unmatched, a group stranded after that test was invisible here. re.compile(r'^wg-foo-[A-Za-z0-9_-]+$'), re.compile(r'^starter-(ws|cl)-test-[A-Za-z0-9_-]+$'), # test_fusion.py fixtures @@ -121,19 +115,17 @@ re.compile(r'^[a-z]-fusion-cluster-[0-9a-f]+$'), re.compile(r'^jobs-fusion-[0-9a-f]+$'), re.compile(r'^stage-fusion-\d-[0-9a-f]+$'), - # test_create_drop_workspace_group's subject. Hex covers the decimal - # id(self) the test used to name it with, so groups stranded by older - # runs -- which this pattern did not match, and which therefore piled up - # invisibly -- are reaped too. + # test_create_drop_workspace_group's subject. Hex also covers the decimal + # id(self) the test used to name it with, so groups stranded by older runs + # are reaped too. re.compile(r'^Create WG Test [0-9a-f]+$'), ] -#: Names the suite used to generate. Kept separate so it is obvious what is -#: only here for cleanup, and matched all the same: a stranded deployment is -#: billed regardless of which revision made it, and ``main`` still creates -#: these -- it carries none of ``utils.track()``, the per-class sweep or this -#: script, so a run there leaks with nothing to reap it. Retire an entry once -#: no branch produces the name and the organization is clean of it. +#: Names the suite used to generate. Kept separate so it is obvious what is only +#: here for cleanup, and matched all the same: a stranded deployment bills +#: whichever revision made it, and ``main`` still creates these with nothing to +#: reap them. Retire an entry once no branch produces the name and the +#: organization is clean of it. LEGACY_PATTERNS = [ # TestStageFusion's two workspace groups, before it moved to v2 clusters # named stage-fusion-- and then to the shared cluster pool @@ -141,11 +133,10 @@ # TestFilesFusion's workspace group, which nothing in the class ever # read; it creates no deployment at all now re.compile(r'^Files Fusion Testing [0-9a-f]+$'), - # 'Group '. No revision of this repo generates this, so it is here - # on the owner's say-so rather than by attribution. Eight hex characters - # minimum, which is what the ones in the organization have: the bare - # 'Group 1' / 'Group 2' that a person or the portal produces is a real - # deployment someone is using, and a plain [0-9a-f]+ would match it. + # 'Group '. No revision of this repo generates this, so it is here on + # the owner's say-so. Eight hex characters minimum, which is what the ones + # in the organization have: a plain [0-9a-f]+ would also match the bare + # 'Group 1' a person or the portal produces. re.compile(r'^Group [0-9a-f]{8,}$'), ] @@ -331,21 +322,19 @@ def keep(obj: Any) -> bool: # # Ledger mode # -# What this exists for: GH Actions run 35631802648, job ``test-coverage``, was -# cancelled 19 minutes into a ``create_cluster(wait_on_active=True, -# wait_timeout=1200)`` and the log ends at ``##[error]The operation was -# canceled.`` with no pytest summary and no sweep output at all. Three clusters -# were live and no in-process handler ever ran. Reading a file written as the -# clusters were created is the only way to know that from another process. +# Why: GH Actions run 35631802648, job ``test-coverage``, was cancelled 19 +# minutes into ``create_cluster(wait_on_active=True)``. The log ends at +# ``##[error]The operation was canceled.`` with no sweep output -- three clusters +# live, no in-process handler ever run. A file written as they are created is the +# only way another process can learn their names. # #: How each ledger kind is resolved back to a live object: the management API #: version that owns it, the point lookup for a record that has an id, and the #: listing to search by name for a ``pending`` record that never got one. #: -#: The kinds are the values of ``utils._KIND_BY_CLASS``; a kind this does not -#: know is reported rather than skipped, since the alternative is silently not -#: reaping it. +#: The kinds are the values of ``utils._KIND_BY_CLASS``. An unknown kind is +#: reported rather than skipped, the alternative being to silently not reap it. LEDGER_KINDS = { 'cluster': ( 'v2', 'get_cluster', lambda mgr: mgr.clusters, @@ -386,27 +375,23 @@ def fold_ledger(lines: Any) -> List[Dict[str, Any]]: """ Reduce ledger records to the deployments that should still be live. - The ledger is append-only and written from several processes (one per xdist - worker), so it is a history, not a state: a deployment shows up as - ``pending``, then ``live`` once it has an id, then ``gone`` once something - terminated it. Folding keeps whatever the last event for a deployment was - not ``gone``. + The ledger is an append-only history, not a state: a deployment shows up as + ``pending``, then ``live`` once it has an id, then ``gone`` once terminated. + Folding keeps every deployment whose last event was not ``gone``. A ``pending`` is keyed by ``(kind, name)`` because that is all it has; the - matching ``live`` retires it and re-keys on the id. So the two records a - normal creation writes collapse to one entry, and a ``pending`` left - standing means the creator was interrupted before it returned -- the - cancelled-mid-``wait_on_active`` case, resolvable only by name. + matching ``live`` retires it and re-keys on the id, so a normal creation's + two records collapse to one entry. A ``pending`` left standing means the + creator was interrupted before returning -- the cancelled-mid-wait case, + resolvable only by name. - Order is creation order, since dicts preserve insertion order and a - deployment's key is first inserted when it first appears. The caller + Order is creation order, since dicts preserve insertion order. The caller reverses it, so a workspace goes before the group that holds it, matching ``utils.cleanup_tracked()``. - Malformed lines are skipped with a warning rather than aborting: this runs - as the last step of a CI job, and one truncated line -- a process killed - between the ``write`` and the ``fsync``, which the per-line fsync makes - unlikely but not impossible -- must not stop the rest from being reaped. + Malformed lines are skipped with a warning rather than aborting: this is the + last step of a CI job, and one truncated line must not stop the rest from + being reaped. """ live: Dict[Any, Dict[str, Any]] = {} @@ -485,11 +470,10 @@ def find_ledger_leftovers( that could not be resolved *and* could still be live, which is what makes the run exit non-zero. - A 404 from the point lookup means the deployment is already gone, which is - the common case: the ledger records every creation, and a run that finished - normally terminated all of them. Anything else -- a transport failure, an - unknown kind -- goes in the third list, because "could not tell" and "not - there" must not read the same when the difference is a cluster billing. + A 404 from the point lookup means the deployment is already gone, the common + case for a run that finished normally. Anything else -- a transport failure, + an unknown kind -- goes in the third list: "could not tell" and "not there" + must not read the same when the difference is a cluster billing. """ from singlestoredb.exceptions import ManagementError @@ -588,7 +572,7 @@ def _run_ledger_sweep(path: str, yes: bool) -> int: if not yes: print('\nDry run; pass --yes to terminate these.') - return 0 + return 1 if unresolved else 0 from singlestoredb.tests import utils @@ -614,10 +598,9 @@ def main(argv: Optional[List[str]] = None) -> int: parser.add_argument( '--ledger', metavar='PATH', help='sweep exactly what the run that wrote this JSONL ledger created ' - '(see SINGLESTOREDB_TEST_DEPLOYMENT_LOG). Replaces both the name ' - 'patterns and the age filter, which a ledger makes unnecessary ' - 'and which would in any case spare everything in it for being ' - 'minutes old. This is the mode CI runs as an if: always() step', + '(see SINGLESTOREDB_TEST_DEPLOYMENT_LOG). Replaces the name ' + 'patterns and the age filter, which would spare everything in it ' + 'for being minutes old. CI runs this as an if: always() step', ) parser.add_argument( '--older-than', type=float, default=DEFAULT_MIN_AGE_HOURS, @@ -666,9 +649,9 @@ def main(argv: Optional[List[str]] = None) -> int: ) args = parser.parse_args(argv) - # --ledger is a different question entirely -- "what did *this* run make?" - # rather than "what looks stranded?" -- so it does not compose with the - # name and age guards, and saying so beats silently ignoring them. + # --ledger asks "what did *this* run make?", not "what looks stranded?", so + # it does not compose with the name and age guards. Erroring beats silently + # ignoring them. if args.ledger: for flag, value in ( ('--older-than', args.older_than != DEFAULT_MIN_AGE_HOURS), diff --git a/singlestoredb/tests/conftest.py b/singlestoredb/tests/conftest.py index c1e3d8190..0de122150 100644 --- a/singlestoredb/tests/conftest.py +++ b/singlestoredb/tests/conftest.py @@ -306,24 +306,17 @@ def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: """ Sweep in an xdist worker, and hand what survived to the controller. - ``addopts`` is ``-n 2`` (``pyproject.toml``), so under the default the - sweep and its ``STILL LIVE`` banner run in a worker, whose stdout the - controller discards. A leak was therefore silent even when the sweep did - run and fail -- the one case the banner exists to make loud. - - ``config.workeroutput`` is the channel xdist provides for exactly this, and - it only exists in a worker: its absence is the ``-n 0`` case, where - ``pytest_unconfigure`` prints directly to a terminal someone is reading and - nothing here is needed. - - Sweeping here rather than leaving it all to ``pytest_unconfigure`` is what - makes the labels available at all. xdist's own - ``pytest_sessionfinish`` is a hookwrapper that sends ``workeroutput`` after - yielding, so anything written to it from this hook is still included -- - but ``pytest_unconfigure`` runs after the send, so a sweep that waited - until then would have nothing left to report. The sweep is idempotent (a - successful one empties ``_tracked``), so the later call simply finds - nothing to do. + Under the parallel default the sweep and its ``STILL LIVE`` banner run in a + worker, whose stdout the controller discards -- so a leak was silent even + when the sweep ran and failed, the one case the banner exists for. + ``config.workeroutput`` is xdist's channel for this; its absence means + ``-n 0``, where ``pytest_unconfigure`` already prints to a real terminal. + + The sweep has to happen here, not in ``pytest_unconfigure``, to have + anything to report: xdist's ``pytest_sessionfinish`` hookwrapper sends + ``workeroutput`` after yielding, which is before ``pytest_unconfigure`` + runs. The sweep is idempotent -- a successful one empties ``_tracked`` -- + so the later call finds nothing to do. """ workeroutput = getattr(session.config, 'workeroutput', None) if workeroutput is None: @@ -341,9 +334,8 @@ def pytest_testnodedown(node: Any, error: Any) -> None: """ Report, on the controller, what a worker could not terminate. - Runs in the controller process, whose output the user actually sees. The - worker's own banner went to a captured stream; this is the copy that gets - read. + The worker's own banner went to a captured stream; this is the copy anyone + actually sees. """ stranded = getattr(node, 'workeroutput', {}).get(_STRANDED_KEY) or [] if not stranded: @@ -397,10 +389,10 @@ def pytest_unconfigure(config: pytest.Config) -> None: #: ``pytest_terminal_summary`` can read it without a fixture. #: #: Every test that ran under an active trace is in here, including the ones that -#: made no management call at all: ``trace_management_api_class`` subtracts this -#: list from the class total to get the fixture share, so a test missing from it -#: has its wall clock charged to ``setUpClass``. The event-less ones are -#: filtered out at report time by :func:`_traced` instead. +#: made no management call: ``trace_management_api_class`` subtracts this list +#: from the class total to get the fixture share, so a test missing from it would +#: have its wall clock charged to ``setUpClass``. :func:`_traced` filters the +#: event-less ones out at report time instead. _management_traces: List[Tuple[str, Any]] = [] #: The same, for the class fixtures rather than the tests. Separate because the diff --git a/singlestoredb/tests/test_fusion.py b/singlestoredb/tests/test_fusion.py index f130e4673..f4337e513 100644 --- a/singlestoredb/tests/test_fusion.py +++ b/singlestoredb/tests/test_fusion.py @@ -982,10 +982,9 @@ def setUpClass(cls): @classmethod def tearDownClass(cls): # Deployments first, and each one guarded. Dropping the database first - # -- as this used to -- meant a database error aborted the teardown - # before a single group was terminated, and an unguarded loop meant a - # failure on the first group abandoned the other two. Three workspace - # groups is the most expensive thing this file leaks. + # meant a database error aborted the teardown before a single group was + # terminated; an unguarded loop meant a failure on the first group + # abandoned the other two. while cls.workspace_groups: group = cls.workspace_groups.pop() try: @@ -1161,14 +1160,11 @@ def test_show_workspaces(self): f'"B Fusion Testing {self.id}" with size S-00', ) - # Wait for the three to be listed, not for them to be ACTIVE. Nothing + # Wait for the three to be listed, not for them to be ACTIVE: nothing # below asserts a state value -- 'State' is checked as a column name, - # never for its contents -- so all this test needs is that SHOW - # WORKSPACES can see them. Requiring ACTIVE cost around 450 seconds a - # run for no assertion, and at a 30 second interval most of that was - # overshoot. Polled through timing.sleep so a traced run accounts for - # it; a bare time.sleep here was invisible to the tracer and landed in - # the unlabelled 'other' bucket. + # never for its contents. Requiring ACTIVE cost around 450 seconds a run + # for no assertion. Polled through timing.sleep so a traced run accounts + # for it; a bare time.sleep landed in the unlabelled 'other' bucket. wanted = ('show-ws-1', 'show-ws-2', 'show-ws-3') deadline = time.time() + 600 while True: @@ -1401,25 +1397,18 @@ class _ClusterFusionMixin: """ Plumbing shared by the CLUSTER fusion suites. - These are the v2 mirror of :class:`TestWorkspaceFusion`, flat rather than - nested. A cluster is created in one statement where a workspace needed - two, so there is no group fixture and no ``IN GROUP`` clause anywhere. - Names are lowercase and hyphenated because ``POST /v2/clusters`` enforces - ``[a-z0-9]([a-z0-9-]*[a-z0-9])?`` at 1-32 characters (audit item 7) -- - the spaced names the v1 suite uses are rejected. - - This was one class deploying three clusters in ``setUpClass``, which every - test then waited out whether or not it touched a cluster: the two - lifecycle tests deploy their own and the region, project and grammar - tests need none at all, yet all of them paid for three. The classes below - declare what they need in :attr:`fixture_prefixes` instead, so the - cluster-less ones start immediately and no class deploys more than it - reads. - - Only :class:`TestClusterFusionSuspendResume` still names a prefix, because - it is the only one left that both needs a cluster up front and mutates it. - :class:`TestClusterFusion` reads without mutating and so borrows from - ``utils.shared_clusters``; the lifecycle suites create their own clusters + A cluster is created in one statement, so there is no group fixture and no + ``IN GROUP`` clause anywhere. Names are lowercase and hyphenated because + ``POST /v2/clusters`` enforces ``[a-z0-9]([a-z0-9-]*[a-z0-9])?`` at 1-32 + characters (audit item 7). + + Each class declares what it needs in :attr:`fixture_prefixes` rather than + the whole file sharing one three-cluster ``setUpClass``, which every test + used to wait out whether or not it touched a cluster. Only + :class:`TestClusterFusionSuspendResume` still names a prefix, being the one + class that needs a cluster up front *and* mutates it: + :class:`TestClusterFusion` reads without mutating and borrows from + ``utils.shared_clusters``, and the lifecycle suites create their own clusters in the test bodies, those creates being the subject under test. Not a ``TestCase``, and named with a leading underscore: pytest collects @@ -1554,23 +1543,18 @@ class TestClusterFusion(_ClusterFusionMixin, unittest.TestCase): ``SHOW CLUSTERS`` against the shared cluster pool. Borrows rather than deploying: nothing here mutates a cluster -- these are - four ``SHOW`` statements -- which is the condition ``utils.shared_clusters`` - asks of a consumer. ``SUSPEND``/``RESUME`` cannot borrow and deploys its own - in :class:`TestClusterFusionSuspendResume`. - - Three of them, which is one more than the pool was built for, so the - ``LIKE``/``ORDER BY``/``LIMIT`` assertions have something to sort. Joining - the Stage group rather than the Jobs one because Stage already asks for two: - the pool grows to the largest request, so this costs that group one extra - cluster instead of three, and the Jobs group is left at one. - - Every assertion here is scoped to ``utils.shared_cluster_pattern()`` and - counted against ``utils.shared_cluster_names()``, never a literal. The pool - is shared and grows to whatever the largest request in the process turns out - to be, so a hardcoded 3 would break the day a class asks for four -- and - would break silently, as a row count, which is the failure this class had - before when its count depended on other classes' clusters leaving the list - endpoint in time. + four ``SHOW`` statements -- which is what ``utils.shared_clusters`` asks of a + consumer. ``SUSPEND``/``RESUME`` cannot borrow and deploys its own in + :class:`TestClusterFusionSuspendResume`. + + Three of them, so the ``LIKE``/``ORDER BY``/``LIMIT`` assertions have + something to sort. In the Stage group rather than the Jobs one because Stage + already asks for two and the pool grows to the largest request: one extra + cluster there instead of three, and Jobs stays at one. + + Assertions are scoped to ``utils.shared_cluster_pattern()`` and counted + against ``utils.shared_cluster_names()``, never a literal, since the pool + grows to whatever the largest request in the process turns out to be. """ #: Borrowed, so kept out of ``clusters``, which ``tearDownClass`` @@ -1609,10 +1593,9 @@ def test_show_clusters_columns(self): assert row[2], row assert row[5], row # ProjectName, not the ID: the column reports the name the project - # listing gives for the ID the cluster was deployed into. Read back - # from the cluster rather than from this class's own project_id -- - # the pool resolves its project independently, and asserting against - # the borrower's copy would be asserting the two resolutions agree. + # listing gives for the ID the cluster was deployed into. Read back from + # the cluster, not from this class's project_id, which the pool resolved + # independently. expected = type(self).manager.get_cluster(cluster.id).project assert row[9] == expected.name, row @@ -1876,7 +1859,6 @@ def test_create_cluster_without_project(self): 'this test is for', ) - live = [] try: with self.assertRaises(Exception): self.cur.execute( @@ -1884,18 +1866,14 @@ def test_create_cluster_without_project(self): f'"{region.region_name}"', ) finally: - # One listing, serving both purposes: the assertion that nothing - # was created, and the cleanup for when something was. The test - # only passes if this comes back empty, so the terminate below - # fires exactly when the assertion is about to fail -- which is - # also the only case where a cluster exists. + # One listing, serving both purposes: the assertion that nothing was + # created, and the cleanup for when something was. The terminate + # below therefore fires only when the assertion is about to fail. # - # Belt and braces rather than the only cleanup, contrary to what - # this used to claim: the handler reaches the API through - # ClusterManager.create_cluster (fusion/handlers/cluster.py), which - # is the method utils._CREATORS wraps, so a cluster created here is - # tracked and ledgered like any other and the sweep would find it. - # Terminating it now just means not waiting for the sweep. + # Belt and braces: the handler goes through + # ClusterManager.create_cluster, which utils._CREATORS wraps, so a + # cluster created here is tracked and the sweep would find it + # anyway. This just means not waiting for the sweep. live = [ x for x in mgr.clusters if x.name == name and x.terminated_at is None @@ -1943,13 +1921,11 @@ def test_create_cluster_named_project(self): assert mgr.get_cluster(cluster_id).project.id == project.id finally: - # utils.terminate, not a bare terminate(force=True). The cluster is + # utils.terminate, not a bare terminate(force=True): the cluster is # PENDING, never having been waited out, and force does not make a - # pre-ACTIVE deployment deletable -- the API refuses it with a 400 - # or a 409 either way (see utils.terminate, which retries exactly - # that). A single forced DELETE here was therefore the likeliest - # outcome, swallowed by the except, leaving the cluster to the - # sweep; utils.terminate retries until it lands. + # pre-ACTIVE deployment deletable -- the API refuses it with a 400 or + # 409 either way. utils.terminate retries until it lands, where a + # single DELETE would be swallowed by the except below. if cluster_id is not None: try: utils.terminate(mgr.get_cluster(cluster_id)) diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index 731734c5c..d48ae2a48 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -2240,9 +2240,9 @@ def test_the_pool_is_built_once(self): self.assertEqual([x.id for x in first], [x.id for x in second]) self.assertEqual(len(self.created), 2) - # The pool build's lock conflict is waited out below this, in - # Manager._doit, which a stand-in manager does not go through: see - # TestManagerLockRetry. + # A lock conflict during the pool build is waited out by the + # @retry_on_lock on create_cluster, which a stand-in manager does not + # have: see TestLockRetry. def test_the_pool_grows_to_the_largest_request(self): with self._patched(): diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index 63e7b9dd0..e0f2357f6 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -89,10 +89,9 @@ def setUpClass(cls): wait_on_active=True, ) except Exception: - # Guarded: an unguarded terminate here replaces the create failure - # with whatever the DELETE raised, which both hides the real error - # and leaves the group live with nothing having reported why. - # utils.cleanup_tracked retries it and says so. + # Guarded: an unguarded terminate here would replace the create + # failure with whatever the DELETE raised. utils.cleanup_tracked + # retries it and reports it. try: cls.workspace_group.terminate(force=True) except Exception: @@ -263,13 +262,11 @@ def setUpClass(cls): name = shared_database_name(secrets.token_urlsafe(20)[:20]) # The starter-tier user name has to be unique across every starter - # deployment in the project, not just within this one: creating the - # same name in a second starter deployment fails while the first is - # live. So it is namespaced like the deployment and the database are, - # or this class collides with TestStarterCluster in test_management_v2 - # -- they run on different xdist workers -- and with any starter - # deployment an earlier failed run leaked. The API answers the - # collision with a bare 500, which names nothing. + # deployment in the project, not just within this one, so it is + # namespaced like the deployment and the database are. Otherwise this + # class collides with TestStarterCluster in test_management_v2 -- they + # run on different xdist workers -- and with anything an earlier failed + # run leaked. The API answers the collision with a bare 500. cls.starter_username = f'starter_user_{name[:8]}' cls.password = secrets.token_urlsafe(20) @@ -957,12 +954,11 @@ def test_get_secret(self): ), ).json() - # The ID comes from the create response rather than from the lookup - # under test: binding it inside the try would leave the cleanup raising - # UnboundLocalError over whatever the lookup actually failed with. - # Without this the secret outlived every run -- it was only ever - # removed opportunistically by the sweep at the top of the *next* one. - # test_management_v2.py's twin already does it this way. + # The ID comes from the create response, not from the lookup under + # test: binding it inside the try would leave the cleanup raising + # UnboundLocalError over whatever the lookup failed with. Without this + # the secret outlived every run, removed only by the sweep at the top of + # the *next* one. test_management_v2.py's twin does it this way. secret_id = created['secret']['secretID'] try: secret = self.manager.organizations.current.get_secret( @@ -1009,10 +1005,9 @@ def setUpClass(cls): wait_on_active=True, ) except Exception: - # Guarded: an unguarded terminate here replaces the create failure - # with whatever the DELETE raised, which both hides the real error - # and leaves the group live with nothing having reported why. - # utils.cleanup_tracked retries it and says so. + # Guarded: an unguarded terminate here would replace the create + # failure with whatever the DELETE raised. utils.cleanup_tracked + # retries it and reports it. try: cls.workspace_group.terminate(force=True) except Exception: diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index 09f8ba3c2..b2ec45ec8 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -290,54 +290,34 @@ def drop_user(name: str) -> None: # # Live deployment tracking # -# Every workspace group, workspace, cluster and starter cluster a test creates -# costs money until it is terminated, and the usual `tearDownClass` is not -# enough on its own: +# Every deployment a test creates costs money until it is terminated, and +# `tearDownClass` is not enough on its own: unittest skips it entirely if +# `setUpClass` raises, so a fixture that dies partway through leaks what it had +# already made, and a test that fails before its own cleanup line leaks too. # -# * unittest does not call `tearDownClass` at all if `setUpClass` raises, so -# a fixture that dies partway through -- two of three clusters created, -# then a dropped connection -- leaks everything it had made so far; -# * a test that creates a deployment in its body and then fails before its -# own cleanup line leaks it too. +# So creations are registered here as well, and `cleanup_tracked()` sweeps what +# is left: per test class as the run moves on, and again for everything at the +# end of the session (see conftest.py). Terminating twice is harmless, so a test +# that cleans up after itself need not untrack. # -# So creations are registered here as well, and `cleanup_tracked()` sweeps -# whatever is left: per test class as the run moves on to the next one, and -# again for everything at the end of the session (see conftest.py). -# Terminating twice is harmless -- the second attempt finds it gone and is -# ignored -- so tracked objects do not have to be untracked by the tests that -# clean up after themselves. -# -# Everything above is in-process, which is the one thing it cannot fix: the -# sweep, the ledger and `tearDownClass` all die with the interpreter. A job -# killed mid-provision -- GitHub force-terminates a cancelled job's remaining -# steps after a five-minute cancellation timeout -- leaves a PENDING cluster -# that no code here will ever get another chance to delete. `expires_at` is -# the answer to that, and only that: it is a property of the deployment, so -# the control plane honours it whether or not this process is still alive. +# All of that is in-process. A job killed mid-provision leaves a PENDING cluster +# nothing here gets another chance to delete; `expires_at` is the answer to that +# and only that, being honoured by the control plane either way. # -#: Expiry to request on every deployment a test creates, as the duration -#: string `POST` accepts (`resources/create_test_cluster.py` has passed one -#: nightly since it was written). The backstop under the sweep and the ledger, -#: not a replacement for either: a test still terminates what it created, and -#: nothing waits for an expiry to fire. -#: -#: Two hours, against a `wait_timeout` of 1200s and a longest test (Fusion -#: `CREATE`/`DROP`, which provisions twice in sequence) of about twenty -#: minutes. Enough headroom that an expiry can never land on a deployment a -#: test is still using, which would show up as an unrelated flake and be read -#: as an API fault. +#: Expiry to request on every deployment a test creates, as the duration string +#: `POST` accepts. A backstop under the sweep and the ledger, not a replacement: +#: a test still terminates what it created and nothing waits for an expiry. #: -#: Applied to what accepts it, which is the deployments that cost: v2 -#: `ClusterManager.create_cluster` and v1 -#: `WorkspaceManager.create_workspace_group`. The rest take no `expires_at` and -#: need none: +#: Two hours, against a `wait_timeout` of 1200s and a longest test of about +#: twenty minutes (Fusion `CREATE`/`DROP`, which provisions twice in sequence): +#: headroom enough that an expiry cannot land on a deployment still in use and +#: read as an unrelated API flake. #: -#: * a v1 workspace -- `expiresAt` is a property of the group, and terminating -#: the group takes its workspaces with it; -#: * the starter deployments -- `create_starter_cluster` and -#: `create_starter_workspace` have no such argument, and being shared tier -#: they are not what a leak costs. +#: Only `ClusterManager.create_cluster` (v2) and +#: `WorkspaceManager.create_workspace_group` (v1) take it, which is also where +#: the cost is. A v1 workspace needs none -- `expiresAt` belongs to the group -- +#: and the starter deployments accept no such argument. DEPLOYMENT_EXPIRES_AT = '2h' #: (owner, label, object) for every deployment created so far and not yet @@ -346,12 +326,11 @@ def drop_user(name: str) -> None: #: than idling -- and billing -- until the session ends. _tracked: List[Tuple[str, str, Any]] = [] -#: (receiver, finder, args, kwargs) for every creation call currently -#: executing. A creator POSTs and only then waits for the deployment to come -#: up, so for the whole ``wait_on_active`` window -- twenty minutes for a -#: cluster -- something billable exists that nothing has registered yet: -#: ``_tracking_wrapper`` tracks on return and recovers in its ``except``, and -#: neither runs if the process is killed. See :func:`recover_in_flight`. +#: (receiver, finder, args, kwargs) for every creation call currently executing. +#: A creator POSTs and only then waits for the deployment to come up, so for the +#: whole ``wait_on_active`` window -- twenty minutes for a cluster -- something +#: billable exists that nothing has registered yet. See +#: :func:`recover_in_flight`. _in_flight: List[Tuple[Any, Any, Tuple[Any, ...], Dict[str, Any]]] = [] #: Test class currently running, as set by conftest. @@ -372,56 +351,36 @@ def set_owner(owner: str) -> None: # # Durable deployment ledger # -# Everything above this point is in-memory only, and that is the one leak the -# sweeps cannot cover. A cancelled CI job is the proven case: GH Actions run -# 35631802648, job ``test-coverage``, was cancelled 19 minutes into -# ``TestClusterFusion.setUpClass``'s ``create_cluster(wait_on_active=True, -# wait_timeout=1200)``. The log ends at ``##[error]The operation was -# canceled.`` with no pytest summary, no "Terminated deployments left behind -# by tests:" and no ``STILL LIVE`` banner -- the process never got to sweep, -# and GitHub's cancellation grace period is nowhere near long enough for -# pytest to unwind three nested class fixtures, list to recover three -# in-flight creations and issue three DELETEs. After the SIGKILL that follows, -# ``_tracked`` and ``_in_flight`` are gone with the process and *nothing on -# disk* records that three clusters were created. +# The leak the in-memory sweeps cannot cover, proven by GH Actions run +# 35631802648: job ``test-coverage`` was cancelled 19 minutes into +# ``TestClusterFusion.setUpClass``'s ``create_cluster(wait_on_active=True)``. The +# log ends at ``##[error]The operation was canceled.`` -- no pytest summary, no +# sweep, no ``STILL LIVE`` banner. After the SIGKILL, ``_tracked`` and +# ``_in_flight`` went with the process and nothing on disk named the three +# clusters. # # So every creation is also appended to a JSONL file, flushed and fsync'd per -# line, which ``cleanup_deployments.py --ledger`` reads afterwards from a -# separate process -- an ``if: always()`` CI step that still runs on -# cancellation. The ledger is the record; the in-memory sweeps stay exactly as -# they were and remain the fast path. +# line, which ``cleanup_deployments.py --ledger`` reads from a separate process +# in an ``if: always()`` CI step. The in-memory sweeps remain the fast path; +# this is the record of last resort. # -# Opt-in, via SINGLESTOREDB_TEST_DEPLOYMENT_LOG. With the variable unset -# nothing is written and behaviour is byte-for-byte what it was: a local run -# has a human watching it and does not need a file to reap from. +# Three events per deployment: ``pending`` before the POST (by name, there being +# no id yet), ``live`` once there is an id, ``gone`` once terminated. The reaper +# folds the file and takes anything whose last event is not ``gone``. +# +# Opt-in via SINGLESTOREDB_TEST_DEPLOYMENT_LOG: unset, nothing is written. # - -#: Ledger event kinds, in the order a deployment normally produces them: -#: -#: * ``pending`` -- the creator is about to be called. Written *before* the -#: POST, from the name argument, because the whole point is the window where -#: the server has a billable deployment and this process has no id for it. -#: * ``live`` -- the creation returned (or an orphan was recovered), so there -#: is an id. -#: * ``gone`` -- it has been terminated. -#: -#: The reaper folds the file: anything whose last event is not ``gone`` is -#: still live. A ``pending`` with no matching ``live`` is the cancelled-mid- -#: wait case, and it is resolved by name rather than by id. #: Environment variable naming the ledger file. Read per write rather than -#: cached at import so a test can point it at a tmp_path with -#: ``mock.patch.dict(os.environ, ...)``. +#: cached at import so a test can point it at a tmp_path. LEDGER_ENV_VAR = 'SINGLESTOREDB_TEST_DEPLOYMENT_LOG' -#: Deployment kind for each created object's class. The ledger records a kind -#: so the reaper knows which manager and which point lookup to resolve a -#: record against, instead of guessing from the name -- ``cl-test-abc`` and -#: ``ws-test-abc`` are only distinguishable by convention, and a ``--ledger`` -#: run deliberately does not consult :data:`cleanup_deployments.PATTERNS`. +#: Deployment kind for each created object's class, so the reaper knows which +#: manager and point lookup to resolve a record against rather than guessing from +#: the name, which is convention only. #: -#: Keyed by class name rather than by the class itself to avoid importing v1 -#: and v2 management just to write a log line. +#: Keyed by class name, not the class, to avoid importing v1 and v2 management +#: just to write a log line. _KIND_BY_CLASS = { 'WorkspaceGroup': 'workspace_group', 'Workspace': 'workspace', @@ -440,19 +399,17 @@ def _ledger_write(**record: Any) -> None: """ Append one record to the deployment ledger. - Opened, written and closed per record, with ``flush()`` and ``os.fsync()`` - before the handle goes: surviving SIGKILL is the entire purpose, and a - line still sitting in a buffer when the process dies records nothing. The - cost is one open per creation, against a creation that takes minutes. + Opened, written and fsync'd per record: surviving SIGKILL is the whole + purpose, and a line still in a buffer records nothing. One open per creation + is nothing against a creation that takes minutes. - ``O_APPEND`` plus one ``write()`` per line is what makes this safe for the - parallel default (``-n 2``): the xdist workers are separate processes - sharing the file, and a single write of well under PIPE_BUF cannot - interleave with another's on Linux. No locking, therefore, and no partial - lines for the reaper to choke on. + That also makes it safe for the parallel default without locking. The xdist + workers are separate processes sharing the file, but each record is one short + ``write()`` to an ``O_APPEND`` handle, which Linux will not interleave, so + the reaper never sees a partial line. - Never raises. This sits on the creation path of every management test, so - a full disk or an unwritable path must cost a warning, not a test failure. + Never raises: this sits on the creation path of every management test, so an + unwritable ledger must cost a warning, not a failed run. """ path = ledger_path() if not path: @@ -483,9 +440,9 @@ def ledger_pending(kind: str, args: Tuple[Any, ...], kwargs: Any) -> None: Record that a deployment of this kind is about to be created. The name is taken the same way :func:`_recover_orphan` takes it -- keyword - first, else the first positional -- because it is the first parameter of - every creator, which ``test_management_utils.py`` pins. A record with no - usable name is skipped: there would be nothing for the reaper to resolve. + first, else the first positional, which every creator's signature makes the + name (pinned by ``test_management_utils.py``). Without a usable name there + is nothing for the reaper to resolve, so no record is written. """ name = kwargs.get('name') or (args[0] if args else None) if not isinstance(name, str): @@ -565,9 +522,8 @@ def track(obj: Any, label: str = '') -> Any: ), obj, )) - # Here rather than in the wrapper, so an orphan that `_recover_orphan` - # digs out of a listing gets an id into the ledger too -- that path - # reaches the server only through this function. + # Here rather than in the wrapper, so an orphan `_recover_orphan` digs + # out of a listing gets its id into the ledger too. ledger_live(obj) return obj @@ -623,24 +579,20 @@ def untrack(obj: Any) -> None: if entry[2] is obj: _tracked.pop(i) found = True - # Only for something that was actually tracked: untracking an object that - # was never registered -- a mocked one, or one already swept -- says - # nothing about whether a real deployment is gone, and a spurious ``gone`` - # would hide a live cluster from the reaper. + # Only for something actually tracked. Untracking an object that was never + # registered -- a mocked one, or one already swept -- says nothing about a + # real deployment, and a spurious ``gone`` hides a live cluster. if found: ledger_gone(obj) #: How long :func:`terminate` keeps retrying a deployment the API will not -#: delete yet, and how long it waits between attempts. Three minutes at 15s -#: spacing: the case being covered is a deployment killed mid-provision, which -#: has to finish coming up before it can be torn down, and an S-00 cluster -#: reaching ACTIVE is ~460s at worst. Waiting the full provision out here would -#: stall the sweep between every test class, so this buys the common case -- -#: a deployment most of the way up -- and leaves the rest to the end-of-session -#: sweep and then to ``cleanup_deployments.py``, which is the end of the line -#: and waits out a full provision with a longer budget of its own -#: (``cleanup_deployments.TERMINATE_TIMEOUT``). +#: delete yet, and the spacing between attempts. Three minutes is deliberately +#: less than a full provision (~460s for an S-00 cluster), because waiting one +#: out here would stall the sweep between every test class. It buys the common +#: case -- a deployment most of the way up -- and leaves the rest to the +#: session-end sweep and then to ``cleanup_deployments.TERMINATE_TIMEOUT``, +#: which is the end of the line and can afford the wait. TERMINATE_RETRY_TIMEOUT = 180.0 TERMINATE_RETRY_INTERVAL = 15.0 @@ -654,11 +606,9 @@ def _terminate_once(obj: Any) -> None: ``StarterCluster.terminate``) take no arguments at all. The signature is inspected rather than discovered by catching ``TypeError`` - from the call, as this used to do. That ``except TypeError`` also caught a - ``TypeError`` raised from *inside* a terminate that did accept ``force``, - and then retried without it -- two DELETEs for one deployment, the second - of them not forced, which is the one shape that leaves a workspace group - behind. + from the call: that also caught a ``TypeError`` raised from *inside* a + terminate which did accept ``force``, and retried without it -- two DELETEs, + the second unforced, which is what leaves a workspace group behind. """ import inspect @@ -682,25 +632,19 @@ def terminate( """ Terminate a deployment, whatever kind it is, retrying a 4xx refusal. - A deployment killed mid-provision is ``PENDING``/``TRANSITIONING``, and the - API refuses to delete it in that state with a 400 or a 409. Nothing retries - that: ``Manager.RETRY_STATUSES`` is ``{429, 500, 502, 503, 504}`` - (``management/manager.py:49``), urllib3 only retries what is in that list, - and there is no wait-until-deletable helper anywhere in the SDK. So the - per-class sweep logged a warning, the session-end sweep tried exactly once - more -- usually still too early -- and the deployment was left running. - - Hence the bounded retry here. Only 4xx other than 404 is retried: - - * 404 means it is already gone, so retrying would burn the whole budget - waiting for something that will never come back. Re-raised, as before, - which is also what ``_is_gone()`` upstream normally prevents. - * 5xx and 429 are already retried inside the transport, so seeing one here - means the transport gave up; another round trip from this layer is not - what fixes it. - - Raises the last error if the budget runs out, so ``cleanup_tracked()`` - keeps the deployment tracked and the session-end sweep gets another go. + A deployment killed mid-provision is ``PENDING``/``TRANSITIONING`` and the + API refuses to delete it, with a 400 or a 409. Nothing else retries that -- + ``Manager.RETRY_STATUSES`` covers only ``{429, 500, 502, 503, 504}`` -- so + the per-class sweep warned, the session-end sweep tried once more, usually + still too early, and the deployment stayed up. Hence the bounded retry. + + Only 4xx other than 404 is retried. A 404 means it is already gone, so + retrying would burn the budget on something that is not coming back; a 5xx + or 429 has already been retried by the transport, and another round trip + from this layer is not what fixes it. + + Raises the last error if the budget runs out, which keeps the deployment in + ``_tracked`` so the session-end sweep gets another go. """ import time @@ -756,10 +700,9 @@ def _creator_is_mocked(target: Any) -> bool: #: automatic, so a new test cannot leak a cluster by forgetting to register it. #: #: ``kind`` is the ledger kind the call produces, and must be a value of -#: :data:`_KIND_BY_CLASS`: it is what lets the ``pending`` record -- written -#: before the POST, when nothing has an id yet -- say which manager the reaper -#: should search. It is stated here rather than derived from ``method_name`` -#: because ``create_workspace`` appears twice, on two different receivers. +#: :data:`_KIND_BY_CLASS`: it tells the reaper which manager to search for a +#: ``pending`` record, which has no id. Stated here rather than derived from +#: ``method_name``, which ``create_workspace`` shares across two receivers. #: #: ``finder`` takes the receiver -- the manager, or the group for #: ``WorkspaceGroup.create_workspace`` -- and returns the collection to search @@ -817,27 +760,20 @@ def _tracking_wrapper(func: Any, kind: str, finder: Any) -> Any: (see :func:`recover_in_flight`). ``_creator_is_mocked``, not ``_is_mocked``: the receiver is the manager (or - the workspace group), and ``_is_mocked`` looks for a ``_manager`` - attribute, which a manager does not have -- so a real manager with a - patched ``_post`` would read as live and the recovery would fire a real - API call from a unit test. ``_creator_is_mocked`` inspects the receiver's - own transport and handles both receiver shapes. - - That same verdict also decides whether the *result* is tracked, rather than - leaving it to ``track()``. ``track()`` can only judge what it is handed, - and it is deliberately biased toward "real" for anything it cannot place -- - including an object whose ``_manager`` is ``None``, which is exactly what a - unit test's stubbed ``get_cluster`` returns. Nothing a mocked creator - returns names a deployment that exists, so the receiver's verdict is the - authoritative one and it is the one used here. - - The ``pending`` ledger record is written here, and deliberately *before* - ``func`` is called rather than after: from the moment the creator POSTs - there is a billable deployment, and everything that could record it -- - ``track()`` on return, ``_recover_orphan()`` in the ``except``, - ``recover_in_flight()`` from a signal handler -- runs after the wait that - a cancelled CI job never survives. A ``pending`` line on disk is the only - thing that outlives a SIGKILL there. + the workspace group), and ``_is_mocked`` looks for a ``_manager`` attribute, + which a manager does not have -- so a real manager with a patched ``_post`` + would read as live and the recovery would fire a real API call from a unit + test. ``_creator_is_mocked`` inspects the receiver's own transport. + + That same verdict decides whether the *result* is tracked, rather than + leaving it to ``track()``, which can only judge what it is handed and is + biased toward "real" for anything it cannot place -- including the + ``_manager is None`` object a stubbed ``get_cluster`` returns. + + The ``pending`` ledger record is written *before* ``func`` is called. From + the POST onward something is billable, and everything else that could record + it -- ``track()`` on return, ``_recover_orphan()``, ``recover_in_flight()`` + -- runs after the wait a cancelled CI job never survives. """ import functools @@ -1001,10 +937,8 @@ def cleanup_tracked(owner: Optional[str] = None) -> List[str]: _, label, obj = entry if _is_gone(obj): _tracked.remove(entry) - # The server says it is gone, which is exactly what the ledger's - # ``gone`` means -- a test that terminated in its own teardown gets - # its record closed here rather than leaving the reaper to look up - # an id that 404s. + # A test that terminated in its own teardown: close the record here + # rather than leaving the reaper to look up an id that 404s. ledger_gone(obj) continue try: @@ -1036,56 +970,47 @@ def tracked_labels() -> List[str]: # # Shared deployment pool # -# Several classes need nothing from a deployment but that it is live: the -# Stage and Job suites read and write through the management API against -# whatever cluster they are handed. Deploying one apiece cost 2190s of the -# 8915s a traced run took, and an S-00 cluster reaching ACTIVE is ~460s that -# cannot be made faster -- so the only lever is deploying fewer of them. +# Several classes need nothing from a deployment but that it is live: the Stage +# and Job suites read and write through the management API against whatever +# cluster they are handed. Deploying one apiece cost 2190s of the 8915s a traced +# run took, and an S-00 cluster reaching ACTIVE is ~460s that cannot be made +# faster -- so the only lever is deploying fewer of them. # # The pool is built on first use and reused for the rest of the process. A -# class must not mutate what it borrows, so anything whose subject *is* the +# borrower must not mutate what it borrows, so anything whose subject *is* the # deployment keeps deploying its own: ``TestCluster`` and ``TestWorkspace`` -# (``test_update`` PATCHes the cluster and cycles it back through PENDING), -# ``TestClusterFusionCreateDrop`` and ``TestClusterFusionSuspendResume``. So -# does ``TestWorkspaceFusion``, whose workspace groups are the subject of its -# ``SHOW WORKSPACE GROUPS`` assertions and cost 40s to deploy unwaited anyway. +# (``test_update`` PATCHes the cluster back through PENDING), +# ``TestClusterFusionCreateDrop``, ``TestClusterFusionSuspendResume`` and +# ``TestWorkspaceFusion``. # -# What makes the borrowers safe is that each scopes its assertions to itself: -# every Stage path is namespaced with the class's ``cls.id``, job listings -# filter by job id rather than listing a deployment's jobs, and none of them -# asserts a row count over an org-wide listing. +# Each borrower also scopes its assertions to itself -- Stage paths namespaced +# with ``cls.id``, job listings filtered by job id -- so none of them asserts a +# row count over an org-wide listing. ``TestClusterFusion`` does count rows, +# ``SHOW CLUSTERS ... LIKE`` being what it tests, and stays inside that rule by +# counting :func:`shared_cluster_pattern` against :func:`shared_cluster_names`. # -# ``TestClusterFusion`` is the one borrower that does count rows, because -# ``SHOW CLUSTERS ... LIKE`` is what it tests. It stays inside that rule by -# counting over :func:`shared_cluster_pattern` -- which matches this process's -# pool and nothing else -- against :func:`shared_cluster_names` rather than a -# literal, so growing the pool cannot break it. -# -# The pool is process-wide, so under ``pytest-xdist`` every worker that gets a -# borrowing class builds a pool of its own. The ``xdist_group`` marks below -# keep the borrowers together on a worker; see ``SHARED_CLUSTER_*_GROUP``. +# The pool is process-wide, so under ``pytest-xdist`` every worker with a +# borrowing class builds one of its own. The ``xdist_group`` marks below keep the +# borrowers together on a worker. # #: ``xdist_group`` names for the classes that borrow from the pool, so -#: ``--dist loadgroup`` puts each set on one worker and each set builds one -#: pool. Two groups rather than one: a single group serialises every borrower -#: behind one pool build, and the groups run concurrently on separate workers, -#: so splitting costs one extra cluster and halves that chain. -#: -#: The split follows what each set borrows -- three for Stage, one for Jobs: +#: ``--dist loadgroup`` puts each set on one worker and each set builds one pool. +#: Two groups rather than one: a single group serialises every borrower behind one +#: pool build, where these two run concurrently on separate workers for the cost +#: of one extra cluster. #: #: * ``SHARED_CLUSTER_STAGE_GROUP`` -- ``TestStageFusion`` (two; it names a #: second in ``IN GROUP``), v2 ``TestStage`` (one), ``TestClusterFusion`` #: (three, for its ``LIKE``/``ORDER BY``/``LIMIT`` rows) #: * ``SHARED_CLUSTER_JOBS_GROUP`` -- ``TestJobsFusion``, v2 ``TestJob`` #: -#: ``TestClusterFusion`` sits with Stage rather than Jobs deliberately: the -#: pool grows to the largest request, so putting the class that wants three -#: with the group that already wants two costs one extra cluster, where -#: putting it with Jobs would cost two and leave Stage's pool untouched. +#: ``TestClusterFusion`` sits with Stage deliberately: the pool grows to the +#: largest request, so the class that wants three costs Stage's pool one extra +#: cluster, against two if it joined Jobs. #: -#: Without ``-n``/``--dist loadgroup`` the marks do nothing: one process, one -#: pool of three, which is the serial behaviour they were added on top of. +#: Without ``-n``/``--dist loadgroup`` the marks do nothing: one process, one pool +#: of three. SHARED_CLUSTER_STAGE_GROUP = 'shared-cluster-stage' SHARED_CLUSTER_JOBS_GROUP = 'shared-cluster-jobs' @@ -1187,13 +1112,11 @@ def shared_cluster_pattern() -> str: """ ``LIKE`` pattern matching this process's pool clusters and nothing else. - The suffix is what scopes it: ``_pool_id`` is minted per process, so a - concurrent run's pool -- or another xdist worker's -- does not match, and - neither does any other ``cl-test-*`` deployment. + The suffix scopes it: ``_pool_id`` is minted per process, so another xdist + worker's pool -- or any other ``cl-test-*`` deployment -- does not match. - For a suite asserting an exact row count over the pool, pair it with - :func:`shared_cluster_names` rather than a literal: the pool grows to the - largest request any class makes, so the number is not fixed at import. + Pair it with :func:`shared_cluster_names`, not a literal count: the pool + grows to the largest request any class makes. """ return f'cl-test-shared-%-{_pool_id}' @@ -1202,9 +1125,8 @@ def shared_cluster_names() -> List[str]: """ Names of every cluster in the pool as it stands right now. - Read at assertion time, not cached: a class that runs later and asks for - more clusters than this one did grows the pool, and an expectation built - from a literal count would go stale the moment that happened. + Read at assertion time, not cached: a later class asking for more clusters + grows the pool, which would leave a cached expectation stale. """ return [x.name for x in _pool] From 67cfceda208ef42ed8c57513bde00813b32d973b Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Fri, 25 Sep 2026 19:00:40 -0400 Subject: [PATCH 26/30] Give each run its own secret, and sweep the ones a kill strands TestSecrets used a fixed secret name and opened by deleting any leftover of it. A secret is org-scoped, so that leftover is a concurrent run's live secret as often as a stranded one: two runs at once each deleted what the other had just created. Name it per-run instead, which makes the leftover-clearing delete unnecessary and the in-test delete the cleanup. Nothing else sweeps a secret as it is made, so a run killed between the POST and the DELETE strands one for good. Add a --secrets mode to cleanup_deployments.py for those, age-guarded so it cannot reap the run calling it, and run it from the always() cleanup step in both workflows. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 7 +- .github/workflows/coverage.yml | 20 +- singlestoredb/tests/cleanup_deployments.py | 209 +++++++++++++++++++ singlestoredb/tests/test_management_utils.py | 200 ++++++++++++++++++ singlestoredb/tests/test_management_v1.py | 29 +-- singlestoredb/tests/test_management_v2.py | 28 +-- 6 files changed, 455 insertions(+), 38 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index bbeb266f3..01d0004c2 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -233,9 +233,14 @@ jobs: # covers a cancel whose clusters are already ACTIVE but not one in the # first few minutes, where DELETE is still refused. That remainder needs # `cleanup_deployments.py --older-than` run by hand. - - name: Terminate any deployment the tests left behind + # + # The secret sweep alongside it is the same rolling janitor coverage.yml + # runs; see the comment on that step. + - name: Clean up what the tests left behind if: always() run: | + python -m singlestoredb.tests.cleanup_deployments --secrets --yes \ + || true python -m singlestoredb.tests.cleanup_deployments \ --ledger "$SINGLESTOREDB_TEST_DEPLOYMENT_LOG" --yes env: diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 0f6e932f6..edec14a54 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -108,9 +108,17 @@ jobs: # covered, being 19 minutes in; a cancel in the first few minutes would be # killed here still getting 400/409, and needs `cleanup_deployments.py # --older-than` run by hand. - - name: Terminate any deployment the tests left behind + - name: Clean up what the tests left behind if: always() run: | + # Secrets first, and its status discarded. TestSecrets creates an + # org-scoped secret that only its own test body deletes, so a killed + # run strands one for good; --secrets is age-guarded, which keeps it + # off this run's and off a concurrent job's, so what it removes is + # what *earlier* runs stranded. A secret bills nothing, and the + # ledger sweep is what this step's status should report. + python -m singlestoredb.tests.cleanup_deployments --secrets --yes \ + || true python -m singlestoredb.tests.cleanup_deployments \ --ledger "$SINGLESTOREDB_TEST_DEPLOYMENT_LOG" --yes env: @@ -184,12 +192,14 @@ jobs: SINGLESTOREDB_MANAGEMENT_TOKEN: ${{ secrets.CLUSTER_API_KEY }} SINGLESTOREDB_FUSION_ENABLE_HIDDEN: "1" - # See the matching step in test-coverage for why this is if: always(). - # The v1 suite deploys workspace groups, which are the kind that force - # exists for. - - name: Terminate any deployment the tests left behind + # See the matching step in test-coverage for why this is if: always(), + # and for what the secret sweep is doing here. The v1 suite deploys + # workspace groups, which are the kind that force exists for. + - name: Clean up what the tests left behind if: always() run: | + python -m singlestoredb.tests.cleanup_deployments --secrets --yes \ + || true python -m singlestoredb.tests.cleanup_deployments \ --ledger "$SINGLESTOREDB_TEST_DEPLOYMENT_LOG" --yes env: diff --git a/singlestoredb/tests/cleanup_deployments.py b/singlestoredb/tests/cleanup_deployments.py index 745069004..478e07e8b 100644 --- a/singlestoredb/tests/cleanup_deployments.py +++ b/singlestoredb/tests/cleanup_deployments.py @@ -55,6 +55,17 @@ minutes old by construction, so the age filter would spare every one of them. What keeps it off other people's deployments instead is that it touches only ids and names the ledger records, and that each CI job writes its own ledger. + +Finally, ``--secrets`` sweeps a different subject: the org-scoped secrets +``TestSecrets.test_get_secret`` creates. They bill nothing, but they are +permanent, and the test deletes its own only if it is not killed mid-test:: + + python -m singlestoredb.tests.cleanup_deployments --secrets --yes + +That is a rolling janitor rather than a run-scoped cleanup -- a secret is named +per-run but a name still says nothing about *which* run, so the age guard is +what keeps this off a live one. It cannot reap the run it is called from; what +it removes is what earlier runs stranded. """ import argparse import datetime @@ -141,6 +152,31 @@ ] +#: Secret names the suite generates. A secret is not a deployment -- it bills +#: nothing and lives on its own route -- so these are swept only when +#: ``--secrets`` asks for it, and never alongside the deployment patterns. +SECRET_PATTERNS = [ + # TestSecrets.test_get_secret, v1 and v2 + re.compile(r'^secret_v[12]_test_[0-9a-f]+$'), +] + +#: Secret names earlier revisions generated. Both are fixed rather than +#: per-run, which is what let two concurrent runs delete each other's secret; +#: ``main`` still creates them, so they are still reaped. +LEGACY_SECRET_PATTERNS = [ + re.compile(r'^secret_name$'), + re.compile(r'^secret_v2_test$'), +] + +#: Hours a secret must have existed before it is treated as stranded. Far lower +#: than :data:`DEFAULT_MIN_AGE_HOURS`, because the window it guards is far +#: shorter: the test creates a secret and deletes it in the same test body, a +#: second or two apart, so no secret a live run owns is even minutes old. Not +#: zero, because a run killed between the POST and the DELETE looks exactly +#: like one that is still between them. +DEFAULT_SECRET_MIN_AGE_HOURS = 1.0 + + def is_test_deployment(name: Optional[str]) -> bool: """Was this name generated by the test suite, now or in the past?""" if not name: @@ -148,6 +184,13 @@ def is_test_deployment(name: Optional[str]) -> bool: return any(x.match(name) for x in PATTERNS + LEGACY_PATTERNS) +def is_test_secret(name: Optional[str]) -> bool: + """Was this secret name generated by the test suite, now or in the past?""" + if not name: + return False + return any(x.match(name) for x in SECRET_PATTERNS + LEGACY_SECRET_PATTERNS) + + def _created_at(obj: Any) -> Optional[datetime.datetime]: """When this deployment was created, or None if the API did not say.""" created = getattr(obj, 'created_at', None) @@ -319,6 +362,138 @@ def keep(obj: Any) -> bool: return found, spared, unmatched +# +# Secret mode +# +# Why this is separate from everything above: a secret is org-scoped and +# permanent, it costs nothing to leave lying around, and it is reached through +# ``secrets`` rather than through any deployment listing. It is here because +# ``TestSecrets.test_get_secret`` is the one test that creates an org-scoped +# named object, and nothing in the suite sweeps one as it is created -- a run +# killed between its POST and its DELETE strands a secret for good. +# + + +def find_stranded_secrets( + mgr: Any, + older_than: float = DEFAULT_SECRET_MIN_AGE_HOURS, + include_unknown_age: bool = False, +) -> Tuple[List[Tuple[str, Any]], List[str], List[str]]: + """ + List the organization's secrets that the test suite stranded. + + Returns the same three lists as :func:`find_leftovers` -- the secrets to + delete, labels for the ones the age guard held back, and labels for the + ones whose names :data:`SECRET_PATTERNS` does not recognize. + + Only one version's manager is needed: ``secrets`` is identical at v1 and + v2 (see ``management/v2/organization.py``). + """ + from singlestoredb.management.organization import Secret + + found: List[Tuple[str, Any]] = [] + spared: List[str] = [] + unmatched: List[str] = [] + + # Verified live only this far: ``GET secrets`` with no parameters is + # accepted and answers with a ``secrets`` array -- the ``?name=`` form is + # all ``Organization.get_secret`` ever sends. UNVERIFIED: that the array is + # *every* secret in the organization rather than a page of them. It could + # not be shown against an organization that has none; if the route turns + # out to paginate, a sweep here is incomplete rather than wrong. + res = mgr._get('secrets') + for item in res.json().get('secrets') or []: + secret = Secret.from_dict(item) + + if secret.deleted_at is not None: + continue + + age = _age_hours(secret) + + if not is_test_secret(secret.name): + unmatched.append( + '{}{}'.format( + secret.name or '', + '' if age is None else f' ({age:.1f}h old)', + ), + ) + continue + + if age is None: + if not include_unknown_age: + spared.append(f'{secret.name} (creation time not reported)') + continue + elif older_than > 0 and age < older_than: + spared.append(f'{secret.name} ({age:.1f}h old, too new)') + continue + + found.append((f'secret {secret.name} ({secret.id})', secret)) + + return found, spared, unmatched + + +def _run_secret_sweep( + older_than: float, + include_unknown_age: bool, + yes: bool, + show_unmatched: bool, +) -> int: + """Report, and with ``yes`` delete, the secrets the suite stranded.""" + mgr = _manager('v2') + + try: + leftovers, spared, unmatched = find_stranded_secrets( + mgr, older_than, include_unknown_age, + ) + except Exception as exc: + # Reported, not raised: this runs as a cleanup step, and a secret bills + # nothing, so failing the job over one is the wrong trade. + print(f'! Could not list secrets: {exc}', file=sys.stderr) + return 1 + + if show_unmatched: + if unmatched: + print( + f'{len(unmatched)} secret(s) not recognized as the suite\'s, ' + 'and so never swept:', + ) + for label in sorted(unmatched): + print(f' ? {label}') + print() + else: + print('Every secret is recognized by SECRET_PATTERNS.\n') + + if spared: + print(f'{len(spared)} match(es) left alone by the age filter:') + for label in spared: + print(f' - {label}') + print() + + if not leftovers: + print('No stranded test secrets found.') + return 0 + + print(f'{len(leftovers)} stranded test secret(s):') + for label, _ in leftovers: + print(f' - {label}') + + if not yes: + print('\nDry run; pass --yes to delete these.') + return 0 + + failed = 0 + for label, secret in leftovers: + try: + mgr._delete(f'secrets/{secret.id}') + except Exception as exc: + failed += 1 + print(f'✗ {label}: {exc}') + else: + print(f'✓ deleted {label}') + + return 1 if failed else 0 + + # # Ledger mode # @@ -602,6 +777,15 @@ def main(argv: Optional[List[str]] = None) -> int: 'patterns and the age filter, which would spare everything in it ' 'for being minutes old. CI runs this as an if: always() step', ) + parser.add_argument( + '--secrets', action='store_true', + help='sweep stranded org-scoped secrets instead of deployments (see ' + 'SECRET_PATTERNS). Honours --yes, --older-than, ' + f'--include-unknown-age and --show-unmatched; --older-than ' + f'defaults to {DEFAULT_SECRET_MIN_AGE_HOURS} here rather than ' + f'{DEFAULT_MIN_AGE_HOURS}, since a secret a live run owns is ' + 'seconds old, not hours', + ) parser.add_argument( '--older-than', type=float, default=DEFAULT_MIN_AGE_HOURS, metavar='HOURS', @@ -659,11 +843,36 @@ def main(argv: Optional[List[str]] = None) -> int: ('--any-name', args.any_name), ('--kind', bool(args.kinds)), ('--show-unmatched', args.show_unmatched), + ('--secrets', args.secrets), ): if value: parser.error(f'{flag} does not apply with --ledger') return _run_ledger_sweep(args.ledger, args.yes) + # A different subject, not a different filter: --secrets sweeps secrets + # *instead of* deployments, so the flags that select deployments do not + # compose with it either. + if args.secrets: + for flag, value in ( + ('--since', args.since is not None), + ('--any-name', args.any_name), + ('--kind', bool(args.kinds)), + ): + if value: + parser.error(f'{flag} does not apply with --secrets') + # An explicit --older-than 6 is indistinguishable from the default + # here, which costs nothing: it is the value the caller asked for + # either way. + older_than = ( + DEFAULT_SECRET_MIN_AGE_HOURS + if args.older_than == DEFAULT_MIN_AGE_HOURS + else args.older_than + ) + return _run_secret_sweep( + older_than, args.include_unknown_age, args.yes, + args.show_unmatched, + ) + kinds = args.kinds or list(KINDS) leftovers, spared, unmatched = find_leftovers( diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index d48ae2a48..337e8bc76 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -2690,6 +2690,206 @@ def test_a_since_that_is_not_a_date_is_rejected(self): self.mod.parse_since('last tuesday') +class TestStrandedSecretPatterns(unittest.TestCase): + """ + ``--secrets`` deletes org-scoped objects in a real organization, and a + secret nobody can read back is not recoverable, so the name gate and the + age gate both matter more here than they do for a deployment. + """ + + def setUp(self): + from singlestoredb.tests import cleanup_deployments + self.mod = cleanup_deployments + + def test_generated_names_match(self): + for name in ( + 'secret_v1_test_deadbeef', + 'secret_v2_test_deadbeef', + ): + self.assertTrue(self.mod.is_test_secret(name), name) + + def test_retired_names_still_match(self): + # The fixed names, which main still creates. Reaping them is the only + # thing that removes one a killed run stranded. + for name in ('secret_name', 'secret_v2_test'): + self.assertTrue(self.mod.is_test_secret(name), name) + + def test_names_a_person_chose_do_not_match(self): + for name in ( + None, + '', + 'secret', + 'openai_api_key', + 'secret_v3_test_deadbeef', + 'prod_secret_name', + 'secret_name_prod', + ): + self.assertFalse(self.mod.is_test_secret(name), name) + + def _secret(self, name, hours=None, deleted=False): + """A secret as the API reports one in the listing.""" + created = None + if hours is not None: + created = ( + datetime.datetime.now(tz=datetime.timezone.utc) + - datetime.timedelta(hours=hours) + ).isoformat() + return dict( + secretID=f'id-{name}', + name=name, + createdBy='someone', + createdAt=created, + lastUpdatedBy='someone', + lastUpdatedAt=created, + deletedAt=created if deleted else None, + ) + + def _manager(self, *items): + mgr = MagicMock() + mgr._get.return_value.json.return_value = dict(secrets=list(items)) + return mgr + + def _find(self, *items, **kwargs): + mgr = self._manager(*items) + found, spared, self.unmatched = self.mod.find_stranded_secrets( + mgr, **kwargs, + ) + self.listed = mgr._get.call_args[0][0] + return [x[1].name for x in found], spared + + def test_the_listing_asks_for_every_secret(self): + # Not ?name=: the point is to find names this process never chose. + self._find() + self.assertEqual(self.listed, 'secrets') + + def test_the_age_filter_spares_a_secret_a_live_run_may_own(self): + names, spared = self._find( + self._secret('secret_v2_test_deadbeef', hours=5), + self._secret('secret_v2_test_beefcafe', hours=0.01), + ) + self.assertEqual(names, ['secret_v2_test_deadbeef']) + self.assertEqual(len(spared), 1) + self.assertIn('secret_v2_test_beefcafe', spared[0]) + + def test_the_default_age_is_shorter_than_the_deployment_one(self): + # The window guarded is a test body, not a suite: a secret is created + # and deleted seconds apart. Still not zero -- a run killed between the + # POST and the DELETE looks like one still between them. + self.assertGreater(self.mod.DEFAULT_SECRET_MIN_AGE_HOURS, 0) + self.assertLess( + self.mod.DEFAULT_SECRET_MIN_AGE_HOURS, + self.mod.DEFAULT_MIN_AGE_HOURS, + ) + + def test_an_unreported_creation_time_is_spared_by_default(self): + names, spared = self._find(self._secret('secret_v2_test_ace0')) + self.assertEqual(names, []) + self.assertIn('secret_v2_test_ace0', spared[0]) + + names, _ = self._find( + self._secret('secret_v2_test_ace0'), include_unknown_age=True, + ) + self.assertEqual(names, ['secret_v2_test_ace0']) + + def test_an_unrecognized_name_is_reported_not_swept(self): + names, _ = self._find( + self._secret('secret_v2_test_cafe', hours=10), + self._secret('openai_api_key', hours=10), + ) + self.assertEqual(names, ['secret_v2_test_cafe']) + self.assertEqual(len(self.unmatched), 1) + self.assertIn('openai_api_key', self.unmatched[0]) + + def test_an_already_deleted_secret_is_ignored_entirely(self): + # Neither swept nor reported as unrecognized: it is already gone, so + # there is nothing for a reader of the output to act on. + names, spared = self._find( + self._secret('secret_v2_test_0ff0', hours=10, deleted=True), + self._secret('someones_deleted_key', hours=10, deleted=True), + ) + self.assertEqual((names, spared, self.unmatched), ([], [], [])) + + +class TestStrandedSecretSweep(unittest.TestCase): + """``--secrets`` end to end, with the management API stubbed out.""" + + def setUp(self): + from singlestoredb.tests import cleanup_deployments + self.mod = cleanup_deployments + self.mgr = MagicMock() + self.mgr._get.return_value.json.return_value = dict( + secrets=[ + dict( + secretID='id-old', name='secret_v2_test_deadbeef', + createdBy='x', lastUpdatedBy='x', lastUpdatedAt=None, + createdAt=( + datetime.datetime.now(tz=datetime.timezone.utc) + - datetime.timedelta(hours=10) + ).isoformat(), + ), + dict( + secretID='id-new', name='secret_v2_test_beefcafe', + createdBy='x', lastUpdatedBy='x', lastUpdatedAt=None, + createdAt=datetime.datetime.now( + tz=datetime.timezone.utc, + ).isoformat(), + ), + ], + ) + patcher = patch.object( + self.mod, '_manager', lambda version: self.mgr, + ) + patcher.start() + self.addCleanup(patcher.stop) + + def deleted(self): + return [x[0][0] for x in self.mgr._delete.call_args_list] + + def test_the_old_secret_is_deleted_and_the_new_one_is_not(self): + self.assertEqual(self.mod.main(['--secrets', '--yes']), 0) + self.assertEqual(self.deleted(), ['secrets/id-old']) + + def test_a_dry_run_deletes_nothing(self): + self.assertEqual(self.mod.main(['--secrets']), 0) + self.assertEqual(self.deleted(), []) + + def test_older_than_is_honoured(self): + self.assertEqual( + self.mod.main(['--secrets', '--older-than', '20', '--yes']), 0, + ) + self.assertEqual(self.deleted(), []) + + def test_no_deployment_listing_is_touched(self): + # --secrets is a different subject, not an extra filter: asking for it + # must not walk the clusters or the workspace groups. + self.mod.main(['--secrets', '--yes']) + self.mgr.clusters.__iter__.assert_not_called() + + def test_a_listing_failure_is_reported_rather_than_raised(self): + # A cleanup step, and a secret bills nothing: failing the job over one + # is the wrong trade. Non-zero, so the log still says something went + # wrong. + self.mgr._get.side_effect = ManagementError(msg='no such route') + self.assertEqual(self.mod.main(['--secrets', '--yes']), 1) + + def test_a_failed_delete_exits_non_zero(self): + self.mgr._delete.side_effect = ManagementError(msg='nope') + self.assertEqual(self.mod.main(['--secrets', '--yes']), 1) + + def test_the_deployment_selectors_do_not_compose_with_it(self): + import contextlib + import io + for argv in ( + ['--secrets', '--kind', 'cluster'], + ['--secrets', '--any-name'], + ['--secrets', '--since', 'today'], + ['--secrets', '--ledger', 'x.jsonl'], + ): + with self.assertRaises(SystemExit, msg=argv), \ + contextlib.redirect_stderr(io.StringIO()): + self.mod.main(argv) + + class TestToDatetime(unittest.TestCase): """ ``to_datetime`` has to read both timestamp shapes the API returns. diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index e0f2357f6..1b03af82a 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -935,37 +935,30 @@ def tearDownClass(cls): cls.manager = None def test_get_secret(self): - # manually create secret and then get secret - # try to delete the secret if it exists - try: - secret = self.manager.organizations.current.get_secret('secret_name') - - secret_id = secret.id - - self.manager._delete(f'secrets/{secret_id}') - except s2.ManagementError: - pass + # Per-run name; see the twin in test_management_v2.py for why the fixed + # 'secret_name' this used to carry -- and the leftover-clearing delete + # that a fixed name required -- had two concurrent runs deleting each + # other's secret. + name = f'secret_v1_test_{secrets.token_hex(4)}' created = self.manager._post( 'secrets', json=dict( - name='secret_name', + name=name, value='secret_value', ), ).json() # The ID comes from the create response, not from the lookup under # test: binding it inside the try would leave the cleanup raising - # UnboundLocalError over whatever the lookup failed with. Without this - # the secret outlived every run, removed only by the sweep at the top of - # the *next* one. test_management_v2.py's twin does it this way. + # UnboundLocalError over whatever the lookup failed with. This delete is + # the only thing that removes the secret now -- nothing else sweeps one + # as it is made. test_management_v2.py's twin does it this way. secret_id = created['secret']['secretID'] try: - secret = self.manager.organizations.current.get_secret( - 'secret_name', - ) + secret = self.manager.organizations.current.get_secret(name) - assert secret.name == 'secret_name' + assert secret.name == name assert secret.value == 'secret_value' finally: self.manager._delete(f'secrets/{secret_id}') diff --git a/singlestoredb/tests/test_management_v2.py b/singlestoredb/tests/test_management_v2.py index fe65e47df..2d50c3a8e 100644 --- a/singlestoredb/tests/test_management_v2.py +++ b/singlestoredb/tests/test_management_v2.py @@ -1765,20 +1765,20 @@ def tearDownClass(cls): cls.manager = None def test_get_secret(self): - # A fixed name, deliberately not one built from id(self): that is a - # process-local address, so a name built from it can never match what - # an interrupted run left behind, which makes the cleanup below dead - # code. A secret is org-scoped and permanent and nothing sweeps them, - # so a leaked one is leaked for good. Distinct from the v1 suite's - # 'secret_name' so the two suites do not delete each other's. - name = 'secret_v2_test' - - # Clear a leftover secret from a previous run - try: - leftover = self.manager.organizations.current.get_secret(name) - self.manager._delete(f'secrets/{leftover.id}') - except s2.ManagementError: - pass + # Per-run name. A secret is org-scoped, so a fixed one is shared with + # every other run in the organization -- and this test used to open by + # deleting any leftover of that fixed name, which is a concurrent run's + # live secret as often as a stranded one. Two runs at once then raced: + # each deleted what the other had just created, and the loser's + # get_secret() failed or read the wrong value. + # + # Not id(self) either: that is a process-local address, so two + # processes can mint the same name. + # + # Nothing sweeps secrets as they are created, so the delete below is + # the cleanup; cleanup_deployments.py --secrets reaps what a run killed + # between the POST and the DELETE strands. + name = f'secret_v2_test_{secrets.token_hex(4)}' created = self.manager._post( 'secrets', From b13dcd0931cc881678c55439b5a1f801dce378cd Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Fri, 25 Sep 2026 19:01:04 -0400 Subject: [PATCH 27/30] Generate admin passwords the API's policy actually accepts The suite built one as secrets.token_urlsafe(20) + '-x&$'. The token half can contain 'abc' or '321', which the API rejects for more than two consecutive sequential characters -- a low enough rate to read as a flake. Generate the whole password instead, redrawing any character that would close a run of three and requiring the mix of cases, digits and a special character the policy wants. Which characters count as special was probed against POST /v1/workspaceGroups, which checks the password before it looks the region up, so a nonexistent region ID makes an accepted password surface as the region's 404 and creates nothing: '-', '_', '$', '!', '@', '#', '%' and '*' all satisfy the rule and '&' does not. The old suffix passed on its '-' and '$'. Draw the specials from '-_$', the three that are also URL-unreserved or a sub-delimiter. Co-Authored-By: Claude Opus 5 --- singlestoredb/tests/test_management_utils.py | 36 +++++++++++++ singlestoredb/tests/test_management_v1.py | 8 +-- singlestoredb/tests/test_management_v2.py | 2 +- singlestoredb/tests/utils.py | 56 ++++++++++++++++++++ 4 files changed, 97 insertions(+), 5 deletions(-) diff --git a/singlestoredb/tests/test_management_utils.py b/singlestoredb/tests/test_management_utils.py index 337e8bc76..de9560239 100644 --- a/singlestoredb/tests/test_management_utils.py +++ b/singlestoredb/tests/test_management_utils.py @@ -23,6 +23,7 @@ from singlestoredb.management.utils import normalize_remote_path from singlestoredb.management.utils import to_datetime from singlestoredb.management.utils import to_datetime_strict +from singlestoredb.tests.utils import admin_password from singlestoredb.tests.utils import counting_file_space from singlestoredb.tests.utils import counting_stage @@ -3018,5 +3019,40 @@ def test_strict_raises_on_the_go_spelling_of_the_sentinel(self): to_datetime_strict('0001-01-01 00:00:00 +0000 UTC') +class TestAdminPassword(unittest.TestCase): + """The generated admin password must satisfy the API's policy on every + draw, not merely most of them: a ``secrets.token_urlsafe`` password + containing ``abc`` or ``321``, or one whose only punctuation is the ``&`` + the API does not count as special, is rejected with a 400 -- which the old + generator hit at a low enough rate to look like an API flake.""" + + #: Enough draws that a per-character rule would have to be enforced, not + #: just usually satisfied, to pass. A 24-character password holds 22 + #: three-character windows. + DRAWS = 2000 + + def test_the_policy_holds_on_every_draw(self): + for _ in range(self.DRAWS): + password = admin_password() + self.assertEqual(len(password), 24) + self.assertTrue(any(x.islower() for x in password), password) + self.assertTrue(any(x.isupper() for x in password), password) + self.assertTrue(any(x.isdigit() for x in password), password) + # A special character the API actually counts as one: `&` is + # accepted in a password but does not satisfy the rule. + self.assertTrue(any(x in '-_$' for x in password), password) + self.assertFalse('&' in password, password) + for i in range(len(password) - 2): + a, b, c = (ord(x) for x in password[i:i+3]) + # No three characters a step of -1, 0 or 1 apart in a row. + self.assertFalse(b - a == c - b and abs(b - a) <= 1, password) + + def test_the_length_is_honoured(self): + self.assertEqual(len(admin_password(32)), 32) + + def test_the_draws_differ(self): + self.assertEqual(len({admin_password() for _ in range(100)}), 100) + + if __name__ == '__main__': unittest.main() diff --git a/singlestoredb/tests/test_management_v1.py b/singlestoredb/tests/test_management_v1.py index 1b03af82a..7a6533d47 100755 --- a/singlestoredb/tests/test_management_v1.py +++ b/singlestoredb/tests/test_management_v1.py @@ -69,7 +69,7 @@ def setUpClass(cls): cls.manager = s2.manage_workspaces(version='v1') us_regions = [x for x in cls.manager.regions if 'US' in x.name] - cls.password = secrets.token_urlsafe(20) + '-x&$' + cls.password = utils.admin_password() name = clean_name(secrets.token_urlsafe(20)[:20]) @@ -268,7 +268,7 @@ def setUpClass(cls): # run on different xdist workers -- and with anything an earlier failed # run leaked. The API answers the collision with a bare 500. cls.starter_username = f'starter_user_{name[:8]}' - cls.password = secrets.token_urlsafe(20) + cls.password = utils.admin_password() cls.database_name = f'starter_db_{name}' @@ -374,7 +374,7 @@ def setUpClass(cls): cls.manager = s2.manage_workspaces(version='v1') us_regions = [x for x in cls.manager.regions if 'US' in x.name] - cls.password = secrets.token_urlsafe(20) + '-x&$' + cls.password = utils.admin_password() name = clean_name(secrets.token_urlsafe(20)[:20]) @@ -978,7 +978,7 @@ def setUpClass(cls): cls.manager = s2.manage_workspaces(version='v1') us_regions = [x for x in cls.manager.regions if 'US' in x.name] - cls.password = secrets.token_urlsafe(20) + '-x&$' + cls.password = utils.admin_password() name = clean_name(secrets.token_urlsafe(20)[:20]) diff --git a/singlestoredb/tests/test_management_v2.py b/singlestoredb/tests/test_management_v2.py index 2d50c3a8e..ce11fefa3 100644 --- a/singlestoredb/tests/test_management_v2.py +++ b/singlestoredb/tests/test_management_v2.py @@ -1517,7 +1517,7 @@ def setUpClass(cls): # xdist worker -- and with anything an earlier failed run leaked. The # API reports the collision as a bare 500. cls.starter_username = f'starter_user_{name[:8]}' - cls.password = secrets.token_urlsafe(20) + cls.password = utils.admin_password() cls.database_name = f'starter_db_{name}' diff --git a/singlestoredb/tests/utils.py b/singlestoredb/tests/utils.py index b2ec45ec8..f0b22e91f 100644 --- a/singlestoredb/tests/utils.py +++ b/singlestoredb/tests/utils.py @@ -8,6 +8,7 @@ import random import re import secrets +import string import unittest import uuid from types import SimpleNamespace @@ -287,6 +288,61 @@ def drop_user(name: str) -> None: cur.execute(f'DROP USER IF EXISTS {name};') +#: The characters the API counts towards `password must contain at least 1 +#: special characters`. Probed one character at a time against +#: `POST /v1/workspaceGroups`, which checks the password before it looks the +#: region up: `-`, `_`, `$`, `!`, `@`, `#`, `%` and `*` all satisfy the rule, +#: and `&` does not -- so a password whose only punctuation is an `&` is a +#: 400. The old hand-appended `-x&$` suffix passed on its `-` and `$`, not on +#: its `&`. These three are the ones that are also URL-unreserved or a +#: sub-delimiter, in case a password ever reaches a connection string. +_PASSWORD_SPECIALS = '-_$' + +#: Characters an admin password is drawn from. +_PASSWORD_ALPHABET = string.ascii_letters + string.digits + _PASSWORD_SPECIALS + + +def _runs_on(a: str, b: str, c: str) -> bool: + """Whether ``a b c`` is three characters in a row of the same step.""" + first, second = ord(b) - ord(a), ord(c) - ord(b) + return first == second and abs(first) <= 1 + + +def admin_password(length: int = 24) -> str: + """ + Return a password the management API will accept. + + The API enforces a policy the obvious ``secrets.token_urlsafe(20)`` does + not satisfy: the password must mix cases, digits and at least one special + character (see :data:`_PASSWORD_SPECIALS`), and it must not contain more + than two consecutive sequential characters -- a + token holding ``abc`` or ``321`` anywhere in it is rejected with a 400, + which made the old generator fail a small fraction of runs rather than + never. Identical runs (``aaa``) are excluded on the same terms, being the + same shape of rule and no loss of entropy worth keeping. + + Sequential is read on code points here, which is stricter than the letter + and digit sequences the API means but simpler, and it costs nothing: a + character that would close a run is redrawn, not the whole password. + + """ + while True: + chars: List[str] = [] + while len(chars) < length: + char = secrets.choice(_PASSWORD_ALPHABET) + if len(chars) >= 2 and _runs_on(chars[-2], chars[-1], char): + continue + chars.append(char) + password = ''.join(chars) + if ( + any(x.islower() for x in password) + and any(x.isupper() for x in password) + and any(x.isdigit() for x in password) + and any(x in _PASSWORD_SPECIALS for x in password) + ): + return password + + # # Live deployment tracking # From bab42505f34e513bd95b1c527d8bb4fd6aa154ca Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 28 Sep 2026 10:19:28 -0400 Subject: [PATCH 28/30] Run host Python 3.14 everywhere a version is pinned [skip ci] The single-version setup-python steps were on 3.10 or 3.11 for no reason that survives inspection: 3.10 is not the floor (pyproject.toml says >=3.9), and the supported range is already covered by the smoke-test and pre-commit matrices. So they now track the newest final release, with a comment at each saying that is what they are. Two kinds of step are involved. publish's cibuildwheel driver, the fusion-docs generator, and smoke-test's setup/shutdown jobs are pure host Pythons -- nothing about the package is exercised under them. coverage and code-check are not: that pin is the interpreter the test suite and the C extension build actually run on, which this moves to 3.14 as well. The smoke-test matrix's 3.11 include entries stay put. Those choose which supported version gets the macOS, Windows, and pure-Python runs, and 3.11 is the only one those platforms see. Co-Authored-By: Claude Opus 5 --- .github/workflows/code-check.yml | 5 ++++- .github/workflows/coverage.yml | 8 ++++++-- .github/workflows/fusion-docs.yml | 6 ++++-- .github/workflows/pre-commit.yml | 1 + .github/workflows/publish.yml | 5 +++-- .github/workflows/smoke-test.yml | 12 ++++++++---- 6 files changed, 26 insertions(+), 11 deletions(-) diff --git a/.github/workflows/code-check.yml b/.github/workflows/code-check.yml index 01d0004c2..ad5f54c9f 100644 --- a/.github/workflows/code-check.yml +++ b/.github/workflows/code-check.yml @@ -50,10 +50,13 @@ jobs: # management step never ran on a PR. fetch-depth: 0 + # One version, not a matrix: the supported range is covered by + # smoke-test.yml and pre-commit.yml, so this tracks the newest final + # release rather than the floor in pyproject.toml. - name: Set up Python uses: actions/setup-python@v7 with: - python-version: "3.10" + python-version: "3.14" cache: "pip" - name: Install dependencies diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index edec14a54..da4188472 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -30,10 +30,13 @@ jobs: steps: - uses: actions/checkout@v7 + # One version, not a matrix: the supported range is covered by + # smoke-test.yml and pre-commit.yml, so this tracks the newest final + # release rather than the floor in pyproject.toml. - name: Set up Python uses: actions/setup-python@v7 with: - python-version: "3.10" + python-version: "3.14" cache: "pip" - name: Install dependencies @@ -167,10 +170,11 @@ jobs: steps: - uses: actions/checkout@v7 + # As in test-coverage above: newest final release, not the floor. - name: Set up Python uses: actions/setup-python@v7 with: - python-version: "3.10" + python-version: "3.14" cache: "pip" - name: Install dependencies diff --git a/.github/workflows/fusion-docs.yml b/.github/workflows/fusion-docs.yml index bfbb35d01..c7b8ac5c8 100644 --- a/.github/workflows/fusion-docs.yml +++ b/.github/workflows/fusion-docs.yml @@ -16,10 +16,12 @@ jobs: steps: - uses: actions/checkout@v7 - - name: Set up Python 3.11 + # Only drives resources/gen_fusion_handlers_doc.py, so this tracks the + # newest final release rather than anything the package supports. + - name: Set up Python 3.14 uses: actions/setup-python@v7 with: - python-version: 3.11 + python-version: "3.14" cache: "pip" - name: Install dependencies diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml index aa6145076..ae1917e24 100644 --- a/.github/workflows/pre-commit.yml +++ b/.github/workflows/pre-commit.yml @@ -14,6 +14,7 @@ jobs: - "3.11" - "3.12" - "3.13" + - "3.14" steps: - uses: actions/checkout@v7 diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 459b30ec3..465cd4e4b 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -84,11 +84,12 @@ jobs: - uses: actions/checkout@v7 # This job's matrix varies only over os; cibuildwheel supplies its own - # interpreters, so 3.10 here is just the host Python that drives it. + # interpreters, so 3.14 here is just the host Python that drives it. + # Host-only pins track the newest final release rather than the floor. - name: Set up Python uses: actions/setup-python@v7 with: - python-version: "3.10" + python-version: "3.14" cache: "pip" - name: Install dependencies diff --git a/.github/workflows/smoke-test.yml b/.github/workflows/smoke-test.yml index 38f63e3f3..87140ffe2 100644 --- a/.github/workflows/smoke-test.yml +++ b/.github/workflows/smoke-test.yml @@ -14,10 +14,13 @@ jobs: steps: - uses: actions/checkout@v7 - - name: Set up Python 3.11 + # This job only drives resources/create_test_cluster.py; the versions the + # package is tested against are the smoke-test matrix below. So it tracks + # the newest final release. + - name: Set up Python 3.14 uses: actions/setup-python@v7 with: - python-version: 3.11 + python-version: "3.14" cache: "pip" - name: Install dependencies @@ -158,10 +161,11 @@ jobs: steps: - uses: actions/checkout@v7 - - name: Set up Python 3.11 + # As in setup-database: this only drives resources/drop_db.py. + - name: Set up Python 3.14 uses: actions/setup-python@v7 with: - python-version: 3.11 + python-version: "3.14" cache: "pip" - name: Install dependencies From 1e9ca15e5def96285cabfacdc6c1e62ae0009a6d Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 28 Sep 2026 12:42:14 -0400 Subject: [PATCH 29/30] Fix the Python 3.14 / numpy 2.5 failures the runner bump exposed Three things, all surfaced by moving CI to Python 3.14 (bab42505): asyncio.iscoroutinefunction is deprecated in 3.14 and slated for removal in 3.16; inspect.iscoroutinefunction is the replacement and has handled functools.partial and markcoroutinefunction for several releases. Seven call sites in the UDF decorator and the ASGI app. numpy 2.5 redefines npt.NDArray as a PEP 695 alias -- type NDArray[ScalarT] = ndarray[_AnyShape, dtype[ScalarT]] -- so get_origin of an NDArray annotation is the alias, not numpy.ndarray, and get_args is the scalar type, not the (shape, dtype) pair the signature code reads. Every typed numpy annotation died with 'unsupported type annotation'. resolve_type_alias expands an alias back to the subscripted-generic form so the introspection is numpy-version independent, and it handles a user-written alias too. The reason a DeprecationWarning was fatal rather than noisy: the vendored MySQLdb capabilities module called warnings.filterwarnings('error') at import, and its package __init__ imports it, so the filter went in process-wide as soon as anything imported those classes. Scoped to the tests that want it, with a reload test to keep it that way. Co-Authored-By: Claude Opus 5 --- singlestoredb/functions/decorator.py | 5 +- singlestoredb/functions/ext/asgi.py | 10 +-- singlestoredb/functions/signature.py | 4 ++ singlestoredb/functions/utils.py | 58 ++++++++++++++++ .../test_MySQLdb/test_MySQLdb_capabilities.py | 26 ++++++- singlestoredb/tests/test_dbapi.py | 20 ++++++ singlestoredb/tests/test_udf_type_aliases.py | 69 +++++++++++++++++++ 7 files changed, 182 insertions(+), 10 deletions(-) create mode 100644 singlestoredb/tests/test_udf_type_aliases.py diff --git a/singlestoredb/functions/decorator.py b/singlestoredb/functions/decorator.py index 3da98ff46..5239f27f6 100644 --- a/singlestoredb/functions/decorator.py +++ b/singlestoredb/functions/decorator.py @@ -1,4 +1,3 @@ -import asyncio import functools import inspect from typing import Any @@ -122,7 +121,7 @@ def _func( if func is None: def decorate(func: UDFType) -> UDFType: - if asyncio.iscoroutinefunction(func): + if inspect.iscoroutinefunction(func): async def async_wrapper(*args: Any, **kwargs: Any) -> UDFType: return await func(*args, **kwargs) # type: ignore async_wrapper._singlestoredb_attrs = _singlestoredb_attrs # type: ignore @@ -136,7 +135,7 @@ def wrapper(*args: Any, **kwargs: Any) -> UDFType: return decorate - if asyncio.iscoroutinefunction(func): + if inspect.iscoroutinefunction(func): async def async_wrapper(*args: Any, **kwargs: Any) -> UDFType: return await func(*args, **kwargs) # type: ignore async_wrapper._singlestoredb_attrs = _singlestoredb_attrs # type: ignore diff --git a/singlestoredb/functions/ext/asgi.py b/singlestoredb/functions/ext/asgi.py index af3dbd385..876cf9eab 100755 --- a/singlestoredb/functions/ext/asgi.py +++ b/singlestoredb/functions/ext/asgi.py @@ -522,7 +522,7 @@ def build_udf_endpoint( """ if returns_data_format in ['scalar', 'list']: - is_async = asyncio.iscoroutinefunction(func) + is_async = inspect.iscoroutinefunction(func) async def do_func( cancel_event: threading.Event, @@ -568,7 +568,7 @@ def build_vector_udf_endpoint( """ masks = get_masked_params(func) array_cls = get_array_class(returns_data_format) - is_async = asyncio.iscoroutinefunction(func) + is_async = inspect.iscoroutinefunction(func) async def do_func( cancel_event: threading.Event, @@ -633,7 +633,7 @@ def build_tvf_endpoint( """ if returns_data_format in ['scalar', 'list']: - is_async = asyncio.iscoroutinefunction(func) + is_async = inspect.iscoroutinefunction(func) async def do_func( cancel_event: threading.Event, @@ -698,7 +698,7 @@ async def do_func( # each result row, so we just have to use the same # row ID for all rows in the result. - is_async = asyncio.iscoroutinefunction(func) + is_async = inspect.iscoroutinefunction(func) # Call function on each column of data async with timer('call_function'): @@ -787,7 +787,7 @@ def make_func( info['timeout'] = max(timeout, 1) # Set async flag - info['is_async'] = asyncio.iscoroutinefunction(func) + info['is_async'] = inspect.iscoroutinefunction(func) # Setup argument types for rowdat_1 parser colspec = [] diff --git a/singlestoredb/functions/signature.py b/singlestoredb/functions/signature.py index 657611a65..0693d4a88 100644 --- a/singlestoredb/functions/signature.py +++ b/singlestoredb/functions/signature.py @@ -276,6 +276,7 @@ def simplify_dtype(dtype: Any) -> List[Any]: list of dtype strings, TupleCollections, and ArrayCollections """ + dtype = utils.resolve_type_alias(dtype) origin = typing.get_origin(dtype) atype = type(dtype) args = [] @@ -889,6 +890,7 @@ def get_schema( function_type = 'udf' udf_parameter = '`returns=`' if mode == 'return' else '`args=`' + spec = utils.resolve_type_alias(spec) spec, is_optional = unwrap_optional(spec) origin = typing.get_origin(spec) args = typing.get_args(spec) @@ -1151,6 +1153,8 @@ def vector_check(obj: Any) -> Tuple[Any, str]: 'scalar', 'list', 'numpy', 'pandas', or 'polars' """ + obj = utils.resolve_type_alias(obj) + if utils.is_numpy(obj): if len(typing.get_args(obj)) < 2: return None, 'numpy' diff --git a/singlestoredb/functions/utils.py b/singlestoredb/functions/utils.py index 5b948e2c4..7aecd0a34 100644 --- a/singlestoredb/functions/utils.py +++ b/singlestoredb/functions/utils.py @@ -36,6 +36,62 @@ def is_union(x: Any) -> bool: return typing.get_origin(x) in _UNION_TYPES +def _is_type_alias(obj: Any) -> bool: + """Check if an object is a PEP 695 type alias.""" + # Duck-typed rather than isinstance(obj, typing.TypeAliasType): the class + # only exists in 3.12+, and a library supporting older versions may be + # using the typing_extensions backport instead. + return hasattr(obj, '__value__') and hasattr(obj, '__type_params__') + + +def resolve_type_alias(obj: Any) -> Any: + """ + Expand a PEP 695 type alias to the type it stands for. + + numpy 2.5 redefined ``npt.NDArray`` as an alias -- + ``type NDArray[ScalarT] = ndarray[_AnyShape, dtype[ScalarT]]`` -- rather + than a subscripted generic. For ``NDArray[np.str_]`` that makes + ``typing.get_origin`` return the alias object instead of ``numpy.ndarray`` + and ``typing.get_args`` return ``(np.str_,)`` instead of the + ``(shape, dtype[...])`` pair the type checks here read. Expanding the alias + puts the annotation back into the subscripted-generic form, so the rest of + the introspection works the same on every numpy version. + + Parameters + ---------- + obj : Any + Python type annotation + + Returns + ------- + Any + The annotation with any type aliases expanded + + """ + while True: + origin = typing.get_origin(obj) + + # A subscripted alias: `NDArray[np.str_]`. Subscripting the alias's own + # value substitutes the arguments for its type parameters. This case is + # tested before the bare one below because a subscripted alias forwards + # attribute lookups to the alias it came from, so it answers to + # __value__ as well -- reading that here would drop the arguments. + if _is_type_alias(origin): + try: + obj = origin.__value__[typing.get_args(obj)] + except TypeError: + # Not substitutable; leave it for the caller to reject + return obj + continue + + # A bare alias used directly as an annotation: `type Vec = NDArray[f64]` + if origin is None and _is_type_alias(obj): + obj = obj.__value__ + continue + + return obj + + def get_annotations(obj: Any) -> Dict[str, Any]: """Get the annotations of an object.""" return typing.get_type_hints(obj) @@ -60,6 +116,8 @@ def get_type_name(obj: Any) -> str: def is_numpy(obj: Any) -> bool: """Check if an object is a numpy array.""" + obj = resolve_type_alias(obj) + if str(obj).startswith('numpy.ndarray['): return True diff --git a/singlestoredb/mysql/tests/thirdparty/test_MySQLdb/test_MySQLdb_capabilities.py b/singlestoredb/mysql/tests/thirdparty/test_MySQLdb/test_MySQLdb_capabilities.py index c1daf6b44..c197ef9a8 100644 --- a/singlestoredb/mysql/tests/thirdparty/test_MySQLdb/test_MySQLdb_capabilities.py +++ b/singlestoredb/mysql/tests/thirdparty/test_MySQLdb/test_MySQLdb_capabilities.py @@ -5,8 +5,6 @@ from . import capabilities from singlestoredb.mysql.tests import base -warnings.filterwarnings('error') - class test_MySQLdb(capabilities.DatabaseTest): @@ -25,6 +23,30 @@ class test_MySQLdb(capabilities.DatabaseTest): leak_test = False + # These tests want warnings raised as exceptions -- test_truncation asks the + # server for an over-long column and expects the driver to complain. Scoped + # to the test rather than set at module import: this package's __init__ + # imports this module, so a module-level warnings.filterwarnings('error') + # was installed process-wide the moment anything imported one of these + # classes (singlestoredb/tests/test_dbapi.py does), and every later warning + # in the session -- a DeprecationWarning from a dependency, say -- became a + # fatal error in an unrelated test. + def setUp(self): + self._warnings = warnings.catch_warnings() + self._warnings.__enter__() + warnings.simplefilter('error') + try: + super().setUp() + except Exception: + self._warnings.__exit__(None, None, None) + raise + + def tearDown(self): + try: + super().tearDown() + finally: + self._warnings.__exit__(None, None, None) + def quote_identifier(self, ident): return '`%s`' % ident diff --git a/singlestoredb/tests/test_dbapi.py b/singlestoredb/tests/test_dbapi.py index 8bacde56d..3242e3e56 100644 --- a/singlestoredb/tests/test_dbapi.py +++ b/singlestoredb/tests/test_dbapi.py @@ -1,8 +1,12 @@ # type: ignore +import importlib import os +import unittest +import warnings import singlestoredb as s2 from . import utils +from singlestoredb.mysql.tests.thirdparty.test_MySQLdb import test_MySQLdb_capabilities from singlestoredb.mysql.tests.thirdparty.test_MySQLdb import test_MySQLdb_dbapi20 @@ -25,3 +29,19 @@ def tearDownClass(cls): def _connect(self): return s2.connect(database=type(self).dbname) + + +class TestWarningFilters(unittest.TestCase): + """Importing the vendored MySQLdb tests must not escalate warnings.""" + + def test_capabilities_import_leaves_filters_alone(self): + # The capabilities module wants warnings raised as errors, but it must + # scope that to its own tests. It used to call + # warnings.filterwarnings('error') at module level, and since this + # module imports that package, the filter was installed process-wide + # and turned any later warning -- a DeprecationWarning from a + # dependency, say -- into a failure in an unrelated test. Reload to + # re-run the module body against a known set of filters. + before = list(warnings.filters) + importlib.reload(test_MySQLdb_capabilities) + self.assertEqual(warnings.filters, before) diff --git a/singlestoredb/tests/test_udf_type_aliases.py b/singlestoredb/tests/test_udf_type_aliases.py new file mode 100644 index 000000000..9a848cf7b --- /dev/null +++ b/singlestoredb/tests/test_udf_type_aliases.py @@ -0,0 +1,69 @@ +# mypy: disable-error-code="attr-defined,type-arg,valid-type,var-annotated" +""" +UDF annotations written with a PEP 695 type alias. + +numpy 2.5 defines ``npt.NDArray`` as an alias of +``ndarray[_AnyShape, dtype[ScalarT]]`` rather than as a subscripted generic, so +``typing.get_origin`` of an NDArray annotation returns the alias object instead +of ``numpy.ndarray``, and ``typing.get_args`` returns the scalar type instead of +the ``(shape, dtype)`` pair the signature machinery reads. The aliases below are +built with ``typing.TypeAliasType`` directly rather than taken from ``npt``, so +the alias form is covered whatever numpy is installed. + +These live in their own module because the type checker cannot follow an alias +built at runtime -- the file-level suppressions above would otherwise apply to +the hand-written annotations in ``test_udf_returns.py``. + +""" +import sys +import typing +import unittest +from typing import Any +from typing import Callable + +import numpy as np +import numpy.typing as npt + +from singlestoredb.functions import udf +from singlestoredb.functions.signature import get_signature +from singlestoredb.functions.signature import signature_to_sql + + +def to_sql(func: Callable[..., Any]) -> str: + """Convert a function signature to SQL.""" + out = signature_to_sql(get_signature(func)) + return out.split('EXTERNAL FUNCTION ')[1].split('AS REMOTE')[0].strip() + + +@unittest.skipIf( + sys.version_info < (3, 12), + 'PEP 695 type aliases require Python 3.12+', +) +class TypeAliasTest(unittest.TestCase): + + def test_subscripted_alias(self) -> None: + ScalarT = typing.TypeVar('ScalarT') + NDArrayAlias = typing.TypeAliasType( # noqa: TYP006 + 'NDArrayAlias', + np.ndarray[Any, np.dtype[ScalarT]], + type_params=(ScalarT,), + ) + + @udf + def foo_a(x: NDArrayAlias[np.str_]) -> NDArrayAlias[np.str_]: + return np.array([f'{i}: {v}' for i, v in enumerate(x)]) + + assert to_sql(foo_a) == '`foo_a`(`x` TEXT NOT NULL) RETURNS TEXT NOT NULL' + + def test_bare_alias(self) -> None: + Vec = typing.TypeAliasType('Vec', npt.NDArray[np.float64]) # noqa: TYP006 + + @udf + def foo_b(x: Vec) -> Vec: + return x * 2 + + assert to_sql(foo_b) == '`foo_b`(`x` DOUBLE NOT NULL) RETURNS DOUBLE NOT NULL' + + +if __name__ == '__main__': + unittest.main() From e88a93adf2125c6919ce676aac141902f4401baf Mon Sep 17 00:00:00 2001 From: Kevin Smith Date: Mon, 28 Sep 2026 13:14:16 -0400 Subject: [PATCH 30/30] Expand a type alias on both sides of the Optional unwrap `get_schema` resolved the annotation's alias once, before `unwrap_optional`, so it only ever saw the outermost layer. An `Optional` is a `Union`, not an alias, so `Optional[NDArray[...]]` -- every typed numpy annotation marked nullable, on numpy 2.5 -- came back from the unwrap still an alias, and the numpy branch read `typing.get_args` as a one-tuple and raised. Resolve again after the unwrap. Regression coverage for both the subscripted and the bare alias form, built with `TypeAliasType` so it exercises the numpy 2.5 shape whatever numpy is installed. Co-Authored-By: Claude Opus 5 --- singlestoredb/functions/signature.py | 4 +++ singlestoredb/tests/test_udf_type_aliases.py | 30 ++++++++++++++++++++ 2 files changed, 34 insertions(+) diff --git a/singlestoredb/functions/signature.py b/singlestoredb/functions/signature.py index 0693d4a88..346116b4f 100644 --- a/singlestoredb/functions/signature.py +++ b/singlestoredb/functions/signature.py @@ -892,6 +892,10 @@ def get_schema( spec = utils.resolve_type_alias(spec) spec, is_optional = unwrap_optional(spec) + # Again: the first pass only sees the outermost layer. An Optional is a + # Union, not an alias, so `Optional[NDArray[...]]` -- every typed numpy + # annotation marked nullable, on numpy 2.5 -- reaches here unexpanded. + spec = utils.resolve_type_alias(spec) origin = typing.get_origin(spec) args = typing.get_args(spec) args_origins = [typing.get_origin(x) if x is not None else None for x in args] diff --git a/singlestoredb/tests/test_udf_type_aliases.py b/singlestoredb/tests/test_udf_type_aliases.py index 9a848cf7b..60c6b721c 100644 --- a/singlestoredb/tests/test_udf_type_aliases.py +++ b/singlestoredb/tests/test_udf_type_aliases.py @@ -20,6 +20,7 @@ import unittest from typing import Any from typing import Callable +from typing import Optional import numpy as np import numpy.typing as npt @@ -64,6 +65,35 @@ def foo_b(x: Vec) -> Vec: assert to_sql(foo_b) == '`foo_b`(`x` DOUBLE NOT NULL) RETURNS DOUBLE NOT NULL' + def test_optional_subscripted_alias(self) -> None: + ScalarT = typing.TypeVar('ScalarT') + NDArrayAlias = typing.TypeAliasType( # noqa: TYP006 + 'NDArrayAlias', + np.ndarray[Any, np.dtype[ScalarT]], + type_params=(ScalarT,), + ) + + @udf + def foo_c( + x: Optional[NDArrayAlias[np.str_]], + ) -> Optional[NDArrayAlias[np.str_]]: + return x + + # NOT NULL despite the Optional: the numpy branch of `get_schema` does + # not thread `is_optional` into its ParamSpec. Pre-existing on every + # numpy version, and asserted here so the alias expansion above is + # provably nullability-neutral. + assert to_sql(foo_c) == '`foo_c`(`x` TEXT NOT NULL) RETURNS TEXT NOT NULL' + + def test_optional_bare_alias(self) -> None: + Vec = typing.TypeAliasType('Vec', npt.NDArray[np.float64]) # noqa: TYP006 + + @udf + def foo_d(x: Optional[Vec]) -> Optional[Vec]: + return x + + assert to_sql(foo_d) == '`foo_d`(`x` DOUBLE NOT NULL) RETURNS DOUBLE NOT NULL' + if __name__ == '__main__': unittest.main()