From 4cf4287653c004a7e94d9ff4abce8073e008977d Mon Sep 17 00:00:00 2001 From: Benjamin Capodanno Date: Mon, 21 Sep 2026 11:43:18 -0700 Subject: [PATCH] fix(score-sets): index functional classification variant fk to unblock score set deletion Deleting a score set cascades through SQLAlchemy to delete each of its variants, and Postgres checks every foreign key into variants.id to enforce referential integrity on each one. score_calibration_functional_classification_variants carries variant_id only as the trailing column of its composite primary key (functional_classification_id, variant_id), so that check has no index it can use and instead sequentially scans the whole table once per variant being deleted. The table is global, growing with every calibration ever created across MaveDB, so the cost compounds with both the deleted score set's variant count and the table's total size. This is the mechanism behind #677 ("unpublished score set deletion often fails"): deletion runs synchronously in the request, and for any score set with a nontrivial number of variants the per-variant sequential scan is slow enough to time out, non-deterministically depending on how large the association table has grown. Confirmed with EXPLAIN against a synthetic 200k-row table: the FK check plan flips from Seq Scan to Index Scan, and deleting 500 variants drops from ~2.6s to ~2.4ms. The migration builds the index CONCURRENTLY, outside a transaction, since the table may already be large enough in production that a blocking build would itself be disruptive. --- ...ndex_functional_classification_variant_.py | 38 +++++++++++++++++++ ...onal_classification_variant_association.py | 3 +- 2 files changed, 40 insertions(+), 1 deletion(-) create mode 100644 alembic/versions/15c40c367731_index_functional_classification_variant_.py diff --git a/alembic/versions/15c40c367731_index_functional_classification_variant_.py b/alembic/versions/15c40c367731_index_functional_classification_variant_.py new file mode 100644 index 000000000..773a5ed9c --- /dev/null +++ b/alembic/versions/15c40c367731_index_functional_classification_variant_.py @@ -0,0 +1,38 @@ +"""index functional classification variant fk + +variant_id is only the trailing column of score_calibration_functional_classification_variants' +composite primary key, so Postgres has no index it can use to look up rows by variant_id alone. +Every Variant delete (including the per-variant cascade when a score set is deleted) forces a +sequential scan of this table to enforce that FK, which dominates the delete for any score set +with a nontrivial number of variants. See mavedb-api#677 ("unpublished score set deletion often +fails"). + +Built CONCURRENTLY, outside a transaction, so this doesn't hold an exclusive lock on the table for +the build's duration. + +Revision ID: 15c40c367731 +Revises: c4b18d0f7a92 +Create Date: 2026-09-21 11:19:39.396343 + +""" + +from alembic import op + +# revision identifiers, used by Alembic. +revision = "15c40c367731" +down_revision = "c4b18d0f7a92" +branch_labels = None +depends_on = None + +INDEX_NAME = "ix_score_calib_func_classification_variants_variant_id" +TABLE_NAME = "score_calibration_functional_classification_variants" + + +def upgrade(): + with op.get_context().autocommit_block(): + op.execute(f"CREATE INDEX CONCURRENTLY IF NOT EXISTS {INDEX_NAME} ON {TABLE_NAME} (variant_id)") + + +def downgrade(): + with op.get_context().autocommit_block(): + op.execute(f"DROP INDEX CONCURRENTLY IF EXISTS {INDEX_NAME}") diff --git a/src/mavedb/models/score_calibration_functional_classification_variant_association.py b/src/mavedb/models/score_calibration_functional_classification_variant_association.py index 61f074bd2..e9475f798 100644 --- a/src/mavedb/models/score_calibration_functional_classification_variant_association.py +++ b/src/mavedb/models/score_calibration_functional_classification_variant_association.py @@ -1,6 +1,6 @@ """SQLAlchemy association table for variants belonging to functional classifications.""" -from sqlalchemy import Column, ForeignKey, Table +from sqlalchemy import Column, ForeignKey, Index, Table from mavedb.db.base import Base @@ -11,4 +11,5 @@ "functional_classification_id", ForeignKey("score_calibration_functional_classifications.id"), primary_key=True ), Column("variant_id", ForeignKey("variants.id"), primary_key=True), + Index("ix_score_calib_func_classification_variants_variant_id", "variant_id"), )