Skip to content

Commit 69692c3

Browse files
committed
perf(db): add hot-path indexes for matching, latest-bbox and frame lookups
Closes #663. Three query shapes on the detection hot path had no index behind them, so each seq-scanned a growing table: - a pose's recently-seen sequences (camera_id, pose_id, last_seen_at), run on every POST /detections during spatial matching - the latest real bbox of a sequence (sequence_id, created_at), run once per candidate sequence per detection, and the same shape the player's sequence reads sort on - sibling rows sharing a frame object (bucket_key), on DELETE /detections/{id} Measured on the (sequence_id, created_at) shape with production-scale synthetic data, index scan vs forced seq scan on identical rows: 7.6ms -> 0.45ms at 200k detections, 34.8ms -> 0.42ms at 2M (roughly production today), 63.6ms -> 1.1ms at 5M. The seq-scan side grows linearly with the table while the index scan stays flat, so the gap widens as detections accumulate. Built with CREATE INDEX CONCURRENTLY inside an autocommit block: detections is the highest-write table and a plain build would hold ACCESS EXCLUSIVE against camera ingest for its whole duration. The migration also self-heals, dropping an index left INVALID by a cancelled build, which if_not_exists would otherwise skip while the upgrade reported success and the planner ignored it. The indexes are declared in models.py as well as the migration so create_all built test databases match production, with tests pinning both sides against one list since drift between them is otherwise invisible.
1 parent 81b7793 commit 69692c3

3 files changed

Lines changed: 127 additions & 0 deletions

File tree

‎src/app/models.py‎

Lines changed: 14 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,7 @@
77
from enum import Enum
88
from typing import Union
99

10+
from sqlalchemy import Index
1011
from sqlmodel import Field, SQLModel
1112

1213
from app.core.config import settings
@@ -83,6 +84,15 @@ class OcclusionMask(SQLModel, table=True):
8384

8485
class Detection(SQLModel, table=True):
8586
__tablename__ = "detections"
87+
# Declared here as well as in the migration so create_all-built databases (the test suite)
88+
# carry the same indexes as production, instead of silently planning every query differently.
89+
__table_args__ = (
90+
# Latest-real-bbox lookups during spatial matching, and the sequence reads the player
91+
# pages through: sequence_id equality then a created_at scan.
92+
Index("ix_detections_sequence_id_created_at", "sequence_id", "created_at"),
93+
# Sibling-row check on deletion (multi-bbox and continuity rows share one frame object).
94+
Index("ix_detections_bucket_key", "bucket_key"),
95+
)
8696
id: int = Field(None, primary_key=True)
8797
camera_id: int = Field(..., foreign_key="cameras.id", nullable=False)
8898
pose_id: int = Field(..., foreign_key="poses.id", nullable=False)
@@ -112,6 +122,10 @@ class Detection(SQLModel, table=True):
112122

113123
class Sequence(SQLModel, table=True):
114124
__tablename__ = "sequences"
125+
__table_args__ = (
126+
# Per-frame lookups of a pose's recently-seen sequences (spatial matching, continuity).
127+
Index("ix_sequences_camera_pose_last_seen", "camera_id", "pose_id", "last_seen_at"),
128+
)
115129
id: int = Field(None, primary_key=True)
116130
camera_id: int = Field(..., foreign_key="cameras.id", nullable=False)
117131
pose_id: Union[int, None] = Field(None, foreign_key="poses.id", nullable=True)
Lines changed: 84 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,84 @@
1+
"""add hot-path indexes for sequence matching, latest-bbox and shared frame lookups
2+
3+
Revision ID: e8f3a6c9d1b7
4+
Revises: c4e9f1a2b3d5
5+
Create Date: 2026-07-30 10:00:00.000000
6+
7+
"""
8+
9+
from typing import Sequence, Union
10+
11+
from alembic import op
12+
from sqlalchemy import text
13+
14+
# revision identifiers, used by Alembic.
15+
revision: str = "e8f3a6c9d1b7"
16+
down_revision: Union[str, None] = "c4e9f1a2b3d5"
17+
branch_labels: Union[str, Sequence[str], None] = None
18+
depends_on: Union[str, Sequence[str], None] = None
19+
20+
# (index name, table, columns). Kept in sync with the __table_args__ declarations in
21+
# app.models, which is what puts these indexes in create_all-built test databases too.
22+
INDEXES = (
23+
("ix_sequences_camera_pose_last_seen", "sequences", ["camera_id", "pose_id", "last_seen_at"]),
24+
("ix_detections_sequence_id_created_at", "detections", ["sequence_id", "created_at"]),
25+
("ix_detections_bucket_key", "detections", ["bucket_key"]),
26+
)
27+
28+
29+
def _drop_if_invalid(index_name: str, table_name: str) -> None:
30+
"""Drop an index left INVALID by a cancelled CONCURRENTLY build.
31+
32+
Without this the migration is not safely re-runnable: if_not_exists sees the leftover
33+
relation and skips creating it, the revision stamps, and the upgrade reports success while
34+
the planner ignores the invalid index, silently leaving these queries on sequential scans.
35+
36+
The lookup joins pg_class by name rather than casting the name to regclass, since that cast
37+
raises when the relation does not exist yet (the common case) instead of returning no rows.
38+
"""
39+
row = (
40+
op
41+
.get_bind()
42+
.execute(
43+
text(
44+
"SELECT idx.indisvalid FROM pg_index idx "
45+
"JOIN pg_class c ON c.oid = idx.indexrelid "
46+
"WHERE c.relname = :name"
47+
),
48+
{"name": index_name},
49+
)
50+
.first()
51+
)
52+
if row is not None and not row[0]:
53+
op.drop_index(index_name, table_name=table_name, if_exists=True, postgresql_concurrently=True)
54+
55+
56+
def upgrade() -> None:
57+
# Three query shapes on the detection hot path had no index backing them:
58+
# - a pose's recently-seen sequences (camera_id, pose_id, last_seen_at), run on every
59+
# POST /detections during spatial matching
60+
# - the latest real bbox of a sequence (sequence_id, created_at), run once per candidate
61+
# sequence per detection, and also the shape the player's sequence reads sort on
62+
# - sibling rows sharing a frame object (bucket_key), on DELETE /detections/{id}
63+
#
64+
# detections is the highest-write table, so a plain CREATE INDEX would hold ACCESS EXCLUSIVE
65+
# against camera ingest for the whole build. CONCURRENTLY cannot run inside a transaction and
66+
# env.py wraps the migration run in one, hence the autocommit block.
67+
with op.get_context().autocommit_block():
68+
for index_name, table_name, columns in INDEXES:
69+
_drop_if_invalid(index_name, table_name)
70+
op.create_index(
71+
index_name,
72+
table_name,
73+
columns,
74+
unique=False,
75+
if_not_exists=True,
76+
postgresql_concurrently=True,
77+
)
78+
79+
80+
def downgrade() -> None:
81+
# DROP INDEX CONCURRENTLY is likewise non-transactional.
82+
with op.get_context().autocommit_block():
83+
for index_name, table_name, _ in reversed(INDEXES):
84+
op.drop_index(index_name, table_name=table_name, if_exists=True, postgresql_concurrently=True)

‎src/tests/test_models.py‎

Lines changed: 29 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,29 @@
1+
import pytest
2+
from sqlmodel import SQLModel, text
3+
from sqlmodel.ext.asyncio.session import AsyncSession
4+
5+
# Hot-path indexes, declared in two places that must not drift: the alembic migration (what
6+
# production runs) and __table_args__ in app.models (what create_all gives a test database built
7+
# without migrations). Drift is invisible at runtime, it just makes one environment plan queries
8+
# differently from the other, so both sides are pinned against this list.
9+
EXPECTED_INDEXES = {
10+
"detections": {"ix_detections_sequence_id_created_at", "ix_detections_bucket_key"},
11+
"sequences": {"ix_sequences_camera_pose_last_seen"},
12+
}
13+
14+
15+
@pytest.mark.parametrize(("table", "expected"), EXPECTED_INDEXES.items())
16+
def test_hot_path_indexes_are_declared_on_the_models(table: str, expected: set):
17+
"""Guards the model side. A DB assertion cannot cover this: the test database is migrated,
18+
so the indexes are present whether or not __table_args__ still declares them."""
19+
declared = {index.name for index in SQLModel.metadata.tables[table].indexes}
20+
assert expected <= declared, f"not declared on {table}: {sorted(expected - declared)}"
21+
22+
23+
@pytest.mark.asyncio
24+
@pytest.mark.parametrize(("table", "expected"), EXPECTED_INDEXES.items())
25+
async def test_hot_path_indexes_exist_in_the_database(async_session: AsyncSession, table: str, expected: set):
26+
"""Guards the migration side: a fresh database must end up with all of them."""
27+
stmt = text("SELECT indexname FROM pg_indexes WHERE tablename = :table").bindparams(table=table)
28+
present = {row[0] for row in (await async_session.exec(stmt)).all()}
29+
assert expected <= present, f"missing on {table}: {sorted(expected - present)}"

0 commit comments

Comments
 (0)