diff --git a/CHANGELOG.md b/CHANGELOG.md index be9b700..bec2b89 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,14 +7,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [0.1.1] - 2026-10-03 + ### Fixed -- PostgreSQL compound index names now fit its 63-byte identifier limit. Long names keep deterministic digests and their index suffix, so typed action indexes are created instead of silently skipped at startup. -- Concurrent first use of a PostgreSQL collection now serializes its schema bootstrap, avoiding races while creating the shared table and base indexes. -- PostgreSQL `$exists` now treats JSON `null` like the in-memory query engine, fixing queries for absent or null nested values. -- File-backed SQLite closes the old `aiosqlite` connection when rebinding across event loops, preventing a leaked worker thread from keeping the process alive. -- Graph transactions isolate their request identity map and invalidate touched parent cache entries after commit or rollback. Index setup is cached per database instance so a second store receives its own indexes. -- Tests now restore mocked walker metadata, bind their own graph context, and enter the test client's lifespan before traversing Root, removing order-dependent suite failures. +- PostgreSQL compound, base, vector, and tenant-policy names now fit its 63-byte identifier limit. Long names keep deterministic digests and their suffix, so indexes are created instead of silently skipped at startup. +- Concurrent first use of a PostgreSQL collection now serializes schema bootstrap across database instances and workers, avoiding races while creating the shared table and base indexes. ## [0.1.0] - 2026-09-27 @@ -39,6 +37,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `AGENTS.md` is now the canonical agent guide. Its former `CLAUDE.md` content has been consolidated there; `CLAUDE.md` is removed. - The deprecated `generate_id_async` alias remains available in 0.1.0; its removal target is 0.2.0. +### Fixed + +- PostgreSQL `$exists` treats JSON `null` like the in-memory query engine, fixing queries for absent or null nested values. +- File-backed SQLite closes the old `aiosqlite` connection when rebinding across event loops, preventing a leaked worker thread from keeping the process alive. +- Graph transactions isolate their request identity map and invalidate touched parent cache entries after commit or rollback. Index setup is cached per database instance so a second store receives its own indexes. +- Tests restore mocked walker metadata, bind their own graph context, and enter the test client's lifespan before traversing Root, removing order-dependent suite failures. + ## [0.0.22] - 2026-09-26 ### Added diff --git a/SPEC.md b/SPEC.md index 8d77627..9a27461 100644 --- a/SPEC.md +++ b/SPEC.md @@ -312,7 +312,7 @@ No built-in migration framework. Adapters do not enforce schemas. Adding optiona ### 5.2 Pushdown vs in-memory -- **Postgres**: `translate_query` pushes the whole operator surface into JSONB SQL; `$text` becomes `to_tsvector('simple', …) @@ plainto_tsquery('simple', …)` (requires `$fields`; the GIN from `@fulltext_index` / `attribute(fulltext=True)` serves it when the fields match in order). Per-class indexes are `(entity, )` (or `WHERE entity = …` with `partial_by_entity`), descending keys `DESC NULLS LAST`, and generated index names stay within PostgreSQL's 63-byte identifier limit using a stable digest when needed, so typed `find(sort=…, limit=…)` can walk an index; first-use collection table and base-index creation is serialized per database instance; the whole-document GIN is optional (`JVSPATIAL_PG_GIN_INDEX`). Untranslatable queries fall back to a full scan + `QueryEngine.match`. +- **Postgres**: `translate_query` pushes the whole operator surface into JSONB SQL; `$text` becomes `to_tsvector('simple', …) @@ plainto_tsquery('simple', …)` (requires `$fields`; the GIN from `@fulltext_index` / `attribute(fulltext=True)` serves it when the fields match in order). Per-class indexes are `(entity, )` (or `WHERE entity = …` with `partial_by_entity`), descending keys `DESC NULLS LAST`, and generated index and tenant-policy names stay within PostgreSQL's 63-byte identifier limit using a stable digest when needed, so typed `find(sort=…, limit=…)` can walk an index; first-use collection table and base-index creation is serialized across database instances using a PostgreSQL transaction-scoped advisory lock; the whole-document GIN is optional (`JVSPATIAL_PG_GIN_INDEX`). Untranslatable queries fall back to a full scan + `QueryEngine.match`. - **MongoDB**: native pushdown; queries run server-side (`$text` uses the collection's text index; `$fields` is stripped). - **SQLite**: translated to SQL via `SQLiteTranslator` (subset; complex `$or` chains may fall back). - **DynamoDB**: limited pushdown via `Select=COUNT` and key conditions; remainder filtered client-side. diff --git a/jvspatial/db/postgres.py b/jvspatial/db/postgres.py index 56f30e0..b588414 100644 --- a/jvspatial/db/postgres.py +++ b/jvspatial/db/postgres.py @@ -606,7 +606,7 @@ async def enable_rls( "tenant_id = NULLIF(current_setting('app.tenant_id', true), '')" ) - policy_name = f"{col}_tenant_isolation" + policy_name = _postgres_index_name(col, "tenant_isolation") # ALTER + CREATE POLICY both error if duplicated; check + drop + # recreate so re-runs are idempotent. ``DROP POLICY IF EXISTS`` # is supported on PG 9.5+. @@ -645,18 +645,30 @@ async def _bootstrap_collection(self, collection: str) -> None: async def _create_collection_schema(self, collection: str) -> None: col = _safe_collection(collection) schema = _safe_collection(self.schema_name) + gin_name = _postgres_index_name(col, "data_gin") + entity_name = _postgres_index_name(col, "entity_idx") + tenant_name = _postgres_index_name(col, "tenant_idx") gin_sql = ( - f"CREATE INDEX IF NOT EXISTS {col}_data_gin " + f"CREATE INDEX IF NOT EXISTS {gin_name} " f"ON {schema}.{col} USING GIN (data jsonb_path_ops);" if self.gin_index == "full" else "" ) pool = await self._ensure_pool() async with pool.acquire() as conn: - # Single round trip — CREATE IF NOT EXISTS is cheap when the - # table already exists. - await conn.execute( - f""" + # An asyncio lock only coordinates callers sharing this instance. + # PostgreSQL must also serialize first use across workers and + # independently constructed PostgresDB instances. Keep the + # advisory lock and DDL in one transaction so the lock survives + # until the table and all base indexes have been created. + async with conn.transaction(): + await conn.execute( + "SELECT pg_advisory_xact_lock(hashtext($1), hashtext($2))", + schema, + col, + ) + await conn.execute( + f""" CREATE TABLE IF NOT EXISTS {schema}.{col} ( id TEXT PRIMARY KEY, entity TEXT NOT NULL, @@ -667,13 +679,13 @@ async def _create_collection_schema(self, collection: str) -> None: updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW() ); {gin_sql} - CREATE INDEX IF NOT EXISTS {col}_entity_idx + CREATE INDEX IF NOT EXISTS {entity_name} ON {schema}.{col} (entity); - CREATE INDEX IF NOT EXISTS {col}_tenant_idx + CREATE INDEX IF NOT EXISTS {tenant_name} ON {schema}.{col} (tenant_id) WHERE tenant_id IS NOT NULL; """ - ) + ) self._collections_bootstrapped.add(collection) # ---- payload helpers --------------------------------------------------- @@ -1608,7 +1620,7 @@ async def create_index( if existing is not None and _STALE_INDEX_DEF_RE.search(existing): # Build the corrected index first so a unique constraint is # never absent, then swap it in under the canonical name. - temp = f"{index_name[:52]}_rebuild" + temp = _postgres_index_name(index_name, "rebuild") await conn.execute(f"DROP INDEX IF EXISTS {schema}.{temp}") await conn.execute(f"CREATE {unique_sql}INDEX {temp} {body}") await conn.execute(f"DROP INDEX {schema}.{index_name}") @@ -1851,7 +1863,7 @@ async def enable_vector_column( await self._bootstrap_collection(collection) col = _safe_collection(collection) schema = _safe_collection(self.schema_name) - index_name = f"{col}_{field_name}_{index}_idx" + index_name = _postgres_index_name(f"{col}_{field_name}_{index}", "idx") async with self._pool_lock: pool = await self._ensure_pool() diff --git a/jvspatial/version.py b/jvspatial/version.py index cf005ef..2f6c572 100644 --- a/jvspatial/version.py +++ b/jvspatial/version.py @@ -9,4 +9,4 @@ # - MAJOR: Breaking changes # - MINOR: New features or breaking changes while pre-1.0 # - PATCH: Bug fixes, backward compatible -__version__ = "0.1.0" +__version__ = "0.1.1" diff --git a/tests/db/test_postgres_indexes_text.py b/tests/db/test_postgres_indexes_text.py index 87bab87..d2bd697 100644 --- a/tests/db/test_postgres_indexes_text.py +++ b/tests/db/test_postgres_indexes_text.py @@ -122,6 +122,66 @@ async def _explain(admin: Any, sql: str, params: List[Any]) -> List[Dict[str, An return _plan_nodes(plan[0]["Plan"]) +async def test_collection_bootstrap_serializes_across_database_instances(): + from jvspatial.db.postgres import PostgresDB + + async with _pg() as (_ctx, db, admin, schema): + other = PostgresDB(dsn=_PG_DSN, schema_name=schema, min_size=1, max_size=4) + collection = "bootstrap_race" + tasks: List[asyncio.Task[None]] = [] + try: + await asyncio.gather(db._ensure_pool(), other._ensure_pool()) + async with admin.transaction(): + await admin.execute( + "SELECT pg_advisory_xact_lock(hashtext($1), hashtext($2))", + schema, + collection, + ) + tasks = [ + asyncio.create_task(instance._bootstrap_collection(collection)) + for instance in (db, other) + ] + await asyncio.sleep(0.1) + assert all(not task.done() for task in tasks) + await asyncio.gather(*tasks) + assert collection in db._collections_bootstrapped + assert collection in other._collections_bootstrapped + indexes = await _indexes(admin, schema, collection) + assert f"{collection}_entity_idx" in indexes + assert f"{collection}_tenant_idx" in indexes + finally: + for task in tasks: + if not task.done(): + task.cancel() + await asyncio.gather(*tasks, return_exceptions=True) + await other.close() + + +async def test_long_collection_gets_distinct_base_indexes_and_rls_policy(): + collection = "collection_" + "x" * 51 + async with _pg(gin_index="full") as (_ctx, db, admin, schema): + await db._bootstrap_collection(collection) + await db.enable_rls(collection) + + indexes = await _indexes(admin, schema, collection) + expected = { + _postgres_index_name(collection, suffix) + for suffix in ("data_gin", "entity_idx", "tenant_idx") + } + assert len(expected) == 3 + assert expected.issubset(indexes) + assert all(len(name.encode("utf-8")) <= 63 for name in expected) + + policies = await admin.fetch( + "SELECT policyname FROM pg_policies WHERE schemaname = $1 AND tablename = $2", + schema, + collection, + ) + assert [row["policyname"] for row in policies] == [ + _postgres_index_name(collection, "tenant_isolation") + ] + + async def test_long_index_names_are_shortened_deterministically(): base = "node_entity_context_agent_id_context_namespace_context_label" name = _postgres_index_name(base, "uniq")