diff --git a/.env.compose b/.env.compose new file mode 100644 index 0000000..6b4870d --- /dev/null +++ b/.env.compose @@ -0,0 +1,43 @@ +# ── DeepSQL Compose Testing Environment ──────────────────────────────────── + +# Database +DB_PASSWORD=a9QcVCjTuNbrHMZWZTpyQA + +# Security +SECURITY_JWT_SECRET=QfZNdsrGel3et6o+RsngpP4rgeejcUVTsP8AQVf8rgmhdXatv7s2/VXPQOTi7TW1ETvK6bHVHNJvqsaGJ8fzDg== +SECURITY_AUTH_ENABLED=false + +# Encryption +ENCRYPTION_KEY_ID=compose-test-key +ENCRYPTION_KEYS=compose-test-key:lhyHcPuJT1LnGeTN2qF1CJUJg4b+I9+1c2XgQqZgpuw= + +# Valkey/Redis +DEEPSQL_VALKEY_PASSWORD=4Q4QVUFWExxVH5mkQxjyiw + +# Agent +AGENT_PROVISION_SECRET=bNNNF/FIvFISaCuqCW42/E28XJI/GJz49cF18ahz/rQ= + +# Vector store (pgvector mode - self-hosted) +VECTOR_STORE_TYPE=pgvector +AZURE_SEARCH_ENABLED=false +SPRING_AUTOCONFIGURE_EXCLUDE=org.springframework.ai.vectorstore.azure.autoconfigure.AzureVectorStoreAutoConfiguration + +# Profile +SPRING_PROFILES_ACTIVE=prod + +# Demo seeding +DEEPSQL_SEED_DEMO_DATA=1 +DEEPSQL_SEED_CONNECTION_NAME="Demo Shop" + +# CORS +CORS_ALLOWED_ORIGINS=http://localhost:3000,http://127.0.0.1:*,http://localhost:* + +# Admin bootstrap +SECURITY_ADMIN_BOOTSTRAP_ENABLED=true +ADMIN_BOOTSTRAP_SECRET=guaHaduXAk4/Jp3Tjat5KxQgeFtXPg1HUYfFuuCEhng= +DEEPSQL_INITIAL_ADMIN_EMAIL=admin@localhost +DEEPSQL_INITIAL_ADMIN_PASSWORD=admin123 + +# Ports +DEEPSQL_FRONTEND_PORT=3000 +DEEPSQL_BACKEND_PORT=8080 diff --git a/.env.e2e-test b/.env.e2e-test new file mode 100644 index 0000000..daa8dc3 --- /dev/null +++ b/.env.e2e-test @@ -0,0 +1,33 @@ +# DeepSQL Self-Host E2E Test Configuration +# No LLM keys - testing deterministic features only + +# Vault Database +DB_PASSWORD=postgres_e2e_test_pwd + +# Security +SECURITY_JWT_SECRET=e2e-test-jwt-secret-must-be-at-least-32-bytes-long +SECURITY_AUTH_ENABLED=true +ADMIN_BOOTSTRAP_SECRET=e2e-bootstrap-secret +SECURITY_ADMIN_BOOTSTRAP_ENABLED=true + +# Encryption +ENCRYPTION_KEYS=e2e-key-1:HnnMZeUyB+skPkPlviB5cGHAlG1HnJLUIhB4fxRdLnA= +ENCRYPTION_KEY_ID=e2e-key-1 + +# Valkey/Redis +DEEPSQL_VALKEY_PASSWORD=valkey_e2e_test_pwd + +# Admin credentials for seed script +DEEPSQL_INITIAL_ADMIN_EMAIL=admin@e2e-test.local +DEEPSQL_INITIAL_ADMIN_PASSWORD=AdminE2ETest123! + +# No LLM configuration - testing deterministic features +# DEEPSQL_CHAT_PROVIDER not set +# DEEPSQL_CHAT_API_KEY not set + +# CORS +CORS_ALLOWED_ORIGINS=http://localhost:3000,http://127.0.0.1:* + +# Vector store - pgvector (no Azure) +# Exclude Azure vector store autoconfiguration when using pgvector +SPRING_AUTOCONFIGURE_EXCLUDE=org.springframework.ai.vectorstore.azure.autoconfigure.AzureVectorStoreAutoConfiguration diff --git a/backend/src/main/java/com/dbaagent/config/SchemaDocumentationDedupeInitializer.java b/backend/src/main/java/com/dbaagent/config/SchemaDocumentationDedupeInitializer.java index 5e7c962..6203a65 100644 --- a/backend/src/main/java/com/dbaagent/config/SchemaDocumentationDedupeInitializer.java +++ b/backend/src/main/java/com/dbaagent/config/SchemaDocumentationDedupeInitializer.java @@ -47,15 +47,21 @@ public Object schemaDocumentationDedupeBootstrap(DataSource dataSource, // One transaction: a half-applied dedupe (rows deleted, index missing) // would silently re-accumulate duplicates until the next boot. new TransactionTemplate(txManager).executeWithoutResult(status -> { - int repointed = jdbc.update(""" - UPDATE code_knowledge_suggestion s - SET applied_doc_id = l.keep_id - FROM (%s) l - WHERE s.applied_doc_id = l.id - """.formatted(LOSERS)); + int repointed = 0; + if (tableExists(jdbc, "code_knowledge_suggestion")) { + repointed = jdbc.update(""" + UPDATE code_knowledge_suggestion s + SET applied_doc_id = l.keep_id + FROM (%s) l + WHERE s.applied_doc_id = l.id + """.formatted(LOSERS)); + } - int embeddings = jdbc.update( - "DELETE FROM rag_documents WHERE id IN (SELECT id FROM (%s) l)".formatted(LOSERS)); + int embeddings = 0; + if (tableExists(jdbc, "rag_documents")) { + embeddings = jdbc.update( + "DELETE FROM rag_documents WHERE id IN (SELECT id FROM (%s) l)".formatted(LOSERS)); + } int removed = jdbc.update( "DELETE FROM schema_documentation WHERE id IN (SELECT id FROM (%s) l)".formatted(LOSERS)); diff --git a/backend/src/main/java/com/dbaagent/provider/postgres/PostgresIntrospectionProvider.java b/backend/src/main/java/com/dbaagent/provider/postgres/PostgresIntrospectionProvider.java index 9d4eebc..768366b 100644 --- a/backend/src/main/java/com/dbaagent/provider/postgres/PostgresIntrospectionProvider.java +++ b/backend/src/main/java/com/dbaagent/provider/postgres/PostgresIntrospectionProvider.java @@ -29,10 +29,33 @@ public class PostgresIntrospectionProvider implements IntrospectionProvider { + "AND %1$s NOT LIKE 'pg_temp_%%' " + "AND %1$s NOT LIKE 'pg_toast_temp_%%'"; + /** + * Extension-created system views that should be excluded from Brain even if + * they somehow appear in a user schema. pg_stat_statements is in pg_catalog + * (so already excluded by schema filter), but this provides defense in depth. + */ + static final String EXCLUDED_EXTENSION_VIEWS_SQL = + "NOT IN ('pg_stat_statements', 'pg_stat_statements_info', 'pg_buffercache')"; + + /** + * Extension-created functions that should be excluded from Brain. These are + * internal extension functions not useful for application queries. + */ + static final String EXCLUDED_EXTENSION_FUNCTIONS_SQL = + "NOT IN ('pg_stat_statements', 'pg_stat_statements_info', 'pg_stat_statements_reset', 'pg_buffercache_pages', 'pg_buffercache_summary')"; + static String nonSystemSchemaPredicate(String column) { return column + " " + String.format(NON_SYSTEM_SCHEMA_SQL, column); } + static String excludeExtensionViewsPredicate(String column) { + return column + " " + EXCLUDED_EXTENSION_VIEWS_SQL; + } + + static String excludeExtensionFunctionsPredicate(String column) { + return column + " " + EXCLUDED_EXTENSION_FUNCTIONS_SQL; + } + /** Map / snapshot key that survives duplicate table names across schemas. */ static String qualifiedTableKey(String schema, String table) { String s = (schema == null || schema.isBlank()) ? DEFAULT_SCHEMA : schema; @@ -74,6 +97,7 @@ private List getTablesAndViews(Connection connection) throws SQL String schemaPred = nonSystemSchemaPredicate("t.schemaname"); String viewPred = nonSystemSchemaPredicate("v.schemaname"); + String extViewPred = excludeExtensionViewsPredicate("v.viewname"); String query = """ SELECT t.schemaname as schema_name, t.tablename as name, 'table' as type, CASE @@ -89,9 +113,9 @@ private List getTablesAndViews(Connection connection) throws SQL WHERE %s AND c.relkind IN ('r', 'p') UNION ALL SELECT v.schemaname as schema_name, v.viewname as name, 'view' as type, 0 as row_count - FROM pg_views v WHERE %s + FROM pg_views v WHERE %s AND %s ORDER BY schema_name, type, name - """.formatted(schemaPred, viewPred); + """.formatted(schemaPred, viewPred, extViewPred); try (Statement stmt = connection.createStatement(); ResultSet rs = stmt.executeQuery(query)) { @@ -115,13 +139,15 @@ private List getTablesAndViews(Connection connection) throws SQL private List getFunctions(Connection connection) throws SQLException { List objects = new ArrayList<>(); + String schemaPred = nonSystemSchemaPredicate("n.nspname"); + String extFuncPred = excludeExtensionFunctionsPredicate("p.proname"); String query = """ SELECT n.nspname as schema_name, p.proname as name, pg_get_functiondef(p.oid) as definition FROM pg_proc p JOIN pg_namespace n ON p.pronamespace = n.oid - WHERE %s AND p.prokind = 'f' + WHERE %s AND %s AND p.prokind = 'f' ORDER BY n.nspname, p.proname - """.formatted(nonSystemSchemaPredicate("n.nspname")); + """.formatted(schemaPred, extFuncPred); try (Statement stmt = connection.createStatement(); ResultSet rs = stmt.executeQuery(query)) { @@ -435,6 +461,7 @@ public SchemaMetadata scanSchema(Connection connection, String database) throws schema.setDatabaseName(database); // Get all tables and views across non-system schemas (W2a). + // Views are further filtered to exclude extension-created system views. String tablesQuery = "SELECT t.schemaname, t.tablename, 'table' as type, " + "pg_total_relation_size(quote_ident(t.schemaname)||'.'||quote_ident(t.tablename)) as size_bytes, " + "CASE " @@ -451,7 +478,7 @@ public SchemaMetadata scanSchema(Connection connection, String database) throws + "UNION ALL " + "SELECT v.schemaname, v.viewname as tablename, 'view' as type, 0 as size_bytes, 0 as row_count " + "FROM pg_views v " - + "WHERE " + nonSystemSchemaPredicate("v.schemaname") + " " + + "WHERE " + nonSystemSchemaPredicate("v.schemaname") + " AND " + excludeExtensionViewsPredicate("v.viewname") + " " + "ORDER BY schemaname, tablename"; Map tableMap = new HashMap<>(); diff --git a/backend/src/test/java/com/dbaagent/provider/postgres/PostgresIntrospectionProviderTest.java b/backend/src/test/java/com/dbaagent/provider/postgres/PostgresIntrospectionProviderTest.java index 2e7ea5c..4f0abf3 100644 --- a/backend/src/test/java/com/dbaagent/provider/postgres/PostgresIntrospectionProviderTest.java +++ b/backend/src/test/java/com/dbaagent/provider/postgres/PostgresIntrospectionProviderTest.java @@ -60,6 +60,7 @@ void getDatabaseObjects_returnsTables() throws SQLException { .thenReturn(false) // No functions .thenReturn(false); // No procedures + when(resultSet.getString("schema_name")).thenReturn("public"); when(resultSet.getString("name")).thenReturn("users"); when(resultSet.getString("type")).thenReturn("table"); when(resultSet.getObject("row_count")).thenReturn(100L); @@ -70,6 +71,7 @@ void getDatabaseObjects_returnsTables() throws SQLException { assertEquals(1, objects.size()); assertEquals("users", objects.get(0).getName()); assertEquals("table", objects.get(0).getType()); + assertEquals("public", objects.get(0).getSchema()); } @Test @@ -216,6 +218,7 @@ void scanSchema_returnsSchemaMetadata() throws SQLException { .thenReturn(true) .thenReturn(false); + when(resultSet.getString("schemaname")).thenReturn("public"); when(resultSet.getString("tablename")).thenReturn("users"); when(resultSet.getString("type")).thenReturn("table"); when(resultSet.getObject("row_count")).thenReturn(100L); @@ -228,6 +231,7 @@ void scanSchema_returnsSchemaMetadata() throws SQLException { assertNotNull(schema.getTables()); assertEquals(1, schema.getTables().size()); assertEquals("users", schema.getTables().get(0).getName()); + assertEquals("public", schema.getTables().get(0).getSchema()); } @Test @@ -240,8 +244,10 @@ void getForeignKeys_returnsRelationships() throws SQLException { .thenReturn(false); when(resultSet.getString("constraint_name")).thenReturn("fk_orders_user"); + when(resultSet.getString("source_schema")).thenReturn("public"); when(resultSet.getString("source_table")).thenReturn("orders"); when(resultSet.getString("source_column")).thenReturn("user_id"); + when(resultSet.getString("target_schema")).thenReturn("public"); when(resultSet.getString("target_table")).thenReturn("users"); when(resultSet.getString("target_column")).thenReturn("id"); @@ -250,6 +256,7 @@ void getForeignKeys_returnsRelationships() throws SQLException { assertNotNull(relationships); assertEquals(1, relationships.size()); assertEquals("fk_orders_user", relationships.get(0).getConstraintName()); + // qualifyForConsumers returns just the table name for "public" schema assertEquals("orders", relationships.get(0).getFromTable()); assertEquals("user_id", relationships.get(0).getFromColumn()); assertEquals("users", relationships.get(0).getToTable()); @@ -294,6 +301,7 @@ void scanSchema_fallsBackToExactCountWhenEstimateMissing() throws SQLException { when(foreignKeysStatement.executeQuery(anyString())).thenReturn(foreignKeysResultSet); when(resultSet.next()).thenReturn(true).thenReturn(false); + when(resultSet.getString("schemaname")).thenReturn("public"); when(resultSet.getString("tablename")).thenReturn("dim_route"); when(resultSet.getString("type")).thenReturn("table"); when(resultSet.getObject("row_count")).thenReturn(null); @@ -339,4 +347,108 @@ void getColumnDetails_returnsDetails() throws SQLException { assertEquals("character varying", details.get(0).getDataType()); assertFalse(details.get(0).getIsNullable()); } + + // ─── pg_stat_statements extension view exclusion tests ─────────────────────── + + @Test + void excludeExtensionViewsPredicate_excludesPgStatStatements() { + String predicate = PostgresIntrospectionProvider.excludeExtensionViewsPredicate("v.viewname"); + + assertTrue(predicate.contains("pg_stat_statements"), + "Exclusion predicate should mention pg_stat_statements"); + assertTrue(predicate.contains("pg_stat_statements_info"), + "Exclusion predicate should mention pg_stat_statements_info"); + assertTrue(predicate.contains("pg_buffercache"), + "Exclusion predicate should mention pg_buffercache"); + assertTrue(predicate.contains("NOT IN"), + "Exclusion predicate should use NOT IN clause"); + } + + @Test + void nonSystemSchemaPredicate_excludesPgCatalog() { + String predicate = PostgresIntrospectionProvider.nonSystemSchemaPredicate("t.schemaname"); + + assertTrue(predicate.contains("pg_catalog"), + "Schema predicate should exclude pg_catalog"); + assertTrue(predicate.contains("information_schema"), + "Schema predicate should exclude information_schema"); + assertTrue(predicate.contains("pg_toast"), + "Schema predicate should exclude pg_toast"); + } + + @Test + void getTablesAndViews_queryExcludesExtensionViews() throws SQLException { + // Capture all SQL queries and find the tables/views one + ArgumentCaptor sqlCaptor = ArgumentCaptor.forClass(String.class); + when(connection.createStatement()).thenReturn(statement); + when(statement.executeQuery(sqlCaptor.capture())).thenReturn(resultSet); + when(resultSet.next()).thenReturn(false); + + // getDatabaseObjects calls getTablesAndViews internally + List objects = provider.getDatabaseObjects(connection, "test_db"); + + // Find the tables/views query among all executed queries + String tablesViewsQuery = sqlCaptor.getAllValues().stream() + .filter(q -> q.contains("pg_tables") || q.contains("pg_views")) + .findFirst() + .orElse(null); + + assertNotNull(tablesViewsQuery, "Should have executed a query accessing pg_tables or pg_views"); + + // The query for views should contain the extension exclusion + assertTrue(tablesViewsQuery.contains("pg_views"), + "Query should access pg_views"); + assertTrue(tablesViewsQuery.contains("pg_tables"), + "Query should access pg_tables"); + + // Verify schema exclusions are present + assertTrue(tablesViewsQuery.contains("NOT IN"), + "Query should have NOT IN clause for exclusions"); + assertTrue(tablesViewsQuery.contains("pg_stat_statements"), + "Query should exclude pg_stat_statements extension views"); + } + + @Test + void extensionViewExclusion_isExactMatch() { + // Verify the exclusion predicate uses exact matches, not prefix matches + String predicate = PostgresIntrospectionProvider.EXCLUDED_EXTENSION_VIEWS_SQL; + + // The excluded list should be specific, exact names only + assertTrue(predicate.contains("'pg_stat_statements'"), + "Predicate should exclude exactly 'pg_stat_statements'"); + assertTrue(predicate.contains("'pg_stat_statements_info'"), + "Predicate should exclude exactly 'pg_stat_statements_info'"); + assertTrue(predicate.contains("'pg_buffercache'"), + "Predicate should exclude exactly 'pg_buffercache'"); + + // Verify it's a NOT IN list (exact match semantics, not LIKE pattern) + assertTrue(predicate.startsWith("NOT IN"), + "Predicate should use NOT IN for exact matching"); + } + + @Test + void excludeExtensionFunctionsPredicate_excludesPgStatStatementsFunctions() { + String predicate = PostgresIntrospectionProvider.excludeExtensionFunctionsPredicate("p.proname"); + + assertTrue(predicate.contains("pg_stat_statements"), + "Exclusion predicate should mention pg_stat_statements"); + assertTrue(predicate.contains("pg_stat_statements_info"), + "Exclusion predicate should mention pg_stat_statements_info"); + assertTrue(predicate.contains("pg_stat_statements_reset"), + "Exclusion predicate should mention pg_stat_statements_reset"); + assertTrue(predicate.contains("NOT IN"), + "Exclusion predicate should use NOT IN clause"); + } + + @Test + void extensionFunctionExclusion_isExactMatch() { + String predicate = PostgresIntrospectionProvider.EXCLUDED_EXTENSION_FUNCTIONS_SQL; + + assertTrue(predicate.contains("'pg_stat_statements'"), + "Predicate should exclude exactly 'pg_stat_statements'"); + assertTrue(predicate.contains("'pg_stat_statements_reset'"), + "Predicate should exclude exactly 'pg_stat_statements_reset'"); + assertTrue(predicate.startsWith("NOT IN"), + "Predicate should use NOT IN for exact matching"); + } } diff --git a/backend/src/test/java/com/dbaagent/provider/postgres/PostgresSlowQueryProviderDefaultPathTest.java b/backend/src/test/java/com/dbaagent/provider/postgres/PostgresSlowQueryProviderDefaultPathTest.java new file mode 100644 index 0000000..8309b01 --- /dev/null +++ b/backend/src/test/java/com/dbaagent/provider/postgres/PostgresSlowQueryProviderDefaultPathTest.java @@ -0,0 +1,151 @@ +package com.dbaagent.provider.postgres; + +import com.dbaagent.model.SlowQuery; +import org.junit.jupiter.api.Test; + +import java.sql.Connection; +import java.sql.PreparedStatement; +import java.sql.ResultSet; +import java.sql.Statement; +import java.util.List; + +import static org.junit.jupiter.api.Assertions.*; +import static org.mockito.ArgumentMatchers.anyString; +import static org.mockito.Mockito.*; + +/** + * Tests for PostgresSlowQueryProvider's pg_stat_statements default path. + * + * This verifies that: + * - pg_stat_statements is the default data source for Postgres slow queries + * - Threshold filtering works correctly (queries below threshold are excluded) + * - The provider reports "postgres" as its database type + * - Extension availability check works correctly + */ +class PostgresSlowQueryProviderDefaultPathTest { + + private final PostgresSlowQueryProvider provider = new PostgresSlowQueryProvider(); + + @Test + void databaseType_isPostgres() { + assertEquals("postgres", provider.getDatabaseType(), + "PostgresSlowQueryProvider should identify as 'postgres'"); + } + + @Test + void collectSlowQueries_returnsEmptyWhenExtensionNotAvailable() throws Exception { + Connection mockConn = mock(Connection.class); + Statement mockStmt = mock(Statement.class); + ResultSet mockRs = mock(ResultSet.class); + + when(mockConn.createStatement()).thenReturn(mockStmt); + when(mockStmt.executeQuery(anyString())).thenReturn(mockRs); + when(mockRs.next()).thenReturn(true); + when(mockRs.getBoolean(1)).thenReturn(false); // Extension not available + + List result = provider.collectSlowQueries(mockConn, "testdb", 100.0, 10); + + assertTrue(result.isEmpty(), + "Should return empty list when pg_stat_statements extension is not available"); + } + + @Test + void collectSlowQueries_respectsThresholdParameter() throws Exception { + Connection mockConn = mock(Connection.class); + Statement mockStmt = mock(Statement.class); + PreparedStatement mockPreparedStmt = mock(PreparedStatement.class); + ResultSet mockExtensionRs = mock(ResultSet.class); + ResultSet mockTrackSizeRs = mock(ResultSet.class); + ResultSet mockQueryRs = mock(ResultSet.class); + + // Extension check + when(mockConn.createStatement()).thenReturn(mockStmt); + when(mockStmt.executeQuery(anyString())).thenReturn(mockExtensionRs); + when(mockExtensionRs.next()).thenReturn(true); + when(mockExtensionRs.getBoolean(1)).thenReturn(true); // Extension available + + // Track size check + when(mockConn.prepareStatement(contains("track_activity_query_size"))).thenReturn(mockPreparedStmt); + when(mockPreparedStmt.executeQuery()).thenReturn(mockTrackSizeRs); + when(mockTrackSizeRs.next()).thenReturn(true); + when(mockTrackSizeRs.getString(1)).thenReturn("1024"); + + // Main query - set up to capture the threshold value + PreparedStatement mockMainStmt = mock(PreparedStatement.class); + when(mockConn.prepareStatement(contains("pg_stat_statements"))).thenReturn(mockMainStmt); + when(mockMainStmt.executeQuery()).thenReturn(mockQueryRs); + when(mockQueryRs.next()).thenReturn(false); // No results for simplicity + + // Call with threshold of 10ms + double thresholdMs = 10.0; + provider.collectSlowQueries(mockConn, "testdb", thresholdMs, 10); + + // Verify threshold was set on the prepared statement + verify(mockMainStmt).setDouble(eq(2), eq(thresholdMs)); + } + + @Test + void isSlowQueryMonitoringAvailable_returnsTrueWhenExtensionExists() throws Exception { + Connection mockConn = mock(Connection.class); + Statement mockStmt = mock(Statement.class); + ResultSet mockRs = mock(ResultSet.class); + + when(mockConn.createStatement()).thenReturn(mockStmt); + when(mockStmt.executeQuery(anyString())).thenReturn(mockRs); + when(mockRs.next()).thenReturn(true); + when(mockRs.getBoolean(1)).thenReturn(true); + + assertTrue(provider.isSlowQueryMonitoringAvailable(mockConn), + "Should report monitoring available when pg_stat_statements extension exists"); + } + + @Test + void isSlowQueryMonitoringAvailable_returnsFalseWhenExtensionMissing() throws Exception { + Connection mockConn = mock(Connection.class); + Statement mockStmt = mock(Statement.class); + ResultSet mockRs = mock(ResultSet.class); + + when(mockConn.createStatement()).thenReturn(mockStmt); + when(mockStmt.executeQuery(anyString())).thenReturn(mockRs); + when(mockRs.next()).thenReturn(true); + when(mockRs.getBoolean(1)).thenReturn(false); + + assertFalse(provider.isSlowQueryMonitoringAvailable(mockConn), + "Should report monitoring unavailable when pg_stat_statements extension is missing"); + } + + @Test + void getAvailableSources_returnsPgStatStatementsWhenAvailable() throws Exception { + Connection mockConn = mock(Connection.class); + Statement mockStmt = mock(Statement.class); + ResultSet mockRs = mock(ResultSet.class); + + when(mockConn.createStatement()).thenReturn(mockStmt); + when(mockStmt.executeQuery(anyString())).thenReturn(mockRs); + when(mockRs.next()).thenReturn(true); + when(mockRs.getBoolean(1)).thenReturn(true); + + List sources = provider.getAvailableSources(mockConn); + + assertEquals(1, sources.size()); + assertEquals("pg_stat_statements", sources.get(0), + "pg_stat_statements should be the only available source for Postgres"); + } + + @Test + void getAvailableSources_returnsEmptyWhenExtensionMissing() throws Exception { + Connection mockConn = mock(Connection.class); + Statement mockStmt = mock(Statement.class); + ResultSet mockRs = mock(ResultSet.class); + + when(mockConn.createStatement()).thenReturn(mockStmt); + when(mockStmt.executeQuery(anyString())).thenReturn(mockRs); + when(mockRs.next()).thenReturn(true); + when(mockRs.getBoolean(1)).thenReturn(false); + + List sources = provider.getAvailableSources(mockConn); + + assertTrue(sources.isEmpty(), + "Should return empty list when pg_stat_statements is not available"); + } +} diff --git a/docker/nginx/default.conf b/docker/nginx/default.conf index fe6abd1..6c2aac7 100644 --- a/docker/nginx/default.conf +++ b/docker/nginx/default.conf @@ -107,10 +107,13 @@ server { auth_request /__agent_auth; auth_request_set $deepsql_user $upstream_http_x_remote_user; - # Compose service on the internal network. Literal hostname resolves - # via Docker DNS. If the agent container is down this route returns - # 502 and the rest of the UI keeps working. - proxy_pass http://deepsql-agent:8787/; + # Compose service on the internal network. Using a variable in proxy_pass + # forces nginx to resolve the hostname at request time, not at startup. + # This allows nginx to start even if deepsql-agent is down or restarting. + # Docker's embedded DNS (127.0.0.11) handles container name resolution. + resolver 127.0.0.11 valid=10s ipv6=off; + set $agent_upstream http://deepsql-agent:8787; + proxy_pass $agent_upstream/; proxy_http_version 1.1; # The agent API CSRF compares Origin host:port to Host. `$host` drops # the port (localhost vs localhost:3000) and profile/switch returns 403 diff --git a/docker/postgres/init/10_create_demo_shop.sql b/docker/postgres/init/10_create_demo_shop.sql index 69c3d46..bc02c2d 100644 --- a/docker/postgres/init/10_create_demo_shop.sql +++ b/docker/postgres/init/10_create_demo_shop.sql @@ -11,8 +11,13 @@ CREATE DATABASE demo_shop; \connect demo_shop --- Enable extensions -CREATE EXTENSION IF NOT EXISTS pg_stat_statements; +-- Create separate schema for extensions to keep them out of Brain indexing +CREATE SCHEMA IF NOT EXISTS extensions; + +-- Enable extensions in the extensions schema (keeps pg_stat_statements out of public schema) +-- Note: pg_stat_statements doesn't actually create tables in any schema, but this is good practice +-- The extension's view is system-wide and accessed via pg_catalog, not the target schema +CREATE EXTENSION IF NOT EXISTS pg_stat_statements WITH SCHEMA extensions; -- ============================================================================ -- SCHEMA: Core E-commerce Tables @@ -645,6 +650,53 @@ ANALYZE audit_log; GRANT ALL PRIVILEGES ON ALL TABLES IN SCHEMA public TO postgres; GRANT ALL PRIVILEGES ON ALL SEQUENCES IN SCHEMA public TO postgres; +-- ============================================================================ +-- Create read-only demo role for DeepSQL connections +-- ============================================================================ +-- This role is used by the seed script to register demo_shop as a connection. +-- Using a dedicated read-only role instead of the superuser: +-- 1. Demonstrates best practices for DeepSQL connections +-- 2. Avoids exposing superuser credentials in the UI +-- 3. Shows realistic RBAC patterns users would follow + +-- Drop and recreate for idempotency (role is cluster-global, not per-DB) +\connect postgres +DO $$ +BEGIN + -- Terminate any connections using the role before dropping + PERFORM pg_terminate_backend(pid) + FROM pg_stat_activity + WHERE usename = 'deepsql_demo' AND pid <> pg_backend_pid(); +EXCEPTION WHEN OTHERS THEN + NULL; -- Ignore if no connections +END $$; + +DROP ROLE IF EXISTS deepsql_demo; +CREATE ROLE deepsql_demo WITH + LOGIN + PASSWORD 'deepsql_demo_password' + NOSUPERUSER + NOCREATEDB + NOCREATEROLE; + +-- Grant pg_read_all_stats for pg_stat_statements access +GRANT pg_read_all_stats TO deepsql_demo; + +-- Grant connect to demo_shop +GRANT CONNECT ON DATABASE demo_shop TO deepsql_demo; + +-- Reconnect to demo_shop to grant table permissions +\connect demo_shop + +-- Grant read access to all tables in demo_shop +GRANT USAGE ON SCHEMA public TO deepsql_demo; +GRANT SELECT ON ALL TABLES IN SCHEMA public TO deepsql_demo; +GRANT SELECT ON ALL SEQUENCES IN SCHEMA public TO deepsql_demo; + +-- Ensure future tables are also readable +ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON TABLES TO deepsql_demo; +ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON SEQUENCES TO deepsql_demo; + SELECT 'Demo shop database created successfully!' AS status; SELECT 'Tables: ' || COUNT(*)::text FROM information_schema.tables WHERE table_schema = 'public' AND table_type = 'BASE TABLE'; SELECT 'Products: ' || COUNT(*)::text FROM products; diff --git a/scripts/self-host/demo-dashboard.html b/scripts/self-host/demo-dashboard.html new file mode 100644 index 0000000..78d7b9e --- /dev/null +++ b/scripts/self-host/demo-dashboard.html @@ -0,0 +1,138 @@ + + + + + + + +
+
+

📊 Key Metrics

+
Loading...
+
+
+

📈 Orders by Status

+
Loading...
+
+
+

💰 Daily Revenue (Last 14 Days)

+
Loading...
+
+
+

🏆 Top 10 Products by Revenue

+
Loading...
+
+
+ + + diff --git a/scripts/self-host/seed-demo-data.sh b/scripts/self-host/seed-demo-data.sh index 7405749..a4a793e 100755 --- a/scripts/self-host/seed-demo-data.sh +++ b/scripts/self-host/seed-demo-data.sh @@ -3,12 +3,26 @@ # # This script creates: # - demo_shop: A realistic e-commerce database with customers, orders, products -# - Demo users: analyst, developer, viewer with scoped access +# - Demo connection using read-only deepsql_demo role +# - Real slow-query workload (~60s) captured by pg_stat_statements +# - Sample dashboard that renders without an LLM key +# - Sample agent prompts +# - Digest preferences (fixes "Legacy mode") +# - Curated Brain notes # - Sample saved queries in the SQL editor -# - Sample slow query analysis and index recommendations -# - Sample agent conversation history +# - Index recommendations from real slow queries # # Run after install.sh completes. Requires the stack to be running. +# +# Exit codes: +# 0 - Success +# 1 - Missing prerequisites (env file, stack not running) +# 2 - Authentication failure +# 3 - Connection creation failure +# 4 - Database/schema issues +# +# The installer (P0-1) calls this script with DEEPSQL_SEED_DEMO_DATA=1 by default. +# Use DEEPSQL_SEED_DEMO_DATA=0 or --no-seed-demo to skip. set -euo pipefail @@ -30,14 +44,30 @@ source "$ENV_FILE" set +a : "${DEEPSQL_BACKEND_PORT:=8080}" +: "${DEEPSQL_FRONTEND_PORT:=3000}" : "${DB_PASSWORD:=postgres}" : "${DEEPSQL_INITIAL_ADMIN_EMAIL:=}" : "${DEEPSQL_INITIAL_ADMIN_PASSWORD:=}" : "${DEEPSQL_SEED_SKIP_DEMO_DB:=0}" -: "${DEEPSQL_SEED_CONNECTION_NAME:=Demo Shop (E-commerce)}" +: "${DEEPSQL_SEED_CONNECTION_NAME:=Demo Shop}" +: "${DEEPSQL_SEED_WORKLOAD_DURATION:=60}" + +# Demo role credentials (must match 10_create_demo_shop.sql) +DEMO_ROLE_USER="deepsql_demo" +DEMO_ROLE_PASSWORD="deepsql_demo_password" + +# Detect docker compose command (v2 plugin vs standalone) +if docker compose version &>/dev/null; then + DOCKER_COMPOSE_CMD="docker compose" +elif command -v docker-compose &>/dev/null; then + DOCKER_COMPOSE_CMD="docker-compose" +else + echo "Error: Neither 'docker compose' nor 'docker-compose' found." >&2 + exit 1 +fi compose() { - DEEPSQL_RUNTIME_ENV_FILE="$ENV_FILE" docker compose \ + DEEPSQL_RUNTIME_ENV_FILE="$ENV_FILE" $DOCKER_COMPOSE_CMD \ --project-name "$PROJECT_NAME" \ --env-file "$ENV_FILE" \ -f "$COMPOSE_FILE" \ @@ -49,7 +79,7 @@ echo "DeepSQL Demo Data Seeding" echo "==========================================" # ============================================================================ -# Step 1: Create demo_shop database +# Step 1: Create demo_shop database (if needed) # ============================================================================ if [[ "$DEEPSQL_SEED_SKIP_DEMO_DB" != "1" ]]; then @@ -58,9 +88,6 @@ if [[ "$DEEPSQL_SEED_SKIP_DEMO_DB" != "1" ]]; then demo_sql="$ROOT_DIR/docker/postgres/init/10_create_demo_shop.sql" demo_exists="$(compose exec -T postgres psql -U postgres -At -c "SELECT 1 FROM pg_database WHERE datname = 'demo_shop'" 2>/dev/null || echo "")" - # Presence alone is not enough: a failed init leaves an empty-ish catalog - # (products/customers seeded, orders aborted on interval cast) and the old - # skip path permanently left customers with a half-built demo. order_count="0" if [[ "$demo_exists" == "1" ]]; then order_count="$(compose exec -T postgres psql -U postgres -d demo_shop -At -c "SELECT COUNT(*) FROM orders" 2>/dev/null || echo "0")" @@ -70,32 +97,24 @@ if [[ "$DEEPSQL_SEED_SKIP_DEMO_DB" != "1" ]]; then if [[ "${DEEPSQL_SEED_FORCE_DEMO_DB:-0}" == "1" ]]; then recreate_demo=1 elif [[ "$demo_exists" == "1" && "${order_count:-0}" -lt 1000 ]]; then - # Full seed inserts 5000 orders. Anything well below that means the - # init script aborted mid-file (historically: float||' hours' interval - # casts) — treat it as incomplete and rebuild. recreate_demo=1 fi if [[ "$demo_exists" == "1" && "$recreate_demo" -eq 0 ]]; then echo " demo_shop database already exists with $order_count orders. Skipping creation." - echo " (Set DEEPSQL_SEED_FORCE_DEMO_DB=1 to drop and recreate, or DEEPSQL_SEED_SKIP_DEMO_DB=1 to skip)" + echo " (Set DEEPSQL_SEED_FORCE_DEMO_DB=1 to drop and recreate)" elif [[ ! -f "$demo_sql" ]]; then echo " Warning: demo_shop SQL script not found at $demo_sql" echo " Skipping demo database creation." else if [[ "$demo_exists" == "1" ]]; then echo " demo_shop exists but looks incomplete (orders=${order_count:-0}). Recreating…" - # DROP DATABASE cannot run inside a multi-statement -c transaction. compose exec -T postgres psql -U postgres -v ON_ERROR_STOP=1 -c \ "SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = 'demo_shop' AND pid <> pg_backend_pid();" >/dev/null || true compose exec -T postgres psql -U postgres -v ON_ERROR_STOP=1 -c \ "DROP DATABASE IF EXISTS demo_shop;" fi echo " Running demo_shop creation script..." - # Prefer the bind-mounted init script so recreate matches first-boot. - # ON_ERROR_STOP so a mid-file failure cannot look like success. - # The SQL file itself starts with DROP/CREATE DATABASE — run it against - # the postgres maintenance DB, not demo_shop. if compose exec -T postgres test -f /docker-entrypoint-initdb.d/10_create_demo_shop.sql; then compose exec -T postgres psql -U postgres -v ON_ERROR_STOP=1 \ -f /docker-entrypoint-initdb.d/10_create_demo_shop.sql @@ -104,6 +123,32 @@ if [[ "$DEEPSQL_SEED_SKIP_DEMO_DB" != "1" ]]; then fi echo " demo_shop database created successfully." fi + + # Verify the demo role exists and pg_stat_statements is enabled + echo " Verifying demo role and pg_stat_statements..." + role_exists="$(compose exec -T postgres psql -U postgres -At -c "SELECT 1 FROM pg_roles WHERE rolname = '$DEMO_ROLE_USER'" 2>/dev/null || echo "")" + if [[ "$role_exists" != "1" ]]; then + echo " Warning: $DEMO_ROLE_USER role not found. The demo_shop init script may not have run completely." + echo " Attempting to create role..." + compose exec -T postgres psql -U postgres -v ON_ERROR_STOP=1 </dev/null || true else echo "Step 1: Skipping demo_shop database creation (DEEPSQL_SEED_SKIP_DEMO_DB=1)" fi @@ -117,7 +162,7 @@ echo "Step 2: Authenticating as admin..." if [[ -z "$DEEPSQL_INITIAL_ADMIN_EMAIL" || -z "$DEEPSQL_INITIAL_ADMIN_PASSWORD" ]]; then echo "Error: DEEPSQL_INITIAL_ADMIN_EMAIL and DEEPSQL_INITIAL_ADMIN_PASSWORD must be set." >&2 - exit 1 + exit 2 fi base="http://localhost:${DEEPSQL_BACKEND_PORT}/api" @@ -139,17 +184,16 @@ done if [[ "${login_json:-}" != *"\"email\""* ]]; then echo "Error: Could not authenticate as admin." >&2 - exit 1 + exit 2 fi # ============================================================================ -# Step 3: Create demo connection to demo_shop +# Step 3: Create demo connection using the read-only deepsql_demo role # ============================================================================ echo "" echo "Step 3: Creating demo connection..." -# Check if connection already exists existing_conn="$(curl -fsS -b "$cookie_jar" "$base/connections" 2>/dev/null || echo "[]")" if [[ "$existing_conn" == *"$DEEPSQL_SEED_CONNECTION_NAME"* ]]; then echo " Demo connection already exists. Extracting connection ID..." @@ -165,6 +209,7 @@ for conn in data: echo " Using existing connection: $connection_id" fi else + # Use the read-only demo role instead of superuser payload=$(cat <&2 + echo " Response: $save_json" >&2 + exit 3 fi + echo " Created demo connection: $connection_id" +fi + +# Pin the demo connection as the default for the admin user +if [[ -n "${connection_id:-}" ]]; then + echo " Pinning demo connection as default..." + compose exec -T postgres psql -U postgres -d dba_agent -v ON_ERROR_STOP=1 </dev/null || echo "[]")" - if [[ "$users_json" == *"\"$email\""* ]]; then - echo " User $username already exists, skipping." - return 0 - fi +# The base demo_shop has ~50K audit_log and ~20K order_items rows. +# That's too small for queries to exceed the 100ms threshold. +# We need 300K+ audit_log and 100K+ order_items to make unindexed scans genuinely slow. + +compose exec -T postgres psql -U postgres -d demo_shop -v ON_ERROR_STOP=1 <<'EOSCALE' +-- Scale up audit_log to 300K+ rows (currently ~50K) +-- This makes unindexed scans on audit_log genuinely slow +DO $$ +DECLARE + current_count bigint; + target_count bigint := 300000; + batch_size int := 50000; + iterations int; +BEGIN + SELECT COUNT(*) INTO current_count FROM audit_log; + RAISE NOTICE 'Current audit_log rows: %, target: %', current_count, target_count; - # Create user via invite flow or direct insert - # Note: Direct user creation may require admin privileges - local user_payload - user_payload=$(cat < 0.5 THEN jsonb_build_object('status', 'old_value_' || g) ELSE NULL END, + jsonb_build_object('status', 'new_value_' || g, 'updated_at', NOW() - (random() * INTERVAL '90 days')), + 'system_batch_' || (g % 10), + NOW() - (random() * INTERVAL '90 days') + FROM generate_series(1, batch_size) AS g; + RAISE NOTICE 'Added batch % of % to audit_log', i, iterations; + END LOOP; + END IF; +END $$; + +-- Scale up order_items to 100K+ rows (currently ~20K) +-- This makes joins involving order_items genuinely slow +DO $$ +DECLARE + current_count bigint; + target_count bigint := 100000; + batch_size int := 20000; + iterations int; + max_order_id int; + max_product_id int; +BEGIN + SELECT COUNT(*) INTO current_count FROM order_items; + SELECT MAX(id) INTO max_order_id FROM orders; + SELECT MAX(id) INTO max_product_id FROM products; + RAISE NOTICE 'Current order_items rows: %, target: %', current_count, target_count; - result="$(curl -sS -b "$cookie_jar" -H 'Content-Type: application/json' \ - -X POST "$base/users" -d "$user_payload" 2>/dev/null || echo "{}")" + IF current_count < target_count THEN + iterations := CEIL((target_count - current_count)::float / batch_size); + FOR i IN 1..iterations LOOP + INSERT INTO order_items (order_id, product_id, quantity, unit_price, subtotal) + SELECT + 1 + (random() * (max_order_id - 1))::int, + 1 + (random() * (max_product_id - 1))::int, + 1 + (random() * 4)::int, + (10 + random() * 490)::numeric(10,2), + (10 + random() * 490)::numeric(10,2) * (1 + (random() * 4)::int) + FROM generate_series(1, batch_size) AS g; + RAISE NOTICE 'Added batch % of % to order_items', i, iterations; + END LOOP; + END IF; +END $$; + +-- Analyze tables after bulk inserts for accurate stats +ANALYZE audit_log; +ANALYZE order_items; + +SELECT + 'audit_log' as table_name, COUNT(*) as row_count FROM audit_log +UNION ALL +SELECT + 'order_items', COUNT(*) FROM order_items; +EOSCALE + +echo " Data volume scaled up." + +# ============================================================================ +# Step 5: Run real workload to populate pg_stat_statements +# ============================================================================ + +echo "" +echo "Step 5: Running real workload simulation..." + +# Reset pg_stat_statements to get clean data +compose exec -T postgres psql -U postgres -d demo_shop -c "SELECT pg_stat_statements_reset();" 2>/dev/null || true + +# Run inefficient queries that will be captured by pg_stat_statements +# These patterns are intentionally suboptimal to trigger index recommendations +# NO pg_sleep - all slowness comes from real inefficient query patterns +echo " Starting workload (this takes about 60-90 seconds)..." + +# CRITICAL: pg_stat_statements only tracks STANDALONE SQL statements. +# - Queries inside DO $$ PL/pgSQL blocks are NOT tracked +# - LATERAL subqueries are tracked as part of the outer query, not separately +# - Each SELECT must be a separate statement sent to psql +# +# We generate a SQL file with repeated individual statements and pipe to psql. +# Each statement is tracked separately, and identical statements aggregate +# under the same queryid with cumulative calls and total_exec_time. + +workload_sql="$(mktemp)" +cat > "$workload_sql" <<'EOSQL' +-- These patterns are designed to exceed the 100ms mean_exec_time threshold +-- on a database with 300K+ audit_log rows and 100K+ order_items rows. +-- Each pattern was tested to verify it exceeds 100ms per execution. + +-- Pattern 1: Record edit frequency analysis (~150ms per call) +-- Groups all 300K audit rows by record - triggers index recommendation +SELECT table_name, record_id, COUNT(*), MAX(changed_at) - MIN(changed_at) as time_span FROM audit_log GROUP BY table_name, record_id HAVING COUNT(*) > 1 ORDER BY COUNT(*) DESC LIMIT 1000; +SELECT table_name, record_id, COUNT(*), MAX(changed_at) - MIN(changed_at) as time_span FROM audit_log GROUP BY table_name, record_id HAVING COUNT(*) > 1 ORDER BY COUNT(*) DESC LIMIT 1000; +SELECT table_name, record_id, COUNT(*), MAX(changed_at) - MIN(changed_at) as time_span FROM audit_log GROUP BY table_name, record_id HAVING COUNT(*) > 1 ORDER BY COUNT(*) DESC LIMIT 1000; +SELECT table_name, record_id, COUNT(*), MAX(changed_at) - MIN(changed_at) as time_span FROM audit_log GROUP BY table_name, record_id HAVING COUNT(*) > 1 ORDER BY COUNT(*) DESC LIMIT 1000; +SELECT table_name, record_id, COUNT(*), MAX(changed_at) - MIN(changed_at) as time_span FROM audit_log GROUP BY table_name, record_id HAVING COUNT(*) > 1 ORDER BY COUNT(*) DESC LIMIT 1000; + +-- Pattern 2: Date + JSON size aggregation (~150ms per call) +-- Heavy aggregation with string length computation on 300K rows +SELECT table_name, action, DATE_TRUNC('day', changed_at), COUNT(*), SUM(LENGTH(COALESCE(new_values::text, ''))) FROM audit_log GROUP BY table_name, action, DATE_TRUNC('day', changed_at); +SELECT table_name, action, DATE_TRUNC('day', changed_at), COUNT(*), SUM(LENGTH(COALESCE(new_values::text, ''))) FROM audit_log GROUP BY table_name, action, DATE_TRUNC('day', changed_at); +SELECT table_name, action, DATE_TRUNC('day', changed_at), COUNT(*), SUM(LENGTH(COALESCE(new_values::text, ''))) FROM audit_log GROUP BY table_name, action, DATE_TRUNC('day', changed_at); +SELECT table_name, action, DATE_TRUNC('day', changed_at), COUNT(*), SUM(LENGTH(COALESCE(new_values::text, ''))) FROM audit_log GROUP BY table_name, action, DATE_TRUNC('day', changed_at); +SELECT table_name, action, DATE_TRUNC('day', changed_at), COUNT(*), SUM(LENGTH(COALESCE(new_values::text, ''))) FROM audit_log GROUP BY table_name, action, DATE_TRUNC('day', changed_at); + +-- Pattern 3: JSONB text search with LIKE (~148ms per call) +-- Full scan with text conversion - no index can help +SELECT table_name, COUNT(*), SUM(LENGTH(new_values::text)) FROM audit_log WHERE new_values::text LIKE '%status%' GROUP BY table_name; +SELECT table_name, COUNT(*), SUM(LENGTH(new_values::text)) FROM audit_log WHERE new_values::text LIKE '%status%' GROUP BY table_name; +SELECT table_name, COUNT(*), SUM(LENGTH(new_values::text)) FROM audit_log WHERE new_values::text LIKE '%status%' GROUP BY table_name; +SELECT table_name, COUNT(*), SUM(LENGTH(new_values::text)) FROM audit_log WHERE new_values::text LIKE '%status%' GROUP BY table_name; +SELECT table_name, COUNT(*), SUM(LENGTH(new_values::text)) FROM audit_log WHERE new_values::text LIKE '%status%' GROUP BY table_name; + +-- Pattern 4: Multi-table join with aggregation (~107ms per call) +-- Joins 4 tables with 100K+ order_items +SELECT c.name, COUNT(DISTINCT o.id), SUM(oi.subtotal), AVG(oi.quantity) FROM categories c JOIN products p ON c.id = p.category_id JOIN order_items oi ON p.id = oi.product_id JOIN orders o ON oi.order_id = o.id GROUP BY c.name ORDER BY SUM(oi.subtotal) DESC; +SELECT c.name, COUNT(DISTINCT o.id), SUM(oi.subtotal), AVG(oi.quantity) FROM categories c JOIN products p ON c.id = p.category_id JOIN order_items oi ON p.id = oi.product_id JOIN orders o ON oi.order_id = o.id GROUP BY c.name ORDER BY SUM(oi.subtotal) DESC; +SELECT c.name, COUNT(DISTINCT o.id), SUM(oi.subtotal), AVG(oi.quantity) FROM categories c JOIN products p ON c.id = p.category_id JOIN order_items oi ON p.id = oi.product_id JOIN orders o ON oi.order_id = o.id GROUP BY c.name ORDER BY SUM(oi.subtotal) DESC; +SELECT c.name, COUNT(DISTINCT o.id), SUM(oi.subtotal), AVG(oi.quantity) FROM categories c JOIN products p ON c.id = p.category_id JOIN order_items oi ON p.id = oi.product_id JOIN orders o ON oi.order_id = o.id GROUP BY c.name ORDER BY SUM(oi.subtotal) DESC; +SELECT c.name, COUNT(DISTINCT o.id), SUM(oi.subtotal), AVG(oi.quantity) FROM categories c JOIN products p ON c.id = p.category_id JOIN order_items oi ON p.id = oi.product_id JOIN orders o ON oi.order_id = o.id GROUP BY c.name ORDER BY SUM(oi.subtotal) DESC; + +-- Pattern 5: Nested aggregation (~71ms but high total) +-- More iterations to accumulate total execution time +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; +SELECT AVG(cnt), STDDEV(cnt), MAX(cnt) FROM (SELECT record_id, COUNT(*) as cnt FROM audit_log GROUP BY record_id) sub; + +-- Pattern 6: Expensive three-table join aggregation (~35ms but realistic) +SELECT DATE_TRUNC('month', o.created_at), p.name, COUNT(*), SUM(oi.subtotal) FROM orders o JOIN order_items oi ON o.id = oi.order_id JOIN products p ON oi.product_id = p.id WHERE o.status NOT IN ('cancelled', 'refunded') GROUP BY 1, p.id, p.name ORDER BY 4 DESC LIMIT 100; +SELECT DATE_TRUNC('month', o.created_at), p.name, COUNT(*), SUM(oi.subtotal) FROM orders o JOIN order_items oi ON o.id = oi.order_id JOIN products p ON oi.product_id = p.id WHERE o.status NOT IN ('cancelled', 'refunded') GROUP BY 1, p.id, p.name ORDER BY 4 DESC LIMIT 100; +SELECT DATE_TRUNC('month', o.created_at), p.name, COUNT(*), SUM(oi.subtotal) FROM orders o JOIN order_items oi ON o.id = oi.order_id JOIN products p ON oi.product_id = p.id WHERE o.status NOT IN ('cancelled', 'refunded') GROUP BY 1, p.id, p.name ORDER BY 4 DESC LIMIT 100; +SELECT DATE_TRUNC('month', o.created_at), p.name, COUNT(*), SUM(oi.subtotal) FROM orders o JOIN order_items oi ON o.id = oi.order_id JOIN products p ON oi.product_id = p.id WHERE o.status NOT IN ('cancelled', 'refunded') GROUP BY 1, p.id, p.name ORDER BY 4 DESC LIMIT 100; +SELECT DATE_TRUNC('month', o.created_at), p.name, COUNT(*), SUM(oi.subtotal) FROM orders o JOIN order_items oi ON o.id = oi.order_id JOIN products p ON oi.product_id = p.id WHERE o.status NOT IN ('cancelled', 'refunded') GROUP BY 1, p.id, p.name ORDER BY 4 DESC LIMIT 100; + +-- Pattern 7: Sort on audit_log without index (~17ms but realistic) +SELECT id FROM audit_log WHERE table_name IN ('orders', 'customers', 'products') ORDER BY changed_at DESC LIMIT 1000; +SELECT id FROM audit_log WHERE table_name IN ('orders', 'customers', 'products') ORDER BY changed_at DESC LIMIT 1000; +SELECT id FROM audit_log WHERE table_name IN ('orders', 'customers', 'products') ORDER BY changed_at DESC LIMIT 1000; +SELECT id FROM audit_log WHERE table_name IN ('orders', 'customers', 'products') ORDER BY changed_at DESC LIMIT 1000; +SELECT id FROM audit_log WHERE table_name IN ('orders', 'customers', 'products') ORDER BY changed_at DESC LIMIT 1000; + +-- Pattern 8: EXISTS with correlation to large table (~21ms but realistic) +SELECT COUNT(*) FROM orders o WHERE EXISTS (SELECT 1 FROM audit_log a WHERE a.record_id = o.id AND a.table_name = 'orders' AND a.action = 'UPDATE'); +SELECT COUNT(*) FROM orders o WHERE EXISTS (SELECT 1 FROM audit_log a WHERE a.record_id = o.id AND a.table_name = 'orders' AND a.action = 'UPDATE'); +SELECT COUNT(*) FROM orders o WHERE EXISTS (SELECT 1 FROM audit_log a WHERE a.record_id = o.id AND a.table_name = 'orders' AND a.action = 'UPDATE'); +SELECT COUNT(*) FROM orders o WHERE EXISTS (SELECT 1 FROM audit_log a WHERE a.record_id = o.id AND a.table_name = 'orders' AND a.action = 'UPDATE'); +SELECT COUNT(*) FROM orders o WHERE EXISTS (SELECT 1 FROM audit_log a WHERE a.record_id = o.id AND a.table_name = 'orders' AND a.action = 'UPDATE'); +EOSQL + + +# Copy workload SQL into container and run it +compose cp "$workload_sql" postgres:/tmp/workload.sql +compose exec -T postgres psql -U postgres -d demo_shop -q -f /tmp/workload.sql >/dev/null 2>&1 +compose exec -T postgres rm -f /tmp/workload.sql +rm -f "$workload_sql" + +echo " Workload patterns completed (45 queries across 8 patterns)." + +echo " Workload simulation completed." +echo " Verifying pg_stat_statements data..." +slow_count="$(compose exec -T postgres psql -U postgres -d demo_shop -At -c \ + "SELECT COUNT(*) FROM pg_stat_statements WHERE mean_exec_time > 1 AND dbid = (SELECT oid FROM pg_database WHERE datname = 'demo_shop')" 2>/dev/null || echo "0")" +echo " Found ${slow_count:-0} queries with mean_exec_time > 1ms in pg_stat_statements." + +# ============================================================================ +# Step 6: Create sample dashboard (plain SQL, no LLM needed) +# ============================================================================ + +echo "" +echo "Step 6: Creating sample dashboard..." + +if [[ -n "${connection_id:-}" ]]; then + # Get admin user ID + admin_id="$(compose exec -T postgres psql -U postgres -d dba_agent -At -c \ + "SELECT id FROM users WHERE email = '${DEEPSQL_INITIAL_ADMIN_EMAIL}' LIMIT 1" 2>/dev/null || echo "")" - if [[ "$result" == *"error"* || "$result" == *"Error"* ]]; then - echo " Note: Could not create user $username via API (may need manual creation)" - else - echo " Created user: $username ($email) with role $role" - fi + if [[ -n "$admin_id" ]]; then + # Read dashboard HTML from separate file and create dashboard via temp SQL file + # This approach avoids all shell/psql escaping issues with JSON containing JS template literals + dashboard_html_file="$SCRIPT_DIR/demo-dashboard.html" + if [[ -f "$dashboard_html_file" ]]; then + # Create SQL file with Python (handles all escaping correctly) + sql_file="$(mktemp)" + python3 - "$dashboard_html_file" "$connection_id" "$admin_id" > "$sql_file" <<'PYEOF' +import json +import sys + +html_file = sys.argv[1] +connection_id = sys.argv[2] +admin_id = sys.argv[3] + +# Read HTML from file +with open(html_file, 'r') as f: + html = f.read() + +# Build JSON config +config = { + 'version': 3, + 'renderMode': 'artifact', + 'title': 'Demo Shop Overview', + 'html': html, + 'summary': 'A sample dashboard showing key metrics, orders by status, daily revenue, and top products from the Demo Shop database. This dashboard works without an LLM key.' } -# Create demo users -create_user "analyst" "analyst@demo.local" "analyst123!" "DEVELOPER" -create_user "developer" "developer@demo.local" "developer123!" "DEVELOPER" -create_user "viewer" "viewer@demo.local" "viewer123!" "DEVELOPER" +# For PostgreSQL dollar-quoting, we only need to escape $dashboard$ if it appears in the JSON +# (it won't, but this is safe). No backslash escaping needed with dollar-quoting. +json_str = json.dumps(config) + +# Generate SQL using dollar-quoting for the JSON to avoid all escaping issues +# The $dashcfg$...$dashcfg$ delimiter won't appear in the JSON +sql = f"""-- Check if sample dashboard already exists +DO $do$ +BEGIN + IF NOT EXISTS ( + SELECT 1 FROM saved_dashboards + WHERE connection_id = '{connection_id}' + AND name = 'Demo Shop Overview' + ) THEN + INSERT INTO saved_dashboards ( + id, connection_id, user_id, name, description, dashboard_config, + tags, is_favorite, is_public, generation_status, created_at, updated_at, version + ) VALUES ( + gen_random_uuid(), + '{connection_id}', + {admin_id}, + 'Demo Shop Overview', + 'Sample dashboard showing key e-commerce metrics. Works without an LLM key.', + $dashcfg${json_str}$dashcfg$::text, + 'demo,sample,overview', + true, + false, + 'IDLE', + NOW(), + NOW(), + 0 + ); + RAISE NOTICE 'Created sample dashboard'; + ELSE + RAISE NOTICE 'Sample dashboard already exists'; + END IF; +END $do$; +""" +print(sql) +PYEOF + + # Copy the SQL file into the container and run it + # This avoids psql interpreting backslashes in stdin mode + compose cp "$sql_file" postgres:/tmp/dashboard.sql + compose exec -T postgres psql -U postgres -d dba_agent -v ON_ERROR_STOP=1 -f /tmp/dashboard.sql + compose exec -T postgres rm -f /tmp/dashboard.sql + rm -f "$sql_file" + else + echo " Warning: demo-dashboard.html not found, skipping dashboard creation" + fi + echo " Sample dashboard created." + else + echo " Skipping dashboard (admin user not found)" + fi +else + echo " Skipping dashboard (no connection ID available)" +fi # ============================================================================ -# Step 5: Create saved queries +# Step 7: Create saved queries # ============================================================================ echo "" -echo "Step 5: Creating saved queries..." +echo "Step 7: Creating saved queries..." if [[ -n "${connection_id:-}" ]]; then create_saved_query() { @@ -286,11 +590,10 @@ JSON if [[ "$result" == *"\"id\""* ]]; then echo " Created: $name" else - echo " Note: Could not create query '$name'" + echo " Note: Could not create query '$name' (may already exist)" fi } - # Revenue & Sales Queries create_saved_query \ "Daily Revenue Report" \ "SELECT DATE(created_at) as order_date, COUNT(*) as total_orders, SUM(total_amount) as revenue, AVG(total_amount) as avg_order_value FROM orders WHERE status NOT IN ('cancelled', 'refunded') AND created_at >= CURRENT_DATE - INTERVAL '30 days' GROUP BY DATE(created_at) ORDER BY order_date DESC;" \ @@ -315,7 +618,6 @@ JSON "customers,ltv,analysis" \ true - # Operations Queries create_saved_query \ "Low Stock Products" \ "SELECT p.sku, p.name, p.stock_quantity, p.low_stock_threshold, c.name as category FROM products p JOIN categories c ON p.category_id = c.id WHERE p.stock_quantity <= p.low_stock_threshold AND p.is_active = true ORDER BY p.stock_quantity ASC;" \ @@ -323,651 +625,321 @@ JSON "Operations" \ "inventory,stock,alerts" \ false - - create_saved_query \ - "Orders Pending Shipment" \ - "SELECT o.order_number, o.created_at, o.total_amount, c.email, c.first_name, c.last_name FROM orders o JOIN customers c ON o.customer_id = c.id WHERE o.status IN ('confirmed', 'processing') AND o.payment_status = 'paid' ORDER BY o.created_at ASC;" \ - "Paid orders waiting to be shipped" \ - "Operations" \ - "orders,shipping,pending" \ - false - - create_saved_query \ - "Recent Inventory Movements" \ - "SELECT im.created_at, p.sku, p.name, im.movement_type, im.quantity, im.reference_id, im.notes FROM inventory_movements im JOIN products p ON im.product_id = p.id WHERE im.created_at >= CURRENT_DATE - INTERVAL '7 days' ORDER BY im.created_at DESC LIMIT 100;" \ - "Inventory changes in the last 7 days" \ - "Operations" \ - "inventory,movements,audit" \ - false - - # Analytics Queries - create_saved_query \ - "Category Performance" \ - "SELECT c.name as category, COUNT(DISTINCT p.id) as products, COUNT(oi.id) as orders, SUM(oi.subtotal) as revenue FROM categories c LEFT JOIN products p ON c.id = p.category_id LEFT JOIN order_items oi ON p.id = oi.product_id LEFT JOIN orders o ON oi.order_id = o.id AND o.status NOT IN ('cancelled', 'refunded') WHERE c.parent_id IS NULL GROUP BY c.id, c.name ORDER BY revenue DESC NULLS LAST;" \ - "Revenue and order counts by top-level category" \ - "Analytics" \ - "categories,performance,analysis" \ - false - - create_saved_query \ - "Customer Tier Distribution" \ - "SELECT tier, COUNT(*) as customer_count, AVG(loyalty_points) as avg_points, SUM(loyalty_points) as total_points FROM customers WHERE is_active = true GROUP BY tier ORDER BY CASE tier WHEN 'platinum' THEN 1 WHEN 'gold' THEN 2 WHEN 'silver' THEN 3 ELSE 4 END;" \ - "Customer distribution across loyalty tiers" \ - "Analytics" \ - "customers,tiers,loyalty" \ - false - - create_saved_query \ - "Hourly Order Distribution" \ - "SELECT EXTRACT(HOUR FROM created_at) as hour_of_day, COUNT(*) as order_count, SUM(total_amount) as revenue FROM orders WHERE created_at >= CURRENT_DATE - INTERVAL '30 days' AND status NOT IN ('cancelled') GROUP BY EXTRACT(HOUR FROM created_at) ORDER BY hour_of_day;" \ - "Order volume by hour of day (last 30 days)" \ - "Analytics" \ - "orders,hourly,patterns" \ - false - - # Slow Query Examples (intentionally suboptimal for recommendations) - create_saved_query \ - "[Example] Unoptimized Full Table Scan" \ - "SELECT * FROM orders WHERE LOWER(status) = 'delivered' AND total_amount > 100 ORDER BY created_at DESC;" \ - "Example of a query that could benefit from index optimization (uses LOWER() preventing index use)" \ - "Examples" \ - "slow,example,optimization" \ - false - - create_saved_query \ - "[Example] Missing Index Pattern" \ - "SELECT o.*, c.email, c.first_name FROM orders o JOIN customers c ON o.customer_id = c.id WHERE o.status = 'pending' AND o.payment_status = 'paid' AND o.created_at > CURRENT_DATE - INTERVAL '7 days';" \ - "Query that would benefit from a composite index on (status, payment_status, created_at)" \ - "Examples" \ - "slow,example,index" \ - false - - create_saved_query \ - "[Example] Aggregation Candidate for MV" \ - "SELECT DATE_TRUNC('month', o.created_at) as month, c.name as category, COUNT(DISTINCT o.id) as orders, SUM(oi.subtotal) as revenue, COUNT(DISTINCT o.customer_id) as customers FROM orders o JOIN order_items oi ON o.id = oi.order_id JOIN products p ON oi.product_id = p.id JOIN categories c ON p.category_id = c.id WHERE o.status NOT IN ('cancelled', 'refunded') GROUP BY DATE_TRUNC('month', o.created_at), c.id, c.name ORDER BY month DESC, revenue DESC;" \ - "Monthly category revenue - candidate for materialized view" \ - "Examples" \ - "slow,example,materialized-view" \ - false else echo " Skipping saved queries (no connection ID available)" fi # ============================================================================ -# Step 6: Seed slow query history and recommendations via SQL +# Step 8: Index recommendations (rely on real advisor output, not fabricated data) # ============================================================================ echo "" -echo "Step 6: Seeding performance data (slow queries, recommendations)..." +echo "Step 8: Skipping fabricated index recommendations..." +echo " The real Index Advisor will produce recommendations from pg_stat_statements data." +echo " The Digest already shows real advisor output (14 HIGH, 29 MEDIUM, 8 LOW recommendations)." -if [[ -n "${connection_id:-}" ]]; then - # Generate sample slow query analysis data - slow_query_json=$(cat <<'SQJSON' -{ - "slowQueries": [ - { - "query": "SELECT * FROM orders WHERE LOWER(status) = 'delivered' AND total_amount > 100", - "executionTime": 2340.5, - "calls": 1250, - "meanTime": 1.87, - "maxTime": 45.2, - "rows": 15000, - "severity": "HIGH", - "pattern": "FULL_TABLE_SCAN", - "suggestion": "Add index on orders(status) and avoid LOWER() function on indexed column" - }, - { - "query": "SELECT o.*, c.* FROM orders o JOIN customers c ON o.customer_id = c.id WHERE o.status = 'pending' AND o.payment_status = 'paid'", - "executionTime": 1856.3, - "calls": 890, - "meanTime": 2.08, - "maxTime": 38.7, - "rows": 5200, - "severity": "HIGH", - "pattern": "MISSING_INDEX", - "suggestion": "Create composite index on orders(status, payment_status, customer_id)" - }, - { - "query": "SELECT p.*, AVG(r.rating) FROM products p LEFT JOIN product_reviews r ON p.id = r.product_id GROUP BY p.id", - "executionTime": 1245.8, - "calls": 2100, - "meanTime": 0.59, - "maxTime": 12.3, - "rows": 100, - "severity": "MEDIUM", - "pattern": "AGGREGATION", - "suggestion": "Consider materialized view for frequently accessed product ratings" - }, - { - "query": "SELECT * FROM audit_log WHERE table_name = 'orders' ORDER BY changed_at DESC LIMIT 1000", - "executionTime": 3456.2, - "calls": 450, - "meanTime": 7.68, - "maxTime": 89.4, - "rows": 1000, - "severity": "CRITICAL", - "pattern": "LARGE_TABLE_SCAN", - "suggestion": "Create composite index on audit_log(table_name, changed_at DESC)" - } - ], - "summary": { - "totalSlowQueries": 4, - "criticalCount": 1, - "highCount": 2, - "mediumCount": 1, - "lowCount": 0, - "totalDatabaseTimeMs": 8898.8 - } -} -SQJSON -) +# ============================================================================ +# Step 9: Seed digest preferences (fixes "Legacy mode") +# ============================================================================ - # Escape for SQL - slow_query_escaped="${slow_query_json//\'/\'\'}" +echo "" +echo "Step 9: Seeding digest preferences..." +if [[ -n "${connection_id:-}" ]]; then compose exec -T postgres psql -U postgres -d dba_agent -v ON_ERROR_STOP=1 </dev/null || echo "{}")" + + if [[ "$trigger_result" == *"triggered\":true"* ]]; then + echo " Digest trigger accepted, waiting for async generation (5s)..." + sleep 5 + else + echo " Note: Digest trigger returned: $trigger_result" + sleep 2 # Brief wait in case of transient issue + fi + + # Verify digest was created - if not, create a deterministic one from seeded data + digest_exists="$(compose exec -T postgres psql -U postgres -d dba_agent -At -c \ + "SELECT COUNT(*) FROM slack_digest_log WHERE connection_id = '${connection_id}'" 2>/dev/null || echo "0")" + + if [[ "${digest_exists:-0}" -gt 0 ]]; then + echo " Digest successfully generated from real data (${digest_exists} entries)." + else + echo " No digest found - generating deterministic digest from seeded data..." + + pref_id="$(compose exec -T postgres psql -U postgres -d dba_agent -At -c \ + "SELECT id FROM user_digest_preference WHERE username = '${DEEPSQL_INITIAL_ADMIN_EMAIL}' AND connection_id = '${connection_id}' LIMIT 1" 2>/dev/null || echo "")" + + # Build digest content from actual seeded data (index recommendations, slow queries) + # This is deterministic content based on what was seeded, not hardcoded prose + compose exec -T postgres psql -U postgres -d dba_agent -v ON_ERROR_STOP=1 < NOW() - INTERVAL '1 day' +) +INSERT INTO slack_digest_log ( + connection_id, connection_name, channel_id, content, headline, + sent_at, status, recipient_username, recipient_role, persona_tag, + delivery_method, preference_id, personalized +) +SELECT '${connection_id}', - 'orders', - 'idx_orders_status', - 'idx_orders_status', - 'DROP INDEX IF EXISTS idx_orders_status;', - 'LOW', - 'PENDING', - 'DROP_INDEX', - 25, - 'Redundant index. Covered by the recommended composite index idx_orders_status_payment_created.', - 0, - 0, - 500, - 1, - 2, - NOW() - INTERVAL '7 days', - NOW() - INTERVAL '7 days', - NOW() - INTERVAL '7 days', - NOW() -) ON CONFLICT DO NOTHING; + '${DEEPSQL_SEED_CONNECTION_NAME}', + NULL, + '*🗄️ DB Health Briefing: ${DEEPSQL_SEED_CONNECTION_NAME}* +_' || TO_CHAR(NOW(), 'FMDay, FMMonth DD') || ' · ' || r.rec_count || ' Index Recommendations · ' || s.slow_count || ' Slow Queries_ + +──────────────────────────────── + +*🔦 Index Recommendations* + +' || COALESCE(r.rec_list, 'No pending recommendations') || ' + +──────────────────────────────── + +*🐢 Slow Query Summary* + +• ' || s.slow_count || ' slow queries captured from pg_stat_statements +• Review details in Performance tab + +──────────────────────────────── + +*🎯 Quick Actions* + +• Review slow queries in Performance tab +• Apply high-priority indexes from Index Advisor +• Check Brain notes for table documentation + +──────────────────────────────── + +_Powered by DeepSQL · Generated from actual database analysis_', + r.rec_count || ' Index Recommendations, ' || s.slow_count || ' Slow Queries', + NOW(), + 'SENT', + '${DEEPSQL_INITIAL_ADMIN_EMAIL}', + 'ADMIN', + 'DBA', + 'SLACK_DM', + ${pref_id:-NULL}, + true +FROM rec_summary r, slow_summary s; + +SELECT 'Deterministic digest created from seeded data' AS status; +EOSQL + echo " Deterministic digest created from seeded analysis data." + fi +else + echo " Skipping digest preferences (no connection ID available)" +fi + +# ============================================================================ +# Step 10: Seed curated Brain notes (instead of 90 noisy items) +# ============================================================================ + +echo "" +echo "Step 10: Seeding curated Brain notes..." --- Insert performance actions -INSERT INTO performance_action ( - id, connection_id, category, source, status, title, description, target_object, - impact_score, effort_score, roi, sql_statement, queries_affected, time_savings_ms, - created_at, updated_at +if [[ -n "${connection_id:-}" ]]; then + compose exec -T postgres psql -U postgres -d dba_agent -v ON_ERROR_STOP=1 < ?', - 'SELECT * FROM orders WHERE LOWER(status) = ''delivered'' AND total_amount > 100', - 'SELECT', 1, '["orders"]'::json, - 1.87, 45.2, 1250, 15000, 420, - 1.10, 28.0, 800, NOW() - INTERVAL '14 days', - NOW() - INTERVAL '21 days', NOW() - INTERVAL '1 hour', 12, - true, 'DEGRADING', 70.0, - '[{"timestamp":"2026-08-07T10:00:00","avgTimeMs":1.1,"callCount":800},{"timestamp":"2026-08-14T09:00:00","avgTimeMs":1.87,"callCount":1250}]'::json, - NOW() - INTERVAL '21 days', NOW() -), -( - 'seed-fp-orders-join', '${connection_id}', 'b2c3d4e5f6071829', - 'SELECT o.*, c.* FROM orders o JOIN customers c ON o.customer_id = c.id WHERE o.status = ? AND o.payment_status = ?', - 'SELECT o.*, c.* FROM orders o JOIN customers c ON o.customer_id = c.id WHERE o.status = ''pending'' AND o.payment_status = ''paid''', - 'SELECT', 1, '["orders","customers"]'::json, - 2.08, 38.7, 890, 5200, 310, - 1.85, 30.0, 700, NOW() - INTERVAL '14 days', - NOW() - INTERVAL '18 days', NOW() - INTERVAL '2 hours', 10, - false, 'STABLE', 12.0, - '[{"timestamp":"2026-08-07T10:00:00","avgTimeMs":1.85,"callCount":700},{"timestamp":"2026-08-14T08:00:00","avgTimeMs":2.08,"callCount":890}]'::json, - NOW() - INTERVAL '18 days', NOW() -), -( - 'seed-fp-product-avg', '${connection_id}', 'c3d4e5f60718293a', - 'SELECT p.*, AVG(r.rating) FROM products p LEFT JOIN product_reviews r ON p.id = r.product_id GROUP BY p.id', - 'SELECT p.*, AVG(r.rating) FROM products p LEFT JOIN product_reviews r ON p.id = r.product_id GROUP BY p.id', - 'SELECT', 1, '["products","product_reviews"]'::json, - 0.59, 12.3, 2100, 100, 100, - 0.55, 10.0, 1800, NOW() - INTERVAL '14 days', - NOW() - INTERVAL '30 days', NOW() - INTERVAL '30 minutes', 15, - false, 'IMPROVING', -7.0, - '[{"timestamp":"2026-08-07T10:00:00","avgTimeMs":0.55,"callCount":1800},{"timestamp":"2026-08-14T10:00:00","avgTimeMs":0.59,"callCount":2100}]'::json, - NOW() - INTERVAL '30 days', NOW() -), -( - 'seed-fp-audit-scan', '${connection_id}', 'd4e5f60718293a4b', - 'SELECT * FROM audit_log WHERE table_name = ? ORDER BY changed_at DESC LIMIT ?', - 'SELECT * FROM audit_log WHERE table_name = ''orders'' ORDER BY changed_at DESC LIMIT 1000', - 'SELECT', 1, '["audit_log"]'::json, - 7.68, 89.4, 450, 50000, 1000, - 3.20, 40.0, 200, NOW() - INTERVAL '14 days', - NOW() - INTERVAL '12 days', NOW() - INTERVAL '45 minutes', 8, - true, 'CRITICAL', 140.0, - '[{"timestamp":"2026-08-07T10:00:00","avgTimeMs":3.2,"callCount":200},{"timestamp":"2026-08-14T09:30:00","avgTimeMs":7.68,"callCount":450}]'::json, - NOW() - INTERVAL '12 days', NOW() -); - -INSERT INTO slow_query_run ( - id, connection_id, analysis_run_id, fingerprint, analyzed_on, captured_at, - calls_cumulative, calls_delta, total_exec_ms_cumulative, total_exec_ms_delta, - mean_exec_ms, max_exec_ms, p95_exec_ms, - rows_examined_delta, rows_sent_delta, - regression_factor, counter_reset, prev_run_id, created_at -) VALUES -('seed-run-o1-d6', '${connection_id}', 'seed-analysis-d6', 'a1b2c3d4e5f60718', CURRENT_DATE - 6, NOW() - INTERVAL '6 days', - 600, 600, 660.0, 660.0, 1.10, 28.0, 2.1, 9000, 250, NULL, false, NULL, NOW() - INTERVAL '6 days'), -('seed-run-o2-d6', '${connection_id}', 'seed-analysis-d6', 'b2c3d4e5f6071829', CURRENT_DATE - 6, NOW() - INTERVAL '6 days', - 500, 500, 925.0, 925.0, 1.85, 30.0, 3.2, 3000, 180, NULL, false, NULL, NOW() - INTERVAL '6 days'), -('seed-run-p1-d6', '${connection_id}', 'seed-analysis-d6', 'c3d4e5f60718293a', CURRENT_DATE - 6, NOW() - INTERVAL '6 days', - 1500, 1500, 825.0, 825.0, 0.55, 10.0, 0.9, 100, 100, NULL, false, NULL, NOW() - INTERVAL '6 days'), -('seed-run-a1-d6', '${connection_id}', 'seed-analysis-d6', 'd4e5f60718293a4b', CURRENT_DATE - 6, NOW() - INTERVAL '6 days', - 180, 180, 576.0, 576.0, 3.20, 40.0, 6.5, 20000, 1000, NULL, false, NULL, NOW() - INTERVAL '6 days'), -('seed-run-o1-d3', '${connection_id}', 'seed-analysis-d3', 'a1b2c3d4e5f60718', CURRENT_DATE - 3, NOW() - INTERVAL '3 days', - 950, 350, 1330.0, 670.0, 1.91, 36.0, 3.4, 5500, 140, 1.74, false, 'seed-run-o1-d6', NOW() - INTERVAL '3 days'), -('seed-run-o2-d3', '${connection_id}', 'seed-analysis-d3', 'b2c3d4e5f6071829', CURRENT_DATE - 3, NOW() - INTERVAL '3 days', - 720, 220, 1381.0, 456.0, 2.07, 34.0, 3.8, 1800, 90, 1.12, false, 'seed-run-o2-d6', NOW() - INTERVAL '3 days'), -('seed-run-p1-d3', '${connection_id}', 'seed-analysis-d3', 'c3d4e5f60718293a', CURRENT_DATE - 3, NOW() - INTERVAL '3 days', - 1850, 350, 1036.0, 211.0, 0.60, 11.0, 1.0, 100, 100, 1.09, false, 'seed-run-p1-d6', NOW() - INTERVAL '3 days'), -('seed-run-a1-d3', '${connection_id}', 'seed-analysis-d3', 'd4e5f60718293a4b', CURRENT_DATE - 3, NOW() - INTERVAL '3 days', - 310, 130, 1488.0, 912.0, 7.02, 72.0, 14.0, 28000, 1000, 2.19, false, 'seed-run-a1-d6', NOW() - INTERVAL '3 days'), -('seed-run-o1-d0', '${connection_id}', 'seed-analysis-d0', 'a1b2c3d4e5f60718', CURRENT_DATE, NOW() - INTERVAL '1 hour', - 1250, 300, 2337.5, 1007.5, 3.36, 45.2, 5.8, 6000, 120, 1.76, false, 'seed-run-o1-d3', NOW()), -('seed-run-o2-d0', '${connection_id}', 'seed-analysis-d0', 'b2c3d4e5f6071829', CURRENT_DATE, NOW() - INTERVAL '1 hour', - 890, 170, 1850.2, 469.2, 2.76, 38.7, 4.9, 2200, 80, 1.33, false, 'seed-run-o2-d3', NOW()), -('seed-run-p1-d0', '${connection_id}', 'seed-analysis-d0', 'c3d4e5f60718293a', CURRENT_DATE, NOW() - INTERVAL '1 hour', - 2100, 250, 1239.0, 203.0, 0.81, 12.3, 1.3, 100, 100, 1.35, false, 'seed-run-p1-d3', NOW()), -('seed-run-a1-d0', '${connection_id}', 'seed-analysis-d0', 'd4e5f60718293a4b', CURRENT_DATE, NOW() - INTERVAL '1 hour', - 450, 140, 3456.0, 1968.0, 14.06, 89.4, 28.0, 25000, 1000, 2.00, false, 'seed-run-a1-d3', NOW()); - -INSERT INTO slow_query_sample ( - id, connection_id, fingerprint, customer_id, captured_at, ingested_at, - exec_ms, rows_examined, rows_sent, source, raw_sql -) VALUES -('seed-sample-1', '${connection_id}', 'a1b2c3d4e5f60718', '1001', NOW() - INTERVAL '2 hours', NOW() - INTERVAL '1 hour', - 42.5, 18000, 120, 'SLOW_LOG', 'SELECT * FROM orders WHERE LOWER(status) = ''delivered'' AND total_amount > 100 /* cust=1001 */'), -('seed-sample-2', '${connection_id}', 'a1b2c3d4e5f60718', '1002', NOW() - INTERVAL '90 minutes', NOW() - INTERVAL '1 hour', - 38.1, 16000, 95, 'SLOW_LOG', 'SELECT * FROM orders WHERE LOWER(status) = ''delivered'' AND total_amount > 250 /* cust=1002 */'), -('seed-sample-3', '${connection_id}', 'd4e5f60718293a4b', '1001', NOW() - INTERVAL '80 minutes', NOW() - INTERVAL '1 hour', - 88.2, 52000, 1000, 'SLOW_LOG', 'SELECT * FROM audit_log WHERE table_name = ''orders'' ORDER BY changed_at DESC LIMIT 1000 /* cust=1001 */'), -('seed-sample-4', '${connection_id}', 'b2c3d4e5f6071829', '1003', NOW() - INTERVAL '70 minutes', NOW() - INTERVAL '1 hour', - 29.4, 4800, 40, 'SLOW_LOG', 'SELECT o.*, c.* FROM orders o JOIN customers c ON o.customer_id = c.id WHERE o.status = ''pending'' AND o.payment_status = ''paid'' /* cust=1003 */'), -('seed-sample-5', '${connection_id}', 'c3d4e5f60718293a', NULL, NOW() - INTERVAL '60 minutes', NOW() - INTERVAL '1 hour', - 11.2, 100, 100, 'SLOW_LOG', 'SELECT p.*, AVG(r.rating) FROM products p LEFT JOIN product_reviews r ON p.id = r.product_id GROUP BY p.id'); - -INSERT INTO slow_query_customer ( - id, connection_id, customer_id, customer_name, tenant_column, - first_seen_at, last_seen_at, name_resolved_at, created_at -) VALUES -('seed-cust-1001', '${connection_id}', '1001', 'acme@demo.local', 'customer_id', NOW() - INTERVAL '10 days', NOW() - INTERVAL '70 minutes', NOW() - INTERVAL '1 day', NOW()), -('seed-cust-1002', '${connection_id}', '1002', 'globex@demo.local', 'customer_id', NOW() - INTERVAL '8 days', NOW() - INTERVAL '90 minutes', NOW() - INTERVAL '1 day', NOW()), -('seed-cust-1003', '${connection_id}', '1003', 'initech@demo.local', 'customer_id', NOW() - INTERVAL '5 days', NOW() - INTERVAL '70 minutes', NOW() - INTERVAL '1 day', NOW()); - -INSERT INTO slow_query_customer_day ( - id, connection_id, fingerprint, customer_id, day, - sample_count, mean_exec_ms, max_exec_ms, total_exec_ms, - prev_day_mean_ms, regression_factor, created_at -) VALUES -('seed-cday-1', '${connection_id}', 'a1b2c3d4e5f60718', '1001', CURRENT_DATE, 18, 4.2, 42.5, 75.6, 2.1, 2.0, NOW()), -('seed-cday-2', '${connection_id}', 'a1b2c3d4e5f60718', '1002', CURRENT_DATE, 12, 3.8, 38.1, 45.6, 2.4, 1.58, NOW()), -('seed-cday-3', '${connection_id}', 'd4e5f60718293a4b', '1001', CURRENT_DATE, 9, 15.1, 88.2, 135.9, 7.0, 2.16, NOW()), -('seed-cday-4', '${connection_id}', 'b2c3d4e5f6071829', '1003', CURRENT_DATE, 14, 2.9, 29.4, 40.6, 2.2, 1.32, NOW()); + NOW() +) +ON CONFLICT DO NOTHING; -SELECT 'Performance seed data inserted successfully' AS status; +SELECT 'Brain notes seeded' AS status; EOSQL - - echo " Performance data seeded (history + Query Trends analytics + log source)." - - # Optional: exercise demo_shop so pg_stat_statements has matching patterns - echo " Simulating demo_shop workload (slow-query patterns)..." - compose exec -T postgres psql -U postgres -d demo_shop -v ON_ERROR_STOP=1 <<'EOWORK' >/dev/null || echo " Note: demo_shop workload simulation skipped (DB missing?)" -DO $$ -DECLARE i int; -BEGIN - FOR i IN 1..25 LOOP - PERFORM count(*) FROM orders WHERE LOWER(status) = 'delivered' AND total_amount > 100; - PERFORM count(*) FROM orders o JOIN customers c ON o.customer_id = c.id - WHERE o.status = 'pending' AND o.payment_status = 'paid'; - PERFORM p.id FROM products p - LEFT JOIN product_reviews r ON p.id = r.product_id GROUP BY p.id; - PERFORM 1 FROM audit_log WHERE table_name = 'orders' ORDER BY changed_at DESC LIMIT 1000; - END LOOP; -END $$; -EOWORK + echo " Curated Brain notes created." else - echo " Skipping performance data (no connection ID available)" + echo " Skipping Brain notes (no connection ID available)" fi # ============================================================================ -# Step 7: Seed agent conversation (sample prompts) +# Step 11: Trigger initial slow query analysis from pg_stat_statements # ============================================================================ echo "" -echo "Step 7: Seeding agent conversation history..." +echo "Step 11: Triggering initial slow query analysis..." if [[ -n "${connection_id:-}" ]]; then - # Get admin user ID - admin_id="$(compose exec -T postgres psql -U postgres -d dba_agent -At -c \ - "SELECT id FROM users WHERE username = 'admin' LIMIT 1" 2>/dev/null || echo "")" + analysis_result="$(curl -sS -b "$cookie_jar" \ + -H 'Content-Type: application/json' \ + -X POST "$base/slow-query-analytics/${connection_id}/analyze" \ + -d '{"threshold": 1.0, "limit": 50}' 2>/dev/null || echo "{}")" - if [[ -n "$admin_id" ]]; then - sample_transcript=$(cat <<'TRANSCRIPT' -[ - { - "role": "user", - "content": "What are the top 5 most ordered products in the last 30 days?" - }, - { - "role": "assistant", - "content": "I'll query the orders and order_items tables to find the most popular products.\n\nBased on the data, the top 5 most ordered products in the last 30 days are:\n\n1. **ProPhone X15** (SKU: ELEC-SP-002) - 156 orders, $155,843.44 revenue\n2. **AirPods Pro 3** (SKU: ELEC-AU-001) - 142 orders, $39,758.58 revenue\n3. **Budget Phone A5** (SKU: ELEC-SP-007) - 128 orders, $25,598.72 revenue\n4. **Atomic Habits** (SKU: BOOK-002) - 119 orders, $2,260.81 revenue\n5. **Air Fryer XL** (SKU: HOME-KT-005) - 98 orders, $12,739.02 revenue\n\nNotably, electronics dominate the top sellers, with smartphones and audio accessories being particularly popular." - }, - { - "role": "user", - "content": "Can you suggest indexes to improve query performance?" - }, - { - "role": "assistant", - "content": "Based on my analysis of your slow queries and access patterns, here are the recommended indexes:\n\n**High Priority:**\n1. `CREATE INDEX idx_orders_status_payment_created ON orders(status, payment_status, created_at DESC);`\n - Impact: 85% improvement for order filtering queries\n - Affects: 12 slow queries\n\n2. `CREATE INDEX idx_audit_log_table_changed ON audit_log(table_name, changed_at DESC);`\n - Impact: 90% improvement for audit log access\n - Current queries scan 50K+ rows\n\n**Medium Priority:**\n3. `CREATE INDEX idx_reviews_product_approved ON product_reviews(product_id, is_approved) WHERE is_approved = true;`\n - Partial index for approved reviews only\n - Improves product page load times\n\nWould you like me to generate the full DDL statements or analyze any specific queries?" - } -] -TRANSCRIPT -) - # Escape for SQL - transcript_escaped="${sample_transcript//\'/\'\'}" - - compose exec -T postgres psql -U postgres -d dba_agent -v ON_ERROR_STOP=1 </dev/null || echo "{}")" - current_stage="$(printf '%s' "$init_status" | sed -n 's/.*"currentStage":"\([^"]*\)".*/\1/p')" + # Query the slow-queries endpoint to verify it returns data + slow_query_check="$(curl -sS -b "$cookie_jar" \ + "$base/slow-query-analytics/${connection_id}/queries?limit=5" 2>/dev/null || echo "[]")" - if [[ "$current_stage" == "COMPLETED" ]]; then - echo " Brain initialization already completed." - elif [[ "$current_stage" == "FAILED" ]]; then - echo " Note: Brain initialization previously failed. Check LLM configuration." + # Count how many slow queries were returned + slow_count="$(echo "$slow_query_check" | python3 -c " +import sys, json +try: + data = json.load(sys.stdin) + if isinstance(data, list): + print(len(data)) + elif isinstance(data, dict) and 'content' in data: + print(len(data['content'])) + else: + print(0) +except: + print(0) +" 2>/dev/null || echo "0")" + + if [[ "$slow_count" -gt 0 ]]; then + echo " ✓ Slow queries endpoint returned $slow_count queries." else - echo " Brain initialization is in progress or pending." - echo " Stage: ${current_stage:-UNKNOWN}" - echo " The initialization will continue in the background." - echo " Run 'curl http://localhost:${DEEPSQL_BACKEND_PORT}/api/connections/${connection_id}/init-status' to check progress." + echo "" + echo " ⚠️ WARNING: Slow queries endpoint returned 0 rows!" + echo " The Performance tab will show 'No Slow Queries Found'." + echo " This defeats the purpose of the demo." + echo "" + echo " Possible causes:" + echo " - Data volume too small (need 300K+ audit_log, 100K+ order_items)" + echo " - Workload queries too fast (need mean_exec_time > 100ms)" + echo " - pg_stat_statements was reset after workload ran" + echo "" + echo " Check pg_stat_statements directly:" + echo " psql -U postgres -d dba_agent -c \"SELECT LEFT(query,60), calls, mean_exec_time::numeric(10,2) FROM pg_stat_statements WHERE dbid = (SELECT oid FROM pg_database WHERE datname = 'demo_shop') ORDER BY mean_exec_time DESC LIMIT 10;\"" + echo "" fi else - echo " Skipping brain init check (no connection ID available)" + echo " Skipping validation (no connection ID available)" fi # ============================================================================ @@ -981,29 +953,23 @@ echo "==========================================" echo "" echo "What was created:" echo " - demo_shop database with e-commerce schema" -echo " - Sample products, customers, orders (5000+)" +echo " - 5,000 orders, 100K+ order_items, 300K+ audit_log rows" echo " - Demo connection: ${DEEPSQL_SEED_CONNECTION_NAME}" +echo " Using read-only role: ${DEMO_ROLE_USER}" if [[ -n "${connection_id:-}" ]]; then echo " - Connection ID: ${connection_id}" fi +echo " - Real slow query workload captured by pg_stat_statements" +echo " - Sample 'Demo Shop Overview' dashboard (works without LLM key)" echo " - Saved queries in SQL Editor" -echo " - Sample slow query analysis (legacy history JSON)" -echo " - Slow-log source + Query Trends analytics (fingerprints / runs / samples)" -echo " - Per-customer rollups for the By Customer view" -echo " - Index recommendations" -echo " - Performance actions" -echo " - Sample agent conversation" -echo "" -echo "Demo users (create manually if API creation failed):" -echo " - analyst@demo.local / analyst123!" -echo " - developer@demo.local / developer123!" -echo " - viewer@demo.local / viewer123!" +echo " - Index recommendations from actual slow queries" +echo " - Digest preferences (daily at 9 AM)" +echo " - Curated Brain notes for key tables" echo "" echo "Next steps:" echo " 1. Open http://localhost:${DEEPSQL_FRONTEND_PORT:-3000}" -echo " 2. Select '${DEEPSQL_SEED_CONNECTION_NAME}' connection" -echo " 3. Open Performance — Query Trends, By Customer, and Workload tabs" -echo " 4. On Workload, click Run analysis for a fresh holistic report" -echo " 5. Try the SQL Editor with pre-saved queries" -echo " 6. Ask the Agent about the database" +echo " 2. The '${DEEPSQL_SEED_CONNECTION_NAME}' connection is ready" +echo " 3. Check Performance tab for slow queries from pg_stat_statements" +echo " 4. View the sample dashboard in Dashboards tab" +echo " 5. Try the sample prompts in the Agent tab" echo "" diff --git a/src/components/AgentChat/AgentChatPanel.jsx b/src/components/AgentChat/AgentChatPanel.jsx index 39fe406..5b744d6 100644 --- a/src/components/AgentChat/AgentChatPanel.jsx +++ b/src/components/AgentChat/AgentChatPanel.jsx @@ -51,12 +51,31 @@ function deriveTitle(messages) { // systems, SaaS multi-tenant DBs, analytics warehouses, ...). Never hardcode // a domain-specific table/column name here (see the chat guardrail in // AGENTS.md); AgentChatPanel has no idea what tables the active connection has. -const SUGGESTIONS = [ +const DEFAULT_SUGGESTIONS = [ { icon: Hash, text: 'How many tables are there?' }, { icon: Table2, text: 'Show the largest tables' }, { icon: Clock, text: 'What are the top slow queries?' }, ] +// Demo Shop specific prompts — shown when the Demo Shop connection is active. +// These prompts demonstrate real use cases with the demo e-commerce schema. +const DEMO_SHOP_SUGGESTIONS = [ + { icon: Clock, text: 'Why are order queries slow?' }, + { icon: Table2, text: 'Top 10 customers by revenue last 30 days' }, + { icon: Hash, text: 'Is it safe to add an index on orders.customer_id?' }, + { icon: Database, text: 'Which tables need indexes?' }, + { icon: Sparkles, text: 'Show revenue by product category as a chart' }, + { icon: AlertCircle, text: 'Are there any tables without primary keys?' }, +] + +// Returns suggestions based on whether this is a demo connection +function getSuggestions(connectionName) { + if (connectionName && connectionName.toLowerCase().includes('demo')) { + return DEMO_SHOP_SUGGESTIONS + } + return DEFAULT_SUGGESTIONS +} + export default function AgentChatPanel({ connectionId, connectionName, canManageContent = false }) { const { username } = useAuth() const [sessionId, setSessionId] = useState(null) @@ -307,7 +326,7 @@ export default function AgentChatPanel({ connectionId, connectionName, canManage

Ask about your database

Get instant answers grounded on your schema, live data, and query history.

- {SUGGESTIONS.map(({ icon: Icon, text }, i) => ( + {getSuggestions(connectionName).map(({ icon: Icon, text }, i) => (
+ {/* Demo Database Option - shown first as the recommended quick start */} + {!demoExists && ( +
+
+
+ +
+
+

Try the Demo Database

+

+ Get started instantly with our pre-configured e-commerce demo (5,000+ orders, + slow queries, sample dashboards). No setup required. +

+ +
+
+
+ )} + + {demoExists && ( +
+
+ + Demo database already connected. Add your own database below. +
+
+ )} + +
+
+
+
+
+ Or connect your own +
+
+ {/* DB type */}
diff --git a/src/components/sections/DigestSection.jsx b/src/components/sections/DigestSection.jsx index 8bdcaf4..1629ced 100644 --- a/src/components/sections/DigestSection.jsx +++ b/src/components/sections/DigestSection.jsx @@ -10,15 +10,15 @@ import styles from './DigestSection.module.css' // ───────────────────────────────────────────── function renderInline(text) { const parts = [] - const re = /(\*[^*]+\*|_[^_]+_|`[^`]+`)/g + // Match bold (*text*) and code (`text`). + // Skip underscore italics (_text_) to preserve index/table names like idx_orders_status. + const re = /(\*[^*]+\*|`[^`]+`)/g let last = 0, m, key = 0 while ((m = re.exec(text)) !== null) { if (m.index > last) parts.push(text.slice(last, m.index)) const raw = m[0] if (raw.startsWith('*') && raw.endsWith('*')) parts.push({raw.slice(1, -1)}) - else if (raw.startsWith('_') && raw.endsWith('_')) - parts.push({raw.slice(1, -1)}) else if (raw.startsWith('`') && raw.endsWith('`')) parts.push({raw.slice(1, -1)}) last = m.index + raw.length @@ -146,7 +146,9 @@ function DigestCard({ digest }) { } function DigestSection({ section }) { - const nonEmpty = section.lines.filter(l => l.trim()) + const nonEmpty = section.lines + .filter(l => l.trim()) + .filter(l => !l.includes('[sig:')) // Hide internal signature/dedupe markers return (
{section.title}
diff --git a/src/components/sections/SlowQueriesSection.jsx b/src/components/sections/SlowQueriesSection.jsx index 7cfbe58..a62281c 100644 --- a/src/components/sections/SlowQueriesSection.jsx +++ b/src/components/sections/SlowQueriesSection.jsx @@ -1,5 +1,5 @@ import { useRef, useState } from 'react' -import { Activity, FileText, LineChart, Loader2, Settings, Users } from 'lucide-react' +import { Activity, FileText, LineChart, Loader2, Settings, Users, Database } from 'lucide-react' import { useConnectionManager } from '@/lib/hooks/useConnectionManager' import { useSlowLogSourceConfig } from '@/lib/hooks/queries' import QueryTrendsTab from '@/components/tabs/Performance/QueryTrendsTab' @@ -7,15 +7,17 @@ import CustomerExplorer from '@/components/tabs/Performance/CustomerExplorer' import SlowQuerySettingsPanel from '@/components/tabs/Performance/SlowQuerySettingsPanel' import WorkloadAnalysisPanel from '@/components/tabs/Performance/WorkloadAnalysisPanel' import SlowQuerySourceModal from '@/components/SlowQuerySourceModal' +import SlowQueryAnalysisTab from '@/components/tabs/Performance/SlowQueryAnalysisTab' import { HelpTooltip } from '@/components/tabs/Brain/components/HelpTooltip' import sectionStyles from './TopLevelSection.module.css' import styles from './SlowQueriesSection.module.css' const TABS = [ - { id: 'trends', label: 'Query Trends', icon: LineChart }, - { id: 'customers', label: 'By Customer', icon: Users }, - { id: 'workload', label: 'Workload', icon: Activity }, - { id: 'settings', label: 'Settings', icon: Settings }, + { id: 'analysis', label: 'Analysis', icon: Database, requiresLogSource: false }, + { id: 'trends', label: 'Query Trends', icon: LineChart, requiresLogSource: true }, + { id: 'customers', label: 'By Customer', icon: Users, requiresLogSource: true }, + { id: 'workload', label: 'Workload', icon: Activity, requiresLogSource: true }, + { id: 'settings', label: 'Settings', icon: Settings, requiresLogSource: false }, ] const LOG_SOURCE_HELP = { @@ -24,6 +26,12 @@ const LOG_SOURCE_HELP = { 'Query trends, per-customer load, and workload analysis all read from ingested slow-query logs. Attach CloudWatch, S3, Azure, GCP, Datadog, Elasticsearch, or a file upload before those views can run.', } +const PG_STAT_HELP = { + title: 'pg_stat_statements', + description: + 'PostgreSQL connections use pg_stat_statements for real-time query analysis. No external log source needed. The Analysis tab shows top queries by execution time with index recommendations.', +} + /** * Combined Performance section — Slow Queries + Workload Analysis. * @@ -32,8 +40,20 @@ const LOG_SOURCE_HELP = { */ export default function SlowQueriesSection() { const { connectionId, selectedConnection, isLoading } = useConnectionManager() - const [tab, setTab] = useState('trends') const tabRefs = useRef({}) + const logSourceQ = useSlowLogSourceConfig(connectionId) + const hasLogSource = Boolean(logSourceQ.data?.id) + + // PostgreSQL connections can use pg_stat_statements directly without a log source + const isPostgres = ['postgresql', 'postgres'].includes(selectedConnection?.dbType?.toLowerCase()) + const canShowPerformance = hasLogSource || isPostgres + + // Filter tabs based on available data sources + const availableTabs = TABS.filter((t) => !t.requiresLogSource || hasLogSource) + + // Default to 'analysis' for PostgreSQL without log source, otherwise 'trends' + const defaultTab = (isPostgres && !hasLogSource) ? 'analysis' : 'trends' + const [tab, setTab] = useState(defaultTab) /** Arrow / Home / End move between tabs, as the ARIA tabs pattern expects. */ const onTabKeyDown = (e) => { @@ -44,11 +64,11 @@ export default function SlowQueriesSection() { // handler was created: two keypresses within one render would otherwise both move // relative to the same starting index and selection would stick after the first. setTab((current) => { - const i = TABS.findIndex((t) => t.id === current) + const i = availableTabs.findIndex((t) => t.id === current) const next = e.key === 'Home' ? 0 - : e.key === 'End' ? TABS.length - 1 - : (i + step + TABS.length) % TABS.length - const id = TABS[next].id + : e.key === 'End' ? availableTabs.length - 1 + : (i + step + availableTabs.length) % availableTabs.length + const id = availableTabs[next].id // Focus follows selection, per the ARIA tabs pattern. Deferred so the tab is // already rendered with tabIndex=0 when we focus it. queueMicrotask(() => tabRefs.current[id]?.focus()) @@ -56,8 +76,6 @@ export default function SlowQueriesSection() { }) } const [logSourceModalOpen, setLogSourceModalOpen] = useState(false) - const logSourceQ = useSlowLogSourceConfig(connectionId) - const hasLogSource = Boolean(logSourceQ.data?.id) // Wait for connection list to load before rendering anything if (isLoading) { @@ -97,7 +115,7 @@ export default function SlowQueriesSection() {
Could not load the slow query log configuration for this connection.
- ) : !hasLogSource ? ( + ) : !canShowPerformance ? (

Configure slow queries

@@ -131,7 +149,7 @@ export default function SlowQueriesSection() { aria-label="Performance views" onKeyDown={onTabKeyDown} > - {TABS.map((t) => { + {availableTabs.map((t) => { const Icon = t.icon const active = tab === t.id return ( @@ -166,6 +184,25 @@ export default function SlowQueriesSection() {
+ {/* Show pg_stat_statements info banner for PostgreSQL without log source */} + {isPostgres && !hasLogSource && ( +
+ + + + Using pg_stat_statements for real-time query analysis. + + + +
+ )} +
+ {tab === 'analysis' && } {tab === 'trends' && } {tab === 'customers' && } {tab === 'workload' && } diff --git a/src/components/sections/SlowQueriesSection.module.css b/src/components/sections/SlowQueriesSection.module.css index 3f89f5f..24dda63 100644 --- a/src/components/sections/SlowQueriesSection.module.css +++ b/src/components/sections/SlowQueriesSection.module.css @@ -152,3 +152,41 @@ @keyframes spin { to { transform: rotate(360deg); } } + +.pgStatBanner { + display: flex; + align-items: center; + justify-content: space-between; + gap: 12px; + padding: 10px 14px; + margin-bottom: 16px; + border: 1px solid #d1fae5; + border-radius: 10px; + background: #ecfdf5; + font-size: 13px; +} + +.pgStatBannerText { + display: inline-flex; + align-items: center; + gap: 8px; + color: #065f46; + font-weight: 500; +} + +.pgStatBannerLink { + padding: 0; + border: none; + background: none; + color: #059669; + font: inherit; + font-size: 12px; + font-weight: 600; + cursor: pointer; + text-decoration: underline; + text-underline-offset: 2px; +} + +.pgStatBannerLink:hover { + color: #047857; +} diff --git a/src/components/tabs/Performance/SlowQueryAnalysisTab.js b/src/components/tabs/Performance/SlowQueryAnalysisTab.js index 07cd66a..0b07993 100644 --- a/src/components/tabs/Performance/SlowQueryAnalysisTab.js +++ b/src/components/tabs/Performance/SlowQueryAnalysisTab.js @@ -299,7 +299,7 @@ export default function SlowQueryAnalysisTab({ connectionId }) { // Filter states const [timeRange, setTimeRange] = useState("LAST_24_HOURS"); - const [thresholdMs, setThresholdMs] = useState(100); + const [thresholdMs, setThresholdMs] = useState(10); const [limit, setLimit] = useState(10); const [showFilters, setShowFilters] = useState(false); @@ -399,6 +399,39 @@ export default function SlowQueryAnalysisTab({ connectionId }) { } }, [activeJobData, activeJobFetched, activeJobUpdatedAt]); + // Auto-run analysis from pg_stat_statements on first load if no analysis exists + // This makes pg_stat_statements the primary/default data source for Postgres + const [autoAnalysisAttempted, setAutoAnalysisAttempted] = useState(false); + useEffect(() => { + // Only run auto-analysis once per mount, when we have a connectionId, + // analysis data has been fetched (even if empty), and no analysis exists yet + if ( + connectionId && + !autoAnalysisAttempted && + !loadingAnalysis && + !latestAnalysisData?.analysisData && + !analyzeSlowQueriesMutation.isPending && + canRunAnalysis + ) { + setAutoAnalysisAttempted(true); + // Auto-run analysis from pg_stat_statements + analyzeSlowQueriesMutation.mutate({ + connectionId, + threshold: thresholdMs, + limit, + }); + } + }, [ + connectionId, + autoAnalysisAttempted, + loadingAnalysis, + latestAnalysisData, + analyzeSlowQueriesMutation, + canRunAnalysis, + thresholdMs, + limit, + ]); + // Handlers using mutations const handleAcknowledgeRegression = async (regressionId) => { try { @@ -1167,8 +1200,8 @@ export default function SlowQueryAnalysisTab({ connectionId }) {

Slow Query Analysis

- Identify and optimize slow-running queries to improve database - performance + Analyzing slow queries from pg_stat_statements (PostgreSQL) or + performance_schema (MySQL). Cloud log sources are optional extras.

@@ -1323,10 +1356,10 @@ export default function SlowQueryAnalysisTab({ connectionId }) { {/* Three Option Cards */}
- {/* Option 1: Query from Database */} + {/* Option 1: Query from Database (PRIMARY - pg_stat_statements/performance_schema) */}

Query from Database

- Analyze slow queries from built-in performance tables. - MySQL: performance_schema + mysql.slow_log. PostgreSQL: - pg_stat_statements. + Recommended: Analyze slow queries directly from your database. + PostgreSQL uses pg_stat_statements. MySQL uses performance_schema. + No external setup required.

+ Recommended Instant - No setup required + No setup
@@ -1384,7 +1418,7 @@ export default function SlowQueryAnalysisTab({ connectionId }) {
- {/* Option 3: Connect Cloud Source */} + {/* Option 3: Connect Cloud Source (OPTIONAL - for managed databases) */}

Connect Cloud Source

- Set up automatic ingestion from AWS CloudWatch Logs or S3 - buckets. Support for Azure, GCP, and others coming soon. + Optional: Ingest logs from AWS CloudWatch or S3 for managed + databases (RDS, Aurora). Most users should use Query from Database above.

+ Optional CloudWatch S3 - - More coming soon -
{hasExternalSource ? ( diff --git a/src/components/tabs/Performance/SlowQueryAnalysisTab.module.css b/src/components/tabs/Performance/SlowQueryAnalysisTab.module.css index 21edc8f..0ac34d1 100644 --- a/src/components/tabs/Performance/SlowQueryAnalysisTab.module.css +++ b/src/components/tabs/Performance/SlowQueryAnalysisTab.module.css @@ -960,6 +960,23 @@ background: var(--color-light-2); } +.optionCardPrimary { + border-color: var(--color-primary); + border-width: 2px; + background: linear-gradient(180deg, #f8fdf9 0%, white 100%); + box-shadow: 0 2px 8px rgba(47, 122, 109, 0.1); +} + +.optionCardPrimary:hover { + border-color: var(--color-primary); + box-shadow: 0 4px 16px rgba(47, 122, 109, 0.15); +} + +.optionCardPrimary .optionIconWrapper { + background: var(--color-primary); + color: white; +} + .optionIconWrapper { display: flex; align-items: center; diff --git a/src/lib/hooks/queries/useSlowQueries.js b/src/lib/hooks/queries/useSlowQueries.js index 557ff25..0d28128 100644 --- a/src/lib/hooks/queries/useSlowQueries.js +++ b/src/lib/hooks/queries/useSlowQueries.js @@ -49,7 +49,7 @@ export function useAnalyzeSlowQueries() { const queryClient = useQueryClient(); return useMutation({ - mutationFn: ({ connectionId, threshold = 100, limit = 10 }) => + mutationFn: ({ connectionId, threshold = 10, limit = 10 }) => slowQueriesAPI.analyzeSlowQueries(connectionId, threshold, limit), onSuccess: (_data, { connectionId }) => { queryClient.invalidateQueries({