From befdb6b31e0a2af68fdd282e696a818cfe750da8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Fredrik=20Adel=C3=B6w?= Date: Mon, 11 May 2026 18:50:49 +0200 Subject: [PATCH] catalog-backend: add migration test for search_indices_and_dedup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tests two scenarios across all supported databases: Preconditions NOT met (dedup runs): inserts five rows with two duplicate (entity_id, key, value) pairs (one with null value), runs the migration, and verifies that duplicates are removed, null values are handled correctly, and the unique constraint is enforced post-migration. The down migration is verified to drop the constraint so duplicates can be inserted again. Preconditions met (PostgreSQL only, dedup skipped): pre-creates the unique index to simulate a user who ran the manual SQL before deploying, re-runs the migration, and verifies the row count is unchanged — confirming the fast path fires correctly. Non-PostgreSQL databases skip the fast-path test with an early return since the pg_index check is PostgreSQL-specific. Signed-off-by: Fredrik Adelöw Co-authored-by: Cursor --- .../src/tests/migrations.test.ts | 158 ++++++++++++++++++ 1 file changed, 158 insertions(+) diff --git a/plugins/catalog-backend/src/tests/migrations.test.ts b/plugins/catalog-backend/src/tests/migrations.test.ts index 6e067a8126..4b1153f3c6 100644 --- a/plugins/catalog-backend/src/tests/migrations.test.ts +++ b/plugins/catalog-backend/src/tests/migrations.test.ts @@ -1094,6 +1094,164 @@ describe('migrations', () => { }, ); + it.each(databases.eachSupportedId())( + '20260510000000_search_indices_and_dedup.js, %p', + async databaseId => { + const knex = await databases.init(databaseId); + + await migrateUntilBefore( + knex, + '20260510000000_search_indices_and_dedup.js', + ); + + // Set up FK dependencies: search → final_entities → refresh_state + await knex('refresh_state').insert({ + entity_id: 'e1', + entity_ref: 'k:ns/n1', + unprocessed_entity: '{}', + errors: '[]', + next_update_at: knex.fn.now(), + last_discovery_at: knex.fn.now(), + }); + await knex('final_entities').insert({ + entity_id: 'e1', + entity_ref: 'k:ns/n1', + hash: 'h1', + final_entity: '{}', + }); + + // Insert search rows with duplicates on (entity_id, key, value). + // Both copies of k1 share the same original_value so the surviving row + // is unambiguous across all database dedup strategies. + // The two null-value rows verify that null dedup does not conflate null + // with non-null or remove more rows than it should. + await knex('search').insert([ + { entity_id: 'e1', key: 'k1', value: 'v1', original_value: 'v1' }, // first copy + { entity_id: 'e1', key: 'k1', value: 'v1', original_value: 'v1' }, // duplicate — removed + { entity_id: 'e1', key: 'k2', value: null, original_value: null }, // null first — kept + { entity_id: 'e1', key: 'k2', value: null, original_value: null }, // null duplicate — removed + { entity_id: 'e1', key: 'k3', value: 'v3', original_value: 'v3' }, // unique — kept + ]); + + // Preconditions NOT met: dedup runs + await migrateUpOnce(knex); + + // 5 rows collapsed to 3 unique (entity_id, key, value) combinations + const rows = await knex('search').orderBy('key'); + expect(rows).toHaveLength(3); + expect(rows.map(r => r.key)).toEqual(['k1', 'k2', 'k3']); + expect(rows[0]).toEqual( + expect.objectContaining({ + key: 'k1', + value: 'v1', + original_value: 'v1', + }), + ); + // null-value row survives correctly + expect(rows[1]).toEqual( + expect.objectContaining({ + key: 'k2', + value: null, + original_value: null, + }), + ); + + // Unique constraint is now in place + await expect( + knex('search').insert({ + entity_id: 'e1', + key: 'k1', + value: 'v1', + original_value: 'extra', + }), + ).rejects.toEqual(expect.anything()); + + await migrateDownOnce(knex); + + // After rollback the unique constraint is gone — duplicates are allowed again + await expect( + knex('search').insert({ + entity_id: 'e1', + key: 'k1', + value: 'v1', + original_value: 'extra', + }), + ).resolves.not.toThrow(); + + await knex.destroy(); + }, + ); + + it.each(databases.eachSupportedId())( + '20260510000000_search_indices_and_dedup.js preconditions met (PG fast path), %p', + async databaseId => { + const knex = await databases.init(databaseId); + + // The fast path (skip dedup when unique index pre-exists) is a + // PostgreSQL-only code path that checks pg_index. + if (!knex.client.config.client.includes('pg')) { + await knex.destroy(); + return; + } + + await migrateUntilBefore( + knex, + '20260510000000_search_indices_and_dedup.js', + ); + + await knex('refresh_state').insert({ + entity_id: 'e1', + entity_ref: 'k:ns/n1', + unprocessed_entity: '{}', + errors: '[]', + next_update_at: knex.fn.now(), + last_discovery_at: knex.fn.now(), + }); + await knex('final_entities').insert({ + entity_id: 'e1', + entity_ref: 'k:ns/n1', + hash: 'h1', + final_entity: '{}', + }); + + // Insert clean rows (no duplicates) so the unique index can be created + await knex('search').insert([ + { entity_id: 'e1', key: 'k1', value: 'v1', original_value: 'V1' }, + { entity_id: 'e1', key: 'k2', value: null, original_value: null }, + ]); + + // Simulate a user who manually deduped and created the unique index + // before deploying this version + await knex.raw( + `CREATE UNIQUE INDEX search_entity_key_value_idx ON search (entity_id, key, value)`, + ); + + // Migration detects the existing unique index and skips the dedup step + await migrateUpOnce(knex); + + // Row count unchanged — no rows were removed by dedup + const rows = await knex('search').orderBy('key'); + expect(rows).toHaveLength(2); + expect(rows[0]).toEqual( + expect.objectContaining({ + key: 'k1', + value: 'v1', + original_value: 'V1', + }), + ); + expect(rows[1]).toEqual( + expect.objectContaining({ + key: 'k2', + value: null, + original_value: null, + }), + ); + + await migrateDownOnce(knex); + await knex.destroy(); + }, + ); + it.each(databases.eachSupportedId())( '20260403000000_add_location_entity_ref.js, %p', async databaseId => {