From 46e9c7314c49f00579fdb1ecef41839dfaae59d5 Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 11:50:54 -0700 Subject: [PATCH 1/4] chore(search): add paced projection replacement and data retirement --- packages/db/drizzle.config.ts | 2 + packages/db/maintenance/index.ts | 18 + .../search-retirement-health.test.ts | 266 +++++ .../maintenance/search-retirement-health.ts | 239 +++++ .../search-retirement.integration.ts | 656 ++++++++++++ packages/db/maintenance/search-retirement.md | 194 ++++ packages/db/maintenance/search-retirement.ts | 972 ++++++++++++++++++ packages/db/package.json | 1 + .../0027_retire_search_embeddings.ts | 41 +- .../search-embedding-retirement.md | 193 +--- packages/db/scripts/push.test.ts | 5 - packages/db/scripts/push.ts | 37 + packages/db/scripts/retire-indexed-search.ts | 208 ++++ packages/db/vitest.config.ts | 7 +- 14 files changed, 2621 insertions(+), 218 deletions(-) create mode 100644 packages/db/maintenance/index.ts create mode 100644 packages/db/maintenance/search-retirement-health.test.ts create mode 100644 packages/db/maintenance/search-retirement-health.ts create mode 100644 packages/db/maintenance/search-retirement.integration.ts create mode 100644 packages/db/maintenance/search-retirement.md create mode 100644 packages/db/maintenance/search-retirement.ts create mode 100644 packages/db/scripts/retire-indexed-search.ts diff --git a/packages/db/drizzle.config.ts b/packages/db/drizzle.config.ts index 42d5b878d2e..cffe391ded6 100644 --- a/packages/db/drizzle.config.ts +++ b/packages/db/drizzle.config.ts @@ -20,5 +20,7 @@ export default { '!script_migrations', '!search_embedding_cleanup_progress', '!search_embedding_cleanup_targets', + '!search_retirement_*', + '!embedding_search_retirement_*', ], } satisfies Config diff --git a/packages/db/maintenance/index.ts b/packages/db/maintenance/index.ts new file mode 100644 index 00000000000..624527cdbe7 --- /dev/null +++ b/packages/db/maintenance/index.ts @@ -0,0 +1,18 @@ +export { + abortSearchRetirement, + advanceSearchRetirement, + beginSearchRetirementPurge, + cutoverSearchRetirement, + finalizeSearchRetirement, + getSearchRetirementStatus, + initializeSearchRetirement, + SearchRetirementError, + type SearchRetirementPhase, + type SearchRetirementStatus, +} from '@sim/db/maintenance/search-retirement' +export { + RetireSearchHealthError, + readSearchRetirementHealth, + readSearchRetirementHealthLimits, + type SearchRetirementHealthLimits, +} from '@sim/db/maintenance/search-retirement-health' diff --git a/packages/db/maintenance/search-retirement-health.test.ts b/packages/db/maintenance/search-retirement-health.test.ts new file mode 100644 index 00000000000..56ea78fedbe --- /dev/null +++ b/packages/db/maintenance/search-retirement-health.test.ts @@ -0,0 +1,266 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + RetireSearchHealthError, + readSearchRetirementHealth, + readSearchRetirementHealthLimits, + type SearchRetirementHealthLimits, +} from '@sim/db/maintenance/search-retirement-health' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const NOW = Date.parse('2026-10-01T12:00:00.000Z') +const LIMITS: SearchRetirementHealthLimits = { + databaseId: 'test-database', + maxReplicaLagBytes: 1_000, + maxReplicaLagSeconds: 2, + maxWalBytesPerSecond: 2_000, + maxDatabaseP95Ms: 100, + maxCpuPercent: 50, + minFreeStorageBytes: 5_000, + maxSampleAgeMs: 10_000, +} +const SAMPLE = { + observedAt: '2026-10-01T12:00:00.000Z', + databaseId: 'test-database', + healthy: true, + replicaLagBytes: 0, + replicaLagSeconds: 0, + walBytesPerSecond: 100, + databaseP95Ms: 10, + cpuPercent: 10, + freeStorageBytes: 10_000, + maintenanceAllowed: true, + cutoverAllowed: false, +} + +describe('Search retirement external health gate', () => { + let directory: string + let path: string + + beforeEach(async () => { + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(NOW) + directory = await mkdtemp(join(tmpdir(), 'search-retirement-health-')) + path = join(directory, 'health.json') + await writeFile(path, JSON.stringify(SAMPLE)) + }) + + afterEach(async () => { + vi.useRealTimers() + await rm(directory, { recursive: true, force: true }) + }) + + it('reads a replacement sample before allowing another page', async () => { + expect(await readSearchRetirementHealth(path, LIMITS)).toBe(NOW) + await writeFile(path, JSON.stringify({ ...SAMPLE, maintenanceAllowed: false })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'maintenance_refused', + }) + }) + + it('requires separate cutover approval while allowing ordinary maintenance', async () => { + expect(await readSearchRetirementHealth(path, LIMITS)).toBe(NOW) + await expect(readSearchRetirementHealth(path, LIMITS, { cutover: true })).rejects.toMatchObject( + { + reason: 'cutover_refused', + } + ) + await writeFile(path, JSON.stringify({ ...SAMPLE, cutoverAllowed: true })) + expect(await readSearchRetirementHealth(path, LIMITS, { cutover: true })).toBe(NOW) + }) + + it('rejects an operator CPU ceiling above one hundred percent', async () => { + await writeFile(path, JSON.stringify({ ...LIMITS, maxCpuPercent: 100.001 })) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + }) + + it.each(['missing', 'directory'] as const)('rejects a %s health file', async (kind) => { + await expect( + readSearchRetirementHealth(kind === 'missing' ? join(directory, 'absent') : directory, LIMITS) + ).rejects.toBeInstanceOf(RetireSearchHealthError) + }) + + it('accepts at most eight KiB, including UTF-8 bytes and trailing whitespace', async () => { + const encoded = JSON.stringify(SAMPLE) + await writeFile(path, encoded.padEnd(8_192, ' ')) + expect(await readSearchRetirementHealth(path, LIMITS)).toBe(NOW) + await writeFile(path, encoded.padEnd(8_193, ' ')) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'health_file_too_large', + }) + await writeFile(path, JSON.stringify({ ...SAMPLE, padding: 'é'.repeat(4_096) })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'health_file_too_large', + }) + }) + + it.each(['', '{', 'null', '[]', 'true', '{"observedAt":123}'])( + 'rejects malformed or incomplete wire content: %s', + async (content) => { + await writeFile(path, content) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + } + ) + + it('rejects invalid UTF-8 instead of substituting a replacement character', async () => { + const encoded = Buffer.from(JSON.stringify(SAMPLE)) + const location = encoded.indexOf('test-database') + encoded[location] = 0xff + await writeFile(path, encoded) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it.each([ + { healthy: 'true' }, + { maintenanceAllowed: 1 }, + { cutoverAllowed: 'true' }, + { cutoverAllowed: undefined }, + { databaseId: '' }, + { unexpected: true }, + { observedAt: null }, + { replicaLagBytes: -1 }, + { replicaLagSeconds: null }, + { walBytesPerSecond: '100' }, + { databaseP95Ms: -1 }, + { cpuPercent: null }, + { freeStorageBytes: -1 }, + ])('rejects malformed sample field %j', async (fields) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, ...fields })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it('rejects a nonfinite JSON number', async () => { + await writeFile( + path, + JSON.stringify(SAMPLE).replace('"replicaLagBytes":0', '"replicaLagBytes":1e999') + ) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it.each([ + '2026-10-01T12:00:00+00:00', + '2026-10-01 12:00:00Z', + '2026-02-30T12:00:00.000Z', + '2026-10-01T24:00:00.000Z', + ])('rejects a noncanonical or impossible UTC timestamp: %s', async (observedAt) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, observedAt })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it.each([ + ['2026-10-01T11:59:49.999Z', 'stale_sample'], + ['2026-10-01T12:00:01.001Z', 'future_sample'], + ])('rejects a sample outside the time budget: %s', async (observedAt, reason) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, observedAt })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ reason }) + }) + + it('does not reuse a previously fresh sample once it expires', async () => { + await readSearchRetirementHealth(path, LIMITS) + vi.setSystemTime(NOW + LIMITS.maxSampleAgeMs + 1) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'stale_sample', + }) + }) + + it.each([ + [{ databaseId: 'other-database' }, 'identity_mismatch'], + [{ healthy: false }, 'unhealthy'], + [{ maintenanceAllowed: false }, 'maintenance_refused'], + [{ replicaLagBytes: 1_001 }, 'replica_lag_bytes'], + [{ replicaLagSeconds: 2.001 }, 'replica_lag_seconds'], + [{ walBytesPerSecond: 2_001 }, 'wal_rate'], + [{ databaseP95Ms: 101 }, 'database_latency'], + [{ cpuPercent: 51 }, 'cpu_usage'], + [{ freeStorageBytes: 4_999 }, 'storage_headroom'], + ] as const)('refuses unsafe sample %j', async (fields, reason) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, ...fields })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ reason }) + }) + + it('permits zero-lag limits and refuses any replication debt', async () => { + const limits = { ...LIMITS, maxReplicaLagBytes: 0, maxReplicaLagSeconds: 0 } + await readSearchRetirementHealth(path, limits) + await writeFile(path, JSON.stringify({ ...SAMPLE, replicaLagBytes: 1 })) + await expect(readSearchRetirementHealth(path, limits)).rejects.toMatchObject({ + reason: 'replica_lag_bytes', + }) + }) + + it.each([ + { databaseId: '' }, + { maxReplicaLagBytes: -1 }, + { maxReplicaLagSeconds: Number.NaN }, + { maxWalBytesPerSecond: 0 }, + { maxDatabaseP95Ms: Number.POSITIVE_INFINITY }, + { maxCpuPercent: 0 }, + { minFreeStorageBytes: 0 }, + { maxSampleAgeMs: 0 }, + { maxSampleAgeMs: 30_001 }, + ])('rejects limits that disable a guard: %j', async (fields) => { + await expect(readSearchRetirementHealth(path, { ...LIMITS, ...fields })).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + await writeFile(path, JSON.stringify({ ...LIMITS, ...fields })) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + }) + + it('loads the bounded operator policy and enforces it against the health sample', async () => { + const policyPath = join(directory, 'policy.json') + await writeFile(policyPath, JSON.stringify({ ...LIMITS, maxReplicaLagBytes: 0 })) + const limits = await readSearchRetirementHealthLimits(policyPath) + await writeFile(path, JSON.stringify({ ...SAMPLE, replicaLagBytes: 1 })) + await expect(readSearchRetirementHealth(path, limits)).rejects.toMatchObject({ + reason: 'replica_lag_bytes', + }) + }) + + it.each(['{}', 'null', '{', JSON.stringify({ ...LIMITS, unexpected: true })])( + 'rejects malformed operator policy content: %s', + async (content) => { + await writeFile(path, content) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + } + ) + + it('enforces the same eight-KiB cap on operator policies', async () => { + await writeFile(path, JSON.stringify(LIMITS).padEnd(8_193, ' ')) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'health_file_too_large', + }) + }) + + it('returns only typed generic reasons for payload and filesystem failures', async () => { + const privateMarker = 'private-operator-marker' + await writeFile(path, JSON.stringify({ ...SAMPLE, databaseId: privateMarker })) + for (const candidate of [path, join(directory, privateMarker)]) { + try { + await readSearchRetirementHealth(candidate, LIMITS) + expect.fail('The health gate should refuse this sample') + } catch (error) { + expect(error).toBeInstanceOf(RetireSearchHealthError) + if (!(error instanceof RetireSearchHealthError)) throw error + expect(error.message).not.toContain(privateMarker) + expect(error.message).not.toContain(directory) + expect(error.cause).toBeUndefined() + } + } + }) +}) diff --git a/packages/db/maintenance/search-retirement-health.ts b/packages/db/maintenance/search-retirement-health.ts new file mode 100644 index 00000000000..61d18109820 --- /dev/null +++ b/packages/db/maintenance/search-retirement-health.ts @@ -0,0 +1,239 @@ +import { constants } from 'node:fs' +import { open } from 'node:fs/promises' +import { isRecordLike } from '@sim/utils/object' + +const MAX_FILE_BYTES = 8_192 +const MAX_SAMPLE_AGE_MS = 30_000 +const MAX_FUTURE_SKEW_MS = 1_000 + +export interface SearchRetirementHealthLimits { + databaseId: string + maxReplicaLagBytes: number + maxReplicaLagSeconds: number + maxWalBytesPerSecond: number + maxDatabaseP95Ms: number + maxCpuPercent: number + minFreeStorageBytes: number + maxSampleAgeMs: number +} + +interface SearchRetirementHealthSample { + observedAt: string + databaseId: string + healthy: boolean + replicaLagBytes: number + replicaLagSeconds: number + walBytesPerSecond: number + databaseP95Ms: number + cpuPercent: number + freeStorageBytes: number + maintenanceAllowed: boolean + cutoverAllowed: boolean +} + +type HealthRefusalReason = + | 'health_file_unreadable' + | 'health_file_too_large' + | 'health_file_changed' + | 'invalid_sample' + | 'invalid_limits' + | 'identity_mismatch' + | 'stale_sample' + | 'future_sample' + | 'unhealthy' + | 'maintenance_refused' + | 'cutover_refused' + | 'replica_lag_bytes' + | 'replica_lag_seconds' + | 'wal_rate' + | 'database_latency' + | 'cpu_usage' + | 'storage_headroom' + +/** A safe refusal reason that never includes operator paths, database identity, or sample values. */ +export class RetireSearchHealthError extends Error { + constructor(readonly reason: HealthRefusalReason) { + super(`Search retirement health gate refused: ${reason}`) + this.name = 'RetireSearchHealthError' + } +} + +const SAMPLE_FIELDS = [ + 'observedAt', + 'databaseId', + 'healthy', + 'replicaLagBytes', + 'replicaLagSeconds', + 'walBytesPerSecond', + 'databaseP95Ms', + 'cpuPercent', + 'freeStorageBytes', + 'maintenanceAllowed', + 'cutoverAllowed', +] as const + +const LIMIT_FIELDS = [ + 'databaseId', + 'maxReplicaLagBytes', + 'maxReplicaLagSeconds', + 'maxWalBytesPerSecond', + 'maxDatabaseP95Ms', + 'maxCpuPercent', + 'minFreeStorageBytes', + 'maxSampleAgeMs', +] as const + +function isNonnegativeFinite(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) && value >= 0 +} + +function isPositiveFinite(value: unknown): value is number { + return isNonnegativeFinite(value) && value > 0 +} + +function hasExactFields(value: Record, fields: readonly string[]): boolean { + return Object.keys(value).length === fields.length && fields.every((field) => field in value) +} + +function assertLimits(value: unknown): asserts value is SearchRetirementHealthLimits { + if ( + !isRecordLike(value) || + !hasExactFields(value, LIMIT_FIELDS) || + typeof value.databaseId !== 'string' || + value.databaseId.trim().length === 0 || + !isNonnegativeFinite(value.maxReplicaLagBytes) || + !isNonnegativeFinite(value.maxReplicaLagSeconds) || + !isPositiveFinite(value.maxWalBytesPerSecond) || + !isPositiveFinite(value.maxDatabaseP95Ms) || + !isPositiveFinite(value.maxCpuPercent) || + value.maxCpuPercent > 100 || + !isPositiveFinite(value.minFreeStorageBytes) || + !isPositiveFinite(value.maxSampleAgeMs) || + value.maxSampleAgeMs > MAX_SAMPLE_AGE_MS + ) { + throw new RetireSearchHealthError('invalid_limits') + } +} + +function assertSample(value: unknown): asserts value is SearchRetirementHealthSample { + if ( + !isRecordLike(value) || + !hasExactFields(value, SAMPLE_FIELDS) || + typeof value.observedAt !== 'string' || + typeof value.databaseId !== 'string' || + value.databaseId.trim().length === 0 || + typeof value.healthy !== 'boolean' || + typeof value.maintenanceAllowed !== 'boolean' || + typeof value.cutoverAllowed !== 'boolean' || + !isNonnegativeFinite(value.replicaLagBytes) || + !isNonnegativeFinite(value.replicaLagSeconds) || + !isNonnegativeFinite(value.walBytesPerSecond) || + !isNonnegativeFinite(value.databaseP95Ms) || + !isNonnegativeFinite(value.cpuPercent) || + !isNonnegativeFinite(value.freeStorageBytes) + ) { + throw new RetireSearchHealthError('invalid_sample') + } +} + +async function readBoundedJson( + path: string, + invalidReason: 'invalid_sample' | 'invalid_limits' +): Promise { + try { + const file = await open(path, constants.O_RDONLY | constants.O_NONBLOCK) + try { + const before = await file.stat() + if (!before.isFile()) throw new RetireSearchHealthError('health_file_unreadable') + if (before.size > MAX_FILE_BYTES) throw new RetireSearchHealthError('health_file_too_large') + + const bytes = Buffer.alloc(MAX_FILE_BYTES) + let length = 0 + while (length < bytes.length) { + const read = await file.read(bytes, length, bytes.length - length, length) + if (read.bytesRead === 0) break + length += read.bytesRead + } + + const after = await file.stat() + if (after.size > MAX_FILE_BYTES) throw new RetireSearchHealthError('health_file_too_large') + if ( + before.size !== after.size || + before.mtimeMs !== after.mtimeMs || + before.ctimeMs !== after.ctimeMs || + length !== after.size + ) { + throw new RetireSearchHealthError('health_file_changed') + } + + try { + const text = new TextDecoder('utf-8', { fatal: true }).decode(bytes.subarray(0, length)) + const value: unknown = JSON.parse(text) + return value + } catch { + throw new RetireSearchHealthError(invalidReason) + } + } finally { + await file.close() + } + } catch (error) { + if (error instanceof RetireSearchHealthError) throw error + throw new RetireSearchHealthError('health_file_unreadable') + } +} + +/** Reads and validates the operator's bounded policy file without exposing its contents in errors. */ +export async function readSearchRetirementHealthLimits( + path: string +): Promise { + const limits = await readBoundedJson(path, 'invalid_limits') + assertLimits(limits) + return limits +} + +/** Reads a fresh external health sample before one bounded maintenance page; unknown health refuses work. */ +export async function readSearchRetirementHealth( + path: string, + limits: SearchRetirementHealthLimits, + options: { cutover?: boolean } = {} +): Promise { + assertLimits(limits) + const sample = await readBoundedJson(path, 'invalid_sample') + assertSample(sample) + if (!/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/.test(sample.observedAt)) { + throw new RetireSearchHealthError('invalid_sample') + } + const observedAt = Date.parse(sample.observedAt) + const canonicalTimestamp = sample.observedAt.includes('.') + ? sample.observedAt + : sample.observedAt.replace('Z', '.000Z') + if (!Number.isFinite(observedAt) || new Date(observedAt).toISOString() !== canonicalTimestamp) { + throw new RetireSearchHealthError('invalid_sample') + } + if (sample.databaseId !== limits.databaseId) + throw new RetireSearchHealthError('identity_mismatch') + const age = Date.now() - observedAt + if (age < -MAX_FUTURE_SKEW_MS) throw new RetireSearchHealthError('future_sample') + if (age > limits.maxSampleAgeMs) throw new RetireSearchHealthError('stale_sample') + if (!sample.healthy) throw new RetireSearchHealthError('unhealthy') + if (!sample.maintenanceAllowed) throw new RetireSearchHealthError('maintenance_refused') + if (options.cutover && !sample.cutoverAllowed) + throw new RetireSearchHealthError('cutover_refused') + if (sample.replicaLagBytes > limits.maxReplicaLagBytes) { + throw new RetireSearchHealthError('replica_lag_bytes') + } + if (sample.replicaLagSeconds > limits.maxReplicaLagSeconds) { + throw new RetireSearchHealthError('replica_lag_seconds') + } + if (sample.walBytesPerSecond > limits.maxWalBytesPerSecond) { + throw new RetireSearchHealthError('wal_rate') + } + if (sample.databaseP95Ms > limits.maxDatabaseP95Ms) { + throw new RetireSearchHealthError('database_latency') + } + if (sample.cpuPercent > limits.maxCpuPercent) throw new RetireSearchHealthError('cpu_usage') + if (sample.freeStorageBytes < limits.minFreeStorageBytes) { + throw new RetireSearchHealthError('storage_headroom') + } + return observedAt +} diff --git a/packages/db/maintenance/search-retirement.integration.ts b/packages/db/maintenance/search-retirement.integration.ts new file mode 100644 index 00000000000..64a061da8d1 --- /dev/null +++ b/packages/db/maintenance/search-retirement.integration.ts @@ -0,0 +1,656 @@ +import { spawnSync } from 'node:child_process' +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { runKnowledgeProjection } from '@sim/db/knowledge-projection' +import { + abortSearchRetirement, + advanceSearchRetirement, + beginSearchRetirementPurge, + cutoverSearchRetirement, + finalizeSearchRetirement, + getSearchRetirementStatus, + initializeSearchRetirement, +} from '@sim/db/maintenance/search-retirement' +import { readTestDatabaseUrl } from '@sim/db/testing/test-infrastructure' +import { generateId } from '@sim/utils/id' +import postgres, { type Sql } from 'postgres' +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +/** + * Real PostgreSQL proof: interrupted copy, late writes behind the cursor, deferred vectors, + * width preservation, generation races, marker invalidation, NOWAIT cutover, cached writers, + * destructive phase gates, and ordinary-KB isolation. The integration runner writes its JSON + * artifact to INTEGRATION_REPORT_PATH (CI uploads test-results/integration.json by default). + */ +describe('operator-driven Search retirement in PostgreSQL', () => { + const databaseUrl = readTestDatabaseUrl() + let admin: Sql + let sql: Sql + let writer: Sql + let database: string + const template = `sim_test_retirement_template_${generateId().replaceAll('-', '')}` + + beforeAll(async () => { + const source = postgres(databaseUrl, { max: 1, onnotice: () => undefined }) + let migrated: boolean + try { + const [row] = await source<{ migrated: boolean }[]>` + SELECT to_regclass('drizzle.__drizzle_migrations') IS NOT NULL AS migrated` + migrated = row.migrated + } finally { + await source.end() + } + const controlUrl = new URL(databaseUrl) + controlUrl.pathname = '/postgres' + const control = postgres(controlUrl.toString(), { max: 1, onnotice: () => undefined }) + try { + await control.unsafe(`CREATE DATABASE "${template}" TEMPLATE template0`) + } finally { + await control.end() + } + const url = new URL(databaseUrl) + url.pathname = `/${template}` + const setup = postgres(url.toString(), { max: 1, onnotice: () => undefined }) + try { + for (const extension of ['vector', 'btree_gin', 'pg_trgm']) { + await setup`CREATE EXTENSION IF NOT EXISTS ${setup(extension)}` + } + } finally { + await setup.end() + } + // A private template avoids cloning the shared fixture while other suites hold connections. + const provision = spawnSync( + 'bun', + ['--no-env-file', `./scripts/${migrated ? 'migrate' : 'push'}.ts`], + { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { + ...process.env, + DATABASE_URL: url.toString(), + MIGRATION_DATABASE_URL: url.toString(), + }, + encoding: 'utf8', + timeout: 90_000, + maxBuffer: 8 * 1_024 * 1_024, + } + ) + expect(provision.status, provision.stdout + provision.stderr).toBe(0) + }, 120_000) + + afterAll(async () => { + const url = new URL(databaseUrl) + url.pathname = '/postgres' + const control = postgres(url.toString(), { max: 1, onnotice: () => undefined }) + try { + await control.unsafe(`DROP DATABASE IF EXISTS "${template}"`) + } finally { + await control.end() + } + }) + + beforeEach(async () => { + database = `sim_test_retirement_${generateId().replaceAll('-', '')}` + const controlUrl = new URL(databaseUrl) + controlUrl.pathname = '/postgres' + admin = postgres(controlUrl.toString(), { max: 1, onnotice: () => undefined }) + await admin.unsafe(`CREATE DATABASE "${database}" TEMPLATE "${template.replaceAll('"', '""')}"`) + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const options = { max: 1, onnotice: () => undefined } + sql = postgres(fixtureUrl.toString(), { ...options, prepare: false }) + writer = postgres(fixtureUrl.toString(), { ...options, prepare: true }) + await sql`INSERT INTO "user" (id, name, email, email_verified, created_at, updated_at) + VALUES ('reader', 'Synthetic reader', 'reader@example.test', true, now(), now())` + await sql`INSERT INTO workspace (id, name, owner_id, billed_account_user_id) + VALUES ('workspace', 'Synthetic workspace', 'reader', 'reader')` + await sql`CREATE TABLE search_embedding_cleanup_progress ( + id integer PRIMARY KEY, knowledge_base_id text, phase text, after_id text)` + await sql`INSERT INTO search_embedding_cleanup_progress VALUES (1, 'search', 'embeddings', '')` + await sql`CREATE TABLE search_embedding_cleanup_targets (knowledge_base_id text PRIMARY KEY)` + await sql`INSERT INTO search_embedding_cleanup_targets VALUES ('search')` + const bases = [ + { id: 'prefix', width: 1536, model: 'text-embedding-3-small' }, + ...[384, 768, 1024, 1536, 3072].map((width) => ({ + id: `full-${width}`, + width, + model: 'full-width-fixture', + })), + { id: 'search', width: 1536, model: 'text-embedding-3-small' }, + ] + for (const base of bases) { + await sql`INSERT INTO knowledge_base (id, user_id, workspace_id, name, embedding_model, embedding_dimension, is_search_index) + VALUES (${base.id}, 'reader', 'workspace', ${base.id}, ${base.model}, ${base.width}, ${base.id === 'search'})` + await sql`INSERT INTO document (id, knowledge_base_id, filename, file_url, file_size, mime_type, + processing_queue_token, acl) + VALUES (${`${base.id}-doc`}, ${base.id}, 'fixture.txt', 'fixture', 10, 'text/plain', 'dispatch', ARRAY['u:reader@example.test'])` + const column = base.width === 1536 ? 'embedding' : `embedding_${base.width}` + await sql.unsafe( + `INSERT INTO embedding + (id, knowledge_base_id, document_id, chunk_index, chunk_hash, content, content_length, token_count, start_offset, end_offset, ${column}) + SELECT $1 || '-' || n, $1, $1 || '-doc', n, 'hash-' || n, 'Synthetic content', 17, 4, 0, 17, + array_fill(0.01::real, ARRAY[${base.width}])::vector(${base.width}) + FROM generate_series(1, 4) n`, + [base.id] + ) + } + // An older deferred writer can leave a canonical chunk without a projected vector. + await sql`DELETE FROM embedding_search WHERE id = 'prefix-4'` + }, 60_000) + + afterEach(async () => { + await writer?.end() + await sql?.end() + if (admin) { + await admin.unsafe(`DROP DATABASE IF EXISTS "${database}"`) + await admin.end() + } + }) + + async function nextPage(pageSize = 5) { + return advanceSearchRetirement(sql, { pageSize }) + } + + async function finishCopy() { + for (let page = 0; page < 100; page++) { + const state = await nextPage() + if (state.phase === 'ready') return state + } + throw new Error('Fixture retirement did not reach the cutover gate') + } + + it('resumes copy, includes late writes, replaces all widths, and only then purges Search chunks', async () => { + expect(await getSearchRetirementStatus(sql)).toBeNull() + await initializeSearchRetirement(sql) + await nextPage() + await nextPage() + await writer`UPDATE embedding SET enabled = false WHERE id = 'full-1024-1'` + await writer`DELETE FROM embedding WHERE id = 'full-384-1'` + await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, + content, content_length, token_count, start_offset, end_offset, embedding) + VALUES ('000-late', 'prefix', 'prefix-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, + array_fill(0.02::real, ARRAY[1536])::vector(1536))` + await initializeSearchRetirement(sql) + await finishCopy() + expect( + (await sql`SELECT count(*)::int AS n FROM embedding WHERE knowledge_base_id = 'search'`)[0].n + ).toBe(4) + await expect(beginSearchRetirementPurge(sql)).rejects.toThrow() + // Warm a prepared reader and trigger writer before replacing the relation. + const preparedRead = () => + writer`SELECT id, enabled FROM embedding_search WHERE id = 'prefix-1'` + await preparedRead() + await writer`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + await finishCopy() + await cutoverSearchRetirement(sql) + expect(await preparedRead()).toEqual([{ id: 'prefix-1', enabled: false }]) + await writer`UPDATE embedding SET enabled = true WHERE id = 'prefix-1'` + expect(await preparedRead()).toEqual([{ id: 'prefix-1', enabled: true }]) + expect((await sql`SELECT count(*)::int AS n FROM embedding_search`)[0].n).toBe(24) + expect( + await sql`SELECT id FROM embedding_search WHERE knowledge_base_id = 'search'` + ).toHaveLength(0) + expect(await sql`SELECT id FROM embedding_search WHERE id = 'prefix-4'`).toHaveLength(1) + expect(await sql`SELECT id FROM embedding_search WHERE id = 'full-384-1'`).toHaveLength(0) + expect( + ( + await sql`SELECT vector_dims(vector_512) AS width FROM embedding_search WHERE id = '000-late'` + )[0].width + ).toBe(512) + for (const width of [384, 768, 1024, 1536, 3072]) { + const column = width === 1536 ? 'vector' : `vector_${width}` + expect( + ( + await sql.unsafe( + `SELECT vector_dims(${column}) AS width FROM embedding_search WHERE id = $1`, + [`full-${width}-2`] + ) + )[0].width + ).toBe(width) + } + await expect(abortSearchRetirement(sql)).rejects.toThrow() + await beginSearchRetirementPurge(sql) + for (let page = 0; page < 100; page++) { + const state = await nextPage() + if (state.phase === 'finalize') break + } + expect((await getSearchRetirementStatus(sql))?.phase).toBe('finalize') + await finalizeSearchRetirement(sql) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('done') + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(24) + expect((await sql`SELECT count(*)::int AS n FROM knowledge_base`)[0].n).toBe(7) + expect( + ( + await sql`SELECT enabled, user_excluded, processing_queue_token FROM document WHERE id = 'search-doc'` + )[0] + ).toEqual({ enabled: false, user_excluded: true, processing_queue_token: null }) + expect( + ( + await sql`SELECT enabled, user_excluded, processing_queue_token, acl FROM document WHERE id = 'prefix-doc'` + )[0] + ).toEqual({ + enabled: true, + user_excluded: false, + processing_queue_token: 'dispatch', + acl: ['u:reader@example.test'], + }) + await writer`DELETE FROM embedding WHERE id = '000-late'` + expect(await sql`SELECT id FROM embedding_search WHERE id = '000-late'`).toHaveLength(0) + }) + + it('refuses cutover immediately while an ordinary reader holds the active projection', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + const [{ oid }] = await sql`SELECT 'embedding_search'::regclass::oid AS oid` + await writer`BEGIN` + try { + await writer`SELECT id FROM embedding_search LIMIT 1` + await expect(cutoverSearchRetirement(sql)).rejects.toMatchObject({ code: '55P03' }) + expect((await sql`SELECT 'embedding_search'::regclass::oid AS oid`)[0].oid).toBe(oid) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + } finally { + await writer`ROLLBACK` + } + await cutoverSearchRetirement(sql) + expect((await sql`SELECT 'embedding_search'::regclass::oid AS oid`)[0].oid).not.toBe(oid) + }) + + it('refuses incomplete copy and invalidates the job when a KB changes classification', async () => { + await initializeSearchRetirement(sql) + await expect(cutoverSearchRetirement(sql)).rejects.toThrow() + await writer`UPDATE knowledge_base SET is_search_index = false WHERE id = 'search'` + await expect(nextPage()).rejects.toThrow() + expect((await getSearchRetirementStatus(sql))?.invalidated).toBe(true) + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + await abortSearchRetirement(sql) + expect(await getSearchRetirementStatus(sql)).toBeNull() + await writer`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + false + ) + }) + + it('rolls back a failed page with its cursor and retries without losing or duplicating rows', async () => { + await initializeSearchRetirement(sql) + for (let n = 0; n < 20; n++) { + if ((await getSearchRetirementStatus(sql))?.phase === 'copy') break + await nextPage() + } + const before = await getSearchRetirementStatus(sql) + await writer`BEGIN` + try { + await writer`LOCK TABLE embedding IN ACCESS EXCLUSIVE MODE` + await expect(nextPage()).rejects.toMatchObject({ code: '55P03' }) + } finally { + await writer`ROLLBACK` + } + expect(await getSearchRetirementStatus(sql)).toEqual(before) + await finishCopy() + await cutoverSearchRetirement(sql) + expect((await sql`SELECT count(*)::int AS n FROM embedding_search`)[0].n).toBe(24) + }) + + it('refuses unknown inbound dependencies instead of silently redirecting only some readers', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await sql`CREATE VIEW external_projection_reader AS SELECT id FROM embedding_search` + await expect(cutoverSearchRetirement(sql)).rejects.toThrow() + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + await sql`DROP VIEW external_projection_reader` + await cutoverSearchRetirement(sql) + }) + + it('fences the old cursor-based command while replacement work exists', async () => { + await initializeSearchRetirement(sql) + await expect( + sql`UPDATE search_embedding_cleanup_progress SET after_id = 'late' WHERE id = 1` + ).rejects.toThrow() + expect( + (await sql`SELECT after_id FROM search_embedding_cleanup_progress WHERE id = 1`)[0].after_id + ).toBe('') + await abortSearchRetirement(sql) + await sql`UPDATE search_embedding_cleanup_progress SET after_id = 'late' WHERE id = 1` + }) + + it('refuses a repeatable-read snapshot taken before copy without any projection lock', async () => { + await writer`BEGIN ISOLATION LEVEL REPEATABLE READ` + try { + await writer`SELECT id FROM workspace LIMIT 1` + await initializeSearchRetirement(sql) + await finishCopy() + await expect(cutoverSearchRetirement(sql)).rejects.toThrow(/Transactions predate/) + expect( + ( + await sql`SELECT count(*)::int AS n FROM embedding_search WHERE knowledge_base_id = 'search'` + )[0].n + ).toBe(4) + } finally { + await writer`ROLLBACK` + } + await cutoverSearchRetirement(sql) + expect((await sql`SELECT count(*)::int AS n FROM embedding_search`)[0].n).toBe(24) + }) + + it('keeps deferred ordinary-vector repair working after cutover and metadata changes', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await cutoverSearchRetirement(sql) + await writer.begin(async (tx) => { + await tx`SET LOCAL sim.projection_mode = 'async'` + await tx`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + }) + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + true + ) + await runKnowledgeProjection(writer, { budgetMs: 5_000, pageSize: 2 }) + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + false + ) + await writer`UPDATE knowledge_base SET embedding_model = 'updated-full-width-fixture' WHERE id = 'full-1536'` + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await getSearchRetirementStatus(sql))?.phase).toBe('finalize') + await finalizeSearchRetirement(sql) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('done') + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(24) + }) + + it('retains a newer dirty generation written while its older image is being copied', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await writer`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + const barrierUrl = new URL(databaseUrl) + barrierUrl.pathname = `/${database}` + const barrier = postgres(barrierUrl.toString(), { max: 1, onnotice: () => undefined }) + await sql`CREATE FUNCTION retirement_test_barrier() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + PERFORM set_config('lock_timeout', '1500ms', true); + PERFORM pg_advisory_xact_lock(791184); + RETURN NEW; + END $$` + await sql`CREATE TRIGGER retirement_test_barrier BEFORE INSERT OR UPDATE ON embedding_search_retirement_shadow + FOR EACH ROW EXECUTE FUNCTION retirement_test_barrier()` + await barrier`SELECT pg_advisory_lock(791184)` + const [{ pid }] = await sql`SELECT pg_backend_pid() AS pid` + const copy = Promise.allSettled([nextPage()]) + try { + await vi.waitFor( + async () => { + const [{ waiting }] = await barrier`SELECT EXISTS ( + SELECT 1 FROM pg_locks WHERE pid = ${pid} AND locktype = 'advisory' AND NOT granted + ) AS waiting` + expect(waiting).toBe(true) + }, + { interval: 5, timeout: 1_000 } + ) + await writer`UPDATE embedding SET enabled = true WHERE id = 'prefix-1'` + await barrier`SELECT pg_advisory_unlock(791184)` + const [result] = await copy + if (result.status === 'rejected') throw result.reason + expect( + await sql`SELECT embedding_id FROM search_retirement_changes WHERE embedding_id = 'prefix-1'` + ).toHaveLength(1) + expect( + (await sql`SELECT enabled FROM embedding_search_retirement_shadow WHERE id = 'prefix-1'`)[0] + .enabled + ).toBe(false) + } finally { + await barrier`SELECT pg_advisory_unlock_all()` + await copy + await barrier.end() + await sql`DROP TRIGGER retirement_test_barrier ON embedding_search_retirement_shadow` + await sql`DROP FUNCTION retirement_test_barrier()` + } + await finishCopy() + await cutoverSearchRetirement(sql) + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + true + ) + }) + + it('catches a late Search write behind the purge cursor before completing document retirement', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await cutoverSearchRetirement(sql) + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'documents') break + } + await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, + content, content_length, token_count, start_offset, end_offset, embedding) + VALUES ('000-late-search', 'search', 'search-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, + array_fill(0.02::real, ARRAY[1536])::vector(1536))` + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await getSearchRetirementStatus(sql))?.phase).toBe('finalize') + await finalizeSearchRetirement(sql) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('done') + expect(await sql`SELECT id FROM embedding WHERE knowledge_base_id = 'search'`).toHaveLength(0) + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(24) + expect( + await sql`SELECT tgname FROM pg_trigger WHERE tgrelid = 'embedding'::regclass + AND tgname = 'search_retirement_capture'` + ).toHaveLength(0) + }) + + it.each(['abort', 'begin-purge', 'finalize'] as const)( + 'rolls back %s if its implicit DROP lock conflicts with a reader', + async (command) => { + await initializeSearchRetirement(sql) + if (command !== 'abort') { + await finishCopy() + await cutoverSearchRetirement(sql) + } + if (command === 'finalize') { + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + } + const before = await getSearchRetirementStatus(sql) + const operation = + command === 'abort' + ? abortSearchRetirement + : command === 'begin-purge' + ? beginSearchRetirementPurge + : finalizeSearchRetirement + await writer`BEGIN` + try { + await writer`SELECT id FROM embedding LIMIT 1` + await expect(operation(sql)).rejects.toMatchObject({ code: '55P03' }) + expect(await getSearchRetirementStatus(sql)).toEqual(before) + expect(await writer`SELECT id FROM embedding_search WHERE id = 'prefix-1'`).toHaveLength(1) + } finally { + await writer`ROLLBACK` + } + await operation(sql) + expect((await getSearchRetirementStatus(sql))?.phase ?? null).toBe( + command === 'abort' ? null : command === 'begin-purge' ? 'purge' : 'done' + ) + } + ) + + it('returns from finalization to bounded purge when a late Search write arrives', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await cutoverSearchRetirement(sql) + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await nextPage()).phase).toBe('finalize') + expect( + await sql`SELECT tgname FROM pg_trigger WHERE tgrelid = 'embedding'::regclass + AND tgname = 'search_retirement_capture'` + ).toHaveLength(1) + await writer`UPDATE document SET enabled = true, user_excluded = false WHERE id = 'search-doc'` + await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, + content, content_length, token_count, start_offset, end_offset, embedding) + VALUES ('000-final-late-search', 'search', 'search-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, + array_fill(0.02::real, ARRAY[1536])::vector(1536))` + expect((await finalizeSearchRetirement(sql)).phase).toBe('purge') + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await finalizeSearchRetirement(sql)).phase).toBe('done') + expect(await sql`SELECT id FROM embedding WHERE knowledge_base_id = 'search'`).toHaveLength(0) + expect( + (await sql`SELECT enabled, user_excluded FROM document WHERE id = 'search-doc'`)[0] + ).toEqual({ enabled: false, user_excluded: true }) + }) + + it('refuses a stored SQL function that would retain the old projection identity', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await sql`CREATE FUNCTION retirement_dependent_function() RETURNS bigint + LANGUAGE SQL RETURN (SELECT count(*) FROM embedding_search)` + await expect(cutoverSearchRetirement(sql)).rejects.toThrow(/dependencies/) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + expect((await sql`SELECT retirement_dependent_function() AS n`)[0].n).toBe('27') + }) + + it('rejects a replacement index changed after preparation', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await sql`DROP INDEX embedding_search_retirement_shadow_512_hnsw_idx` + await sql`CREATE INDEX embedding_search_retirement_shadow_512_hnsw_idx + ON embedding_search_retirement_shadow (enabled)` + await expect(cutoverSearchRetirement(sql)).rejects.toThrow() + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + expect( + ( + await sql`SELECT count(*)::int AS n FROM embedding_search WHERE knowledge_base_id = 'search'` + )[0].n + ).toBe(4) + }) + + it('refuses a disabled canonical writer before creating maintenance objects', async () => { + await sql`ALTER TABLE embedding DISABLE TRIGGER embedding_search_sync` + await expect(initializeSearchRetirement(sql)).rejects.toThrow() + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + }) + + it('serializes with another maintenance operator without advancing its checkpoint', async () => { + await initializeSearchRetirement(sql) + const before = await getSearchRetirementStatus(sql) + await writer`SELECT pg_advisory_lock(hashtextextended('sim:search-retirement-maintenance', 0))` + try { + await expect(nextPage()).rejects.toThrow(/Another migration or retirement operation/) + } finally { + await writer`SELECT pg_advisory_unlock_all()` + } + expect(await getSearchRetirementStatus(sql)).toEqual(before) + }) + + it('refuses db:push before schema reconciliation once replacement retirement exists', async () => { + await initializeSearchRetirement(sql) + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const script = fileURLToPath(new URL('../scripts/push.ts', import.meta.url)) + const result = spawnSync('bun', ['--no-env-file', script, '--retirement-test-invalid-option'], { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { ...process.env, NODE_ENV: 'development', DATABASE_URL: fixtureUrl.toString() }, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + }) + expect(result.status).toBe(1) + expect(`${result.stdout}${result.stderr}`).toContain( + 'Schema push is disabled after Search retirement starts' + ) + expect(`${result.stdout}${result.stderr}`).not.toContain('Unrecognized options') + expect((await getSearchRetirementStatus(sql))?.phase).toBe('snapshot') + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + }) + + it('the CLI refuses missing or unhealthy telemetry before initializing and status stays read-only', async () => { + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const script = fileURLToPath(new URL('../scripts/retire-indexed-search.ts', import.meta.url)) + const environment = { + ...process.env, + NODE_ENV: 'development', + MIGRATION_DATABASE_URL: fixtureUrl.toString(), + } + const invoke = (...args: string[]) => + spawnSync('bun', ['--no-env-file', script, ...args], { + env: environment, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + }) + expect(invoke('status').status).toBe(0) + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect(invoke('prepare', '--ack-release-drained').status).toBe(1) + expect(await getSearchRetirementStatus(sql)).toBeNull() + const identity = invoke('identity') + expect(identity.status).toBe(0) + const databaseId = identity.stdout.trim() + const directory = await mkdtemp(join(tmpdir(), 'retirement-cli-')) + try { + const policy = join(directory, 'policy.json') + const health = join(directory, 'health.json') + await writeFile( + policy, + JSON.stringify({ + databaseId, + maxReplicaLagBytes: 1, + maxReplicaLagSeconds: 1, + maxWalBytesPerSecond: 1_000, + maxDatabaseP95Ms: 100, + maxCpuPercent: 50, + minFreeStorageBytes: 100, + maxSampleAgeMs: 30_000, + }) + ) + await writeFile( + health, + JSON.stringify({ + databaseId, + observedAt: new Date().toISOString(), + healthy: false, + maintenanceAllowed: true, + cutoverAllowed: false, + replicaLagBytes: 0, + replicaLagSeconds: 0, + walBytesPerSecond: 0, + databaseP95Ms: 1, + cpuPercent: 1, + freeStorageBytes: 1_000, + }) + ) + const refused = invoke( + 'prepare', + '--ack-release-drained', + '--health-file', + health, + '--health-policy', + policy + ) + expect(refused.status).toBe(1) + expect(`${refused.stdout}${refused.stderr}`).toContain('unhealthy') + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect(invoke('status').status).toBe(0) + } finally { + await rm(directory, { recursive: true, force: true }) + } + const oldScript = fileURLToPath( + new URL('../script-migrations/0027_retire_search_embeddings.ts', import.meta.url) + ) + expect( + spawnSync('bun', ['--no-env-file', oldScript, '--maintenance'], { + env: environment, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + }).status + ).toBe(1) + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + }) +}) diff --git a/packages/db/maintenance/search-retirement.md b/packages/db/maintenance/search-retirement.md new file mode 100644 index 00000000000..9a21e098823 --- /dev/null +++ b/packages/db/maintenance/search-retirement.md @@ -0,0 +1,194 @@ +# Retire indexed Search data without a bulk rebuild + +This is an **operator-run workflow**, outside `db:migrate`, `db:push`, deployment jobs, cron, and +Trigger.dev. Merging or deploying the code starts no copy, deletion, vacuum, or reindex. Deploy the +indexed-Search code removal first, verify every app/worker uses live Search, and drain old indexing, +projection-maintenance, and cleanup commands. Do not roll back to an indexed-Search release during +or after retirement. + +The command replaces only `embedding_search`. It reconstructs ordinary-KB candidate vectors from +canonical `embedding` rows, including rows missing from the old deferred projection. Full-precision +vectors, content, keyword search, document ACLs, provenance and table identities used by ordinary KBs +remain in place. Search KB shells and source/credential configuration remain because live Search +uses them. Search chunks are deleted only after replacement cutover and a separate approval to +retire the old projection. Search documents are excluded/disabled and dispatch tokens cleared; +documents/files are not hard-deleted. Their eventual deletion must use the storage-outbox/accounting +path, not raw SQL. + +This reduces risk; it cannot guarantee zero latency impact. Shadow inserts still build HNSW links, +consume CPU/I/O, and emit replicated WAL. Change capture adds a small write to each affected chunk +transaction while copying. Relation DDL can briefly exclude conflicting operations, and replayed DDL +can conflict with replica queries. Health sampling cannot react to a spike that begins inside a page. +Start with a canary page and prioritize serving traffic over migration speed. + +## Release and operator sequence + +1. **Deploy the code-removal PR first.** Deploy this stacked PR only after its parent. The command + requires an explicit `--ack-release-drained` for preparation and cutover. It cannot prove which + application binaries or external jobs remain running; verify that outside PostgreSQL. +2. **Rehearse on a disposable database**, using the current schema and realistic retained-vector + widths. Run the integration suite, then ordinary-KB retrieval and live Search smoke tests against + the staged application. Configure a dedicated maintenance role and primary direct/session-pooled + connection. PostgreSQL 17+ is required for the transaction deadline. Never use transaction pooling. +3. **Configure health collection and disk/WAL headroom.** Feed fresh primary, every serving replica, + and application SLO observations into the health file described below. Establish normal baselines; + choose numeric limits from those baselines and provision space for old and new projections plus + WAL. Merely having space for the final table is insufficient. Do not refresh a stale metric's + timestamp or manufacture healthy samples. No valid fresh sample means no maintenance. +4. **Prepare and canary.** `prepare` creates a logged empty replacement with the six shared HNSW + indexes already present, installs capture, and creates durable state. `run` advances one small page + by default. Inspect application latency, query errors, CPU/I/O, WAL and every replica after the + first pages. Leave the health collector's maintenance switch off when attention or headroom is + unavailable. Increase only the number of scheduled bounded invocations once impact is acceptable. +5. **Copy, catch up and verify.** Continue `run` until `ready`. Each page commits with its cursor; + retries resume committed work. Validation compares retained identities, enabled state, binary + compatibility and vector values in both directions. Concurrent canonical writes enqueue IDs with + generations; catchup acknowledges only the generation it processed. Metadata changes during + construction invalidate the copy instead of silently changing its scope. Abort and prepare again + after investigating. Do not change the schema or run other index maintenance during this workflow. +6. **Cut over explicitly.** First verify replicas have replayed the copy and drain old snapshots on + primary and replicas, or divert replica reads and wait for existing transactions to drain. The + collector's separate `cutoverAllowed` must attest this. Run a fresh bounded catchup page if changes + remain, then `cutover`. It checks relation ownership/privileges, known writer triggers, dependencies + and indexes, and uses `LOCK ... NOWAIT`. It refuses busy tables or old primary transactions rather + than cancelling traffic or waiting in a DDL lock queue. Refusal is expected on busy systems: stop, + inspect, and retry in a suitable window. Do not automatically loop cutover attempts. +7. **Observe before allowing deletion.** Verify ordinary-KB semantic/keyword retrieval, document ACL + denials, connector ingestion/deferred repair, live Search, and error/replica metrics. The retained + old projection is an observation backup, **not an instant rollback**: new writes target the new + table. Never swap a stale backup back into service. A rollback requires a separately validated + reconstruction from canonical embeddings, which remain intact at this point. +8. **Retire the backup, then purge slowly.** `begin-purge --ack-retire-backup` requires separate + cutover clearance. It drops the old projection with `RESTRICT` before canonical chunk deletion, + so its outgoing FK cannot maintain the old HNSW graph on each delete. Continue bounded `run` + invocations until `finalize`. Every destructive page locks/rechecks Search markers. Ordinary rows + and live configuration are preserved. Existing keyword/provenance FKs still perform bounded + per-chunk cleanup. These indexes have a cost; leave telemetry gates enabled throughout. +9. **Finalize explicitly.** Clear another primary/replica DDL window and run `finalize`. It removes + the capture triggers only after checking for late writes. If it returns `purge`, resume bounded + pages and revisit this gate. Backup removal, abort and finalization may require implicit PostgreSQL + lock upgrades: those waits are capped at 1 ms, but a brief lock queue is still possible. +10. **Verify completion.** Inspect the durable status, smoke-test retrieval again, and observe normal + autovacuum/replica recovery. The fresh vector projection needs no final rebuild. Ordinary vacuum + can reuse dead canonical/keyword heap space; it does not promise to shrink files. This command + does not run `REINDEX`, `VACUUM FULL`, or an unbounded final absence scan. + +## Commands + +Supply the migration writer DSN through `MIGRATION_DATABASE_URL` in the process environment. There +is no fallback to the app DSN. The command uses one connection with +`application_name=sim-search-data-retirement`; configure a provider Traffic Control budget for that +application name when available. Keep credentials, provider identifiers, policy and health files +outside the repository. + +```sh +bun --no-env-file packages/db/scripts/retire-indexed-search.ts --help +bun --no-env-file packages/db/scripts/retire-indexed-search.ts identity +bun --no-env-file packages/db/scripts/retire-indexed-search.ts status + +bun --no-env-file packages/db/scripts/retire-indexed-search.ts prepare \ + --ack-release-drained --health-file /secure/health.json --health-policy /secure/policy.json + +# One page by default. More pages remain bounded and health is checked before each one. +bun --no-env-file packages/db/scripts/retire-indexed-search.ts run \ + --health-file /secure/health.json --health-policy /secure/policy.json --pages 10 --seconds 60 + +# Separate operational decisions; never put these in an automatic run loop. +bun --no-env-file packages/db/scripts/retire-indexed-search.ts cutover \ + --ack-release-drained --health-file /secure/health.json --health-policy /secure/policy.json +bun --no-env-file packages/db/scripts/retire-indexed-search.ts begin-purge \ + --ack-retire-backup --health-file /secure/health.json --health-policy /secure/policy.json +bun --no-env-file packages/db/scripts/retire-indexed-search.ts finalize \ + --health-file /secure/health.json --health-policy /secure/policy.json + +# Before cutover only: stop capture and discard this replacement after a NOWAIT lock attempt. +bun --no-env-file packages/db/scripts/retire-indexed-search.ts abort +``` + +`status` is read-only and never initializes work. Ctrl-C or connection loss leaves committed pages +intact; an interrupted transaction rolls back its writes and cursor together. Any database error, +health refusal, or capacity throttle stops the invocation with a nonzero exit code. No automatic +retry widens a page or raises a timeout. Check status before retrying an ambiguous connection loss. + +The default page is 25 source rows; copying/validation can be explicitly lowered or raised to at +most 100. Destructive pages never exceed 25 source rows. IDs/vectors remain in PostgreSQL except a +bounded queue page of IDs/generations. The CLI runs at most 120 pages/600 seconds per invocation, +with at least five seconds and nine times the previous page duration between pages. The time budget +stops *starting* pages; the last transaction and cooldown can finish afterward. Core transactions +have short statement, lock, and transaction deadlines. There is no dynamic batch growth. A metadata +swap does no bulk data work while it holds locks. + +Pausing the worker does not stop change capture. If a pause will be long, or the dirty queue itself +causes pressure, use `abort` before cutover. Capture must never reject application writes merely +because its queue grew. Monitor its storage too. A dedicated Traffic Control budget is an additional +backstop; an external scheduler may repeat bounded `run` commands, but must never repeat cutover or +begin-purge or finalize automatically. Trigger.dev is optional orchestration, not a substitute for these guards. + +## Health contract + +The repository deliberately has **no fabricated provider-health fallback**. A trusted external +collector must atomically replace an at-most-8-KiB UTF-8 JSON file. Its observation time is the oldest +underlying required metric timestamp, not the time it wrote the file. Aggregate worst lag/CPU across +all relevant nodes and minimum free storage. Require application SLO/error checks as part of +`healthy`; a successful database ping or static `/api/health` response is insufficient. + +Required sample fields: + +| Field | Meaning | +| --- | --- | +| `observedAt` | UTC ISO timestamp, seconds or three-digit milliseconds, ending `Z` | +| `databaseId` | Output of `identity`, SHA-256 of endpoint/port/database/role, excluding password | +| `healthy` | Collector verified the expected node inventory and application SLOs | +| `maintenanceAllowed` | Operator/alert kill switch; false stops the next page | +| `cutoverAllowed` | Separate confirmation that serving replicas and old snapshots are safe for DDL; normally false | +| `replicaLagBytes`, `replicaLagSeconds` | Worst replay backlog/lag across every serving replica | +| `walBytesPerSecond` | Measured WAL generation rate over a defined recent interval | +| `databaseP95Ms`, `cpuPercent`, `freeStorageBytes` | Current workload latency, worst CPU, and minimum available storage | + +The policy file has exactly `databaseId`, `maxReplicaLagBytes`, `maxReplicaLagSeconds`, +`maxWalBytesPerSecond`, `maxDatabaseP95Ms`, `maxCpuPercent`, `minFreeStorageBytes`, and +`maxSampleAgeMs`. Supply finite numeric limits chosen for the deployment, not strings. Lag limits may +be zero; other limits must be positive, CPU at most 100, and sample age at most 30,000 milliseconds. +Missing/unknown fields, nonfinite/negative metrics, wrong identity, stale/future observations, +unhealthy/refused maintenance, or exceeded limits fail closed. There is no `--force` bypass. + +PlanetScale's metrics API exposes replica lag, retained WAL, CPU and query latency/error series, but +collector mapping and freshness must be verified against the actual deployment. Retained WAL bytes +are **not** automatically replica replay backlog or WAL generation rate. If telemetry resolution or +permissions cannot satisfy the contract, keep maintenance paused and improve collection; do not fill +missing measurements with zero. Primary transaction visibility requires sufficient statistics +permissions; cutover refuses to infer safety from a partially visible activity view. + +## Compatibility and follow-up contracts + +The replacement retains nullable legacy source/ACL and binary columns because the deployed shared +projector still writes them. It preserves the ordinary shared vector indexes and matching query +shape, omitting source-specific/ACL indexes from the replacement. The old primary table and its +unused indexes are reclaimed together when the operator ends backup retention. + +This workflow leaves legacy keyword tables, activity tables, shared dirty-queue state and historical +migration receipts in the schema. Search chunk deletes remove their referencing projection rows; +removing the empty tables, obsolete trigger installers and compatible columns is a later schema +contract after all remaining shared writers are removed. The complete inventory is in +[Indexed Search retirement inventory](../script-migrations/indexed-search-retirement.md). Never drop +canonical document ACLs or ordinary-KB content projections just because indexed Search is retired. + +The runner owns `search_retirement_*` and `embedding_search_retirement_*`; Drizzle push excludes +those objects. `db:push` refuses databases with retirement state, including completed jobs, because +its historical reconcilers would recreate the retired writers and backfills. Use reviewed versioned +migrations after starting retirement; do not delete the receipt to bypass this guard. Existing `search_embedding_cleanup_*` checkpoints remain unchanged. Its progress-row +fence prevents the old cursor command from advancing while replacement work exists. Stop old +maintenance binaries first; this cannot prevent an arbitrary operator from issuing SQL independently. + +## Verification and references + +`packages/db/maintenance/search-retirement.integration.ts` exercises the real PostgreSQL boundary. +The integration reporter writes to `INTEGRATION_REPORT_PATH` (default +`test-results/integration.json`, uploaded by CI). The health boundary has separate invalid-input and +admission tests. Rehearsal against the running app and production-shaped workload remains an operator +prerequisite, not a claim made by these small fixtures. + +- [pgvector HNSW](https://github.com/pgvector/pgvector#hnsw): empty indexes are supported; incremental building trades throughput for bounded work. +- [PostgreSQL locks](https://www.postgresql.org/docs/17/sql-lock.html) and [MVCC caveats](https://www.postgresql.org/docs/17/mvcc-caveats.html): fail-fast locking and old snapshots both matter. +- [Hot standby conflicts](https://www.postgresql.org/docs/17/hot-standby.html#HOT-STANDBY-CONFLICT): primary DDL may affect replica queries. +- [PlanetScale metrics API](https://planetscale.com/docs/api/reference/get_branch_metrics) and [Traffic Control](https://planetscale.com/docs/postgres/traffic-control/concepts): external observations and supplementary workload budgets. diff --git a/packages/db/maintenance/search-retirement.ts b/packages/db/maintenance/search-retirement.ts new file mode 100644 index 00000000000..6dac2422670 --- /dev/null +++ b/packages/db/maintenance/search-retirement.ts @@ -0,0 +1,972 @@ +import type { Sql, TransactionSql } from 'postgres' + +const STATE = 'public.search_retirement_state' +const SHADOW = 'public.embedding_search_retirement_shadow' +const BACKUP = 'public.embedding_search_retirement_backup' +const JOB_LOCK = 'sim:search-retirement-replacement' +const MAINTENANCE_LOCK = 'sim:search-retirement-maintenance' +const MIGRATION_LOCK = '4961002270' +const WIDTHS = [1536, 384, 512, 768, 1024, 3072] as const +const SOURCE_WIDTHS = [1536, 384, 768, 1024, 3072] as const +const column = (prefix: string, width: number) => (width === 1536 ? prefix : `${prefix}_${width}`) +const VECTOR_COLUMNS = WIDTHS.map((width) => column('vector', width)) +const BINARY_COLUMNS = SOURCE_WIDTHS.map((width) => column('binary', width)) +const PROJECTED_COLUMNS = [ + 'id', + 'knowledge_base_id', + 'document_id', + 'enabled', + ...BINARY_COLUMNS, + ...VECTOR_COLUMNS, +] +const SHARED_INDEXES = [ + { name: 'embedding_search_pkey', suffix: 'pkey', definition: 'UNIQUE (id)' }, + { name: 'embedding_search_kb_idx', suffix: 'kb_idx', definition: '(knowledge_base_id)' }, + { + name: 'embedding_search_document_lookup_idx', + suffix: 'document_lookup_idx', + definition: '(document_id, knowledge_base_id, id) WHERE enabled', + }, + ...WIDTHS.map((width) => ({ + name: + width === 1536 + ? 'embedding_search_cosine_hnsw_idx' + : `embedding_search_${width}_cosine_hnsw_idx`, + suffix: `${width}_hnsw_idx`, + definition: `USING hnsw (${column('vector', width)} halfvec_cosine_ops) WITH (m = 16, ef_construction = 64)`, + })), +] as const + +export type SearchRetirementPhase = + | 'snapshot' + | 'copy' + | 'catchup' + | 'validate-source' + | 'validate-shadow' + | 'ready' + | 'cutover' + | 'purge' + | 'documents' + | 'finalize' + | 'done' + +/** Deliberate operator messages never include database values, IDs, or statement parameters. */ +export class SearchRetirementError extends Error { + override name = 'SearchRetirementError' +} + +export interface SearchRetirementStatus { + phase: SearchRetirementPhase + invalidated: boolean + invalidationReason: string | null + sourceScanned: number + copied: number + reconciled: number + validatedSource: number + validatedShadow: number + purgedEmbeddings: number + retiredDocuments: number + pendingChanges: boolean + changeBacklogAtLimit: boolean + backupRetained: boolean +} + +interface StateRow { + version: number + phase: SearchRetirementPhase + resume_phase: 'validate-source' | 'ready' + after_id: string + invalidated: boolean + invalidation_reason: string | null + original_oid: number + replacement_oid: number + source_scanned: string + copied: string + reconciled: string + validated_source: string + validated_shadow: string + purged_embeddings: string + retired_documents: string + round_mutations: string + ready_at: Date | null +} + +interface PageResult { + after_id: string | null + scanned: number + changed: number +} + +const identifier = (value: string) => `"${value.replaceAll('"', '""')}"` +const columns = PROJECTED_COLUMNS.map(identifier).join(', ') +const assignments = PROJECTED_COLUMNS.filter((name) => name !== 'id') + .map((name) => `${identifier(name)} = EXCLUDED.${identifier(name)}`) + .join(', ') +const comparison = (left: string, right: string) => + `(${PROJECTED_COLUMNS.map((name) => `${left}.${identifier(name)}`).join(', ')}) IS DISTINCT FROM (${PROJECTED_COLUMNS.map((name) => `${right}.${identifier(name)}`).join(', ')})` + +const REPLACEMENT_SHAPE = `SELECT jsonb_build_object( + 'columns', (SELECT jsonb_agg(to_jsonb(a) ORDER BY a.attnum) FROM ( + SELECT a.attnum, a.attname, a.atttypid, a.atttypmod, a.attnotnull, a.attidentity, + a.attgenerated, a.attcollation, a.attacl, pg_get_expr(d.adbin, d.adrelid) AS default_value + FROM pg_attribute a LEFT JOIN pg_attrdef d ON d.adrelid = a.attrelid AND d.adnum = a.attnum + WHERE a.attrelid = c.oid AND a.attnum > 0 AND NOT a.attisdropped ORDER BY a.attnum LIMIT 32 + ) a), + 'constraints', (SELECT jsonb_agg(to_jsonb(k) ORDER BY k.conname) FROM ( + SELECT conname, contype, convalidated, condeferrable, condeferred, pg_get_constraintdef(oid) AS definition + FROM pg_constraint WHERE conrelid = c.oid ORDER BY conname LIMIT 32 + ) k), + 'table', jsonb_build_array(c.relkind, c.relpersistence, c.relrowsecurity, + c.relforcerowsecurity, c.relreplident, c.reloptions, c.reltablespace), + 'unexpected_dependencies', EXISTS (SELECT 1 FROM pg_constraint WHERE confrelid = c.oid) + OR EXISTS (SELECT 1 FROM pg_depend WHERE refobjid = c.oid + AND refclassid = 'pg_class'::regclass + AND classid IN ('pg_rewrite'::regclass, 'pg_proc'::regclass, 'pg_policy'::regclass)) + OR EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = c.oid OR inhparent = c.oid) + OR EXISTS (SELECT 1 FROM pg_policy WHERE polrelid = c.oid) + OR EXISTS (SELECT 1 FROM pg_trigger WHERE tgrelid = c.oid AND NOT tgisinternal) + OR EXISTS (SELECT 1 FROM pg_publication_tables WHERE schemaname = 'public' + AND tablename = 'embedding_search_retirement_shadow') +) FROM pg_class c WHERE c.oid = 'public.embedding_search_retirement_shadow'::regclass` + +/** Matches the current synchronous writer, including OpenAI's 512-dimensional prefix. */ +function projectedValues(source: string, model: string): string { + const shortened = `${model} IN ('text-embedding-3-small', 'text-embedding-3-large') AND ${source}.embedding_384 IS NULL` + return [ + `${source}.id`, + `${source}.knowledge_base_id`, + `${source}.document_id`, + `${source}.enabled`, + ...SOURCE_WIDTHS.map( + (width) => + `binary_quantize(${source}.${column('embedding', width)})::bit(${width}) AS ${identifier(column('binary', width))}` + ), + ...WIDTHS.map((width) => + width === 512 + ? `CASE WHEN ${shortened} THEN subvector(coalesce(${SOURCE_WIDTHS.map((size) => `${source}.${column('embedding', size)}`).join(', ')}), 1, 512)::halfvec(512) END AS vector_512` + : `CASE WHEN NOT (${shortened}) THEN ${source}.${column('embedding', width)}::halfvec(${width}) END AS ${column('vector', width)}` + ), + ].join(', ') +} + +async function relationExists(tx: Sql | TransactionSql, name: string): Promise { + const [row] = await tx<{ present: boolean }[]>`SELECT to_regclass(${name}) IS NOT NULL AS present` + return row.present +} + +async function stateOf(tx: Sql | TransactionSql): Promise { + const [row] = await tx`SELECT * FROM public.search_retirement_state WHERE id = 1` + if (!row || row.version !== 1) + throw new SearchRetirementError('Unrecognized retirement state; no changes were made') + return row +} + +async function statusOf(tx: Sql | TransactionSql): Promise { + const state = await stateOf(tx) + const [relations] = await tx<{ pending: boolean; backlog: boolean; backup: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM public.search_retirement_changes) AS pending, + EXISTS (SELECT 1 FROM public.search_retirement_changes OFFSET 10000 LIMIT 1) AS backlog, + to_regclass('public.embedding_search_retirement_backup') IS NOT NULL AS backup` + return { + phase: state.phase, + invalidated: state.invalidated, + invalidationReason: state.invalidation_reason, + sourceScanned: Number(state.source_scanned), + copied: Number(state.copied), + reconciled: Number(state.reconciled), + validatedSource: Number(state.validated_source), + validatedShadow: Number(state.validated_shadow), + purgedEmbeddings: Number(state.purged_embeddings), + retiredDocuments: Number(state.retired_documents), + pendingChanges: relations.pending, + changeBacklogAtLimit: relations.backlog, + backupRetained: relations.backup, + } +} + +/** Read-only: inspecting an uninitialized database never creates maintenance objects. */ +export async function getSearchRetirementStatus(sql: Sql): Promise { + if (!(await relationExists(sql, STATE))) return null + return statusOf(sql) +} + +async function operation( + sql: Sql, + work: (tx: TransactionSql) => Promise, + ddl = false +): Promise { + return sql.begin('isolation level read committed', async (tx) => { + const [version] = await tx<{ supported: boolean }[]>` + SELECT current_setting('server_version_num')::int >= 170000 AS supported` + if (!version.supported) + throw new SearchRetirementError( + 'Retirement requires PostgreSQL 17 or newer for a bounded transaction deadline' + ) + await tx.unsafe("SET LOCAL transaction_timeout = '3s'") + await tx.unsafe(`SET LOCAL statement_timeout = '${ddl ? '5s' : '2s'}'`) + await tx.unsafe("SET LOCAL lock_timeout = '100ms'") + await tx.unsafe('SET LOCAL search_path = pg_catalog, public') + const [locks] = await tx<{ job: boolean; maintenance: boolean; migration: boolean }[]>` + SELECT pg_try_advisory_xact_lock(hashtextextended(${JOB_LOCK}, 0)) AS job, + pg_try_advisory_xact_lock(hashtextextended(${MAINTENANCE_LOCK}, 0)) AS maintenance, + pg_try_advisory_xact_lock(${MIGRATION_LOCK}::bigint) AS migration` + if (!locks.job || !locks.maintenance || !locks.migration) { + throw new SearchRetirementError( + 'Another migration or retirement operation is active; retry later' + ) + } + if (await relationExists(tx, 'public.search_embedding_cleanup_progress')) { + await tx`SELECT id FROM public.search_embedding_cleanup_progress WHERE id = 1 FOR UPDATE NOWAIT` + } + return work(tx) + }) as Promise +} + +function requireValid(state: StateRow): void { + if (state.invalidated) { + if (['cutover', 'purge', 'documents', 'finalize', 'done'].includes(state.phase)) { + throw new SearchRetirementError( + 'A captured target changed after cutover; investigate its scope before resuming. Retirement state is preserved' + ) + } + throw new SearchRetirementError( + 'Knowledge-base indexing metadata changed; abort and prepare a new replacement' + ) + } +} + +async function verifyRelationIdentity(tx: TransactionSql, state: StateRow): Promise { + const swapped = ['cutover', 'purge', 'documents', 'finalize', 'done'].includes(state.phase) + const [row] = await tx<{ active: number; shadow: number | null; backup: number | null }[]>` + SELECT to_regclass('public.embedding_search')::oid AS active, + to_regclass('public.embedding_search_retirement_shadow')::oid AS shadow, + to_regclass('public.embedding_search_retirement_backup')::oid AS backup` + if ( + row.active !== (swapped ? state.replacement_oid : state.original_oid) || + (!swapped && row.shadow !== state.replacement_oid) || + (state.phase === 'cutover' && row.backup !== state.original_oid) + ) { + throw new SearchRetirementError( + 'Retirement relation identity changed; refusing to adopt or replace it' + ) + } +} + +/** Catalog checks are bounded and reject dependencies that a name swap would strand on the old OID. */ +async function inspectProjection(tx: TransactionSql): Promise { + const [relation] = await tx<{ safe: boolean }[]>` + SELECT c.relkind = 'r' AND c.relpersistence = 'p' AND NOT c.relrowsecurity + AND NOT c.relforcerowsecurity + AND NOT EXISTS (SELECT 1 FROM pg_policy WHERE polrelid = c.oid) + AND NOT EXISTS (SELECT 1 FROM pg_attribute WHERE attrelid = c.oid AND attacl IS NOT NULL) + AND NOT EXISTS (SELECT 1 FROM pg_constraint WHERE confrelid = c.oid) + AND NOT EXISTS (SELECT 1 FROM pg_depend WHERE refobjid = c.oid + AND refclassid = 'pg_class'::regclass + AND classid IN ('pg_rewrite'::regclass, 'pg_proc'::regclass, 'pg_policy'::regclass)) + AND NOT EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = c.oid OR inhparent = c.oid) + AND NOT EXISTS (SELECT 1 FROM pg_publication_tables + WHERE schemaname = 'public' AND tablename IN ('embedding', 'embedding_search', 'knowledge_base')) + AS safe + FROM pg_class c WHERE c.oid = 'public.embedding_search'::regclass` + if (!relation?.safe) + throw new SearchRetirementError( + 'Projection dependencies, publication, or row security require a separate migration' + ) + const [constraints] = await tx<{ safe: boolean }[]>` + SELECT NOT EXISTS (SELECT 1 FROM pg_constraint c WHERE c.conrelid = 'public.embedding_search'::regclass + AND NOT ( + (c.contype = 'p' AND c.conkey = ARRAY[1]::smallint[]) + OR (c.contype = 'f' AND c.conkey = ARRAY[1]::smallint[] + AND c.confrelid = 'public.embedding'::regclass AND c.confkey = ARRAY[1]::smallint[] + AND c.confdeltype = 'c' AND c.confupdtype = 'a') + OR (c.contype = 'c' AND c.conname = 'embedding_search_width_check') + OR c.contype = 'n' + )) AS safe` + if (!constraints.safe) + throw new SearchRetirementError( + 'Projection constraints differ from the reviewed compatible shape' + ) + const actual = await tx<{ name: string; type: string; required: boolean }[]>` + SELECT attname AS name, format_type(atttypid, atttypmod) AS type, attnotnull AS required + FROM pg_attribute WHERE attrelid = 'public.embedding_search'::regclass + AND attnum > 0 AND NOT attisdropped ORDER BY attnum LIMIT 32` + const expected = new Map([ + ['id', 'text:true'], + ['knowledge_base_id', 'text:true'], + ['document_id', 'text:true'], + ['enabled', 'boolean:true'], + ['connector_id', 'text:false'], + ['acl', 'text[]:false'], + ...SOURCE_WIDTHS.map((width) => [column('binary', width), `bit(${width}):false`] as const), + ...WIDTHS.map((width) => [column('vector', width), `halfvec(${width}):false`] as const), + ]) + if ( + actual.length !== expected.size || + actual.some((entry) => expected.get(entry.name) !== `${entry.type}:${entry.required}`) + ) { + throw new SearchRetirementError('Projection columns differ from the reviewed compatible shape') + } + const [triggers] = await tx<{ unknown: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM pg_trigger t JOIN pg_proc p ON p.oid = t.tgfoid + WHERE t.tgrelid = 'public.embedding_search'::regclass AND NOT t.tgisinternal + AND (t.tgname <> 'embedding_search_source_acl_set' OR p.proname <> 'set_projection_source_acl' + OR p.pronamespace <> 'public'::regnamespace OR t.tgenabled <> 'O')) AS unknown` + if (triggers.unknown) + throw new SearchRetirementError( + 'Projection has an unrecognized trigger; no replacement was prepared' + ) + const [writer] = await tx<{ safe: boolean; guard: string | null }[]>` + SELECT t.tgenabled = 'O' AND t.tgtype = 21 AND t.tgnargs = 0 + AND t.tgfoid = to_regprocedure('public.sync_embedding_search()') + AND ARRAY(SELECT a.attname::text FROM pg_attribute a + WHERE a.attrelid = t.tgrelid AND a.attnum = ANY(t.tgattr) ORDER BY a.attname) + = ARRAY['document_id', 'embedding', 'embedding_1024', 'embedding_3072', + 'embedding_384', 'embedding_768', 'enabled', 'knowledge_base_id']::text[] AS safe, + pg_get_expr(t.tgqual, t.tgrelid) AS guard + FROM pg_trigger t WHERE t.tgrelid = 'public.embedding'::regclass + AND t.tgname = 'embedding_search_sync' AND NOT t.tgisinternal` + const guard = writer?.guard?.replace(/[\s()]/g, '') + const setting = "current_setting'sim.projection_mode'::text,true" + if ( + !writer?.safe || + (guard && + ![ + `${setting}ISDISTINCTFROM'async'::text`, + `NOT${setting}ISNOTDISTINCTFROM'async'::text`, + `NOTNOT${setting}ISDISTINCTFROM'async'::text`, + ].includes(guard)) + ) { + throw new SearchRetirementError( + 'Canonical vector synchronization trigger differs from the reviewed writer' + ) + } +} + +async function clonePrivileges(tx: TransactionSql): Promise { + const [owner] = await tx<{ name: string }[]>` + SELECT pg_get_userbyid(relowner) AS name FROM pg_class WHERE oid = 'public.embedding_search'::regclass` + await tx.unsafe(`ALTER TABLE ${SHADOW} OWNER TO ${identifier(owner.name)}`) + const previousGrants = await tx<{ role: string }[]>` + SELECT DISTINCT CASE WHEN a.grantee = 0 THEN 'PUBLIC' ELSE pg_get_userbyid(a.grantee) END AS role + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search_retirement_shadow'::regclass AND a.grantee <> c.relowner LIMIT 129` + if (previousGrants.length > 128) + throw new SearchRetirementError( + 'Default table privileges exceed the reviewed maintenance bound' + ) + for (const grant of previousGrants) { + await tx.unsafe( + `REVOKE ALL PRIVILEGES ON TABLE ${SHADOW} FROM ${grant.role === 'PUBLIC' ? 'PUBLIC' : identifier(grant.role)}` + ) + } + const grants = await tx<{ role: string; privilege: string; grantable: boolean }[]>` + SELECT CASE WHEN a.grantee = 0 THEN 'PUBLIC' ELSE pg_get_userbyid(a.grantee) END AS role, + a.privilege_type AS privilege, a.is_grantable AS grantable + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search'::regclass AND a.grantee <> c.relowner LIMIT 129` + if (grants.length > 128) + throw new SearchRetirementError( + 'Projection privilege set exceeds the reviewed maintenance bound' + ) + for (const grant of grants) { + if ( + ![ + 'SELECT', + 'INSERT', + 'UPDATE', + 'DELETE', + 'TRUNCATE', + 'REFERENCES', + 'TRIGGER', + 'MAINTAIN', + ].includes(grant.privilege) + ) { + throw new SearchRetirementError('Projection has an unrecognized privilege') + } + await tx.unsafe( + `GRANT ${grant.privilege} ON TABLE ${SHADOW} TO ${grant.role === 'PUBLIC' ? 'PUBLIC' : identifier(grant.role)}${grant.grantable ? ' WITH GRANT OPTION' : ''}` + ) + } +} + +async function installCapture(tx: TransactionSql): Promise { + await tx.unsafe(`CREATE FUNCTION public.capture_search_retirement_change() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + IF TG_OP <> 'INSERT' THEN + INSERT INTO public.search_retirement_changes (embedding_id) VALUES (OLD.id) + ON CONFLICT (embedding_id) DO UPDATE SET generation = search_retirement_changes.generation + 1; + END IF; + IF TG_OP <> 'DELETE' AND (TG_OP <> 'UPDATE' OR OLD.id IS DISTINCT FROM NEW.id) THEN + INSERT INTO public.search_retirement_changes (embedding_id) VALUES (NEW.id) + ON CONFLICT (embedding_id) DO UPDATE SET generation = search_retirement_changes.generation + 1; + END IF; + RETURN NULL; + END $$`) + await tx.unsafe(`CREATE TRIGGER search_retirement_capture AFTER INSERT OR UPDATE OR DELETE + ON public.embedding FOR EACH ROW EXECUTE FUNCTION public.capture_search_retirement_change()`) + await tx.unsafe(`CREATE FUNCTION public.invalidate_search_retirement() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + UPDATE public.search_retirement_state SET invalidated = true, + invalidation_reason = 'knowledge-base-metadata-changed' WHERE id = 1; + RETURN NULL; + END $$`) + await tx.unsafe(`CREATE TRIGGER search_retirement_invalidate + AFTER UPDATE OF is_search_index, embedding_model, embedding_dimension ON public.knowledge_base + FOR EACH ROW WHEN (OLD.is_search_index IS DISTINCT FROM NEW.is_search_index + OR OLD.embedding_model IS DISTINCT FROM NEW.embedding_model + OR OLD.embedding_dimension IS DISTINCT FROM NEW.embedding_dimension) + EXECUTE FUNCTION public.invalidate_search_retirement()`) + if (await relationExists(tx, 'public.search_embedding_cleanup_progress')) { + await tx.unsafe( + 'LOCK TABLE public.search_embedding_cleanup_progress IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe(`CREATE FUNCTION public.guard_legacy_search_retirement() RETURNS trigger + LANGUAGE plpgsql AS $$ BEGIN + RAISE EXCEPTION 'Legacy cleanup is fenced by replacement retirement' USING ERRCODE = '55000'; + END $$`) + await tx.unsafe(`CREATE TRIGGER search_retirement_legacy_guard BEFORE INSERT OR UPDATE OR DELETE + ON public.search_embedding_cleanup_progress FOR EACH ROW + EXECUTE FUNCTION public.guard_legacy_search_retirement()`) + } +} + +/** Only empty-object DDL runs here. Copying, validation, cutover, and deletion require separate calls. */ +export async function initializeSearchRetirement(sql: Sql): Promise { + return operation( + sql, + async (tx) => { + if (await relationExists(tx, STATE)) { + const state = await stateOf(tx) + await verifyRelationIdentity(tx, state) + return statusOf(tx) + } + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe('LOCK TABLE public.embedding_search IN ACCESS SHARE MODE NOWAIT') + await inspectProjection(tx) + for (const relation of [ + SHADOW, + BACKUP, + 'public.search_retirement_targets', + 'public.search_retirement_changes', + ]) { + if (await relationExists(tx, relation)) + throw new SearchRetirementError( + 'Unowned retirement objects already exist; refusing to replace them' + ) + } + await tx.unsafe(`CREATE TABLE ${STATE} ( + id integer PRIMARY KEY CHECK (id = 1), version integer NOT NULL CHECK (version = 1), + phase text NOT NULL, resume_phase text NOT NULL DEFAULT 'validate-source', after_id text NOT NULL DEFAULT '', + invalidated boolean NOT NULL DEFAULT false, invalidation_reason text, + original_oid oid NOT NULL, replacement_oid oid NOT NULL, + source_scanned bigint NOT NULL DEFAULT 0, copied bigint NOT NULL DEFAULT 0, + reconciled bigint NOT NULL DEFAULT 0, validated_source bigint NOT NULL DEFAULT 0, + validated_shadow bigint NOT NULL DEFAULT 0, purged_embeddings bigint NOT NULL DEFAULT 0, + retired_documents bigint NOT NULL DEFAULT 0, round_mutations bigint NOT NULL DEFAULT 0 + , ready_at timestamptz, index_manifest jsonb NOT NULL DEFAULT '{}', relation_manifest jsonb NOT NULL DEFAULT '{}' + )`) + await tx.unsafe( + 'CREATE TABLE public.search_retirement_targets (knowledge_base_id text PRIMARY KEY)' + ) + await tx.unsafe( + 'CREATE TABLE public.search_retirement_changes (embedding_id text PRIMARY KEY, generation bigint NOT NULL DEFAULT 1)' + ) + await tx.unsafe( + `CREATE TABLE ${SHADOW} (LIKE public.embedding_search INCLUDING DEFAULTS INCLUDING GENERATED INCLUDING CONSTRAINTS INCLUDING STORAGE)` + ) + await tx.unsafe(`ALTER TABLE ${SHADOW} ADD CONSTRAINT embedding_search_retirement_shadow_pkey PRIMARY KEY (id), + ADD CONSTRAINT embedding_search_retirement_shadow_embedding_fk FOREIGN KEY (id) REFERENCES public.embedding(id) ON DELETE CASCADE`) + for (const index of SHARED_INDEXES.slice(1)) { + await tx.unsafe( + `CREATE INDEX ${identifier(`embedding_search_retirement_shadow_${index.suffix}`)} ON ${SHADOW} ${index.definition}` + ) + } + await clonePrivileges(tx) + await tx`INSERT INTO public.search_retirement_state (id, version, phase, original_oid, replacement_oid) + VALUES (1, 1, 'snapshot', 'public.embedding_search'::regclass, 'public.embedding_search_retirement_shadow'::regclass)` + await tx`UPDATE public.search_retirement_state SET index_manifest = ( + SELECT jsonb_object_agg(c.relname, pg_get_indexdef(i.indexrelid)) FROM pg_index i + JOIN pg_class c ON c.oid = i.indexrelid WHERE i.indrelid = 'public.embedding_search_retirement_shadow'::regclass + ) WHERE id = 1` + await tx.unsafe(`UPDATE ${STATE} SET relation_manifest = (${REPLACEMENT_SHAPE}) WHERE id = 1`) + await installCapture(tx) + return statusOf(tx) + }, + true + ) +} + +async function copyPage( + tx: TransactionSql, + state: StateRow, + pageSize: number +): Promise { + const [page] = await tx.unsafe( + `WITH page AS MATERIALIZED ( + SELECT id FROM public.embedding WHERE id > $1 ORDER BY id LIMIT $2 + ), expected AS MATERIALIZED ( + SELECT ${projectedValues('e', 'k.embedding_model')} FROM page p + JOIN public.embedding e ON e.id = p.id JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE NOT k.is_search_index + ), written AS ( + INSERT INTO ${SHADOW} AS s (${columns}) SELECT ${columns} FROM expected + ON CONFLICT (id) DO UPDATE SET ${assignments} WHERE ${comparison('s', 'EXCLUDED')} + RETURNING 1 + ) SELECT max(id) AS after_id, count(*)::int AS scanned, + (SELECT count(*)::int FROM written) AS changed FROM page`, + [state.after_id, pageSize] + ) + return page +} + +/** Read a generation before source state; a newer committed write keeps its queue entry for retry. */ +async function reconcilePage(tx: TransactionSql, pageSize: number): Promise { + const queued = await tx<{ embedding_id: string; generation: string }[]>` + SELECT embedding_id, generation::text FROM public.search_retirement_changes ORDER BY embedding_id LIMIT ${pageSize}` + if (queued.length === 0) return 0 + const ids = queued.map((row) => row.embedding_id) + await tx.unsafe( + `WITH expected AS MATERIALIZED ( + SELECT ${projectedValues('e', 'k.embedding_model')} FROM public.embedding e + JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE e.id = ANY($1::text[]) AND NOT k.is_search_index + ) INSERT INTO ${SHADOW} AS s (${columns}) SELECT ${columns} FROM expected + ON CONFLICT (id) DO UPDATE SET ${assignments} WHERE ${comparison('s', 'EXCLUDED')}`, + [ids] + ) + await tx.unsafe( + `DELETE FROM ${SHADOW} s WHERE s.id = ANY($1::text[]) AND NOT EXISTS ( + SELECT 1 FROM public.embedding e JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE e.id = s.id AND NOT k.is_search_index)`, + [ids] + ) + await tx`DELETE FROM public.search_retirement_changes q USING + unnest(${ids}::text[], ${queued.map((row) => row.generation)}::bigint[]) AS done(id, generation) + WHERE q.embedding_id = done.id AND q.generation = done.generation` + return queued.length +} + +async function validatePage( + tx: TransactionSql, + state: StateRow, + pageSize: number +): Promise { + const source = state.phase === 'validate-source' + const [page] = await tx.unsafe<(PageResult & { invalid: boolean })[]>( + `WITH page AS MATERIALIZED ( + SELECT id FROM ${source ? 'public.embedding' : SHADOW} WHERE id > $1 ORDER BY id LIMIT $2 + ), expected AS MATERIALIZED ( + SELECT ${projectedValues('e', 'k.embedding_model')} FROM page p + JOIN public.embedding e ON e.id = p.id JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE NOT k.is_search_index + ) SELECT max(p.id) AS after_id, count(*)::int AS scanned, 0 AS changed, + coalesce(bool_or(NOT EXISTS (SELECT 1 FROM public.search_retirement_changes q WHERE q.embedding_id = p.id) + AND ${source ? `e.id IS NOT NULL AND (s.id IS NULL OR ${comparison('s', 'e')})` : `(e.id IS NULL OR ${comparison('s', 'e')})`}), false) AS invalid + FROM page p LEFT JOIN expected e ON e.id = p.id LEFT JOIN ${SHADOW} s ON s.id = p.id`, + [state.after_id, pageSize] + ) + if (page.invalid) + throw new SearchRetirementError( + 'Replacement validation found an uncaptured mismatch; cutover is not permitted' + ) + return page +} + +async function transition(tx: TransactionSql, phase: SearchRetirementPhase): Promise { + await tx`UPDATE public.search_retirement_state SET phase = ${phase}, after_id = '', round_mutations = 0 WHERE id = 1` +} + +/** Late writes retain their generation until the corresponding canonical row is safely retired. */ +async function purgeCapturedPage(tx: TransactionSql, pageSize: number): Promise { + const queued = await tx<{ embedding_id: string; generation: string }[]>` + SELECT embedding_id, generation::text FROM public.search_retirement_changes + ORDER BY embedding_id LIMIT ${Math.min(pageSize, 25)}` + if (queued.length === 0) return false + const ids = queued.map((row) => row.embedding_id) + const [page] = await tx<{ changed: number; invalid: boolean }[]>` + WITH targets AS MATERIALIZED ( + SELECT e.id, e.knowledge_base_id FROM public.embedding e + JOIN public.search_retirement_targets t ON t.knowledge_base_id = e.knowledge_base_id + WHERE e.id = ANY(${ids}::text[]) + ), bases AS MATERIALIZED ( + SELECT k.id, k.is_search_index FROM public.knowledge_base k + WHERE k.id IN (SELECT knowledge_base_id FROM targets) ORDER BY k.id FOR SHARE OF k + ), changed AS ( + DELETE FROM public.embedding e USING targets t, bases b + WHERE e.id = t.id AND e.knowledge_base_id = t.knowledge_base_id + AND b.id = t.knowledge_base_id AND b.is_search_index RETURNING e.id + ) SELECT (SELECT count(*)::int FROM changed) AS changed, + EXISTS (SELECT 1 FROM targets t LEFT JOIN bases b ON b.id = t.knowledge_base_id + WHERE b.id IS NULL OR NOT b.is_search_index) AS invalid` + if (page.invalid) + throw new SearchRetirementError( + 'A captured cleanup target is no longer Search-marked; no page was committed' + ) + await tx`DELETE FROM public.search_retirement_changes q USING + unnest(${ids}::text[], ${queued.map((row) => row.generation)}::bigint[]) AS done(id, generation) + WHERE q.embedding_id = done.id AND q.generation = done.generation` + await tx`UPDATE public.search_retirement_state SET purged_embeddings = purged_embeddings + ${page.changed} + WHERE id = 1` + return true +} + +async function removeCapture(tx: TransactionSql): Promise { + // Trigger removal can upgrade its relation lock; refuse a busy reader within one millisecond. + await tx.unsafe("SET LOCAL lock_timeout = '1ms'") + await tx.unsafe('DROP TRIGGER search_retirement_capture ON public.embedding') + await tx.unsafe('DROP TRIGGER search_retirement_invalidate ON public.knowledge_base') + await tx.unsafe( + 'DROP FUNCTION public.capture_search_retirement_change(), public.invalidate_search_retirement()' + ) +} + +async function purgePage(tx: TransactionSql, state: StateRow, pageSize: number): Promise { + if (await purgeCapturedPage(tx, pageSize)) { + if (state.phase === 'documents') await transition(tx, 'purge') + return + } + const documents = state.phase === 'documents' + const mutation = documents + ? `UPDATE public.document d SET user_excluded = true, enabled = false, processing_queue_token = NULL, + processing_queued_at = NULL, processing_deferred_until = NULL + FROM targets t WHERE d.id = t.id AND d.knowledge_base_id = t.knowledge_base_id + AND (NOT d.user_excluded OR d.enabled OR d.processing_queue_token IS NOT NULL + OR d.processing_queued_at IS NOT NULL OR d.processing_deferred_until IS NOT NULL) RETURNING d.id` + : `DELETE FROM public.embedding e USING targets t + WHERE e.id = t.id AND e.knowledge_base_id = t.knowledge_base_id RETURNING e.id` + const [page] = await tx.unsafe<(PageResult & { invalid: boolean })[]>( + `WITH page AS MATERIALIZED ( + SELECT id, knowledge_base_id FROM public.${documents ? 'document' : 'embedding'} + WHERE id > $1 ORDER BY id LIMIT $2 + ), targets AS MATERIALIZED ( + SELECT p.* FROM page p JOIN public.search_retirement_targets t ON t.knowledge_base_id = p.knowledge_base_id + ), bases AS MATERIALIZED ( + SELECT k.id, k.is_search_index FROM public.knowledge_base k + WHERE k.id IN (SELECT knowledge_base_id FROM targets) ORDER BY k.id FOR SHARE OF k + ), safe_targets AS MATERIALIZED ( + SELECT t.* FROM targets t JOIN bases b ON b.id = t.knowledge_base_id WHERE b.is_search_index + ), changed AS (${mutation.replace('FROM targets t', 'FROM safe_targets t').replace('USING targets t', 'USING safe_targets t')}) + SELECT max(id) AS after_id, count(*)::int AS scanned, + (SELECT count(*)::int FROM changed) AS changed, + EXISTS (SELECT 1 FROM targets t LEFT JOIN bases b ON b.id = t.knowledge_base_id + WHERE b.id IS NULL OR NOT b.is_search_index) AS invalid FROM page`, + [state.after_id, Math.min(pageSize, 25)] + ) + if (page.invalid) + throw new SearchRetirementError( + 'A captured cleanup target is no longer Search-marked; no page was committed' + ) + if (page.after_id !== null) { + await tx.unsafe( + `UPDATE ${STATE} SET after_id = $1, round_mutations = round_mutations + $2, + ${documents ? 'retired_documents' : 'purged_embeddings'} = ${documents ? 'retired_documents' : 'purged_embeddings'} + $2 WHERE id = 1`, + [page.after_id, page.changed] + ) + return + } + if (Number(state.round_mutations) > 0) { + await transition(tx, state.phase) + return + } + await transition(tx, documents ? 'finalize' : 'documents') +} + +/** Commits at most one bounded page. A timeout rolls back both its writes and cursor. */ +export async function advanceSearchRetirement( + sql: Sql, + options: { pageSize: number } +): Promise { + if (!Number.isInteger(options.pageSize) || options.pageSize < 1 || options.pageSize > 100) { + throw new SearchRetirementError('Page size must be an integer from 1 to 100') + } + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + const size = options.pageSize + const before = await statusOf(tx) + if ( + before.changeBacklogAtLimit && + !['ready', 'catchup', 'cutover', 'purge', 'documents', 'finalize', 'done'].includes( + state.phase + ) + ) { + const reconciled = await reconcilePage(tx, size) + await tx`UPDATE public.search_retirement_state SET reconciled = reconciled + ${reconciled} WHERE id = 1` + return statusOf(tx) + } + if (state.phase === 'snapshot') { + const [page] = await tx`WITH page AS MATERIALIZED ( + SELECT id, is_search_index FROM public.knowledge_base WHERE id > ${state.after_id} ORDER BY id LIMIT ${size} + ), inserted AS (INSERT INTO public.search_retirement_targets (knowledge_base_id) + SELECT id FROM page WHERE is_search_index ON CONFLICT DO NOTHING RETURNING 1) + SELECT max(id) AS after_id, count(*)::int AS scanned, (SELECT count(*)::int FROM inserted) AS changed FROM page` + if (page.after_id === null) await transition(tx, 'copy') + else + await tx`UPDATE public.search_retirement_state SET after_id = ${page.after_id} WHERE id = 1` + } else if (state.phase === 'copy') { + const page = await copyPage(tx, state, size) + if (page.after_id === null) await transition(tx, 'catchup') + else + await tx`UPDATE public.search_retirement_state SET after_id = ${page.after_id}, + source_scanned = source_scanned + ${page.scanned}, copied = copied + ${page.changed} WHERE id = 1` + } else if (state.phase === 'catchup' || state.phase === 'ready') { + const reconciled = await reconcilePage(tx, size) + if (reconciled > 0) { + await tx`UPDATE public.search_retirement_state SET reconciled = reconciled + ${reconciled}, ready_at = NULL WHERE id = 1` + if (state.phase === 'ready') await transition(tx, 'catchup') + } + if (reconciled === 0) { + if (state.resume_phase === 'ready' && !state.ready_at) { + await tx.unsafe(`ANALYZE ${SHADOW} (id, knowledge_base_id, document_id, enabled)`) + await tx`UPDATE public.search_retirement_state SET ready_at = clock_timestamp() WHERE id = 1` + } + if (state.phase === 'catchup') await transition(tx, state.resume_phase) + } + } else if (state.phase === 'validate-source' || state.phase === 'validate-shadow') { + const page = await validatePage(tx, state, size) + if (page.after_id === null) { + if (state.phase === 'validate-source') await transition(tx, 'validate-shadow') + else { + await tx`UPDATE public.search_retirement_state SET resume_phase = 'ready' WHERE id = 1` + await transition(tx, 'catchup') + } + } else { + const counter = state.phase === 'validate-source' ? 'validated_source' : 'validated_shadow' + await tx.unsafe( + `UPDATE ${STATE} SET after_id = $1, ${counter} = ${counter} + $2 WHERE id = 1`, + [page.after_id, page.scanned] + ) + } + } else if (state.phase === 'purge' || state.phase === 'documents') { + await purgePage(tx, state, size) + } + return statusOf(tx) + }) +} + +async function replaceVectorWriter(tx: TransactionSql): Promise { + await tx.unsafe(`CREATE OR REPLACE FUNCTION public.sync_embedding_search() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + IF NOT EXISTS (SELECT 1 FROM public.knowledge_base WHERE id = NEW.knowledge_base_id AND NOT is_search_index) THEN + DELETE FROM public.embedding_search WHERE id = NEW.id; + RETURN NEW; + END IF; + INSERT INTO public.embedding_search AS s (${columns}) + SELECT ${projectedValues('NEW', 'k.embedding_model')} FROM public.knowledge_base k WHERE k.id = NEW.knowledge_base_id + ON CONFLICT (id) DO UPDATE SET ${assignments}; + RETURN NEW; + END $$`) +} + +/** Ordinary writes no longer need capture once the replacement receives synchronous projections. */ +async function retainSearchCapture(tx: TransactionSql): Promise { + await tx.unsafe(`CREATE OR REPLACE FUNCTION public.capture_search_retirement_change() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + IF TG_OP <> 'DELETE' AND EXISTS (SELECT 1 FROM public.search_retirement_targets + WHERE knowledge_base_id = NEW.knowledge_base_id) THEN + INSERT INTO public.search_retirement_changes (embedding_id) VALUES (NEW.id) + ON CONFLICT (embedding_id) DO UPDATE SET generation = search_retirement_changes.generation + 1; + END IF; + RETURN NULL; + END $$`) + await tx.unsafe(`CREATE OR REPLACE FUNCTION public.invalidate_search_retirement() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + IF OLD.is_search_index IS DISTINCT FROM NEW.is_search_index + AND EXISTS (SELECT 1 FROM public.search_retirement_targets WHERE knowledge_base_id = NEW.id) THEN + UPDATE public.search_retirement_state SET invalidated = true, + invalidation_reason = 'captured-target-marker-changed' WHERE id = 1; + END IF; + RETURN NULL; + END $$`) +} + +/** A zero-backlog metadata swap; busy relations cause an immediate rollback, never a wait queue. */ +export async function cutoverSearchRetirement(sql: Sql): Promise { + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + if (state.phase !== 'ready') + throw new SearchRetirementError( + 'Replacement must complete copy, catchup, and validation before cutover' + ) + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe(`LOCK TABLE public.embedding_search, ${SHADOW} IN ACCESS EXCLUSIVE MODE NOWAIT`) + requireValid(await stateOf(tx)) + await inspectProjection(tx) + if ((await statusOf(tx)).pendingChanges) + throw new SearchRetirementError( + 'Concurrent changes remain; run another bounded catchup page before cutover' + ) + const [snapshots] = await tx<{ observable: boolean; stale: boolean; prepared: boolean }[]>` + SELECT (SELECT rolsuper FROM pg_roles WHERE rolname = current_user) + OR pg_has_role(current_user, 'pg_read_all_stats', 'USAGE') AS observable, + EXISTS (SELECT 1 FROM pg_stat_activity a WHERE a.datid = (SELECT oid FROM pg_database WHERE datname = current_database()) + AND a.pid <> pg_backend_pid() AND a.xact_start <= (SELECT ready_at FROM public.search_retirement_state WHERE id = 1)) AS stale, + (SELECT ready_at IS NOT NULL FROM public.search_retirement_state WHERE id = 1) AS prepared` + if (!snapshots.observable) + throw new SearchRetirementError( + 'Cutover requires pg_read_all_stats to verify the transaction snapshot barrier' + ) + if (!snapshots.prepared || snapshots.stale) + throw new SearchRetirementError( + 'Transactions predate the replacement readiness barrier; let them finish and retry cutover' + ) + const [indexes] = await tx<{ valid: boolean }[]>` + SELECT coalesce(bool_and(i.indisvalid AND i.indisready), false) + AND jsonb_object_agg(c.relname, pg_get_indexdef(i.indexrelid)) = + (SELECT index_manifest FROM public.search_retirement_state WHERE id = 1) AS valid + FROM pg_index i JOIN pg_class c ON c.oid = i.indexrelid + WHERE i.indrelid = 'public.embedding_search_retirement_shadow'::regclass` + if (!indexes.valid) + throw new SearchRetirementError('Replacement index definitions changed after preparation') + const [shape] = await tx.unsafe<{ same: boolean }[]>(`SELECT (${REPLACEMENT_SHAPE}) = + (SELECT relation_manifest FROM ${STATE} WHERE id = 1) AS same`) + if (!shape.same) { + throw new SearchRetirementError( + 'Replacement columns, constraints, or dependencies changed after preparation' + ) + } + const [privileges] = await tx<{ same: boolean }[]>` + WITH source AS (SELECT a.grantee, a.privilege_type, a.is_grantable + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search'::regclass), + replacement AS (SELECT a.grantee, a.privilege_type, a.is_grantable + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search_retirement_shadow'::regclass) + SELECT a.relowner = b.relowner AND NOT EXISTS (SELECT * FROM source EXCEPT SELECT * FROM replacement) + AND NOT EXISTS (SELECT * FROM replacement EXCEPT SELECT * FROM source) AS same + FROM pg_class a, pg_class b WHERE a.oid = 'public.embedding_search'::regclass + AND b.oid = 'public.embedding_search_retirement_shadow'::regclass` + if (!privileges.same) + throw new SearchRetirementError( + 'Projection privileges changed; replacement cutover is not safe' + ) + const [aclTrigger] = await tx<{ present: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM pg_trigger WHERE tgrelid = 'public.embedding_search'::regclass + AND tgname = 'embedding_search_source_acl_set' AND NOT tgisinternal) AS present` + for (const index of SHARED_INDEXES) { + const [installed] = await tx<{ present: boolean; valid: boolean; original: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM pg_class WHERE oid = to_regclass(${`public.${index.name}`})) AS present, + EXISTS (SELECT 1 FROM pg_index WHERE indexrelid = to_regclass(${`public.${index.name}`}) + AND indrelid = 'public.embedding_search'::regclass) AS original, + EXISTS (SELECT 1 FROM pg_index WHERE indexrelid = to_regclass(${`public.embedding_search_retirement_shadow_${index.suffix}`}) + AND indrelid = 'public.embedding_search_retirement_shadow'::regclass AND indisvalid AND indisready) AS valid` + if (!installed.valid) + throw new SearchRetirementError( + 'Replacement index is missing or invalid; cutover is not permitted' + ) + if (installed.present && !installed.original) + throw new SearchRetirementError('A shared index name belongs to an unrelated relation') + if (installed.present) + await tx.unsafe( + `ALTER INDEX public.${identifier(index.name)} RENAME TO ${identifier(`embedding_search_retirement_backup_${index.suffix}`)}` + ) + await tx.unsafe( + `ALTER INDEX public.${identifier(`embedding_search_retirement_shadow_${index.suffix}`)} RENAME TO ${identifier(index.name)}` + ) + } + await tx.unsafe( + `ALTER TABLE public.embedding_search RENAME TO embedding_search_retirement_backup` + ) + await tx.unsafe(`ALTER TABLE ${SHADOW} RENAME TO embedding_search`) + await tx.unsafe(`ALTER TABLE public.embedding_search + RENAME CONSTRAINT embedding_search_retirement_shadow_embedding_fk TO embedding_search_id_embedding_id_fk`) + if (aclTrigger.present) { + await tx.unsafe(`CREATE TRIGGER embedding_search_source_acl_set + BEFORE INSERT OR UPDATE OF document_id, enabled ON public.embedding_search + FOR EACH ROW WHEN (current_setting('sim.projection_mode', true) IS DISTINCT FROM 'async') + EXECUTE FUNCTION public.set_projection_source_acl()`) + } + await replaceVectorWriter(tx) + await retainSearchCapture(tx) + await transition(tx, 'cutover') + return statusOf(tx) + }) +} + +/** Explicitly ends the backup observation window before any canonical deletion can touch its HNSW graphs. */ +export async function beginSearchRetirementPurge(sql: Sql): Promise { + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + if (state.phase !== 'cutover') + throw new SearchRetirementError( + 'Purge requires a completed cutover and an explicit end to backup retention' + ) + await tx.unsafe('LOCK TABLE public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT') + await tx.unsafe(`LOCK TABLE ${BACKUP} IN ACCESS EXCLUSIVE MODE NOWAIT`) + // Removing the backup FK can upgrade its parent lock; keep that wait below a normal page's limit. + await tx.unsafe("SET LOCAL lock_timeout = '1ms'") + await tx.unsafe(`DROP TABLE ${BACKUP} RESTRICT`) + await transition(tx, 'purge') + return statusOf(tx) + }) +} + +/** Explicitly removes capture after operators clear the final primary and replica DDL window. */ +export async function finalizeSearchRetirement(sql: Sql): Promise { + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + if (state.phase !== 'finalize') { + throw new SearchRetirementError( + 'Finalization requires completed canonical and document retirement' + ) + } + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + requireValid(await stateOf(tx)) + if ((await statusOf(tx)).pendingChanges) { + await transition(tx, 'purge') + return statusOf(tx) + } + await removeCapture(tx) + await transition(tx, 'done') + return statusOf(tx) + }) +} + +/** Before cutover, abort removes only this job's empty-or-partial replacement and capture machinery. */ +export async function abortSearchRetirement(sql: Sql): Promise { + return operation(sql, async (tx) => { + if (!(await relationExists(tx, STATE))) return + const state = await stateOf(tx) + await verifyRelationIdentity(tx, state) + if (['cutover', 'purge', 'documents', 'finalize', 'done'].includes(state.phase)) { + throw new SearchRetirementError( + 'A cut-over replacement cannot be rolled back by renaming a stale backup' + ) + } + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe(`LOCK TABLE ${SHADOW} IN ACCESS EXCLUSIVE MODE NOWAIT`) + await removeCapture(tx) + if (await relationExists(tx, 'public.search_embedding_cleanup_progress')) { + await tx.unsafe( + 'LOCK TABLE public.search_embedding_cleanup_progress IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe( + 'DROP TRIGGER IF EXISTS search_retirement_legacy_guard ON public.search_embedding_cleanup_progress' + ) + } + await tx.unsafe('DROP FUNCTION IF EXISTS public.guard_legacy_search_retirement()') + await tx.unsafe( + `DROP TABLE ${SHADOW}, public.search_retirement_changes, public.search_retirement_targets, ${STATE} RESTRICT` + ) + }) +} diff --git a/packages/db/package.json b/packages/db/package.json index 381203f99a1..31cf5145650 100644 --- a/packages/db/package.json +++ b/packages/db/package.json @@ -36,6 +36,7 @@ "db:migrate": "bun --env-file=.env run ./scripts/migrate.ts", "db:reconcile-fork-kb-file-ownership": "bun --env-file=.env run ./scripts/reconcile-fork-kb-file-ownership.ts", "db:reconcile-workspace-storage": "bun --env-file=.env run ./scripts/reconcile-workspace-storage.ts", + "db:retire-indexed-search": "bun --no-env-file ./scripts/retire-indexed-search.ts", "db:studio": "bunx drizzle-kit studio --config=./drizzle.config.ts", "test": "vitest run", "type-check": "tsc --noEmit", diff --git a/packages/db/script-migrations/0027_retire_search_embeddings.ts b/packages/db/script-migrations/0027_retire_search_embeddings.ts index f58f5b14b01..e9022b633f1 100644 --- a/packages/db/script-migrations/0027_retire_search_embeddings.ts +++ b/packages/db/script-migrations/0027_retire_search_embeddings.ts @@ -1,11 +1,9 @@ -import { parseArgs } from 'node:util' -import { resolveMigrationDatabaseUrl } from '@sim/db/script-migrations/database-url' import type { ScriptMigration } from '@sim/db/script-migrations/types' import { retryOnLockTimeout } from '@sim/db/scripts/lock-timeout-retry' import { createLogger } from '@sim/logger' import { getPostgresCancellationReason } from '@sim/utils/errors' import { sleep } from '@sim/utils/helpers' -import postgres, { type Sql, type TransactionSql } from 'postgres' +import type { Sql, TransactionSql } from 'postgres' const logger = createLogger('RetireSearchEmbeddings') /** Most IDs one page reads in primary-key order; reading is cheap next to the mutation. */ @@ -409,37 +407,10 @@ async function validateTargetMarkers(tx: TransactionSql): Promise { } } -/** - * The operator entry: resumes the saved cursor and, with `--maintenance`, also rebuilds the indexes, - * vacuums, and journals the completed cleanup. `--pause-ratio` and `--max-rows` set the pacing. - */ +/** Historical implementation remains for migration replay tests; the unbounded CLI is retired. */ if (import.meta.main) { - const { values } = parseArgs({ - options: { - maintenance: { type: 'boolean', default: false }, - 'pause-ratio': { type: 'string' }, - 'max-rows': { type: 'string' }, - }, - }) - const pacing: RetirementPacing = { - pauseRatio: Number(values['pause-ratio'] ?? DEFAULT_RETIREMENT_PACING.pauseRatio), - maxRows: Number(values['max-rows'] ?? DEFAULT_RETIREMENT_PACING.maxRows), - } - const url = resolveMigrationDatabaseUrl() - if (!url) throw new Error('DATABASE_URL is required for Search retirement') - const sql = postgres(url, { max: 1, max_lifetime: null, onnotice: () => undefined }) - try { - if (values.maintenance) { - const { runScriptMigrations } = await import('@sim/db/script-migrations/index') - const { retireAllSearchEmbeddings } = await import( - '@sim/db/script-migrations/0029_retire_all_search_embeddings' - ) - await runScriptMigrations(sql, [retireAllSearchEmbeddings(pacing)]) - } else { - await retireSearchEmbeddings(sql, pacing) - logger.info('Search retirement pass finished; run with --maintenance off-peak to complete it') - } - } finally { - await sql.end() - } + logger.error( + 'Use packages/db/scripts/retire-indexed-search.ts. The legacy delete/reindex command is disabled.' + ) + process.exitCode = 1 } diff --git a/packages/db/script-migrations/search-embedding-retirement.md b/packages/db/script-migrations/search-embedding-retirement.md index 7b8864c4499..bef76e1feed 100644 --- a/packages/db/script-migrations/search-embedding-retirement.md +++ b/packages/db/script-migrations/search-embedding-retirement.md @@ -1,179 +1,18 @@ # Retiring legacy Search indexes -The retirement is an **operator-run maintenance command, not a deploy step**. Deploy migrations no -longer register it: a long cleanup inside the deploy migration generated heavy WAL and stalled -application writes, and it held the release until it finished. It is optional storage reclamation -once live Search is on, so it runs separately, paced, at a time the operator chooses. Self-hosted -operators can run the same command. - -`0029_retire_all_search_embeddings` snapshots every knowledge base whose persisted `is_search_index` -marker is true and supersedes the single-KB retirement and maintenance receipts (`0027`/`0028`), -including databases that already recorded either. No Search KB is a completed no-op. Once saved, the -snapshot stays fixed across runs even if another Search KB is created. Ordinary KBs and the selected -KBs' live source/credential configuration, document metadata, and backing files are preserved. - -## Before running - -The app and workers must already use live Search, and older indexing jobs must be drained. -Enterprise Search no longer has an indexed backend or an environment toggle to re-enable it. -Older releases could re-enable indexed Search, so their workers must be drained before retirement. Live source setup may still -create a Search KB for configuration; it does not index content. Document uploads, dispatch and -queued processing also honor the indexed-search gate. - -## Running it - -From the repository root, with the migration role's writer DSN on a **direct or session-pooled** -connection (the run holds a session advisory lock and session settings; PgBouncer transaction pooling -is unsupported, and reserving a postgres.js client does not pin a backend through it): - -```sh -# Retire documents and delete their chunks, resuming the saved cursor. Safe to stop and rerun. -MIGRATION_DATABASE_URL= bun run packages/db/script-migrations/0027_retire_search_embeddings.ts - -# Off-peak: finish any remaining retirement, rebuild the HNSW indexes, vacuum, and record completion. -MIGRATION_DATABASE_URL= bun run packages/db/script-migrations/0027_retire_search_embeddings.ts --maintenance -``` - -| Flag | Default | Effect | -| --- | --- | --- | -| `--pause-ratio N` | `2` | After each page, pause N × the page's duration (at most one minute), so the run is busy at most `1 / (1 + N)` of the time. Raise it to go gentler. | -| `--max-rows N` | `2000` | The most rows one page may update or delete (25–8,000). Lower it to make each page lighter. | -| `--maintenance` | off | After retirement, run the index rebuilds and vacuums and journal `0029` with its superseded names. | - -Run it as the migration role: maintenance needs `pg_maintain`, which the application roles lack. Run -it outside peak traffic, and run `--maintenance` in the quietest window you have: concurrent HNSW -rebuilds are long and write a lot of WAL (GitLab, for example, schedules automatic reindexing for -weekends). Keep one run at a time. - -**Pausing.** Ctrl-C is safe at any point. The in-flight page rolls back with its cursor, and an -interrupted concurrent rebuild's leftover index is removed on the next run. Rerun the same command to -resume; completed pages stay committed. - -**Watching.** Every ten pages the run logs the phase, cursor, rows mutated so far and current row -limit; it also logs each halving after a slow page, each phase change, and the start of the completion -recheck. In PostgreSQL, watch for `checkpoint starting: wal` in quick succession, slow checkpoint -sync times, `canceling wait for synchronous replication`, and replica lag. If they appear, stop the -run and resume later with a higher `--pause-ratio` or lower `--max-rows`. - -```sql -SELECT * FROM search_embedding_cleanup_progress; -SELECT name, applied_at FROM script_migrations -WHERE name IN ('0027_retire_search_embeddings', '0028_maintain_search_retirement', - '0029_retire_all_search_embeddings'); -``` - -A plain run does not journal anything; only a `--maintenance` run that finishes records `0029` and its -superseded names. On upgrading a legacy single-KB checkpoint, the snapshot and cursor reset commit -atomically. The scan starts at the beginning once so it includes other KBs behind the old cursor; -previous deletes remain committed. Maintenance checkpoints also reset once because the expanded -cleanup creates new dead entries. A completed legacy checkpoint does not require its former KB to -still exist or remain Search-marked; the new snapshot selects current Search KBs and preserves any KB -now marked ordinary. An unfinished legacy checkpoint still requires its target to remain -Search-marked. - -## How a run paces itself - -Each page mutates at most a row limit of target rows and reads at most four IDs per row of that -limit, never more than 25,000 IDs. Pages execute -sequentially, and each is followed by a pause of `--pause-ratio` times its duration, up to one -minute. Because a page is timed through its commit, a slow synchronous replica or a checkpoint stall -lengthens the following pause by the same factor. Retiring a document is a non-HOT update that writes every index on -`document`, and deleting a chunk cascades into its projections, so a page's cost follows the target -rows it mutates, not the IDs it reads. A page that reaches the row limit advances the cursor only to -its last mutated row; the rest of its scan is read again by the next page. Tying the scan window to -the limit keeps that re-reading proportional to the work, even after the limit shrinks. Documents -that are already retired never count against the limit. - -The row limit starts at 2,000 rows, or `--max-rows` if lower. A page is timed from the start of its transaction through its -commit, including the synchronous-replication wait and any lock-timeout retries. A page slower than -30 seconds halves the limit. A fast page, one under 7.5 seconds, doubles it up to `--max-rows`, which -also widens the scan window, so sparse stretches are not crawled in small windows. The limit never drops below 25 rows. Phase changes do not adjust it. - -Materialized SQL pages keep the IDs inside PostgreSQL; the migration process receives only a cursor -and a validation result. Each page uses a two-minute statement timeout and a one-second lock -timeout. If a page's mutating statement exceeds the statement timeout, the page rolls back with its -cursor and is retried with half the row limit after the usual pause. From then on, fast pages grow -the limit only up to that halved size, so a size that timed out is never tried again. A page that -still times out at 25 rows fails the migration. Any other statement timeout fails the migration at once, because a smaller page cannot -speed it up. The completion rechecks, which walk every captured KB once, run with a 30-minute -timeout. Brief lock timeouts retry the rolled-back page with bounded backoff for up to one minute. -Other errors, or exhausted lock retries, fail the migration without a completion receipt. - -The runner-owned `search_embedding_cleanup_targets` table stores the frozen KB set, populated in -bounded SQL pages within one repeatable-read transaction. The existing `search_embedding_cleanup_progress` -row stores the shared phase and ID cursor; its legacy `knowledge_base_id` remains an informational -anchor, not the full deletion scope. Page mutations and cursor advancement commit together. -The one-off migration journal records only completion. `db:push` excludes both bookkeeping tables -from schema diffing. - -The documents phase fences queued and in-flight processing by marking only target documents -excluded/disabled and clearing their dispatch stamps. The embeddings phase deletes only target -chunks, and refuses a page whose target document was not retired or has inconsistent ownership. -The existing foreign keys cascade to vector/keyword projections and chunk provenance. Both phases -walk the primary key once for the entire captured set in bounded pages; ordinary rows are never -updated. Each page locks and rechecks the Search markers for its target KBs before mutation, and -fails atomically if any target changed to an ordinary KB. This avoids a separate full-table scan -per KB or sorting a whole KB when no suitable composite cleanup index exists. Resumption continues the saved scan, including across pages containing only -unrelated rows. Before completion, the cleanup checks for unretired documents and remaining chunks -behind either cursor and restarts the affected phase if needed. A final bounded pass validates all -captured KB markers, including empty KBs and KBs whose rows were already scanned, holding shared -marker locks until the completion checkpoint commits. Resuming a completed cleanup before maintenance -also revalidates the captured set, with the same 30-minute timeout as the completion recheck. Keep target writers stopped and -do not change their Search markers during the pass. - -Inspect progress with: - -```sql -SELECT * FROM search_embedding_cleanup_progress; -SELECT name, applied_at FROM script_migrations -WHERE name IN ('0027_retire_search_embeddings', '0028_maintain_search_retirement', - '0029_retire_all_search_embeddings'); -``` - -After completion, check the selected KBs have no `embedding` rows, verify their live Search, and verify -ordinary KB retrieval. A zero-row absence check may still scan index entries; use an appropriate -timeout. The cleanup is destructive and not reversible by flipping the search flag. Re-enabling -indexed Search requires deliberately restoring document eligibility and fully resyncing its sources. - -## Storage maintenance - -With `--maintenance`, after deletion, `0029` invokes the existing maintenance implementation to run `REINDEX INDEX CONCURRENTLY` on each HNSW index of `embedding_search`, -then `VACUUM (ANALYZE, TRUNCATE FALSE)` on the vector and keyword projections, chunk provenance, -embeddings, and documents. These operations execute sequentially outside transactions. Rebuilds -keep ordinary reads and writes available and require temporary index space and WAL capacity. -They wait for older transactions and can dominate total runtime. PostgreSQL's `pg_stat_progress_create_index` and `pg_stat_progress_vacuum` expose progress. - -The existing progress row gains `reindexed_through` and `vacuumed_tables` checkpoints. Completed -indexes and tables are skipped on retry; interruption between an operation and its checkpoint may -repeat that one operation. Invalid `_ccnew`/`_ccold` siblings from an interrupted concurrent rebuild -are removed concurrently before retrying their original index. One advisory lock serializes the -maintenance worker. Do not run other index maintenance on these tables at the same time. - -Ordinary vacuum makes dead space reusable and refreshes planner statistics; it generally does not -shrink table files. This migration does not run `VACUUM FULL` or rewrite tables. See the -[PostgreSQL vacuum documentation](https://www.postgresql.org/docs/17/sql-vacuum.html), -[concurrent reindex recovery](https://www.postgresql.org/docs/17/sql-reindex.html#SQL-REINDEX-CONCURRENTLY), -and [pgvector maintenance guidance](https://github.com/pgvector/pgvector#vacuuming). - -Document counts are historical until a later document cleanup. Do not raw-delete documents or -bucket objects: their application hard-delete path also enqueues identity-bound storage cleanup -and applies accounting. Its ordinary scoped mode excludes retired documents, so a follow-up must -explicitly support these rows while preserving those side effects. Do not delete source accounts, -integration policies or permission grants used by live Search. - -## Why it runs this way - -Long data changes belong outside deploy migrations, in batches, throttled on database health, and -resumable from a cursor: - -- [strong_migrations: Backfilling data](https://github.com/ankane/strong_migrations#backfilling-data) -- [GitLab batched background migrations](https://docs.gitlab.com/development/database/batched_background_migrations/) - and [automatic reindexing](https://docs.gitlab.com/omnibus/settings/database/) -- [Shopify maintenance_tasks](https://github.com/Shopify/maintenance_tasks) -- [gh-ost throttling](https://github.com/github/gh-ost/blob/master/doc/throttle.md) and - [pt-online-schema-change](https://docs.percona.com/percona-toolkit/pt-online-schema-change.html) -- [Stripe: online migrations at scale](https://stripe.com/blog/online-migrations) -- PostgreSQL 17: [WAL configuration](https://www.postgresql.org/docs/17/wal-configuration.html), - [synchronous replication](https://www.postgresql.org/docs/17/warm-standby.html#SYNCHRONOUS-REPLICATION), - [replication statistics](https://www.postgresql.org/docs/17/monitoring-stats.html) -- [PlanetScale: the only scalable delete](https://planetscale.com/blog/the-only-scalable-delete) +The old `0027` CLI is disabled. Its `--maintenance` path performed large concurrent HNSW rebuilds +without workload-health gates. The historical implementations and receipts remain for compatibility; +none of `0027`–`0029` is registered in deployment migrations. + +Use the [operator-run replacement workflow](../maintenance/search-retirement.md). It builds empty +HNSW indexes first, copies ordinary-KB vectors in bounded pages, validates and cuts over separately, +then slowly removes Search chunks. It never invokes `0028` or `REINDEX`/`VACUUM FULL`. + +Existing `search_embedding_cleanup_progress`, `search_embedding_cleanup_targets`, and migration +receipts are retained. Already-deleted chunks need no further work. The replacement reconstructs +ordinary vectors from canonical embeddings, including deferred projections. It captures the current +Search-marked KB set anew and rechecks markers before mutation; live KB/source configuration survives. + +Do not run older cleanup binaries alongside the replacement. Stop old workers and maintenance jobs +before preparing it. Its guard also prevents an existing legacy cursor from advancing while the +replacement exists. diff --git a/packages/db/scripts/push.test.ts b/packages/db/scripts/push.test.ts index acca74d8f55..c1e95771e60 100644 --- a/packages/db/scripts/push.test.ts +++ b/packages/db/scripts/push.test.ts @@ -33,11 +33,6 @@ afterEach(() => { }) describe('db:push policy and process boundaries', () => { - it('does not implicitly approve data loss', async () => { - expect(await runPush([])).toBe(0) - expect(spawn.mock.calls[0][0]).not.toContain('--force') - }) - it('rejects interactive renames without a terminal before any database commands', async () => { expect(await runPush(['--interactive-renames', '--force'])).toBe(1) expect(spawn).not.toHaveBeenCalled() diff --git a/packages/db/scripts/push.ts b/packages/db/scripts/push.ts index 2d10c1c4e0e..16fbf5e6ba9 100644 --- a/packages/db/scripts/push.ts +++ b/packages/db/scripts/push.ts @@ -1,4 +1,6 @@ import { createLogger } from '@sim/logger' +import { getPostgresErrorCode } from '@sim/utils/errors' +import postgres, { type Sql } from 'postgres' const logger = createLogger('DatabasePush') const RECONCILIATION_COMMANDS = [ @@ -12,6 +14,39 @@ const RECONCILIATION_COMMANDS = [ ['bun', '--env-file=.env', 'run', './script-migrations/0026_user_table_schema_for_write.ts'], ] +/** Historical push reconcilers recreate retired projections and cannot run after replacement starts. */ +async function retirementAllowsPush(): Promise { + const url = process.env.DATABASE_URL + if (!url) { + logger.error('DATABASE_URL is required for schema push') + return false + } + let sql: Sql | undefined + try { + sql = postgres(url, { + max: 1, + prepare: false, + connect_timeout: 5, + connection: { statement_timeout: 2_000, lock_timeout: 100 }, + onnotice: () => undefined, + }) + const [state] = await sql<{ present: boolean }[]>` + SELECT to_regclass('public.search_retirement_state') IS NOT NULL AS present` + if (state.present) { + logger.error( + 'Schema push is disabled after Search retirement starts, including completed retirement. Use reviewed versioned migrations; push reconcilers would restore retired projections' + ) + return false + } + return true + } catch (error) { + logger.error('Unable to verify schema-push safety', { code: getPostgresErrorCode(error) }) + return false + } finally { + await sql?.end({ timeout: 1 }).catch(() => undefined) + } +} + /** * Push treats additions and removals as distinct objects by default. The pinned * Drizzle patch reads this policy only in the push subprocess; generation keeps @@ -31,6 +66,8 @@ export async function runPush(args: string[]): Promise { return 1 } + if (!help && !(await retirementAllowsPush())) return 1 + if (!help && args.includes('--force')) { const preparation = Bun.spawn(['bun', '--env-file=.env', 'run', './scripts/prepare-push.ts'], { stdin: 'inherit', diff --git a/packages/db/scripts/retire-indexed-search.ts b/packages/db/scripts/retire-indexed-search.ts new file mode 100644 index 00000000000..77318201e32 --- /dev/null +++ b/packages/db/scripts/retire-indexed-search.ts @@ -0,0 +1,208 @@ +import { createHash } from 'node:crypto' +import { parseArgs } from 'node:util' +import { + abortSearchRetirement, + advanceSearchRetirement, + beginSearchRetirementPurge, + cutoverSearchRetirement, + finalizeSearchRetirement, + getSearchRetirementStatus, + initializeSearchRetirement, + RetireSearchHealthError, + readSearchRetirementHealth, + readSearchRetirementHealthLimits, + SearchRetirementError, +} from '@sim/db/maintenance' +import { createLogger, LogLevel } from '@sim/logger' +import { getPostgresErrorCode } from '@sim/utils/errors' +import { sleep } from '@sim/utils/helpers' +import postgres from 'postgres' + +class RetirementCommandError extends Error {} + +const logger = createLogger('SearchDataRetirement', { enabled: true, logLevel: LogLevel.INFO }) +const HELP = `Operator-only indexed Search retirement. No work runs on deployment. + +bun --no-env-file packages/db/scripts/retire-indexed-search.ts [options] + +Commands: + identity Print the non-secret connection fingerprint for the health policy/collector. + status Read progress; never initialize or resume work. + prepare Create empty indexed replacement and capture writes (requires --ack-release-drained). + run Advance bounded pages; stops at each manual gate (default: one page). + cutover Attempt NOWAIT swap after validation (requires --ack-release-drained). + begin-purge Retire the backup and allow chunk deletion (requires --ack-retire-backup). + finalize Remove capture after deletion and a separate DDL health clearance. + abort Remove capture and discard the replacement before cutover. + +Every write except abort requires --health-file PATH --health-policy PATH. +Policy databaseId must equal the identity command's fingerprint. +Run options: --page-size 25 (1–100), --pages 1 (1–120), --seconds 60 (1–600). +A fixed pause of at least 5 seconds follows each page. Timeouts stop; no automatic retry. +Connection: MIGRATION_DATABASE_URL, direct primary or session pooling only. +See packages/db/maintenance/search-retirement.md before use.` + +function integerOption(value: string | undefined, fallback: number, ceiling: number): number { + const parsed = value === undefined ? fallback : Number(value) + if (!Number.isInteger(parsed) || parsed < 1 || parsed > ceiling) { + throw new RetirementCommandError(`Expected an integer between 1 and ${ceiling}`) + } + return parsed +} + +async function main(): Promise { + const { values, positionals } = parseArgs({ + args: process.argv.slice(2), + allowPositionals: true, + options: { + help: { type: 'boolean', default: false }, + 'health-file': { type: 'string' }, + 'health-policy': { type: 'string' }, + 'page-size': { type: 'string' }, + pages: { type: 'string' }, + seconds: { type: 'string' }, + 'ack-release-drained': { type: 'boolean', default: false }, + 'ack-retire-backup': { type: 'boolean', default: false }, + }, + }) + if (values.help || positionals.length === 0) { + process.stdout.write(`${HELP}\n`) + return + } + const [command] = positionals + if ( + positionals.length !== 1 || + ![ + 'identity', + 'status', + 'prepare', + 'run', + 'cutover', + 'begin-purge', + 'finalize', + 'abort', + ].includes(command) + ) { + throw new RetirementCommandError('Unknown command; use --help') + } + const rawUrl = process.env.MIGRATION_DATABASE_URL + if (!rawUrl) + throw new RetirementCommandError( + 'MIGRATION_DATABASE_URL is required; there is no application DSN fallback' + ) + const url = new URL(rawUrl) + if (!['postgres:', 'postgresql:'].includes(url.protocol)) + throw new RetirementCommandError('Expected a PostgreSQL URL') + const databaseId = createHash('sha256') + .update( + JSON.stringify([url.hostname.toLowerCase(), url.port || '5432', url.pathname, url.username]) + ) + .digest('hex') + if (command === 'identity') { + process.stdout.write(`${databaseId}\n`) + return + } + if (url.hostname.endsWith('.pg.psdb.cloud') && url.port === '6432') { + throw new RetirementCommandError( + 'PlanetScale transaction pooling is unsupported; use a direct primary connection' + ) + } + const pageSize = integerOption(values['page-size'], 25, 100) + const pages = integerOption(values.pages, 1, 120) + const budgetMs = integerOption(values.seconds, 60, 600) * 1_000 + if (['prepare', 'cutover'].includes(command) && !values['ack-release-drained']) { + throw new RetirementCommandError( + 'Confirm the code-removal release is fully deployed and old workers/maintenance jobs drained with --ack-release-drained' + ) + } + if (command === 'begin-purge' && !values['ack-retire-backup']) { + throw new RetirementCommandError( + 'Confirm ordinary KB retrieval and live Search after cutover with --ack-retire-backup' + ) + } + let checkHealth = async () => {} + if (!['status', 'abort'].includes(command)) { + const healthFile = values['health-file'] + const policyFile = values['health-policy'] + if (!healthFile || !policyFile) + throw new RetirementCommandError('--health-file and --health-policy are required') + const limits = await readSearchRetirementHealthLimits(policyFile) + if (limits.databaseId !== databaseId) { + throw new RetirementCommandError('Health policy does not match this database connection') + } + checkHealth = async () => { + await readSearchRetirementHealth(healthFile, limits, { + cutover: ['cutover', 'begin-purge', 'finalize'].includes(command), + }) + } + await checkHealth() + } + const sql = postgres(rawUrl, { + max: 1, + prepare: false, + connect_timeout: 5, + max_lifetime: null, + connection: { + application_name: 'sim-search-data-retirement', + statement_timeout: 2_000, + lock_timeout: 100, + idle_in_transaction_session_timeout: 5_000, + }, + onnotice: () => undefined, + }) + try { + if (command === 'status') { + logger.info('Search retirement status', { progress: await getSearchRetirementStatus(sql) }) + return + } + const [{ writable }] = await sql<{ writable: boolean }[]>` + SELECT NOT pg_is_in_recovery() AND current_setting('transaction_read_only') = 'off' AS writable` + if (!writable) + throw new RetirementCommandError('Maintenance requires a writable primary connection') + const [{ acquired }] = await sql<{ acquired: boolean }[]>` + SELECT pg_try_advisory_lock(hashtextextended('sim:search-retirement-maintenance', 0)) AS acquired` + if (!acquired) + throw new RetirementCommandError('Another retirement or index-maintenance command is running') + await checkHealth() + if (command === 'prepare') await initializeSearchRetirement(sql) + else if (command === 'cutover') await cutoverSearchRetirement(sql) + else if (command === 'begin-purge') await beginSearchRetirementPurge(sql) + else if (command === 'finalize') await finalizeSearchRetirement(sql) + else if (command === 'abort') await abortSearchRetirement(sql) + else { + const started = performance.now() + for (let page = 0; page < pages && performance.now() - started < budgetMs; page++) { + await checkHealth() + const before = performance.now() + const progress = await advanceSearchRetirement(sql, { pageSize }) + logger.info('Search retirement page committed', { page: page + 1, progress }) + if (['ready', 'cutover', 'finalize', 'done'].includes(progress.phase)) break + await sleep(Math.max(5_000, (performance.now() - before) * 9)) + } + } + logger.info('Search retirement command finished', { + progress: await getSearchRetirementStatus(sql), + }) + } finally { + await sql.end({ timeout: 5 }) + } +} + +try { + await main() +} catch (error) { + // PostgreSQL errors may contain row values or credentials; never log their query, detail or stack. + const code = getPostgresErrorCode(error) + logger.error('Search retirement stopped; committed pages remain resumable', { + reason: + error instanceof RetireSearchHealthError || + error instanceof RetirementCommandError || + error instanceof SearchRetirementError + ? error.message + : code + ? 'Database refused the operation' + : 'Preflight or command failed; check the runbook and options', + ...(code ? { code } : {}), + }) + process.exitCode = 1 +} diff --git a/packages/db/vitest.config.ts b/packages/db/vitest.config.ts index 8ab9cee962d..f8159f6b59c 100644 --- a/packages/db/vitest.config.ts +++ b/packages/db/vitest.config.ts @@ -14,7 +14,12 @@ export default defineConfig(({ mode }) => { test: { include: integration ? ['**/*.integration.ts'] - : ['scripts/**/*.test.ts', 'script-migrations/**/*.test.ts', '*.test.ts'], + : [ + 'scripts/**/*.test.ts', + 'script-migrations/**/*.test.ts', + 'maintenance/**/*.test.ts', + '*.test.ts', + ], ...(integration && { setupFiles: ['./vitest.integration.setup.ts'] }), }, }) From ce207dd9787e001f56cd0f5d6bf5782212177a6c Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 12:08:04 -0700 Subject: [PATCH 2/4] chore(search): keep retirement usage in the CLI --- packages/db/maintenance/search-retirement.md | 194 ------------------ packages/db/script-migrations/index.ts | 2 +- .../search-embedding-retirement.md | 18 -- packages/db/scripts/retire-indexed-search.ts | 24 ++- 4 files changed, 22 insertions(+), 216 deletions(-) delete mode 100644 packages/db/maintenance/search-retirement.md delete mode 100644 packages/db/script-migrations/search-embedding-retirement.md diff --git a/packages/db/maintenance/search-retirement.md b/packages/db/maintenance/search-retirement.md deleted file mode 100644 index 9a21e098823..00000000000 --- a/packages/db/maintenance/search-retirement.md +++ /dev/null @@ -1,194 +0,0 @@ -# Retire indexed Search data without a bulk rebuild - -This is an **operator-run workflow**, outside `db:migrate`, `db:push`, deployment jobs, cron, and -Trigger.dev. Merging or deploying the code starts no copy, deletion, vacuum, or reindex. Deploy the -indexed-Search code removal first, verify every app/worker uses live Search, and drain old indexing, -projection-maintenance, and cleanup commands. Do not roll back to an indexed-Search release during -or after retirement. - -The command replaces only `embedding_search`. It reconstructs ordinary-KB candidate vectors from -canonical `embedding` rows, including rows missing from the old deferred projection. Full-precision -vectors, content, keyword search, document ACLs, provenance and table identities used by ordinary KBs -remain in place. Search KB shells and source/credential configuration remain because live Search -uses them. Search chunks are deleted only after replacement cutover and a separate approval to -retire the old projection. Search documents are excluded/disabled and dispatch tokens cleared; -documents/files are not hard-deleted. Their eventual deletion must use the storage-outbox/accounting -path, not raw SQL. - -This reduces risk; it cannot guarantee zero latency impact. Shadow inserts still build HNSW links, -consume CPU/I/O, and emit replicated WAL. Change capture adds a small write to each affected chunk -transaction while copying. Relation DDL can briefly exclude conflicting operations, and replayed DDL -can conflict with replica queries. Health sampling cannot react to a spike that begins inside a page. -Start with a canary page and prioritize serving traffic over migration speed. - -## Release and operator sequence - -1. **Deploy the code-removal PR first.** Deploy this stacked PR only after its parent. The command - requires an explicit `--ack-release-drained` for preparation and cutover. It cannot prove which - application binaries or external jobs remain running; verify that outside PostgreSQL. -2. **Rehearse on a disposable database**, using the current schema and realistic retained-vector - widths. Run the integration suite, then ordinary-KB retrieval and live Search smoke tests against - the staged application. Configure a dedicated maintenance role and primary direct/session-pooled - connection. PostgreSQL 17+ is required for the transaction deadline. Never use transaction pooling. -3. **Configure health collection and disk/WAL headroom.** Feed fresh primary, every serving replica, - and application SLO observations into the health file described below. Establish normal baselines; - choose numeric limits from those baselines and provision space for old and new projections plus - WAL. Merely having space for the final table is insufficient. Do not refresh a stale metric's - timestamp or manufacture healthy samples. No valid fresh sample means no maintenance. -4. **Prepare and canary.** `prepare` creates a logged empty replacement with the six shared HNSW - indexes already present, installs capture, and creates durable state. `run` advances one small page - by default. Inspect application latency, query errors, CPU/I/O, WAL and every replica after the - first pages. Leave the health collector's maintenance switch off when attention or headroom is - unavailable. Increase only the number of scheduled bounded invocations once impact is acceptable. -5. **Copy, catch up and verify.** Continue `run` until `ready`. Each page commits with its cursor; - retries resume committed work. Validation compares retained identities, enabled state, binary - compatibility and vector values in both directions. Concurrent canonical writes enqueue IDs with - generations; catchup acknowledges only the generation it processed. Metadata changes during - construction invalidate the copy instead of silently changing its scope. Abort and prepare again - after investigating. Do not change the schema or run other index maintenance during this workflow. -6. **Cut over explicitly.** First verify replicas have replayed the copy and drain old snapshots on - primary and replicas, or divert replica reads and wait for existing transactions to drain. The - collector's separate `cutoverAllowed` must attest this. Run a fresh bounded catchup page if changes - remain, then `cutover`. It checks relation ownership/privileges, known writer triggers, dependencies - and indexes, and uses `LOCK ... NOWAIT`. It refuses busy tables or old primary transactions rather - than cancelling traffic or waiting in a DDL lock queue. Refusal is expected on busy systems: stop, - inspect, and retry in a suitable window. Do not automatically loop cutover attempts. -7. **Observe before allowing deletion.** Verify ordinary-KB semantic/keyword retrieval, document ACL - denials, connector ingestion/deferred repair, live Search, and error/replica metrics. The retained - old projection is an observation backup, **not an instant rollback**: new writes target the new - table. Never swap a stale backup back into service. A rollback requires a separately validated - reconstruction from canonical embeddings, which remain intact at this point. -8. **Retire the backup, then purge slowly.** `begin-purge --ack-retire-backup` requires separate - cutover clearance. It drops the old projection with `RESTRICT` before canonical chunk deletion, - so its outgoing FK cannot maintain the old HNSW graph on each delete. Continue bounded `run` - invocations until `finalize`. Every destructive page locks/rechecks Search markers. Ordinary rows - and live configuration are preserved. Existing keyword/provenance FKs still perform bounded - per-chunk cleanup. These indexes have a cost; leave telemetry gates enabled throughout. -9. **Finalize explicitly.** Clear another primary/replica DDL window and run `finalize`. It removes - the capture triggers only after checking for late writes. If it returns `purge`, resume bounded - pages and revisit this gate. Backup removal, abort and finalization may require implicit PostgreSQL - lock upgrades: those waits are capped at 1 ms, but a brief lock queue is still possible. -10. **Verify completion.** Inspect the durable status, smoke-test retrieval again, and observe normal - autovacuum/replica recovery. The fresh vector projection needs no final rebuild. Ordinary vacuum - can reuse dead canonical/keyword heap space; it does not promise to shrink files. This command - does not run `REINDEX`, `VACUUM FULL`, or an unbounded final absence scan. - -## Commands - -Supply the migration writer DSN through `MIGRATION_DATABASE_URL` in the process environment. There -is no fallback to the app DSN. The command uses one connection with -`application_name=sim-search-data-retirement`; configure a provider Traffic Control budget for that -application name when available. Keep credentials, provider identifiers, policy and health files -outside the repository. - -```sh -bun --no-env-file packages/db/scripts/retire-indexed-search.ts --help -bun --no-env-file packages/db/scripts/retire-indexed-search.ts identity -bun --no-env-file packages/db/scripts/retire-indexed-search.ts status - -bun --no-env-file packages/db/scripts/retire-indexed-search.ts prepare \ - --ack-release-drained --health-file /secure/health.json --health-policy /secure/policy.json - -# One page by default. More pages remain bounded and health is checked before each one. -bun --no-env-file packages/db/scripts/retire-indexed-search.ts run \ - --health-file /secure/health.json --health-policy /secure/policy.json --pages 10 --seconds 60 - -# Separate operational decisions; never put these in an automatic run loop. -bun --no-env-file packages/db/scripts/retire-indexed-search.ts cutover \ - --ack-release-drained --health-file /secure/health.json --health-policy /secure/policy.json -bun --no-env-file packages/db/scripts/retire-indexed-search.ts begin-purge \ - --ack-retire-backup --health-file /secure/health.json --health-policy /secure/policy.json -bun --no-env-file packages/db/scripts/retire-indexed-search.ts finalize \ - --health-file /secure/health.json --health-policy /secure/policy.json - -# Before cutover only: stop capture and discard this replacement after a NOWAIT lock attempt. -bun --no-env-file packages/db/scripts/retire-indexed-search.ts abort -``` - -`status` is read-only and never initializes work. Ctrl-C or connection loss leaves committed pages -intact; an interrupted transaction rolls back its writes and cursor together. Any database error, -health refusal, or capacity throttle stops the invocation with a nonzero exit code. No automatic -retry widens a page or raises a timeout. Check status before retrying an ambiguous connection loss. - -The default page is 25 source rows; copying/validation can be explicitly lowered or raised to at -most 100. Destructive pages never exceed 25 source rows. IDs/vectors remain in PostgreSQL except a -bounded queue page of IDs/generations. The CLI runs at most 120 pages/600 seconds per invocation, -with at least five seconds and nine times the previous page duration between pages. The time budget -stops *starting* pages; the last transaction and cooldown can finish afterward. Core transactions -have short statement, lock, and transaction deadlines. There is no dynamic batch growth. A metadata -swap does no bulk data work while it holds locks. - -Pausing the worker does not stop change capture. If a pause will be long, or the dirty queue itself -causes pressure, use `abort` before cutover. Capture must never reject application writes merely -because its queue grew. Monitor its storage too. A dedicated Traffic Control budget is an additional -backstop; an external scheduler may repeat bounded `run` commands, but must never repeat cutover or -begin-purge or finalize automatically. Trigger.dev is optional orchestration, not a substitute for these guards. - -## Health contract - -The repository deliberately has **no fabricated provider-health fallback**. A trusted external -collector must atomically replace an at-most-8-KiB UTF-8 JSON file. Its observation time is the oldest -underlying required metric timestamp, not the time it wrote the file. Aggregate worst lag/CPU across -all relevant nodes and minimum free storage. Require application SLO/error checks as part of -`healthy`; a successful database ping or static `/api/health` response is insufficient. - -Required sample fields: - -| Field | Meaning | -| --- | --- | -| `observedAt` | UTC ISO timestamp, seconds or three-digit milliseconds, ending `Z` | -| `databaseId` | Output of `identity`, SHA-256 of endpoint/port/database/role, excluding password | -| `healthy` | Collector verified the expected node inventory and application SLOs | -| `maintenanceAllowed` | Operator/alert kill switch; false stops the next page | -| `cutoverAllowed` | Separate confirmation that serving replicas and old snapshots are safe for DDL; normally false | -| `replicaLagBytes`, `replicaLagSeconds` | Worst replay backlog/lag across every serving replica | -| `walBytesPerSecond` | Measured WAL generation rate over a defined recent interval | -| `databaseP95Ms`, `cpuPercent`, `freeStorageBytes` | Current workload latency, worst CPU, and minimum available storage | - -The policy file has exactly `databaseId`, `maxReplicaLagBytes`, `maxReplicaLagSeconds`, -`maxWalBytesPerSecond`, `maxDatabaseP95Ms`, `maxCpuPercent`, `minFreeStorageBytes`, and -`maxSampleAgeMs`. Supply finite numeric limits chosen for the deployment, not strings. Lag limits may -be zero; other limits must be positive, CPU at most 100, and sample age at most 30,000 milliseconds. -Missing/unknown fields, nonfinite/negative metrics, wrong identity, stale/future observations, -unhealthy/refused maintenance, or exceeded limits fail closed. There is no `--force` bypass. - -PlanetScale's metrics API exposes replica lag, retained WAL, CPU and query latency/error series, but -collector mapping and freshness must be verified against the actual deployment. Retained WAL bytes -are **not** automatically replica replay backlog or WAL generation rate. If telemetry resolution or -permissions cannot satisfy the contract, keep maintenance paused and improve collection; do not fill -missing measurements with zero. Primary transaction visibility requires sufficient statistics -permissions; cutover refuses to infer safety from a partially visible activity view. - -## Compatibility and follow-up contracts - -The replacement retains nullable legacy source/ACL and binary columns because the deployed shared -projector still writes them. It preserves the ordinary shared vector indexes and matching query -shape, omitting source-specific/ACL indexes from the replacement. The old primary table and its -unused indexes are reclaimed together when the operator ends backup retention. - -This workflow leaves legacy keyword tables, activity tables, shared dirty-queue state and historical -migration receipts in the schema. Search chunk deletes remove their referencing projection rows; -removing the empty tables, obsolete trigger installers and compatible columns is a later schema -contract after all remaining shared writers are removed. The complete inventory is in -[Indexed Search retirement inventory](../script-migrations/indexed-search-retirement.md). Never drop -canonical document ACLs or ordinary-KB content projections just because indexed Search is retired. - -The runner owns `search_retirement_*` and `embedding_search_retirement_*`; Drizzle push excludes -those objects. `db:push` refuses databases with retirement state, including completed jobs, because -its historical reconcilers would recreate the retired writers and backfills. Use reviewed versioned -migrations after starting retirement; do not delete the receipt to bypass this guard. Existing `search_embedding_cleanup_*` checkpoints remain unchanged. Its progress-row -fence prevents the old cursor command from advancing while replacement work exists. Stop old -maintenance binaries first; this cannot prevent an arbitrary operator from issuing SQL independently. - -## Verification and references - -`packages/db/maintenance/search-retirement.integration.ts` exercises the real PostgreSQL boundary. -The integration reporter writes to `INTEGRATION_REPORT_PATH` (default -`test-results/integration.json`, uploaded by CI). The health boundary has separate invalid-input and -admission tests. Rehearsal against the running app and production-shaped workload remains an operator -prerequisite, not a claim made by these small fixtures. - -- [pgvector HNSW](https://github.com/pgvector/pgvector#hnsw): empty indexes are supported; incremental building trades throughput for bounded work. -- [PostgreSQL locks](https://www.postgresql.org/docs/17/sql-lock.html) and [MVCC caveats](https://www.postgresql.org/docs/17/mvcc-caveats.html): fail-fast locking and old snapshots both matter. -- [Hot standby conflicts](https://www.postgresql.org/docs/17/hot-standby.html#HOT-STANDBY-CONFLICT): primary DDL may affect replica queries. -- [PlanetScale metrics API](https://planetscale.com/docs/api/reference/get_branch_metrics) and [Traffic Control](https://planetscale.com/docs/postgres/traffic-control/concepts): external observations and supplementary workload budgets. diff --git a/packages/db/script-migrations/index.ts b/packages/db/script-migrations/index.ts index 7700a8f80a3..084c707c333 100644 --- a/packages/db/script-migrations/index.ts +++ b/packages/db/script-migrations/index.ts @@ -60,7 +60,7 @@ export const scriptMigrations: readonly ScriptMigration[] = [ userTableSchemaForWriteMigration, /** * Search retirement (0027–0029) is an operator-run maintenance command, not a deploy step: - * see `search-embedding-retirement.md`. + * run `packages/db/scripts/retire-indexed-search.ts --help` for usage. */ ] diff --git a/packages/db/script-migrations/search-embedding-retirement.md b/packages/db/script-migrations/search-embedding-retirement.md deleted file mode 100644 index bef76e1feed..00000000000 --- a/packages/db/script-migrations/search-embedding-retirement.md +++ /dev/null @@ -1,18 +0,0 @@ -# Retiring legacy Search indexes - -The old `0027` CLI is disabled. Its `--maintenance` path performed large concurrent HNSW rebuilds -without workload-health gates. The historical implementations and receipts remain for compatibility; -none of `0027`–`0029` is registered in deployment migrations. - -Use the [operator-run replacement workflow](../maintenance/search-retirement.md). It builds empty -HNSW indexes first, copies ordinary-KB vectors in bounded pages, validates and cuts over separately, -then slowly removes Search chunks. It never invokes `0028` or `REINDEX`/`VACUUM FULL`. - -Existing `search_embedding_cleanup_progress`, `search_embedding_cleanup_targets`, and migration -receipts are retained. Already-deleted chunks need no further work. The replacement reconstructs -ordinary vectors from canonical embeddings, including deferred projections. It captures the current -Search-marked KB set anew and rechecks markers before mutation; live KB/source configuration survives. - -Do not run older cleanup binaries alongside the replacement. Stop old workers and maintenance jobs -before preparing it. Its guard also prevents an existing legacy cursor from advancing while the -replacement exists. diff --git a/packages/db/scripts/retire-indexed-search.ts b/packages/db/scripts/retire-indexed-search.ts index 77318201e32..b477b11a426 100644 --- a/packages/db/scripts/retire-indexed-search.ts +++ b/packages/db/scripts/retire-indexed-search.ts @@ -39,8 +39,26 @@ Every write except abort requires --health-file PATH --health-policy PATH. Policy databaseId must equal the identity command's fingerprint. Run options: --page-size 25 (1–100), --pages 1 (1–120), --seconds 60 (1–600). A fixed pause of at least 5 seconds follows each page. Timeouts stop; no automatic retry. -Connection: MIGRATION_DATABASE_URL, direct primary or session pooling only. -See packages/db/maintenance/search-retirement.md before use.` +Connection: MIGRATION_DATABASE_URL, PostgreSQL 17+, direct primary or session pooling only. + +Deploy the code-removal release and drain old workers first. Then: + prepare -> run until ready -> cutover -> verify retrieval -> begin-purge + -> run until finalize -> finalize. Repeat run if late writes return the phase to purge. +Reads use the old projection until cutover. The retained backup is not an instant rollback. +Cutover requires pg_read_all_stats and no old snapshots; never automatically retry DDL gates. +Abort, backup removal and finalization cap implicit DROP lock waits at 1 ms. +After retirement starts, use versioned migrations; db:push would restore old projection machinery. + +Health files are UTF-8 JSON, at most 8 KiB, atomically refreshed by trusted telemetry. +Sample fields: observedAt (oldest metric timestamp, UTC ISO), databaseId, healthy, + maintenanceAllowed, cutoverAllowed, replicaLagBytes, replicaLagSeconds, + walBytesPerSecond, databaseP95Ms, cpuPercent, freeStorageBytes. +Policy fields: databaseId, maxReplicaLagBytes, maxReplicaLagSeconds, maxWalBytesPerSecond, + maxDatabaseP95Ms, maxCpuPercent, minFreeStorageBytes, maxSampleAgeMs (at most 30000). +Choose limits from actual capacity and latency requirements; missing/stale telemetry stops work. +Use worst replica lag and CPU, and minimum free storage; healthy includes application error/latency checks. +Set cutoverAllowed only after primary and replica snapshots are clear for DDL. +Pacing limits load but cannot eliminate latency or replica-conflict risk.` function integerOption(value: string | undefined, fallback: number, ceiling: number): number { const parsed = value === undefined ? fallback : Number(value) @@ -201,7 +219,7 @@ try { ? error.message : code ? 'Database refused the operation' - : 'Preflight or command failed; check the runbook and options', + : 'Preflight or command failed; use --help to check options', ...(code ? { code } : {}), }) process.exitCode = 1 From 6ec913c09f19d9becf23d5af55d78de949e65a78 Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 12:31:23 -0700 Subject: [PATCH 3/4] fix(db): fence schema push and require direct retirement sessions --- .../search-retirement.integration.ts | 222 ++++++++++++++++-- packages/db/scripts/push.ts | 125 ++++++---- packages/db/scripts/retire-indexed-search.ts | 50 +++- 3 files changed, 319 insertions(+), 78 deletions(-) diff --git a/packages/db/maintenance/search-retirement.integration.ts b/packages/db/maintenance/search-retirement.integration.ts index 64a061da8d1..a28873e79d3 100644 --- a/packages/db/maintenance/search-retirement.integration.ts +++ b/packages/db/maintenance/search-retirement.integration.ts @@ -1,4 +1,4 @@ -import { spawnSync } from 'node:child_process' +import { spawn, spawnSync } from 'node:child_process' import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -163,10 +163,15 @@ describe('operator-driven Search retirement in PostgreSQL', () => { it('resumes copy, includes late writes, replaces all widths, and only then purges Search chunks', async () => { expect(await getSearchRetirementStatus(sql)).toBeNull() await initializeSearchRetirement(sql) - await nextPage() - await nextPage() + for (let page = 0; page < 10; page++) { + if ((await nextPage()).copied > 0) break + } + expect((await getSearchRetirementStatus(sql))?.copied).toBeGreaterThan(0) + expect( + await sql`SELECT id FROM embedding_search_retirement_shadow WHERE id = 'full-1024-1'` + ).toHaveLength(1) await writer`UPDATE embedding SET enabled = false WHERE id = 'full-1024-1'` - await writer`DELETE FROM embedding WHERE id = 'full-384-1'` + await writer`DELETE FROM embedding WHERE id = 'full-1024-3'` await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, content, content_length, token_count, start_offset, end_offset, embedding) VALUES ('000-late', 'prefix', 'prefix-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, @@ -192,7 +197,10 @@ describe('operator-driven Search retirement in PostgreSQL', () => { await sql`SELECT id FROM embedding_search WHERE knowledge_base_id = 'search'` ).toHaveLength(0) expect(await sql`SELECT id FROM embedding_search WHERE id = 'prefix-4'`).toHaveLength(1) - expect(await sql`SELECT id FROM embedding_search WHERE id = 'full-384-1'`).toHaveLength(0) + expect(await sql`SELECT id FROM embedding_search WHERE id = 'full-1024-3'`).toHaveLength(0) + expect(await sql`SELECT enabled FROM embedding_search WHERE id = 'full-1024-1'`).toEqual([ + { enabled: false }, + ]) expect( ( await sql`SELECT vector_dims(vector_512) AS width FROM embedding_search WHERE id = '000-late'` @@ -568,7 +576,124 @@ describe('operator-driven Search retirement in PostgreSQL', () => { expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) }) - it('the CLI refuses missing or unhealthy telemetry before initializing and status stays read-only', async () => { + it('refuses db:push while retirement holds its maintenance fence before creating a receipt', async () => { + await writer`SELECT pg_advisory_lock(hashtextextended('sim:search-retirement-maintenance', 0))` + try { + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const result = spawnSync( + 'bun', + ['--no-env-file', './scripts/push.ts', '--retirement-test-invalid-option'], + { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { ...process.env, NODE_ENV: 'development', DATABASE_URL: fixtureUrl.toString() }, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + } + ) + expect(result.status).toBe(1) + expect(`${result.stdout}${result.stderr}`).toContain( + 'Another migration or retirement operation' + ) + expect(`${result.stdout}${result.stderr}`).not.toContain('Unrecognized options') + expect(await getSearchRetirementStatus(sql)).toBeNull() + } finally { + await writer`SELECT pg_advisory_unlock_all()` + } + }) + + it('fences retirement during db:push and stops database preparation when its fence connection closes', async () => { + await sql`ALTER TABLE workspace_files ADD COLUMN size bigint` + await writer`BEGIN` + await writer`LOCK TABLE workspace_files IN ACCESS EXCLUSIVE MODE` + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const child = spawn( + 'bun', + ['--no-env-file', './scripts/push.ts', '--force', '--retirement-test-invalid-option'], + { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { ...process.env, NODE_ENV: 'development', DATABASE_URL: fixtureUrl.toString() }, + stdio: ['ignore', 'pipe', 'pipe'], + timeout: 10_000, + } + ) + let output = '' + child.stdout.on('data', (chunk: Buffer) => { + output += chunk.toString() + }) + child.stderr.on('data', (chunk: Buffer) => { + output += chunk.toString() + }) + const exited = new Promise((resolve, reject) => { + child.once('error', reject) + child.once('close', resolve) + }) + try { + await vi.waitFor( + async () => { + expect( + await sql`SELECT pid FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND query = 'LOCK TABLE public.workspace_files IN ACCESS EXCLUSIVE MODE'` + ).toHaveLength(1) + }, + { timeout: 5_000 } + ) + await expect(initializeSearchRetirement(sql)).rejects.toThrow( + /Another migration or retirement operation/ + ) + const [guard] = await sql`SELECT pid, backend_xmin FROM pg_stat_activity + WHERE datname = current_database() AND application_name = 'sim-db-push'` + expect(guard.backend_xmin).toBeNull() + await sql`SELECT pg_terminate_backend(${guard.pid})` + expect(await exited).toBe(1) + expect(output).toContain('Schema-push lock connection closed') + expect(output).not.toContain('Unrecognized options') + await writer`ROLLBACK` + await vi.waitFor(async () => { + expect( + await sql`SELECT pid FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND query = 'LOCK TABLE public.workspace_files IN ACCESS EXCLUSIVE MODE'` + ).toHaveLength(0) + }) + expect( + await sql`SELECT attname FROM pg_attribute + WHERE attrelid = 'workspace_files'::regclass AND attname = 'size' AND NOT attisdropped` + ).toHaveLength(1) + expect(await getSearchRetirementStatus(sql)).toBeNull() + } finally { + await writer`ROLLBACK` + await exited + } + }) + + it('the operator CLI refuses unverified endpoints and connection overrides before connecting', () => { + const script = fileURLToPath(new URL('../scripts/retire-indexed-search.ts', import.meta.url)) + for (const url of [ + 'postgresql://reader@pool.example.invalid:5432/postgres', + 'postgresql://reader@fixture.pg.psdb.cloud:6432/postgres?sslmode=verify-full&sslrootcert=system', + 'postgresql://reader@fixture.pg.psdb.cloud.example.invalid:5432/postgres', + 'postgresql://reader@localhost:5432/production', + 'postgresql://reader@fixture.pg.psdb.cloud:5432/postgres?sslmode=disable', + 'postgresql://fixture.pg.psdb.cloud:5432/postgres?sslmode=verify-full&sslrootcert=system', + 'postgresql://reader@fixture.pg.psdb.cloud:5432/?sslmode=verify-full&sslrootcert=system', + `${databaseUrl}?statement_timeout=0`, + ]) { + const result = spawnSync('bun', ['--no-env-file', script, 'identity'], { + env: { ...process.env, NODE_ENV: 'development', MIGRATION_DATABASE_URL: url }, + encoding: 'utf8', + timeout: 5_000, + maxBuffer: 64 * 1_024, + }) + expect(result.status).toBe(1) + expect(`${result.stdout}${result.stderr}`).not.toContain(url) + } + }) + + it('the CLI respects health gates, keeps status read-only, and stops after losing its session', async () => { const fixtureUrl = new URL(databaseUrl) fixtureUrl.pathname = `/${database}` const script = fileURLToPath(new URL('../scripts/retire-indexed-search.ts', import.meta.url)) @@ -608,22 +733,20 @@ describe('operator-driven Search retirement in PostgreSQL', () => { maxSampleAgeMs: 30_000, }) ) - await writeFile( - health, - JSON.stringify({ - databaseId, - observedAt: new Date().toISOString(), - healthy: false, - maintenanceAllowed: true, - cutoverAllowed: false, - replicaLagBytes: 0, - replicaLagSeconds: 0, - walBytesPerSecond: 0, - databaseP95Ms: 1, - cpuPercent: 1, - freeStorageBytes: 1_000, - }) - ) + const sample = { + databaseId, + observedAt: new Date().toISOString(), + healthy: false, + maintenanceAllowed: true, + cutoverAllowed: false, + replicaLagBytes: 0, + replicaLagSeconds: 0, + walBytesPerSecond: 0, + databaseP95Ms: 1, + cpuPercent: 1, + freeStorageBytes: 1_000, + } + await writeFile(health, JSON.stringify(sample)) const refused = invoke( 'prepare', '--ack-release-drained', @@ -636,6 +759,61 @@ describe('operator-driven Search retirement in PostgreSQL', () => { expect(`${refused.stdout}${refused.stderr}`).toContain('unhealthy') expect(await getSearchRetirementStatus(sql)).toBeNull() expect(invoke('status').status).toBe(0) + await writeFile(health, JSON.stringify({ ...sample, healthy: true })) + expect( + invoke( + 'prepare', + '--ack-release-drained', + '--health-file', + health, + '--health-policy', + policy + ).status + ).toBe(0) + const runner = spawn( + 'bun', + [ + '--no-env-file', + script, + 'run', + '--pages', + '2', + '--health-file', + health, + '--health-policy', + policy, + ], + { + env: environment, + stdio: 'ignore', + timeout: 20_000, + } + ) + const exited = new Promise((resolve, reject) => { + runner.once('error', reject) + runner.once('close', resolve) + }) + try { + await vi.waitFor( + async () => { + expect( + (await sql`SELECT after_id FROM search_retirement_state WHERE id = 1`)[0].after_id + ).not.toBe('') + }, + { timeout: 5_000 } + ) + const before = await getSearchRetirementStatus(sql) + const [session] = await sql`SELECT pid FROM pg_stat_activity + WHERE datname = current_database() AND application_name = 'sim-search-data-retirement'` + await sql`SELECT pg_terminate_backend(${session.pid})` + const exitCode = await exited + expect(await getSearchRetirementStatus(sql)).toEqual(before) + expect(exitCode).toBe(1) + } finally { + runner.kill('SIGKILL') + await exited + } + await abortSearchRetirement(sql) } finally { await rm(directory, { recursive: true, force: true }) } diff --git a/packages/db/scripts/push.ts b/packages/db/scripts/push.ts index 16fbf5e6ba9..5c2725097f2 100644 --- a/packages/db/scripts/push.ts +++ b/packages/db/scripts/push.ts @@ -1,3 +1,4 @@ +import { fileURLToPath } from 'node:url' import { createLogger } from '@sim/logger' import { getPostgresErrorCode } from '@sim/utils/errors' import postgres, { type Sql } from 'postgres' @@ -14,34 +15,56 @@ const RECONCILIATION_COMMANDS = [ ['bun', '--env-file=.env', 'run', './script-migrations/0026_user_table_schema_for_write.ts'], ] -/** Historical push reconcilers recreate retired projections and cannot run after replacement starts. */ -async function retirementAllowsPush(): Promise { - const url = process.env.DATABASE_URL - if (!url) { +/** Keep preparation, Drizzle and reconcilers fenced without a session lock or a retained snapshot. */ +async function withRetirementFence(run: (signal: AbortSignal) => Promise): Promise { + const rawUrl = process.env.DATABASE_URL + if (!rawUrl) { logger.error('DATABASE_URL is required for schema push') - return false + return 1 } + const abort = new AbortController() let sql: Sql | undefined try { - sql = postgres(url, { + const url = new URL(rawUrl) + for (const parameter of ['application_name', 'statement_timeout', 'lock_timeout', 'options']) { + url.searchParams.delete(parameter) + } + sql = postgres(url.toString(), { max: 1, prepare: false, connect_timeout: 5, - connection: { statement_timeout: 2_000, lock_timeout: 100 }, + max_lifetime: null, + connection: { application_name: 'sim-db-push', statement_timeout: 2_000, lock_timeout: 100 }, onnotice: () => undefined, + onclose: () => abort.abort(), }) - const [state] = await sql<{ present: boolean }[]>` - SELECT to_regclass('public.search_retirement_state') IS NOT NULL AS present` - if (state.present) { - logger.error( - 'Schema push is disabled after Search retirement starts, including completed retirement. Use reviewed versioned migrations; push reconcilers would restore retired projections' - ) - return false - } - return true + return (await sql.begin('isolation level read committed read only', async (tx) => { + // Simple protocol closes SELECT portals so concurrent index builds do not wait on their snapshots. + const [locks] = await tx<{ maintenance: boolean; migration: boolean }[]>` + SELECT pg_try_advisory_xact_lock(hashtextextended('sim:search-retirement-maintenance', 0)) AS maintenance, + pg_try_advisory_xact_lock(4961002270::bigint) AS migration`.simple() + if (!locks.maintenance || !locks.migration) { + logger.error('Another migration or retirement operation is active; retry schema push later') + return 1 + } + const [state] = await tx<{ present: boolean }[]>` + SELECT to_regclass('public.search_retirement_state') IS NOT NULL AS present`.simple() + if (state.present) { + logger.error( + 'Schema push is disabled after Search retirement starts, including completed retirement. Use reviewed versioned migrations; push reconcilers would restore retired projections' + ) + return 1 + } + return run(abort.signal) + })) as number } catch (error) { - logger.error('Unable to verify schema-push safety', { code: getPostgresErrorCode(error) }) - return false + logger.error( + abort.signal.aborted + ? 'Schema-push lock connection closed; stopped all commands' + : 'Schema push stopped', + { code: getPostgresErrorCode(error) } + ) + return 1 } finally { await sql?.end({ timeout: 1 }).catch(() => undefined) } @@ -66,41 +89,45 @@ export async function runPush(args: string[]): Promise { return 1 } - if (!help && !(await retirementAllowsPush())) return 1 - - if (!help && args.includes('--force')) { - const preparation = Bun.spawn(['bun', '--env-file=.env', 'run', './scripts/prepare-push.ts'], { - stdin: 'inherit', - stdout: 'inherit', - stderr: 'inherit', - }) - const preparationExit = await preparation.exited - if (preparationExit !== 0) return preparationExit - } + async function runCommands(signal?: AbortSignal): Promise { + async function runCommand(command: string[]): Promise { + signal?.throwIfAborted() + const child = Bun.spawn(command, { + env: { ...process.env, SIM_DB_PUSH_RENAME_MODE: interactiveRenames ? 'prompt' : 'create' }, + stdin: 'inherit', + stdout: 'inherit', + stderr: 'inherit', + signal, + killSignal: 'SIGKILL', + }) + const code = await child.exited + signal?.throwIfAborted() + return code + } - const pushArgs = args.filter((arg) => arg !== '--interactive-renames') - const child = Bun.spawn( - ['bunx', '--no-install', 'drizzle-kit', 'push', '--config=./drizzle.config.ts', ...pushArgs], - { - env: { ...process.env, SIM_DB_PUSH_RENAME_MODE: interactiveRenames ? 'prompt' : 'create' }, - stdin: 'inherit', - stdout: 'inherit', - stderr: 'inherit', + if (!help && args.includes('--force')) { + const code = await runCommand(['bun', '--env-file=.env', 'run', './scripts/prepare-push.ts']) + if (code !== 0) return code } - ) - const exitCode = await child.exited - if (exitCode !== 0 || help) return exitCode + const pushArgs = args.filter((arg) => arg !== '--interactive-renames') + // Launch the installed CLI directly so losing the fence also terminates its database connection. + const cli = fileURLToPath(new URL('bin.cjs', import.meta.resolve('drizzle-kit'))) + const exitCode = await runCommand([ + 'bun', + cli, + 'push', + '--config=./drizzle.config.ts', + ...pushArgs, + ]) + if (exitCode !== 0 || help) return exitCode - for (const command of RECONCILIATION_COMMANDS) { - const reconciliation = Bun.spawn(command, { - stdin: 'inherit', - stdout: 'inherit', - stderr: 'inherit', - }) - const reconciliationExit = await reconciliation.exited - if (reconciliationExit !== 0) return reconciliationExit + for (const command of RECONCILIATION_COMMANDS) { + const code = await runCommand(command) + if (code !== 0) return code + } + return 0 } - return 0 + return help ? runCommands() : withRetirementFence(runCommands) } if (import.meta.main) { diff --git a/packages/db/scripts/retire-indexed-search.ts b/packages/db/scripts/retire-indexed-search.ts index b477b11a426..0692e1390a9 100644 --- a/packages/db/scripts/retire-indexed-search.ts +++ b/packages/db/scripts/retire-indexed-search.ts @@ -39,11 +39,14 @@ Every write except abort requires --health-file PATH --health-policy PATH. Policy databaseId must equal the identity command's fingerprint. Run options: --page-size 25 (1–100), --pages 1 (1–120), --seconds 60 (1–600). A fixed pause of at least 5 seconds follows each page. Timeouts stop; no automatic retry. -Connection: MIGRATION_DATABASE_URL, PostgreSQL 17+, direct primary or session pooling only. +Connection: MIGRATION_DATABASE_URL, PostgreSQL 17+, verified PlanetScale primary on port 5432. +Other endpoints are refused; loopback databases with a test name segment are for local fixtures only. Deploy the code-removal release and drain old workers first. Then: - prepare -> run until ready -> cutover -> verify retrieval -> begin-purge + prepare -> run until ready -> cutover -> verify behavior -> begin-purge -> run until finalize -> finalize. Repeat run if late writes return the phase to purge. +Start with default one-page runs and watch telemetry before requesting longer runs. +Before begin-purge, verify ordinary-KB retrieval, ACL denials, connector ingestion and live Search. Reads use the old projection until cutover. The retained backup is not an instant rollback. Cutover requires pg_read_all_stats and no old snapshots; never automatically retry DDL gates. Abort, backup removal and finalization cap implicit DROP lock waits at 1 ms. @@ -111,6 +114,39 @@ async function main(): Promise { const url = new URL(rawUrl) if (!['postgres:', 'postgresql:'].includes(url.protocol)) throw new RetirementCommandError('Expected a PostgreSQL URL') + if (!url.username || !url.pathname.slice(1)) { + throw new RetirementCommandError('The connection URL must include its role and database name') + } + const localFixture = + ['localhost', '127.0.0.1', '[::1]'].includes(url.hostname) && + /(^|_)test(_|$)/.test(decodeURIComponent(url.pathname.slice(1))) + const hosted = + /^[a-z0-9-]+\.(?:pg|horizon)\.psdb\.cloud$/.test(url.hostname) && + (url.port || '5432') === '5432' && + !decodeURIComponent(url.username).includes('|') + if (!hosted && !localFixture) { + throw new RetirementCommandError( + 'Use a direct PlanetScale primary endpoint on port 5432; unverified endpoints and poolers are unsupported' + ) + } + if ( + [...url.searchParams.keys()].some( + (key) => !['sslmode', 'sslrootcert', 'sslnegotiation'].includes(key) + ) + ) { + throw new RetirementCommandError( + 'Connection overrides are unsupported; only TLS URL parameters are allowed' + ) + } + if ( + hosted && + (url.searchParams.get('sslmode') !== 'verify-full' || + url.searchParams.get('sslrootcert') !== 'system') + ) { + throw new RetirementCommandError( + 'PlanetScale requires sslmode=verify-full and sslrootcert=system' + ) + } const databaseId = createHash('sha256') .update( JSON.stringify([url.hostname.toLowerCase(), url.port || '5432', url.pathname, url.username]) @@ -120,11 +156,6 @@ async function main(): Promise { process.stdout.write(`${databaseId}\n`) return } - if (url.hostname.endsWith('.pg.psdb.cloud') && url.port === '6432') { - throw new RetirementCommandError( - 'PlanetScale transaction pooling is unsupported; use a direct primary connection' - ) - } const pageSize = integerOption(values['page-size'], 25, 100) const pages = integerOption(values.pages, 1, 120) const budgetMs = integerOption(values.seconds, 60, 600) * 1_000 @@ -156,6 +187,8 @@ async function main(): Promise { await checkHealth() } const sql = postgres(rawUrl, { + port: Number(url.port || '5432'), + ssl: hosted ? 'verify-full' : false, max: 1, prepare: false, connect_timeout: 5, @@ -167,6 +200,9 @@ async function main(): Promise { idle_in_transaction_session_timeout: 5_000, }, onnotice: () => undefined, + onclose: () => { + void sql.end({ timeout: 0 }).catch(() => undefined) + }, }) try { if (command === 'status') { From 5593bea1e4424c4408b87fa1a2416a90eb732a3a Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 12:52:33 -0700 Subject: [PATCH 4/4] fix(db): preserve test connection URL parameters --- packages/db/maintenance/search-retirement.integration.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/packages/db/maintenance/search-retirement.integration.ts b/packages/db/maintenance/search-retirement.integration.ts index a28873e79d3..0ab85b505ef 100644 --- a/packages/db/maintenance/search-retirement.integration.ts +++ b/packages/db/maintenance/search-retirement.integration.ts @@ -672,6 +672,8 @@ describe('operator-driven Search retirement in PostgreSQL', () => { it('the operator CLI refuses unverified endpoints and connection overrides before connecting', () => { const script = fileURLToPath(new URL('../scripts/retire-indexed-search.ts', import.meta.url)) + const overrideUrl = new URL(databaseUrl) + overrideUrl.searchParams.set('statement_timeout', '0') for (const url of [ 'postgresql://reader@pool.example.invalid:5432/postgres', 'postgresql://reader@fixture.pg.psdb.cloud:6432/postgres?sslmode=verify-full&sslrootcert=system', @@ -680,7 +682,7 @@ describe('operator-driven Search retirement in PostgreSQL', () => { 'postgresql://reader@fixture.pg.psdb.cloud:5432/postgres?sslmode=disable', 'postgresql://fixture.pg.psdb.cloud:5432/postgres?sslmode=verify-full&sslrootcert=system', 'postgresql://reader@fixture.pg.psdb.cloud:5432/?sslmode=verify-full&sslrootcert=system', - `${databaseUrl}?statement_timeout=0`, + overrideUrl.toString(), ]) { const result = spawnSync('bun', ['--no-env-file', script, 'identity'], { env: { ...process.env, NODE_ENV: 'development', MIGRATION_DATABASE_URL: url },