|
| 1 | +import { writeFileSync } from 'node:fs' |
| 2 | +import { db } from '@sim/db' |
| 3 | +import { document, knowledgeBase, organization, user, workspace } from '@sim/db/schema' |
| 4 | +import { generateId } from '@sim/utils/id' |
| 5 | +import { eq, inArray, sql } from 'drizzle-orm' |
| 6 | +import { afterAll, beforeAll, describe, expect, it } from 'vitest' |
| 7 | +import { |
| 8 | + createKnowledgeAclFixtureIds, |
| 9 | + seedKnowledgeAclFixture, |
| 10 | +} from '@/lib/knowledge/__integration__/seed-source-access-fixture' |
| 11 | +import { type KnowledgeAccessScope, WORKSPACE_ACCESS_TOKENS } from '@/lib/knowledge/access/types' |
| 12 | +import { getWorkspaceKnowledgeBases } from '@/lib/knowledge/service' |
| 13 | + |
| 14 | +/** |
| 15 | + * A small KB page must not count the rest of its workspace or other tenants before applying |
| 16 | + * its limit. Real query plans catch this even when warm caches hide it from a timing test. |
| 17 | + * The same read must retain ACL/lifecycle filtering, empty bases, and keyset continuity. |
| 18 | + */ |
| 19 | +const ids = createKnowledgeAclFixtureIds() |
| 20 | +const foreign = createKnowledgeAclFixtureIds() |
| 21 | +const offPageId = generateId() |
| 22 | +const emptyId = generateId() |
| 23 | +const archivedId = generateId() |
| 24 | +const access: KnowledgeAccessScope = { kind: 'workspace', tokens: WORKSPACE_ACCESS_TOKENS } |
| 25 | +const reports: Array<Record<string, unknown>> = [] |
| 26 | + |
| 27 | +interface CapturedQuery { |
| 28 | + query: string |
| 29 | + parameters: NonNullable<Parameters<typeof db.$client.unsafe>[1]> |
| 30 | +} |
| 31 | + |
| 32 | +interface ExplainNode { |
| 33 | + 'Relation Name'?: string |
| 34 | + 'Index Name'?: string |
| 35 | + 'Actual Rows': number |
| 36 | + 'Actual Loops': number |
| 37 | + 'Rows Removed by Filter'?: number |
| 38 | + 'Rows Removed by Index Recheck'?: number |
| 39 | + Plans?: ExplainNode[] |
| 40 | +} |
| 41 | + |
| 42 | +function documentVisits(node: ExplainNode): number { |
| 43 | + const readsDocuments = |
| 44 | + node['Relation Name'] === 'document' || node['Index Name']?.startsWith('doc_') |
| 45 | + const own = readsDocuments |
| 46 | + ? (node['Actual Rows'] + |
| 47 | + (node['Rows Removed by Filter'] ?? 0) + |
| 48 | + (node['Rows Removed by Index Recheck'] ?? 0)) * |
| 49 | + node['Actual Loops'] |
| 50 | + : 0 |
| 51 | + return own + (node.Plans ?? []).reduce((total, child) => total + documentVisits(child), 0) |
| 52 | +} |
| 53 | + |
| 54 | +beforeAll(async () => { |
| 55 | + await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) |
| 56 | + await seedKnowledgeAclFixture(foreign, { connectorType: 'google_drive' }) |
| 57 | + await db |
| 58 | + .update(knowledgeBase) |
| 59 | + .set({ name: 'A small', createdAt: new Date('2026-01-01') }) |
| 60 | + .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) |
| 61 | + await db.insert(knowledgeBase).values([ |
| 62 | + { |
| 63 | + id: emptyId, |
| 64 | + workspaceId: ids.workspaceId, |
| 65 | + userId: ids.aliceId, |
| 66 | + name: 'B empty', |
| 67 | + createdAt: new Date('2026-01-02'), |
| 68 | + }, |
| 69 | + { |
| 70 | + id: offPageId, |
| 71 | + workspaceId: ids.workspaceId, |
| 72 | + userId: ids.aliceId, |
| 73 | + name: 'C large', |
| 74 | + createdAt: new Date('2026-01-03'), |
| 75 | + }, |
| 76 | + { |
| 77 | + id: archivedId, |
| 78 | + workspaceId: ids.workspaceId, |
| 79 | + userId: ids.aliceId, |
| 80 | + name: 'D archived', |
| 81 | + deletedAt: new Date(), |
| 82 | + }, |
| 83 | + ]) |
| 84 | + await db.insert(document).values( |
| 85 | + [ |
| 86 | + { tokenCount: 7 }, |
| 87 | + { tokenCount: 11 }, |
| 88 | + { tokenCount: 100, acl: ['u:hidden@fixture.test'] }, |
| 89 | + { tokenCount: 100, archivedAt: new Date() }, |
| 90 | + { tokenCount: 100, deletedAt: new Date() }, |
| 91 | + { tokenCount: 100, userExcluded: true }, |
| 92 | + ].map((row) => ({ |
| 93 | + id: generateId(), |
| 94 | + knowledgeBaseId: ids.knowledgeBaseId, |
| 95 | + filename: 'fixture.txt', |
| 96 | + fileUrl: 'https://fixture.invalid/document', |
| 97 | + fileSize: 1, |
| 98 | + mimeType: 'text/plain', |
| 99 | + acl: ['ws'], |
| 100 | + ...row, |
| 101 | + })) |
| 102 | + ) |
| 103 | + for (const baseId of [offPageId, foreign.knowledgeBaseId]) { |
| 104 | + await db.execute(sql`INSERT INTO document |
| 105 | + (id, knowledge_base_id, filename, file_url, file_size, mime_type, acl, token_count) |
| 106 | + SELECT ${baseId} || '-' || n, ${baseId}, 'bulk.txt', 'https://fixture.invalid/bulk', |
| 107 | + 1, 'text/plain', ARRAY['ws'], 1 FROM generate_series(1, 10000) AS n`) |
| 108 | + } |
| 109 | + await db.execute(sql`ANALYZE knowledge_base`) |
| 110 | + await db.execute(sql`ANALYZE document`) |
| 111 | +}, 60_000) |
| 112 | + |
| 113 | +afterAll(async () => { |
| 114 | + const reportPath = process.env.KNOWLEDGE_BASE_LIST_REPORT_PATH |
| 115 | + if (reportPath) writeFileSync(reportPath, JSON.stringify(reports, null, 2)) |
| 116 | + try { |
| 117 | + for (const fixture of [ids, foreign]) { |
| 118 | + await db.delete(workspace).where(eq(workspace.id, fixture.workspaceId)) |
| 119 | + await db.delete(organization).where(eq(organization.id, fixture.organizationId)) |
| 120 | + await db.delete(user).where(inArray(user.id, [fixture.aliceId, fixture.bobId])) |
| 121 | + } |
| 122 | + } finally { |
| 123 | + await db.$client.end() |
| 124 | + } |
| 125 | +}) |
| 126 | + |
| 127 | +describe('knowledge base list counts on real Postgres', () => { |
| 128 | + it.each(['name', 'createdAt'] as const)( |
| 129 | + 'bounds document reads to a small page ordered by %s', |
| 130 | + async (sortBy) => { |
| 131 | + const captured: CapturedQuery[] = [] |
| 132 | + const previousDebug = db.$client.options.debug |
| 133 | + db.$client.options.debug = (_connection, query, parameters) => { |
| 134 | + if (captured.length < 30) captured.push({ query, parameters: [...parameters] }) |
| 135 | + } |
| 136 | + try { |
| 137 | + const page = await getWorkspaceKnowledgeBases(ids.workspaceId, 'active', { |
| 138 | + countsFor: access, |
| 139 | + limit: 1, |
| 140 | + sortBy, |
| 141 | + }) |
| 142 | + expect( |
| 143 | + page.data.map(({ id, docCount, tokenCount }) => ({ id, docCount, tokenCount })) |
| 144 | + ).toEqual([{ id: ids.knowledgeBaseId, docCount: 2, tokenCount: 18 }]) |
| 145 | + expect(page.nextCursorKeys).not.toBeNull() |
| 146 | + } finally { |
| 147 | + db.$client.options.debug = previousDebug |
| 148 | + } |
| 149 | + const plans = [] |
| 150 | + for (const statement of captured.filter(({ query }) => query.includes('"document"'))) { |
| 151 | + const [result] = await db.$client.unsafe< |
| 152 | + Array<{ 'QUERY PLAN': Array<{ Plan: ExplainNode }> }> |
| 153 | + >(`EXPLAIN (ANALYZE, BUFFERS, FORMAT JSON) ${statement.query}`, statement.parameters) |
| 154 | + plans.push(...result['QUERY PLAN']) |
| 155 | + } |
| 156 | + const visits = plans.reduce((total, plan) => total + documentVisits(plan.Plan), 0) |
| 157 | + reports.push({ sortBy, visits, plans }) |
| 158 | + expect(plans.length).toBeGreaterThan(0) |
| 159 | + expect(visits).toBeLessThan(100) |
| 160 | + } |
| 161 | + ) |
| 162 | + |
| 163 | + it('keeps empty KBs and count visibility through pagination and unpaged reads', async () => { |
| 164 | + const first = await getWorkspaceKnowledgeBases(ids.workspaceId, 'active', { |
| 165 | + countsFor: access, |
| 166 | + limit: 1, |
| 167 | + sortBy: 'name', |
| 168 | + }) |
| 169 | + if (!first.nextCursorKeys) throw new Error('Expected a second knowledge-base page') |
| 170 | + const second = await getWorkspaceKnowledgeBases(ids.workspaceId, 'active', { |
| 171 | + countsFor: access, |
| 172 | + limit: 1, |
| 173 | + sortBy: 'name', |
| 174 | + cursorKeys: first.nextCursorKeys, |
| 175 | + }) |
| 176 | + expect( |
| 177 | + second.data.map(({ id, docCount, tokenCount }) => ({ id, docCount, tokenCount })) |
| 178 | + ).toEqual([{ id: emptyId, docCount: 0, tokenCount: 0 }]) |
| 179 | + const all = await getWorkspaceKnowledgeBases(ids.workspaceId, 'active', { |
| 180 | + countsFor: access, |
| 181 | + sortBy: 'name', |
| 182 | + }) |
| 183 | + expect(all.data.map(({ id, docCount, tokenCount }) => ({ id, docCount, tokenCount }))).toEqual([ |
| 184 | + { id: ids.knowledgeBaseId, docCount: 2, tokenCount: 18 }, |
| 185 | + { id: emptyId, docCount: 0, tokenCount: 0 }, |
| 186 | + { id: offPageId, docCount: 10000, tokenCount: 10000 }, |
| 187 | + ]) |
| 188 | + expect(all.nextCursorKeys).toBeNull() |
| 189 | + const archived = await getWorkspaceKnowledgeBases(ids.workspaceId, 'archived', { |
| 190 | + countsFor: access, |
| 191 | + }) |
| 192 | + expect(archived.data.map(({ id }) => id)).toEqual([archivedId]) |
| 193 | + }) |
| 194 | +}) |
0 commit comments