From 884cdd90a8ceff92ce5b608eaf0f9fa975088ac8 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Fri, 25 Sep 2026 17:24:31 -0700 Subject: [PATCH] improvement(search): consolidate knowledge search around live Sim Search and document-decided workspace retrieval - Move indexed organization search under lib/sim-search/indexed, dormant behind the single isIndexedOrgSearchEnabled() gate, with a check:indexed-org-search-boundary audit keeping callers on its public entry - Decide workspace knowledge base search on the document for every principal, so workspace retrieval never waits on the projector - Remove the knowledge-async-projection, knowledge-projection-fill and knowledge-tin-keyword flags and their code paths - Trim the projector to a single periodic sweep that releases workspace marks before deciding whether a pass is owed - Remove per-source vector index builds - Scope member sync and processing recovery to the indexed search path - Remove dead code left behind by the consolidation - Add an ops runbook and scripts for recovering and maintaining a dormant search index --- .../api/cron/knowledge-projection/route.ts | 5 +- .../member-sync/route.integration.ts | 79 + .../knowledge/connectors/member-sync/route.ts | 16 +- .../github/installations/route.test.ts | 168 - .../knowledge/github/installations/route.ts | 53 - .../api/knowledge/member-connectors/route.ts | 22 - apps/sim/app/api/knowledge/search/route.ts | 7 +- .../app/api/knowledge/search/utils.test.ts | 490 --- apps/sim/app/api/knowledge/utils.ts | 14 - apps/sim/app/api/v1/knowledge/search/route.ts | 6 +- .../github-member-integration.tsx | 0 .../integrations/indexed/index.ts | 1 + .../{ => indexed}/member-integration-row.tsx | 12 +- .../member-integrations-list.tsx | 8 +- .../integrations/indexed}/source-status.ts | 0 .../indexed}/use-member-enrollment.test.tsx | 2 +- .../indexed}/use-member-enrollment.ts | 63 +- .../integrations/integrations.test.tsx | 10 +- .../integrations/integrations.tsx | 2 +- .../[knowledgeBaseId]/[documentId]/page.tsx | 4 +- .../components/integrations/indexed/index.ts | 1 + ...xed-organization-integrations-settings.tsx | 139 + .../organization-integrations-setup.tsx | 0 .../organization-search-stats-period.tsx | 0 .../organization-source-people.tsx | 0 .../organization-source-stats.tsx | 2 +- ...rganization-integrations-settings.test.tsx | 11 +- .../organization-integrations-settings.tsx | 133 +- .../integrations/search-source-setup.test.tsx | 14 + .../providers/[connectorType]/page.test.tsx | 6 +- .../knowledge-search-results/index.ts | 5 +- .../knowledge-search-results/indexed/index.ts | 1 + .../indexed/indexed-search-results.tsx | 214 ++ .../knowledge-search-results.tsx | 302 +- .../search-transitions.test.tsx | 11 + .../knowledge-search-results/utils.ts | 85 + apps/sim/background/knowledge-projection.ts | 16 +- .../queries/github-search-installations.ts | 53 - .../use-github-installation-setup.test.tsx | 4 +- .../hooks/use-github-installation-setup.ts | 2 - .../lib/api/contracts/knowledge/connectors.ts | 26 - .../knowledge/github-installations.ts | 63 - apps/sim/lib/core/config/env.ts | 3 - .../sim/lib/core/config/feature-flags.test.ts | 45 - apps/sim/lib/core/config/feature-flags.ts | 37 +- .../credential-groups/slack-provider.test.ts | 1 + ...esolve-organization-personal-token.test.ts | 8 + .../resolve-organization-personal-token.ts | 63 +- .../lib/folders/application/resource-vfs.ts | 438 --- ...async-projection-processing.integration.ts | 162 - .../__integration__/coda-live.integration.ts | 8 + .../confluence-enrollment.integration.ts | 48 +- .../directory-sync.integration.ts | 14 +- ...dormant-processing-recovery.integration.ts | 152 + .../excluded-member-documents.integration.ts | 6 + .../filtered-search.integration.ts | 5 +- .../github-member.integration.ts | 8 +- .../gitlab-live.integration.ts | 10 +- .../gmail-member.integration.ts | 6 + .../google-calendar-member.integration.ts | 6 + .../jira-member.integration.ts | 6 + .../kb-block-search.integration.ts | 26 +- .../knowledge-projection.integration.ts | 468 +-- .../organization-mcp-search.integration.ts | 17 +- ...rovider-processing-recovery.integration.ts | 8 +- .../read-indexed-document.integration.ts | 14 +- .../__integration__/scale.integration.ts | 14 +- .../search-index-policy.integration.ts | 7 - .../search-latency.integration.ts | 8 + .../search-source-setup.integration.ts | 7 + .../stored-document-recovery.integration.ts | 8 +- .../unfilled-projection-source.integration.ts | 61 +- ...orkspace-kb-document-access.integration.ts | 426 +++ .../knowledge/access/predicate.integration.ts | 13 +- .../lib/knowledge/access/predicate.test.ts | 10 +- apps/sim/lib/knowledge/access/predicate.ts | 294 +- apps/sim/lib/knowledge/access/scope.ts | 13 - apps/sim/lib/knowledge/api/route-policies.ts | 21 +- .../lib/knowledge/application/batch-policy.ts | 5 - .../application/connector-access.test.ts | 1 - .../knowledge/application/connectors.test.ts | 56 +- .../lib/knowledge/application/connectors.ts | 149 +- .../knowledge/application/documents.test.ts | 67 +- .../lib/knowledge/application/documents.ts | 137 +- .../knowledge/application/knowledge-bases.ts | 53 - .../knowledge/application/knowledge-vfs.ts | 263 -- .../knowledge/application/operations.test.ts | 1 - .../lib/knowledge/application/operations.ts | 60 - .../organization-search-overview.ts | 4 +- .../organization-search-stats.test.ts | 17 +- .../application/organization-search-stats.ts | 3 + .../personal-search-integration-pages.ts | 49 + .../personal-search-integrations.test.ts | 2 + .../personal-search-integrations.ts | 262 +- .../application/search-integrations.test.ts | 3 + .../lib/knowledge/application/search.test.ts | 31 +- apps/sim/lib/knowledge/application/search.ts | 10 +- .../knowledge/application/sim-search.test.ts | 25 +- .../lib/knowledge/application/sim-search.ts | 4 + .../slack-search/assistant.test.ts | 5 - .../slack-search/onboarding.test.ts | 5 - .../application/slack-search/source-status.ts | 56 - .../connectors/external-group-sync.test.ts | 2 - .../knowledge/connectors/indexing-policy.ts | 6 +- .../connectors/member-observations.ts | 7 +- .../knowledge/connectors/member-queue.test.ts | 2 - ...ber-sync-engine-content-credential.test.ts | 2 - .../lib/knowledge/connectors/queue.test.ts | 2 - .../knowledge/connectors/sync-engine.test.ts | 2 - .../lib/knowledge/connectors/sync-engine.ts | 16 +- .../lib/knowledge/connectors/sync-limits.ts | 4 +- .../sim/lib/knowledge/connectors/sync-lock.ts | 20 +- .../connectors/user-document-visibility.ts | 4 +- .../knowledge/documents/processing-payload.ts | 14 - .../documents/processing-recovery.ts | 13 +- apps/sim/lib/knowledge/documents/service.ts | 13 - apps/sim/lib/knowledge/embeddings.test.ts | 81 +- .../lib/knowledge/mcp/server.protocol.test.ts | 6 +- apps/sim/lib/knowledge/mcp/server.test.ts | 6 +- apps/sim/lib/knowledge/mcp/server.ts | 206 +- apps/sim/lib/knowledge/mcp/tool-runner.ts | 47 + .../orchestration/connector-access.test.ts | 2 - .../orchestration/connectors.test.ts | 8 - .../lib/knowledge/orchestration/documents.ts | 56 - apps/sim/lib/knowledge/orchestration/index.ts | 1 - .../projection/enqueue-inline.test.ts | 23 +- .../lib/knowledge/projection/enqueue.test.ts | 37 +- apps/sim/lib/knowledge/projection/enqueue.ts | 96 +- apps/sim/lib/knowledge/projection/run.test.ts | 76 +- apps/sim/lib/knowledge/projection/run.ts | 59 +- apps/sim/lib/knowledge/search/candidates.ts | 541 +++ .../lib/knowledge/search/filter-conditions.ts | 23 +- .../lib/knowledge/search/keyword-ranking.ts | 49 + apps/sim/lib/knowledge/search/queries.test.ts | 1540 ++------- apps/sim/lib/knowledge/search/queries.ts | 2937 +++-------------- apps/sim/lib/knowledge/search/search-index.ts | 4 - .../search/source-vector-indexes.test.ts | 58 +- .../knowledge/search/source-vector-indexes.ts | 150 +- apps/sim/lib/knowledge/search/tag-filters.ts | 263 ++ apps/sim/lib/knowledge/search/vector-leg.ts | 340 ++ .../sim/lib/knowledge/transfer/bundle.test.ts | 24 +- apps/sim/lib/knowledge/transfer/bundle.ts | 24 - apps/sim/lib/knowledge/types.ts | 4 - .../load-search-integrations.test.ts | 4 +- .../application/load-search-integrations.ts | 80 +- apps/sim/lib/mothership/chat/payload.test.ts | 3 +- .../server/knowledge/workspace-search.test.ts | 10 +- .../server/knowledge/workspace-search.ts | 22 +- apps/sim/lib/sim-search/connectors.ts | 26 - apps/sim/lib/sim-search/indexed/README.md | 38 + .../documents}/read-indexed-document.ts | 2 + .../documents}/read-search-document.test.ts | 10 +- .../documents}/read-search-document.ts | 2 + apps/sim/lib/sim-search/indexed/gate.ts | 42 + apps/sim/lib/sim-search/indexed/index.ts | 18 + .../personal-account-ownership.ts | 29 + .../integrations/personal-inventory.ts | 43 + .../personal-search-integrations.ts | 164 + .../sim-search/indexed/mcp/register-tools.ts | 148 + .../indexed/retrieval/access-plan.ts} | 77 +- .../lib/sim-search/indexed/retrieval/index.ts | 10 + .../sim-search/indexed/retrieval/keyword.ts | 325 ++ .../sim-search/indexed/retrieval/legs.test.ts | 995 ++++++ .../lib/sim-search/indexed/retrieval/legs.ts | 128 + .../sim-search/indexed/retrieval/permitted.ts | 424 +++ .../indexed/retrieval/projection-access.ts | 234 ++ .../indexed/retrieval/projection-fill.test.ts | 78 + .../indexed/retrieval/projection-fill.ts | 102 + .../retrieval/source-vector-indexes.ts | 33 + .../retrieval}/tin-keyword-readiness.test.ts | 14 +- .../indexed/retrieval}/tin-keyword.test.ts | 23 +- .../indexed/retrieval}/tin-keyword.ts | 18 +- .../indexed/retrieval}/tin-query.test.ts | 2 +- .../indexed/retrieval}/tin-query.ts | 0 .../sim-search/indexed/retrieval/vector.ts | 491 +++ .../search/scoped-search.activity.test.ts} | 10 +- .../indexed/search/scoped-search.test.ts} | 19 +- .../indexed/search/scoped-search.ts} | 12 +- apps/sim/lib/sim-search/live/README.md | 2 +- .../lib/sim-search/live/application.test.ts | 2 - apps/sim/lib/table/application/operations.ts | 21 - apps/sim/lib/table/application/table-vfs.ts | 259 +- packages/db/knowledge-projection.test.ts | 66 +- packages/db/knowledge-projection.ts | 199 +- packages/db/schema.ts | 11 +- .../0022_projection_source_acl_backfill.ts | 7 +- .../src/mocks/deployment-shape.mock.ts | 5 +- packages/testing/src/mocks/env-flags.mock.ts | 3 +- .../src/mocks/indexed-org-search.mock.ts | 22 + .../src/mocks/knowledge-access-scope.mock.ts | 2 - .../src/mocks/knowledge-api-utils.mock.ts | 2 - .../mocks/knowledge-base-use-cases.mock.ts | 7 - .../src/mocks/sim-search-connectors.mock.ts | 5 +- 193 files changed, 7925 insertions(+), 8851 deletions(-) create mode 100644 apps/sim/app/api/knowledge/connectors/member-sync/route.integration.ts delete mode 100644 apps/sim/app/api/knowledge/github/installations/route.test.ts delete mode 100644 apps/sim/app/api/knowledge/github/installations/route.ts delete mode 100644 apps/sim/app/api/knowledge/member-connectors/route.ts delete mode 100644 apps/sim/app/api/knowledge/search/utils.test.ts rename apps/sim/app/o/[organizationId]/integrations/{ => indexed}/github-member-integration.tsx (100%) create mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/index.ts rename apps/sim/app/o/[organizationId]/integrations/{ => indexed}/member-integration-row.tsx (97%) rename apps/sim/app/o/[organizationId]/integrations/{ => indexed}/member-integrations-list.tsx (98%) rename apps/sim/{lib/sim-search => app/o/[organizationId]/integrations/indexed}/source-status.ts (100%) rename apps/sim/{hooks => app/o/[organizationId]/integrations/indexed}/use-member-enrollment.test.tsx (99%) rename apps/sim/{hooks => app/o/[organizationId]/integrations/indexed}/use-member-enrollment.ts (84%) create mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts create mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx rename apps/sim/app/o/[organizationId]/settings/components/integrations/{ => indexed}/organization-integrations-setup.tsx (100%) rename apps/sim/app/o/[organizationId]/settings/components/integrations/{ => indexed}/organization-search-stats-period.tsx (100%) rename apps/sim/app/o/[organizationId]/settings/components/integrations/{ => indexed}/organization-source-people.tsx (100%) rename apps/sim/app/o/[organizationId]/settings/components/integrations/{ => indexed}/organization-source-stats.tsx (99%) create mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/index.ts create mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/indexed-search-results.tsx create mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/utils.ts delete mode 100644 apps/sim/hooks/queries/github-search-installations.ts delete mode 100644 apps/sim/lib/folders/application/resource-vfs.ts delete mode 100644 apps/sim/lib/knowledge/__integration__/async-projection-processing.integration.ts create mode 100644 apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts create mode 100644 apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts delete mode 100644 apps/sim/lib/knowledge/application/knowledge-vfs.ts create mode 100644 apps/sim/lib/knowledge/application/personal-search-integration-pages.ts delete mode 100644 apps/sim/lib/knowledge/application/slack-search/source-status.ts create mode 100644 apps/sim/lib/knowledge/mcp/tool-runner.ts create mode 100644 apps/sim/lib/knowledge/search/candidates.ts create mode 100644 apps/sim/lib/knowledge/search/keyword-ranking.ts create mode 100644 apps/sim/lib/knowledge/search/tag-filters.ts create mode 100644 apps/sim/lib/knowledge/search/vector-leg.ts create mode 100644 apps/sim/lib/sim-search/indexed/README.md rename apps/sim/lib/{knowledge/application => sim-search/indexed/documents}/read-indexed-document.ts (98%) rename apps/sim/lib/{knowledge/application => sim-search/indexed/documents}/read-search-document.test.ts (95%) rename apps/sim/lib/{knowledge/application => sim-search/indexed/documents}/read-search-document.ts (98%) create mode 100644 apps/sim/lib/sim-search/indexed/gate.ts create mode 100644 apps/sim/lib/sim-search/indexed/index.ts create mode 100644 apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts create mode 100644 apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts create mode 100644 apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts create mode 100644 apps/sim/lib/sim-search/indexed/mcp/register-tools.ts rename apps/sim/lib/{knowledge/access/connector-eligibility.ts => sim-search/indexed/retrieval/access-plan.ts} (64%) create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/index.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/keyword.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/legs.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/permitted.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts rename apps/sim/lib/{knowledge/search => sim-search/indexed/retrieval}/tin-keyword-readiness.test.ts (58%) rename apps/sim/lib/{knowledge/search => sim-search/indexed/retrieval}/tin-keyword.test.ts (54%) rename apps/sim/lib/{knowledge/search => sim-search/indexed/retrieval}/tin-keyword.ts (71%) rename apps/sim/lib/{knowledge/search => sim-search/indexed/retrieval}/tin-query.test.ts (91%) rename apps/sim/lib/{knowledge/search => sim-search/indexed/retrieval}/tin-query.ts (100%) create mode 100644 apps/sim/lib/sim-search/indexed/retrieval/vector.ts rename apps/sim/lib/{knowledge/application/workspace-search.activity.test.ts => sim-search/indexed/search/scoped-search.activity.test.ts} (92%) rename apps/sim/lib/{knowledge/application/workspace-search.test.ts => sim-search/indexed/search/scoped-search.test.ts} (77%) rename apps/sim/lib/{knowledge/application/workspace-search.ts => sim-search/indexed/search/scoped-search.ts} (94%) create mode 100644 packages/testing/src/mocks/indexed-org-search.mock.ts diff --git a/apps/sim/app/api/cron/knowledge-projection/route.ts b/apps/sim/app/api/cron/knowledge-projection/route.ts index 5e5369afcce..f400c4f31e2 100644 --- a/apps/sim/app/api/cron/knowledge-projection/route.ts +++ b/apps/sim/app/api/cron/knowledge-projection/route.ts @@ -11,9 +11,8 @@ export const dynamic = 'force-dynamic' export const maxDuration = 60 /** - * The knowledge projector's periodic sweep: enqueues one pass per window while there is work, and - * returns once Trigger.dev accepts it. Writers ask for passes as they commit; this converges - * whatever those requests missed. + * The knowledge projector's periodic sweep: enqueues one pass per window while documents are + * marked, and returns once Trigger.dev accepts it. It is the only thing that starts a pass. */ export const GET = withRouteHandler(async (request: NextRequest) => { const authError = verifyCronAuth(request, 'Knowledge projection sweep') diff --git a/apps/sim/app/api/knowledge/connectors/member-sync/route.integration.ts b/apps/sim/app/api/knowledge/connectors/member-sync/route.integration.ts new file mode 100644 index 00000000000..66e77f9ebec --- /dev/null +++ b/apps/sim/app/api/knowledge/connectors/member-sync/route.integration.ts @@ -0,0 +1,79 @@ +/** + * The member sync scheduler's reclaim against real PostgreSQL: a members-mode connector whose + * member lease went stale is put back on the failure ladder, and a connector in any other access + * mode is left exactly as it is, whatever its member columns say. + */ + +import { db } from '@sim/db' +import { knowledgeConnector, organization, user, workspace } from '@sim/db/schema' +import { createMockRequest } from '@sim/testing' +import { generateId } from '@sim/utils/id' +import { eq, inArray } from 'drizzle-orm' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +vi.mock('@/lib/auth/internal', () => ({ verifyCronAuth: () => null })) +vi.mock('@/lib/knowledge/connectors/member-queue', async (importOriginal) => ({ + ...(await importOriginal()), + dispatchMemberSync: vi.fn(async () => undefined), +})) + +import { + createKnowledgeAclFixtureIds, + seedKnowledgeAclFixture, +} from '@/lib/knowledge/__integration__/seed-source-access-fixture' +import { GET } from '@/app/api/knowledge/connectors/member-sync/route' + +describe('member sync reclaim in PostgreSQL', () => { + const ids = createKnowledgeAclFixtureIds() + const members = generateId() + const admin = generateId() + const staleLease = new Date(Date.now() - 24 * 60 * 60 * 1000) + + beforeAll(async () => { + await seedKnowledgeAclFixture(ids) + await db.insert(knowledgeConnector).values( + [ + { id: members, accessMode: 'members' }, + { id: admin, accessMode: 'admin' }, + ].map(({ id, accessMode }) => ({ + id, + knowledgeBaseId: ids.knowledgeBaseId, + connectorType: 'google_drive', + sourceConfig: {}, + accessMode, + status: 'active', + credentialId: ids.credentialId, + memberSyncStatus: 'running', + memberSyncLockToken: generateId(), + memberSyncLockLeaseAt: staleLease, + })) + ) + }) + + afterAll(async () => { + await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) + await db.delete(organization).where(eq(organization.id, ids.organizationId)) + await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) + await db.$client.end() + }) + + it('reclaims a stale members-mode lease and leaves every other access mode alone', async () => { + const response = await GET(createMockRequest('GET'), undefined) + expect(response.status).toBe(200) + const rows = await db + .select({ + id: knowledgeConnector.id, + memberSyncStatus: knowledgeConnector.memberSyncStatus, + memberSyncLockToken: knowledgeConnector.memberSyncLockToken, + }) + .from(knowledgeConnector) + .where(inArray(knowledgeConnector.id, [members, admin])) + const byId = new Map(rows.map((row) => [row.id, row])) + expect(byId.get(members)).toMatchObject({ + memberSyncStatus: 'error', + memberSyncLockToken: null, + }) + expect(byId.get(admin)?.memberSyncStatus).toBe('running') + expect(byId.get(admin)?.memberSyncLockToken).not.toBeNull() + }) +}) diff --git a/apps/sim/app/api/knowledge/connectors/member-sync/route.ts b/apps/sim/app/api/knowledge/connectors/member-sync/route.ts index 15891cf596b..a05b3b847ef 100644 --- a/apps/sim/app/api/knowledge/connectors/member-sync/route.ts +++ b/apps/sim/app/api/knowledge/connectors/member-sync/route.ts @@ -58,6 +58,18 @@ function reclaimedNextMemberSyncAt(): SQL { return sql`CASE WHEN ${reclaimedFailureCount()} >= ${MAX_CONSECUTIVE_FAILURES} THEN NULL ELSE now() + LEAST(${reclaimedFailureCount()} * ${CONNECTOR_FAILURE_BACKOFF_STEP_MINUTES}, ${CONNECTOR_FAILURE_BACKOFF_CAP_MINUTES}) * INTERVAL '1 minute' END` } +/** + * Only the member engine takes the member lease, and only on a members-mode connector, which a + * mode switch cannot leave while the lease is held; so both reclaims match `access_mode` too, the + * predicate `kc_member_sync_due_idx` is partial on, and read that index instead of the table. + */ +function reclaimableMemberSync(status: 'running' | 'pending'): SQL | undefined { + return and( + eq(knowledgeConnector.accessMode, 'members'), + eq(knowledgeConnector.memberSyncStatus, status) + ) +} + /** * The write shared by both reclaims: a run that stopped making progress * re-enters the member failure ladder, which is the content engine's ladder @@ -108,7 +120,7 @@ export const GET = withRouteHandler(async (request: NextRequest) => { .set(reclaimPayload(STALE_LOCK_ERROR_MESSAGE)) .where( and( - eq(knowledgeConnector.memberSyncStatus, 'running'), + reclaimableMemberSync('running'), sql`${memberSyncLockLease()} <= ${sql.param(staleCutoff, knowledgeConnector.memberSyncLockLeaseAt)}`, isNull(knowledgeConnector.archivedAt), isNull(knowledgeConnector.deletedAt) @@ -120,7 +132,7 @@ export const GET = withRouteHandler(async (request: NextRequest) => { .set(reclaimPayload(LOST_DISPATCH_ERROR_MESSAGE)) .where( and( - eq(knowledgeConnector.memberSyncStatus, 'pending'), + reclaimableMemberSync('pending'), sql`${memberSyncLockLease()} <= ${sql.param(staleCutoff, knowledgeConnector.memberSyncLockLeaseAt)}`, isNull(knowledgeConnector.archivedAt), isNull(knowledgeConnector.deletedAt) diff --git a/apps/sim/app/api/knowledge/github/installations/route.test.ts b/apps/sim/app/api/knowledge/github/installations/route.test.ts deleted file mode 100644 index a924b73ae59..00000000000 --- a/apps/sim/app/api/knowledge/github/installations/route.test.ts +++ /dev/null @@ -1,168 +0,0 @@ -import { authMockFns } from '@sim/testing' -import { credentialsManagedOauthMock } from '@sim/testing/mocks/credentials-managed-oauth.mock' -import { githubInstallationMock } from '@sim/testing/mocks/github-installation.mock' -import { rateLimiterMock, rateLimiterMockFns } from '@sim/testing/mocks/rate-limiter.mock' -import { createMockRequest } from '@sim/testing/mocks/request.mock' -import { NextRequest, NextResponse } from 'next/server' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -const mocks = vi.hoisted(() => ({ list: vi.fn(), connect: vi.fn() })) - -vi.mock('@/lib/core/rate-limiter', () => rateLimiterMock) -vi.mock('@/lib/knowledge/application/github-installations', () => ({ - listGitHubSearchInstallations: { - operation: { id: 'knowledge.github.installations.list' }, - execute: mocks.list, - }, - connectGitHubSearchInstallation: { - operation: { id: 'knowledge.github.installations.connect' }, - execute: mocks.connect, - }, -})) -vi.mock('@/lib/oauth/github-installation', () => githubInstallationMock) -vi.mock('@/lib/credentials/managed-oauth', () => credentialsManagedOauthMock) - -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { ManagedOAuthCredentialError } from '@/lib/credentials/managed-oauth' -import { GitHubInstallationError } from '@/lib/oauth/github-installation' -import { GET, POST } from '@/app/api/knowledge/github/installations/route' - -const URL = 'http://localhost/api/knowledge/github/installations' -const installation = { - installationId: '123', - accountId: '456', - accountLogin: 'acme', - accountType: 'Organization', -} - -beforeEach(() => { - authMockFns.mockGetSession.mockResolvedValue({ - user: { id: 'admin-1' }, - session: { id: 'session-1' }, - }) - rateLimiterMockFns.mockEnforceUserRateLimit.mockResolvedValue(null) - mocks.list.mockResolvedValue({ - available: true, - installUrl: 'https://github.com/apps/sim-search/installations/new', - needsUserConnection: false, - installations: [installation], - }) - mocks.connect.mockResolvedValue({ credential: { id: 'cred-1', displayName: 'GitHub · acme' } }) -}) - -describe('GitHub installation route boundary', () => { - it.each(['GET', 'POST'] as const)( - 'authenticates %s before parsing or calling the use case', - async (method) => { - authMockFns.mockGetSession.mockResolvedValue(null) - const request = new NextRequest(URL, method === 'POST' ? { method, body: '{' } : undefined) - const json = vi.spyOn(request, 'json') - const response = await (method === 'GET' ? GET(request) : POST(request)) - expect(response.status).toBe(401) - expect(response.headers.get('Cache-Control')).toBe('private, no-store') - expect(json).not.toHaveBeenCalled() - expect(rateLimiterMockFns.mockEnforceUserRateLimit).not.toHaveBeenCalled() - expect(mocks.list).not.toHaveBeenCalled() - expect(mocks.connect).not.toHaveBeenCalled() - } - ) - - it('applies admission before parsing the POST body', async () => { - rateLimiterMockFns.mockEnforceUserRateLimit.mockResolvedValue( - NextResponse.json({ error: 'Rate limit exceeded' }, { status: 429 }) - ) - const request = new NextRequest(URL, { method: 'POST', body: '{' }) - const json = vi.spyOn(request, 'json') - expect((await POST(request)).status).toBe(429) - expect(json).not.toHaveBeenCalled() - expect(mocks.connect).not.toHaveBeenCalled() - expect(rateLimiterMockFns.mockEnforceUserRateLimit).toHaveBeenCalledWith( - 'github-search-installations', - 'admin-1', - undefined - ) - }) - - it.each(['0', '-1', '1.5', '123/path', ''])( - 'rejects invalid installation ID %s before the use case', - async (installationId) => { - const response = await POST( - new NextRequest(URL, { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ organizationId: 'org-1', installationId }), - }) - ) - expect(response.status).toBe(400) - expect(mocks.connect).not.toHaveBeenCalled() - } - ) - - it('forwards POST cancellation and only returns the safe credential projection', async () => { - const request = new NextRequest(URL, { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ organizationId: 'org-1', installationId: '123' }), - }) - mocks.connect.mockResolvedValue({ - credential: { - id: 'cred-1', - displayName: 'GitHub · acme', - encryptedServiceAccountKey: 'private', - }, - created: true, - }) - const response = await POST(request) - expect(response.status).toBe(200) - expect(response.headers.get('Cache-Control')).toBe('private, no-store') - expect(mocks.connect).toHaveBeenCalledWith( - expect.objectContaining({ - input: { organizationId: 'org-1', installationId: '123', signal: request.signal }, - }) - ) - expect(await response.json()).toEqual({ - success: true, - credential: { id: 'cred-1', displayName: 'GitHub · acme' }, - }) - }) - - it.each([ - [ - new OrchestrationError('forbidden', 'Organization administrator access is required'), - 403, - 'Organization administrator access is required', - ], - [ - new GitHubInstallationError('Installation permission denied', 403), - 403, - 'Installation permission denied', - ], - [ - new GitHubInstallationError('GitHub is temporarily unavailable', 503), - 502, - 'GitHub is temporarily unavailable', - ], - [ - new ManagedOAuthCredentialError( - 'MANAGED_CREDENTIAL_NEEDS_REAUTH', - 'private refresh details', - 401 - ), - 401, - 'Reconnect your GitHub account to continue installation setup', - ], - [new Error('private database details'), 500, 'Internal server error'], - ] as const)( - 'projects %s without successful installation data', - async (error, status, message) => { - mocks.list.mockRejectedValue(error) - const response = await GET(createMockRequest({ url: `${URL}?organizationId=org-1` })) - expect(response.status).toBe(status) - expect(response.headers.get('Cache-Control')).toBe('private, no-store') - const body = await response.json() - expect(body.error).toBe(message) - expect(body).not.toHaveProperty('installations') - expect(body).not.toHaveProperty('credential') - } - ) -}) diff --git a/apps/sim/app/api/knowledge/github/installations/route.ts b/apps/sim/app/api/knowledge/github/installations/route.ts deleted file mode 100644 index 1ec3700a125..00000000000 --- a/apps/sim/app/api/knowledge/github/installations/route.ts +++ /dev/null @@ -1,53 +0,0 @@ -import { - connectGitHubSearchInstallationContract, - listGitHubSearchInstallationsContract, -} from '@/lib/api/contracts/knowledge/github-installations' -import { - defineInternalJsonRoute, - extendInternalErrorPolicy, - internalErrorResponse, - internalOrchestrationErrorPolicy, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { ManagedOAuthCredentialError } from '@/lib/credentials/managed-oauth' -import { - connectGitHubSearchInstallation, - listGitHubSearchInstallations, -} from '@/lib/knowledge/application/github-installations' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { GitHubInstallationError } from '@/lib/oauth/github-installation' - -const errorPolicy = extendInternalErrorPolicy(internalOrchestrationErrorPolicy, (error) => { - if (error instanceof GitHubInstallationError) - return internalErrorResponse(error.status === 403 ? 403 : 502, { error: error.message }) - if (error instanceof ManagedOAuthCredentialError) - return internalErrorResponse(error.statusCode, { - error: 'Reconnect your GitHub account to continue installation setup', - }) - return null -}) - -export const GET = defineInternalJsonRoute({ - contract: listGitHubSearchInstallationsContract, - auth: internalSessionAuth, - operation: knowledgeOperations.listGitHubInstallations, - rateLimit: internalRateLimits.user({ bucketName: 'github-search-installations' }), - errorPolicy, - mapInput: ({ query }, { request }) => ({ ...query, signal: request.signal }), - useCase: listGitHubSearchInstallations, - present: (result) => ({ success: true, ...result }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) - -export const POST = defineInternalJsonRoute({ - contract: connectGitHubSearchInstallationContract, - auth: internalSessionAuth, - operation: knowledgeOperations.connectGitHubInstallation, - rateLimit: internalRateLimits.user({ bucketName: 'github-search-installations' }), - errorPolicy, - mapInput: ({ body }, { request }) => ({ ...body, signal: request.signal }), - useCase: connectGitHubSearchInstallation, - present: ({ credential }) => ({ success: true, credential }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) diff --git a/apps/sim/app/api/knowledge/member-connectors/route.ts b/apps/sim/app/api/knowledge/member-connectors/route.ts deleted file mode 100644 index 3b7d453ea6e..00000000000 --- a/apps/sim/app/api/knowledge/member-connectors/route.ts +++ /dev/null @@ -1,22 +0,0 @@ -import { listWorkspaceMemberConnectorsContract } from '@/lib/api/contracts/knowledge' -import { - defineInternalJsonRoute, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { listWorkspaceMemberConnectors } from '@/lib/knowledge/application/connectors' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' - -export const GET = defineInternalJsonRoute({ - contract: listWorkspaceMemberConnectorsContract, - auth: internalSessionAuth, - operation: knowledgeOperations.listWorkspaceMemberConnectors, - rateLimit: internalRateLimits.none({ - reason: 'Preserve existing internal connector listing behavior', - }), - errorPolicy: internalKnowledgeErrorPolicies.connectors, - mapInput: ({ query }) => ({ workspaceId: query.workspaceId }), - useCase: listWorkspaceMemberConnectors, - present: ({ connectors }) => ({ success: true as const, data: connectors }), -}) diff --git a/apps/sim/app/api/knowledge/search/route.ts b/apps/sim/app/api/knowledge/search/route.ts index e0004abafb5..1186b6da0f0 100644 --- a/apps/sim/app/api/knowledge/search/route.ts +++ b/apps/sim/app/api/knowledge/search/route.ts @@ -4,12 +4,12 @@ import { internalRateLimits, internalSessionAuth, } from '@/lib/api/server/routes' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { searchScopedKnowledge } from '@/lib/knowledge/application/workspace-search' import { DEFAULT_RERANKER_MODEL } from '@/lib/knowledge/reranker-models' import { sourceAuthor } from '@/lib/knowledge/search/author' +import { searchScopedKnowledge } from '@/lib/sim-search/indexed' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { searchLiveKnowledge } from '@/lib/sim-search/live/application' const DIRECT_SEARCH_VECTOR_BUDGET_MS = 3000 @@ -81,4 +81,5 @@ const liveSearchRoute = defineInternalJsonRoute({ present: (data) => ({ success: true as const, data }), }) -export const POST = isLiveEnterpriseSearchEnabled ? liveSearchRoute : indexedSearchRoute +/** Indexed organization search is dormant unless its gate is on; Live Search serves otherwise. */ +export const POST = isIndexedOrgSearchEnabled() ? indexedSearchRoute : liveSearchRoute diff --git a/apps/sim/app/api/knowledge/search/utils.test.ts b/apps/sim/app/api/knowledge/search/utils.test.ts deleted file mode 100644 index 3ada166b485..00000000000 --- a/apps/sim/app/api/knowledge/search/utils.test.ts +++ /dev/null @@ -1,490 +0,0 @@ -/** - * Tests for knowledge search utility functions - * Focuses on testing core functionality with simplified mocking - */ -import { - dbChainMockFns, - queueTableRows, - resetDbChainMock, - schemaMock, - setupGlobalFetchMock, -} from '@sim/testing/mocks' -import { - afterAll, - afterEach, - beforeEach, - describe, - expect, - it, - type MockInstance, - vi, -} from 'vitest' -import { env } from '@/lib/core/config/env' -import * as documentsUtilsModule from '@/lib/knowledge/documents/utils' -import { runWithKnowledgeModelInputProvenance } from '@/lib/knowledge/model-input-provenance' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -vi.mock('@/lib/core/rate-limiter/provider-admission', () => ({ - PROVIDER_QUOTA_COOLDOWN_MS: 300_000, - ProviderQuotaExhaustedError: class ProviderQuotaExhaustedError extends Error {}, - ProviderAdmissionTimeoutError: class ProviderAdmissionTimeoutError extends Error {}, - isProviderQuotaExhausted: vi.fn().mockResolvedValue(false), - recordProviderCooldown: vi.fn().mockResolvedValue(undefined), - waitForProviderAdmission: vi.fn().mockResolvedValue(undefined), -})) - -/** - * Spy on the real documents/utils namespace instead of vi.mock: the shared - * `@/lib/knowledge/embeddings` module may be cached bound to the real module, - * so patching the namespace is the only wiring that always applies. - */ -let retrySpy: MockInstance -beforeEach(() => { - retrySpy = vi - .spyOn(documentsUtilsModule, 'retryWithExponentialBackoff') - .mockImplementation(((fn: () => unknown) => fn()) as never) -}) - -afterAll(() => { - retrySpy.mockRestore() -}) - -/** - * Under `isolate: false` the shared `@/lib/knowledge/embeddings` module may be - * cached bound to the REAL env module, so tests mutate the real `env` object - * (the tests below clear and assign it per case) instead of vi.mock'ing a - * file-local replacement that a cached consumer would never see. The snapshot - * restores whatever the worker started with after every test. - */ -const envSnapshot = { ...env } - -afterEach(() => { - for (const key of Object.keys(env)) { - delete (env as Record)[key] - } - Object.assign(env, envSnapshot) -}) - -import { WORKSPACE_ACCESS_SCOPE } from '@/lib/knowledge/access/scope' -import { generateSearchEmbedding, type KbEmbeddingTarget } from '@/lib/knowledge/embeddings' - -/** The platform default model and vector width, as a knowledge base records them. */ -const DEFAULT_EMBEDDING_TARGET: KbEmbeddingTarget = { - model: 'text-embedding-3-small', - dimensions: 1536, -} - -import { - executeKeywordSearch, - executeKnowledgeSearch, - fuseByReciprocalRank, - getQueryStrategy, - handleTagAndVectorSearch, - type SearchResult, -} from '@/lib/knowledge/search/queries' -import { RRF_K } from '@/lib/knowledge/search/recency' - -/** Minimal SearchResult builder — only the fields fusion and ordering read. */ -function makeResult(id: string, distance = 0.1): SearchResult { - return { - id, - content: `content-${id}`, - documentId: `doc-${id}`, - chunkIndex: 0, - tag1: null, - tag2: null, - tag3: null, - tag4: null, - tag5: null, - tag6: null, - tag7: null, - number1: null, - number2: null, - number3: null, - number4: null, - number5: null, - date1: null, - date2: null, - boolean1: null, - boolean2: null, - boolean3: null, - distance, - knowledgeBaseId: 'kb-123', - } -} - -const TEST_EMBEDDING = [0.1, 0.2, 0.3, ...Array.from({ length: 1533 }, () => 0)].map(Math.fround) - -function mockNextEmbeddingResponse(): void { - vi.mocked(fetch).mockImplementationOnce(async (_url, init) => { - const request = JSON.parse(String(init?.body)) - const embedding = - request.encoding_format === 'base64' - ? Buffer.from(new Float32Array(TEST_EMBEDDING).buffer).toString('base64') - : TEST_EMBEDDING - return new Response( - JSON.stringify({ - data: [{ embedding, index: 0 }], - usage: { prompt_tokens: 1, total_tokens: 1 }, - }), - { status: 200, headers: { 'Content-Type': 'application/json' } } - ) - }) -} - -describe('Knowledge Search Utils', () => { - beforeEach(() => { - // The worker-level fetch stub from vitest.setup.ts is removed after the - // first test by `unstubGlobals: true`; re-stub it per test so - // `vi.mocked(fetch)` always operates on a mocked fetch. - setupGlobalFetchMock({ json: {} }) - retrySpy.mockImplementation(((fn: () => unknown) => fn()) as never) - }) - - describe('handleTagAndVectorSearch', () => { - it('returns only bounded ranked rows without first materializing every matching tag ID', async () => { - resetDbChainMock() - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = (query as { toSQL: () => { sql: string } }).toSQL().sql - if (statement.includes('AS visible')) return [] - if (statement.includes(') + 0 LIMIT')) return [{ id: 'first' }, { id: 'second' }] - /** The page reads the pool slice's identities; the walk's order is kept client-side. */ - if (statement.includes('AS "connectorId"') && statement.includes('= ANY(')) - return [makeResult('second', 0.2), makeResult('first', 0.1)] - return [{ id: 'doc-first' }, { id: 'doc-second' }] - }) - queueTableRows(schemaMock.embedding, [makeResult('second', 0.2), makeResult('first', 0.1)]) - - const results = await handleTagAndVectorSearch({ - knowledgeBaseIds: ['kb-1', 'kb-2'], - access: WORKSPACE_ACCESS_SCOPE, - topK: 2, - structuredFilters: [ - { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'common' }, - ], - queryVector: { vector: JSON.stringify(TEST_EMBEDDING), dimensions: 1536 }, - distanceThreshold: 0.8, - }) - - expect(results.map((row) => row.id)).toEqual(['first', 'second']) - /** Only hydration reads through the query builder; ranking never materializes tag IDs. */ - expect(dbChainMockFns.select).toHaveBeenCalledTimes(1) - expect(dbChainMockFns.select.mock.calls[0][0]).toHaveProperty('distance') - const exact = dbChainMockFns.execute.mock.calls - .map(([query]) => (query as { toSQL: () => { sql: string; params: unknown[] } }).toSQL()) - .find((statement) => statement.sql.includes(') + 0 LIMIT'))! - expect(exact.params).toContain(200) - }) - }) - - describe('fuseByReciprocalRank', () => { - it('ranks a row found by both legs above rows found by only one', () => { - const shared = makeResult('shared') - const vectorOnly = makeResult('vector-only') - const keywordOnly = makeResult('keyword-only') - - const fused = fuseByReciprocalRank( - [ - [vectorOnly, shared], - [keywordOnly, shared], - ], - 10 - ) - - expect(fused[0].id).toBe('shared') - // `shared` is credited to both legs, so the following tie is even and - // resolves to the earliest list. - expect(fused.map((r) => r.id)).toEqual(['shared', 'vector-only', 'keyword-only']) - }) - - it('dedupes by chunk id, keeping the first occurrence', () => { - const fromVector = makeResult('chunk-1', 0.2) - const fromKeyword = { ...makeResult('chunk-1', 0.9), content: 'stale copy' } - - const fused = fuseByReciprocalRank([[fromVector], [fromKeyword]], 10) - - expect(fused).toHaveLength(1) - expect(fused[0].content).toBe('content-chunk-1') - expect(fused[0].distance).toBe(0.2) - }) - - it('scores by reciprocal rank so a deep double hit beats a shallow single hit', () => { - const deepShared = makeResult('deep-shared') - const topSingle = makeResult('top-single') - - /** - * `deep-shared` sits at rank 2 in both legs: 2 / (RRF_K + 2). - * `top-single` sits at rank 1 in one leg only: 1 / (RRF_K + 1). - * With RRF_K = 60 the double hit wins. - */ - expect(2 / (RRF_K + 2)).toBeGreaterThan(1 / (RRF_K + 1)) - - const fused = fuseByReciprocalRank( - [ - [topSingle, deepShared], - [makeResult('other'), deepShared], - ], - 10 - ) - - expect(fused[0].id).toBe('deep-shared') - }) - - it('does not let the first leg starve the second at small topK', () => { - const lexicalOnly = makeResult('lexical-only') - const vectorOnly = makeResult('vector-only') - - /** - * Rank 1 in each leg scores identically. Ordering by score alone would - * always emit the first list's row, so a `topK: 1` hybrid search would - * return exactly what vector-only search already returned. - */ - expect(fuseByReciprocalRank([[lexicalOnly], [vectorOnly]], 1).map((r) => r.id)).toEqual([ - 'lexical-only', - ]) - expect(fuseByReciprocalRank([[lexicalOnly], [vectorOnly]], 2).map((r) => r.id)).toEqual([ - 'lexical-only', - 'vector-only', - ]) - }) - - it('interleaves tied ranks so neither leg monopolizes the head', () => { - const legA = [makeResult('a1'), makeResult('a2'), makeResult('a3')] - const legB = [makeResult('b1'), makeResult('b2'), makeResult('b3')] - - expect(fuseByReciprocalRank([legA, legB], 6).map((r) => r.id)).toEqual([ - 'a1', - 'b1', - 'a2', - 'b2', - 'a3', - 'b3', - ]) - }) - - it('does not let a shared top hit evict the lexical-only row at topK 2', () => { - const shared = makeResult('shared') - const lexicalOnly = makeResult('lexical-only') - const vectorOnly = makeResult('vector-only') - - /** - * `shared` is rank 1 in both legs. Crediting it to only one leg would - * leave the round-robin owing the other leg the remaining slot, evicting - * the row that only the shared hit's leg could produce. - */ - const fused = fuseByReciprocalRank( - [ - [shared, lexicalOnly], - [shared, vectorOnly], - ], - 2 - ) - - expect(fused.map((r) => r.id)).toEqual(['shared', 'lexical-only']) - }) - }) - - describe('executeKeywordSearch', () => { - beforeEach(() => { - resetDbChainMock() - }) - - it('returns nothing for a whitespace-only query without touching the database', async () => { - const results = await executeKeywordSearch({ - knowledgeBaseIds: ['kb-123'], - access: WORKSPACE_ACCESS_SCOPE, - topK: 10, - query: ' ', - queryVector: JSON.stringify([0.1, 0.2, 0.3]), - }) - - expect(results).toEqual([]) - expect(dbChainMockFns.select).not.toHaveBeenCalled() - }) - - it('issues one query per knowledge base once the parallel threshold is crossed', async () => { - const knowledgeBaseIds = ['kb-1', 'kb-2', 'kb-3', 'kb-4', 'kb-5'] - expect(getQueryStrategy(knowledgeBaseIds.length, 10).useParallel).toBe(true) - - await executeKeywordSearch({ - knowledgeBaseIds, - access: WORKSPACE_ACCESS_SCOPE, - topK: 10, - query: 'PROJ-1234', - queryVector: JSON.stringify([0.1, 0.2, 0.3]), - }) - - /** Keyword retrieval preserves its existing per-base lexical candidate selection. */ - expect(dbChainMockFns.select).toHaveBeenCalledTimes(knowledgeBaseIds.length) - }) - }) - - describe('executeKnowledgeSearch', () => { - beforeEach(() => { - resetDbChainMock() - }) - - it('runs both legs and fuses them in hybrid mode', async () => { - /** - * Vector ranking is raw SQL throughout and consumes no table chain. Keyword ranking and - * hydration complete before vector content hydration. - */ - dbChainMockFns.execute.mockResolvedValue([{ id: 'vector-hit' }]) - queueTableRows(schemaMock.embedding, [{ id: 'keyword-hit', keywordRank: 0.9 }]) - queueTableRows(schemaMock.embedding, [makeResult('keyword-hit')]) - queueTableRows(schemaMock.embedding, [{ id: 'vector-hit' }]) - queueTableRows(schemaMock.embedding, [makeResult('vector-hit')]) - - const results = await executeKnowledgeSearch({ - knowledgeBaseIds: ['kb-123'], - access: WORKSPACE_ACCESS_SCOPE, - topK: 10, - searchMode: 'hybrid', - query: 'PROJ-1234', - queryVector: { vector: JSON.stringify(TEST_EMBEDDING), dimensions: 1536 }, - }) - - expect(results.map((r) => r.id).sort()).toEqual(['keyword-hit', 'vector-hit']) - expect(dbChainMockFns.select).toHaveBeenCalledTimes(3) - }) - - it('propagates unexpected keyword errors after the vector leg finishes', async () => { - /** The failing ranking chain is still built first and takes the first queued set. */ - dbChainMockFns.execute.mockResolvedValue([{ id: 'vector-hit' }]) - queueTableRows(schemaMock.embedding, [{ id: 'never-ranked', keywordRank: 0 }]) - queueTableRows(schemaMock.embedding, [{ id: 'vector-hit' }]) - queueTableRows(schemaMock.embedding, [makeResult('vector-hit')]) - - const failure = new Error('tsquery failed') - dbChainMockFns.orderBy.mockImplementationOnce(() => { - throw failure - }) - - await expect( - executeKnowledgeSearch({ - knowledgeBaseIds: ['kb-123'], - access: WORKSPACE_ACCESS_SCOPE, - topK: 10, - searchMode: 'hybrid', - query: 'PROJ-1234', - queryVector: { vector: JSON.stringify(TEST_EMBEDDING), dimensions: 1536 }, - }) - ).rejects.toBe(failure) - expect(dbChainMockFns.select).toHaveBeenCalledTimes(2) - }) - - it('skips both query legs when only tag filters are provided', async () => { - queueTableRows(schemaMock.embedding, [makeResult('tag-hit')]) - - const results = await executeKnowledgeSearch({ - knowledgeBaseIds: ['kb-123'], - access: WORKSPACE_ACCESS_SCOPE, - topK: 10, - searchMode: 'hybrid', - structuredFilters: [ - { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'api' } as never, - ], - }) - - expect(results.map((r) => r.id)).toEqual(['tag-hit']) - expect(dbChainMockFns.select).toHaveBeenCalledTimes(1) - }) - }) - - describe('generateSearchEmbedding', () => { - it('should use Azure OpenAI when KB-specific config is provided', async () => { - const { env } = await import('@/lib/core/config/env') - Object.keys(env).forEach((key) => delete (env as any)[key]) - Object.assign(env, { - AZURE_OPENAI_API_KEY: 'test-azure-key', - AZURE_OPENAI_ENDPOINT: 'https://test.openai.azure.com', - AZURE_OPENAI_API_VERSION: '2024-12-01-preview', - KB_OPENAI_MODEL_NAME: 'text-embedding-ada-002', - OPENAI_API_KEY: 'test-openai-key', - }) - - mockNextEmbeddingResponse() - - const result = await generateSearchEmbedding('test query', DEFAULT_EMBEDDING_TARGET) - - expect(vi.mocked(fetch)).toHaveBeenCalledWith( - 'https://test.openai.azure.com/openai/deployments/text-embedding-ada-002/embeddings?api-version=2024-12-01-preview', - expect.objectContaining({ - headers: expect.objectContaining({ - 'api-key': 'test-azure-key', - }), - }) - ) - expect(result.embedding).toEqual(TEST_EMBEDDING) - - // Clean up - Object.keys(env).forEach((key) => delete (env as any)[key]) - }) - - it('falls back to OpenAI when AZURE_OPENAI_API_VERSION is not set', async () => { - const { env } = await import('@/lib/core/config/env') - Object.keys(env).forEach((key) => delete (env as any)[key]) - Object.assign(env, { - AZURE_OPENAI_API_KEY: 'test-azure-key', - AZURE_OPENAI_ENDPOINT: 'https://test.openai.azure.com', - KB_OPENAI_MODEL_NAME: 'custom-embedding-model', - OPENAI_API_KEY: 'test-openai-key', - }) - - mockNextEmbeddingResponse() - - await generateSearchEmbedding('test query', DEFAULT_EMBEDDING_TARGET) - - expect(vi.mocked(fetch)).toHaveBeenCalledWith( - 'https://api.openai.com/v1/embeddings', - expect.any(Object) - ) - - // Clean up - Object.keys(env).forEach((key) => delete (env as any)[key]) - }) - - it('should throw error when no API configuration provided', async () => { - const { env } = await import('@/lib/core/config/env') - Object.keys(env).forEach((key) => delete (env as any)[key]) - Object.assign(env, { - OPENAI_API_KEY: undefined, - OPENAI_API_KEY_1: undefined, - OPENAI_API_KEY_2: undefined, - OPENAI_API_KEY_3: undefined, - OPENROUTER_API_KEY: undefined, - }) - - await expect(generateSearchEmbedding('test query', DEFAULT_EMBEDDING_TARGET)).rejects.toThrow( - 'Semantic retrieval is unavailable because its embedding provider is not configured.' - ) - }) - - it('projects verified provenance only in the model-bound embedding payload', async () => { - Object.keys(env).forEach((key) => delete (env as any)[key]) - Object.assign(env, { OPENAI_API_KEY: 'test-openai-key' }) - mockNextEmbeddingResponse() - - const registry = new ResolvedSecretTraceRegistry([ - { name: 'TOKEN', plaintext: 'secret-value', encryptedValue: 'encrypted-token' }, - ]) - registry.recordResolved('TOKEN', 'secret-value') - - await runWithKnowledgeModelInputProvenance(registry, () => - generateSearchEmbedding('prefix secret-value suffix', DEFAULT_EMBEDDING_TARGET) - ) - - expect(vi.mocked(fetch)).toHaveBeenCalledWith( - 'https://api.openai.com/v1/embeddings', - expect.objectContaining({ - body: JSON.stringify({ - input: ['prefix {{TOKEN}} suffix'], - model: 'text-embedding-3-small', - encoding_format: 'base64', - dimensions: 1536, - }), - }) - ) - }) - }) -}) diff --git a/apps/sim/app/api/knowledge/utils.ts b/apps/sim/app/api/knowledge/utils.ts index 8e5ccfd7c57..a1a704c24dc 100644 --- a/apps/sim/app/api/knowledge/utils.ts +++ b/apps/sim/app/api/knowledge/utils.ts @@ -99,17 +99,3 @@ export async function checkKnowledgeBaseAccess( ): Promise { return resolveKnowledgeBaseAccess(knowledgeBaseId, userId, false) } - -/** - * Check if a user has write access to a knowledge base. - * - * Write access is granted if: - * 1. KB has a workspace: user has write or admin permissions on that workspace - * 2. KB has no workspace (legacy): user owns the KB directly - */ -export async function checkKnowledgeBaseWriteAccess( - knowledgeBaseId: string, - userId: string -): Promise { - return resolveKnowledgeBaseAccess(knowledgeBaseId, userId, true) -} diff --git a/apps/sim/app/api/v1/knowledge/search/route.ts b/apps/sim/app/api/v1/knowledge/search/route.ts index 0ea92e767f0..2a1b97b5b3e 100644 --- a/apps/sim/app/api/v1/knowledge/search/route.ts +++ b/apps/sim/app/api/v1/knowledge/search/route.ts @@ -15,15 +15,16 @@ import { recordSearchEmbeddingUsage, } from '@/lib/knowledge/embeddings' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' +import type { SearchResult } from '@/lib/knowledge/search/candidates' import { resolveKnowledgeSearchDefaults } from '@/lib/knowledge/search/defaults' import { type KnowledgeRetrievalResult, retrieveKnowledgeSearch, - type SearchResult, } from '@/lib/knowledge/search/queries' import { getDocumentTagDefinitions } from '@/lib/knowledge/tags/service' import { buildUndefinedTagsError, validateTagValue } from '@/lib/knowledge/tags/utils' import type { StructuredFilter } from '@/lib/knowledge/types' +import { usesIndexedRetrieval } from '@/lib/sim-search/indexed/gate' import { checkKnowledgeBaseAccess, type KnowledgeBaseAccessResult } from '@/app/api/knowledge/utils' import { handleError, resolveV1KnowledgeReadAccess } from '@/app/api/v1/knowledge/utils' import { @@ -250,6 +251,7 @@ export const POST = withRouteHandler(async (request: NextRequest) => { accessProvider, searchMode, boostRecency, + indexedRetrieval: usesIndexedRetrieval(accessibleKbs), structuredFilters, }) } else if (hasQuery) { @@ -266,7 +268,7 @@ export const POST = withRouteHandler(async (request: NextRequest) => { accessProvider, searchMode, boostRecency, - searchIndexOnly: accessibleKbs.every((kb) => kb.isSearchIndex), + indexedRetrieval: usesIndexedRetrieval(accessibleKbs), query, queryVector: { vector: JSON.stringify(queryEmbeddingResult.embedding), diff --git a/apps/sim/app/o/[organizationId]/integrations/github-member-integration.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/github-member-integration.tsx similarity index 100% rename from apps/sim/app/o/[organizationId]/integrations/github-member-integration.tsx rename to apps/sim/app/o/[organizationId]/integrations/indexed/github-member-integration.tsx diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/index.ts b/apps/sim/app/o/[organizationId]/integrations/indexed/index.ts new file mode 100644 index 00000000000..9b1281fd906 --- /dev/null +++ b/apps/sim/app/o/[organizationId]/integrations/indexed/index.ts @@ -0,0 +1 @@ +export { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/indexed/member-integrations-list' diff --git a/apps/sim/app/o/[organizationId]/integrations/member-integration-row.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx similarity index 97% rename from apps/sim/app/o/[organizationId]/integrations/member-integration-row.tsx rename to apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx index ea8cd4a3a22..6619a285d71 100644 --- a/apps/sim/app/o/[organizationId]/integrations/member-integration-row.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx @@ -3,18 +3,18 @@ import { Chip, ChipLink } from '@sim/emcn' import { organizationRoutes } from '@/lib/navigation/paths' import { connectorDisplayName } from '@/lib/sim-search/connectors' -import { getSearchSourceStatus } from '@/lib/sim-search/source-status' import { DisconnectAccountMenu } from '@/app/o/[organizationId]/integrations/disconnect-account-menu' +import { getSearchSourceStatus } from '@/app/o/[organizationId]/integrations/indexed/source-status' +import { + CONNECTABLE_MEMBERSHIPS, + enrollmentActionLabel, + type useMemberEnrollment, +} from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' import { IntegrationTile } from '@/app/workspace/[workspaceId]/integrations/components/integrations-showcase' import type { RowAction } from '@/app/workspace/[workspaceId]/settings/components/row-actions-menu' import { SettingsResourceRow } from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' import type { useSearchSources } from '@/hooks/queries/kb/connectors' -import { - CONNECTABLE_MEMBERSHIPS, - enrollmentActionLabel, - type useMemberEnrollment, -} from '@/hooks/use-member-enrollment' interface MemberIntegrationRowProps { organizationId: string diff --git a/apps/sim/app/o/[organizationId]/integrations/member-integrations-list.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx similarity index 98% rename from apps/sim/app/o/[organizationId]/integrations/member-integrations-list.tsx rename to apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx index debfcbd6819..bcb1692d8eb 100644 --- a/apps/sim/app/o/[organizationId]/integrations/member-integrations-list.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx @@ -10,8 +10,9 @@ import { SEARCH_SOURCE_TYPES, type SearchConnector, } from '@/lib/sim-search/connectors' -import { GitHubMemberIntegration } from '@/app/o/[organizationId]/integrations/github-member-integration' -import { MemberIntegrationRow } from '@/app/o/[organizationId]/integrations/member-integration-row' +import { GitHubMemberIntegration } from '@/app/o/[organizationId]/integrations/indexed/github-member-integration' +import { MemberIntegrationRow } from '@/app/o/[organizationId]/integrations/indexed/member-integration-row' +import { useMemberEnrollment } from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' import { SourceSetupModal } from '@/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal' import { @@ -27,7 +28,6 @@ import { import { usePersonalSearchIntegrations } from '@/hooks/queries/personal-search-integrations' import { useSearchIntegrations } from '@/hooks/queries/search-integrations' import { searchSourceKeys } from '@/hooks/queries/utils/search-source-keys' -import { useMemberEnrollment } from '@/hooks/use-member-enrollment' import { usePermissionConfig } from '@/hooks/use-permission-config' interface MemberIntegrationsListProps { @@ -243,7 +243,7 @@ function MemberIntegration({ mirroredAccessAvailable={mirroredAccessAvailable} onCreate={ canCreate && connector - ? () => enrollment.connectSearchSource(scope, connector, undefined) + ? () => enrollment.connectSearchSource(scope, connector) : undefined } addLabel={ diff --git a/apps/sim/lib/sim-search/source-status.ts b/apps/sim/app/o/[organizationId]/integrations/indexed/source-status.ts similarity index 100% rename from apps/sim/lib/sim-search/source-status.ts rename to apps/sim/app/o/[organizationId]/integrations/indexed/source-status.ts diff --git a/apps/sim/hooks/use-member-enrollment.test.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx similarity index 99% rename from apps/sim/hooks/use-member-enrollment.test.tsx rename to apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx index 690423054f9..7bba56306ae 100644 --- a/apps/sim/hooks/use-member-enrollment.test.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx @@ -24,7 +24,7 @@ const mocks = vi.hoisted(() => ({ vi.mock('@tanstack/react-query', () => reactQueryMock) vi.mock('@/hooks/queries/kb/connectors', () => kbConnectorsQueriesMock) -import { useMemberEnrollment } from '@/hooks/use-member-enrollment' +import { useMemberEnrollment } from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' type Enrollment = ReturnType diff --git a/apps/sim/hooks/use-member-enrollment.ts b/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts similarity index 84% rename from apps/sim/hooks/use-member-enrollment.ts rename to apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts index 0b62b215c8a..e6d0376bf55 100644 --- a/apps/sim/hooks/use-member-enrollment.ts +++ b/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts @@ -4,7 +4,6 @@ import { useCallback, useEffect, useRef, useState } from 'react' import { createLogger } from '@sim/logger' import { generateId } from '@sim/utils/id' import { type QueryKey, useQueryClient } from '@tanstack/react-query' -import type { WorkspaceMemberConnector } from '@/lib/api/contracts/knowledge/connectors' import { type ResourceScope, resourceScopeFields, @@ -15,7 +14,6 @@ import { credentialGroupOAuthCompletionChannel, isCredentialGroupOAuthFailure, } from '@/lib/credential-groups/oauth-completion' -import type { MemberSyncStatus } from '@/lib/knowledge/types' import type { SearchConnector } from '@/lib/sim-search/connectors' import { useConnectSimSearchConnector, @@ -47,52 +45,6 @@ export function enrollmentActionLabel( return membership === 'needs_reauth' ? 'Reconnect' : 'Connect' } -interface DescribeMembershipInput { - membership: ViewerConnectorMembership - memberSyncStatus: MemberSyncStatus - /** Whether this surface opened an enrollment tab that has not connected yet. */ - waiting: boolean - /** The connector's display name. */ - name: string -} - -/** - * One sentence on where the viewer stands with a per-member connector, shared - * by every surface that shows it so the wording cannot drift between them. - * Null once the viewer is connected and nothing is happening for them. - */ -export function describeMembership({ - membership, - memberSyncStatus, - waiting, - name, -}: DescribeMembershipInput): string | null { - switch (membership) { - case 'connected': - switch (memberSyncStatus) { - case 'pending': - case 'running': - return `Syncing the ${name} documents shared with you. They appear when the sync completes.` - case 'error': - return `The last ${name} sync failed; the documents you already have stay visible while it retries.` - case 'disabled': - return `Syncing ${name} per member is turned off. Ask a workspace admin to turn it back on.` - default: - return null - } - case 'needs_reauth': - return `Reconnect your ${name} account to keep seeing the documents shared with you.` - case 'unverified_email': - return `Verify your email address to see the ${name} documents shared with you.` - case 'revoked': - return `A workspace admin removed your access to ${name} documents.` - default: - return waiting - ? `Finish connecting your ${name} account in the other tab.` - : `Connect your ${name} account to see the documents shared with you.` - } -} - /** An enrollment tab this surface opened that has not connected yet. */ interface AwaitingEnrollment { since: number @@ -371,19 +323,10 @@ export function useMemberEnrollment({ const [setupConnector, setSetupConnector] = useState(null) /** - * One click on a Sim Search source: enroll in its connector when someone - * already connected it, ask for its setup fields when it needs them, and - * otherwise create it and enroll in one step. + * One click on a new Sim Search source: ask for its setup fields when it + * needs them, and otherwise create it and enroll in one step. */ - const connectSearchSource = ( - owner: string | ResourceScope, - connector: SearchConnector, - connection: WorkspaceMemberConnector | undefined - ) => { - if (connection) { - connect(connection.knowledgeBaseId, connection.connectorId) - return - } + const connectSearchSource = (owner: string | ResourceScope, connector: SearchConnector) => { if (connector.setupFields.length > 0) { setSetupConnector(connector) return diff --git a/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx b/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx index 43d842383f1..aad02cc2093 100644 --- a/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx @@ -104,7 +104,7 @@ vi.mock('@/app/o/[organizationId]/integrations/disconnect-account-menu', () => ( }, })) vi.mock('@/hooks/queries/kb/connectors', () => kbConnectorsQueriesMock) -vi.mock('@/hooks/use-member-enrollment', () => ({ +vi.mock('@/app/o/[organizationId]/integrations/indexed/use-member-enrollment', () => ({ enrollmentActionLabel: (membership: string, waiting: boolean) => waiting ? 'Open again' : membership === 'needs_reauth' ? 'Reconnect' : 'Connect', CONNECTABLE_MEMBERSHIPS: new Set(['invited', 'not_enrolled', 'needs_reauth']), @@ -126,8 +126,8 @@ vi.mock('@/hooks/use-oauth-return', () => ({ useOAuthReturnRouter: () => undefined, })) +import { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/indexed' import { OrganizationIntegrations } from '@/app/o/[organizationId]/integrations/integrations' -import { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/member-integrations-list' import { type RowAction, RowActionsMenu, @@ -803,8 +803,7 @@ describe('grouped member integrations', () => { await act(async () => buttons('Connect')[0].click()) expect(mocks.connectSearchSource).toHaveBeenCalledWith( scope, - expect.objectContaining({ type: 'gmail' }), - undefined + expect.objectContaining({ type: 'gmail' }) ) expect(document.querySelector('[role="dialog"]')).toBeNull() }) @@ -857,8 +856,7 @@ describe('grouped member integrations', () => { } else { expect(mocks.connectSearchSource).toHaveBeenCalledExactlyOnceWith( scope, - expect.objectContaining({ type: 'slack' }), - undefined + expect.objectContaining({ type: 'slack' }) ) expect(mocks.connect).not.toHaveBeenCalled() } diff --git a/apps/sim/app/o/[organizationId]/integrations/integrations.tsx b/apps/sim/app/o/[organizationId]/integrations/integrations.tsx index 8627de69e2a..5a92f870368 100644 --- a/apps/sim/app/o/[organizationId]/integrations/integrations.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/integrations.tsx @@ -5,8 +5,8 @@ import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-t import { SEARCH_DEBOUNCE_MS } from '@/lib/url-state' import { OrganizationPage } from '@/app/o/[organizationId]/components/organization-page' import { useOrganizationPageFilters } from '@/app/o/[organizationId]/components/organization-page/use-organization-page-filters' +import { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/indexed' import { LiveMemberIntegrations } from '@/app/o/[organizationId]/integrations/live-member-integrations' -import { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/member-integrations-list' import { SlackSearchActions } from '@/app/o/[organizationId]/integrations/slack-search-actions' import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' import { SearchIntegrationConnection } from '@/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/search-integration-connection' diff --git a/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx b/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx index 6cd806b52bc..35ad5bdf2f7 100644 --- a/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx +++ b/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx @@ -4,7 +4,8 @@ import type { SearchParams } from 'nuqs/server' import { readSearchDocumentResultSchema } from '@/lib/api/contracts/knowledge/documents' import { getSession } from '@/lib/auth' import { OrchestrationError } from '@/lib/core/orchestration/types' -import { readSearchDocument } from '@/lib/knowledge/application/read-search-document' +import { readSearchDocument } from '@/lib/sim-search/indexed' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { buildAuthCrossLink } from '@/app/(auth)/auth-redirect' import { loadDocumentReadParams, @@ -22,6 +23,7 @@ export default async function OrganizationDocumentPage({ params, searchParams, }: OrganizationDocumentPageProps) { + if (!isIndexedOrgSearchEnabled()) notFound() const { organizationId, knowledgeBaseId, documentId } = await params const position = await loadDocumentReadParams(searchParams, { strict: true }).catch(() => notFound() diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts new file mode 100644 index 00000000000..658a877fbc1 --- /dev/null +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts @@ -0,0 +1 @@ +export { IndexedOrganizationIntegrationsSettings } from '@/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings' diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx new file mode 100644 index 00000000000..714b6b164ec --- /dev/null +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx @@ -0,0 +1,139 @@ +'use client' + +import { useState } from 'react' +import { Chip, ChipConfirmModal, ChipModalError, ChipSwitch, toast } from '@sim/emcn' +import { useQueryStates } from 'nuqs' +import { getOrganizationAccountUpdateOptions } from '@/lib/credential-groups/organization-account-options' +import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' +import { OrganizationIntegrationsSetup } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup' +import { OrganizationSourcePeople } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people' +import { OrganizationSourceStats } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats' +import { + organizationIntegrationsTabParam, + organizationPeopleIntegrationParam, +} from '@/app/o/[organizationId]/settings/components/integrations/search-params' +import { RowActionsMenu } from '@/app/workspace/[workspaceId]/settings/components/row-actions-menu' +import { + SettingsEmptyState, + SettingsQueryErrorState, +} from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' +import { + useOrganizationAccounts, + useUpdateOrganizationAccounts, +} from '@/hooks/queries/organization-accounts' + +/** + * Organization Integrations settings for indexed organization search: the Sources, People, and + * Stats tabs over the connectors that crawl the search index. Rendered only while the deployment + * serves indexed search (`features.liveEnterpriseSearch === false`). + */ +export function IndexedOrganizationIntegrationsSettings() { + const { organization, viewer } = useOrganizationContext() + const [{ tab }, setNavigation] = useQueryStates({ + [organizationIntegrationsTabParam.key]: organizationIntegrationsTabParam.parser, + [organizationPeopleIntegrationParam.key]: organizationPeopleIntegrationParam.parser, + }) + const accounts = useOrganizationAccounts(viewer.isAdmin ? organization.id : undefined) + const update = useUpdateOrganizationAccounts() + const [refreshOpen, setRefreshOpen] = useState(false) + const group = accounts.data?.credentialGroup + const refreshConnections = () => { + if (!group || update.isPending) return + update.mutate( + { + organizationId: organization.id, + groupId: group.id, + update: { options: getOrganizationAccountUpdateOptions(group) }, + }, + { + onSuccess: () => { + setRefreshOpen(false) + toast.success('Sign-in settings updated') + }, + } + ) + } + if (!viewer.isAdmin) return null + + const tabs = ( + void setNavigation({ tab: value, integration: null })} + options={[ + { value: 'providers', label: 'Sources' }, + { value: 'people', label: 'People' }, + { value: 'stats', label: 'Stats' }, + ]} + /> + ) + + return ( +
+ {tab === 'providers' && ( +
+ {tabs} + {!accounts.error && group && group.options.length > 0 && ( + { + update.reset() + setRefreshOpen(true) + }, + }, + ]} + /> + )} +
+ )} + { + if (!update.isPending) setRefreshOpen(open) + }} + title='Update sign-in settings?' + text='Apply Sim’s current OAuth app and permission settings to member connections. People whose settings changed must reconnect. This does not sync content.' + confirm={{ label: 'Update', pending: update.isPending, onClick: refreshConnections }} + > + {update.error?.message} + + {tab === 'providers' && } + {tab === 'stats' && } + {tab === 'people' && ( + void accounts.refetch()} + variant='inline' + /> + ) : !accounts.data ? ( + Loading connected accounts + ) : !accounts.data.credentialGroup ? ( +
+ + Add a source that uses member accounts before requesting connections. + + void setNavigation({ tab: 'providers', integration: null })}> + View sources + +
+ ) : undefined + } + /> + )} +
+ ) +} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-setup.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup.tsx similarity index 100% rename from apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-setup.tsx rename to apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup.tsx diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-search-stats-period.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period.tsx similarity index 100% rename from apps/sim/app/o/[organizationId]/settings/components/integrations/organization-search-stats-period.tsx rename to apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period.tsx diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-source-people.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people.tsx similarity index 100% rename from apps/sim/app/o/[organizationId]/settings/components/integrations/organization-source-people.tsx rename to apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people.tsx diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-source-stats.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx similarity index 99% rename from apps/sim/app/o/[organizationId]/settings/components/integrations/organization-source-stats.tsx rename to apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx index 79c91fbe163..0048318bfaa 100644 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-source-stats.tsx +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx @@ -9,7 +9,7 @@ import { SEARCH_STATS_SURFACE_LABELS, SEARCH_STATS_SURFACES, } from '@/lib/knowledge/search/stats' -import { OrganizationSearchStatsPeriod } from '@/app/o/[organizationId]/settings/components/integrations/organization-search-stats-period' +import { OrganizationSearchStatsPeriod } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period' import { SettingsEmptyState } from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' import { SettingsPanel } from '@/app/workspace/[workspaceId]/settings/components/settings-panel' import { diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx index ae8a8853050..d5c61f0370a 100644 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx @@ -1,6 +1,7 @@ /** @vitest-environment jsdom */ import { act } from 'react' import { toast } from '@sim/emcn' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { organizationAccountsQueriesMock, organizationAccountsQueriesMockFns, @@ -23,11 +24,11 @@ const mocks = vi.hoisted(() => ({ })) vi.mock('@/app/o/[organizationId]/providers/organization-provider', () => organizationProviderMock) vi.mock( - '@/app/o/[organizationId]/settings/components/integrations/organization-integrations-setup', + '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup', () => ({ OrganizationIntegrationsSetup: () =>
Provider setup
}) ) vi.mock( - '@/app/o/[organizationId]/settings/components/integrations/organization-source-stats', + '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats', () => ({ OrganizationSourceStats: ({ organizationId }: { organizationId: string }) => (
Stats for {organizationId}
@@ -37,6 +38,7 @@ vi.mock( vi.mock('@/hooks/queries/organization-accounts', () => organizationAccountsQueriesMock) import { SettingsHeaderProvider, SettingsHeaderShell } from '@/components/settings/settings-header' +import { resetDeploymentShape } from '@/lib/core/config/deployment-shape' import { OrganizationIntegrationsSettings } from '@/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings' const mockContext = organizationProviderMockFns.mockUseOrganizationContext @@ -66,6 +68,9 @@ describe('organization integration invitations', () => { let container: HTMLDivElement beforeEach(() => { + /** These cover the indexed organization Integrations settings. */ + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + resetDeploymentShape() vi.spyOn(toast, 'success').mockReturnValue('toast-id') vi.spyOn(toast, 'error').mockReturnValue('toast-id') mocks.updatePending = false @@ -94,6 +99,8 @@ describe('organization integration invitations', () => { afterEach(async () => { await act(async () => root.unmount()) container.remove() + resetEnvFlagsMock() + resetDeploymentShape() }) async function render(searchParams = '') { diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx index 16e047b9bc5..58942c418ad 100644 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx @@ -1,28 +1,8 @@ 'use client' -import { useState } from 'react' -import { Chip, ChipConfirmModal, ChipModalError, ChipSwitch, toast } from '@sim/emcn' -import { useQueryStates } from 'nuqs' import { useDeploymentShape } from '@/lib/core/config/deployment-shape' -import { getOrganizationAccountUpdateOptions } from '@/lib/credential-groups/organization-account-options' -import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' +import { IndexedOrganizationIntegrationsSettings } from '@/app/o/[organizationId]/settings/components/integrations/indexed' import { LiveSearchSettings } from '@/app/o/[organizationId]/settings/components/integrations/live-search-settings' -import { OrganizationIntegrationsSetup } from '@/app/o/[organizationId]/settings/components/integrations/organization-integrations-setup' -import { OrganizationSourcePeople } from '@/app/o/[organizationId]/settings/components/integrations/organization-source-people' -import { OrganizationSourceStats } from '@/app/o/[organizationId]/settings/components/integrations/organization-source-stats' -import { - organizationIntegrationsTabParam, - organizationPeopleIntegrationParam, -} from '@/app/o/[organizationId]/settings/components/integrations/search-params' -import { RowActionsMenu } from '@/app/workspace/[workspaceId]/settings/components/row-actions-menu' -import { - SettingsEmptyState, - SettingsQueryErrorState, -} from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' -import { - useOrganizationAccounts, - useUpdateOrganizationAccounts, -} from '@/hooks/queries/organization-accounts' export function OrganizationIntegrationsSettings() { const { features } = useDeploymentShape() @@ -32,114 +12,3 @@ export function OrganizationIntegrationsSettings() { ) } - -function IndexedOrganizationIntegrationsSettings() { - const { organization, viewer } = useOrganizationContext() - const [{ tab }, setNavigation] = useQueryStates({ - [organizationIntegrationsTabParam.key]: organizationIntegrationsTabParam.parser, - [organizationPeopleIntegrationParam.key]: organizationPeopleIntegrationParam.parser, - }) - const accounts = useOrganizationAccounts(viewer.isAdmin ? organization.id : undefined) - const update = useUpdateOrganizationAccounts() - const [refreshOpen, setRefreshOpen] = useState(false) - const group = accounts.data?.credentialGroup - const refreshConnections = () => { - if (!group || update.isPending) return - update.mutate( - { - organizationId: organization.id, - groupId: group.id, - update: { options: getOrganizationAccountUpdateOptions(group) }, - }, - { - onSuccess: () => { - setRefreshOpen(false) - toast.success('Sign-in settings updated') - }, - } - ) - } - if (!viewer.isAdmin) return null - - const tabs = ( - void setNavigation({ tab: value, integration: null })} - options={[ - { value: 'providers', label: 'Sources' }, - { value: 'people', label: 'People' }, - { value: 'stats', label: 'Stats' }, - ]} - /> - ) - - return ( -
- {tab === 'providers' && ( -
- {tabs} - {tab === 'providers' && !accounts.error && group && group.options.length > 0 && ( - { - update.reset() - setRefreshOpen(true) - }, - }, - ]} - /> - )} -
- )} - { - if (!update.isPending) setRefreshOpen(open) - }} - title='Update sign-in settings?' - text='Apply Sim’s current OAuth app and permission settings to member connections. People whose settings changed must reconnect. This does not sync content.' - confirm={{ label: 'Update', pending: update.isPending, onClick: refreshConnections }} - > - {update.error?.message} - - {tab === 'providers' && } - {tab === 'stats' && } - {tab === 'people' && ( - void accounts.refetch()} - variant='inline' - /> - ) : !accounts.data ? ( - Loading connected accounts - ) : !accounts.data.credentialGroup ? ( -
- - Add a source that uses member accounts before requesting connections. - - void setNavigation({ tab: 'providers', integration: null })}> - View sources - -
- ) : undefined - } - /> - )} -
- ) -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx index db898b601a4..e228074fefb 100644 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx @@ -4,6 +4,7 @@ import { act, cloneElement, type ReactNode } from 'react' import { authClientMock, authClientMockFns } from '@sim/testing/mocks/auth-client.mock' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { kbConnectorsQueriesMock, kbConnectorsQueriesMockFns, @@ -405,9 +406,16 @@ afterEach(async () => { container?.remove() root = null container = null + resetEnvFlagsMock() resetDeploymentShape() }) +/** Selects the indexed backend, whose arms of these dialogs index and mirror sources. */ +function selectIndexedSearch() { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + resetDeploymentShape() +} + describe('Search source setup with real connector dialogs', () => { it.each([ ['source-one', '/o/org-1/settings/integrations/sources/source-one'], @@ -433,6 +441,7 @@ describe('Search source setup with real connector dialogs', () => { ) it('prepares organization connected-account indexing in members mode even when central access is available', async () => { + selectIndexedSearch() mocks.bases = [] await render( { it.each(['admin', 'members'] as const)( 'shows and saves Gmail’s Search default date window in %s mode', async (accessMode) => { + selectIndexedSearch() mocks.credentials = [ { id: 'gmail-service', @@ -919,6 +929,7 @@ describe('administrator source prerequisites in real connector dialogs', () => { ])( 'requires the Directory administrator email in $type administrator mode and refuses empty or blank subjects', async ({ type, provider }) => { + selectIndexedSearch() mocks.credentials = [{ ...driveCredential, provider }] await render( { ])( 'excludes personal OAuth accounts and stale OAuth drafts from $type administrator setup', async ({ type, provider, name }) => { + selectIndexedSearch() const oauthCredential = { id: 'drive-personal', name: 'Personal Drive account', @@ -1086,6 +1098,7 @@ describe('administrator source prerequisites in real connector dialogs', () => { ) it('does not let an administrator erase the crawl subject from an existing mirrored Drive source', async () => { + selectIndexedSearch() await render( { }) it('defaults an OAuth source to member accounts and never offers workspace-wide access', async () => { + selectIndexedSearch() await render( ({ authorize: vi.fn() })) vi.mock('@/lib/settings/application/organization-section-access', () => ({ @@ -26,10 +27,13 @@ import OrganizationProviderPage from '@/app/o/[organizationId]/settings/integrat const mockRedirect = nextNavigationMockFns.mockRedirect const mockGetSession = authMockFns.mockGetSession +/** Member providers such as Jira keep a provider page only under indexed organization search. */ beforeEach(() => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) mockGetSession.mockResolvedValue({ user: { id: 'admin-1' } }) mocks.authorize.mockResolvedValue(true) }) +afterEach(resetEnvFlagsMock) it.each(['jira', 'confluence'])( 'moves legacy %s Accounts links to filtered People and preserves the search', diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/index.ts b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/index.ts index 87b655dba7b..b3bb6255fa5 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/index.ts +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/index.ts @@ -1,4 +1 @@ -export { - groupResultsByDocument, - KnowledgeSearchResults, -} from './knowledge-search-results' +export { KnowledgeSearchResults } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results' diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/index.ts b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/index.ts new file mode 100644 index 00000000000..9478d3fe8a5 --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/index.ts @@ -0,0 +1 @@ +export { IndexedSearchResults } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/indexed-search-results' diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/indexed-search-results.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/indexed-search-results.tsx new file mode 100644 index 00000000000..44518944580 --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/indexed-search-results.tsx @@ -0,0 +1,214 @@ +'use client' + +import { useEffect, useMemo, useState } from 'react' +import { Chip, ChipLink, cn } from '@sim/emcn' +import { useQueryStates } from 'nuqs' +import { ActivityStatus } from '@/components/ui/activity-status' +import { WORKSPACE_KNOWLEDGE_SEARCH_LIMITS } from '@/lib/api/contracts/knowledge' +import { connectorDisplayName } from '@/lib/sim-search/connectors' +import { SearchFilters } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-filters' +import { + groupResultsByDocument, + handleResultsKeyDown, + type SearchResultsProps, + toSource, +} from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/utils' +import { SourceCard } from '@/app/workspace/[workspaceId]/home/components/message-content/components/source-card' +import { + resourceUrlKeys, + searchFilterParsers, + searchFiltersFromParams, +} from '@/app/workspace/[workspaceId]/home/search-params' +import { useSearchIndex, useSearchSourceOverview } from '@/hooks/queries/kb/connectors' +import { useWorkspaceKnowledgeSearch } from '@/hooks/queries/kb/knowledge' + +/** Every result without a connector is an upload; the filter names them so. */ +const UPLOAD_SOURCE = 'upload' + +/** Results from indexed organization search, rendered only while the deployment serves it. */ +export function IndexedSearchResults({ + scope, + query, + onSummarize, + onSearchChange, + suppliedFilters, + topK, +}: SearchResultsProps) { + const [hasShownFilters, setHasShownFilters] = useState(false) + const [searchedAt] = useState(Date.now) + /** + * More results are a second, wider search: the first paint stays as quick as it is, and a + * refinement of the filters starts over at the first page. + */ + const [expandedFor, setExpandedFor] = useState(null) + const { + data: index, + isPending: basesPending, + isError: basesFailed, + isFetching: basesFetching, + refetch: refetchIndex, + } = useSearchIndex(scope) + const [filters] = useQueryStates(searchFilterParsers, resourceUrlKeys) + const custom = filters.updated === 'custom' + const pageFilters = useMemo( + () => searchFiltersFromParams(filters, searchedAt), + [filters.source, filters.updated, filters.from, filters.to, searchedAt] + ) + const searchFilters = suppliedFilters ?? pageFilters + const scopeId = scope.kind === 'organization' ? scope.organizationId : scope.workspaceId + useEffect(() => { + onSearchChange?.({ scope, query, filters: searchFilters, ...(topK ? { topK } : {}) }) + }, [scope.kind, scopeId, query, searchFilters, topK, onSearchChange]) + const filtersKey = JSON.stringify(searchFilters) + const expanded = expandedFor === filtersKey + /** A custom window is two-ended: until both days are chosen, nothing is searched. */ + const awaitingRange = !suppliedFilters && custom && !(filters.from && filters.to) + const { + data: search, + isPending, + isFetching, + isPlaceholderData, + isError: searchFailed, + refetch: refetchSearch, + } = useWorkspaceKnowledgeSearch( + scope, + awaitingRange ? '' : query, + searchFilters, + topK ?? + (expanded + ? WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.expanded + : WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.initial), + { retainAcrossLimits: topK === undefined } + ) + /** A full first page may collapse to few cards, yet more documents may still match. */ + const mayHaveMore = + topK === undefined && + !expanded && + (search?.results.length ?? 0) >= WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.initial + const { data: overview } = useSearchSourceOverview(scope) + const indexing = (overview?.providers ?? []) + .filter((provider) => provider.isSyncing) + .map((provider) => connectorDisplayName(provider.connectorType)) + const documents = groupResultsByDocument(search?.results ?? []) + const sourceTypes = [ + ...new Set([ + ...(filters.source ? [filters.source] : []), + ...(overview?.providers.map((provider) => provider.connectorType) ?? []), + UPLOAD_SOURCE, + ]), + ].sort((left, right) => connectorDisplayName(left).localeCompare(connectorDisplayName(right))) + const failed = basesFailed || searchFailed + const pending = basesPending || isPending + const fetching = basesFetching || isFetching + const noSources = !basesPending && !basesFailed && !index?.knowledgeBaseId + const partial = search?.retrieval.status === 'partial' + const documentCount = documents.length === 1 ? '1 document' : `${documents.length} documents` + + const indexingNote = + indexing.length > 0 + ? `Still indexing ${indexing.join(', ')}; results grow as documents land.` + : null + + const showResults = !noSources && !failed && !basesPending && documents.length > 0 + /** A custom window waiting for its days must show the filters, or the picker is unreachable. */ + const showFilters = + hasShownFilters || + showResults || + awaitingRange || + (!noSources && !pending && !failed && !!search && !partial) + if (showFilters && !hasShownFilters) setHasShownFilters(true) + + return noSources ? ( +
+

No sources are set up yet.

+ + View sources + +
+ ) : ( +
+
+
+ {awaitingRange ? ( +

+ Choose the days to search. +

+ ) : fetching ? ( + + ) : pending && !failed ? null : ( +

+ {failed + ? 'Search couldn’t run.' + : partial + ? documents.length === 0 + ? 'Search timed out.' + : `${documentCount} · some results may be missing.` + : documents.length === 0 + ? 'Search found no results.' + : `${documentCount} · searched as you`} +

+ )} + {indexingNote && !failed && !partial && ( +

{indexingNote}

+ )} +
+ {(failed || partial) && ( + void (basesFailed ? refetchIndex() : refetchSearch())} + > + {fetching ? 'Retrying' : 'Try again'} + + )} +
+ {suppliedFilters === undefined && showFilters && } + {showResults && ( +
+ {documents.map((result) => { + const source = toSource(result, query, scope) + return ( + + onSummarize(`Summarize "${cited.title ?? cited.url}"`, { + ...searchFilters, + documentIds: [result.documentId], + }) + } + /> + ) + })} + {mayHaveMore && ( +
+ setExpandedFor(filtersKey)} + > + Show more + +
+ )} +
+ )} +
+ ) +} diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx index afd2f6e3c84..0505039e054 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx @@ -1,111 +1,37 @@ 'use client' import { useEffect, useMemo, useState } from 'react' -import { Chip, ChipLink, cn } from '@sim/emcn' +import { Chip } from '@sim/emcn' import { useQueryStates } from 'nuqs' import { ActivityStatus } from '@/components/ui/activity-status' -import { - WORKSPACE_KNOWLEDGE_SEARCH_LIMITS, - type WorkspaceKnowledgeSearchResult, - type WorkspaceSearchFilters, -} from '@/lib/api/contracts/knowledge' +import type { WorkspaceSearchFilters } from '@/lib/api/contracts/knowledge' import { useSession } from '@/lib/auth/auth-client' import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import { type ResourceScope, resourceScopeKey } from '@/lib/core/resource-scope' -import { getBaseUrl } from '@/lib/core/utils/urls' -import { matchSnippet } from '@/lib/knowledge/search/snippet' -import type { SearchResource } from '@/lib/mothership/generated/resources' -import { connectorDisplayName } from '@/lib/sim-search/connectors' +import { IndexedSearchResults } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed' import { SearchFilters } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-filters' -import { SourceCard } from '@/app/workspace/[workspaceId]/home/components/message-content/components/source-card' import { - isHttpUrl, - type SourceTagData, -} from '@/app/workspace/[workspaceId]/home/components/message-content/components/special-tags' + groupResultsByDocument, + handleResultsKeyDown, + type SearchResultsProps, + toSource, +} from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/utils' +import { SourceCard } from '@/app/workspace/[workspaceId]/home/components/message-content/components/source-card' import { resourceUrlKeys, searchFilterParsers, searchFiltersFromParams, } from '@/app/workspace/[workspaceId]/home/search-params' -import { useSearchIndex, useSearchSourceOverview } from '@/hooks/queries/kb/connectors' import { useWorkspaceKnowledgeSearch } from '@/hooks/queries/kb/knowledge' -/** Every result without a connector is an upload; the filter names them so. */ -const UPLOAD_SOURCE = 'upload' - -/** - * One card per document, keeping the best-ranked chunk of each: the list is - * already in rank order, so the first chunk seen for a document is its best. - */ -export function groupResultsByDocument( - results: readonly WorkspaceKnowledgeSearchResult[] -): WorkspaceKnowledgeSearchResult[] { - const seen = new Set() - const grouped: WorkspaceKnowledgeSearchResult[] = [] - for (const result of results) { - if (seen.has(result.documentId)) continue - seen.add(result.documentId) - grouped.push(result) - } - return grouped -} - -/** - * A result as the source card renders it: the row's second line names the - * source app, or the knowledge base for an upload. Without an HTTP(S) source - * URL, the link opens the canonical document in Sim. - */ -function toSource( - result: WorkspaceKnowledgeSearchResult, - query: string, - scope: ResourceScope -): SourceTagData { - return { - url: isHttpUrl(result.sourceUrl) - ? result.sourceUrl - : `${getBaseUrl()}${scope.kind === 'organization' ? `/o/${encodeURIComponent(scope.organizationId)}` : `/workspace/${encodeURIComponent(scope.workspaceId)}`}/knowledge/${encodeURIComponent(result.knowledgeBaseId)}/${encodeURIComponent(result.documentId)}`, - title: result.documentName ?? undefined, - siteName: result.connectorType - ? connectorDisplayName(result.connectorType) - : result.knowledgeBaseName || undefined, - connectorType: result.connectorType ?? undefined, - snippet: matchSnippet(result.content, query), - author: result.author ?? undefined, - updatedAt: result.sourceDate ?? result.sourceModifiedAt ?? undefined, - } -} - -/** - * Arrow keys walk the result links, the way a search page does; Enter on a - * focused link opens it natively. Focus stops at either end. - */ -function handleResultsKeyDown(event: React.KeyboardEvent) { - if (event.key !== 'ArrowDown' && event.key !== 'ArrowUp') return - const links = [...event.currentTarget.querySelectorAll('a[data-source-link]')] - if (links.length === 0) return - const index = links.findIndex((link) => link === document.activeElement) - if (index < 0) return - const next = - event.key === 'ArrowDown' ? Math.min(index + 1, links.length - 1) : Math.max(index - 1, 0) - if (next === index) return - event.preventDefault() - links[next].focus() -} - type KnowledgeSearchResultsProps = ( | { workspaceId: string; scope?: never } | { scope: ResourceScope; workspaceId?: never } -) & { - query: string - /** A tool-owned search keeps its exact scope instead of inheriting page filters. */ - filters?: WorkspaceSearchFilters - topK?: number - nativeQueries?: SearchResource['nativeQueries'] - reuseFreshResult?: boolean - /** Binds the Assistant turn to the selected canonical document. */ - onSummarize: (prompt: string, filters: WorkspaceSearchFilters) => void - onSearchChange?: (search: SearchResource) => void -} +) & + Omit & { + /** A tool-owned search keeps its exact scope instead of inheriting page filters. */ + filters?: WorkspaceSearchFilters + } /** A new query or access scope starts a fresh search and rolling-date anchor. */ export function KnowledgeSearchResults({ @@ -123,7 +49,7 @@ export function KnowledgeSearchResults({ const { data: session } = useSession() const trimmed = query.trim() const { features } = useDeploymentShape() - const Results = features.liveEnterpriseSearch ? LiveSearchResults : SearchResults + const Results = features.liveEnterpriseSearch ? LiveSearchResults : IndexedSearchResults return ( (null) - const { - data: index, - isPending: basesPending, - isError: basesFailed, - isFetching: basesFetching, - refetch: refetchIndex, - } = useSearchIndex(scope) - const [filters] = useQueryStates(searchFilterParsers, resourceUrlKeys) - const custom = filters.updated === 'custom' - const pageFilters = useMemo( - () => searchFiltersFromParams(filters, searchedAt), - [filters.source, filters.updated, filters.from, filters.to, searchedAt] - ) - const searchFilters = suppliedFilters ?? pageFilters - const scopeId = scope.kind === 'organization' ? scope.organizationId : scope.workspaceId - useEffect(() => { - onSearchChange?.({ scope, query, filters: searchFilters, ...(topK ? { topK } : {}) }) - }, [scope.kind, scopeId, query, searchFilters, topK, onSearchChange]) - const filtersKey = JSON.stringify(searchFilters) - const expanded = expandedFor === filtersKey - /** A custom window is two-ended: until both days are chosen, nothing is searched. */ - const awaitingRange = !suppliedFilters && custom && !(filters.from && filters.to) - const { - data: search, - isPending, - isFetching, - isPlaceholderData, - isError: searchFailed, - refetch: refetchSearch, - } = useWorkspaceKnowledgeSearch( - scope, - awaitingRange ? '' : query, - searchFilters, - topK ?? - (expanded - ? WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.expanded - : WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.initial), - { retainAcrossLimits: topK === undefined } - ) - /** A full first page may collapse to few cards, yet more documents may still match. */ - const mayHaveMore = - topK === undefined && - !expanded && - (search?.results.length ?? 0) >= WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.initial - const { data: overview } = useSearchSourceOverview(scope) - const indexing = (overview?.providers ?? []) - .filter((provider) => provider.isSyncing) - .map((provider) => connectorDisplayName(provider.connectorType)) - const documents = groupResultsByDocument(search?.results ?? []) - const sourceTypes = [ - ...new Set([ - ...(filters.source ? [filters.source] : []), - ...(overview?.providers.map((provider) => provider.connectorType) ?? []), - UPLOAD_SOURCE, - ]), - ].sort((left, right) => connectorDisplayName(left).localeCompare(connectorDisplayName(right))) - const failed = basesFailed || searchFailed - const pending = basesPending || isPending - const fetching = basesFetching || isFetching - const noSources = !basesPending && !basesFailed && !index?.knowledgeBaseId - const partial = search?.retrieval.status === 'partial' - const documentCount = documents.length === 1 ? '1 document' : `${documents.length} documents` - - const indexingNote = - indexing.length > 0 - ? `Still indexing ${indexing.join(', ')}; results grow as documents land.` - : null - - const showResults = !noSources && !failed && !basesPending && documents.length > 0 - /** A custom window waiting for its days must show the filters, or the picker is unreachable. */ - const showFilters = - hasShownFilters || - showResults || - awaitingRange || - (!noSources && !pending && !failed && !!search && !partial) - if (showFilters && !hasShownFilters) setHasShownFilters(true) - - return noSources ? ( -
-

No sources are set up yet.

- - View sources - -
- ) : ( -
-
-
- {awaitingRange ? ( -

- Choose the days to search. -

- ) : fetching ? ( - - ) : pending && !failed ? null : ( -

- {failed - ? 'Search couldn’t run.' - : partial - ? documents.length === 0 - ? 'Search timed out.' - : `${documentCount} · some results may be missing.` - : documents.length === 0 - ? 'Search found no results.' - : `${documentCount} · searched as you`} -

- )} - {indexingNote && !failed && !partial && ( -

{indexingNote}

- )} -
- {(failed || partial) && ( - void (basesFailed ? refetchIndex() : refetchSearch())} - > - {fetching ? 'Retrying' : 'Try again'} - - )} -
- {suppliedFilters === undefined && showFilters && } - {showResults && ( -
- {documents.map((result) => { - const source = toSource(result, query, scope) - return ( - - onSummarize(`Summarize "${cited.title ?? cited.url}"`, { - ...searchFilters, - documentIds: [result.documentId], - }) - } - /> - ) - })} - {mayHaveMore && ( -
- setExpandedFor(filtersKey)} - > - Show more - -
- )} -
- )} -
- ) -} - /** Live results do not mount index or sync-status queries. */ function LiveSearchResults({ scope, diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-transitions.test.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-transitions.test.tsx index 36d5c2b07a4..77e85f9f3ed 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-transitions.test.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-transitions.test.tsx @@ -6,6 +6,7 @@ import { apiClientRequestMockFns, } from '@sim/testing/mocks/api-client-request.mock' import { authClientMock, authClientMockFns } from '@sim/testing/mocks/auth-client.mock' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { kbConnectorsQueriesMock, kbConnectorsQueriesMockFns, @@ -235,6 +236,16 @@ async function complete( } describe('search refinement with the real query cache and URL state', () => { + /** The source refinement chips belong to the indexed search results. */ + beforeEach(() => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + resetDeploymentShape() + }) + afterEach(() => { + resetEnvFlagsMock() + resetDeploymentShape() + }) + it('does not restore cleared access data as a placeholder', async () => { await render() await complete(0) diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/utils.ts b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/utils.ts new file mode 100644 index 00000000000..1eea3ee3a9c --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/utils.ts @@ -0,0 +1,85 @@ +import type { + WorkspaceKnowledgeSearchResult, + WorkspaceSearchFilters, +} from '@/lib/api/contracts/knowledge' +import type { ResourceScope } from '@/lib/core/resource-scope' +import { getBaseUrl } from '@/lib/core/utils/urls' +import { matchSnippet } from '@/lib/knowledge/search/snippet' +import type { SearchResource } from '@/lib/mothership/generated/resources' +import { connectorDisplayName } from '@/lib/sim-search/connectors' +import { + isHttpUrl, + type SourceTagData, +} from '@/app/workspace/[workspaceId]/home/components/message-content/components/special-tags' + +/** + * One card per document, keeping the best-ranked chunk of each: the list is + * already in rank order, so the first chunk seen for a document is its best. + */ +export function groupResultsByDocument( + results: readonly WorkspaceKnowledgeSearchResult[] +): WorkspaceKnowledgeSearchResult[] { + const seen = new Set() + const grouped: WorkspaceKnowledgeSearchResult[] = [] + for (const result of results) { + if (seen.has(result.documentId)) continue + seen.add(result.documentId) + grouped.push(result) + } + return grouped +} + +/** + * A result as the source card renders it: the row's second line names the + * source app, or the knowledge base for an upload. Without an HTTP(S) source + * URL, the link opens the canonical document in Sim. + */ +export function toSource( + result: WorkspaceKnowledgeSearchResult, + query: string, + scope: ResourceScope +): SourceTagData { + return { + url: isHttpUrl(result.sourceUrl) + ? result.sourceUrl + : `${getBaseUrl()}${scope.kind === 'organization' ? `/o/${encodeURIComponent(scope.organizationId)}` : `/workspace/${encodeURIComponent(scope.workspaceId)}`}/knowledge/${encodeURIComponent(result.knowledgeBaseId)}/${encodeURIComponent(result.documentId)}`, + title: result.documentName ?? undefined, + siteName: result.connectorType + ? connectorDisplayName(result.connectorType) + : result.knowledgeBaseName || undefined, + connectorType: result.connectorType ?? undefined, + snippet: matchSnippet(result.content, query), + author: result.author ?? undefined, + updatedAt: result.sourceDate ?? result.sourceModifiedAt ?? undefined, + } +} + +/** + * Arrow keys walk the result links, the way a search page does; Enter on a + * focused link opens it natively. Focus stops at either end. + */ +export function handleResultsKeyDown(event: React.KeyboardEvent) { + if (event.key !== 'ArrowDown' && event.key !== 'ArrowUp') return + const links = [...event.currentTarget.querySelectorAll('a[data-source-link]')] + if (links.length === 0) return + const index = links.findIndex((link) => link === document.activeElement) + if (index < 0) return + const next = + event.key === 'ArrowDown' ? Math.min(index + 1, links.length - 1) : Math.max(index - 1, 0) + if (next === index) return + event.preventDefault() + links[next].focus() +} + +/** What each backend's results view is handed by `KnowledgeSearchResults`. */ +export interface SearchResultsProps { + suppliedFilters?: WorkspaceSearchFilters + topK?: number + nativeQueries?: SearchResource['nativeQueries'] + reuseFreshResult?: boolean + scope: ResourceScope + query: string + /** Binds the Assistant turn to the selected canonical document. */ + onSummarize: (prompt: string, filters: WorkspaceSearchFilters) => void + onSearchChange?: (search: SearchResource) => void +} diff --git a/apps/sim/background/knowledge-projection.ts b/apps/sim/background/knowledge-projection.ts index d3f52a23f6e..e2194598fff 100644 --- a/apps/sim/background/knowledge-projection.ts +++ b/apps/sim/background/knowledge-projection.ts @@ -7,7 +7,6 @@ import { import { KNOWLEDGE_PROJECTION_PASS_BUDGET_MS, KNOWLEDGE_PROJECTION_TASK_ID, - requestKnowledgeProjection, } from '@/lib/knowledge/projection/enqueue' import { runKnowledgeProjectionPass } from '@/lib/knowledge/projection/run' @@ -23,9 +22,10 @@ export const KNOWLEDGE_PROJECTION_RETRY_POLICY: BackgroundRetryPolicy = { /** * Runs one knowledge projector pass. One pass runs at a time and projects several documents at - * once itself; the prompt requests and the sweep collapse into whichever pass is queued. A pass - * that ran out of budget with marks left asks for the next one. Retry-safe: a pass writes only rows - * that differ from their source and removes a mark only on the generation it read. + * once itself; the sweep enqueues at most one per minute, and marks a pass left are taken by the + * next. Retry-safe: a pass writes only rows that differ from their source and removes a mark only + * on the generation it read, or, for a workspace document with no content to project, once no + * writer holds it. */ export const knowledgeProjectionTask = task({ id: KNOWLEDGE_PROJECTION_TASK_ID, @@ -35,11 +35,5 @@ export const knowledgeProjectionTask = task({ queue: { name: KNOWLEDGE_PROJECTION_TASK_ID, concurrencyLimit: 1 }, catchError: async ({ error, ctx }) => getBackgroundRetryDecision(error, ctx.attempt.number, KNOWLEDGE_PROJECTION_RETRY_POLICY), - run: async () => { - const result = await runKnowledgeProjectionPass({ - budgetMs: KNOWLEDGE_PROJECTION_PASS_BUDGET_MS, - }) - if (result.remaining) await requestKnowledgeProjection() - return result - }, + run: async () => runKnowledgeProjectionPass({ budgetMs: KNOWLEDGE_PROJECTION_PASS_BUDGET_MS }), }) diff --git a/apps/sim/hooks/queries/github-search-installations.ts b/apps/sim/hooks/queries/github-search-installations.ts deleted file mode 100644 index f7788e2d199..00000000000 --- a/apps/sim/hooks/queries/github-search-installations.ts +++ /dev/null @@ -1,53 +0,0 @@ -'use client' - -import { useMutation, useQuery, useQueryClient } from '@tanstack/react-query' -import { requestJson } from '@/lib/api/client/request' -import { - type ConnectGitHubSearchInstallationBody, - connectGitHubSearchInstallationContract, - listGitHubSearchInstallationsContract, -} from '@/lib/api/contracts/knowledge/github-installations' -import { oauthCredentialKeys } from '@/hooks/queries/oauth/oauth-credentials' - -export const GITHUB_SEARCH_INSTALLATIONS_STALE_TIME = 30_000 - -export const githubSearchInstallationKeys = { - all: ['github-search-installations'] as const, - lists: () => [...githubSearchInstallationKeys.all, 'list'] as const, - list: (organizationId?: string) => - [...githubSearchInstallationKeys.lists(), organizationId ?? ''] as const, -} - -export function useGitHubSearchInstallations(organizationId?: string) { - return useQuery({ - queryKey: githubSearchInstallationKeys.list(organizationId), - queryFn: ({ signal }) => { - if (!organizationId) throw new Error('Organization is required') - return requestJson(listGitHubSearchInstallationsContract, { - query: { organizationId }, - signal, - }) - }, - enabled: Boolean(organizationId), - staleTime: GITHUB_SEARCH_INSTALLATIONS_STALE_TIME, - refetchOnWindowFocus: 'always', - retry: false, - }) -} - -export function useConnectGitHubSearchInstallation() { - const queryClient = useQueryClient() - return useMutation({ - mutationFn: (body: ConnectGitHubSearchInstallationBody) => - requestJson(connectGitHubSearchInstallationContract, { body }), - onSuccess: (_result, { organizationId }) => - Promise.all([ - queryClient.invalidateQueries({ - queryKey: githubSearchInstallationKeys.list(organizationId), - }), - queryClient.invalidateQueries({ - queryKey: oauthCredentialKeys.list('github-repositories', '', '', organizationId), - }), - ]), - }) -} diff --git a/apps/sim/hooks/use-github-installation-setup.test.tsx b/apps/sim/hooks/use-github-installation-setup.test.tsx index c8873aa2423..ce3a4a3f25f 100644 --- a/apps/sim/hooks/use-github-installation-setup.test.tsx +++ b/apps/sim/hooks/use-github-installation-setup.test.tsx @@ -102,10 +102,12 @@ describe('GitHub installation setup handoff', () => { expect(current.pending).toBe(false) expect(mocks.connected).toHaveBeenCalledExactlyOnceWith('installation-1') expect(tab.close).toHaveBeenCalledOnce() - expect(mockInvalidate).toHaveBeenCalledTimes(4) expect(mockInvalidate).toHaveBeenCalledWith({ queryKey: ['oauthCredentials', 'list', 'github-repositories', '', '', 'org-1', 'browsing'], }) + expect(mockInvalidate).toHaveBeenCalledWith({ + queryKey: ['organization-accounts', 'detail', 'org-1'], + }) act(() => root.render()) expect(mocks.connected).toHaveBeenCalledOnce() }) diff --git a/apps/sim/hooks/use-github-installation-setup.ts b/apps/sim/hooks/use-github-installation-setup.ts index 3be9057e6f1..3ace14a876a 100644 --- a/apps/sim/hooks/use-github-installation-setup.ts +++ b/apps/sim/hooks/use-github-installation-setup.ts @@ -10,7 +10,6 @@ import { isCredentialGroupOAuthFailure, } from '@/lib/credential-groups/oauth-completion' import { resolveGitHubSetupUrl } from '@/lib/knowledge/github-setup-navigation' -import { githubSearchInstallationKeys } from '@/hooks/queries/github-search-installations' import { isGitHubSetupTerminalError, useCancelGitHubSearchSetup, @@ -105,7 +104,6 @@ export function useGitHubInstallationSetup({ ), }) } - void client.invalidateQueries({ queryKey: githubSearchInstallationKeys.list(organizationId) }) void client.invalidateQueries({ queryKey: organizationAccountsKeys.detail(organizationId) }) callback.current(result.credential.id) } else if ( diff --git a/apps/sim/lib/api/contracts/knowledge/connectors.ts b/apps/sim/lib/api/contracts/knowledge/connectors.ts index 4c524701a6b..962d101a731 100644 --- a/apps/sim/lib/api/contracts/knowledge/connectors.ts +++ b/apps/sim/lib/api/contracts/knowledge/connectors.ts @@ -12,7 +12,6 @@ import { booleanQueryFlagSchema, organizationIdSchema, resourceOwnerSchema, - workspaceIdSchema, } from '@/lib/api/contracts/primitives' import { defineRouteContract } from '@/lib/api/contracts/types' import { CONNECTOR_ACCESS_MODES } from '@/lib/knowledge/connectors/access-modes' @@ -341,21 +340,6 @@ export const startKnowledgeConnectorMemberEnrollmentContract = defineRouteContra }, }) -/** A source's personal account or mirrored-ACL identity connection for the current viewer. */ -export const workspaceMemberConnectorSchema = z.object({ - knowledgeBaseId: z.string(), - knowledgeBaseName: z.string(), - knowledgeBaseIsSearchIndex: z.boolean().optional(), - sourceDescription: z.string().max(240).optional(), - connectorId: z.string(), - connectorType: z.string(), - memberSyncStatus: z.enum(MEMBER_SYNC_STATUSES), - viewerMembership: viewerConnectorMembershipSchema, - /** Documents of this connector the viewer may read right now. */ - viewerDocumentCount: z.number().int().nonnegative(), -}) -export type WorkspaceMemberConnector = z.output - const searchSourceSummaryFields = { knowledgeBaseId: knowledgeBaseParamsSchema.shape.id, connectorId: knowledgeConnectorParamsSchema.shape.connectorId, @@ -585,16 +569,6 @@ export const connectSimSearchConnectorContract = defineRouteContract({ }, }) -export const listWorkspaceMemberConnectorsContract = defineRouteContract({ - method: 'GET', - path: '/api/knowledge/member-connectors', - query: z.object({ workspaceId: workspaceIdSchema }), - response: { - mode: 'json', - schema: successResponseSchema(z.array(workspaceMemberConnectorSchema)), - }, -}) - export const deleteKnowledgeConnectorContract = defineRouteContract({ method: 'DELETE', path: '/api/knowledge/[id]/connectors/[connectorId]', diff --git a/apps/sim/lib/api/contracts/knowledge/github-installations.ts b/apps/sim/lib/api/contracts/knowledge/github-installations.ts index 28fcbdc57ab..00518f5056d 100644 --- a/apps/sim/lib/api/contracts/knowledge/github-installations.ts +++ b/apps/sim/lib/api/contracts/knowledge/github-installations.ts @@ -1,6 +1,4 @@ import { z } from 'zod' -import { organizationIdSchema } from '@/lib/api/contracts/primitives' -import { defineRouteContract } from '@/lib/api/contracts/types' export const githubInstallationIdSchema = z .string() @@ -16,64 +14,3 @@ export const githubSearchInstallationSchema = z.object({ accountLogin: z.string().min(1).max(100), accountType: z.enum(['User', 'Organization']), }) -export type GitHubSearchInstallation = z.output - -export const listGitHubSearchInstallationsQuerySchema = z.object({ - organizationId: organizationIdSchema, -}) -export type ListGitHubSearchInstallationsQuery = z.input< - typeof listGitHubSearchInstallationsQuerySchema -> - -export const listGitHubSearchInstallationsResponseSchema = z.object({ - success: z.literal(true), - available: z.boolean(), - installUrl: z - .string() - .max(2000) - .regex( - /^https:\/\/github\.com\/apps\/[a-z0-9-]+\/installations\/new$/, - 'GitHub installation URL must use the configured GitHub App' - ) - .nullable(), - needsUserConnection: z.boolean(), - installations: z.array(githubSearchInstallationSchema).max(1000), -}) -export type ListGitHubSearchInstallationsResponse = z.output< - typeof listGitHubSearchInstallationsResponseSchema -> - -export const listGitHubSearchInstallationsContract = defineRouteContract({ - method: 'GET', - path: '/api/knowledge/github/installations', - query: listGitHubSearchInstallationsQuerySchema, - response: { mode: 'json', schema: listGitHubSearchInstallationsResponseSchema }, -}) - -export const connectGitHubSearchInstallationBodySchema = z - .object({ - organizationId: organizationIdSchema, - installationId: githubInstallationIdSchema, - }) - .strict() -export type ConnectGitHubSearchInstallationBody = z.input< - typeof connectGitHubSearchInstallationBodySchema -> - -export const connectGitHubSearchInstallationResponseSchema = z.object({ - success: z.literal(true), - credential: z.object({ - id: z.string().min(1).max(200), - displayName: z.string().min(1).max(500), - }), -}) -export type ConnectGitHubSearchInstallationResponse = z.output< - typeof connectGitHubSearchInstallationResponseSchema -> - -export const connectGitHubSearchInstallationContract = defineRouteContract({ - method: 'POST', - path: '/api/knowledge/github/installations', - body: connectGitHubSearchInstallationBodySchema, - response: { mode: 'json', schema: connectGitHubSearchInstallationResponseSchema }, -}) diff --git a/apps/sim/lib/core/config/env.ts b/apps/sim/lib/core/config/env.ts index 933f829cccd..d213f8fe180 100644 --- a/apps/sim/lib/core/config/env.ts +++ b/apps/sim/lib/core/config/env.ts @@ -639,9 +639,6 @@ export const env = createEnv({ AGENT_MEMORY_HISTORY: z.boolean().optional(), CREDENTIAL_GROUPS: z.boolean().optional(), // Enable enterprise Credential Groups globally KNOWLEDGE_MEMBER_ACCESS: z.boolean().optional(), // Enable per-member knowledge connectors and hybrid-by-default retrieval globally - KNOWLEDGE_TIN_KEYWORD: z.boolean().optional(), // Rank large-scope keyword retrieval through the Tin text index where it exists - KNOWLEDGE_ASYNC_PROJECTION: z.boolean().optional(), // Knowledge writers leave search projection rows to the background projector - KNOWLEDGE_PROJECTION_FILL: z.boolean().optional(), // The knowledge projector fills projection rows written before they carried a source and ACL // Organizations - for self-hosted deployments ORGANIZATIONS_ENABLED: z.boolean().optional(), // Enable organizations on self-hosted (bypasses plan requirements) diff --git a/apps/sim/lib/core/config/feature-flags.test.ts b/apps/sim/lib/core/config/feature-flags.test.ts index c630218e5b5..c0cfd64730c 100644 --- a/apps/sim/lib/core/config/feature-flags.test.ts +++ b/apps/sim/lib/core/config/feature-flags.test.ts @@ -38,8 +38,6 @@ import { const envRef = mockEnvObject setEnv({ APPCONFIG_APPLICATION: 'sim-staging', - KNOWLEDGE_PROJECTION_FILL: undefined, - KNOWLEDGE_ASYNC_PROJECTION: undefined, APPCONFIG_ENVIRONMENT: 'staging', TABLES_V2_API: undefined, TABLE_ROW_TTL: undefined, @@ -48,7 +46,6 @@ setEnv({ AGENT_MEMORY_HISTORY: undefined, CREDENTIAL_GROUPS: undefined, KNOWLEDGE_MEMBER_ACCESS: undefined, - KNOWLEDGE_TIN_KEYWORD: undefined, SLACK_SEARCH_SHARED_APP: undefined, }) @@ -142,9 +139,6 @@ describe('isFeatureEnabled', () => { setEnvFlags({ isAppConfigEnabled: false }) envRef.CREDENTIAL_GROUPS = undefined envRef.KNOWLEDGE_MEMBER_ACCESS = undefined - envRef.KNOWLEDGE_TIN_KEYWORD = undefined - envRef.KNOWLEDGE_ASYNC_PROJECTION = undefined - envRef.KNOWLEDGE_PROJECTION_FILL = undefined envRef.SLACK_SEARCH_SHARED_APP = undefined }) @@ -181,45 +175,6 @@ describe('isFeatureEnabled', () => { }) }) - describe('knowledge-tin-keyword flag', () => { - it('is a global switch', async () => { - expect(await isFeatureEnabled('knowledge-tin-keyword')).toBe(false) - envRef.KNOWLEDGE_TIN_KEYWORD = true - expect(await isFeatureEnabled('knowledge-tin-keyword')).toBe(true) - }) - - it('follows an AppConfig global rule', async () => { - withAppConfig({ 'knowledge-tin-keyword': { enabled: true } }) - expect(await isFeatureEnabled('knowledge-tin-keyword')).toBe(true) - }) - }) - - describe('knowledge-async-projection flag', () => { - it('is a global switch', async () => { - expect(await isFeatureEnabled('knowledge-async-projection')).toBe(false) - envRef.KNOWLEDGE_ASYNC_PROJECTION = true - expect(await isFeatureEnabled('knowledge-async-projection')).toBe(true) - }) - - it('follows an AppConfig global rule', async () => { - withAppConfig({ 'knowledge-async-projection': { enabled: true } }) - expect(await isFeatureEnabled('knowledge-async-projection')).toBe(true) - }) - }) - - describe('knowledge-projection-fill flag', () => { - it('is a global switch', async () => { - expect(await isFeatureEnabled('knowledge-projection-fill')).toBe(false) - envRef.KNOWLEDGE_PROJECTION_FILL = true - expect(await isFeatureEnabled('knowledge-projection-fill')).toBe(true) - }) - - it('follows an AppConfig global rule', async () => { - withAppConfig({ 'knowledge-projection-fill': { enabled: true } }) - expect(await isFeatureEnabled('knowledge-projection-fill')).toBe(true) - }) - }) - describe('knowledge-member-access flag', () => { it('uses a global fallback switch off AppConfig', async () => { expect(await isFeatureEnabled('knowledge-member-access')).toBe(false) diff --git a/apps/sim/lib/core/config/feature-flags.ts b/apps/sim/lib/core/config/feature-flags.ts index db2a3c6b33a..91c105338c9 100644 --- a/apps/sim/lib/core/config/feature-flags.ts +++ b/apps/sim/lib/core/config/feature-flags.ts @@ -102,38 +102,15 @@ const FEATURE_FLAGS = { }, 'knowledge-member-access': { description: - 'Permission-aware indexing and retrieval. Organization Search UI, MCP, and search APIs ' + - 'require this flag and credential-groups for the canonical orgId; user/admin/workspace ' + - 'targeting cannot enable another organization. Workspace member sync uses workspaceId; ' + - 'workspace retrieval defaults may additionally use user/admin targeting. Source ACL ' + - 'mirroring remains independent of managed identities. Off-AppConfig falls back to ' + - 'KNOWLEDGE_MEMBER_ACCESS.', + 'Organization Search (live) and the permission-aware workspace connector modes: members ' + + '(per-member sync, which also requires credential-groups) and admin (source ACL ' + + 'mirroring, independent of managed identities). Organization Search UI, MCP, and ' + + 'search APIs require this flag and credential-groups for the canonical orgId; ' + + 'user/admin/workspace targeting cannot enable another organization. Workspace connector ' + + 'modes use workspaceId; workspace retrieval defaults may additionally use user/admin ' + + 'targeting. Off-AppConfig falls back to KNOWLEDGE_MEMBER_ACCESS.', fallback: 'KNOWLEDGE_MEMBER_ACCESS', }, - 'knowledge-tin-keyword': { - description: - 'Rank keyword retrieval for members whose permitted set is too large to enumerate through ' + - 'the Tin text index instead of GIN. Has no effect where the Tin keyword index is absent or ' + - 'invalid. Off-AppConfig falls back to KNOWLEDGE_TIN_KEYWORD.', - fallback: 'KNOWLEDGE_TIN_KEYWORD', - }, - 'knowledge-async-projection': { - description: - 'Knowledge writers (document processing and connector ACL writes) leave search projection ' + - 'rows to the background knowledge projector instead of rewriting them in their own ' + - 'transaction. Global on/off only; turn it on only once no release older than the ' + - 'projector serves search. Off-AppConfig falls back to KNOWLEDGE_ASYNC_PROJECTION.', - fallback: 'KNOWLEDGE_ASYNC_PROJECTION', - }, - 'knowledge-projection-fill': { - description: - 'The knowledge projector also fills search projection rows written before they carried ' + - "their document's source and ACL, marking at most 100 documents at once so fresh writes " + - 'never wait behind much of it. Global on/off only; off pauses the fill, and search keeps ' + - 'deciding unfilled rows on their document. Off-AppConfig falls back to ' + - 'KNOWLEDGE_PROJECTION_FILL.', - fallback: 'KNOWLEDGE_PROJECTION_FILL', - }, } satisfies Record /** diff --git a/apps/sim/lib/credential-groups/slack-provider.test.ts b/apps/sim/lib/credential-groups/slack-provider.test.ts index 024c69b0f0a..51da5603c64 100644 --- a/apps/sim/lib/credential-groups/slack-provider.test.ts +++ b/apps/sim/lib/credential-groups/slack-provider.test.ts @@ -105,6 +105,7 @@ describe('Slack member scope policy', () => { }) it('uses the option policy for enrollment instead of widening it', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) const scopes = SLACK_MANAGED_USER_SCOPES const current = context(scopes) const policy = await adapter.getPolicy(current.option, { diff --git a/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts b/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts index d771e87f86b..4cf02aca8b1 100644 --- a/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts +++ b/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts @@ -55,6 +55,12 @@ vi.mock('@/lib/sim-search/connectors', () => ({ ], })) vi.mock('@/lib/oauth/utils', () => oauthUtilsMock) +/** The barrel's other use cases need the application layer mocked above; ownership is exercised as is. */ +vi.mock('@/lib/sim-search/indexed', async () => ({ + ownsIndexedPersonalSearchAccount: ( + await import('@/lib/sim-search/indexed/integrations/personal-account-ownership') + ).ownsIndexedPersonalSearchAccount, +})) import { prepareOrganizationPersonalConnection, @@ -167,6 +173,7 @@ describe('organization personal token authorization', () => { expect(mocks.token).not.toHaveBeenCalled() }) it('uses the authenticated person inventory and organization token scope without a workspace', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) await expect(resolveOrganizationPersonalToken.execute({ principal, input })).resolves.toEqual({ accessToken: 'secret', refreshed: false, @@ -213,6 +220,7 @@ describe('organization personal token authorization', () => { }) it('does not confuse paused indexing with account authorization', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) mocks.inventory.mockResolvedValue({ connections: [ { indexingStatus: 'paused', accounts: [{ credentialId: 'own', status: 'connected' }] }, diff --git a/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts b/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts index d8b00d7e211..a1c7f8f8b29 100644 --- a/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts +++ b/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts @@ -4,7 +4,6 @@ import { recordProjectedUseCaseAuditEntries } from '@/lib/core/application' import { authorizeOrganizationOperation } from '@/lib/core/application/organization-authorization' import { defineOrganizationOperation } from '@/lib/core/application/organization-operation' import { getBlockVisibility } from '@/lib/core/config/block-visibility' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { OrchestrationError } from '@/lib/core/orchestration/types' import { isManagedCredentialGroupBindingLive, @@ -12,11 +11,12 @@ import { } from '@/lib/credential-groups/credentials' import { resolveManagedOAuthToken } from '@/lib/credentials/managed-oauth' import { projectIntegrationToolsForViewer } from '@/lib/integrations/tool-projection' -import { listPersonalSearchIntegrations } from '@/lib/knowledge/application/personal-search-integrations' +import { personalSearchIntegrationPages } from '@/lib/knowledge/application/personal-search-integration-pages' import { requireOrganizationSearchApproval } from '@/lib/knowledge/search/integration-policy' import { providerIdsForService } from '@/lib/oauth/utils' import { getUserPermissionConfigForOrganization } from '@/lib/permission-groups/resolve.server' import { SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { listLiveAccounts } from '@/lib/sim-search/live/accounts' import { requiresScopedRetrieval } from '@/lib/sim-search/live/policy-schema' import { livePolicyFor, loadLiveSearchPolicies } from '@/lib/sim-search/live/policy-store' @@ -84,42 +84,29 @@ export const resolveOrganizationPersonalToken = { ) { throw new OrchestrationError('forbidden', 'This integration operation is unavailable.') } - let cursor: string | undefined - let owned = isLiveEnterpriseSearchEnabled - ? (await listLiveAccounts({ organizationId: context.organizationId }, context.userId)).some( + const indexed = isIndexedOrgSearchEnabled() + /** Loaded lazily: this resolver is on the executor's credential path, which never needs it otherwise. */ + const owned = indexed + ? await (await import('@/lib/sim-search/indexed')).ownsIndexedPersonalSearchAccount( + principal, + { + organizationId: context.organizationId, + connectorType: connector.type, + credentialId: input.credentialId, + } + ) + : (await listLiveAccounts({ organizationId: context.organizationId }, context.userId)).some( (account) => account.id === input.credentialId && account.type === 'managed_oauth' && account.providerId === binding.providerId ) - : false - const seen = new Set() - for (let page = 0; !isLiveEnterpriseSearchEnabled && page < 100; page++) { - const inventory = await listPersonalSearchIntegrations.execute({ - principal, - input: { - organizationId: context.organizationId, - connectorType: connector.type, - ...(cursor ? { cursor } : {}), - }, - }) - owned = inventory.connections.some((connection) => - connection.accounts.some( - (account) => account.credentialId === input.credentialId && account.status === 'connected' - ) - ) - if (owned || inventory.nextCursor === null) break - if (seen.has(inventory.nextCursor)) - throw new Error('Personal account pagination did not advance') - seen.add(inventory.nextCursor) - cursor = inventory.nextCursor - } if (!owned) throw new OrchestrationError( 'forbidden', 'Assistant can only use your own connected account for this integration.' ) - if (isLiveEnterpriseSearchEnabled) { + if (!indexed) { const policies = await loadLiveSearchPolicies({ organizationId: context.organizationId }) if (requiresScopedRetrieval(connector.type, livePolicyFor(policies, connector.type))) throw new OrchestrationError( @@ -190,28 +177,16 @@ export const prepareOrganizationPersonalConnection = { ) if (!connector) throw new OrchestrationError('validation', 'This integration is unavailable in Search.') - let cursor: string | undefined - const seen = new Set() - for (let page = 0; page < 100; page++) { - const inventory = await listPersonalSearchIntegrations.execute({ - principal, - input: { - organizationId: context.organizationId, - connectorType: connector.type, - ...(cursor ? { cursor } : {}), - }, - }) + for await (const inventory of personalSearchIntegrationPages({ + principal, + input: { organizationId: context.organizationId, connectorType: connector.type }, + })) { const target = input.credentialId ? inventory.connections .flatMap((connection) => connection.accounts) .find((account) => account.credentialId === input.credentialId)?.action : inventory.available[0]?.target if (target) return { provider: connector.meta.name, providerId: connector.providerId, target } - if (inventory.nextCursor === null) break - if (seen.has(inventory.nextCursor)) - throw new Error('Personal account pagination did not advance') - seen.add(inventory.nextCursor) - cursor = inventory.nextCursor } throw new OrchestrationError( 'validation', diff --git a/apps/sim/lib/folders/application/resource-vfs.ts b/apps/sim/lib/folders/application/resource-vfs.ts deleted file mode 100644 index 30ef3e27294..00000000000 --- a/apps/sim/lib/folders/application/resource-vfs.ts +++ /dev/null @@ -1,438 +0,0 @@ -import type { FolderResourceType } from '@/lib/api/contracts/folders' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { - createFolderAtPath, - deleteFolderByPath, - relocateFolderByPath, -} from '@/lib/folders/orchestration' -import { buildFolderPath } from '@/lib/folders/paths' -import { listFoldersForWorkspace } from '@/lib/folders/queries' - -/** - * Shared VFS folder semantics for flat-row resources (tables, knowledge - * bases): both are a single DB row with a `folderId`, so mkdir/mv/rm over - * their `{root}/{...folders}/{name}` paths is identical logic parameterized - * by how rows are listed, moved, and renamed. Folder mutations delegate to - * `lib/folders/orchestration`, which owns naming invariants, tree locks, and - * delete cascades — this module only resolves paths and moves rows. - * - * Authorization is deliberately NOT here: every entry point is called from a - * per-resource authorized use case (table-vfs / knowledge-vfs), the same - * layering workflow-vfs uses. - */ -export interface FolderedResourceRow { - id: string - name: string - folderId: string | null -} - -export interface FolderedResourceAdapter { - resourceType: Extract - rootSegment: 'tables' | 'knowledgebases' - /** Human label for error messages, e.g. "table" / "knowledge base". */ - label: string - listRows(workspaceId: string): Promise - moveRow(row: FolderedResourceRow, folderId: string | null, workspaceId: string): Promise - renameRow( - row: FolderedResourceRow, - newName: string, - workspaceId: string - ): Promise<{ id: string; name: string }> -} - -export interface ResourceVfsOutcome { - source: string - /** Path segments under the resource root the item landed at (folders + leaf). */ - targetSegments?: string[] - kind: 'resource' | 'folder' - resourceId?: string - error?: string -} - -interface FolderNode { - id: string - name: string - parentId: string | null -} - -interface FolderIndex { - byId: Map - /** parentKey(parentId) + "\u0000" + name → folderId */ - byParentAndName: Map -} - -const ROOT_PARENT_KEY = '' - -function parentKey(parentId: string | null): string { - return parentId ?? ROOT_PARENT_KEY -} - -function childKey(parentId: string | null, name: string): string { - return `${parentKey(parentId)}\u0000${name}` -} - -async function loadFolderIndex( - workspaceId: string, - resourceType: FolderResourceType -): Promise { - const folders = await listFoldersForWorkspace(workspaceId, 'active', resourceType) - const byId = new Map() - const byParentAndName = new Map() - for (const folder of folders) { - byId.set(folder.id, { id: folder.id, name: folder.name, parentId: folder.parentId }) - byParentAndName.set(childKey(folder.parentId, folder.name), folder.id) - } - return { byId, byParentAndName } -} - -function folderSegments(index: FolderIndex, folderId: string | null): string[] { - const segments: string[] = [] - let current = folderId - const visited = new Set() - while (current && !visited.has(current)) { - visited.add(current) - const node = index.byId.get(current) - if (!node) break - segments.unshift(node.name) - current = node.parentId - } - return segments -} - -/** Resolves an existing folder path to its id; null = root; undefined = missing. */ -function resolveFolderId( - index: FolderIndex, - segments: readonly string[] -): string | null | undefined { - let current: string | null = null - for (const segment of segments) { - const next = index.byParentAndName.get(childKey(current, segment)) - if (!next) return undefined - current = next - } - return current -} - -/** - * mkdir -p: creates every missing folder along each path. Ancestors created by - * an earlier path in the same batch are found via the reloaded index. - */ -export async function createResourceVfsFolders( - adapter: FolderedResourceAdapter, - params: { - workspaceId: string - userId: string - paths: Array<{ source: string; segments: string[] }> - } -): Promise { - let index = await loadFolderIndex(params.workspaceId, adapter.resourceType) - const outcomes: ResourceVfsOutcome[] = [] - for (const { source, segments } of params.paths) { - if (segments.length === 0) { - outcomes.push({ - source, - kind: 'folder', - error: 'Path must include at least one folder segment', - }) - continue - } - try { - const folderId = await ensureFolderPath( - adapter, - params.workspaceId, - params.userId, - index, - segments - ) - index = await loadFolderIndex(params.workspaceId, adapter.resourceType) - outcomes.push({ - source, - kind: 'folder', - resourceId: folderId ?? undefined, - targetSegments: [...segments], - }) - } catch (error) { - outcomes.push({ - source, - kind: 'folder', - error: messageFor(error, `${adapter.label} folder creation failed`), - }) - } - } - return outcomes -} - -async function ensureFolderPath( - adapter: FolderedResourceAdapter, - workspaceId: string, - userId: string, - index: FolderIndex, - segments: readonly string[] -): Promise { - let folderId: string | null = null - for (let position = 0; position < segments.length; position += 1) { - const existing = resolveFolderId(index, segments.slice(0, position + 1)) - if (existing !== undefined) { - folderId = existing - continue - } - const result = await createFolderAtPath({ - resourceType: adapter.resourceType, - workspaceId, - userId, - path: buildFolderPath(segments.slice(0, position + 1)), - effects: false, - throwInfrastructure: true, - }) - if (!result.success || !result.folder) { - if (result.errorCode === 'conflict') { - index = await loadFolderIndex(workspaceId, adapter.resourceType) - const concurrent = resolveFolderId(index, segments.slice(0, position + 1)) - if (concurrent !== undefined) { - folderId = concurrent - continue - } - } - throw new OrchestrationError( - result.errorCode === 'forbidden' ? 'forbidden' : 'validation', - result.error ?? `${adapter.label} folder creation failed` - ) - } - folderId = result.folder.id - index.byId.set(folderId, { - id: folderId, - name: result.folder.name, - parentId: result.folder.parentId, - }) - index.byParentAndName.set(childKey(result.folder.parentId, result.folder.name), folderId) - } - return folderId -} - -type ResolvedSource = - | { kind: 'resource'; row: FolderedResourceRow } - | { kind: 'folder'; folderId: string } - -/** - * Resolves one source path: the leaf is preferred as a resource row inside the - * resolved parent folder; a folder of that name is the fallback. A bare leaf - * (no folder segments) also matches a uniquely-named resource anywhere in the - * tree, so pre-folders paths keep working after rows move into folders. - */ -function resolveSource( - adapter: FolderedResourceAdapter, - index: FolderIndex, - rows: FolderedResourceRow[], - segments: readonly string[] -): ResolvedSource { - const root = adapter.rootSegment - if (segments.length === 0) { - throw new OrchestrationError( - 'validation', - `Path must name a ${adapter.label} or folder under ${root}/` - ) - } - const leaf = segments[segments.length - 1] - const parentSegments = segments.slice(0, -1) - const parentId = resolveFolderId(index, parentSegments) - if (parentId !== undefined) { - const inParent = rows.filter((row) => row.name === leaf && row.folderId === (parentId ?? null)) - if (inParent.length === 1) return { kind: 'resource', row: inParent[0] } - const asFolder = resolveFolderId(index, segments) - if (asFolder !== undefined && asFolder !== null) return { kind: 'folder', folderId: asFolder } - } - if (parentSegments.length === 0) { - const anywhere = rows.filter((row) => row.name === leaf) - if (anywhere.length === 1) return { kind: 'resource', row: anywhere[0] } - if (anywhere.length > 1) { - throw new OrchestrationError( - 'conflict', - `${root}/${leaf} is ambiguous — several ${adapter.label}s share that name. Use the full folder path.` - ) - } - } - throw new OrchestrationError( - 'not_found', - `No ${adapter.label} or folder found at ${root}/${segments.join('/')}` - ) -} - -/** - * mv: with a trailing-slash destination, moves every source (resource rows and - * whole folders) into that folder path, creating it as needed. Without one, - * exactly one source is renamed and/or moved to the destination's parent + - * leaf name. Mirrors the workflows/ contract. - */ -export async function transferResourceVfsItems( - adapter: FolderedResourceAdapter, - params: { - workspaceId: string - userId: string - sources: Array<{ source: string; segments: string[] }> - destination: { segments: string[]; trailingSlash: boolean } - } -): Promise { - const { workspaceId, userId } = params - let index = await loadFolderIndex(workspaceId, adapter.resourceType) - let rows = await adapter.listRows(workspaceId) - const outcomes: ResourceVfsOutcome[] = [] - - const moveIntoFolder = params.destination.trailingSlash - if (!moveIntoFolder && params.sources.length > 1) { - throw new OrchestrationError( - 'validation', - `A rename destination takes exactly one source; to move several items, end the destination with "/" (a folder path).` - ) - } - - const destFolderSegments = moveIntoFolder - ? params.destination.segments - : params.destination.segments.slice(0, -1) - const renameTo = moveIntoFolder - ? null - : params.destination.segments[params.destination.segments.length - 1] - if (!moveIntoFolder && !renameTo) { - throw new OrchestrationError('validation', 'destination must include a name') - } - - for (const { source, segments } of params.sources) { - try { - const resolved = resolveSource(adapter, index, rows, segments) - if (resolved.kind === 'resource') { - const targetFolderId = await ensureFolderPath( - adapter, - workspaceId, - userId, - index, - destFolderSegments - ) - let row = resolved.row - if (row.folderId !== (targetFolderId ?? null)) { - await adapter.moveRow(row, targetFolderId ?? null, workspaceId) - row = { ...row, folderId: targetFolderId ?? null } - } - let finalName = row.name - if (renameTo && renameTo !== row.name) { - const renamed = await adapter.renameRow(row, renameTo, workspaceId) - finalName = renamed.name - } - rows = rows.map((r) => (r.id === row.id ? { ...row, name: finalName } : r)) - outcomes.push({ - source, - kind: 'resource', - resourceId: row.id, - targetSegments: [...destFolderSegments, finalName], - }) - continue - } - - const sourcePath = buildFolderPath(folderSegments(index, resolved.folderId)) - const destinationPath = buildFolderPath( - moveIntoFolder - ? [...destFolderSegments, ...folderSegments(index, resolved.folderId).slice(-1)] - : [...destFolderSegments, renameTo as string] - ) - const result = await relocateFolderByPath({ - resourceType: adapter.resourceType, - workspaceId, - userId, - path: sourcePath, - destinationPath, - effects: false, - throwInfrastructure: true, - }) - if (!result.success || !result.folder) { - throw new OrchestrationError( - result.errorCode === 'forbidden' ? 'forbidden' : 'validation', - result.error ?? `${adapter.label} folder move failed` - ) - } - index = await loadFolderIndex(workspaceId, adapter.resourceType) - rows = await adapter.listRows(workspaceId) - outcomes.push({ - source, - kind: 'folder', - resourceId: result.folder.id, - targetSegments: [...folderSegments(index, result.folder.id)], - }) - } catch (error) { - outcomes.push({ - source, - kind: 'resource', - error: messageFor(error, `${adapter.label} move failed`), - }) - } - } - return outcomes -} - -/** rm of folder paths: recursive delete through the shared cascade. */ -export async function deleteResourceVfsFolders( - adapter: FolderedResourceAdapter, - params: { - workspaceId: string - userId: string - paths: Array<{ source: string; segments: string[] }> - } -): Promise { - const outcomes: ResourceVfsOutcome[] = [] - for (const { source, segments } of params.paths) { - try { - const index = await loadFolderIndex(params.workspaceId, adapter.resourceType) - const folderId = resolveFolderId(index, segments) - if (folderId === undefined || folderId === null) { - throw new OrchestrationError( - 'not_found', - `No ${adapter.label} folder found at ${adapter.rootSegment}/${segments.join('/')}` - ) - } - const result = await deleteFolderByPath({ - resourceType: adapter.resourceType, - workspaceId: params.workspaceId, - userId: params.userId, - path: buildFolderPath(segments), - recursive: true, - effects: false, - throwInfrastructure: true, - }) - if (!result.success) { - throw new OrchestrationError( - result.errorCode === 'forbidden' ? 'forbidden' : 'validation', - result.error ?? `${adapter.label} folder deletion failed` - ) - } - outcomes.push({ source, kind: 'folder', resourceId: folderId }) - } catch (error) { - outcomes.push({ - source, - kind: 'folder', - error: messageFor(error, `${adapter.label} folder deletion failed`), - }) - } - } - return outcomes -} - -/** Resolves a resource (never a folder) for the segments-aware rename/delete paths. */ -export async function resolveResourceRowBySegments( - adapter: FolderedResourceAdapter, - workspaceId: string, - segments: readonly string[] -): Promise { - const index = await loadFolderIndex(workspaceId, adapter.resourceType) - const rows = await adapter.listRows(workspaceId) - const resolved = resolveSource(adapter, index, rows, segments) - if (resolved.kind !== 'resource') { - throw new OrchestrationError( - 'validation', - `${adapter.rootSegment}/${segments.join('/')} is a folder; this operation takes a ${adapter.label}.` - ) - } - return resolved.row -} - -function messageFor(error: unknown, fallback: string): string { - if (error instanceof OrchestrationError) return error.message - if (error instanceof Error && error.message) return error.message - return fallback -} diff --git a/apps/sim/lib/knowledge/__integration__/async-projection-processing.integration.ts b/apps/sim/lib/knowledge/__integration__/async-projection-processing.integration.ts deleted file mode 100644 index 46f7a1bbb3e..00000000000 --- a/apps/sim/lib/knowledge/__integration__/async-projection-processing.integration.ts +++ /dev/null @@ -1,162 +0,0 @@ -/** - * Document processing with `knowledge-async-projection` on, through the real service: the commit - * writes the chunks and a mark on their document and no projection row, asks for a projector pass - * after it commits, and the pass then writes every projection row from the chunks and the document. - */ -import { mkdtempSync } from 'node:fs' -import { rm } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import path from 'node:path' -import { db } from '@sim/db' -import { runKnowledgeProjection } from '@sim/db/knowledge-projection' -import { - document, - embeddingSearch, - knowledgeBase, - knowledgeProjectionDirty, - organization, - user, - workspace, -} from '@sim/db/schema' -import { eq, inArray } from 'drizzle-orm' -import postgres from 'postgres' -import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' - -const fixtures = vi.hoisted(() => ({ - root: '', - process: vi.fn(), - embeddings: vi.fn(), - requestKnowledgeProjection: vi.fn(), -})) -vi.mock('@/lib/uploads/core/setup.server', () => ({ - get UPLOAD_DIR_SERVER() { - return fixtures.root - }, -})) -vi.mock('@/lib/knowledge/documents/document-processor', () => ({ - processDocument: fixtures.process, -})) -vi.mock('@/lib/knowledge/embeddings', () => ({ generateEmbeddings: fixtures.embeddings })) -vi.mock('@/lib/core/config/feature-flags', () => ({ - isFeatureEnabled: async (flag: string) => flag === 'knowledge-async-projection', -})) -/** The pass is run by the test itself, so the commit can be read before any pass has run. */ -vi.mock('@/lib/knowledge/projection/enqueue', () => ({ - requestKnowledgeProjection: fixtures.requestKnowledgeProjection, -})) - -import { resolveBillingAttribution } from '@/lib/billing/core/billing-attribution' -import * as embeddingClient from '@/lib/embeddings/client' -import { - createKnowledgeAclFixtureIds, - seedKnowledgeAclFixture, -} from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' -import { addDocument } from '@/lib/knowledge/connectors/sync-persistence' -import { processDocumentAsync } from '@/lib/knowledge/documents/service' - -/** The projections a processing commit writes where the Tin extension is absent. */ -const PROJECTIONS = ['embedding_search', 'embedding_keyword_search'] as const - -describe('document processing with the asynchronous projection', () => { - const ids = createKnowledgeAclFixtureIds() - let projector: postgres.Sql - - beforeAll(async () => { - fixtures.root = mkdtempSync(path.join(tmpdir(), 'sim-async-projection-')) - projector = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) - await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) - vi.spyOn(embeddingClient, 'assertKnowledgeEmbeddingCapacity').mockResolvedValue(undefined) - }) - - afterAll(async () => { - vi.restoreAllMocks() - await db.delete(knowledgeBase).where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) - await db.delete(organization).where(eq(organization.id, ids.organizationId)) - await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) - await rm(fixtures.root, { recursive: true, force: true }) - await projector?.end() - await db.$client.end() - }) - - const count = async (table: string, documentId: string) => { - const [row] = await db.$client.unsafe>( - `SELECT count(*)::int AS count FROM ${table} WHERE document_id = $1`, - [documentId] - ) - return row?.count - } - - it('commits the chunks with a mark and no projection rows, and a pass then projects them', async () => { - const file = await addDocument( - ids.knowledgeBaseId, - ids.connectorId, - 'google_drive', - { - externalId: 'async-projection-fixture', - title: 'Synthetic async projection fixture.txt', - content: 'Synthetic text', - mimeType: 'text/plain', - contentHash: 'synthetic-async-projection', - }, - { userId: ids.aliceId, workspaceId: ids.workspaceId }, - undefined, - 'workspace', - createContentSyncLease(ids.connectorId, ids.lockId) - ) - const chunks = Array.from({ length: 12 }, (_, index) => ({ - text: `Synthetic chunk ${index}`, - metadata: { startIndex: index * 20, endIndex: index * 20 + 19 }, - })) - fixtures.process.mockResolvedValue({ - chunks, - metadata: { chunkCount: chunks.length, tokenCount: 36, characterCount: 240 }, - }) - fixtures.embeddings.mockResolvedValue({ - embeddings: chunks.map(() => Array(1536).fill(0.2)), - billableTokens: 0, - modelName: 'text-embedding-3-small', - pricingId: 'text-embedding-3-small', - }) - const billing = await resolveBillingAttribution({ - actorUserId: ids.aliceId, - workspaceId: ids.workspaceId, - }) - fixtures.requestKnowledgeProjection.mockClear() - - await processDocumentAsync(ids.knowledgeBaseId, file.documentId, file, {}, billing) - - expect(await count('embedding', file.documentId)).toBe(chunks.length) - for (const table of PROJECTIONS) expect(await count(table, file.documentId)).toBe(0) - expect( - await db - .select({ content: knowledgeProjectionDirty.content }) - .from(knowledgeProjectionDirty) - .where(eq(knowledgeProjectionDirty.documentId, file.documentId)) - ).toEqual([{ content: true }]) - expect(fixtures.requestKnowledgeProjection).toHaveBeenCalledTimes(1) - - await runKnowledgeProjection(projector) - - for (const table of PROJECTIONS) { - expect(await count(table, file.documentId)).toBe(chunks.length) - } - const [source] = await db - .select({ connectorId: document.connectorId, acl: document.acl }) - .from(document) - .where(eq(document.id, file.documentId)) - expect( - await db - .selectDistinct({ connectorId: embeddingSearch.connectorId, acl: embeddingSearch.acl }) - .from(embeddingSearch) - .where(eq(embeddingSearch.documentId, file.documentId)) - ).toEqual([source]) - expect( - await db - .select() - .from(knowledgeProjectionDirty) - .where(eq(knowledgeProjectionDirty.documentId, file.documentId)) - ).toEqual([]) - }) -}) diff --git a/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts b/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts index 08d3adaa19b..ec084696138 100644 --- a/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts @@ -24,6 +24,14 @@ import { generateId } from '@sim/utils/id' import { serializeSignedCookie } from 'better-call' import { eq } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +/** Its search-index knowledge bases are read through indexed organization search, dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) + import { z } from 'zod' const metrics = vi.hoisted(() => ({ embeddingCalls: 0 })) diff --git a/apps/sim/lib/knowledge/__integration__/confluence-enrollment.integration.ts b/apps/sim/lib/knowledge/__integration__/confluence-enrollment.integration.ts index 8bea389ab89..d9709792656 100644 --- a/apps/sim/lib/knowledge/__integration__/confluence-enrollment.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/confluence-enrollment.integration.ts @@ -25,7 +25,7 @@ import { seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { startKnowledgeConnectorMemberEnrollment } from '@/lib/knowledge/application/connector-access' -import { listWorkspaceMemberConnectors } from '@/lib/knowledge/application/connectors' +import { resolveViewerConnectorMemberships } from '@/lib/knowledge/connectors/member-provisioning' describe('Confluence mirrored-identity self-enrollment', () => { const previousClient = { id: env.CONFLUENCE_CLIENT_ID, secret: env.CONFLUENCE_CLIENT_SECRET } @@ -106,15 +106,25 @@ describe('Confluence mirrored-identity self-enrollment', () => { await db.$client.end() }) - function discover(principal = bob, workspaceId = ids.workspaceId) { - return listWorkspaceMemberConnectors.execute({ principal, input: { workspaceId } }) + /** Bob's enrollment state for the fixture source, or null when none is offered. */ + async function viewerMembership() { + const [connector] = await db + .select() + .from(knowledgeConnector) + .where(eq(knowledgeConnector.id, ids.connectorId)) + const memberships = await resolveViewerConnectorMemberships({ + userId: ids.bobId, + workspaceId: ids.workspaceId, + connectors: [connector!], + }) + return memberships.get(ids.connectorId) ?? null } it('hides enrollment when the Confluence OAuth client is not configured', async () => { const client = { id: env.CONFLUENCE_CLIENT_ID, secret: env.CONFLUENCE_CLIENT_SECRET } try { Object.assign(env, { CONFLUENCE_CLIENT_ID: undefined, CONFLUENCE_CLIENT_SECRET: undefined }) - await expect(discover()).resolves.toEqual({ connectors: [] }) + await expect(viewerMembership()).resolves.toBeNull() } finally { Object.assign(env, { CONFLUENCE_CLIENT_ID: client.id, @@ -162,18 +172,7 @@ describe('Confluence mirrored-identity self-enrollment', () => { expect(before.credentials).toEqual([]) expect(before.policies).toHaveLength(1) expect(before.roles.find((role) => role.userId === ids.bobId)?.permissionType).toBe('read') - await expect(discover()).resolves.toMatchObject({ - connectors: [ - { - connectorId: ids.connectorId, - knowledgeBaseIsSearchIndex: true, - knowledgeBaseName: 'Renamed search index', - connectorType: 'confluence', - viewerMembership: 'not_enrolled', - memberSyncStatus: 'idle', - }, - ], - }) + await expect(viewerMembership()).resolves.toBe('not_enrolled') const { url } = await enroll() const link = new URL(url) @@ -191,9 +190,7 @@ describe('Confluence mirrored-identity self-enrollment', () => { invitationTokenHash: createHash('sha256').update(token).digest('hex'), }) expect(pending[0]!.invitationExpiresAt!.getTime()).toBeGreaterThan(Date.now()) - await expect(discover()).resolves.toMatchObject({ - connectors: [{ connectorId: ids.connectorId, viewerMembership: 'invited' }], - }) + await expect(viewerMembership()).resolves.toBe('invited') expect(await crawlerAuthority()).toEqual(before) }) @@ -269,7 +266,7 @@ describe('Confluence mirrored-identity self-enrollment', () => { .set({ options, status: state === 'disabled-group' ? 'disabled' : 'active' }) .where(eq(credentialGroup.id, groupId)) const before = await crawlerAuthority() - await expect(discover()).resolves.toEqual({ connectors: [] }) + await expect(viewerMembership()).resolves.toBeNull() await expect(enroll()).rejects.toMatchObject({ code: 'validation' }) expect(await enrollments()).toEqual([]) expect(await crawlerAuthority()).toEqual(before) @@ -284,9 +281,7 @@ describe('Confluence mirrored-identity self-enrollment', () => { .where(eq(credentialGroupEnrollment.credentialGroupId, groupId)) const revoked = await enrollments() const before = await crawlerAuthority() - await expect(discover()).resolves.toMatchObject({ - connectors: [{ connectorId: ids.connectorId, viewerMembership: 'revoked' }], - }) + await expect(viewerMembership()).resolves.toBe('revoked') await expect(enroll()).rejects.toMatchObject({ code: 'forbidden' }) expect(await enrollments()).toEqual(revoked) expect(await crawlerAuthority()).toEqual(before) @@ -294,9 +289,7 @@ describe('Confluence mirrored-identity self-enrollment', () => { it('requires the actor to verify their own email before issuing an invitation', async () => { await db.update(user).set({ emailVerified: false }).where(eq(user.id, ids.bobId)) - await expect(discover()).resolves.toMatchObject({ - connectors: [{ connectorId: ids.connectorId, viewerMembership: 'unverified_email' }], - }) + await expect(viewerMembership()).resolves.toBe('unverified_email') const before = await crawlerAuthority() await expect(enroll()).rejects.toThrow('Verify your email address') expect(await enrollments()).toEqual([]) @@ -315,7 +308,7 @@ describe('Confluence mirrored-identity self-enrollment', () => { .set(source) .where(eq(knowledgeConnector.id, ids.connectorId)) const before = await crawlerAuthority() - await expect(discover()).resolves.toEqual({ connectors: [] }) + await expect(viewerMembership()).resolves.toBeNull() await expect(enroll()).rejects.toMatchObject({ code: 'validation' }) expect(await enrollments()).toEqual([]) expect(await crawlerAuthority()).toEqual(before) @@ -325,7 +318,6 @@ describe('Confluence mirrored-identity self-enrollment', () => { it('rejects another workspace actor and an asserted foreign workspace without changing grants', async () => { const outsider: Principal = { ...bob, userId: foreign.bobId } const before = await crawlerAuthority() - await expect(discover(outsider)).rejects.toThrow('Insufficient workspace permissions') await expect(enroll(outsider)).rejects.toThrow('Insufficient workspace permissions') await expect( startKnowledgeConnectorMemberEnrollment.execute({ diff --git a/apps/sim/lib/knowledge/__integration__/directory-sync.integration.ts b/apps/sim/lib/knowledge/__integration__/directory-sync.integration.ts index 3ea03a7024d..357cf09fdb0 100644 --- a/apps/sim/lib/knowledge/__integration__/directory-sync.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/directory-sync.integration.ts @@ -56,7 +56,7 @@ import { seedKnowledgeAclFixture, seedKnowledgeMemberFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { resolveUserKnowledgeAccessScope } from '@/lib/knowledge/access/scope' +import { createUserKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' import { groupToken } from '@/lib/knowledge/access/tokens' import { persistExternalGroupMembership, @@ -793,14 +793,14 @@ describe('directory failure visibility in PostgreSQL', () => { groupId: 'identity-only', })! await syncExternalDirectoryGroups({ workspaceId: ids.workspaceId, directory, force: true }) + /** A fresh provider per read, so every check resolves the person's tokens again. */ + const tokensOf = async (userId: string) => + (await createUserKnowledgeAccessProvider(userId, { workspaceId: ids.workspaceId }).get()) + .tokens const hasGroup = async () => - (await resolveUserKnowledgeAccessScope(ids.aliceId, ids.workspaceId)).tokens.some( - (token) => token === expectedGroup - ) + (await tokensOf(ids.aliceId)).some((token) => token === expectedGroup) expect(await hasGroup()).toBe(true) - expect( - (await resolveUserKnowledgeAccessScope(ids.bobId, ids.workspaceId)).tokens - ).not.toContain(expectedGroup) + expect(await tokensOf(ids.bobId)).not.toContain(expectedGroup) await db .update(credential) diff --git a/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts b/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts new file mode 100644 index 00000000000..fea881f3e4b --- /dev/null +++ b/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts @@ -0,0 +1,152 @@ +/** + * Processing recovery while indexed organization search is dormant (the default: Live Search on). + * A search index is neither crawled nor projected in that state, so recovery must not re-admit its + * failed documents from stored bytes; a workspace knowledge base's failed documents are recovered + * exactly as before. + */ +import { mkdtempSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import path from 'node:path' +import { db } from '@sim/db' +import { + document, + knowledgeBase, + member, + organization, + outboxEvent, + user, + workspace, +} from '@sim/db/schema' +import { generateId } from '@sim/utils/id' +import { and, eq, inArray, sql } from 'drizzle-orm' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +const fixture = vi.hoisted(() => ({ root: '' })) +vi.mock('@/lib/core/config/trigger-runtime', () => ({ isInsideTriggerRun: () => false })) +/** Pinned to Live Search, whatever `SIM_SEARCH_LIVE` the run was started with. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => ({ + ...(await importOriginal>()), + isLiveEnterpriseSearchEnabled: true, +})) +vi.mock('@/lib/uploads/core/setup.server', () => ({ + get UPLOAD_DIR_SERVER() { + return fixture.root + }, +})) + +import { + createKnowledgeAclFixtureIds, + seedKnowledgeAclFixture, +} from '@/lib/knowledge/__integration__/seed-source-access-fixture' +import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' +import { addDocument } from '@/lib/knowledge/connectors/sync-persistence' +import { + KNOWLEDGE_DOCUMENT_RECOVERY_OUTBOX_EVENT, + recoverKnowledgeDocumentProcessing, +} from '@/lib/knowledge/documents/processing-recovery' +import { QUEUED_DISPATCH_GRACE_MS } from '@/lib/knowledge/documents/types' + +type FixtureIds = ReturnType + +const fixtures: FixtureIds[] = [] +const old = () => new Date(Date.now() - QUEUED_DISPATCH_GRACE_MS - 60_000) + +/** A connector document whose processing failed long enough ago to be recoverable. */ +async function failedFile(ids: FixtureIds, organizationOwned: boolean) { + const file = await addDocument( + ids.knowledgeBaseId, + ids.connectorId, + 'google_drive', + { + externalId: generateId(), + title: 'Retained fixture.txt', + content: 'Recovery fixture content retained from the source.', + mimeType: 'text/plain', + contentHash: 'fixture-retained-v1', + }, + organizationOwned + ? { userId: ids.aliceId, workspaceId: null, organizationId: ids.organizationId } + : { userId: ids.aliceId, workspaceId: ids.workspaceId }, + undefined, + 'admin', + createContentSyncLease(ids.connectorId, ids.lockId) + ) + await db + .update(document) + .set({ + processingStatus: 'failed', + processingAttempts: 1, + uploadedAt: old(), + processingCompletedAt: old(), + processingQueuedAt: old(), + processingQueueToken: 'old-fixture-generation', + processingError: 'Synthetic prior failure', + }) + .where(eq(document.id, file.documentId)) + return file +} + +async function recoveryEvents(ids: FixtureIds) { + return db + .select({ id: outboxEvent.id }) + .from(outboxEvent) + .where( + and( + eq(outboxEvent.eventType, KNOWLEDGE_DOCUMENT_RECOVERY_OUTBOX_EVENT), + sql`${outboxEvent.payload}->>'knowledgeBaseId' = ${ids.knowledgeBaseId}` + ) + ) +} + +beforeAll(() => { + fixture.root = mkdtempSync(path.join(tmpdir(), 'sim-dormant-recovery-')) +}) + +afterAll(async () => { + for (const ids of fixtures) { + await db + .delete(outboxEvent) + .where(sql`${outboxEvent.payload}->>'knowledgeBaseId' = ${ids.knowledgeBaseId}`) + await db.delete(knowledgeBase).where(eq(knowledgeBase.id, ids.knowledgeBaseId)) + await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) + await db.delete(organization).where(eq(organization.id, ids.organizationId)) + await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) + } + await rm(fixture.root, { recursive: true, force: true }) + await db.$client.end() +}) + +describe('processing recovery while indexed organization search is dormant', () => { + it('re-admits a workspace document and leaves a search-index document alone', async () => { + const workspaceIds = createKnowledgeAclFixtureIds() + const indexIds = createKnowledgeAclFixtureIds() + fixtures.push(workspaceIds, indexIds) + await seedKnowledgeAclFixture(workspaceIds, { connectorType: 'google_drive' }) + await seedKnowledgeAclFixture(indexIds, { connectorType: 'google_drive' }) + await db.insert(member).values({ + id: generateId(), + organizationId: indexIds.organizationId, + userId: indexIds.aliceId, + role: 'owner', + }) + await db + .update(knowledgeBase) + .set({ workspaceId: null, organizationId: indexIds.organizationId, isSearchIndex: true }) + .where(eq(knowledgeBase.id, indexIds.knowledgeBaseId)) + const workspaceFile = await failedFile(workspaceIds, false) + const indexFile = await failedFile(indexIds, true) + + await recoverKnowledgeDocumentProcessing() + + expect(await recoveryEvents(workspaceIds)).toHaveLength(1) + expect(await recoveryEvents(indexIds)).toEqual([]) + const rows = await db + .select({ id: document.id, attempts: document.processingAttempts }) + .from(document) + .where(inArray(document.id, [workspaceFile.documentId, indexFile.documentId])) + const attempts = new Map(rows.map((row) => [row.id, row.attempts])) + expect(attempts.get(workspaceFile.documentId)).toBe(2) + expect(attempts.get(indexFile.documentId)).toBe(1) + }) +}) diff --git a/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts b/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts index 8a6348eb04d..451f96df73b 100644 --- a/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts @@ -17,6 +17,12 @@ import { eq, inArray } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' const provider = vi.hoisted(() => ({ list: vi.fn(), get: vi.fn(), changes: vi.fn() })) +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/connectors/registry.server', () => ({ CONNECTOR_REGISTRY: { google_drive: { diff --git a/apps/sim/lib/knowledge/__integration__/filtered-search.integration.ts b/apps/sim/lib/knowledge/__integration__/filtered-search.integration.ts index b1418400258..e08292154f5 100644 --- a/apps/sim/lib/knowledge/__integration__/filtered-search.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/filtered-search.integration.ts @@ -18,7 +18,7 @@ import { } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { confluencePageAcl } from '@/lib/knowledge/access/confluence-permissions' import { resolveKnowledgeAccessScope } from '@/lib/knowledge/access/scope' -import { executeKnowledgeSearch } from '@/lib/knowledge/search/queries' +import { retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' import { embeddingVectorValues } from '@/lib/knowledge/vector-columns' afterAll(async () => { @@ -130,7 +130,7 @@ describe.each([384, 768, 1024, 1536, 3072] as const)( async function search(principal: Principal, mode: 'vector' | 'hybrid' = 'vector') { const access = await resolveKnowledgeAccessScope(principal, { workspaceId: ids.workspaceId }) - const rows = await executeKnowledgeSearch({ + const { rows, retrieval } = await retrieveKnowledgeSearch({ knowledgeBaseIds: [ids.knowledgeBaseId, secondBaseId], topK: 3, access, @@ -145,6 +145,7 @@ describe.each([384, 768, 1024, 1536, 3072] as const)( { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'common' }, ], }) + expect(retrieval.status).toBe('complete') return rows.map((row) => fixtures.find((fixture) => fixture.embeddingId === row.id)!.name) } diff --git a/apps/sim/lib/knowledge/__integration__/github-member.integration.ts b/apps/sim/lib/knowledge/__integration__/github-member.integration.ts index 3f064fc2810..b680357e2f6 100644 --- a/apps/sim/lib/knowledge/__integration__/github-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/github-member.integration.ts @@ -32,6 +32,12 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, isNull, sql } from 'drizzle-orm' import { afterAll, afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, @@ -85,7 +91,6 @@ import { listKnowledgeConnectorDocuments, } from '@/lib/knowledge/application/connectors' import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' -import { readIndexedKnowledgeDocument } from '@/lib/knowledge/application/read-indexed-document' import { searchKnowledge } from '@/lib/knowledge/application/search' import { readSearchSourceOverview } from '@/lib/knowledge/application/search-source-overview' import { listSearchSources } from '@/lib/knowledge/application/search-sources' @@ -97,6 +102,7 @@ import { } from '@/lib/knowledge/connectors/sync-limits' import { getDocuments } from '@/lib/knowledge/documents/service' import { getTagUsageStats } from '@/lib/knowledge/tags/service' +import { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' import { deleteFile } from '@/lib/uploads/core/storage-service' import { downloadFileFromUrl } from '@/lib/uploads/utils/file-utils.server' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' diff --git a/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts b/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts index e515978e548..774b7d00846 100644 --- a/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts @@ -36,6 +36,14 @@ import { serializeSignedCookie } from 'better-call' import { and, eq, inArray, isNull, or } from 'drizzle-orm' import { NextRequest } from 'next/server' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +/** Its search-index knowledge bases are read through indexed organization search, dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) + import type { EmbedOptions } from '@/lib/embeddings/types' const fixture = vi.hoisted(() => ({ embeddingCalls: 0 })) @@ -86,9 +94,9 @@ import { listKnowledgeChunks } from '@/lib/knowledge/application/chunks' import { updateKnowledgeConnectorAccess } from '@/lib/knowledge/application/connector-access' import { updateKnowledgeConnector } from '@/lib/knowledge/application/connectors' import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' -import { readSearchDocument } from '@/lib/knowledge/application/read-search-document' import { searchKnowledge } from '@/lib/knowledge/application/search' import { executeSync } from '@/lib/knowledge/connectors/sync-engine' +import { readSearchDocument } from '@/lib/sim-search/indexed/documents/read-search-document' import * as storage from '@/lib/uploads/core/storage-service' import { downloadFileFromUrl } from '@/lib/uploads/utils/file-utils.server' import { PATCH as updateConnectorRoute } from '@/app/api/knowledge/[id]/connectors/[connectorId]/route' diff --git a/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts b/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts index 02bebeabca4..716af7b1b66 100644 --- a/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts @@ -24,6 +24,12 @@ import { eq, inArray } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' const counters = vi.hoisted(() => ({ embeddedTexts: 0 })) +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, diff --git a/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts b/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts index f00ea34e458..2a66fdbf151 100644 --- a/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts @@ -24,6 +24,12 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, isNull } from 'drizzle-orm' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, diff --git a/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts b/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts index a8f970c6b87..e9d76507746 100644 --- a/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts @@ -25,6 +25,12 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, isNull } from 'drizzle-orm' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, diff --git a/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts b/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts index 22490631337..9eff697046f 100644 --- a/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts @@ -10,13 +10,12 @@ import { seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' -import { - forgetProjectionFilled, - resolvePermittedDocuments, - retrieveKnowledgeSearch, - VECTOR_PROBE_DOCUMENT_LIMIT, -} from '@/lib/knowledge/search/queries' +import { VECTOR_PROBE_DOCUMENT_LIMIT } from '@/lib/knowledge/search/candidates' +import { retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' import { embeddingVectorValues } from '@/lib/knowledge/vector-columns' +import { resolveSearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' +import { resolvePermittedDocuments } from '@/lib/sim-search/indexed/retrieval/permitted' +import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' describe('API-key KB block fan-out', () => { const ids = createKnowledgeAclFixtureIds() @@ -135,10 +134,10 @@ describe('API-key KB block fan-out', () => { statements.filter((query) => query.includes(fragment)) /** * Every statement runs under the leg's deadline: the candidate search applies it with the - * scan settings in one statement, and the probe, exact ranking, document-backed page - * and hydration each open with one of their own. + * scan settings in one statement, and the probe, exact ranking and hydration each open + * with one of their own. */ - expect(matching('statement_timeout')).toHaveLength(bases.length * 5) + expect(matching('statement_timeout')).toHaveLength(bases.length * 4) expect(matching('IS NOT NULL AS unfilled')).toHaveLength(0) /** * A scope this small leaves the bounded traversal short of its candidate limit, so every @@ -147,8 +146,8 @@ describe('API-key KB block fan-out', () => { expect(matching('hnsw.iterative_scan')).toHaveLength(bases.length) expect(matching('AS visible')).toHaveLength(bases.length) expect(matching(') + 0 LIMIT')).toHaveLength(bases.length) - /** Ordinary KBs read page identities from documents without requiring a filled projection. */ - expect(matching('"embedding_search"."id" = ANY(')).toHaveLength(bases.length) + /** The walk and the rescue carry each candidate's document, so no page read follows them. */ + expect(matching('"embedding_search"."id" = ANY(')).toHaveLength(0) /** The probe enumerates visible documents and reports saturation; it never ranks them. */ expect( statements.filter( @@ -178,9 +177,12 @@ describe('API-key KB block fan-out', () => { 'text/plain', 'completed', ARRAY['ws']::text[] FROM generate_series(1, ${VECTOR_PROBE_DOCUMENT_LIMIT + 1}) AS n `) + const access = { kind: 'user' as const, userId: ids.bobId, tokens: ['pub', 'ws'] } const permitted = await resolvePermittedDocuments({ knowledgeBaseIds: [bases[0].id], - access: { kind: 'user', userId: ids.bobId, tokens: ['pub', 'ws'] }, + access, + accessPlan: await resolveSearchAccessPlan([bases[0].id], access), + filtered: false, }) expect(permitted).toEqual({ kind: 'bounded', diff --git a/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts b/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts index be65e08111f..804b90ac24c 100644 --- a/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts @@ -1,21 +1,21 @@ /** * The knowledge projector and the readers that must stay correct while it lags. A GitHub member - * source's document is changed by a writer in either projection mode — synchronous, as every - * writer before the projector, or asynchronous, where the writer only marks the document — and - * search is checked before the projector runs: a revoked member is refused and a granted one is - * served, on the vector and keyword legs and under the source filter, and a disabled or deleted - * chunk is gone at once. The projector's own contract follows: it converges the rows and removes - * the mark, keeps a mark that a write bumped during its pass, survives a document deleted under - * it, writes in pages bounded by chunk rows, fills rows written before they carried a source and - * ACL, and an asynchronous commit writes no projection row at all. + * source's document in a search index is changed by a writer in either projection mode — + * synchronous, as every writer now is, or deferred, as writers of releases that carried the + * `knowledge-async-projection` flag were, leaving only a mark — and search is checked before the + * projector runs: a revoked member is refused and a granted one is served, on the vector and + * keyword legs and under the source filter, and a disabled or deleted chunk is gone at once. The + * projector's own contract follows: it converges the rows and removes the mark, keeps a mark that + * a write bumped during its pass, survives a document deleted under it, writes in pages bounded by + * chunk rows, releases workspace marks with nothing to project without a pass, and a deferred + * commit writes no projection row at all. */ import { createHash } from 'node:crypto' import { db } from '@sim/db' import { - DEFER_KNOWLEDGE_PROJECTION, - FILL_MARK_CEILING, + hasKnowledgeProjectionWork, type KnowledgeProjection, - markUnfilledProjectionDocuments, + releaseSettledMarks, runKnowledgeProjection, } from '@sim/db/knowledge-projection' import { @@ -37,50 +37,41 @@ import { } from '@sim/db/schema' import { sleep } from '@sim/utils/helpers' import { generateId } from '@sim/utils/id' -import { and, eq, inArray, isNull, sql } from 'drizzle-orm' +import { and, eq, inArray, sql } from 'drizzle-orm' import postgres from 'postgres' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) /** The TINQL `resolveTinKeywordQuery` renders for `fixture`: its `english` stem, quoted. */ -vi.mock('@/lib/knowledge/search/tin-keyword', () => ({ +vi.mock('@/lib/sim-search/indexed/retrieval/tin-keyword', () => ({ resolveTinKeywordQuery: async () => '"fixtur"', })) -/** The flags a test turns on; connector writers read `knowledge-async-projection` from here. */ -const { enabledFlags, requestKnowledgeProjection } = vi.hoisted(() => ({ - enabledFlags: new Set(), - requestKnowledgeProjection: vi.fn(async () => {}), -})) -vi.mock('@/lib/core/config/feature-flags', () => ({ - isFeatureEnabled: async (flag: string) => enabledFlags.has(flag), -})) -/** - * Writers ask for a pass once they commit; here that request is only recorded, so the passes each - * test runs are the only ones and a background pass cannot converge rows a test is inspecting. - */ -vi.mock('@/lib/knowledge/projection/enqueue', () => ({ requestKnowledgeProjection })) +vi.mock('@/lib/core/config/feature-flags', () => ({ isFeatureEnabled: async () => false })) import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { - projectionCandidateAccessCondition, - type SearchAccessPlan, -} from '@/lib/knowledge/access/predicate' import type { GitHubInstallationReadGrant, KnowledgeAccessProvider, UserAccessScope, } from '@/lib/knowledge/access/types' import { leaseTransaction } from '@/lib/knowledge/connectors/sync-lock' -import { - executeKeywordSearch, - forgetProjectionFilled, - handleVectorOnlySearch, - liveSourceAccessFor, -} from '@/lib/knowledge/search/queries' +import { liveSourceAccessForConnectors } from '@/lib/knowledge/search/candidates' import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' +import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' +import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' +import type { IndexedRetrievalContext } from '@/lib/sim-search/indexed/retrieval/permitted' +import { projectionCandidateAccessCondition } from '@/lib/sim-search/indexed/retrieval/projection-access' +import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' +import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' const ids = createKnowledgeAclFixtureIds() const connectorId = generateId() @@ -140,7 +131,15 @@ const grant: GitHubInstallationReadGrant = { repositoryId, } -function searchInputs() { +const searchInputs = { + knowledgeBaseIds: [ids.knowledgeBaseId], + topK: 20, + access: scope, + queryVector, +} + +/** A narrow reader of the search index, who proves the installation grant live. */ +function searchContext(): IndexedRetrievalContext { const granted = { ...scope, githubInstallationGrants: [grant] } const accessProvider: KnowledgeAccessProvider = { get: async () => scope, @@ -149,25 +148,19 @@ function searchInputs() { liveSourceConnectorCondition: async () => null, } return { - knowledgeBaseIds: [ids.knowledgeBaseId], - topK: 20, access: scope, - accessProvider, accessPlan: plan, - liveSourceAccess: liveSourceAccessFor(scope, plan, accessProvider), - queryVector, + filtered: false, + permitted: { kind: 'unbounded', broad: false }, + liveSourceAccess: liveSourceAccessForConnectors( + plan.connectors.liveProofRequired, + accessProvider + ), } } const keywordIds = async () => - ( - await executeKeywordSearch({ - ...searchInputs(), - query: 'fixture', - permitted: { kind: 'unbounded', broad: false }, - searchIndexOnly: true, - }) - ) + (await executeIndexedKeywordSearch({ ...searchInputs, query: 'fixture' }, searchContext())) .map((row) => row.id) .sort() @@ -177,13 +170,7 @@ const keywordIds = async () => * a chunk can be pruned from every neighbour list and never be reached, however far the walk goes. */ const vectorIds = async () => - ( - await handleVectorOnlySearch({ - ...searchInputs(), - distanceThreshold: 2, - permitted: { kind: 'unbounded', broad: false }, - }) - ) + (await selectIndexedVectorResults({ ...searchInputs, distanceThreshold: 2 }, searchContext())) .map((row) => row.id) .sort() @@ -217,13 +204,19 @@ async function admitted(): Promise> { type Mode = 'sync' | 'async' +/** + * What a writer of a release that deferred its projection selected first: the setting that skips + * the synchronous projection triggers, which the database still honours. + */ +const DEFER_PROJECTION = `SELECT set_config('sim.projection_mode', 'async', true)` + /** Runs a write in a transaction of the given projection mode, as a knowledge writer would. */ function write( mode: Mode, work: (tx: Parameters[0]>[0]) => Promise ) { return db.transaction(async (tx) => { - if (mode === 'async') await tx.execute(sql.raw(`SELECT ${DEFER_KNOWLEDGE_PROJECTION}`)) + if (mode === 'async') await tx.execute(sql.raw(DEFER_PROJECTION)) await work(tx) }) } @@ -263,8 +256,8 @@ const rowAcl = async (table: typeof embeddingSearch | typeof embeddingKeywordTin /** The projector's connection: a pass holds its per-document advisory locks on it. */ let projector: postgres.Sql -const project = (options: Parameters[1] = {}) => - runKnowledgeProjection(projector, options) +const project = (options: Partial[1]> = {}) => + runKnowledgeProjection(projector, { searchIndexes: true, ...options }) /** Shims for the Tin extension, which the test database does not carry. */ let createdTinShims = false @@ -427,7 +420,6 @@ afterAll(async () => { * so no synchronous trigger would write the Tin row. */ beforeEach(async () => { - requestKnowledgeProjection.mockClear() await db.delete(embedding).where(eq(embedding.documentId, documentId)) await db .update(document) @@ -662,30 +654,19 @@ describe('the projector', () => { ).toEqual([]) }) - it.each([false, true])( - "leaves a connector ACL page's projection rows to the projector only while the flag is on (%s)", - async (flagOn) => { - if (flagOn) enabledFlags.add('knowledge-async-projection') - try { - await leaseTransaction(connectorId)((tx) => - tx - .update(document) - .set({ acl: aclOf('bob') }) - .where(eq(document.id, documentId)) - ) - } finally { - enabledFlags.delete('knowledge-async-projection') - } - expect(requestKnowledgeProjection).toHaveBeenCalledOnce() - expect(await markOf()).toMatchObject({ content: false }) - expect((await rowAcl(embeddingSearch))?.acl).toEqual( - flagOn ? aclOf('alice', 'bob') : aclOf('bob') - ) - expect(await admitted()).toEqual({ vector: [], keyword: [] }) - await project() - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) - } - ) + it("rewrites a connector ACL page's projection rows in the page's own statement", async () => { + await leaseTransaction(connectorId)((tx) => + tx + .update(document) + .set({ acl: aclOf('bob') }) + .where(eq(document.id, documentId)) + ) + expect(await markOf()).toMatchObject({ content: false }) + expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) + expect(await admitted()).toEqual({ vector: [], keyword: [] }) + await project() + expect(await markOf()).toBeUndefined() + }) it('splits the marks between passes that start together', async () => { const documents = await Promise.all( @@ -716,7 +697,7 @@ describe('the projector', () => { const second = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) try { /** Each page lingers, so neither pass can finish the batch before the other starts. */ - const lingering = { onPage: () => sleep(25) } + const lingering = { onPage: () => sleep(25), searchIndexes: true } const passes = await Promise.all([ project(lingering), runKnowledgeProjection(second, lingering), @@ -749,148 +730,6 @@ describe('the projector', () => { expect(await markOf()).toBeUndefined() }) - it('fills rows written before they carried a source and ACL, and converges them', async () => { - for (const table of [embeddingSearch, embeddingKeywordTin]) { - await db - .update(table) - .set({ connectorId: null, acl: null }) - .where(eq(table.documentId, documentId)) - } - expect(await markOf()).toBeUndefined() - let cursor: Parameters[1] | null - let marked = 0 - while (cursor !== null) { - const fill = await markUnfilledProjectionDocuments(projector, cursor) - marked += fill.marked - cursor = fill.cursor - await project() - } - expect(marked).toBeGreaterThanOrEqual(1) - expect(await rowAcl(embeddingSearch)).toMatchObject({ acl: aclOf('alice', 'bob'), connectorId }) - expect(await rowAcl(embeddingKeywordTin)).toMatchObject({ - acl: aclOf('alice', 'bob'), - connectorId, - }) - expect(await markOf()).toBeUndefined() - }) - - it('fills every document when more are unfilled than the fill may mark at once', async () => { - const documents = Array.from({ length: FILL_MARK_CEILING + 20 }, () => generateId()) - await db.insert(document).values( - documents.map((id, index) => ({ - id, - connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: `fill-${index}`, - filename: `fill-${index}.md`, - fileUrl: `https://fixture.test/fill-${index}`, - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed' as const, - acl: aclOf('alice', 'bob'), - })) - ) - /** Two chunks each, whose random ids interleave the documents' rows in the unfilled index. */ - await write('async', (tx) => - tx - .insert(embedding) - .values( - documents.flatMap((id) => - [0, 1].map((chunkIndex) => ({ ...chunkRow(generateId(), chunkIndex), documentId: id })) - ) - ) - ) - await project() - /** Only the vector rows, so no other projection's unfilled rows lead the fill back to them. */ - await db - .update(embeddingSearch) - .set({ connectorId: null, acl: null }) - .where(inArray(embeddingSearch.documentId, documents)) - let cursor: Parameters[1] | null - let marked = 0 - while (cursor !== null) { - const fill = await markUnfilledProjectionDocuments(projector, cursor) - marked += fill.marked - cursor = fill.cursor - await project() - } - expect(marked).toBeGreaterThanOrEqual(documents.length) - expect( - await db - .select({ id: embeddingSearch.id }) - .from(embeddingSearch) - .where(and(inArray(embeddingSearch.documentId, documents), isNull(embeddingSearch.acl))) - ).toEqual([]) - await db.delete(document).where(inArray(document.id, documents)) - }) - - it('marks what it can while a document it chose is deleted under it', async () => { - const [deleted, kept] = [generateId(), generateId()] - await db.insert(document).values( - [deleted, kept].map((id, index) => ({ - id, - connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: `fill-race-${index}`, - filename: `fill-race-${index}.md`, - fileUrl: `https://fixture.test/fill-race-${index}`, - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed' as const, - acl: aclOf('alice', 'bob'), - })) - ) - await write('async', (tx) => - tx - .insert(embedding) - .values([deleted, kept].map((id) => ({ ...chunkRow(generateId(), 0), documentId: id }))) - ) - await project() - for (const table of [embeddingSearch, embeddingKeywordTin]) { - await db - .update(table) - .set({ connectorId: null, acl: null }) - .where(inArray(table.documentId, [deleted, kept])) - } - const deleter = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) - try { - /** The deletion is under way when the fill reads, and commits while the fill still runs. */ - let fill: ReturnType | undefined - let settled = false - await deleter.begin(async (tx) => { - await tx`DELETE FROM document WHERE id = ${deleted}` - fill = markUnfilledProjectionDocuments(projector) - void fill.then( - () => { - settled = true - }, - () => { - settled = true - } - ) - await vi.waitFor( - async () => { - const [row] = await db.execute<{ waiting: boolean }>( - sql`SELECT EXISTS (SELECT 1 FROM pg_locks WHERE NOT granted) AS waiting` - ) - expect(settled || Boolean(row?.waiting)).toBe(true) - }, - { timeout: 5_000, interval: 10 } - ) - }) - await expect(fill).resolves.toMatchObject({ marked: expect.any(Number) }) - const marks = await db - .select({ documentId: knowledgeProjectionDirty.documentId }) - .from(knowledgeProjectionDirty) - .where(inArray(knowledgeProjectionDirty.documentId, [deleted, kept])) - expect(marks.map((mark) => mark.documentId)).toEqual([kept]) - } finally { - await deleter.end() - await project() - await db.delete(document).where(inArray(document.id, [deleted, kept])) - } - }) - it.each(['sync', 'async'] as const)( 'writes %s projection rows from a chunk commit only when the writer did not defer them', async (mode) => { @@ -921,4 +760,177 @@ describe('the projector', () => { expect(await markOf()).toMatchObject({ content: mode === 'async' }) } ) + + it('owes a pass for content and, while indexed search is on, search-index marks, asking only for content behind an undrained release', async () => { + const workspaceBaseId = generateId() + const [workspaceDocument, contentDocument] = [generateId(), generateId()] + await db.insert(knowledgeBase).values({ + id: workspaceBaseId, + userId: ids.aliceId, + workspaceId: ids.workspaceId, + name: 'Workspace work fixture', + chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, + }) + /** Thrown to roll the transaction back, so the marks of other suites are never touched for good. */ + const rollback = new Error('rollback') + try { + await db.insert(document).values( + [workspaceDocument, contentDocument].map((id, index) => ({ + id, + knowledgeBaseId: workspaceBaseId, + filename: `work-${index}.md`, + fileUrl: `https://fixture.test/work-${index}`, + fileSize: 12, + mimeType: 'text/plain', + processingStatus: 'completed' as const, + })) + ) + const answers: Record = {} + const indexed = { searchIndexes: true } + const answer = async (tx: postgres.TransactionSql, label: string) => { + answers[label] = { + drained: await hasKnowledgeProjectionWork(tx, { drained: true }, indexed), + undrained: await hasKnowledgeProjectionWork(tx, { drained: false }, indexed), + dormant: await hasKnowledgeProjectionWork( + tx, + { drained: true }, + { searchIndexes: false } + ), + } + } + await projector + .begin(async (tx) => { + await tx`DELETE FROM knowledge_projection_dirty` + await tx`SELECT mark_knowledge_projection(ARRAY[${workspaceDocument}]::text[], false)` + await answer(tx, 'workspace') + await tx`SELECT mark_knowledge_projection(ARRAY[${documentId}]::text[], false)` + await answer(tx, 'search index') + await tx`SELECT mark_knowledge_projection(ARRAY[${contentDocument}]::text[], true)` + await answer(tx, 'content') + throw rollback + }) + .catch((error) => { + if (error !== rollback) throw error + }) + expect(answers).toEqual({ + workspace: { drained: false, undrained: false, dormant: false }, + 'search index': { drained: true, undrained: false, dormant: false }, + content: { drained: true, undrained: true, dormant: true }, + }) + } finally { + await db.delete(knowledgeBase).where(eq(knowledgeBase.id, workspaceBaseId)) + } + }) + + it('releases marks with nothing to project, keeping search-index marks for a pass only while indexed search is on', async () => { + const workspaceBaseId = generateId() + const [synced, deferred, held] = [generateId(), generateId(), generateId()] + await db.insert(knowledgeBase).values({ + id: workspaceBaseId, + userId: ids.aliceId, + workspaceId: ids.workspaceId, + name: 'Workspace fixture', + chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, + }) + try { + await db.insert(document).values( + [synced, deferred, held].map((id, index) => ({ + id, + knowledgeBaseId: workspaceBaseId, + filename: `workspace-${index}.md`, + fileUrl: `https://fixture.test/workspace-${index}`, + fileSize: 12, + mimeType: 'text/plain', + processingStatus: 'completed' as const, + })) + ) + const workspaceChunk = (id: string, documentId: string) => ({ + ...chunkRow(id, 0), + documentId, + knowledgeBaseId: workspaceBaseId, + }) + const deferredChunk = generateId() + await write('sync', (tx) => + tx + .insert(embedding) + .values([workspaceChunk(generateId(), synced), workspaceChunk(generateId(), held)]) + ) + await write('async', (tx) => + tx.insert(embedding).values(workspaceChunk(deferredChunk, deferred)) + ) + /** A search-index mark with nothing but a source and ACL change is still a pass's. */ + await db + .update(document) + .set({ acl: aclOf('bob') }) + .where(eq(document.id, documentId)) + const marked = async () => + ( + await db + .select({ documentId: knowledgeProjectionDirty.documentId }) + .from(knowledgeProjectionDirty) + .where( + inArray(knowledgeProjectionDirty.documentId, [synced, deferred, held, documentId]) + ) + ) + .map((row) => row.documentId) + .sort() + + /** A writer re-marking `held` holds its mark until it commits; the release passes it over. */ + const writer = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) + let release: () => void = () => {} + const holding = new Promise((resolve) => { + release = resolve + }) + let locked: () => void = () => {} + const marking = new Promise((resolve) => { + locked = resolve + }) + try { + const writing = writer.begin(async (tx) => { + await tx`SELECT mark_knowledge_projection(ARRAY[${held}]::text[], false)` + locked() + await holding + }) + await marking + expect( + (await releaseSettledMarks(projector, Number.POSITIVE_INFINITY, { searchIndexes: true })) + .released + ).toBeGreaterThanOrEqual(1) + expect(await marked()).toEqual([deferred, held, documentId].sort()) + release() + await writing + } finally { + release() + await writer.end() + } + expect( + (await releaseSettledMarks(projector, Number.POSITIVE_INFINITY, { searchIndexes: true })) + .released + ).toBeGreaterThanOrEqual(1) + expect(await marked()).toEqual([deferred, documentId].sort()) + + /** The deferred chunk has no row until a pass writes it, and the pass settles both marks. */ + const deferredRow = async () => + db + .select({ id: embeddingSearch.id }) + .from(embeddingSearch) + .where(eq(embeddingSearch.id, deferredChunk)) + expect(await deferredRow()).toEqual([]) + await project() + expect(await deferredRow()).toEqual([{ id: deferredChunk }]) + expect(await marked()).toEqual([]) + expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) + + /** While indexed search is dormant, nothing reads a search-index mark, so it is released too. */ + await db + .update(document) + .set({ acl: aclOf('alice') }) + .where(eq(document.id, documentId)) + expect(await marked()).toEqual([documentId]) + await releaseSettledMarks(projector, Number.POSITIVE_INFINITY, { searchIndexes: false }) + expect(await marked()).toEqual([]) + } finally { + await db.delete(knowledgeBase).where(eq(knowledgeBase.id, workspaceBaseId)) + } + }) }) diff --git a/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts b/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts index 76aab3ba344..7b2d0f2a111 100644 --- a/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts @@ -43,10 +43,11 @@ const fixtures = vi.hoisted(() => ({ afterResponse: [] as Array<() => Promise>, })) /** This ingestion suite covers the explicit rollback backend; live Search has its own suites. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => ({ - ...(await importOriginal()), - isLiveEnterpriseSearchEnabled: false, -})) +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/core/utils/after-response', () => ({ afterResponse: (task: () => Promise) => fixtures.afterResponse.push(task), })) @@ -84,15 +85,15 @@ import { import { confluencePageAcl } from '@/lib/knowledge/access/confluence-permissions' import { listKnowledgeChunks } from '@/lib/knowledge/application/chunks' import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' -import { listKnowledgeBaseCatalog } from '@/lib/knowledge/application/knowledge-bases' -import { readIndexedKnowledgeDocument } from '@/lib/knowledge/application/read-indexed-document' +import { listKnowledgeBases } from '@/lib/knowledge/application/knowledge-bases' import { prepareSearchSource } from '@/lib/knowledge/application/sim-search' -import { searchScopedKnowledge } from '@/lib/knowledge/application/workspace-search' import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' import { addDocument, persistDocumentAcls } from '@/lib/knowledge/connectors/sync-persistence' import { processDocumentAsync } from '@/lib/knowledge/documents/service' import { getSearchMcpUrl } from '@/lib/knowledge/mcp/urls' import { replaceKnowledgeEmbeddingSecretProvenanceInTx } from '@/lib/knowledge/secret-provenance' +import { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' +import { searchScopedKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' import { DELETE, GET, POST } from '@/app/api/mcp/search/organizations/[organizationId]/route' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' @@ -455,7 +456,7 @@ describe('organization Search MCP rollback backend with real ingestion and curre userId: otherAdminId, }), ]) - const catalog = await listKnowledgeBaseCatalog.execute({ + const catalog = await listKnowledgeBases.execute({ principal: alicePrincipal, input: { workspaceId }, }) diff --git a/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts b/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts index c653e69eac0..2003f25e31e 100644 --- a/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts @@ -25,6 +25,12 @@ import { and, eq, inArray, sql } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' const fixtureStorage = vi.hoisted(() => ({ root: '' })) +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/uploads/core/setup.server', () => ({ get UPLOAD_DIR_SERVER() { return fixtureStorage.root @@ -49,7 +55,6 @@ import { seedKnowledgeMemberFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { searchKnowledge } from '@/lib/knowledge/application/search' -import { searchScopedKnowledge } from '@/lib/knowledge/application/workspace-search' import { materializeDocumentAcls, recordMemberObservations, @@ -61,6 +66,7 @@ import { knowledgeDocumentProcessingOutboxHandlers } from '@/lib/knowledge/docum import { assertDocumentProcessingPayload } from '@/lib/knowledge/documents/processing-payload' import * as providerContinuation from '@/lib/knowledge/documents/processing-provider-continuation' import { processDocumentsWithQueue } from '@/lib/knowledge/documents/service' +import { searchScopedKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' const PNG = Buffer.from( 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+jRZkAAAAASUVORK5CYII=', diff --git a/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts b/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts index a4cb6ff0afa..9eea5b6cc4a 100644 --- a/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts @@ -21,6 +21,12 @@ import { eq, inArray } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' const fixtures = vi.hoisted(() => ({ storageRoot: '' })) +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/uploads/core/setup.server', () => ({ get UPLOAD_DIR_SERVER() { return fixtures.storageRoot @@ -45,13 +51,13 @@ import { seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { confluencePageAcl } from '@/lib/knowledge/access/confluence-permissions' -import { - type ReadIndexedKnowledgeDocumentInput, - readIndexedKnowledgeDocument, -} from '@/lib/knowledge/application/read-indexed-document' import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' import { addDocument, persistDocumentAcls } from '@/lib/knowledge/connectors/sync-persistence' import { processDocumentAsync } from '@/lib/knowledge/documents/service' +import { + type ReadIndexedKnowledgeDocumentInput, + readIndexedKnowledgeDocument, +} from '@/lib/sim-search/indexed/documents/read-indexed-document' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' describe('indexed document references', () => { diff --git a/apps/sim/lib/knowledge/__integration__/scale.integration.ts b/apps/sim/lib/knowledge/__integration__/scale.integration.ts index 7019a02b51e..7035ec30c80 100644 --- a/apps/sim/lib/knowledge/__integration__/scale.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/scale.integration.ts @@ -21,7 +21,8 @@ import { runConnectorContentPass } from '@/lib/knowledge/connectors/sync-content import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' import { persistDocumentAcls } from '@/lib/knowledge/connectors/sync-persistence' import { loadPageCorpus } from '@/lib/knowledge/connectors/sync-primitives' -import { executeKnowledgeSearch, getStructuredTagFilters } from '@/lib/knowledge/search/queries' +import { retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' +import { getStructuredTagFilters } from '@/lib/knowledge/search/tag-filters' import type { StructuredFilter } from '@/lib/knowledge/types' import { embeddingDistance } from '@/lib/knowledge/vector-columns' import { CONNECTOR_REGISTRY } from '@/connectors/registry.server' @@ -38,6 +39,13 @@ const distribution = z .parse(process.env.KNOWLEDGE_SCALE_DISTRIBUTION ?? 'periodic-stress') const rows = Number(process.env.KNOWLEDGE_SCALE_DOCUMENTS ?? 250_000) const SEED_BATCH_SIZE = 2_000 + +/** Every leg must finish inside its deadline: a partial answer is not a result to measure. */ +async function completeSearch(params: Parameters[0]) { + const { rows, retrieval } = await retrieveKnowledgeSearch(params) + expect(retrieval.status).toBe('complete') + return rows +} const PAGE_SIZE = 500 const DIMENSIONS = 1536 /** @@ -446,7 +454,7 @@ describe.skipIf(!enabled)('knowledge scale: isolated real PostgreSQL, no provide ) const vector = z.string().parse(queryChunk.vector) const workspaceResults = await measure('search.workspace.denied', () => - executeKnowledgeSearch({ + completeSearch({ knowledgeBaseIds: [ids.knowledgeBaseId], topK: 10, access: { kind: 'workspace', tokens: WORKSPACE_ACCESS_TOKENS }, @@ -478,7 +486,7 @@ describe.skipIf(!enabled)('knowledge scale: isolated real PostgreSQL, no provide const queryLabel = `search.${label}.${filtered ? 'tag' : 'all'}.${mode}` for (let sample = 0; sample < 3; sample++) { const result = await measure(`${queryLabel}.${sample}`, () => - executeKnowledgeSearch({ + completeSearch({ knowledgeBaseIds: [ids.knowledgeBaseId], topK: 10, access, diff --git a/apps/sim/lib/knowledge/__integration__/search-index-policy.integration.ts b/apps/sim/lib/knowledge/__integration__/search-index-policy.integration.ts index 988ef9fe054..184fb5524e9 100644 --- a/apps/sim/lib/knowledge/__integration__/search-index-policy.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-index-policy.integration.ts @@ -25,7 +25,6 @@ import { readKnowledgeBase, restoreKnowledgeBase, } from '@/lib/knowledge/application/knowledge-bases' -import { deleteKnowledgeBaseByVfsPath } from '@/lib/knowledge/application/knowledge-vfs' import { deleteKnowledgeBase, updateKnowledgeBase } from '@/lib/knowledge/service' const ids = createKnowledgeAclFixtureIds() @@ -182,12 +181,6 @@ describe('canonical search knowledge-base policy', () => { ] as Principal[]) { await expect(deleteKnowledgeBaseOperation.execute({ principal, input })).rejects.toThrow() } - await expect( - deleteKnowledgeBaseByVfsPath.execute({ - principal: copilot(ids.bobId), - input: { workspaceId: ids.workspaceId, sourceName: indexName }, - }) - ).rejects.toThrow('Insufficient workspace permissions') await expectIndexActive() }) diff --git a/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts b/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts index de927b794fd..5eb6e667a39 100644 --- a/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts @@ -23,6 +23,14 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' import { NextRequest } from 'next/server' import { afterAll, beforeAll, describe, expect, it, type MockInstance, vi } from 'vitest' + +/** Turns indexed organization search on: its search-index knowledge bases are read through it. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) + import { z } from 'zod' import { workspaceKnowledgeSearchDataSchema } from '@/lib/api/contracts/knowledge/search' import { internalSessionAuth } from '@/lib/api/server/routes' diff --git a/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts b/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts index 5f6a14890b5..a339cd91cbe 100644 --- a/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts @@ -14,6 +14,13 @@ import { generateId } from '@sim/utils/id' import { and, eq, isNull } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' +/** A Search source crawls into a search index, which only indexed organization search reads. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) + const fixture = vi.hoisted(() => ({ dispatch: vi.fn() })) vi.mock('@/lib/credential-groups/provider-registry', () => ({ getCredentialGroupProviderAdapter: (provider: string) => ({ diff --git a/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts b/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts index dc7b1dcecd0..33daeb09e50 100644 --- a/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts @@ -26,6 +26,12 @@ const fixture = vi.hoisted(() => ({ listRuns: vi.fn(), batchTrigger: vi.fn(), })) +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) vi.mock('@/lib/core/config/trigger-runtime', () => ({ isInsideTriggerRun: () => fixture.useTrigger, })) @@ -78,7 +84,6 @@ import { seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { searchKnowledge } from '@/lib/knowledge/application/search' -import { searchScopedKnowledge } from '@/lib/knowledge/application/workspace-search' import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' import { addDocument } from '@/lib/knowledge/connectors/sync-persistence' import { sweepStuckDocuments } from '@/lib/knowledge/connectors/sync-primitives' @@ -99,6 +104,7 @@ import { retryDocumentProcessing, } from '@/lib/knowledge/documents/service' import { MAX_PROCESSING_ATTEMPTS, QUEUED_DISPATCH_GRACE_MS } from '@/lib/knowledge/documents/types' +import { searchScopedKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' import type { SyncResult } from '@/connectors/types' const fixtures: ReturnType[] = [] diff --git a/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts b/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts index bd932b01c7f..0cb20102915 100644 --- a/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts @@ -27,8 +27,14 @@ import { isRecordLike } from '@sim/utils/object' import { eq, inArray, sql } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( + importOriginal + ) +) /** The TINQL `resolveTinKeywordQuery` renders for `fixture`: its `english` stem, quoted. */ -vi.mock('@/lib/knowledge/search/tin-keyword', () => ({ +vi.mock('@/lib/sim-search/indexed/retrieval/tin-keyword', () => ({ resolveTinKeywordQuery: async () => '"fixtur"', })) @@ -36,19 +42,18 @@ import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import type { SearchAccessPlan } from '@/lib/knowledge/access/predicate' import type { GitHubInstallationReadGrant, KnowledgeAccessProvider, UserAccessScope, } from '@/lib/knowledge/access/types' -import { - executeKeywordSearch, - forgetProjectionFilled, - handleVectorOnlySearch, - liveSourceAccessFor, -} from '@/lib/knowledge/search/queries' +import { liveSourceAccessForConnectors } from '@/lib/knowledge/search/candidates' import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' +import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' +import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' +import type { IndexedRetrievalContext } from '@/lib/sim-search/indexed/retrieval/permitted' +import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' +import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' const ids = createKnowledgeAclFixtureIds() const connectorId = generateId() @@ -106,7 +111,15 @@ const aliceGrant: GitHubInstallationReadGrant = { repositoryId, } -function searchInputs(who: 'alice' | 'bob') { +const searchInputs = (who: 'alice' | 'bob') => ({ + knowledgeBaseIds: [ids.knowledgeBaseId], + topK: 5, + access: scopeFor(who), + queryVector, +}) + +/** A narrow reader of the search index, whose live installation grant is resolved on demand. */ +function searchContext(who: 'alice' | 'bob'): IndexedRetrievalContext { const access = scopeFor(who) const accessPlan = planFor(who) const granted = who === 'alice' ? { ...access, githubInstallationGrants: [aliceGrant] } : access @@ -117,24 +130,23 @@ function searchInputs(who: 'alice' | 'bob') { liveSourceConnectorCondition: async () => null, } return { - knowledgeBaseIds: [ids.knowledgeBaseId], - topK: 5, access, - accessProvider, accessPlan, - liveSourceAccess: liveSourceAccessFor(access, accessPlan, accessProvider), - queryVector, + filtered: false, + permitted: { kind: 'unbounded', broad: false }, + liveSourceAccess: liveSourceAccessForConnectors( + accessPlan.connectors.liveProofRequired, + accessProvider + ), } } const keywordIds = async (who: 'alice' | 'bob') => ( - await executeKeywordSearch({ - ...searchInputs(who), - query: 'fixture', - permitted: { kind: 'unbounded', broad: false }, - searchIndexOnly: true, - }) + await executeIndexedKeywordSearch( + { ...searchInputs(who), query: 'fixture' }, + searchContext(who) + ) ).map((row) => row.id) /** @@ -144,11 +156,10 @@ const keywordIds = async (who: 'alice' | 'bob') => */ const vectorIds = async (who: 'alice' | 'bob') => ( - await handleVectorOnlySearch({ - ...searchInputs(who), - distanceThreshold: 2, - permitted: { kind: 'unbounded', broad: false }, - }) + await selectIndexedVectorResults( + { ...searchInputs(who), distanceThreshold: 2 }, + searchContext(who) + ) ).map((row) => row.id) async function setProjection(state: 'filled' | 'unfilled') { diff --git a/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts b/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts new file mode 100644 index 00000000000..c0fc6cfd66b --- /dev/null +++ b/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts @@ -0,0 +1,426 @@ +/** + * Workspace knowledge base search decides readability on the document for every principal. A + * signed-in reader and an actorless run over the same workspace base get the same rows, and + * neither consults the ranking projections' mirrored source and ACL, the projector's marks, or + * the keyword projection — so projection rows that are stale, unfilled, or missing change + * nothing about what a workspace search returns. A signed-in reader's own grants widen what they + * read, and a source whose reader must be proven live admits a candidate only once the proof + * holds, with the readable rows below it filling the page when it does not. A search index named + * by id while indexed organization search is dormant is searched the same way. + */ +import { createHash } from 'node:crypto' +import type { Principal } from '@sim/auth/principal' +import { db } from '@sim/db' +import { + credential, + credentialGroup, + credentialGroupEnrollment, + document, + embedding, + embeddingKeywordSearch, + embeddingSearch, + knowledgeBase, + knowledgeConnector, + knowledgeConnectorMember, + knowledgeDocumentObservation, + organization, + user, + workspace, +} from '@sim/db/schema' +import { generateId } from '@sim/utils/id' +import { eq, inArray } from 'drizzle-orm' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +/** Pinned to Live Search, so indexed organization search is dormant whatever the run's environment. */ +vi.mock('@/lib/core/config/env-flags', async (importOriginal) => ({ + ...(await importOriginal>()), + isLiveEnterpriseSearchEnabled: true, +})) + +import { + createKnowledgeAclFixtureIds, + seedKnowledgeAclFixture, +} from '@/lib/knowledge/__integration__/seed-source-access-fixture' +import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' +import type { + GitHubInstallationReadGrant, + KnowledgeAccessProvider, + UserAccessScope, +} from '@/lib/knowledge/access/types' +import { type KnowledgeSearchMode, retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' +import { embeddingVectorValues } from '@/lib/knowledge/vector-columns' +import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' +import { usesIndexedRetrieval } from '@/lib/sim-search/indexed/gate' + +afterAll(async () => { + await db.$client.end() +}) + +describe.each([ + ['a workspace knowledge base', false], + ['a dormant search index', true], +] as const)('search of %s decides access on the document', (_label, isSearchIndex) => { + const ids = createKnowledgeAclFixtureIds() + const baseId = generateId() + const readable = [generateId(), generateId(), generateId()] + const personal = generateId() + const adminConnectorId = generateId() + const chunkOf = new Map([...readable, personal].map((documentId) => [documentId, generateId()])) + const vector = [0, 0, 1, ...Array(1533).fill(0)] + const queryVector = { + vector: JSON.stringify(vector), + dimensions: 1536 as const, + model: 'text-embedding-3-small', + } + const reader: Principal = { kind: 'session', userId: ids.bobId, sessionId: 'fixture-session' } + const actorless: Principal = { + kind: 'workspace_api_key', + workspaceId: ids.workspaceId, + keyId: 'fixture-key', + } + + beforeAll(async () => { + await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) + await db.insert(knowledgeBase).values({ + id: baseId, + userId: ids.aliceId, + workspaceId: ids.workspaceId, + name: 'Workspace handbook', + isSearchIndex, + }) + await db.insert(knowledgeConnector).values({ + id: adminConnectorId, + knowledgeBaseId: baseId, + connectorType: 'google_drive', + sourceConfig: {}, + accessMode: 'admin', + status: 'active', + credentialId: ids.credentialId, + }) + await db.insert(document).values( + [...readable, personal].map((id) => ({ + id, + knowledgeBaseId: baseId, + filename: `${id}.txt`, + fileUrl: `https://fixture.invalid/${id}`, + fileSize: 12, + mimeType: 'text/plain', + processingStatus: 'completed' as const, + /** + * Uploads carry the workspace token; one source document is shared with a directory group + * only Alice belongs to, its permissions freshly verified. + */ + ...(id === personal + ? { + connectorId: adminConnectorId, + acl: ['g:google-drive:fixture-tenant:page'], + aclVerifiedAt: new Date(), + } + : { acl: ['ws'] }), + })) + ) + await db.insert(embedding).values( + [...readable, personal].map((documentId, index) => ({ + id: chunkOf.get(documentId)!, + documentId, + knowledgeBaseId: baseId, + chunkIndex: 0, + chunkHash: documentId, + content: `Handbook onboarding chapter ${index}`, + contentLength: 30, + tokenCount: 4, + startOffset: 0, + endOffset: 30, + ...embeddingVectorValues(1536, vector), + })) + ) + /** + * The projections behind the removed per-row path, made wrong: every row's mirrored ACL names + * nobody the reader is, its source is gone, and the keyword projection holds nothing. A search + * that consulted any of them would drop the readable rows for the signed-in reader. + */ + await db + .update(embeddingSearch) + .set({ acl: ['s:nobody:-:stale'], connectorId: null }) + .where(eq(embeddingSearch.knowledgeBaseId, baseId)) + await db + .delete(embeddingKeywordSearch) + .where(eq(embeddingKeywordSearch.knowledgeBaseId, baseId)) + }) + + afterAll(async () => { + await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) + await db.delete(organization).where(eq(organization.id, ids.organizationId)) + await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) + }) + + async function search(principal: Principal, searchMode: KnowledgeSearchMode) { + const accessProvider = createKnowledgeAccessProvider(principal, { + workspaceId: ids.workspaceId, + knowledgeBaseIds: [baseId], + }) + const result = await retrieveKnowledgeSearch({ + knowledgeBaseIds: [baseId], + topK: 10, + access: await accessProvider.get(), + accessProvider, + searchMode, + indexedRetrieval: usesIndexedRetrieval([{ isSearchIndex }]), + query: 'handbook onboarding', + queryVector, + }) + expect(result.retrieval).toEqual({ status: 'complete', timedOutLegs: [] }) + return result.rows.map((row) => row.documentId).sort() + } + + it.each(['vector', 'hybrid'] as const)( + 'returns a signed-in reader the rows an actorless run gets (%s)', + async (mode) => { + const previousDebug = db.$client.options.debug + const statements: string[] = [] + db.$client.options.debug = (_connection, query) => { + statements.push(query) + } + try { + const signedIn = await search(reader, mode) + const workspaceKey = await search(actorless, mode) + expect(signedIn).toEqual([...readable].sort()) + expect(workspaceKey).toEqual(signedIn) + /** Nothing read the per-row machinery the search-index path keeps for itself. */ + for (const fragment of [ + 'embedding_keyword_search', + 'knowledge_projection_dirty', + '"embedding_search"."acl"', + '"embedding_search"."connector_id"', + ]) { + expect(statements.filter((statement) => statement.includes(fragment))).toEqual([]) + } + } finally { + db.$client.options.debug = previousDebug + } + } + ) + + it.each(['vector', 'hybrid'] as const)( + 'returns a signed-in reader the document shared with them alone as well (%s)', + async (mode) => { + const owner: Principal = { kind: 'session', userId: ids.aliceId, sessionId: 'fixture-owner' } + expect(await search(owner, mode)).toEqual([...readable, personal].sort()) + } + ) +}) + +describe('a workspace source whose reader must be proven live', () => { + const ids = createKnowledgeAclFixtureIds() + const baseId = generateId() + const connectorId = generateId() + const contentCredentialId = generateId() + const readerCredentialId = generateId() + const groupId = generateId() + const optionId = generateId() + const repositoryId = '4343' + const readerSubject = 'reader-gh' + const readerToken = `s:github-repositories:-:${readerSubject}` + const gated = generateId() + const readable = [generateId(), generateId()] + const vector = [0, 0, 0, 1, ...Array(1532).fill(0)] + const queryVector = { + vector: JSON.stringify(vector), + dimensions: 1536 as const, + model: 'text-embedding-3-small', + } + /** The installation document is the nearest; the readable ones trail it. */ + const vectorOf = (documentId: string, index: number) => + documentId === gated ? vector : [0, 0, 0, 1, 0.1 * (index + 1), ...Array(1531).fill(0)] + + beforeAll(async () => { + await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) + const now = new Date() + await db.insert(knowledgeBase).values({ + id: baseId, + userId: ids.aliceId, + workspaceId: ids.workspaceId, + name: 'Workspace repositories', + }) + await db.insert(credential).values({ + id: contentCredentialId, + workspaceId: ids.workspaceId, + type: 'service_account', + displayName: 'Fixture GitHub installation', + createdBy: ids.aliceId, + providerId: GITHUB_INSTALLATION_PROVIDER_ID, + }) + await db.insert(credentialGroup).values({ + id: groupId, + workspaceId: ids.workspaceId, + publicId: generateId(), + name: 'GitHub readers', + options: [ + { + id: optionId, + provider: 'github-repositories', + label: 'GitHub fixture', + authorizationAppId: 'fixture-app', + requiredScopes: ['repo'], + scopeVersion: 1, + required: false, + status: 'active', + }, + ], + } as typeof credentialGroup.$inferInsert) + const [enrollment] = await db + .insert(credentialGroupEnrollment) + .values({ + id: generateId(), + credentialGroupId: groupId, + userId: ids.bobId, + email: `${ids.bobId}@fixture.test`, + status: 'completed', + invitationTokenHash: createHash('sha256').update(generateId()).digest('hex'), + invitationExpiresAt: new Date(Date.now() + 60 * 60 * 1000), + invitedAt: now, + }) + .returning({ id: credentialGroupEnrollment.id }) + await db.insert(credential).values({ + id: readerCredentialId, + workspaceId: ids.workspaceId, + type: 'managed_oauth', + displayName: 'Fixture GitHub reader', + providerId: 'github-repositories', + authorizationAppId: 'fixture-app', + credentialGroupEnrollmentId: enrollment!.id, + credentialGroupOptionId: optionId, + managedOauthScopeVersion: 1, + providerSubjectId: readerSubject, + providerTenantId: '', + managedOauthStatus: 'active', + grantedScopes: ['repo'], + encryptedOauthTokenSet: 'fixture-not-an-oauth-token', + grantedAt: now, + createdBy: ids.bobId, + }) + await db.insert(knowledgeConnector).values({ + id: connectorId, + knowledgeBaseId: baseId, + connectorType: 'github', + sourceConfig: { githubRepositoryId: repositoryId }, + accessMode: 'members', + status: 'active', + credentialId: contentCredentialId, + credentialGroupId: groupId, + credentialGroupOptionId: optionId, + }) + const memberId = generateId() + await db.insert(knowledgeConnectorMember).values({ + id: memberId, + workspaceId: ids.workspaceId, + connectorId, + credentialId: readerCredentialId, + subjectToken: readerToken, + status: 'active', + memberSyncedThrough: now, + }) + await db.insert(document).values([ + { + id: gated, + connectorId, + knowledgeBaseId: baseId, + externalId: 'fixture-readme', + filename: 'readme.md', + fileUrl: 'https://fixture.invalid/readme', + fileSize: 12, + mimeType: 'text/plain', + processingStatus: 'completed' as const, + acl: [readerToken], + }, + ...readable.map((id) => ({ + id, + knowledgeBaseId: baseId, + filename: `${id}.txt`, + fileUrl: `https://fixture.invalid/${id}`, + fileSize: 12, + mimeType: 'text/plain', + processingStatus: 'completed' as const, + acl: ['ws'], + })), + ]) + await db + .insert(knowledgeDocumentObservation) + .values({ documentId: gated, memberId, lastSeenAt: now, runId: generateId() }) + await db.insert(embedding).values( + [gated, ...readable].map((documentId, index) => ({ + id: generateId(), + documentId, + knowledgeBaseId: baseId, + chunkIndex: 0, + chunkHash: documentId, + content: `Repository release notes ${index}`, + contentLength: 30, + tokenCount: 4, + startOffset: 0, + endOffset: 30, + ...embeddingVectorValues(1536, vectorOf(documentId, index)), + })) + ) + }) + + afterAll(async () => { + await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) + await db.delete(credentialGroup).where(eq(credentialGroup.id, groupId)) + await db.delete(organization).where(eq(organization.id, ids.organizationId)) + await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) + }) + + /** + * A reader whose stored tokens match the installation document's mirrored permissions, and who + * either proves the installation grant live or does not. + */ + async function search(proven: boolean, searchMode: KnowledgeSearchMode) { + const reader: UserAccessScope = { + kind: 'user', + userId: ids.bobId, + tokens: ['pub', readerToken, `u:${ids.bobId}@fixture.test`, 'ws'].sort(), + } + const grant: GitHubInstallationReadGrant = { + connectorId, + contentCredentialId, + readerCredentialId, + readerSubjectToken: readerToken, + repositoryId, + } + const authorized = proven ? { ...reader, githubInstallationGrants: [grant] } : reader + const accessProvider: KnowledgeAccessProvider = { + get: async () => reader, + getForConnectors: async () => authorized, + getForDocuments: async () => authorized, + liveSourceConnectorCondition: async () => eq(knowledgeConnector.id, connectorId), + } + const result = await retrieveKnowledgeSearch({ + knowledgeBaseIds: [baseId], + topK: 2, + access: reader, + accessProvider, + searchMode, + query: 'repository release notes', + queryVector, + }) + expect(result.retrieval).toEqual({ status: 'complete', timedOutLegs: [] }) + return result.rows.map((row) => row.documentId) + } + + it.each(['vector', 'hybrid'] as const)( + 'drops the candidate at hydration without the grant, and fills the page with readable rows (%s)', + async (mode) => { + expect((await search(false, mode)).sort()).toEqual([...readable].sort()) + } + ) + + it.each(['vector', 'hybrid'] as const)( + 'returns the candidate to a reader who proves the grant (%s)', + async (mode) => { + const rows = await search(true, mode) + expect(rows).toHaveLength(2) + expect(rows).toContain(gated) + } + ) +}) diff --git a/apps/sim/lib/knowledge/access/predicate.integration.ts b/apps/sim/lib/knowledge/access/predicate.integration.ts index 3c9c4d3596c..c42d9108c8b 100644 --- a/apps/sim/lib/knowledge/access/predicate.integration.ts +++ b/apps/sim/lib/knowledge/access/predicate.integration.ts @@ -23,13 +23,12 @@ const { mergeMirroredAcls, hideUnlistedDocuments } = await import( '@/lib/knowledge/connectors/mirrored-acls' ) const { PgDialect } = await import('drizzle-orm/pg-core') -const { - knowledgeAccessCondition, - knowledgeCandidateAccessConditionForConnectors, - projectionCandidateAccessCondition, - restrictSearchAccessPlan, - knowledgeMetadataCandidateAccessCondition, -} = await import('@/lib/knowledge/access/predicate') +const { knowledgeAccessCondition, knowledgeMetadataCandidateAccessCondition } = await import( + '@/lib/knowledge/access/predicate' +) +const { restrictSearchAccessPlan } = await import('@/lib/sim-search/indexed/retrieval/access-plan') +const { knowledgeCandidateAccessConditionForConnectors, projectionCandidateAccessCondition } = + await import('@/lib/sim-search/indexed/retrieval/projection-access') const { confluencePageAcl } = await import('@/lib/knowledge/access/confluence-permissions') /** Every table and index belongs to an isolated disposable schema. */ diff --git a/apps/sim/lib/knowledge/access/predicate.test.ts b/apps/sim/lib/knowledge/access/predicate.test.ts index 411cd31f0e4..06b4e248481 100644 --- a/apps/sim/lib/knowledge/access/predicate.test.ts +++ b/apps/sim/lib/knowledge/access/predicate.test.ts @@ -16,12 +16,10 @@ const { PgDialect } = await import('drizzle-orm/pg-core') const { embeddingSearch } = await import('@sim/db/schema') const { sql: rawSql } = await import('drizzle-orm') const sqlColumn = (name: string) => rawSql.raw(`"row"."${name}"`) -const { - knowledgeAccessCondition, - knowledgeCandidateAccessConditionForConnectors, - projectionCandidateAccessCondition, - restrictSearchAccessPlan, -} = await import('@/lib/knowledge/access/predicate') +const { knowledgeAccessCondition } = await import('@/lib/knowledge/access/predicate') +const { restrictSearchAccessPlan } = await import('@/lib/sim-search/indexed/retrieval/access-plan') +const { knowledgeCandidateAccessConditionForConnectors, projectionCandidateAccessCondition } = + await import('@/lib/sim-search/indexed/retrieval/projection-access') const { SYSTEM_ACCESS_SCOPE } = await import('@/lib/knowledge/access/types') function render(condition: ReturnType) { diff --git a/apps/sim/lib/knowledge/access/predicate.ts b/apps/sim/lib/knowledge/access/predicate.ts index 2517af6f483..a316783db85 100644 --- a/apps/sim/lib/knowledge/access/predicate.ts +++ b/apps/sim/lib/knowledge/access/predicate.ts @@ -7,12 +7,10 @@ import { knowledgeConnector, knowledgeConnectorMember, knowledgeDocumentObservation, - knowledgeProjectionDirty, member, user, } from '@sim/db/schema' import { type SQL, sql } from 'drizzle-orm' -import type { AnyPgColumn } from 'drizzle-orm/pg-core' import { EXTERNAL_GROUP_STALE_AFTER_MS } from '@/lib/knowledge/access/external-groups' import { SOURCE_ACL_MAX_AGE_MS } from '@/lib/knowledge/access/freshness' import { confluenceReaderGroupCondition } from '@/lib/knowledge/access/group-membership' @@ -194,7 +192,7 @@ export function knowledgeMetadataCandidateAccessCondition( } /** Every requirement clause must reach the caller, which preserves source permission intersections. */ -function aclRequirementsSatisfied(tokens: SQL): SQL { +export function aclRequirementsSatisfied(tokens: SQL): SQL { return sql`NOT EXISTS ( SELECT 1 FROM jsonb_array_elements(${document.aclRequirements}) AS required_clause(tokens) WHERE NOT (required_clause.tokens ?| ${tokens}) @@ -202,228 +200,10 @@ function aclRequirementsSatisfied(tokens: SQL): SQL { } /** The live source proofs this request carries, as the connector-scoped clause both shapes apply. */ -export function liveSourceAccessCondition(scope: KnowledgeAccessScope): SQL { +function liveSourceAccessCondition(scope: KnowledgeAccessScope): SQL { return sql`(${githubInstallationAccessCondition(scope)} AND ${confluenceSiteAccessCondition(scope)})` } -/** - * The connectors a search may read from, resolved once per query: their ids grouped by the shape - * their documents' ACLs take, and separately those whose reader access is proven live per request. - */ -/** - * The caller's active member identities on the connectors a search reads, by what makes their - * observations current: `confirmed` members drained their change feed inside the freshness window, - * so every observation they hold stands; `observed` members are trusted only where the observation - * itself is recent. - */ -/** One of the caller's member identities and the connector it belongs to. */ -export interface KnowledgeMemberObserver { - id: string - connectorId: string -} - -export interface KnowledgeMemberObservers { - confirmed: readonly KnowledgeMemberObserver[] - observed: readonly KnowledgeMemberObserver[] -} - -/** What a search resolves once about its sources and the caller's standing in them. */ -export interface SearchAccessPlan { - connectors: KnowledgeConnectorEligibility - observers: KnowledgeMemberObservers - /** Connectors the caller is an active member of, whose documents they read broadly. */ - memberSources: readonly string[] - /** Each eligible connector's type, so a search may be confined to one kind of source. */ - connectorTypes: ReadonlyMap - /** Whether documents without a source — uploads — are in scope. */ - uploads: boolean -} - -/** - * The plan confined to one kind of source: the connectors of that type keep their eligibility and - * the rest lose it, so every predicate built from the plan — on the row and on the document — and - * every source the legs walk or rank are that kind alone. `upload` keeps only source-less documents. - */ -export function restrictSearchAccessPlan(plan: SearchAccessPlan, source: string): SearchAccessPlan { - const keep = (id: string) => source !== 'upload' && plan.connectorTypes.get(id) === source - const kept = (ids: readonly string[]) => ids.filter(keep) - return { - connectors: { - workspace: kept(plan.connectors.workspace), - admin: kept(plan.connectors.admin), - members: kept(plan.connectors.members), - liveProofRequired: kept(plan.connectors.liveProofRequired), - }, - observers: { - confirmed: plan.observers.confirmed.filter((observer) => keep(observer.connectorId)), - observed: plan.observers.observed.filter((observer) => keep(observer.connectorId)), - }, - memberSources: kept(plan.memberSources), - connectorTypes: plan.connectorTypes, - uploads: source === 'upload', - } -} - -export interface KnowledgeConnectorEligibility { - /** Documents carry the workspace ACL. */ - workspace: readonly string[] - /** Documents carry mirrored source permissions verified as a whole. */ - admin: readonly string[] - /** Documents carry the subject tokens of the members who observe them. */ - members: readonly string[] - /** Of the above, those that additionally require this request's live source proof. */ - liveProofRequired: readonly string[] -} - -/** - * The candidate predicate with connector state resolved ahead of the query instead of per row. - * - * Deletion, archival, a pending access rewrite, the organization's integration approval and the - * access mode are facts about a connector, not a document, so checking them once per query leaves - * each candidate an id comparison plus its own columns. - * - * `liveSourceAccess` is the caller's live source proof, and defaults to admitting everything: - * candidate ranking defers that proof until after ranking, exactly as - * {@link knowledgeMetadataCandidateAccessCondition} does, and only a reader that already holds the - * grants — content hydration — passes it. The connectors it would gate are listed separately so - * that clause is applied to those alone. - * - * Either way it narrows exactly as the predicate it stands in for: the eligible ids are the - * connectors that predicate's `EXISTS` would admit, and every document-level clause is carried - * over unchanged. - */ -export function knowledgeCandidateAccessConditionForConnectors( - scope: KnowledgeAccessScope | SystemAccessScope, - plan: SearchAccessPlan, - liveSourceAccess: SQL = sql`true` -): SQL { - const eligibility = plan.connectors - if (scope.kind === 'system') return documentConnectorIsActive() - if (scope.tokens.length === 0) return sql`false` - const tokens = textArrayLiteral(scope.tokens) - const cutoff = sql`statement_timestamp() - (${SOURCE_ACL_MAX_AGE_MS} * interval '1 millisecond')` - const liveProof = new Set(eligibility.liveProofRequired) - const inConnectors = (ids: readonly string[]): SQL => - ids.length === 0 - ? sql`false` - : sql`${document.connectorId} = ANY(${textArrayLiteral([...ids])})` - const mirrored = (ids: readonly string[], current: SQL): SQL => { - const direct = ids.filter((id) => !liveProof.has(id)) - const gated = ids.filter((id) => liveProof.has(id)) - const currentAndMirrored = sql`${document.acl} <> ARRAY['ws']::text[] AND ${current}` - return sql`( - (${inConnectors(direct)} AND ${currentAndMirrored}) - OR (${inConnectors(gated)} AND ${currentAndMirrored} AND EXISTS ( - SELECT 1 FROM ${knowledgeConnector} - WHERE ${knowledgeConnector.id} = ${document.connectorId} - AND ${liveSourceAccess} - )) - )` - } - const workspaceOwned = plan.uploads - ? sql`(${document.connectorId} IS NULL OR ${inConnectors(eligibility.workspace)})` - : inConnectors(eligibility.workspace) - return sql`( - ${aclOverlap(tokens)} - AND ${aclRequirementsSatisfied(tokens)} - AND ( - (${workspaceOwned} AND ${document.acl} = ARRAY['ws']::text[]) - OR ${mirrored(eligibility.admin, sql`${document.aclVerifiedAt} > ${cutoff}`)} - OR ${mirrored(eligibility.members, resolvedObservationCondition(plan.observers, cutoff))} - ) - )` -} - -/** - * Whether a projection row belongs to a document marked for the knowledge projector: its source, - * ACL, or chunks changed and its rows may not show it yet. A probe of the marks' primary key: the - * planner may instead hash the whole set once per statement, which is as cheap while the marks are - * few, and an `IN` would risk re-reading them per row once they outgrow the hash. - */ -export function projectionPending(documentId: AnyPgColumn | SQL): SQL { - return sql`(EXISTS (SELECT 1 FROM ${knowledgeProjectionDirty} WHERE ${knowledgeProjectionDirty.documentId} = ${documentId}))` -} - -/** - * Whether a projection row is decided on its document rather than on its own columns: its - * document is marked for the projector, or, while the source and ACL fill runs, the row has not - * been filled. - */ -export function projectionDecidedOnDocument( - projection: { acl: AnyPgColumn | SQL; documentId: AnyPgColumn | SQL }, - filled: boolean -): SQL { - const pending = projectionPending(projection.documentId) - return filled ? pending : sql`(${projection.acl} IS NULL OR ${pending})` -} - -/** - * The candidate predicate on a ranking projection's own row, for a scope whose connectors were - * resolved: `connectorId` and `acl` are mirrored there from the document, so a walk or a keyword - * window decides readability on the row it scores instead of joining `document` per candidate. - * - * It admits a superset of the document predicate, never a subset: a mirrored ACL names the members - * who observe a document, so overlap with the caller's tokens is the per-row test without the - * observation's freshness, and requirement clauses live on the document. Both are refused there, - * under the full predicate, before content is returned — this predicate only decides what is worth - * ranking. - * - * A row whose columns may be behind its document is decided on the document instead, under - * {@link knowledgeCandidateAccessConditionForConnectors} — the join per candidate that every row - * paid before the columns existed: a row the source and ACL fill has not reached (`acl IS NULL`), - * and every row of a document marked for the knowledge projector. A revoked grant still on such a - * row never admits it, and a new grant not yet on it never hides it from a statement that reaches - * the row. A source-scoped walk or slice reaches rows by the source on the row, though, so a - * document that moved to another source joins that source's ranking once the projector has - * rewritten its rows; until then it can be missing there, never shown where it is not readable. - * The projector and the fill run in the background, so search never waits on either. - */ -export function projectionCandidateAccessCondition( - projection: { - connectorId: AnyPgColumn | SQL - acl: AnyPgColumn | SQL - documentId: AnyPgColumn | SQL - }, - scope: KnowledgeAccessScope | SystemAccessScope, - plan: SearchAccessPlan, - options: { - /** - * Whether every row of the projection carries its mirrored source and ACL. While the fill - * is under way, a row it has not reached is decided on its document; once it is complete only - * a marked document's rows are. - */ - filled?: boolean - } = {} -): SQL { - if (scope.kind === 'system') return sql`true` - if (scope.tokens.length === 0) return sql`false` - const tokens = textArrayLiteral(scope.tokens) - const inSources = (ids: readonly string[]): SQL => - ids.length === 0 - ? sql`false` - : sql`${projection.connectorId} = ANY(${textArrayLiteral([...ids])})` - const mirrored = [ - ...plan.connectors.workspace, - ...plan.connectors.admin, - ...plan.connectors.members, - ] - const owned = plan.uploads - ? sql`(${projection.connectorId} IS NULL OR ${inSources(mirrored)})` - : inSources(mirrored) - const onRow = sql`(${projection.acl} && ${tokens} AND ${owned})` - /** - * A scalar subquery rather than `EXISTS`: the planner may turn an `EXISTS` into one hash of every - * readable document, a sequential scan of `document` for a statement that only needs a few rows - * decided. A scalar subquery is only ever a primary-key probe per row that needs it. - */ - const onDocument = sql`(SELECT ${document.id} FROM ${document} - WHERE ${document.id} = ${projection.documentId} - AND ${knowledgeCandidateAccessConditionForConnectors(scope, plan)} - LIMIT 1) IS NOT NULL` - return sql`((${projectionDecidedOnDocument(projection, options.filled ?? false)} AND ${onDocument}) - OR (${onRow} AND NOT ${projectionPending(projection.documentId)}))` -} - /** * A members-mode document is readable while one of the caller's active member identities on its * connector still observes it, freshly. Correlated on the document so each check is a lookup on @@ -431,36 +211,6 @@ export function projectionCandidateAccessCondition( * inside the access predicate's `OR`, PostgreSQL instead hashes every observation in the table * once per statement, a fixed cost paid by every query that carries the predicate. */ -/** - * The same membership, resolved ahead of the query: each candidate costs one lookup on the - * observation key instead of a join to the member behind it. Equivalent by construction — the ids - * are the members that join would have matched, and each one's freshness rule is carried over. - */ -function resolvedObservationCondition(observers: KnowledgeMemberObservers, cutoff: SQL): SQL { - if (observers.confirmed.length === 0 && observers.observed.length === 0) return sql`false` - /** - * An observation vouches for a document only from a member of the document's own connector: a - * document that changed hands keeps its old observations, which must not carry it. - */ - const byMember = (members: readonly KnowledgeMemberObserver[]): SQL => - sql`(${knowledgeDocumentObservation.memberId}, ${document.connectorId}) IN (${sql.join( - members.map((member) => sql`(${member.id}, ${member.connectorId})`), - sql`, ` - )})` - const current = - observers.confirmed.length === 0 - ? sql`${byMember(observers.observed)} AND ${knowledgeDocumentObservation.lastSeenAt} > ${cutoff}` - : observers.observed.length === 0 - ? byMember(observers.confirmed) - : sql`(${byMember(observers.confirmed)} - OR (${byMember(observers.observed)} AND ${knowledgeDocumentObservation.lastSeenAt} > ${cutoff}))` - return sql`EXISTS ( - SELECT 1 FROM ${knowledgeDocumentObservation} - WHERE ${knowledgeDocumentObservation.documentId} = ${document.id} - AND ${current} - )` -} - function memberObservationCondition(tokens: SQL, cutoff: SQL): SQL { return sql`EXISTS ( SELECT 1 FROM ${knowledgeDocumentObservation} @@ -474,6 +224,21 @@ function memberObservationCondition(tokens: SQL, cutoff: SQL): SQL { )` } +/** Source-derived grants count only while their evidence is younger than the freshness window. */ +export function sourceAclFreshnessCutoff(): SQL { + return sql`statement_timestamp() - (${SOURCE_ACL_MAX_AGE_MS} * interval '1 millisecond')` +} + +/** A document readable by the whole workspace: the ACL is exactly the workspace token. */ +export function documentHasWorkspaceAcl(): SQL { + return sql`${document.acl} = ARRAY['ws']::text[]` +} + +/** A document carrying mirrored source permissions rather than the workspace token. */ +export function documentHasMirroredAcl(): SQL { + return sql`${document.acl} <> ARRAY['ws']::text[]` +} + function storedKnowledgeAccessCondition( scope: KnowledgeAccessScope | SystemAccessScope, liveSourceAccess: SQL @@ -481,12 +246,12 @@ function storedKnowledgeAccessCondition( if (scope.kind === 'system') return documentConnectorIsActive() if (scope.tokens.length === 0) return sql`false` const tokens = textArrayLiteral(scope.tokens) - const cutoff = sql`statement_timestamp() - (${SOURCE_ACL_MAX_AGE_MS} * interval '1 millisecond')` + const cutoff = sourceAclFreshnessCutoff() return sql`( ${aclOverlap(tokens)} AND ${aclRequirementsSatisfied(tokens)} AND ( - (${document.connectorId} IS NULL AND ${document.acl} = ARRAY['ws']::text[]) + (${document.connectorId} IS NULL AND ${documentHasWorkspaceAcl()}) OR EXISTS ( SELECT 1 FROM ${knowledgeConnector} WHERE ${knowledgeConnector.id} = ${document.connectorId} @@ -496,8 +261,8 @@ function storedKnowledgeAccessCondition( AND ${searchIntegrationAccessCondition()} AND ${liveSourceAccess} AND ( - (${knowledgeConnector.accessMode} = 'workspace' AND ${document.acl} = ARRAY['ws']::text[]) - OR (${document.acl} <> ARRAY['ws']::text[] AND ( + (${knowledgeConnector.accessMode} = 'workspace' AND ${documentHasWorkspaceAcl()}) + OR (${documentHasMirroredAcl()} AND ( (${knowledgeConnector.accessMode} = 'admin' AND ${document.aclVerifiedAt} > ${cutoff}) OR (${knowledgeConnector.accessMode} = 'members' AND ${memberObservationCondition(tokens, cutoff)}) )) @@ -508,21 +273,10 @@ function storedKnowledgeAccessCondition( } /** - * The token half of the stored access predicate: the documents a caller's tokens reach before - * any source, freshness, or requirement check narrows them. It is a necessary condition of - * {@link knowledgeAccessCondition}, never a substitute for it. - * - * Paired with `deleted_at IS NULL` it matches `doc_acl_gin_idx` exactly, so a query can enumerate - * a member's reachable documents from that index alone. PostgreSQL cannot estimate array-overlap - * selectivity, so left to itself it intersects this highly selective bitmap with base-wide ones. + * The token half of the stored access predicate: one spelling of the overlap, so the indexed + * search probe's reach and the full predicate cannot drift. */ -export function knowledgeAclOverlapCondition(scope: KnowledgeAccessScope): SQL { - if (scope.tokens.length === 0) return sql`false` - return aclOverlap(textArrayLiteral(scope.tokens)) -} - -/** One spelling of the token overlap, so the probe's reach and the full predicate cannot drift. */ -function aclOverlap(tokens: SQL): SQL { +export function aclOverlap(tokens: SQL): SQL { return sql`${document.acl} && ${tokens}` } diff --git a/apps/sim/lib/knowledge/access/scope.ts b/apps/sim/lib/knowledge/access/scope.ts index b5120b36458..9b07e36bacc 100644 --- a/apps/sim/lib/knowledge/access/scope.ts +++ b/apps/sim/lib/knowledge/access/scope.ts @@ -374,19 +374,6 @@ async function resolveUserKnowledgeIdentity(userId: string, context: KnowledgeAc } } -/** - * The scope of a person identified only by user id — the shape session-backed - * routes outside the application layer have in hand. Never call this with a - * user id that stands in for an actorless run (a workflow owner, a billing - * owner); those callers use {@link WORKSPACE_ACCESS_SCOPE}. - */ -export async function resolveUserKnowledgeAccessScope( - userId: string, - workspaceId: string | undefined -): Promise { - return { kind: 'user', userId, tokens: (await loadUserAccess(userId, { workspaceId })).tokens } -} - /** Memoises {@link resolveKnowledgeAccessScope} for one operation; a failed lookup is retried on the next call. */ export function createKnowledgeAccessProvider( principal: Principal, diff --git a/apps/sim/lib/knowledge/api/route-policies.ts b/apps/sim/lib/knowledge/api/route-policies.ts index 5f046e65909..30f841a41a5 100644 --- a/apps/sim/lib/knowledge/api/route-policies.ts +++ b/apps/sim/lib/knowledge/api/route-policies.ts @@ -16,6 +16,7 @@ import { KnowledgeUsageLimitExceededError } from '@/lib/knowledge/application/bi import { KnowledgeDocumentNotReadyError } from '@/lib/knowledge/application/chunk-errors' import { KnowledgeSearchProvenanceUnavailableError } from '@/lib/knowledge/application/search' import { KnowledgeDocumentUnsupportedMediaTypeError } from '@/lib/knowledge/application/upload-sessions' +import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' import { v2Error } from '@/app/api/v2/lib/response' function internalKnowledgeErrorPolicy(unhandledMessage: string): InternalErrorPolicy { @@ -58,6 +59,18 @@ export const internalKnowledgeSessionOrExecutorAuth = createInternalSessionOrExe export const KNOWLEDGE_BASE_NOT_FOUND_MESSAGE = 'Knowledge base not found' +/** + * Answers an indexed-only surface refused while indexed organization search is dormant with a + * `409`: the request is well formed and authorized, and the deployment's state is what refuses it. + */ +function refuseDormantSearchIndex(base: InternalErrorPolicy): InternalErrorPolicy { + return extendInternalErrorPolicy(base, (error) => + error instanceof SearchIndexDormantError + ? internalErrorResponse(409, { error: error.message }) + : null + ) +} + /** * Conceals a knowledge-base-scoped internal policy the way the v2 knowledge * routes conceal theirs. The workspace-level `list` and `create` policies are @@ -111,8 +124,12 @@ export const internalKnowledgeErrorPolicies = { tags: concealKnowledgeBase( internalKnowledgeErrorPolicy('Failed to process knowledge tag request') ), - connectors: concealKnowledgeBase(internalKnowledgeErrorPolicy('Internal server error')), - connectAccount: concealKnowledgeBase(internalPersonalCredentialConnectionErrorPolicy), + connectors: concealKnowledgeBase( + refuseDormantSearchIndex(internalKnowledgeErrorPolicy('Internal server error')) + ), + connectAccount: concealKnowledgeBase( + refuseDormantSearchIndex(internalPersonalCredentialConnectionErrorPolicy) + ), uploads: concealKnowledgeBase(internalKnowledgeUploadErrorPolicy), } as const diff --git a/apps/sim/lib/knowledge/application/batch-policy.ts b/apps/sim/lib/knowledge/application/batch-policy.ts index 03fb857c2d2..01a5f30c2e9 100644 --- a/apps/sim/lib/knowledge/application/batch-policy.ts +++ b/apps/sim/lib/knowledge/application/batch-policy.ts @@ -27,11 +27,6 @@ export const BULK_DELETE_KNOWLEDGE_BASES_COST_POLICY = { execution: 'sequential_best_effort', } as const -export const BULK_DELETE_KNOWLEDGE_DOCUMENTS_COST_POLICY = { - maxItems: MAX_KNOWLEDGE_BATCH_ITEMS, - execution: 'sequential_best_effort', -} as const - /** Domain names for the shared batch shapes, so call sites read in knowledge terms. */ export type KnowledgeBatchTerminalFailure = BatchTerminalFailure export type KnowledgeBatchExecutionResult = BatchExecutionResult diff --git a/apps/sim/lib/knowledge/application/connector-access.test.ts b/apps/sim/lib/knowledge/application/connector-access.test.ts index 99514850f97..11f031f2a64 100644 --- a/apps/sim/lib/knowledge/application/connector-access.test.ts +++ b/apps/sim/lib/knowledge/application/connector-access.test.ts @@ -68,7 +68,6 @@ vi.mock('@/lib/knowledge/connectors/mirrored-access', () => ({ assertConnectorMirrorsSourceAcls: hoisted.mirror, })) vi.mock('@/lib/knowledge/application/connectors', () => ({ - requireConnectorWorkspaceId: (context: { workspaceId: string }) => context.workspaceId, requireSuccessfulOutcome: vi.fn(), resolveConnectorCredentialAccessToken: hoisted.token, validateConnectorSourceConfig: hoisted.validate, diff --git a/apps/sim/lib/knowledge/application/connectors.test.ts b/apps/sim/lib/knowledge/application/connectors.test.ts index 2debf77463f..0763cf0c38f 100644 --- a/apps/sim/lib/knowledge/application/connectors.test.ts +++ b/apps/sim/lib/knowledge/application/connectors.test.ts @@ -1,4 +1,4 @@ -import { document, knowledgeConnector, member } from '@sim/db/schema' +import { member } from '@sim/db/schema' import { dbChainMockFns, queueTableRows, resetDbChainMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' import { auditMock, auditMockFns } from '@sim/testing/mocks/audit.mock' @@ -40,8 +40,6 @@ const hoisted = vi.hoisted(() => ({ provision: vi.fn(), decryptApiKey: vi.fn(), viewerMemberships: vi.fn(), - getAccess: vi.fn(), - getForConnectors: vi.fn(), })) vi.mock('@sim/audit', () => auditMock) @@ -125,7 +123,6 @@ import { createApprovedSearchSource, createKnowledgeConnector, deleteKnowledgeConnector, - listWorkspaceMemberConnectors, resolveConnectorCredentialAccessToken, syncKnowledgeConnector, updateKnowledgeConnector, @@ -153,8 +150,6 @@ const mocks = { mocks.canUseCredential.mockReturnValue(undefined) authOAuthUtilsMockFns.mockResolveOAuthAccountId.mockResolvedValue(null) knowledgeAccessScopeMockFns.mockCreateKnowledgeAccessProvider.mockImplementation(() => ({ - get: mocks.getAccess, - getForConnectors: mocks.getForConnectors, liveSourceConnectorCondition: async () => ({ type: 'live-sources' }), })) @@ -322,55 +317,6 @@ describe('knowledge connector application use cases', () => { afterAll(resetDbChainMock) - it('counts workspace central Confluence documents only after candidate site admission', async () => { - const identity = { - kind: 'user' as const, - userId: 'reader', - tokens: ['ws', 's:confluence:-:alice'], - } - mocks.getAccess.mockResolvedValue(identity) - mocks.getForConnectors.mockResolvedValue({ - ...identity, - confluenceSiteGrants: [ - { - connectorId: 'cf-source', - contentCredentialId: 'crawler', - readerCredentialId: 'personal', - readerSubjectToken: 's:confluence:-:alice', - domain: 'company.atlassian.net', - cloudId: 'cloud-1', - }, - ], - }) - mocks.viewerMemberships.mockResolvedValue(new Map([['cf-source', 'connected']])) - queueTableRows(knowledgeConnector, [ - { - id: 'cf-source', - knowledgeBaseId: 'knowledge-b', - knowledgeBaseName: 'Search', - knowledgeBaseIsSearchIndex: true, - connectorType: 'confluence', - accessMode: 'admin', - sourceConfig: { domain: 'company.atlassian.net', spaceKey: ['DEMO'] }, - memberSyncStatus: 'idle', - }, - ]) - queueTableRows(document, []) - queueTableRows(knowledgeConnector, [{ connectorId: 'cf-source' }]) - queueTableRows(document, [{ connectorId: 'cf-source', count: 2 }]) - const result = await listWorkspaceMemberConnectors.execute({ - principal: createSessionPrincipal({ userId: 'reader', sessionId: 'test' }), - input: { workspaceId: 'workspace-b' }, - }) - expect(mocks.getForConnectors).toHaveBeenCalledWith(['cf-source'], undefined) - expect(result.connectors).toEqual([ - expect.objectContaining({ connectorId: 'cf-source', viewerDocumentCount: 2 }), - ]) - expect(JSON.stringify(dbChainMockFns.where.mock.calls.at(-1)?.[0])).toContain( - 'confluence_read_grant' - ) - }) - it('rejects a forged OAuth credential for central Drive creation before using its token', async () => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('admin') knowledgeContextsMockFns.mockResolveActiveKnowledgeBaseContext.mockResolvedValue({ diff --git a/apps/sim/lib/knowledge/application/connectors.ts b/apps/sim/lib/knowledge/application/connectors.ts index f479c8e2378..c9ef81eb794 100644 --- a/apps/sim/lib/knowledge/application/connectors.ts +++ b/apps/sim/lib/knowledge/application/connectors.ts @@ -3,23 +3,19 @@ import type { Principal } from '@sim/auth/principal' import { resolvePrincipalSubjectUserId } from '@sim/auth/principal' import { db } from '@sim/db' import { - credentialGroup, document, - knowledgeBase, knowledgeConnector, knowledgeConnectorMember, knowledgeConnectorMemberSyncLog, knowledgeConnectorSyncLog, } from '@sim/db/schema' import { toError } from '@sim/utils/errors' -import { truncate } from '@sim/utils/string' import { and, asc, desc, eq, inArray, isNull, lt, or, sql } from 'drizzle-orm' import type { ConnectorDocumentFilter } from '@/lib/api/contracts/knowledge/connectors' import type { BillingAttributionSnapshot } from '@/lib/billing/core/billing-attribution' import { requireCurrentHumanRole } from '@/lib/core/application' import { resolvePrincipalEnvironmentVariable } from '@/lib/core/application/environment-reference' import { requireOrganizationMembership } from '@/lib/core/application/organization-authorization' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { OrchestrationError, type OrchestrationRequestContext, @@ -36,7 +32,6 @@ import { parseExactEnvironmentReference } from '@/lib/environment/reference' import { resolveEffectiveEnvironmentVariables } from '@/lib/environment/utils' import { requireKnowledgeMemberAccessAvailable } from '@/lib/knowledge/access/availability' import { knowledgeAccessCondition } from '@/lib/knowledge/access/predicate' -import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' import { resolveKnowledgeAttributedUserId, @@ -47,7 +42,6 @@ import { type ActiveKnowledgeResourceBaseContext, resolveActiveKnowledgeConnectorContext, resolveActiveKnowledgeResourceContext, - resolveKnowledgeWorkspaceContext, } from '@/lib/knowledge/application/contexts' import { rethrowGitHubInstallationSourceError } from '@/lib/knowledge/application/github-installation-error' import { prepareGitHubInstallationSource } from '@/lib/knowledge/application/github-installation-source' @@ -63,6 +57,7 @@ import { resolveConnectorAccessToken, syncContextForToken, } from '@/lib/knowledge/connectors/access-token' +import { requiresConnectorIndexing } from '@/lib/knowledge/connectors/indexing-policy' import { provisionKnowledgeConnectorMembersBinding, resolveViewerConnectorMemberships, @@ -110,10 +105,8 @@ import type { KnowledgeOperationSource, KnowledgeOrchestrationResult, } from '@/lib/knowledge/orchestration/shared' -import { type KnowledgeReadAccess, knowledgeReadAccessBatches } from '@/lib/knowledge/read-access' import { requireOrganizationSearchApproval } from '@/lib/knowledge/search/integration-policy' import { escapeLikePattern } from '@/lib/knowledge/tags/utils' -import { isMemberSyncStatus } from '@/lib/knowledge/types' import { credentialProviderMatchesService, type ServiceProviderIdentity } from '@/lib/oauth' import { ServiceAccountTokenError } from '@/lib/oauth/credential-service' import { CAPABILITY_RULES, refuseCapability } from '@/lib/permission-groups/capabilities' @@ -125,9 +118,8 @@ import { withSearchSourceDefaults, } from '@/lib/sim-search/connectors' import { SIM_SEARCH_SYNC_INTERVAL_MINUTES } from '@/lib/sim-search/constants' -import { describeSearchSource } from '@/lib/sim-search/source-identity' import { getConnectorApiKeyConfig, isConnectorCredentialTypeAllowed } from '@/connectors/auth' -import { CONNECTOR_META_REGISTRY, getConnectorMeta } from '@/connectors/registry' +import { getConnectorMeta } from '@/connectors/registry' import type { ConnectorAuthConfig } from '@/connectors/types' import { PER_MEMBER_LISTING_CONTEXT } from '@/connectors/utils' @@ -266,13 +258,6 @@ function connectorTarget(context: ActiveKnowledgeResourceBaseContext) { } } -export function requireConnectorWorkspaceId(context: ActiveKnowledgeResourceBaseContext): string { - if (!context.workspaceId) { - throw new OrchestrationError('conflict', 'Knowledge base is missing workspace billing context') - } - return context.workspaceId -} - async function resolveAuthorizedConnectorCredentialIdentity(input: { principal: Principal requestId: string @@ -578,126 +563,6 @@ export const listKnowledgeConnectors = defineAuthorizedKnowledgeUseCase({ }, }) -export interface ListWorkspaceMemberConnectorsInput { - workspaceId: string -} - -/** Live documents per connector that the viewer's tokens match, for the Search tab's counts. */ -async function countViewerDocuments( - connectorIds: readonly string[], - access: KnowledgeReadAccess -): Promise> { - const counts = new Map() - if (connectorIds.length === 0) return counts - const conditions = [ - inArray(document.connectorId, [...connectorIds]), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - ] - for await (const accessCondition of knowledgeReadAccessBatches(access, conditions)) { - const rows = await db - .select({ connectorId: document.connectorId, count: sql`count(*)::int` }) - .from(document) - .where(and(...conditions, accessCondition)) - .groupBy(document.connectorId) - for (const row of rows) { - if (row.connectorId) - counts.set(row.connectorId, (counts.get(row.connectorId) ?? 0) + row.count) - } - } - return counts -} - -/** Live workspace sources that let the viewer connect a crawl account or a mirrored-ACL identity. */ -export const listWorkspaceMemberConnectors = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.listWorkspaceMemberConnectors, - resolveContext: ({ input }: { input: ListWorkspaceMemberConnectorsInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ principal, context }) { - const viewerUserId = resolvePrincipalSubjectUserId(principal) - if (!viewerUserId) return { connectors: [] } - const rows = await db - .select({ - knowledgeBaseId: knowledgeConnector.knowledgeBaseId, - knowledgeBaseName: knowledgeBase.name, - knowledgeBaseIsSearchIndex: knowledgeBase.isSearchIndex, - id: knowledgeConnector.id, - connectorType: knowledgeConnector.connectorType, - accessMode: knowledgeConnector.accessMode, - sourceConfig: knowledgeConnector.sourceConfig, - credentialGroupName: credentialGroup.name, - memberSyncStatus: knowledgeConnector.memberSyncStatus, - credentialGroupId: knowledgeConnector.credentialGroupId, - credentialGroupOptionId: knowledgeConnector.credentialGroupOptionId, - }) - .from(knowledgeConnector) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) - .leftJoin(credentialGroup, eq(credentialGroup.id, knowledgeConnector.credentialGroupId)) - .where( - and( - eq(knowledgeBase.workspaceId, context.workspaceId), - isNull(knowledgeBase.deletedAt), - or( - eq(knowledgeConnector.accessMode, 'members'), - and( - eq(knowledgeConnector.accessMode, 'admin'), - inArray( - knowledgeConnector.connectorType, - Object.values(CONNECTOR_META_REGISTRY) - .filter((meta) => meta.mirrorsSourceAcls && meta.requiresMemberIdentity) - .map((meta) => meta.id) - ) - ) - ), - isNull(knowledgeConnector.archivedAt), - isNull(knowledgeConnector.deletedAt) - ) - ) - .orderBy(asc(knowledgeBase.name), asc(knowledgeConnector.createdAt)) - const [memberships, documentCounts] = await Promise.all([ - resolveViewerConnectorMemberships({ - userId: viewerUserId, - workspaceId: context.workspaceId, - connectors: rows, - }), - countViewerDocuments( - rows.map((row) => row.id), - createKnowledgeAccessProvider(principal, { workspaceId: context.workspaceId }) - ), - ]) - return { - connectors: rows.flatMap((row) => { - const viewerMembership = memberships.get(row.id) - if (!isMemberSyncStatus(row.memberSyncStatus)) { - throw new OrchestrationError( - 'conflict', - `Unexpected member sync status ${row.memberSyncStatus}` - ) - } - return viewerMembership - ? [ - { - knowledgeBaseId: row.knowledgeBaseId, - knowledgeBaseName: row.knowledgeBaseName, - knowledgeBaseIsSearchIndex: row.knowledgeBaseIsSearchIndex, - connectorId: row.id, - connectorType: row.connectorType, - sourceDescription: - (getConnectorMeta(row.connectorType) - ? describeSearchSource(getConnectorMeta(row.connectorType)!, row.sourceConfig) - : '') || truncate(row.credentialGroupName ?? '', 237), - memberSyncStatus: row.accessMode === 'members' ? row.memberSyncStatus : 'idle', - viewerMembership, - viewerDocumentCount: documentCounts.get(row.id) ?? 0, - }, - ] - : [] - }), - } - }, -}) - export const readKnowledgeConnector = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.readConnector, resolveContext: ({ @@ -710,9 +575,11 @@ export const readKnowledgeConnector = defineAuthorizedKnowledgeUseCase({ async execute({ principal, context }) { const connector = await getKnowledgeConnector(context.knowledgeBaseId, context.connectorId) if (!connector) throw new OrchestrationError('not_found', 'Connector not found') - const liveSearchConnector = isLiveEnterpriseSearchEnabled && context.knowledgeBase.isSearchIndex + const dormantSearchIndexConnector = !requiresConnectorIndexing( + context.knowledgeBase.isSearchIndex + ) const [syncLogs, memberSyncLogs, members] = await Promise.all([ - liveSearchConnector + dormantSearchIndexConnector ? [] : db .select() @@ -720,7 +587,7 @@ export const readKnowledgeConnector = defineAuthorizedKnowledgeUseCase({ .where(eq(knowledgeConnectorSyncLog.connectorId, context.connectorId)) .orderBy(desc(knowledgeConnectorSyncLog.startedAt)) .limit(10), - liveSearchConnector + dormantSearchIndexConnector ? [] : db .select() @@ -728,7 +595,7 @@ export const readKnowledgeConnector = defineAuthorizedKnowledgeUseCase({ .where(eq(knowledgeConnectorMemberSyncLog.connectorId, context.connectorId)) .orderBy(desc(knowledgeConnectorMemberSyncLog.startedAt)) .limit(10), - !liveSearchConnector && connector.accessMode === 'members' + !dormantSearchIndexConnector && connector.accessMode === 'members' ? summarizeConnectorMembers(context.connectorId, connector.syncIntervalMinutes) : { active: 0, suspended: 0, stale: 0 }, ]) diff --git a/apps/sim/lib/knowledge/application/documents.test.ts b/apps/sim/lib/knowledge/application/documents.test.ts index 441cb54f51d..31ceaca2b64 100644 --- a/apps/sim/lib/knowledge/application/documents.test.ts +++ b/apps/sim/lib/knowledge/application/documents.test.ts @@ -21,7 +21,7 @@ import { knowledgeTagsServiceMock, knowledgeTagsServiceMockFns, } from '@sim/testing/mocks/knowledge-tags-service.mock' -import { posthogServerMock, posthogServerMockFns } from '@sim/testing/mocks/posthog-server.mock' +import { posthogServerMock } from '@sim/testing/mocks/posthog-server.mock' import { storageServiceMockFns } from '@sim/testing/mocks/storage-service.mock' import { uploadsMock } from '@sim/testing/mocks/uploads.mock' import { @@ -71,7 +71,6 @@ vi.mock('@/lib/posthog/server', () => posthogServerMock) import { OrchestrationError } from '@/lib/core/orchestration/types' import { WORKSPACE_ACCESS_SCOPE } from '@/lib/knowledge/access/scope' import { - bulkDeleteKnowledgeDocuments, createKnowledgeDocuments, deleteKnowledgeDocument, listKnowledgeDocuments, @@ -517,68 +516,4 @@ describe('knowledge document application use cases', () => { expect(mocks.performBulkUpload).not.toHaveBeenCalled() expect(auditMockFns.mockRecordAudit).not.toHaveBeenCalled() }) - - it('conceals a cross-knowledge-base bulk document before mutation for a dual-workspace subject', async () => { - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockRejectedValueOnce( - new OrchestrationError('not_found', 'Document not found') - ) - - const result = await bulkDeleteKnowledgeDocuments.execute({ - principal: { - kind: 'delegated', - serviceId: 'copilot', - subjectUserId: 'dual-workspace-user', - workspaceId: 'workspace-1', - delegationId: 'tool-call-1', - audience: 'sim:knowledge', - issuedAt: new Date(), - expiresAt: new Date(Date.now() + 60_000), - }, - input: { - knowledgeBaseId: 'knowledge-1', - assertedWorkspaceId: 'workspace-1', - documentIds: ['workspace-2-document'], - }, - }) - - expect(result).toMatchObject({ deleted: [], failed: ['workspace-2-document'] }) - expect(workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission).toHaveBeenCalledWith( - 'dual-workspace-user', - 'workspace-1', - null, - undefined, - { forUpdate: undefined } - ) - expect(mocks.deleteDocument).not.toHaveBeenCalled() - expect(auditMockFns.mockRecordAudit).not.toHaveBeenCalled() - }) - - it('audits completed document deletions before propagating infrastructure failure', async () => { - const failure = new Error('document store unavailable') - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockImplementation( - async ({ documentId }) => ({ - ...context, - documentId, - document: { ...document, id: documentId }, - }) - ) - mocks.deleteDocument.mockResolvedValueOnce(undefined).mockRejectedValueOnce(failure) - - await expect( - bulkDeleteKnowledgeDocuments.execute({ - principal: createSessionPrincipal(), - input: { - knowledgeBaseId: 'knowledge-1', - assertedWorkspaceId: 'workspace-1', - documentIds: ['document-1', 'document-2'], - }, - }) - ).rejects.toBe(failure) - - expect(auditMockFns.mockRecordAudit).toHaveBeenCalledOnce() - expect(auditMockFns.mockRecordAudit).toHaveBeenCalledWith( - expect.objectContaining({ resourceId: 'document-1' }) - ) - expect(posthogServerMockFns.mockCaptureServerEvent).not.toHaveBeenCalled() - }) }) diff --git a/apps/sim/lib/knowledge/application/documents.ts b/apps/sim/lib/knowledge/application/documents.ts index 036aa97aea2..c2de94c0d5b 100644 --- a/apps/sim/lib/knowledge/application/documents.ts +++ b/apps/sim/lib/knowledge/application/documents.ts @@ -9,16 +9,10 @@ import { checkAttributedUsageLimits, } from '@/lib/billing/core/billing-attribution' import { authorizeWorkspaceOperation } from '@/lib/core/application' -import { asOrchestrationError, OrchestrationError } from '@/lib/core/orchestration/types' +import { OrchestrationError } from '@/lib/core/orchestration/types' import { generateRequestId } from '@/lib/core/utils/request' import { knowledgeDelegationPolicy } from '@/lib/knowledge/application/authorization' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { - BULK_DELETE_KNOWLEDGE_DOCUMENTS_COST_POLICY, - type KnowledgeBatchExecutionResult, - requireBoundedKnowledgeBatch, - rethrowKnowledgeBatchTerminalFailure, -} from '@/lib/knowledge/application/batch-policy' import { KnowledgeUsageLimitExceededError, resolveKnowledgeAttributedUserId, @@ -27,7 +21,6 @@ import { } from '@/lib/knowledge/application/billing' import { type ActiveKnowledgeDocumentContext, - type ActiveKnowledgeResourceBaseContext, resolveActiveKnowledgeBaseContext, resolveActiveKnowledgeDocumentContext, resolveActiveKnowledgeResourceContext, @@ -160,35 +153,6 @@ export interface DeleteKnowledgeDocumentInput extends ReadKnowledgeDocumentInput source?: string } -export interface BulkDeleteKnowledgeDocumentsInput extends UploadKnowledgeDocumentAdmissionInput { - documentIds: string[] - cancellationSignal?: AbortSignal - source?: string -} - -interface DeletedKnowledgeDocument { - id: string - filename: string - fileSize: number - mimeType: string -} - -export interface BulkDeleteKnowledgeDocumentsResult { - knowledgeBaseId: string - deleted: string[] - failed: string[] - deletedDocuments: DeletedKnowledgeDocument[] - cancelled: boolean -} - -interface BulkDeleteKnowledgeDocumentsExecutionResult - extends BulkDeleteKnowledgeDocumentsResult, - KnowledgeBatchExecutionResult {} - -type BulkDeleteKnowledgeDocumentsContext = ActiveKnowledgeResourceBaseContext & { - documentIds: string[] -} - export interface UpdateKnowledgeDocumentInput extends ReadKnowledgeDocumentInput { filename?: string enabled?: boolean @@ -830,105 +794,6 @@ export const deleteKnowledgeDocument = defineAuthorizedKnowledgeUseCase({ }), }) -export const bulkDeleteKnowledgeDocuments = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.bulkDeleteDocuments, - async resolveContext({ - principal, - input, - }: { - principal: Principal - input: BulkDeleteKnowledgeDocumentsInput - }): Promise { - const documentIds = requireBoundedKnowledgeBatch( - input.documentIds, - 'document IDs', - BULK_DELETE_KNOWLEDGE_DOCUMENTS_COST_POLICY.maxItems - ) - return { - ...(await resolveActiveKnowledgeResourceContext(input, principal)), - documentIds, - } - }, - async execute({ - principal, - input, - context, - }): Promise { - const deletedDocuments: DeletedKnowledgeDocument[] = [] - const failed: string[] = [] - let terminalFailure: KnowledgeBatchExecutionResult['terminalFailure'] - - for (const documentId of context.documentIds) { - if (input.cancellationSignal?.aborted) break - try { - const canonical = await resolveCanonicalActiveKnowledgeDocumentContext( - { - knowledgeBaseId: context.knowledgeBaseId, - documentId, - assertedWorkspaceId: context.workspaceId, - }, - principal - ) - if (canonical.workspaceId) { - await authorizeWorkspaceOperation( - principal, - knowledgeOperations.bulkDeleteDocuments, - canonical, - { delegation: knowledgeDelegationPolicy } - ) - } - if (input.cancellationSignal?.aborted) break - await deleteKnowledgeDocumentInKnowledgeBase( - canonical.knowledgeBaseId, - canonical.documentId, - generateRequestId(), - await canonical.access.get() - ) - deletedDocuments.push({ - id: canonical.documentId, - filename: canonical.document.filename, - fileSize: canonical.document.fileSize, - mimeType: canonical.document.mimeType, - }) - } catch (error) { - const classified = asOrchestrationError(error) - if (classified && classified.code !== 'internal') { - failed.push(documentId) - continue - } - terminalFailure = { error } - break - } - } - - return { - knowledgeBaseId: context.knowledgeBaseId, - deleted: deletedDocuments.map((document) => document.id), - failed, - deletedDocuments, - cancelled: input.cancellationSignal?.aborted ?? false, - ...(terminalFailure && { terminalFailure }), - } - }, - projectAudit: ({ input, context, result }) => - result.deletedDocuments.map((document) => ({ - action: AuditAction.DOCUMENT_DELETED, - resourceType: AuditResourceType.DOCUMENT, - resourceId: document.id, - resourceName: document.filename, - description: `Deleted document "${document.filename}" from knowledge base "${context.knowledgeBase.name}"`, - metadata: { - source: input.source, - knowledgeBaseId: context.knowledgeBaseId, - knowledgeBaseName: context.knowledgeBase.name, - fileName: document.filename, - fileSize: document.fileSize, - mimeType: document.mimeType, - }, - })), - afterSuccess: ({ result }) => rethrowKnowledgeBatchTerminalFailure(result), -}) - export const updateKnowledgeDocument = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.updateDocument, resolveContext: ({ diff --git a/apps/sim/lib/knowledge/application/knowledge-bases.ts b/apps/sim/lib/knowledge/application/knowledge-bases.ts index 76fea95ae96..d85652917ba 100644 --- a/apps/sim/lib/knowledge/application/knowledge-bases.ts +++ b/apps/sim/lib/knowledge/application/knowledge-bases.ts @@ -1,9 +1,6 @@ import { AuditAction, AuditResourceType } from '@sim/audit' import type { Principal, SessionPrincipal } from '@sim/auth/principal' -import { db } from '@sim/db' -import { knowledgeBaseTagDefinitions } from '@sim/db/schema' import { createLogger } from '@sim/logger' -import { inArray } from 'drizzle-orm' import type { CursorKey } from '@/lib/api/list-query' import { authorizeWorkspaceOperation, @@ -116,20 +113,6 @@ export interface RestoreKnowledgeBaseResult extends KnowledgeBaseResult { restored: boolean } -export interface KnowledgeBaseCatalogTagDefinition { - id: string - knowledgeBaseId: string - tagSlot: string - displayName: string - fieldType: string -} - -export interface ListKnowledgeBaseCatalogResult { - knowledgeBases: Array< - KnowledgeBaseResult & { tagDefinitions: KnowledgeBaseCatalogTagDefinition[] } - > -} - export interface CreateKnowledgeBaseInput { workspaceId: string name: string @@ -417,42 +400,6 @@ export const listKnowledgeBases = defineAuthorizedKnowledgeUseCase({ execute: executeListKnowledgeBases, }) -export const listKnowledgeBaseCatalog = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.list, - resolveContext: ({ input }: { input: ListKnowledgeBasesInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ principal, input, context }): Promise { - const result = await executeListKnowledgeBases({ principal, input, context }) - const knowledgeBaseIds = result.knowledgeBases.map(({ knowledgeBase }) => knowledgeBase.id) - const tagDefinitions = - knowledgeBaseIds.length === 0 - ? [] - : await db - .select({ - id: knowledgeBaseTagDefinitions.id, - knowledgeBaseId: knowledgeBaseTagDefinitions.knowledgeBaseId, - tagSlot: knowledgeBaseTagDefinitions.tagSlot, - displayName: knowledgeBaseTagDefinitions.displayName, - fieldType: knowledgeBaseTagDefinitions.fieldType, - }) - .from(knowledgeBaseTagDefinitions) - .where(inArray(knowledgeBaseTagDefinitions.knowledgeBaseId, knowledgeBaseIds)) - .orderBy(knowledgeBaseTagDefinitions.tagSlot) - const tagsByKnowledgeBase = new Map() - for (const definition of tagDefinitions) { - const existing = tagsByKnowledgeBase.get(definition.knowledgeBaseId) - if (existing) existing.push(definition) - else tagsByKnowledgeBase.set(definition.knowledgeBaseId, [definition]) - } - return { - knowledgeBases: result.knowledgeBases.map((entry) => ({ - ...entry, - tagDefinitions: tagsByKnowledgeBase.get(entry.knowledgeBase.id) ?? [], - })), - } - }, -}) - /** * Un-archives a knowledge base and reports it as it now stands. * diff --git a/apps/sim/lib/knowledge/application/knowledge-vfs.ts b/apps/sim/lib/knowledge/application/knowledge-vfs.ts deleted file mode 100644 index 9b7da74bd8c..00000000000 --- a/apps/sim/lib/knowledge/application/knowledge-vfs.ts +++ /dev/null @@ -1,263 +0,0 @@ -import { AuditAction, AuditResourceType } from '@sim/audit' -import { resolvePrincipalAttribution } from '@sim/auth/principal' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { generateRequestId } from '@/lib/core/utils/request' -import { - createResourceVfsFolders, - deleteResourceVfsFolders, - type FolderedResourceAdapter, - resolveResourceRowBySegments, - transferResourceVfsItems, -} from '@/lib/folders/application/resource-vfs' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { - type KnowledgeWorkspaceContext, - resolveKnowledgeWorkspaceContext, -} from '@/lib/knowledge/application/contexts' -import { authorizeSearchIndexDeletion } from '@/lib/knowledge/application/knowledge-base-access' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { - deleteKnowledgeBase, - findActiveKnowledgeBasesByExactName, - getWorkspaceKnowledgeBases, - updateKnowledgeBase, -} from '@/lib/knowledge/service' -import type { KnowledgeBaseWithCounts } from '@/lib/knowledge/types' - -interface KnowledgeVfsReferenceInput { - workspaceId: string - sourceName: string - /** Folder segments + leaf name; when present the nested-aware resolver is used. */ - sourceSegments?: string[] -} - -const knowledgeVfsAdapter: FolderedResourceAdapter = { - resourceType: 'knowledge_base', - rootSegment: 'knowledgebases', - label: 'knowledge base', - async listRows(workspaceId) { - const { data: rows } = await getWorkspaceKnowledgeBases(workspaceId, 'active', {}) - return rows.map((kb) => ({ id: kb.id, name: kb.name, folderId: kb.folderId ?? null })) - }, - async moveRow(row, folderId, workspaceId) { - await updateKnowledgeBase(row.id, { folderId }, generateRequestId(), { - assertedWorkspaceId: workspaceId, - }) - }, - async renameRow(row, newName, workspaceId) { - const updated = await updateKnowledgeBase(row.id, { name: newName }, generateRequestId(), { - assertedWorkspaceId: workspaceId, - }) - return { id: updated.id, name: updated.name } - }, -} - -export interface RenameKnowledgeBaseByVfsPathInput extends KnowledgeVfsReferenceInput { - newName: string -} - -export type DeleteKnowledgeBaseByVfsPathInput = KnowledgeVfsReferenceInput - -async function resolveKnowledgeBaseByVfsName( - context: KnowledgeWorkspaceContext, - sourceName: string, - sourceSegments?: string[] -): Promise> { - if (sourceSegments && sourceSegments.length > 1) { - const row = await resolveResourceRowBySegments( - knowledgeVfsAdapter, - context.workspaceId, - sourceSegments - ) - /** - * Resolved by folder path, so the name may be shared with knowledge bases in - * other folders. The exact-name lookup caps its result set, so the full list - * is read here and narrowed by id instead. - */ - const { data: rows } = await getWorkspaceKnowledgeBases(context.workspaceId, 'active', { - search: row.name, - }) - const match = rows.find((kb) => kb.id === row.id) - if (!match) { - throw new OrchestrationError( - 'not_found', - `Knowledge base not found at knowledgebases/${sourceSegments.join('/')}` - ) - } - return match - } - const matches = await findActiveKnowledgeBasesByExactName(context.workspaceId, sourceName) - if (matches.length > 1) { - throw new OrchestrationError( - 'conflict', - `Knowledge base path is ambiguous: knowledgebases/${sourceName}` - ) - } - const knowledgeBase = matches[0] - if (!knowledgeBase) { - throw new OrchestrationError( - 'not_found', - `Knowledge base not found at knowledgebases/${sourceName}` - ) - } - return knowledgeBase -} - -export const renameKnowledgeBaseByVfsPath = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.renameByVfsPath, - resolveContext: ({ input }: { input: RenameKnowledgeBaseByVfsPathInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ input, context }) { - const knowledgeBase = await resolveKnowledgeBaseByVfsName( - context, - input.sourceName, - input.sourceSegments - ) - const updated = await updateKnowledgeBase( - knowledgeBase.id, - { name: input.newName }, - generateRequestId(), - { assertedWorkspaceId: context.workspaceId } - ) - return { - id: updated.id, - name: updated.name, - previousName: knowledgeBase.name, - workspaceId: context.workspaceId, - } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.KNOWLEDGE_BASE_UPDATED, - resourceType: AuditResourceType.KNOWLEDGE_BASE, - resourceId: result.id, - resourceName: result.name, - description: `Renamed knowledge base to "${result.name}"`, - metadata: { source: 'copilot_vfs', previousName: result.previousName, updatedFields: ['name'] }, - }), -}) - -export const deleteKnowledgeBaseByVfsPath = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.deleteByVfsPath, - resolveContext: ({ input }: { input: DeleteKnowledgeBaseByVfsPathInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ principal, input, context }) { - const knowledgeBase = await resolveKnowledgeBaseByVfsName( - context, - input.sourceName, - input.sourceSegments - ) - const allowSearchIndexDelete = await authorizeSearchIndexDeletion( - principal, - context, - knowledgeBase - ) - await deleteKnowledgeBase(knowledgeBase.id, generateRequestId(), { - allowSearchIndexDelete, - assertedWorkspaceId: context.workspaceId, - }) - return { - id: knowledgeBase.id, - name: knowledgeBase.name, - workspaceId: context.workspaceId, - deleted: true as const, - } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.KNOWLEDGE_BASE_DELETED, - resourceType: AuditResourceType.KNOWLEDGE_BASE, - resourceId: result.id, - resourceName: result.name, - description: `Deleted knowledge base "${result.name}"`, - metadata: { source: 'copilot_vfs', knowledgeBaseName: result.name }, - }), -}) - -export interface KnowledgeVfsPathsInput { - workspaceId: string - paths: Array<{ source: string; segments: string[] }> -} - -export interface TransferKnowledgeVfsItemsInput { - workspaceId: string - sources: Array<{ source: string; segments: string[] }> - destination: { segments: string[]; trailingSlash: boolean } -} - -/** mkdir -p under knowledgebases/ — folder invariants live in lib/folders. */ -export const createKnowledgeVfsFolders = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.manageVfsFolders, - resolveContext: ({ input }: { input: KnowledgeVfsPathsInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ principal, input, context }) { - const userId = resolvePrincipalAttribution(principal, { - workspaceBillingOwnerUserId: context.billedAccountUserId, - }).attributedUserId - const outcomes = await createResourceVfsFolders(knowledgeVfsAdapter, { - workspaceId: context.workspaceId, - userId, - paths: input.paths, - }) - return { outcomes, workspaceId: context.workspaceId } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.KNOWLEDGE_BASE_UPDATED, - resourceType: AuditResourceType.KNOWLEDGE_BASE, - resourceId: result.workspaceId, - resourceName: 'knowledgebases', - description: 'Created knowledge base folders', - metadata: { op: 'vfs_mkdir', count: result.outcomes.length, source: 'copilot_vfs' }, - }), -}) - -/** mv under knowledgebases/: rows into folders, folder moves/renames, leaf renames. */ -export const transferKnowledgeVfsItems = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.moveByVfsPath, - resolveContext: ({ input }: { input: TransferKnowledgeVfsItemsInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ principal, input, context }) { - const userId = resolvePrincipalAttribution(principal, { - workspaceBillingOwnerUserId: context.billedAccountUserId, - }).attributedUserId - const outcomes = await transferResourceVfsItems(knowledgeVfsAdapter, { - workspaceId: context.workspaceId, - userId, - sources: input.sources, - destination: input.destination, - }) - return { outcomes, workspaceId: context.workspaceId } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.KNOWLEDGE_BASE_UPDATED, - resourceType: AuditResourceType.KNOWLEDGE_BASE, - resourceId: result.workspaceId, - resourceName: 'knowledgebases', - description: 'Moved knowledge base VFS items', - metadata: { op: 'vfs_mv', count: result.outcomes.length, source: 'copilot_vfs' }, - }), -}) - -/** rm of knowledgebases/ folder paths — recursive via the shared cascade. */ -export const deleteKnowledgeVfsFolders = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.manageVfsFolders, - resolveContext: ({ input }: { input: KnowledgeVfsPathsInput }) => - resolveKnowledgeWorkspaceContext(input), - async execute({ principal, input, context }) { - const userId = resolvePrincipalAttribution(principal, { - workspaceBillingOwnerUserId: context.billedAccountUserId, - }).attributedUserId - const outcomes = await deleteResourceVfsFolders(knowledgeVfsAdapter, { - workspaceId: context.workspaceId, - userId, - paths: input.paths, - }) - return { outcomes, workspaceId: context.workspaceId } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.KNOWLEDGE_BASE_DELETED, - resourceType: AuditResourceType.KNOWLEDGE_BASE, - resourceId: result.workspaceId, - resourceName: 'knowledgebases', - description: 'Deleted knowledge base folders', - metadata: { op: 'vfs_rm_folder', count: result.outcomes.length, source: 'copilot_vfs' }, - }), -}) diff --git a/apps/sim/lib/knowledge/application/operations.test.ts b/apps/sim/lib/knowledge/application/operations.test.ts index f8a9fabcd26..0f91cbc3c8d 100644 --- a/apps/sim/lib/knowledge/application/operations.test.ts +++ b/apps/sim/lib/knowledge/application/operations.test.ts @@ -66,7 +66,6 @@ describe('knowledge operation registry', () => { const operations = [ knowledgeOperations.updateDocument, knowledgeOperations.addWorkspaceFiles, - knowledgeOperations.bulkDeleteDocuments, knowledgeOperations.createTag, knowledgeOperations.updateTag, knowledgeOperations.deleteTag, diff --git a/apps/sim/lib/knowledge/application/operations.ts b/apps/sim/lib/knowledge/application/operations.ts index cd8232201cd..c341f2cb6be 100644 --- a/apps/sim/lib/knowledge/application/operations.ts +++ b/apps/sim/lib/knowledge/application/operations.ts @@ -82,10 +82,6 @@ const ALL_PRINCIPAL_POLICY = { ], delegatedServices: ['copilot'], } as const -const COPILOT_PRINCIPAL_POLICY = { - principalKinds: ['delegated'], - delegatedServices: ['copilot'], -} as const const ALL_PRINCIPAL_WITH_EXECUTOR_POLICY = { principalKinds: [ @@ -396,42 +392,6 @@ export const knowledgeOperations = { ...ALL_PRINCIPAL_POLICY, }) ), - renameByVfsPath: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.vfs.rename', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - ...COPILOT_PRINCIPAL_POLICY, - }) - ), - moveByVfsPath: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.vfs.move', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - ...COPILOT_PRINCIPAL_POLICY, - }) - ), - manageVfsFolders: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.vfs.folders.manage', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - ...COPILOT_PRINCIPAL_POLICY, - }) - ), - deleteByVfsPath: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.vfs.delete', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - ...COPILOT_PRINCIPAL_POLICY, - }) - ), search: defineKnowledgeOperation( defineWorkspaceOperation({ id: 'knowledge.search', @@ -540,16 +500,6 @@ export const knowledgeOperations = { ...ALL_PRINCIPAL_WITH_EXECUTOR_POLICY, }) ), - bulkDeleteDocuments: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.documents.bulk_delete', - oauthScope: 'api:write', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - ...HUMAN_AND_COPILOT_PRINCIPAL_POLICY, - }) - ), updateDocument: defineKnowledgeOperation( defineWorkspaceOperation({ id: 'knowledge.documents.update', @@ -874,16 +824,6 @@ export const knowledgeOperations = { }), { organizationDelegation: 'allow' } ), - /** Sources with a personal connection, including identities used by mirrored ACLs. */ - listWorkspaceMemberConnectors: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.connectors.members.list', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), /** * A workspace reader connecting their own account for a member crawl or a * mirrored-ACL identity. Enrollment never creates crawler access grants. diff --git a/apps/sim/lib/knowledge/application/organization-search-overview.ts b/apps/sim/lib/knowledge/application/organization-search-overview.ts index 085e0537462..f1d0b114158 100644 --- a/apps/sim/lib/knowledge/application/organization-search-overview.ts +++ b/apps/sim/lib/knowledge/application/organization-search-overview.ts @@ -11,7 +11,7 @@ import { import { and, eq, exists, inArray, isNotNull, isNull, type SQL, sql } from 'drizzle-orm' import { OrchestrationError } from '@/lib/core/orchestration/types' import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' -import { SOURCE_ACL_MAX_AGE_MS } from '@/lib/knowledge/access/freshness' +import { sourceAclFreshnessCutoff } from '@/lib/knowledge/access/predicate' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' @@ -106,7 +106,7 @@ export const readOrganizationSearchOverview = defineAuthorizedKnowledgeUseCase({ OR coalesce(${knowledgeConnector.nextMemberSyncAt} <= statement_timestamp(), false) )) )` - const cutoff = sql`statement_timestamp() - (${SOURCE_ACL_MAX_AGE_MS} * interval '1 millisecond')` + const cutoff = sourceAclFreshnessCutoff() /** * A member whose last run only had per-document content failures carries * {@link SOURCE_CONTENT_ERROR} as a marker so its next run lists fully; the diff --git a/apps/sim/lib/knowledge/application/organization-search-stats.test.ts b/apps/sim/lib/knowledge/application/organization-search-stats.test.ts index 67865c975bd..050cf7e7bf3 100644 --- a/apps/sim/lib/knowledge/application/organization-search-stats.test.ts +++ b/apps/sim/lib/knowledge/application/organization-search-stats.test.ts @@ -4,6 +4,7 @@ import { createPersonalApiKeyPrincipal, createSessionPrincipal, } from '@sim/testing/factories/principal.factory' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -17,7 +18,7 @@ import { permissionGroupsResolveMockFns, } from '@sim/testing/mocks/permission-groups-resolve.mock' import { workspaceAuthzMock } from '@sim/testing/mocks/workspace-authz.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const mocks = vi.hoisted(() => ({ load: vi.fn(), @@ -31,12 +32,17 @@ vi.mock('@/lib/knowledge/search/activity-stats', () => ({ })) import { readOrganizationSearchStats } from '@/lib/knowledge/application/organization-search-stats' +import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' const principal = createSessionPrincipal({ userId: 'admin', sessionId: 'session' }) const input = { organizationId: 'organization', period: '7d', surface: 'mcp' } as const +afterEach(resetEnvFlagsMock) + beforeEach(() => { resetDbChainMock() + /** The Stats tab reports indexed organization search, so these cases run with it on. */ + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue({ organizationId: 'organization', }) @@ -66,6 +72,15 @@ describe('organization Search stats authorization', () => { expect(knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext).not.toHaveBeenCalled() expect(mocks.load).not.toHaveBeenCalled() }) + it('refuses while indexed organization search is dormant, before aggregation', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) + queueTableRows(member, [{ role: 'admin' }]) + await expect(readOrganizationSearchStats.execute({ principal, input })).rejects.toBeInstanceOf( + SearchIndexDormantError + ) + expect(mocks.load).not.toHaveBeenCalled() + }) + it('fails closed when Search is disabled', async () => { queueTableRows(member, [{ role: 'admin' }]) knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable.mockRejectedValueOnce( diff --git a/apps/sim/lib/knowledge/application/organization-search-stats.ts b/apps/sim/lib/knowledge/application/organization-search-stats.ts index 98aabe2973c..6b79404d65b 100644 --- a/apps/sim/lib/knowledge/application/organization-search-stats.ts +++ b/apps/sim/lib/knowledge/application/organization-search-stats.ts @@ -7,12 +7,15 @@ import { loadOrganizationSearchStats, type SearchStatsInput, } from '@/lib/knowledge/search/activity-stats' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' +/** The indexed Stats tab's activity report; refused while indexed organization search is dormant. */ export const readOrganizationSearchStats = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.readOrganizationSearchStats, resolveContext: ({ input }: { input: SearchStatsInput }) => resolveKnowledgeOwnerContext({ organizationId: input.organizationId }), async execute({ context, input }) { + assertIndexedOrgSearchEnabled() if (!context.organizationId) throw new OrchestrationError('validation', 'Organization is required') await requireOrganizationSearchAvailable(context.organizationId) diff --git a/apps/sim/lib/knowledge/application/personal-search-integration-pages.ts b/apps/sim/lib/knowledge/application/personal-search-integration-pages.ts new file mode 100644 index 00000000000..709ee5f81b2 --- /dev/null +++ b/apps/sim/lib/knowledge/application/personal-search-integration-pages.ts @@ -0,0 +1,49 @@ +import type { Principal } from '@sim/auth/principal' +import { + type ListPersonalSearchIntegrationsInput, + listPersonalSearchIntegrations, +} from '@/lib/knowledge/application/personal-search-integrations' + +/** One page of the viewer's personal Search integrations. */ +export type PersonalSearchIntegrationsPage = Awaited< + ReturnType +> + +/** The most pages one walk of the personal inventory reads before it stops. */ +const MAX_PERSONAL_SEARCH_INTEGRATION_PAGES = 100 + +/** + * Every page of the viewer's personal Search integrations, in order: Live Search answers in one + * page, indexed search in one per batch of sources. A cursor that repeats, or a walk past + * {@link MAX_PERSONAL_SEARCH_INTEGRATION_PAGES}, is a defect and throws rather than answering from + * part of the inventory. + */ +export async function* personalSearchIntegrationPages({ + principal, + input, + signal, +}: { + principal: Principal + input: Omit + signal?: AbortSignal +}): AsyncGenerator { + const seen = new Set() + let cursor: string | undefined + for (let page = 0; page < MAX_PERSONAL_SEARCH_INTEGRATION_PAGES; page++) { + signal?.throwIfAborted() + const inventory = await listPersonalSearchIntegrations.execute({ + principal, + input: { ...input, ...(cursor ? { cursor } : {}) }, + }) + signal?.throwIfAborted() + yield inventory + if (inventory.nextCursor === null) return + if (seen.has(inventory.nextCursor)) + throw new Error('Personal Search integration pagination did not advance') + seen.add(inventory.nextCursor) + cursor = inventory.nextCursor + } + throw new Error( + `Personal Search integration pagination exceeded ${MAX_PERSONAL_SEARCH_INTEGRATION_PAGES} pages` + ) +} diff --git a/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts b/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts index 61e4626ccc1..aeaf4296001 100644 --- a/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts +++ b/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts @@ -128,6 +128,8 @@ beforeEach(() => { mockGetConnectorAccessAvailability.mockReturnValue({ members: true }) }) describe('personal Search inventory', () => { + beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) + it.each([ [{}, 'connected', 'not_indexed'], [{ isSyncing: true }, 'connected', 'indexing'], diff --git a/apps/sim/lib/knowledge/application/personal-search-integrations.ts b/apps/sim/lib/knowledge/application/personal-search-integrations.ts index 525069a23b8..5121c564f0c 100644 --- a/apps/sim/lib/knowledge/application/personal-search-integrations.ts +++ b/apps/sim/lib/knowledge/application/personal-search-integrations.ts @@ -2,28 +2,21 @@ import { requirePrincipalSubjectUserId } from '@sim/auth/principal' import { db } from '@sim/db' import { credentialGroup, user } from '@sim/db/schema' import { eq } from 'drizzle-orm' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { OrchestrationError } from '@/lib/core/orchestration/types' import { findCredentialGroupProviderFromProviderId } from '@/lib/credential-groups/providers' import { isScopedCredentialGroupsAvailable } from '@/lib/credential-groups/scoped-availability' import { readSearchConnectionCompletion } from '@/lib/credential-groups/search-connection-completion' import { getOrganizationAccountsGroup } from '@/lib/credential-groups/service' import { listViewerOrganizationAccounts } from '@/lib/credential-groups/viewer-accounts' -import { - getIntegrationAvailability, - isOAuthServiceDeploymentAvailable, -} from '@/lib/integrations/availability.server' -import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' import { resolveKnowledgeOrganizationContext } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { listConfiguredSearchProviderTypes } from '@/lib/knowledge/application/search-source-overview' -import { listSearchSources } from '@/lib/knowledge/application/search-sources' import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' -import { getConnectorAccessAvailability, SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' +import { SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' +import { listIndexedPersonalSearchIntegrations } from '@/lib/sim-search/indexed/integrations/personal-search-integrations' import { LIVE_SEARCH_SCOPE_FIELDS } from '@/lib/sim-search/live/policy-schema' -import { findSharedSlackSearchInstallation } from '@/lib/slack-search/shared-app' export interface ListPersonalSearchIntegrationsInput { organizationId: string @@ -46,209 +39,78 @@ export const listPersonalSearchIntegrations = defineAuthorizedKnowledgeUseCase({ .where(eq(user.id, userId)) .limit(1) if (!viewer) throw new OrchestrationError('forbidden', 'The current person is unavailable') - if (isLiveEnterpriseSearchEnabled) { - if (input.connectorId || input.cursor) - throw new OrchestrationError('validation', 'Refresh your live account connections') - const scope = { kind: 'organization', organizationId: context.organizationId } as const - if (!(await isScopedCredentialGroupsAvailable(scope))) - return { completedCredentialId: null, connections: [], available: [], nextCursor: null } - const [group, approvals] = await Promise.all([ - getOrganizationAccountsGroup(context.organizationId), - listOrganizationSearchApprovals(context.organizationId), - ]) - const accounts = group - ? await listViewerOrganizationAccounts({ - organizationId: context.organizationId, - userId, - matching: eq(credentialGroup.id, group.id), - }) - : [] - const connections = (group?.options ?? []).flatMap((option) => { - const connector = SEARCH_CONNECTORS.find( - (entry) => findCredentialGroupProviderFromProviderId(entry.providerId) === option.provider - ) - if ( - !connector || - !LIVE_SEARCH_SCOPE_FIELDS[connector.type] || - (input.connectorType && connector.type !== input.connectorType) - ) - return [] - const ready = Boolean( - viewer.emailVerified && - group?.status === 'active' && - option.status === 'active' && - option.configurationStatus === 'ready' && - approvals.get(connector.type) - ) - const target: SearchConnectionTarget = { - type: 'link', - provider: option.provider, - connectorType: connector.type, - connectionMode: 'live', - optionId: option.id, - } - const own = accounts - .filter((account) => account.optionId === option.id) - .map((account) => ({ - credentialId: account.credentialId, - displayName: account.displayName, - status: - account.status === 'active' ? ('connected' as const) : ('reconnect_needed' as const), - action: ready ? { ...target, credentialId: account.credentialId } : null, - })) - return [ - { - name: connector.meta.name, - providerId: option.provider, - connectorType: connector.type, - connectorId: undefined, - knowledgeBaseId: undefined, - indexingStatus: undefined, - description: '', - accounts: own, - connectionStatus: !ready - ? ('unavailable' as const) - : own.some((account) => account.status === 'reconnect_needed') - ? ('reconnect_needed' as const) - : own.length - ? ('connected' as const) - : ('not_connected' as const), - action: ready ? target : null, - }, - ] - }) - return { - completedCredentialId: input.completionId - ? await readSearchConnectionCompletion({ - organizationId: context.organizationId, - userId, - completionId: input.completionId, - }) - : null, - connections: connections.filter((entry) => entry.accounts.length > 0), - available: connections.flatMap((entry) => - entry.action - ? [{ name: entry.name, description: entry.description, target: entry.action }] - : [] - ), - nextCursor: null, - } - } - const [page, configuredTypes, approvals, access, sharedSlack] = await Promise.all([ - listSearchSources.execute({ principal, input }), - listConfiguredSearchProviderTypes({ organizationId: context.organizationId }), + if (isIndexedOrgSearchEnabled()) + return listIndexedPersonalSearchIntegrations({ principal, input, context, userId, viewer }) + if (input.connectorId || input.cursor) + throw new OrchestrationError('validation', 'Refresh your live account connections') + const scope = { kind: 'organization', organizationId: context.organizationId } as const + if (!(await isScopedCredentialGroupsAvailable(scope))) + return { completedCredentialId: null, connections: [], available: [], nextCursor: null } + const [group, approvals] = await Promise.all([ + getOrganizationAccountsGroup(context.organizationId), listOrganizationSearchApprovals(context.organizationId), - resolveKnowledgeAccessAvailability(context), - findSharedSlackSearchInstallation(context.organizationId), ]) - const deployment = new Map( - getIntegrationAvailability().map((entry) => [entry.type.toLowerCase(), entry]) - ) - const oauth = new Map( - SEARCH_CONNECTORS.map((entry) => [ - entry.providerId, - isOAuthServiceDeploymentAvailable(entry.providerId), - ]) - ) - const configured = new Set(configuredTypes) - const eligible = (connectorType: string) => { - const connector = SEARCH_CONNECTORS.find((entry) => entry.type === connectorType) - return Boolean( + const accounts = group + ? await listViewerOrganizationAccounts({ + organizationId: context.organizationId, + userId, + matching: eq(credentialGroup.id, group.id), + }) + : [] + const connections = (group?.options ?? []).flatMap((option) => { + const connector = SEARCH_CONNECTORS.find( + (entry) => findCredentialGroupProviderFromProviderId(entry.providerId) === option.provider + ) + if ( + !connector || + !LIVE_SEARCH_SCOPE_FIELDS[connector.type] || + (input.connectorType && connector.type !== input.connectorType) + ) + return [] + const ready = Boolean( viewer.emailVerified && - connector && - approvals.get(connectorType) && - getConnectorAccessAvailability(connector.meta, deployment, { - memberAccessAvailable: access.memberScoped, - mirroredAccessAvailable: access.sourceMirrored, - oauthServiceAvailability: oauth, - isIntegrationAvailabilityReady: true, - }).members + group?.status === 'active' && + option.status === 'active' && + option.configurationStatus === 'ready' && + approvals.get(connector.type) ) - } - const projected = page.sources.flatMap((source) => { - const connector = SEARCH_CONNECTORS.find((entry) => entry.type === source.connectorType) - if (!connector) return [] const target: SearchConnectionTarget = { type: 'link', - provider: connector.providerId, - connectorType: source.connectorType, - connectorId: source.connectorId, + provider: option.provider, + connectorType: connector.type, + connectionMode: 'live', + optionId: option.id, } - const canConnect = - eligible(source.connectorType) && - source.enabled && - source.availability === 'available' && - source.viewerEmailVerified && - source.connectionRequired && - source.viewerMembership !== null && - !['revoked', 'unverified_email'].includes(source.viewerMembership) - const accounts = source.viewerAccounts.map((account) => { - if (!account.status) throw new Error('Personal Search account status is missing') - return { + const own = accounts + .filter((account) => account.optionId === option.id) + .map((account) => ({ credentialId: account.credentialId, displayName: account.displayName, status: account.status === 'active' ? ('connected' as const) : ('reconnect_needed' as const), - action: - canConnect && account.status === 'needs_reauth' - ? { ...target, credentialId: account.credentialId } - : null, - } - }) + action: ready ? { ...target, credentialId: account.credentialId } : null, + })) return [ { name: connector.meta.name, - providerId: connector.providerId, + providerId: option.provider, connectorType: connector.type, - connectorId: source.connectorId, - knowledgeBaseId: source.knowledgeBaseId, - description: source.sourceDescription, - accounts, - connectionStatus: accounts.some((account) => account.status === 'reconnect_needed') - ? ('reconnect_needed' as const) - : accounts.length - ? ('connected' as const) - : canConnect - ? ('not_connected' as const) - : ('unavailable' as const), - indexingStatus: - !source.enabled || source.availability !== 'available' || source.approved === false - ? ('paused' as const) - : source.isSyncing - ? ('indexing' as const) - : source.hasSyncError || source.viewerFailedDocumentCount > 0 - ? ('sync_failed' as const) - : source.hasViewerDocuments - ? ('indexed' as const) - : ('not_indexed' as const), - action: canConnect && !accounts.length ? target : null, + connectorId: undefined, + knowledgeBaseId: undefined, + indexingStatus: undefined, + description: '', + accounts: own, + connectionStatus: !ready + ? ('unavailable' as const) + : own.some((account) => account.status === 'reconnect_needed') + ? ('reconnect_needed' as const) + : own.length + ? ('connected' as const) + : ('not_connected' as const), + action: ready ? target : null, }, ] }) - const available: Array<{ name: string; description: string; target: SearchConnectionTarget }> = - [ - ...projected.flatMap((entry) => - entry.action - ? [{ name: entry.name, description: entry.description, target: entry.action }] - : [] - ), - ...SEARCH_CONNECTORS.filter( - (connector) => - !input.connectorId && - (!input.connectorType || connector.type === input.connectorType) && - (connector.type !== 'slack' || sharedSlack !== null) && - (!configured.has(connector.type) || connector.setupFields.length > 0) && - eligible(connector.type) - ).map((connector) => ({ - name: connector.meta.name, - description: '', - target: { - type: 'link' as const, - provider: connector.providerId, - connectorType: connector.type, - }, - })), - ] return { completedCredentialId: input.completionId ? await readSearchConnectionCompletion({ @@ -257,9 +119,13 @@ export const listPersonalSearchIntegrations = defineAuthorizedKnowledgeUseCase({ completionId: input.completionId, }) : null, - connections: projected.filter((entry) => entry.accounts.length > 0), - available, - nextCursor: page.nextCursor, + connections: connections.filter((entry) => entry.accounts.length > 0), + available: connections.flatMap((entry) => + entry.action + ? [{ name: entry.name, description: entry.description, target: entry.action }] + : [] + ), + nextCursor: null, } }, }) diff --git a/apps/sim/lib/knowledge/application/search-integrations.test.ts b/apps/sim/lib/knowledge/application/search-integrations.test.ts index ac35eec4829..1f00c17269a 100644 --- a/apps/sim/lib/knowledge/application/search-integrations.test.ts +++ b/apps/sim/lib/knowledge/application/search-integrations.test.ts @@ -153,6 +153,7 @@ describe('organization Search approval', () => { }) it('preserves existing sources while an explicit deactivation overrides them', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) queueTableRows(member, [{ role: 'member' }]) queueTableRows(organizationSearchIntegration, [{ connectorType: 'gmail', approved: false }]) queueTableRows(knowledgeConnector, [ @@ -223,6 +224,7 @@ describe('organization Search controls through Mothership', () => { }) it('allows delegated members to read approval state without granting writes', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) queueTableRows(member, [{ role: 'member' }]) queueTableRows(organizationSearchIntegration, [{ connectorType: 'gmail', approved: true }]) queueTableRows(knowledgeConnector, []) @@ -377,6 +379,7 @@ describe('live organization search policies', () => { ) it('rejects policy writes when the rollout flag is off', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) queueTableRows(member, [{ role: 'owner' }]) await expect( approveSearchIntegration.execute({ diff --git a/apps/sim/lib/knowledge/application/search.test.ts b/apps/sim/lib/knowledge/application/search.test.ts index 4ec52416185..48f1b6e06ee 100644 --- a/apps/sim/lib/knowledge/application/search.test.ts +++ b/apps/sim/lib/knowledge/application/search.test.ts @@ -17,6 +17,7 @@ import { billingUsageMonitorMock, billingUsageMonitorMockFns, } from '@sim/testing/mocks/billing-usage-monitor.mock' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -40,7 +41,7 @@ import { import { permissionGroupsResolveMock } from '@sim/testing/mocks/permission-groups-resolve.mock' import { getMockPlatformEvent, telemetryMock } from '@sim/testing/mocks/telemetry.mock' import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { OrchestrationError } from '@/lib/core/orchestration/types' const hoisted = vi.hoisted(() => ({ @@ -238,6 +239,7 @@ describe('knowledge search application use case', () => { describe.each(['workspace', 'organization'] as const)('%s ranking policy', (scope) => { beforeEach(() => { if (scope === 'organization') { + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) mocks.getKnowledgeBase.mockResolvedValue({ ...knowledgeBase, workspaceId: null, @@ -247,6 +249,7 @@ describe('knowledge search application use case', () => { queueTableRows(member, [{ role: 'member' }]) } }) + afterEach(resetEnvFlagsMock) it('meters only successful organization calls under the acting person', async () => { await searchKnowledge.execute({ @@ -294,6 +297,32 @@ describe('knowledge search application use case', () => { expect(mocks.executeSearch).not.toHaveBeenCalled() }) + describe('while indexed organization search is dormant', () => { + const principal = createSessionPrincipal() + beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: true })) + afterEach(resetEnvFlagsMock) + + it.each([ + ['an organization', { workspaceId: null, organizationId: 'org-canonical' }], + ['a workspace', {}], + ])('still searches %s search index named by id', async (_owner, owner) => { + mocks.getKnowledgeBase.mockResolvedValue({ + ...knowledgeBase, + ...owner, + isSearchIndex: true, + }) + queueTableRows(member, [{ role: 'member' }]) + const result = await searchKnowledge.execute({ + principal, + input: { knowledgeBaseIds: ['knowledge-1'], query: 'answer', topK: 5 }, + }) + expect(result.results).toHaveLength(1) + expect(mocks.executeSearch).toHaveBeenCalledWith( + expect.objectContaining({ knowledgeBaseIds: ['knowledge-1'], indexedRetrieval: false }) + ) + }) + }) + it('authorizes every canonical knowledge base before billing and search', async () => { const result = await searchKnowledge.execute({ principal: createSessionPrincipal(), diff --git a/apps/sim/lib/knowledge/application/search.ts b/apps/sim/lib/knowledge/application/search.ts index 706071b894b..319e2622ee1 100644 --- a/apps/sim/lib/knowledge/application/search.ts +++ b/apps/sim/lib/knowledge/application/search.ts @@ -39,14 +39,11 @@ import { hasRerankerCredential, rerank } from '@/lib/knowledge/reranker' import type { RerankerStatus } from '@/lib/knowledge/reranker-models' import { recordOrganizationSearchActivity } from '@/lib/knowledge/search/activity' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' +import type { SearchResult } from '@/lib/knowledge/search/candidates' import { resolveKnowledgeSearchDefaults } from '@/lib/knowledge/search/defaults' import { annotateSearchDiagnostics, measureSearchStage } from '@/lib/knowledge/search/diagnostics' import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' -import { - type RetrievalStatus, - retrieveKnowledgeSearch, - type SearchResult, -} from '@/lib/knowledge/search/queries' +import { type RetrievalStatus, retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' import { importKnowledgeSearchResultSecretProvenance } from '@/lib/knowledge/secret-provenance' import { getActiveKnowledgeBaseReferences } from '@/lib/knowledge/service' import { @@ -56,6 +53,7 @@ import { import { getDocumentTagDefinitionsByKnowledgeBaseIds } from '@/lib/knowledge/tags/service' import type { DocumentTagDefinition } from '@/lib/knowledge/tags/types' import type { StructuredFilter } from '@/lib/knowledge/types' +import { usesIndexedRetrieval } from '@/lib/sim-search/indexed/gate' import { estimateTokenCount } from '@/lib/tokenization/estimators' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' import { getRerankModelPricing } from '@/providers/models' @@ -475,7 +473,7 @@ export async function runKnowledgeSearch({ } : undefined, structuredFilters: structuredFilters.length > 0 ? structuredFilters : undefined, - searchIndexOnly: context.knowledgeBases.every((knowledgeBase) => knowledgeBase.isSearchIndex), + indexedRetrieval: usesIndexedRetrieval(context.knowledgeBases), }) ) diff --git a/apps/sim/lib/knowledge/application/sim-search.test.ts b/apps/sim/lib/knowledge/application/sim-search.test.ts index 8147185f2dc..b06c4ef7087 100644 --- a/apps/sim/lib/knowledge/application/sim-search.test.ts +++ b/apps/sim/lib/knowledge/application/sim-search.test.ts @@ -6,6 +6,7 @@ import { credentialGroupsServiceMock, credentialGroupsServiceMockFns, } from '@sim/testing/mocks/credential-groups-service.mock' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -35,7 +36,7 @@ import { permissionGroupsResolveMockFns, } from '@sim/testing/mocks/permission-groups-resolve.mock' import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' +import { afterAll, afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const hoisted = vi.hoisted(() => ({ createConnector: vi.fn(), @@ -129,6 +130,7 @@ import { prepareSearchSource, } from '@/lib/knowledge/application/sim-search' import { DEFAULT_PERMISSION_GROUP_CONFIG } from '@/lib/permission-groups/fields' +import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' const mocks = { ...hoisted, @@ -173,9 +175,12 @@ function queueConnectorLookups(...results: Array { afterAll(resetDbChainMock) + afterEach(resetEnvFlagsMock) beforeEach(() => { resetDbChainMock() + /** A Search source crawls into the search index, which only indexed organization search reads. */ + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue(workspaceContext) permissionGroupsResolveMockFns.mockGetUserPermissionConfig.mockResolvedValue( DEFAULT_PERMISSION_GROUP_CONFIG @@ -256,6 +261,22 @@ describe('connectSimSearchConnector', () => { expect(mocks.createKnowledgeBase).not.toHaveBeenCalled() }) + it('refuses while indexed organization search is dormant, before creating anything', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) + workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('admin') + queueConnectorLookups(null) + + await expect( + connectSimSearchConnector.execute({ + principal, + input: { workspaceId: 'workspace-1', connectorType: 'google_drive' }, + }) + ).rejects.toBeInstanceOf(SearchIndexDormantError) + expect(mocks.createKnowledgeBase).not.toHaveBeenCalled() + expect(mocks.createConnector).not.toHaveBeenCalled() + expect(mocks.enroll).not.toHaveBeenCalled() + }) + it('refuses before creating anything when per-member access is unavailable', async () => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('admin') knowledgeAvailabilityMockFns.mockIsKnowledgeMemberAccessAvailable.mockResolvedValue(false) @@ -337,8 +358,10 @@ describe('connectSimSearchConnector', () => { describe('organization Search setup', () => { const owner = { organizationId: 'org-1' } + afterEach(resetEnvFlagsMock) beforeEach(() => { resetDbChainMock() + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue(owner) knowledgeAvailabilityMockFns.mockIsKnowledgeMemberAccessAvailable.mockResolvedValue(true) mocks.ensureAccounts.mockResolvedValue({ id: 'org-accounts' }) diff --git a/apps/sim/lib/knowledge/application/sim-search.ts b/apps/sim/lib/knowledge/application/sim-search.ts index 94508a94d9b..75c10b00e4c 100644 --- a/apps/sim/lib/knowledge/application/sim-search.ts +++ b/apps/sim/lib/knowledge/application/sim-search.ts @@ -53,6 +53,7 @@ import { withSearchSourceDefaults, } from '@/lib/sim-search/connectors' import { SIM_SEARCH_SYNC_INTERVAL_MINUTES } from '@/lib/sim-search/constants' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { searchSourceIdentity } from '@/lib/sim-search/source-identity' import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' @@ -296,12 +297,15 @@ async function requireSimSearchSetupAdmin( * The database identifies one active search index per owner. Local * singleflight also coalesces repeated setup clicks for each source; concurrent * source creation is serialized by the connector insert transaction before enrollment. + * The source crawls into the owner's search index, so it is refused while indexed organization + * search is dormant. */ export const configureSimSearchConnector = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.simSearchConnect, resolveContext: ({ input }: { input: ConnectSimSearchConnectorInput }) => resolveKnowledgeOwnerContext(input), async execute({ principal, input, context, request }) { + assertIndexedOrgSearchEnabled() const meta = CONNECTOR_META_REGISTRY[input.connectorType] if (!meta || !canConnectPersonally(meta)) { throw new OrchestrationError( diff --git a/apps/sim/lib/knowledge/application/slack-search/assistant.test.ts b/apps/sim/lib/knowledge/application/slack-search/assistant.test.ts index 97e81c30872..7c131c04567 100644 --- a/apps/sim/lib/knowledge/application/slack-search/assistant.test.ts +++ b/apps/sim/lib/knowledge/application/slack-search/assistant.test.ts @@ -34,7 +34,6 @@ const hoisted = vi.hoisted(() => ({ outcome: vi.fn(), lease: vi.fn(), stopped: vi.fn(), - sources: vi.fn(), onboarding: vi.fn(), title: vi.fn(), stoppedMessage: vi.fn((message: unknown) => message), @@ -68,9 +67,6 @@ vi.mock('@/lib/knowledge/application/slack-search/turns', () => ({ requireSlackSearchTurnLease: hoisted.lease, wasSlackSearchTurnStopped: hoisted.stopped, })) -vi.mock('@/lib/knowledge/application/slack-search/source-status', () => ({ - getSlackSearchSourceStatus: { execute: hoisted.sources }, -})) vi.mock('@/lib/knowledge/application/slack-search/onboarding', () => ({ sendSlackSearchOnboarding: hoisted.onboarding, })) @@ -188,7 +184,6 @@ beforeEach(() => { m.run.mockResolvedValue({ success: true, content: 'Answer', contentBlocks: [], toolCalls: [] }) m.finalize.mockResolvedValue({ appendedAssistant: true }) m.stopped.mockResolvedValue(false) - m.sources.mockResolvedValue({ hasSearchableDocuments: true }) m.onboarding.mockResolvedValue({ text: 'Connect sources', url: 'https://sim.test/slack-search/connect/token', diff --git a/apps/sim/lib/knowledge/application/slack-search/onboarding.test.ts b/apps/sim/lib/knowledge/application/slack-search/onboarding.test.ts index 5653cec3b61..4340a8c4e50 100644 --- a/apps/sim/lib/knowledge/application/slack-search/onboarding.test.ts +++ b/apps/sim/lib/knowledge/application/slack-search/onboarding.test.ts @@ -15,7 +15,6 @@ const m = vi.hoisted(() => ({ authorize: vi.fn(), sender: vi.fn(), member: vi.fn(), - sources: vi.fn(), persist: vi.fn(), dispatch: vi.fn(), post: vi.fn(), @@ -52,9 +51,6 @@ vi.mock('@/lib/knowledge/application/slack-search/identity', () => ({ }, })) vi.mock('@/lib/core/application/organization-authorization', () => organizationAuthorizationMock) -vi.mock('@/lib/knowledge/application/slack-search/source-status', () => ({ - getSlackSearchSourceStatus: { execute: m.sources }, -})) vi.mock('@/lib/knowledge/application/slack-search/turns', () => ({ persistSlackSearchTurn: m.persist, requireSlackSearchTurnLease: m.lease, @@ -145,7 +141,6 @@ beforeEach(() => { organizationAuthorizationMockFns.mockAuthorizeOrganizationOperation.mockResolvedValue({ role: 'member', }) - m.sources.mockResolvedValue({ hasSearchableDocuments: true }) m.persist.mockResolvedValue('retry1') m.api.mockResolvedValue({ status: 200, data: { ok: true, permalink: state.slackUrl } }) m.post.mockResolvedValue({ status: 200, data: { ok: true } }) diff --git a/apps/sim/lib/knowledge/application/slack-search/source-status.ts b/apps/sim/lib/knowledge/application/slack-search/source-status.ts deleted file mode 100644 index d9a5e764f39..00000000000 --- a/apps/sim/lib/knowledge/application/slack-search/source-status.ts +++ /dev/null @@ -1,56 +0,0 @@ -import { db } from '@sim/db' -import { document, embedding, knowledgeBase } from '@sim/db/schema' -import { and, eq, exists, isNull } from 'drizzle-orm' -import type { OperationUseCase } from '@/lib/core/application/operation' -import { authorizeOrganizationOperation } from '@/lib/core/application/organization-authorization' -import { defineOrganizationOperation } from '@/lib/core/application/organization-operation' -import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' -import { knowledgeReadAccessBatches } from '@/lib/knowledge/read-access' - -const operation = defineOrganizationOperation({ - id: 'knowledge.slack.sources.status', - capability: 'knowledge.use', - minimumRole: 'member', - principalKinds: ['session', 'organization_delegated'], - delegatedServices: ['slack-search'], - delegationAudience: 'sim:knowledge', -}) - -/** Checks accessible indexed content, never treating another member's connection as the sender's. */ -export const getSlackSearchSourceStatus: OperationUseCase< - typeof operation, - { organizationId: string }, - { hasSearchableDocuments: boolean } -> = { - operation, - async execute({ principal, input }) { - await authorizeOrganizationOperation(principal, operation, input) - const access = createKnowledgeAccessProvider(principal, input) - const conditions = [ - eq(knowledgeBase.organizationId, input.organizationId), - eq(knowledgeBase.isSearchIndex, true), - isNull(knowledgeBase.deletedAt), - eq(document.processingStatus, 'completed'), - eq(document.enabled, true), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - exists( - db - .select({ id: embedding.id }) - .from(embedding) - .where(and(eq(embedding.documentId, document.id), eq(embedding.enabled, true))) - ), - ] - for await (const accessCondition of knowledgeReadAccessBatches(access, conditions)) { - const [visible] = await db - .select({ id: document.id }) - .from(document) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, document.knowledgeBaseId)) - .where(and(...conditions, accessCondition)) - .limit(1) - if (visible) return { hasSearchableDocuments: true } - } - return { hasSearchableDocuments: false } - }, -} diff --git a/apps/sim/lib/knowledge/connectors/external-group-sync.test.ts b/apps/sim/lib/knowledge/connectors/external-group-sync.test.ts index 7d9dc0a76a3..14495ddfcbf 100644 --- a/apps/sim/lib/knowledge/connectors/external-group-sync.test.ts +++ b/apps/sim/lib/knowledge/connectors/external-group-sync.test.ts @@ -4,7 +4,6 @@ import { resetDbChainMock, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { knowledgeAvailabilityMock, @@ -180,7 +179,6 @@ describe('refreshConnectorDirectory', () => { }) it('skips previously queued Search directory refreshes before resolving credentials in live mode', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(schemaMock.knowledgeConnector, [connectorRow({ isSearchIndex: true })]) await expect(refreshConnectorDirectory('connector-1', 'req-1')).resolves.toBe('skipped') diff --git a/apps/sim/lib/knowledge/connectors/indexing-policy.ts b/apps/sim/lib/knowledge/connectors/indexing-policy.ts index 37b680847a2..bbb61df298f 100644 --- a/apps/sim/lib/knowledge/connectors/indexing-policy.ts +++ b/apps/sim/lib/knowledge/connectors/indexing-policy.ts @@ -1,13 +1,13 @@ import { knowledgeBase } from '@sim/db/schema' import { eq } from 'drizzle-orm' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' /** Federated Search keeps source configuration but does not crawl content into a knowledge base. */ export function requiresConnectorIndexing(isSearchIndex?: boolean | null): boolean { - return !isLiveEnterpriseSearchEnabled || isSearchIndex !== true + return isIndexedOrgSearchEnabled() || isSearchIndex !== true } /** Keeps federated sources out of bounded indexing scheduler pages. */ export function connectorIndexingCondition() { - return isLiveEnterpriseSearchEnabled ? eq(knowledgeBase.isSearchIndex, false) : undefined + return isIndexedOrgSearchEnabled() ? undefined : eq(knowledgeBase.isSearchIndex, false) } diff --git a/apps/sim/lib/knowledge/connectors/member-observations.ts b/apps/sim/lib/knowledge/connectors/member-observations.ts index 28ec4ee92ac..3acdc5359e3 100644 --- a/apps/sim/lib/knowledge/connectors/member-observations.ts +++ b/apps/sim/lib/knowledge/connectors/member-observations.ts @@ -364,11 +364,8 @@ interface ConnectorDocumentCursor { * trigger. Every page holds at least one document, so one larger than the cap still makes progress * alone. Documents keep their order. * - * Cleanup boundary: the row bound, and {@link lockProjectionPage} and - * {@link writeProjectionPages} built on it, exist only because a synchronous ACL write rewrites - * projection rows in its own statement. With `knowledge-async-projection` on, an ACL write only - * marks its documents; once the flag has been on everywhere and the synchronous triggers are - * dropped, these collapse to plain pages of {@link ACL_CHANGE_BATCH_SIZE} documents. + * The row bound, and {@link lockProjectionPage} and {@link writeProjectionPages} built on it, + * exist because an ACL write rewrites the documents' projection rows in its own statement. */ export function pagesByProjectionRows( documents: readonly { id: string; chunkCount: number }[] diff --git a/apps/sim/lib/knowledge/connectors/member-queue.test.ts b/apps/sim/lib/knowledge/connectors/member-queue.test.ts index 01cbf775b18..b9af1468b4b 100644 --- a/apps/sim/lib/knowledge/connectors/member-queue.test.ts +++ b/apps/sim/lib/knowledge/connectors/member-queue.test.ts @@ -4,7 +4,6 @@ import { resetDbChainMock, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { asyncJobsRegionMock, @@ -76,7 +75,6 @@ describe('member sync queue', () => { }) it('does not dispatch a live Search source to member indexing', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(schemaMock.knowledgeConnector, [{ ...CONNECTOR_ROW, isSearchIndex: true }]) expect(await dispatchMemberSync('c-1', { billingAttribution: BILLING })).toEqual({ queued: false, diff --git a/apps/sim/lib/knowledge/connectors/member-sync-engine-content-credential.test.ts b/apps/sim/lib/knowledge/connectors/member-sync-engine-content-credential.test.ts index 9fbce36f005..c1b990c381f 100644 --- a/apps/sim/lib/knowledge/connectors/member-sync-engine-content-credential.test.ts +++ b/apps/sim/lib/knowledge/connectors/member-sync-engine-content-credential.test.ts @@ -4,7 +4,6 @@ import { resetDbChainMock, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { billingAttributionMock } from '@sim/testing/mocks/billing-attribution.mock' import { @@ -370,7 +369,6 @@ describe('member engine with a dedicated content credential', () => { }) it('refuses an already queued live Search crawl before resolving credentials or taking a lock', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) const result = await arrange({ isSearchIndex: true, members: true })() expect(result.skipReason).toBe('connector_not_syncable') expect(dbChainMockFns.update).not.toHaveBeenCalled() diff --git a/apps/sim/lib/knowledge/connectors/queue.test.ts b/apps/sim/lib/knowledge/connectors/queue.test.ts index 45d3a082c5e..44c35db48d8 100644 --- a/apps/sim/lib/knowledge/connectors/queue.test.ts +++ b/apps/sim/lib/knowledge/connectors/queue.test.ts @@ -6,7 +6,6 @@ import { resetDbChainMock, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { asyncJobsRegionMock, @@ -125,7 +124,6 @@ describe('connector sync queue', () => { ) it('does not dispatch a live Search source to Trigger or the inline indexer', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) resetDbChainMock() queueTableRows(schemaMock.knowledgeConnector, [{ isSearchIndex: true }]) expect(await dispatchSync('connector-1', { billingAttribution: BILLING_ATTRIBUTION })).toEqual({ diff --git a/apps/sim/lib/knowledge/connectors/sync-engine.test.ts b/apps/sim/lib/knowledge/connectors/sync-engine.test.ts index 3e62e04098e..45d6ba8eacc 100644 --- a/apps/sim/lib/knowledge/connectors/sync-engine.test.ts +++ b/apps/sim/lib/knowledge/connectors/sync-engine.test.ts @@ -9,7 +9,6 @@ import { resetDbChainMock as resetDatabaseMock, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { billingAttributionMock } from '@sim/testing/mocks/billing-attribution.mock' import { @@ -245,7 +244,6 @@ describe('connector content replacement processing state', () => { }) it('refuses a queued live Search source before locking or indexing content', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(schemaMock.knowledgeConnector, [CONNECTOR]) queueTableRows(schemaMock.knowledgeBase, [{ isSearchIndex: true }]) const result = await executeSync('connector-1', { diff --git a/apps/sim/lib/knowledge/connectors/sync-engine.ts b/apps/sim/lib/knowledge/connectors/sync-engine.ts index 731c2210186..96fac9f029e 100644 --- a/apps/sim/lib/knowledge/connectors/sync-engine.ts +++ b/apps/sim/lib/knowledge/connectors/sync-engine.ts @@ -7,7 +7,6 @@ import { } from '@sim/db/schema' import { createLogger } from '@sim/logger' import { - getErrorMessage, getTransientDatabaseFailure, type TransientDatabaseFailureClass, toError, @@ -96,7 +95,6 @@ import { } from '@/lib/knowledge/connectors/sync-primitives' import { hardDeleteDocuments } from '@/lib/knowledge/documents/service' import { getRetryAfterMs, isRateLimitError } from '@/lib/knowledge/documents/utils' -import { ensureSourceVectorIndex } from '@/lib/knowledge/search/source-vector-indexes' import { getCredentialTerminalRefreshError } from '@/lib/oauth/credential-service' import { isCredentialRevocationError } from '@/lib/oauth/terminal-errors' import { connectorHasAuthSource } from '@/connectors/auth' @@ -444,7 +442,7 @@ export async function completeSuccessfulSync( return null }) try { - const completed = await db.transaction(async (tx) => { + return await db.transaction(async (tx) => { const [lockedKnowledgeBase] = await tx .select({ id: knowledgeBase.id }) .from(knowledgeBase) @@ -536,18 +534,6 @@ export async function completeSuccessfulSync( return true }) - /** - * A source that has grown past the threshold gets its own vector index, so a member who reads - * it whole is ranked through a walk of their own documents. Retrieval ranks exactly without - * it, so a failure here is logged and left for the next sync. - */ - await ensureSourceVectorIndex(connectorId).catch((error: unknown) => { - logger.warn('Could not ensure the source vector index', { - connectorId, - error: getErrorMessage(error), - }) - }) - return completed } catch (error) { if (error instanceof SyncCompletionOwnershipLost) return false throw error diff --git a/apps/sim/lib/knowledge/connectors/sync-limits.ts b/apps/sim/lib/knowledge/connectors/sync-limits.ts index 4c1adba9d35..23e64b7d122 100644 --- a/apps/sim/lib/knowledge/connectors/sync-limits.ts +++ b/apps/sim/lib/knowledge/connectors/sync-limits.ts @@ -179,9 +179,7 @@ export const LEASE_PAGE_STATEMENT_TIMEOUT_MS = 30_000 * document trigger that copies it onto every chunk's search projection rows, and * each of those rows is re-inserted into the vector index, so one statement costs * the chunks of every document in it rather than the documents. Kept small so a - * page of changed documents cannot outrun the statement timeout; with - * `knowledge-async-projection` on, the trigger only marks the documents and the bound is the - * cleanup boundary noted at `pagesByProjectionRows` in `member-observations.ts`. Also the page + * page of changed documents cannot outrun the statement timeout. Also the page * of the transactions that remove observations and rematerialise the ACLs they * decide together, which must commit as one and so cannot be split by rows. */ diff --git a/apps/sim/lib/knowledge/connectors/sync-lock.ts b/apps/sim/lib/knowledge/connectors/sync-lock.ts index cefb67b1544..3c7b198f5d8 100644 --- a/apps/sim/lib/knowledge/connectors/sync-lock.ts +++ b/apps/sim/lib/knowledge/connectors/sync-lock.ts @@ -1,15 +1,12 @@ import { db } from '@sim/db' -import { DEFER_KNOWLEDGE_PROJECTION } from '@sim/db/knowledge-projection' import { knowledgeConnector } from '@sim/db/schema' import { and, eq, isNull, sql } from 'drizzle-orm' -import { isFeatureEnabled } from '@/lib/core/config/feature-flags' import type { DbOrTx } from '@/lib/db/types' import { LEASE_PAGE_LOCK_TIMEOUT_MS, LEASE_PAGE_STATEMENT_TIMEOUT_MS, SYNC_LOCK_HEARTBEAT_INTERVAL_MS, } from '@/lib/knowledge/connectors/sync-limits' -import { requestKnowledgeProjection } from '@/lib/knowledge/projection/enqueue' /** * Raised when a run discovers mid-flight that it no longer holds its sync lock. @@ -246,27 +243,20 @@ export async function assertSyncLeaseHeldInTx( /** * Runs `write` as one connector-lease ACL page: a short transaction of its own whose first * statement sets its bounds, so a page that waits on a lock or runs long fails within them and - * rolls back only itself, and its projection mode. While `knowledge-async-projection` is on, the - * page's ACL writes only mark their documents and the knowledge projector rewrites their search - * projection rows; off, the `document` ACL trigger still rewrites every filled row in the page's - * own statement, which is what the row-bounded paging of these pages exists for. The flag is read - * before the transaction opens, and a projector pass is requested once it commits. Every page that - * assigns a connector document's ACL runs here, so this is the one place those writers choose the - * mode. + * rolls back only itself. The `document` ACL trigger rewrites every filled search projection row of + * the page's documents in the page's own statement, which is what the row-bounded paging of these + * pages exists for. Every page that assigns a connector document's ACL runs here. */ export async function aclPageTransaction( write: (tx: DbOrTx) => Promise, executor: Pick = db ): Promise { - const deferProjection = await isFeatureEnabled('knowledge-async-projection') - const written = await executor.transaction(async (tx) => { + return executor.transaction(async (tx) => { await tx.execute( - sql`SELECT set_config('lock_timeout', ${`${LEASE_PAGE_LOCK_TIMEOUT_MS}ms`}, true), set_config('statement_timeout', ${`${LEASE_PAGE_STATEMENT_TIMEOUT_MS}ms`}, true)${deferProjection ? sql.raw(`, ${DEFER_KNOWLEDGE_PROJECTION}`) : sql``}` + sql`SELECT set_config('lock_timeout', ${`${LEASE_PAGE_LOCK_TIMEOUT_MS}ms`}, true), set_config('statement_timeout', ${`${LEASE_PAGE_STATEMENT_TIMEOUT_MS}ms`}, true)` ) return write(tx) }) - await requestKnowledgeProjection() - return written } /** Runs one bounded page of writes in a short transaction of its own. */ diff --git a/apps/sim/lib/knowledge/connectors/user-document-visibility.ts b/apps/sim/lib/knowledge/connectors/user-document-visibility.ts index f2034965306..21678885616 100644 --- a/apps/sim/lib/knowledge/connectors/user-document-visibility.ts +++ b/apps/sim/lib/knowledge/connectors/user-document-visibility.ts @@ -1,7 +1,7 @@ import { db } from '@sim/db' import { document } from '@sim/db/schema' import { sql } from 'drizzle-orm' -import { SOURCE_ACL_MAX_AGE_MS } from '@/lib/knowledge/access/freshness' +import { sourceAclFreshnessCutoff } from '@/lib/knowledge/access/predicate' import { userToken } from '@/lib/knowledge/access/tokens' /** @@ -27,7 +27,7 @@ export async function hasVisibleUserDocuments( WHERE ${document.connectorId} = ${connectorId} AND ${document.userExcluded} = false AND ${document.archivedAt} IS NULL - AND ${document.aclVerifiedAt} > statement_timestamp() - (${SOURCE_ACL_MAX_AGE_MS} * interval '1 millisecond') + AND ${document.aclVerifiedAt} > ${sourceAclFreshnessCutoff()} LIMIT 1 `) return rows.length > 0 diff --git a/apps/sim/lib/knowledge/documents/processing-payload.ts b/apps/sim/lib/knowledge/documents/processing-payload.ts index 8cd6a4a797b..f5dd1466a84 100644 --- a/apps/sim/lib/knowledge/documents/processing-payload.ts +++ b/apps/sim/lib/knowledge/documents/processing-payload.ts @@ -199,20 +199,6 @@ export function createWorkspaceDocumentProcessingBillingContext( } } -export function createNonWorkspaceDocumentProcessingBillingContext( - actorUserId: string -): NonWorkspaceDocumentProcessingBillingContext { - const billingContext = assertDocumentProcessingBillingContext({ - billingScope: 'non-workspace', - actorUserId, - workspaceId: null, - }) - if (billingContext.billingScope !== 'non-workspace') { - throw new Error('Non-workspace document processing context could not be created') - } - return billingContext -} - /** Identifies a durable handoff independently of its original indexing pass and scheduled time. */ export function createDocumentProcessingContinuationToken( payload: Pick, diff --git a/apps/sim/lib/knowledge/documents/processing-recovery.ts b/apps/sim/lib/knowledge/documents/processing-recovery.ts index 08c9a18e087..836d63e5835 100644 --- a/apps/sim/lib/knowledge/documents/processing-recovery.ts +++ b/apps/sim/lib/knowledge/documents/processing-recovery.ts @@ -11,6 +11,7 @@ import { import { enqueueOutboxEvent } from '@/lib/core/outbox/service' import { withinDeadline } from '@/lib/core/utils/deadline' import { getConnectorFailureDiagnostic } from '@/lib/knowledge/connectors/connector-error' +import { connectorIndexingCondition } from '@/lib/knowledge/connectors/indexing-policy' import { createDocumentProcessingPayload, createOrganizationDocumentProcessingBillingContext, @@ -35,7 +36,8 @@ const RECOVERABLE_CONNECTOR_STATUSES = ['active', 'error', 'pending', 'syncing'] /** * Re-admits bounded, abandoned connector documents from our retained bytes, independently * of source sync schedules and credentials. The generation, attempt and outbox event commit - * together; no source-provider call or source lease is needed. Paused/deleted sources stay paused. + * together; no source-provider call or source lease is needed. Paused/deleted sources stay paused, + * and search-index knowledge bases are not indexed while indexed organization search is dormant. */ export async function recoverKnowledgeDocumentProcessing(now = new Date()): Promise { const deadlineAt = Date.now() + RECOVERY_RUNTIME_MS @@ -102,6 +104,7 @@ async function recoverStoredDocumentBatch( ? notInArray(document.connectorId, [...attemptedConnectors]) : undefined, isNull(knowledgeBase.deletedAt), + connectorIndexingCondition(), isNull(knowledgeConnector.deletedAt), isNull(knowledgeConnector.archivedAt), inArray(knowledgeConnector.status, RECOVERABLE_CONNECTOR_STATUSES) @@ -151,7 +154,13 @@ async function recoverStoredDocumentBatch( organizationId: knowledgeBase.organizationId, }) .from(knowledgeBase) - .where(and(eq(knowledgeBase.id, knowledgeBaseId), isNull(knowledgeBase.deletedAt))) + .where( + and( + eq(knowledgeBase.id, knowledgeBaseId), + isNull(knowledgeBase.deletedAt), + connectorIndexingCondition() + ) + ) .for('share', { skipLocked: true }) signal.throwIfAborted() if (!kb) { diff --git a/apps/sim/lib/knowledge/documents/service.ts b/apps/sim/lib/knowledge/documents/service.ts index e6d16a5ba53..6373e269d69 100644 --- a/apps/sim/lib/knowledge/documents/service.ts +++ b/apps/sim/lib/knowledge/documents/service.ts @@ -1,5 +1,4 @@ import { db } from '@sim/db' -import { DEFER_KNOWLEDGE_PROJECTION } from '@sim/db/knowledge-projection' import { document, documentSecretProvenance, @@ -53,7 +52,6 @@ import type { ChunkingStrategy, StrategyOptions } from '@/lib/chunkers/types' import { resolveTriggerRegion } from '@/lib/core/async-jobs/region' import { env, envNumber } from '@/lib/core/config/env' import { getCostMultiplier } from '@/lib/core/config/env-flags' -import { isFeatureEnabled } from '@/lib/core/config/feature-flags' import { isTriggerAvailable } from '@/lib/core/config/trigger-availability' import { withResourceOutboundScope } from '@/lib/core/network/resource-scope.server' import { OrchestrationError } from '@/lib/core/orchestration/types' @@ -166,7 +164,6 @@ import { } from '@/lib/knowledge/embedding-models' import { generateEmbeddings, type KbEmbeddingTarget } from '@/lib/knowledge/embeddings' import { runWithKnowledgeModelInputProvenance } from '@/lib/knowledge/model-input-provenance' -import { requestKnowledgeProjection } from '@/lib/knowledge/projection/enqueue' import { type KnowledgeReadAccess, knowledgeReadAccessBatches } from '@/lib/knowledge/read-access' import { bindKnowledgeDocumentFieldSecretProvenance, @@ -2011,17 +2008,9 @@ export async function processDocumentAsync( ) .limit(1) if (!sourceActive) return - /** - * While `knowledge-async-projection` is on, the chunks are committed with a mark on - * their document and no search projection rows; the knowledge projector writes those - * after the commit. Read before the transaction opens. - */ - const deferProjection = await isFeatureEnabled('knowledge-async-projection') processingCommitted = await db .transaction(async (tx) => { signal.throwIfAborted() - if (deferProjection) - await tx.execute(sql.raw(`SELECT ${DEFER_KNOWLEDGE_PROJECTION}`)) /** * Reads only the document row. Connector activity is checked by * the completion write at the end instead: reading @@ -2157,8 +2146,6 @@ export async function processDocumentAsync( logger.info(`[${documentId}] Discarded output from an obsolete processing attempt`) return { outcome: 'skipped', reason: 'superseded' } } - /** The commit marked the document in either mode; a pass writes or verifies its rows. */ - await requestKnowledgeProjection() const processingTime = Date.now() - startTime logger.info(`[${documentId}] Successfully processed document in ${processingTime}ms`) diff --git a/apps/sim/lib/knowledge/embeddings.test.ts b/apps/sim/lib/knowledge/embeddings.test.ts index 2d3e7915e73..4acea7d8ccb 100644 --- a/apps/sim/lib/knowledge/embeddings.test.ts +++ b/apps/sim/lib/knowledge/embeddings.test.ts @@ -1,12 +1,34 @@ -import { afterAll, beforeEach, describe, expect, it, type MockInstance, vi } from 'vitest' +import { setupGlobalFetchMock } from '@sim/testing/mocks' +import { mockEnvObject, resetEnvMock } from '@sim/testing/mocks/env.mock' +import { + afterAll, + afterEach, + beforeEach, + describe, + expect, + it, + type MockInstance, + vi, +} from 'vitest' import * as billingAttributionModule from '@/lib/billing/core/billing-attribution' import * as usageLogModule from '@/lib/billing/core/usage-log' import * as thresholdBillingModule from '@/lib/billing/threshold-billing' import * as embeddingModelsModule from '@/lib/knowledge/embedding-models' -import { recordSearchEmbeddingUsage } from '@/lib/knowledge/embeddings' +import { generateSearchEmbedding, recordSearchEmbeddingUsage } from '@/lib/knowledge/embeddings' +import { runWithKnowledgeModelInputProvenance } from '@/lib/knowledge/model-input-provenance' import * as tokenizationModule from '@/lib/tokenization' +import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' import * as providersUtilsModule from '@/providers/utils' +vi.mock('@/lib/core/rate-limiter/provider-admission', () => ({ + PROVIDER_QUOTA_COOLDOWN_MS: 300_000, + ProviderQuotaExhaustedError: class ProviderQuotaExhaustedError extends Error {}, + ProviderAdmissionTimeoutError: class ProviderAdmissionTimeoutError extends Error {}, + isProviderQuotaExhausted: vi.fn().mockResolvedValue(false), + recordProviderCooldown: vi.fn().mockResolvedValue(undefined), + waitForProviderAdmission: vi.fn().mockResolvedValue(undefined), +})) + /** * Spy on the real module namespaces instead of vi.mock: under `isolate: false` * `@/lib/knowledge/embeddings` is a shared consumer cached across test files, @@ -97,3 +119,58 @@ describe('recordSearchEmbeddingUsage', () => { }) }) }) + +describe('generateSearchEmbedding', () => { + const target = { model: 'text-embedding-3-small', dimensions: 1536 } as const + const unconfigured = { + AZURE_OPENAI_API_KEY: undefined, + OPENAI_API_KEY: undefined, + OPENAI_API_KEY_1: undefined, + OPENAI_API_KEY_2: undefined, + OPENAI_API_KEY_3: undefined, + OPENROUTER_API_KEY: undefined, + } + + beforeEach(() => { + setupGlobalFetchMock({ json: {} }) + Object.assign(mockEnvObject, unconfigured) + }) + + afterEach(() => { + resetEnvMock() + }) + + it('projects verified provenance only in the model-bound embedding payload', async () => { + mockEnvObject.OPENAI_API_KEY = 'test-openai-key' + const embedding = Buffer.from(new Float32Array(1536).buffer).toString('base64') + vi.mocked(fetch).mockResolvedValueOnce( + new Response( + JSON.stringify({ + data: [{ embedding, index: 0 }], + usage: { prompt_tokens: 1, total_tokens: 1 }, + }), + { status: 200, headers: { 'Content-Type': 'application/json' } } + ) + ) + const registry = new ResolvedSecretTraceRegistry([ + { name: 'TOKEN', plaintext: 'secret-value', encryptedValue: 'encrypted-token' }, + ]) + registry.recordResolved('TOKEN', 'secret-value') + + await runWithKnowledgeModelInputProvenance(registry, () => + generateSearchEmbedding('prefix secret-value suffix', target) + ) + + expect(vi.mocked(fetch)).toHaveBeenCalledWith( + 'https://api.openai.com/v1/embeddings', + expect.objectContaining({ + body: JSON.stringify({ + input: ['prefix {{TOKEN}} suffix'], + model: 'text-embedding-3-small', + encoding_format: 'base64', + dimensions: 1536, + }), + }) + ) + }) +}) diff --git a/apps/sim/lib/knowledge/mcp/server.protocol.test.ts b/apps/sim/lib/knowledge/mcp/server.protocol.test.ts index ae3db927117..a0ea9244d10 100644 --- a/apps/sim/lib/knowledge/mcp/server.protocol.test.ts +++ b/apps/sim/lib/knowledge/mcp/server.protocol.test.ts @@ -21,9 +21,13 @@ vi.mock('@/lib/api/server/routes/v2-json-route', () => ({ v2RateLimits: { publicApi: { enforce: vi.fn().mockResolvedValue(null) } }, })) vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) -vi.mock('@/lib/knowledge/application/read-indexed-document', () => ({ +vi.mock('@/lib/sim-search/indexed/documents/read-indexed-document', () => ({ readIndexedKnowledgeDocument: { execute: hoisted.indexedRead }, })) +vi.mock('@/lib/sim-search/indexed', async () => ({ + registerIndexedKnowledgeMcpTools: (await import('@/lib/sim-search/indexed/mcp/register-tools')) + .registerIndexedKnowledgeMcpTools, +})) vi.mock('@/lib/sim-search/live/application', () => ({ searchLiveKnowledge: { execute: hoisted.liveSearch }, readLiveDocument: { execute: hoisted.liveRead }, diff --git a/apps/sim/lib/knowledge/mcp/server.test.ts b/apps/sim/lib/knowledge/mcp/server.test.ts index 6d0e710cbbe..aca1577fd0d 100644 --- a/apps/sim/lib/knowledge/mcp/server.test.ts +++ b/apps/sim/lib/knowledge/mcp/server.test.ts @@ -50,9 +50,13 @@ vi.mock('@/lib/api/server/routes/v2-json-route', () => ({ v2RateLimits: { publicApi: { enforce: hoisted.rateLimit } }, })) vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) -vi.mock('@/lib/knowledge/application/read-indexed-document', () => ({ +vi.mock('@/lib/sim-search/indexed/documents/read-indexed-document', () => ({ readIndexedKnowledgeDocument: { execute: hoisted.read }, })) +vi.mock('@/lib/sim-search/indexed', async () => ({ + registerIndexedKnowledgeMcpTools: (await import('@/lib/sim-search/indexed/mcp/register-tools')) + .registerIndexedKnowledgeMcpTools, +})) vi.mock('@/lib/sim-search/live/application', () => ({ searchLiveKnowledge: { execute: hoisted.liveSearch }, readLiveDocument: { execute: hoisted.liveRead }, diff --git a/apps/sim/lib/knowledge/mcp/server.ts b/apps/sim/lib/knowledge/mcp/server.ts index 818472e2dca..e3a9c600186 100644 --- a/apps/sim/lib/knowledge/mcp/server.ts +++ b/apps/sim/lib/knowledge/mcp/server.ts @@ -1,5 +1,4 @@ import { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js' -import type { CallToolResult } from '@modelcontextprotocol/sdk/types.js' import { resolvePrincipalSubjectUserId } from '@sim/auth/principal' import { createLogger } from '@sim/logger' import { isPlainRecord } from '@sim/utils/object' @@ -8,41 +7,31 @@ import type { NextRequest } from 'next/server' import { chatSearchMcpSchema, liveSearchMcpSchema, - readDocumentMcpSchema, readLiveDocumentMcpSchema, - searchMcpSchema, } from '@/lib/api/contracts/knowledge/mcp' import type { V2ApiKeyAuthContext } from '@/lib/api/server/routes/v2-api-key-auth' import { v2RateLimits } from '@/lib/api/server/routes/v2-json-route' -import type { ApplicationOperation } from '@/lib/core/application' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' -import type { ResourceScope } from '@/lib/core/resource-scope' import { afterResponse } from '@/lib/core/utils/after-response' -import { getBaseUrl } from '@/lib/core/utils/urls' import { organizationSearchChatOperation } from '@/lib/knowledge/application/chat-operations' import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { readIndexedKnowledgeDocument } from '@/lib/knowledge/application/read-indexed-document' -import { searchKnowledge } from '@/lib/knowledge/application/search' import { recordOrganizationSearchMcpActivity, type SearchMcpActivityInput, } from '@/lib/knowledge/mcp/activity' -import { createKnowledgeDocumentCitation, liveCitationId } from '@/lib/knowledge/search/citation' +import { + KNOWLEDGE_MCP_READ_ONLY, + type KnowledgeMcpToolRunner, + projectResult, +} from '@/lib/knowledge/mcp/tool-runner' +import { liveCitationId } from '@/lib/knowledge/search/citation' import { toolError } from '@/lib/mcp/tool-result' +import { registerIndexedKnowledgeMcpTools } from '@/lib/sim-search/indexed' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { readLiveDocument, searchLiveKnowledge } from '@/lib/sim-search/live/application' import { v2CaughtOrchestrationError } from '@/app/api/v2/lib/response' -import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' const logger = createLogger('KnowledgeMcp') -const MAX_RESULT_BYTES = 1024 * 1024 -const READ_ONLY = { - readOnlyHint: true, - destructiveHint: false, - idempotentHint: true, - openWorldHint: false, -} as const - interface KnowledgeMcpContext { organizationId: string request: NextRequest @@ -50,36 +39,13 @@ interface KnowledgeMcpContext { searchIndexId: string | null } -function projectResult(value: unknown, registry: ResolvedSecretTraceRegistry): CallToolResult { - if (!registry.isComplete()) { - return toolError( - 'Document secret provenance is unavailable. The content cannot be returned safely.' - ) - } - const projected = projectResolvedSecretModelContent(value, registry, MAX_RESULT_BYTES) - if (!projected.safe) { - return toolError('This result cannot be safely returned. Try a smaller result page.') - } - const text = JSON.stringify(projected.value) - if (Buffer.byteLength(text) > MAX_RESULT_BYTES) { - return toolError('Result is too large. Request fewer results or a smaller page.') - } - return { content: [{ type: 'text', text }] } -} - /** A request owns its server; no credential or principal survives into another HTTP request. */ export function createKnowledgeMcpServer(context: KnowledgeMcpContext): McpServer { const { request, auth, searchIndexId, organizationId } = context - const scope: ResourceScope = { kind: 'organization', organizationId } const principal = auth.principal const server = new McpServer({ name: 'Sim Search', version: '1.0.0' }) - async function execute( - toolName: SearchMcpActivityInput['toolName'], - operation: ApplicationOperation, - toolSignal: AbortSignal, - run: (registry: ResolvedSecretTraceRegistry, signal: AbortSignal) => Promise - ): Promise { + const execute: KnowledgeMcpToolRunner = async (toolName, operation, toolSignal, run) => { const startedAt = performance.now() const signal = AbortSignal.any([request.signal, toolSignal]) let outcome: SearchMcpActivityInput['outcome'] = 'error' @@ -141,19 +107,27 @@ export function createKnowledgeMcpServer(context: KnowledgeMcpContext): McpServe } } - server.registerTool( - 'search', - { - title: 'Search', - description: isLiveEnterpriseSearchEnabled - ? 'Search this organization’s sources through their live APIs, within your access and the admin’s source settings. Use source and date filters to narrow results, or nativeQueries for provider queries and pagination. Inspect live.accounts for provider status and continuation cursors, and live.guidance for query syntax. Results are candidates, not proof of complete coverage. Use read_document with the exact returned documentId for context and cite citationUrl when available.' - : 'Search accessible passages in this organization’s Search index. Use source (for example, jira), modifiedAfter (an ISO timestamp), or documentIds to narrow results. Results are candidates; score is similarity, not answer confidence. Use read_document for context and cite citationUrl.', - inputSchema: isLiveEnterpriseSearchEnabled ? liveSearchMcpSchema : searchMcpSchema, - annotations: READ_ONLY, - }, - async (input: unknown, extra: { signal: AbortSignal }) => - execute('search', knowledgeOperations.search, extra.signal, async (registry, signal) => { - if (isLiveEnterpriseSearchEnabled) { + if (isIndexedOrgSearchEnabled()) { + registerIndexedKnowledgeMcpTools({ + server, + principal, + request, + organizationId, + searchIndexId, + execute, + }) + } else { + server.registerTool( + 'search', + { + title: 'Search', + description: + 'Search this organization’s sources through their live APIs, within your access and the admin’s source settings. Use source and date filters to narrow results, or nativeQueries for provider queries and pagination. Inspect live.accounts for provider status and continuation cursors, and live.guidance for query syntax. Results are candidates, not proof of complete coverage. Use read_document with the exact returned documentId for context and cite citationUrl when available.', + inputSchema: liveSearchMcpSchema, + annotations: KNOWLEDGE_MCP_READ_ONLY, + }, + async (input: unknown, extra: { signal: AbortSignal }) => + execute('search', knowledgeOperations.search, extra.signal, async (registry, signal) => { const { query, topK, nativeQueries, ...filters } = liveSearchMcpSchema.parse(input) const result = await searchLiveKnowledge.execute({ principal, @@ -189,74 +163,23 @@ export function createKnowledgeMcpServer(context: KnowledgeMcpContext): McpServe }, registry ) - } - const { query, topK, ...filters } = searchMcpSchema.parse(input) - if (!searchIndexId) { - return projectResult( - { - results: [], - message: 'No Search index is configured. Ask an admin to connect a source.', - }, - registry - ) - } - const result = await searchKnowledge.execute({ - principal, - input: { - organizationId, - knowledgeBaseIds: [searchIndexId], - query, - topK, - filters, - resultSecretRegistry: registry, - surface: 'mcp', - signal, - }, - request, }) - return projectResult( - { - results: result.results.map((row) => ({ - documentId: row.documentId, - title: row.documentName, - sourceUrl: row.sourceUrl, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: row.knowledgeBaseId, - documentId: row.documentId, - sourceUrl: row.sourceUrl, - baseUrl: getBaseUrl(), - }), - sourceModifiedAt: row.sourceModifiedAt?.toISOString() ?? null, - connectorType: row.connectorType, - content: row.content, - chunkIndex: row.chunkIndex, - score: row.similarity, - })), - }, - result.resultSecretRegistry ?? registry - ) - }) - ) - server.registerTool( - 'read_document', - { - title: 'Read document', - description: isLiveEnterpriseSearchEnabled - ? 'Read a live document using the exact documentId returned by search. Access and the admin’s source settings are checked again on every read. When hasMore is true, pass next.startChunkIndex and next.startOffset with the same documentId to continue. Cite citationUrl when available.' - : 'Read an indexed document by documentId from search or its original URL. URLs must match an accessible indexed source; this tool does not browse the web. Set aroundChunkIndex to a search hit’s chunkIndex for nearby context, or use offset for sequential pages. When pagination.hasMore is true, continue with pagination.offset + pagination.limit. Cite citationUrl. Documents still indexing return metadata only.', - inputSchema: isLiveEnterpriseSearchEnabled - ? readLiveDocumentMcpSchema - : readDocumentMcpSchema, - annotations: READ_ONLY, - }, - async (raw: unknown, extra: { signal: AbortSignal }) => - execute( - 'read_document', - knowledgeOperations.readDocument, - extra.signal, - async (registry, signal) => { - if (isLiveEnterpriseSearchEnabled) { + ) + server.registerTool( + 'read_document', + { + title: 'Read document', + description: + 'Read a live document using the exact documentId returned by search. Access and the admin’s source settings are checked again on every read. When hasMore is true, pass next.startChunkIndex and next.startOffset with the same documentId to continue. Cite citationUrl when available.', + inputSchema: readLiveDocumentMcpSchema, + annotations: KNOWLEDGE_MCP_READ_ONLY, + }, + async (raw: unknown, extra: { signal: AbortSignal }) => + execute( + 'read_document', + knowledgeOperations.readDocument, + extra.signal, + async (registry, signal) => { const input = readLiveDocumentMcpSchema.parse(raw) const result = await readLiveDocument.execute({ principal, @@ -278,40 +201,9 @@ export function createKnowledgeMcpServer(context: KnowledgeMcpContext): McpServe registry ) } - const input = readDocumentMcpSchema.parse(raw) - if (!input.url && !input.documentId) return toolError('Document not found') - const result = await readIndexedKnowledgeDocument.execute({ - principal, - input: { - organizationId, - target: input.url - ? { kind: 'url', url: input.url } - : { kind: 'id', documentId: input.documentId! }, - limit: input.limit, - offset: input.offset, - aroundChunkIndex: input.aroundChunkIndex, - resultSecretRegistry: registry, - signal, - }, - request, - }) - const { knowledgeBaseId, ...document } = result - return projectResult( - { - ...document, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId, - documentId: result.documentId, - sourceUrl: result.sourceUrl, - baseUrl: getBaseUrl(), - }), - }, - registry - ) - } - ) - ) + ) + ) + } server.registerTool( 'chat', diff --git a/apps/sim/lib/knowledge/mcp/tool-runner.ts b/apps/sim/lib/knowledge/mcp/tool-runner.ts new file mode 100644 index 00000000000..8b03011a7fa --- /dev/null +++ b/apps/sim/lib/knowledge/mcp/tool-runner.ts @@ -0,0 +1,47 @@ +import type { CallToolResult } from '@modelcontextprotocol/sdk/types.js' +import type { ApplicationOperation } from '@/lib/core/application' +import type { SearchMcpActivityInput } from '@/lib/knowledge/mcp/activity' +import { toolError } from '@/lib/mcp/tool-result' +import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' +import type { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' + +const MAX_RESULT_BYTES = 1024 * 1024 +/** The annotations of every read-only Search MCP tool. */ +export const KNOWLEDGE_MCP_READ_ONLY = { + readOnlyHint: true, + destructiveHint: false, + idempotentHint: true, + openWorldHint: false, +} as const + +/** + * Runs one Search MCP tool call: rate limit, cancellation, error projection and activity. Handed + * to every tool registration so each backend's tools share it. + */ +export type KnowledgeMcpToolRunner = ( + toolName: SearchMcpActivityInput['toolName'], + operation: ApplicationOperation, + toolSignal: AbortSignal, + run: (registry: ResolvedSecretTraceRegistry, signal: AbortSignal) => Promise +) => Promise + +/** A tool result projected for its resolved secrets and bounded in size. */ +export function projectResult( + value: unknown, + registry: ResolvedSecretTraceRegistry +): CallToolResult { + if (!registry.isComplete()) { + return toolError( + 'Document secret provenance is unavailable. The content cannot be returned safely.' + ) + } + const projected = projectResolvedSecretModelContent(value, registry, MAX_RESULT_BYTES) + if (!projected.safe) { + return toolError('This result cannot be safely returned. Try a smaller result page.') + } + const text = JSON.stringify(projected.value) + if (Buffer.byteLength(text) > MAX_RESULT_BYTES) { + return toolError('Result is too large. Request fewer results or a smaller page.') + } + return { content: [{ type: 'text', text }] } +} diff --git a/apps/sim/lib/knowledge/orchestration/connector-access.test.ts b/apps/sim/lib/knowledge/orchestration/connector-access.test.ts index 0006716b27f..164e5c9de0a 100644 --- a/apps/sim/lib/knowledge/orchestration/connector-access.test.ts +++ b/apps/sim/lib/knowledge/orchestration/connector-access.test.ts @@ -9,7 +9,6 @@ import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, } from '@sim/testing/mocks/knowledge-availability.mock' -import { knowledgeDocumentsServiceMock } from '@sim/testing/mocks/knowledge-documents-service.mock' import { knowledgeMemberAccessMock, knowledgeMemberAccessMockFns, @@ -41,7 +40,6 @@ vi.mock('@/lib/knowledge/connectors/member-observations', () => ({ vi.mock('@sim/audit', () => auditMock) vi.mock('@/lib/api-key/crypto', () => ({ encryptApiKey: vi.fn() })) vi.mock('@/lib/billing/core/subscription', () => billingSubscriptionMock) -vi.mock('@/lib/knowledge/documents/service', () => knowledgeDocumentsServiceMock) vi.mock('@/lib/knowledge/tags/service', () => knowledgeTagsServiceMock) vi.mock('@/lib/posthog/server', () => posthogServerMock) vi.mock('@/lib/knowledge/connectors/member-access', () => knowledgeMemberAccessMock) diff --git a/apps/sim/lib/knowledge/orchestration/connectors.test.ts b/apps/sim/lib/knowledge/orchestration/connectors.test.ts index 9a3c7f54768..05efa5d5d21 100644 --- a/apps/sim/lib/knowledge/orchestration/connectors.test.ts +++ b/apps/sim/lib/knowledge/orchestration/connectors.test.ts @@ -7,7 +7,6 @@ import { resetDbChainMock, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { auditMock, auditMockFns } from '@sim/testing/mocks/audit.mock' import { billingStorageMock, billingStorageMockFns } from '@sim/testing/mocks/billing-storage.mock' @@ -15,10 +14,6 @@ import { billingSubscriptionMock, billingSubscriptionMockFns, } from '@sim/testing/mocks/billing-subscription.mock' -import { - knowledgeDocumentsServiceMock, - knowledgeDocumentsServiceMockFns, -} from '@sim/testing/mocks/knowledge-documents-service.mock' import { knowledgeMemberAccessMock, knowledgeMemberAccessMockFns, @@ -66,7 +61,6 @@ vi.mock('@/lib/knowledge/connectors/detachment', () => ({ vi.mock('@/lib/knowledge/connectors/queue', () => ({ dispatchSync: mockDispatchSync })) vi.mock('@/lib/knowledge/connectors/member-queue', () => knowledgeMemberQueueMock) vi.mock('@/lib/knowledge/connectors/member-access', () => knowledgeMemberAccessMock) -vi.mock('@/lib/knowledge/documents/service', () => knowledgeDocumentsServiceMock) vi.mock('@/lib/knowledge/tags/service', () => knowledgeTagsServiceMock) vi.mock('@/lib/posthog/server', () => posthogServerMock) vi.mock('@/connectors/registry.server', () => ({ @@ -126,7 +120,6 @@ const mockRevoke = knowledgeMemberAccessMockFns.mockRevokeKnowledgeConnectorCred const mockResolveStorageBillingContext = billingStorageMockFns.mockResolveStorageBillingContext const mockIncrementStorage = billingStorageMockFns.mockIncrementStorageUsageForBillingContextInTx const mockNotifyStorage = billingStorageMockFns.mockMaybeNotifyStorageLimitForBillingContext -knowledgeDocumentsServiceMockFns.mockDeleteDocumentStorageFiles.mockResolvedValue(undefined) knowledgeTagsServiceMockFns.mockCleanupUnusedTagDefinitions.mockResolvedValue(undefined) const mockRecordAudit = auditMockFns.mockRecordAudit @@ -349,7 +342,6 @@ describe('performUpdateKnowledgeConnector', () => { afterAll(resetDbChainMock) it('saves live permissions without indexing while protecting stale indexed ACLs', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(schemaMock.knowledgeConnector, [ { id: 'conn-1', diff --git a/apps/sim/lib/knowledge/orchestration/documents.ts b/apps/sim/lib/knowledge/orchestration/documents.ts index 4436933ecfd..0dff7ca4055 100644 --- a/apps/sim/lib/knowledge/orchestration/documents.ts +++ b/apps/sim/lib/knowledge/orchestration/documents.ts @@ -15,7 +15,6 @@ import { markDocumentAsFailedTimeout, type ProcessingOptions, retryDocumentProcessing, - updateDocument, } from '@/lib/knowledge/documents/service' import type { DocumentProcessingOutcome, @@ -372,61 +371,6 @@ export async function performUploadKnowledgeDocuments( return { success: true, documents: created } } -export interface PerformUpdateKnowledgeDocumentParams extends KnowledgeOperationContext { - knowledgeBase: KnowledgeBaseTarget - document: { id: string; filename: string } - updates: Parameters[1] -} - -export type PerformUpdateKnowledgeDocumentResult = KnowledgeOrchestrationResult<{ - document: Awaited> -}> - -/** Renames a document, toggles it, or edits its tags, and records the change. */ -export async function performUpdateKnowledgeDocument( - params: PerformUpdateKnowledgeDocumentParams -): Promise { - const { knowledgeBase, document, updates, request, source } = params - const requestId = params.requestId ?? generateRequestId() - - const updatedFields = Object.keys(updates).filter( - (key) => updates[key as keyof typeof updates] !== undefined - ) - if (updatedFields.length === 0) { - return fail('No updates specified', 'validation') - } - - let updated: Awaited> - try { - updated = await updateDocument(document.id, updates, requestId) - } catch (error) { - return classifyKnowledgeFailure(error, requestId, `Update document ${document.id}`) - } - - const filename = updates.filename ?? document.filename - - recordAudit({ - workspaceId: knowledgeBase.workspaceId, - ...auditActorFields(params), - action: AuditAction.DOCUMENT_UPDATED, - resourceType: AuditResourceType.DOCUMENT, - resourceId: document.id, - resourceName: filename, - description: `Updated document "${filename}" in knowledge base "${knowledgeBase.name ?? knowledgeBase.id}"`, - metadata: { - source, - knowledgeBaseId: knowledgeBase.id, - knowledgeBaseName: knowledgeBase.name, - fileName: filename, - updatedFields, - ...(updates.enabled !== undefined && { enabled: updates.enabled }), - }, - ...(request ? { request } : {}), - }) - - return { success: true, document: updated } -} - export interface PerformDeleteKnowledgeDocumentParams extends KnowledgeOperationContext { knowledgeBase: KnowledgeBaseTarget document: { id: string; filename: string; fileSize?: number; mimeType?: string } diff --git a/apps/sim/lib/knowledge/orchestration/index.ts b/apps/sim/lib/knowledge/orchestration/index.ts index 2f9d033a355..8e03fe20141 100644 --- a/apps/sim/lib/knowledge/orchestration/index.ts +++ b/apps/sim/lib/knowledge/orchestration/index.ts @@ -16,7 +16,6 @@ export { performDeleteKnowledgeDocument, performMarkKnowledgeDocumentTimedOut, performRetryKnowledgeDocumentProcessing, - performUpdateKnowledgeDocument, performUploadKnowledgeDocument, performUploadKnowledgeDocuments, } from './documents' diff --git a/apps/sim/lib/knowledge/projection/enqueue-inline.test.ts b/apps/sim/lib/knowledge/projection/enqueue-inline.test.ts index a06be98aef9..04e5fa5b338 100644 --- a/apps/sim/lib/knowledge/projection/enqueue-inline.test.ts +++ b/apps/sim/lib/knowledge/projection/enqueue-inline.test.ts @@ -1,20 +1,21 @@ -import { dbChainMockFns } from '@sim/testing/mocks/database.mock' import { setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { featureFlagsMock } from '@sim/testing/mocks/feature-flags.mock' import { sleep } from '@sim/utils/helpers' import { describe, expect, it, vi } from 'vitest' const mocks = vi.hoisted(() => ({ runPass: vi.fn() })) -vi.mock('@/lib/core/config/feature-flags', () => featureFlagsMock) +vi.mock('@sim/db/knowledge-projection', () => ({ + releaseSettledMarks: async () => ({ released: 0, drained: true, empty: false }), + MARK_RELEASE_BUDGET_MS: 10_000, + hasKnowledgeProjectionWork: async () => true, +})) vi.mock('@/lib/core/config/trigger-runtime', () => ({ isInsideTriggerRun: () => false })) vi.mock('@/lib/knowledge/projection/run', () => ({ runKnowledgeProjectionPass: mocks.runPass })) -import { requestKnowledgeProjection } from '@/lib/knowledge/projection/enqueue' +import { enqueueKnowledgeProjectionSweep } from '@/lib/knowledge/projection/enqueue' setEnvFlags({ isTriggerDevEnabled: false }) -dbChainMockFns.execute.mockImplementation(async () => [{ pending: true }]) /** A pass that runs until the test finishes it. */ function heldPass() { @@ -29,20 +30,20 @@ function heldPass() { } describe('knowledge projection without a Trigger.dev worker', () => { - it('runs a pass for a request that arrives at any point while the last pass is finishing', async () => { + it('runs a pass for a sweep that arrives at any point while the last pass is finishing', async () => { for (let hops = 0; hops < 8; hops++) { mocks.runPass.mockReset() - /** The pass asks for another once it has settled, `hops` microtasks later. */ + /** A sweep arrives once the pass has settled, `hops` microtasks later. */ mocks.runPass .mockImplementationOnce(() => { void (async () => { for (let hop = 0; hop < hops; hop++) await Promise.resolve() - await requestKnowledgeProjection() + await enqueueKnowledgeProjectionSweep() })() return Promise.resolve() }) .mockResolvedValue(undefined) - await requestKnowledgeProjection() + await enqueueKnowledgeProjectionSweep() await vi.waitFor(() => expect(mocks.runPass.mock.calls.length).toBeGreaterThanOrEqual(2), { timeout: 200, }) @@ -55,9 +56,9 @@ describe('knowledge projection without a Trigger.dev worker', () => { mocks.runPass.mockReset() const failFirst = heldPass() mocks.runPass.mockResolvedValue(undefined) - await requestKnowledgeProjection() + await enqueueKnowledgeProjectionSweep() await vi.waitFor(() => expect(mocks.runPass).toHaveBeenCalledTimes(1)) - await requestKnowledgeProjection() + await enqueueKnowledgeProjectionSweep() failFirst(new Error('database unavailable')) await vi.waitFor(() => expect(mocks.runPass).toHaveBeenCalledTimes(2)) }) diff --git a/apps/sim/lib/knowledge/projection/enqueue.test.ts b/apps/sim/lib/knowledge/projection/enqueue.test.ts index 2807a76e14a..b477b39b041 100644 --- a/apps/sim/lib/knowledge/projection/enqueue.test.ts +++ b/apps/sim/lib/knowledge/projection/enqueue.test.ts @@ -2,33 +2,32 @@ import { asyncJobsRegionMock, asyncJobsRegionMockFns, } from '@sim/testing/mocks/async-jobs-region.mock' -import { dbChainMockFns } from '@sim/testing/mocks/database.mock' import { mockEnvObject } from '@sim/testing/mocks/env.mock' import { setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { featureFlagsMock, featureFlagsMockFns } from '@sim/testing/mocks/feature-flags.mock' import { tasks } from '@trigger.dev/sdk' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const hoisted = vi.hoisted(() => ({ runPass: vi.fn(), insideRun: vi.fn(), + pending: vi.fn(), })) -vi.mock('@/lib/core/config/feature-flags', () => featureFlagsMock) +vi.mock('@sim/db/knowledge-projection', () => ({ + releaseSettledMarks: async () => ({ released: 0, drained: true, empty: false }), + MARK_RELEASE_BUDGET_MS: 10_000, + hasKnowledgeProjectionWork: hoisted.pending, +})) vi.mock('@/lib/core/async-jobs/region', () => asyncJobsRegionMock) vi.mock('@/lib/core/config/trigger-runtime', () => ({ isInsideTriggerRun: hoisted.insideRun })) vi.mock('@/lib/knowledge/projection/run', () => ({ runKnowledgeProjectionPass: hoisted.runPass })) -import { - enqueueKnowledgeProjectionSweep, - requestKnowledgeProjection, -} from '@/lib/knowledge/projection/enqueue' +import { enqueueKnowledgeProjectionSweep } from '@/lib/knowledge/projection/enqueue' const mocks = { ...hoisted, resolveRegion: asyncJobsRegionMockFns.mockResolveTriggerRegion, - isFeatureEnabled: featureFlagsMockFns.mockIsFeatureEnabled, } setEnvFlags({ isTriggerDevEnabled: true }) @@ -41,8 +40,7 @@ describe('knowledge projection enqueue', () => { vi.setSystemTime(new Date('2026-09-23T12:34:45.000Z')) mocks.resolveRegion.mockResolvedValue('us-east-1') mockTrigger.mockResolvedValue({ id: 'run-1' }) - dbChainMockFns.execute.mockResolvedValue([{ pending: true }]) - mocks.isFeatureEnabled.mockResolvedValue(false) + mocks.pending.mockResolvedValue(true) mockEnvObject.TRIGGER_SECRET_KEY = 'fixture-key' mocks.insideRun.mockReturnValue(false) }) @@ -65,23 +63,4 @@ describe('knowledge projection enqueue', () => { }) expect(mocks.runPass).not.toHaveBeenCalled() }) - - it('debounces prompt requests across processes and collapses them within one', async () => { - await requestKnowledgeProjection() - await requestKnowledgeProjection() - expect(mockTrigger).toHaveBeenCalledTimes(1) - expect(mockTrigger).toHaveBeenCalledWith('knowledge-projection', undefined, { - debounce: { key: 'knowledge-projection', delay: '5s', maxDelay: '1m' }, - region: 'us-east-1', - }) - vi.advanceTimersByTime(5_000) - await requestKnowledgeProjection() - expect(mockTrigger).toHaveBeenCalledTimes(2) - }) - - it('never fails the write that asked when the request is refused', async () => { - vi.advanceTimersByTime(60_000) - mockTrigger.mockRejectedValueOnce(new Error('trigger unavailable')) - await expect(requestKnowledgeProjection()).resolves.toBeUndefined() - }) }) diff --git a/apps/sim/lib/knowledge/projection/enqueue.ts b/apps/sim/lib/knowledge/projection/enqueue.ts index 5420aae6b1f..93b8887c845 100644 --- a/apps/sim/lib/knowledge/projection/enqueue.ts +++ b/apps/sim/lib/knowledge/projection/enqueue.ts @@ -1,10 +1,13 @@ import { db } from '@sim/db' -import { SOURCE_ACL_PROJECTIONS } from '@sim/db/knowledge-projection' +import { + hasKnowledgeProjectionWork, + MARK_RELEASE_BUDGET_MS, + releaseSettledMarks, +} from '@sim/db/knowledge-projection' import { createLogger } from '@sim/logger' import { getErrorMessage } from '@sim/utils/errors' -import { sql } from 'drizzle-orm' -import { isFeatureEnabled } from '@/lib/core/config/feature-flags' import { isTriggerAvailable } from '@/lib/core/config/trigger-availability' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' const logger = createLogger('KnowledgeProjectionEnqueue') @@ -20,28 +23,17 @@ export const KNOWLEDGE_PROJECTION_PASS_BUDGET_MS = 8 * 60 * 1000 /** The periodic sweep's window: at most one sweep run is enqueued per window. */ const KNOWLEDGE_PROJECTION_SWEEP_INTERVAL_MS = 60 * 1000 -/** - * A prompt request waits this long for more writes before its run starts, and is never pushed - * back past the maximum, so a steady stream of writes still gets a run each window. - */ -const PROMPT_DEBOUNCE = { key: KNOWLEDGE_PROJECTION_TASK_ID, delay: '5s', maxDelay: '1m' } as const - -/** Trigger.dev requests from one process closer together than this collapse into the first. */ -const PROMPT_REQUEST_INTERVAL_MS = 5_000 - -let lastPromptAt = 0 - /** Whether an inline pass is running when no Trigger.dev worker is configured, and whether another is owed. */ let inlineRunning = false let inlinePassOwed = false /** - * Starts a pass in this process without waiting for it: at most one runs at a time, a request while + * Starts a pass in this process without waiting for it: at most one runs at a time, a sweep while * one runs is folded into a single pass after it, and a failed pass is logged without dropping one * owed after it. The loop reads the owed flag and clears the running flag in the same synchronous - * step, so a request can never land between the two and be dropped. The pass module loads on first - * use. For deployments without a Trigger.dev worker, whose sweep and writes run passes here, as - * their document processing does. + * step, so a sweep can never land between the two and be dropped. The pass module loads on first + * use. For deployments without a Trigger.dev worker, whose sweep runs passes here, as their + * document processing does. */ function runInline(): void { if (inlineRunning) { @@ -66,67 +58,31 @@ function runInline(): void { })() } -/** - * Asks for a projector pass soon after a knowledge write commits, so its marked documents are - * converged within seconds rather than at the next sweep. Debounced twice: in this process, and - * across processes by the task's debounce key. Without a Trigger.dev worker the pass runs in this - * process, one at a time, as document processing does there. Never throws: a request that fails - * leaves the marks to the sweep, which keeps enqueueing a pass every minute until one runs. - */ -export async function requestKnowledgeProjection(): Promise { - if (!isTriggerAvailable()) { - runInline() - return - } - const now = Date.now() - if (now - lastPromptAt < PROMPT_REQUEST_INTERVAL_MS) return - lastPromptAt = now - try { - const [{ tasks }, { resolveTriggerRegion }] = await Promise.all([ - import('@trigger.dev/sdk'), - import('@/lib/core/async-jobs/region'), - ]) - await tasks.trigger(KNOWLEDGE_PROJECTION_TASK_ID, undefined, { - debounce: PROMPT_DEBOUNCE, - region: await resolveTriggerRegion(), - }) - } catch (error) { - logger.warn('Knowledge projection request failed; the sweep will pick the marks up', { - error: getErrorMessage(error), - }) - } -} - export interface KnowledgeProjectionSweepResult { - /** Whether a pass was started; the sweep starts none when nothing is marked or left to fill. */ + /** Whether a pass was started; the sweep starts none when nothing is marked. */ triggered: boolean backend: 'trigger-dev' | 'inline' | null jobId: string | null } /** - * Whether a pass would find anything to do: a marked document, or, while the fill is on, a - * projection row it has not reached. Each is one probe of an index that is empty once the - * projector has caught up. - */ -async function hasKnowledgeProjectionWork(): Promise { - const fill = await isFeatureEnabled('knowledge-projection-fill') - const unfilled = SOURCE_ACL_PROJECTIONS.map( - (projection) => `EXISTS (SELECT 1 FROM ${projection} WHERE acl IS NULL)` - ).join(' OR ') - const [row] = await db.execute<{ pending: boolean }>( - sql`SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty)${fill ? sql.raw(` OR ${unfilled}`) : sql``} AS pending` - ) - return Boolean(row?.pending) -} - -/** - * The periodic sweep behind the prompt requests: one pass per window while there is work, so a - * mark whose request was lost, or a document a pass gave up, is still converged, and an idle - * deployment starts no pass at all. + * The knowledge projector's only trigger: one pass per window while marks need one, so marks are + * settled within about a minute of their write, a document a pass gave up is retried by the next, + * and an idle deployment starts no pass at all. The sweep first releases, on the pooled database, + * the marks no pass is owed, so the always-on marking of writes never starts one. While indexed + * organization search is off that is every mark without content to project, search-index ones + * included, since nothing reads their mirrored source and ACL. */ export async function enqueueKnowledgeProjectionSweep(): Promise { - if (!(await hasKnowledgeProjectionWork())) { + const scope = { searchIndexes: isIndexedOrgSearchEnabled() } + let release = { drained: false, empty: false } + try { + release = await releaseSettledMarks(db.$client, Date.now() + MARK_RELEASE_BUDGET_MS, scope) + } catch (error) { + /** A release that failed leaves its marks for the next sweep; whether a pass is owed still stands. */ + logger.warn('Releasing settled projection marks failed', { error: getErrorMessage(error) }) + } + if (release.empty || !(await hasKnowledgeProjectionWork(db.$client, release, scope))) { return { triggered: false, backend: null, jobId: null } } if (!isTriggerAvailable()) { diff --git a/apps/sim/lib/knowledge/projection/run.test.ts b/apps/sim/lib/knowledge/projection/run.test.ts index 176e6a6f298..b18555ae1cb 100644 --- a/apps/sim/lib/knowledge/projection/run.test.ts +++ b/apps/sim/lib/knowledge/projection/run.test.ts @@ -1,10 +1,9 @@ import { databaseMockFns } from '@sim/testing/mocks/database.mock' -import { featureFlagsMock, featureFlagsMockFns } from '@sim/testing/mocks/feature-flags.mock' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' -const hoisted = vi.hoisted(() => ({ +const mocks = vi.hoisted(() => ({ runProjection: vi.fn(), - markUnfilled: vi.fn(), + release: vi.fn(), end: vi.fn(), marks: vi.fn(), })) @@ -15,22 +14,17 @@ await vi.hoisted(async () => { }) vi.mock('@sim/db/knowledge-projection', () => ({ - runKnowledgeProjection: hoisted.runProjection, - markUnfilledProjectionDocuments: hoisted.markUnfilled, + runKnowledgeProjection: mocks.runProjection, + releaseSettledMarks: mocks.release, + MARK_RELEASE_BUDGET_MS: 10_000, })) /** A session answering the backlog count; the projection itself is mocked above. */ vi.mock('postgres', () => ({ - default: () => Object.assign(async () => [{ marks: hoisted.marks() }], { end: hoisted.end }), + default: () => Object.assign(async () => [{ marks: mocks.marks() }], { end: mocks.end }), })) -vi.mock('@/lib/core/config/feature-flags', () => featureFlagsMock) import { runKnowledgeProjectionPass } from '@/lib/knowledge/projection/run' -const mocks = { - ...hoisted, - isFeatureEnabled: featureFlagsMockFns.mockIsFeatureEnabled, -} - databaseMockFns.mockResolveDbUrl.mockReturnValue('postgresql://fixture/sim_acl_test') const drained = { settled: 1, deferred: 0, pages: 2, written: 3, remaining: false } @@ -38,7 +32,7 @@ const drained = { settled: 1, deferred: 0, pages: 2, written: 3, remaining: fals describe('runKnowledgeProjectionPass', () => { beforeEach(() => { mocks.runProjection.mockResolvedValue(drained) - mocks.isFeatureEnabled.mockResolvedValue(false) + mocks.release.mockResolvedValue({ released: 0, drained: true, empty: false }) mocks.marks.mockReturnValue(2) }) @@ -75,60 +69,6 @@ describe('runKnowledgeProjectionPass', () => { await expect(pass).rejects.toThrow('connection lost') }) - it('marks unfilled documents and converges them until none are left', async () => { - mocks.isFeatureEnabled.mockResolvedValue(true) - mocks.markUnfilled - .mockResolvedValueOnce({ marked: 2, cursor: { projection: 0, afterId: 'row-2' } }) - .mockResolvedValueOnce({ marked: 0, cursor: null }) - const result = await runKnowledgeProjectionPass({ budgetMs: 60_000 }) - expect(result).toMatchObject({ filled: 2, remaining: false }) - expect(mocks.markUnfilled).toHaveBeenNthCalledWith(1, expect.anything(), undefined) - expect(mocks.markUnfilled).toHaveBeenNthCalledWith(2, expect.anything(), { - projection: 0, - afterId: 'row-2', - }) - }) - - describe('at the budget', () => { - beforeEach(() => { - vi.useFakeTimers() - mocks.isFeatureEnabled.mockResolvedValue(true) - }) - - afterEach(() => { - vi.useRealTimers() - }) - - it('reports a fill the budget cut off as remaining once its marks are settled', async () => { - mocks.markUnfilled.mockResolvedValueOnce({ - marked: 2, - cursor: { projection: 0, afterId: 'row-2' }, - }) - mocks.runProjection - .mockResolvedValueOnce(drained) - .mockResolvedValueOnce(drained) - .mockImplementation(async () => { - vi.advanceTimersByTime(60_001) - return drained - }) - await expect(runKnowledgeProjectionPass({ budgetMs: 60_000 })).resolves.toMatchObject({ - remaining: true, - filled: 2, - }) - expect(mocks.markUnfilled).toHaveBeenCalledOnce() - }) - }) - - it('does not fill while marks remain, so writers are converged first', async () => { - mocks.isFeatureEnabled.mockResolvedValue(true) - mocks.runProjection.mockResolvedValue({ ...drained, settled: 0, remaining: true }) - await expect(runKnowledgeProjectionPass({ budgetMs: 60_000 })).resolves.toMatchObject({ - remaining: true, - filled: 0, - }) - expect(mocks.markUnfilled).not.toHaveBeenCalled() - }) - it('closes its connections when a pass fails', async () => { mocks.runProjection.mockRejectedValue(new Error('connection lost')) await expect(runKnowledgeProjectionPass({ budgetMs: 60_000 })).rejects.toThrow( diff --git a/apps/sim/lib/knowledge/projection/run.ts b/apps/sim/lib/knowledge/projection/run.ts index 853bd8e5ce2..b0f2c6a5cad 100644 --- a/apps/sim/lib/knowledge/projection/run.ts +++ b/apps/sim/lib/knowledge/projection/run.ts @@ -1,14 +1,15 @@ import { resolveDbUrl } from '@sim/db' import { type KnowledgeProjectionProgress, - markUnfilledProjectionDocuments, + MARK_RELEASE_BUDGET_MS, + releaseSettledMarks, runKnowledgeProjection, } from '@sim/db/knowledge-projection' import { withUtcTimestamps } from '@sim/db/timestamps' import { createLogger } from '@sim/logger' import postgres, { type Sql } from 'postgres' import { env, envNumber } from '@/lib/core/config/env' -import { isFeatureEnabled } from '@/lib/core/config/feature-flags' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' const logger = createLogger('KnowledgeProjectionPass') @@ -36,22 +37,25 @@ async function workersFor(session: Sql): Promise { } export interface KnowledgeProjectionPassResult extends KnowledgeProjectionProgress { - /** Documents the source and ACL fill marked during the pass. */ - filled: number + /** Marks removed without a pass over their rows, since no pass was owed them. */ + released: number } /** - * One pass of the knowledge projector: converges every marked document, then, while time is - * left and the fill is on, marks the documents of projection rows the source and ACL fill has not - * reached and converges those too, keeping the outstanding marks under the fill's ceiling so fresh - * writes never queue behind much of it and search keeps probing a small set of marks. The fill is - * its own flag so an operator can pause its extra index writes. + * One pass of the knowledge projector: settles every marked document. Each round first releases, + * for no longer than a release's own short budget, the marks no pass is owed (see + * `releaseSettledMarks`), so a backlog of them shrinks every round without holding up the content + * behind it. What is left is projected: content a writer deferred, and, while indexed organization + * search is on, search-index documents, whose rows mirror their source and ACL. * * Workers project documents in parallel, each on a connection of its own that holds its * per-document advisory locks; a pass this long should not hold the pool's connections. Workers * read the same oldest marks and split them at those locks. A round ends when every worker found * nothing more it could take; the pass goes on while rounds settle documents, and `remaining` - * reports marks it left or a fill it did not finish, so the caller can schedule another. + * reports marks it left for the next sweep. + * + * The pass writes Tin keyword rows only while indexed organization search is enabled: only that + * search reads them. */ export async function runKnowledgeProjectionPass(options: { budgetMs: number @@ -71,24 +75,31 @@ export async function runKnowledgeProjectionPass(options: { ) ) const deadline = Date.now() + options.budgetMs + const scope = { searchIndexes: isIndexedOrgSearchEnabled() } const result: KnowledgeProjectionPassResult = { settled: 0, deferred: 0, pages: 0, written: 0, remaining: false, - filled: 0, + released: 0, } try { - /** `null` ends the fill: it is off, or every unfilled row has been read. */ - let fillCursor: Parameters[1] | null = - (await isFeatureEnabled('knowledge-projection-fill')) ? undefined : null while (Date.now() < deadline) { + const release = await releaseSettledMarks( + sessions[0], + Math.min(deadline, Date.now() + MARK_RELEASE_BUDGET_MS), + scope + ) + result.released += release.released const workers = sessions.slice(0, await workersFor(sessions[0])) /** Every worker finishes its document before a failure ends the pass, so none is cut off. */ const outcomes = await Promise.allSettled( workers.map((session) => - runKnowledgeProjection(session, { budgetMs: Math.max(0, deadline - Date.now()) }) + runKnowledgeProjection(session, { + budgetMs: Math.max(0, deadline - Date.now()), + ...scope, + }) ) ) const round: KnowledgeProjectionProgress[] = [] @@ -101,21 +112,11 @@ export async function runKnowledgeProjectionPass(options: { result.pages += round.reduce((sum, progress) => sum + progress.pages, 0) result.written += round.reduce((sum, progress) => sum + progress.written, 0) result.remaining = round.some((progress) => progress.remaining) - if (result.remaining) { - if (settled > 0) continue - result.deferred = round.reduce((most, progress) => Math.max(most, progress.deferred), 0) - break - } - /** The fill starts only inside the budget, and what it marks is left for a round to settle. */ - if (fillCursor === null || Date.now() >= deadline) break - const fill = await markUnfilledProjectionDocuments(sessions[0], fillCursor) - result.filled += fill.marked - fillCursor = fill.cursor - if (fill.marked > 0) result.remaining = true - if (fill.marked === 0 && fillCursor === null) break + if (!result.remaining) break + if (settled > 0) continue + result.deferred = round.reduce((most, progress) => Math.max(most, progress.deferred), 0) + break } - /** A fill stopped partway by the budget has rows left to read, so another pass is owed. */ - if (fillCursor) result.remaining = true logger.info('Knowledge projection pass finished', result) return result } finally { diff --git a/apps/sim/lib/knowledge/search/candidates.ts b/apps/sim/lib/knowledge/search/candidates.ts new file mode 100644 index 00000000000..2eb6fbdb854 --- /dev/null +++ b/apps/sim/lib/knowledge/search/candidates.ts @@ -0,0 +1,541 @@ +import { document, embedding, embeddingSearch, knowledgeConnector } from '@sim/db/schema' +import { and, eq, inArray, isNull, type SQL, sql } from 'drizzle-orm' +import { mapWithConcurrency } from '@/lib/core/utils/concurrency' +import { + type ConfluenceSiteReadGrant, + type GitHubInstallationReadGrant, + type KnowledgeAccessProvider, + type KnowledgeAccessScope, + MAX_KNOWLEDGE_ACCESS_CANDIDATES, +} from '@/lib/knowledge/access/types' +import type { KbEmbeddingDimensions } from '@/lib/knowledge/embedding-models' +import { type RetrievalLeg, runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' +import { measureSearchStage } from '@/lib/knowledge/search/diagnostics' +import { workspaceSearchFilterConditions } from '@/lib/knowledge/search/filter-conditions' +import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' +import type { StructuredFilter } from '@/lib/knowledge/types' + +/** Bound candidate pages retained while live permissions are checked. */ +const MAX_AUTHORIZED_SEARCH_CANDIDATES = 20_000 +/** + * The probe's share of the leg. It ranks nothing, so it must never be why the leg misses its + * own deadline. + * + * Its share comes out of what the rescue can claim from the narrowest live budget, + * `DIRECT_SEARCH_VECTOR_BUDGET_MS`, before live authorization, hydration and the exact rerank + * need the rest. + */ +export const VECTOR_PROBE_BUDGET_MS = 600 +/** + * What one document costs the probe, measured on a corpus shaped like a search index under + * comparable cache pressure: the access predicate, evaluated once per document. + */ +const VECTOR_PROBE_MICROSECONDS_PER_DOCUMENT = 6 +/** + * Documents the probe enumerates before it concludes the permitted set is too large to rank + * exactly. Derived so that reaching it is what spends the probe's budget, rather than a separate + * number that a change to that budget could silently invalidate. + * + * It bounds the rescue's second step too: exact ranking of the `halfvec` projection measures at + * around half the probe's per-document cost, so a permitted set within this bound is affordable + * by construction. + */ +export const VECTOR_PROBE_DOCUMENT_LIMIT = Math.round( + (VECTOR_PROBE_BUDGET_MS * 1000) / VECTOR_PROBE_MICROSECONDS_PER_DOCUMENT +) + +export interface SearchResult { + id: string + content: string + documentId: string + chunkIndex: number + tag1: string | null + tag2: string | null + tag3: string | null + tag4: string | null + tag5: string | null + tag6: string | null + tag7: string | null + number1: number | null + number2: number | null + number3: number | null + number4: number | null + number5: number | null + date1: Date | null + date2: Date | null + boolean1: boolean | null + boolean2: boolean | null + boolean3: boolean | null + /** + * The score this row's position in the returned list comes from: the + * reciprocal-rank-fusion score in hybrid mode, the cosine similarity + * (`1 - distance`) in vector mode, and 1 for a tag-only search. Stamped by + * retrieval on every row it returns; absent on rows straight from a single + * retrieval leg. Recency may reorder rows without changing this score. + */ + rankScore?: number + /** 1-based position in the returned order, stamped alongside `rankScore`. */ + rank?: number + distance: number + knowledgeBaseId: string + /** When the source last changed the document; NULL for uploads and sources that do not say. */ + sourceModifiedAt: Date | null + filename: string + sourceUrl: string | null + /** The connector type behind the document; NULL for an upload. */ + connectorType: string | null +} + +/** + * A query embedding and the width it was produced at. The two travel together + * because the width selects both the pgvector column the comparison reads and + * the index form it has to be written in; a vector without it cannot be + * compared against anything. + */ +export interface KnowledgeQueryVector { + /** JSON array literal of the embedding, in pgvector's text input format. */ + vector: string + dimensions: KbEmbeddingDimensions + model: string +} + +/** What every retrieval leg is handed, whichever strategy ranks its candidates. */ +export interface SearchParams { + knowledgeBaseIds: string[] + topK: number + /** What the caller may read; every leg applies it. Required so no leg can be written without it. */ + access: KnowledgeAccessScope + signal?: AbortSignal + budget?: SearchBudget + structuredFilters?: StructuredFilter[] + filters?: WorkspaceSearchFilters + queryVector?: KnowledgeQueryVector + distanceThreshold?: number +} + +export interface KeywordSearchParams { + knowledgeBaseIds: string[] + topK: number + access: KnowledgeAccessScope + signal?: AbortSignal + budget?: SearchBudget + query: string + /** Query embedding, so keyword-only hits still carry a real cosine distance. */ + queryVector: KnowledgeQueryVector + structuredFilters?: StructuredFilter[] + filters?: WorkspaceSearchFilters +} + +/** + * The three ranked legs of one search. Workspace knowledge bases decide readability on the + * document; the dormant search-index strategies bind the caller's resolved plan instead. Retrieval + * picks one set per search and runs, fuses, and reports on it the same way either way. + */ +export interface RetrievalLegs { + tags(params: SearchParams): Promise + vector(params: SearchParams): Promise + keyword(params: KeywordSearchParams): Promise +} + +/** Common fields selected for search results */ +export const getSearchResultFields = (distanceExpr: SQL | SQL.Aliased) => ({ + id: embedding.id, + content: embedding.content, + documentId: embedding.documentId, + chunkIndex: embedding.chunkIndex, + tag1: embedding.tag1, + tag2: embedding.tag2, + tag3: embedding.tag3, + tag4: embedding.tag4, + tag5: embedding.tag5, + tag6: embedding.tag6, + tag7: embedding.tag7, + number1: embedding.number1, + number2: embedding.number2, + number3: embedding.number3, + number4: embedding.number4, + number5: embedding.number5, + date1: embedding.date1, + date2: embedding.date2, + boolean1: embedding.boolean1, + boolean2: embedding.boolean2, + boolean3: embedding.boolean3, + distance: distanceExpr, + knowledgeBaseId: embedding.knowledgeBaseId, + sourceModifiedAt: document.sourceModifiedAt, + filename: document.filename, + sourceUrl: document.sourceUrl, + connectorType: knowledgeConnector.connectorType, +}) + +/** + * Match the normalization used by the stored text vectors so query terms and + * document terms resolve to the same lexemes in both keyword retrieval paths. + */ +export const FTS_CONFIG = 'english' + +/** + * Row visibility predicates shared by every search leg: a chunk is only + * retrievable when both it and its document are enabled, the document finished + * processing, it has not been excluded, archived, or soft-deleted, and the + * caller may read it. Every leg spreads this helper rather than listing the + * predicates itself, so no leg can drift from the others. + */ +export function getVisibilityConditions( + filters: WorkspaceSearchFilters | undefined, + accessCondition: SQL, + enabledColumn: typeof embedding.enabled | typeof embeddingSearch.enabled = embedding.enabled +) { + return [eq(enabledColumn, true), ...getDocumentVisibilityConditions(filters, accessCondition)] +} + +function getDocumentVisibilityConditions( + filters: WorkspaceSearchFilters | undefined, + accessCondition: SQL +) { + return [ + eq(document.enabled, true), + eq(document.processingStatus, 'completed'), + eq(document.userExcluded, false), + isNull(document.archivedAt), + isNull(document.deletedAt), + accessCondition, + ...workspaceSearchFilterConditions(filters), + ] +} + +/** + * The document-level candidate predicate every ranked leg applies. A permitted set is resolved + * with the same list, which is what lets a leg rank inside it without admitting anything more. + */ +export function candidateDocumentConditions( + knowledgeBaseIds: string[], + filters: WorkspaceSearchFilters | undefined, + accessCondition: SQL +) { + return [ + inArray(document.knowledgeBaseId, knowledgeBaseIds), + ...getDocumentVisibilityConditions(filters, accessCondition), + ] +} + +export interface SearchReadCandidatePage { + candidates: SearchReadCandidate[] + nextOffset: number +} + +export type SearchReadCandidate = { + id: string + documentId: string + connectorId: string | null +} + +/** Only opaque identifiers leave candidate ranking; content stays behind the full read predicate. */ +export const SEARCH_READ_CANDIDATE_FIELDS = { + id: embedding.id, + documentId: document.id, + connectorId: document.connectorId, +} + +/** + * The caller's proof of reader access to the sources that require one, resolved at most once per + * search and only when a candidate from such a source is about to be read. + */ +export interface LiveSourceAccess { + gates: (connectorId: string) => boolean + /** The caller's scope with its grants, and the gated sources those grants do not cover. */ + resolve: () => Promise<{ access: KnowledgeAccessScope; denied: ReadonlySet }> +} + +/** Sources ranked or authorized at once; each holds a connection or a request of its own. */ +export const SOURCE_RANKING_CONCURRENCY = 3 + +/** + * Binds a search's gated sources to one memoized resolution of the caller's grants; nothing when + * the search holds no gated source. + */ +export function liveSourceAccessForConnectors( + gatedConnectorIds: readonly string[], + accessProvider: KnowledgeAccessProvider, + signal?: AbortSignal +): LiveSourceAccess | undefined { + const gated = new Set(gatedConnectorIds) + if (gated.size === 0) return undefined + let pending: Promise<{ access: KnowledgeAccessScope; denied: ReadonlySet }> | undefined + /** The provider authorizes connectors in bounded pages, so a wide scope resolves page by page. */ + const pages: string[][] = [] + for (const id of gated) { + const last = pages.at(-1) + if (!last || last.length === MAX_KNOWLEDGE_ACCESS_CANDIDATES) pages.push([id]) + else last.push(id) + } + return { + gates: (connectorId) => gated.has(connectorId), + resolve: () => { + pending ??= measureSearchStage('live_source_grants', async () => { + const scopes = await mapWithConcurrency(pages, SOURCE_RANKING_CONCURRENCY, (page) => + accessProvider.getForConnectors(page, signal) + ) + const [first] = scopes + const githubInstallationGrants: GitHubInstallationReadGrant[] = [] + const confluenceSiteGrants: ConfluenceSiteReadGrant[] = [] + for (const scope of scopes) { + if (scope.kind !== 'user') continue + if (scope.githubInstallationGrants) + githubInstallationGrants.push(...scope.githubInstallationGrants) + if (scope.confluenceSiteGrants) confluenceSiteGrants.push(...scope.confluenceSiteGrants) + } + const granted = new Set([ + ...githubInstallationGrants.map((grant) => grant.connectorId), + ...confluenceSiteGrants.map((grant) => grant.connectorId), + ]) + const denied = new Set([...gated].filter((id) => !granted.has(id))) + const merged = + first.kind === 'user' + ? { ...first, githubInstallationGrants, confluenceSiteGrants } + : first + return { access: merged, denied } + }) + return pending + }, + } +} + +/** Keeps the candidates of sources the caller turned out not to hold out of a page. */ +export function excludeSearchSources(sourceIds: readonly string[]): SQL | undefined { + return sourceIds.length + ? sql`(${document.connectorId} IS NULL OR NOT (${inArray(document.connectorId, [...sourceIds])}))` + : undefined +} + +const AUTHORIZED_SEARCH_PAGE_SIZE = 200 +const AUTHORIZED_SEARCH_BUDGET_MS = 8000 + +/** + * Verification follows ranked candidates, never the organization's source order. Denied + * sources are excluded on refill, so many matches from one revoked source cannot + * consume every result slot. Candidate pages and the shared deadline bound authorization work. + */ +export async function selectAuthorizedSearchResults(input: { + leg: 'vector' | 'keyword' | 'tags' + access: KnowledgeAccessScope + signal?: AbortSignal + budget?: SearchBudget + topK: number + selectPage: ( + limit: number, + offset: number, + excludedSources: readonly string[] + ) => Promise + compareResults?: (a: SearchResult, b: SearchResult) => number + hydrate: (ids: string[], access: KnowledgeAccessScope) => Promise + liveSourceAccess?: LiveSourceAccess +}): Promise { + const deadline = Date.now() + AUTHORIZED_SEARCH_BUDGET_MS + const pageSize = Math.min(AUTHORIZED_SEARCH_PAGE_SIZE, Math.max(input.topK, 20)) + const results = new Map() + const considered = new Set() + /** Gated sources the caller turned out not to hold: left out of every page once that is known. */ + let excluded: ReadonlySet = new Set() + let scanned = 0 + let offset = 0 + /** Candidates a ranking returned beyond the current hydration slice. */ + let pending: SearchReadCandidate[] = [] + let lastPageShort = false + try { + while ( + results.size < input.topK && + scanned < MAX_AUTHORIZED_SEARCH_CANDIDATES && + (input.budget !== undefined || Date.now() < deadline) + ) { + input.signal?.throwIfAborted() + input.budget?.remaining() + /** + * A ranking may hand back more candidates than one hydration should read — a narrow + * reader's keyword window is ranked once for several pages' worth — so a page is drained in + * hydration-sized slices, and what is left waits, unread, until the results still need it. + */ + if (!pending.length) { + const page = await measureSearchStage(`${input.leg}.candidates`, () => + input.selectPage(pageSize, offset, [...excluded]) + ) + if (!page.candidates.length) break + scanned += page.candidates.length + /** Short means the ranking had fewer to give, not that a read recovered fewer than it asked. */ + lastPageShort = page.nextOffset - offset < pageSize + offset = page.nextOffset + pending = page.candidates.filter((candidate) => !considered.has(candidate.id)) + if (!pending.length) { + if (lastPageShort) break + continue + } + } + /** + * A candidate counts as considered only once its slice is read: the slices a refill discards + * were never read, so the rebuilt pages may hand their readable candidates back. + */ + const candidates = pending + .slice(0, pageSize) + .filter((candidate) => !considered.has(candidate.id)) + pending = pending.slice(pageSize) + for (const candidate of candidates) considered.add(candidate.id) + if (!candidates.length) continue + /** + * A source that proves its reader live is asked for that proof only once a candidate of + * its own reaches this page, and then once for the whole search: a scope that ranks none + * of them — most scopes — never asks, and one that ranks many asks once. + */ + const proof = input.liveSourceAccess + const gatedOnPage = + proof !== undefined && + candidates.some((candidate) => candidate.connectorId && proof.gates(candidate.connectorId)) + let access = input.access + let refill = false + if (gatedOnPage && proof) { + const resolved = await proof.resolve() + access = resolved.access + /** + * A denied source's candidates cannot hydrate, yet they took the slots of sources the + * caller does hold. Once the denial is known the pages are rebuilt without that source, + * from the start; the ids already seen are not read twice. + */ + if (resolved.denied.size > excluded.size) { + excluded = resolved.denied + refill = true + } + } + const hydrated = await measureSearchStage(`${input.leg}.hydration`, () => + input.hydrate( + candidates.map((candidate) => candidate.id), + access + ) + ) + const byId = new Map(hydrated.map((row) => [row.id, row])) + for (const candidate of candidates) { + const row = byId.get(candidate.id) + if (row) results.set(row.id, row) + if (!input.compareResults && results.size === input.topK) break + } + if (refill) { + /** The rebuilt pages are a new stream of candidates, so the scan budget starts over. */ + offset = 0 + scanned = 0 + pending = [] + continue + } + /** A short page is the end of the candidates, whether or not they were reordered. */ + if (!pending.length && lastPageShort) break + } + } catch (error) { + if (!input.budget?.isTimeout(error)) throw error + } + input.signal?.throwIfAborted() + const rows = [...results.values()] + /** A reordered leg keeps every page's rows until the end: a later page cannot displace what an earlier one ranked. */ + return input.compareResults ? rows.sort(input.compareResults).slice(0, input.topK) : rows +} + +/** + * Loads the content of candidates that survived ranking, under the read predicate. + * + * A page of ranked identifiers is small, so its content is read under the full predicate, which + * re-reads each connector's own lifecycle and approval: a source deleted, archived or unapproved + * while the search was running stops answering here, at the gate that returns content. + */ +export function hydrateSearchCandidates( + ids: string[], + accessCondition: SQL, + distance: SQL | SQL.Aliased, + filters: WorkspaceSearchFilters | undefined, + conditions: (SQL | undefined)[], + leg: RetrievalLeg, + budget?: SearchBudget, + /** Whether a condition reads the projection's stored halfvec, which only the vector leg's threshold does. */ + joinProjection = false +) { + /** + * The projection joins so a condition on its stored halfvec — the candidate threshold — can be + * tested here; the returned score is whatever the leg passes as `distance`. Both legs pass the + * original vector's cosine distance: one out-of-line read per hydrated row, the page's size, + * where scoring the whole candidate pool that way read one per candidate on every novel query. + */ + return runSearchQuery(budget, `${leg}.sql`, (executor) => { + const read = executor + .select(getSearchResultFields(distance)) + .from(embedding) + .innerJoin(document, eq(embedding.documentId, document.id)) + .leftJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) + return ( + joinProjection ? read.leftJoin(embeddingSearch, eq(embeddingSearch.id, embedding.id)) : read + ).where( + and( + inArray(embedding.id, ids), + ...getVisibilityConditions(filters, accessCondition), + ...conditions + ) + ) + }) +} + +/** A document a caller may rank, with the source a live authorization pass may later exclude. */ +export type PermittedDocument = { + id: string + connectorId: string | null +} + +export type ProbeOutcome = + | { kind: 'documents'; documents: PermittedDocument[] } + /** The caller reads more documents than an exact ranking can afford. */ + | { kind: 'saturated' } + /** The probe spent its own deadline before finding out. */ + | { kind: 'timed_out' } + +/** + * Enumerate the documents the caller may read, stopping once there are more of them than an exact + * ranking can afford. The bound is documents examined, not chunks accumulated: the access + * predicate is evaluated once per document, and a search index holds only a few chunks per + * document, so a chunk-bounded enumeration walks many times more documents than its limit says. + * + * `query` returns at most one row past {@link VECTOR_PROBE_DOCUMENT_LIMIT}, or a lone + * `saturated` sentinel. Neither saturation nor a timeout is a failure of the leg, which keeps the + * candidates it already has. + */ +export async function probeVisibleDocuments( + query: SQL, + budget: SearchBudget | undefined, + stage: 'vector.probe' | 'permitted_documents', + probeBudgetMs: number = VECTOR_PROBE_BUDGET_MS +): Promise { + const probeBudget = budget?.capped(probeBudgetMs) + try { + const probed = await runSearchQuery(probeBudget, stage, (executor) => + executor.execute(query) + ) + /** The saturation sentinel is only ever emitted alone. */ + if (probed.length > VECTOR_PROBE_DOCUMENT_LIMIT || probed[0]?.saturated) { + return { kind: 'saturated' } + } + return { + kind: 'documents', + documents: probed.map(({ id, connectorId }) => ({ id, connectorId })), + } + } catch (error) { + if (!budget || !probeBudget?.isTimeout(error)) throw error + /** Only the probe's share was spent; the leg's own deadline still governs. */ + budget.remaining() + return { kind: 'timed_out' } + } +} + +/** + * The probe's SQL when the conditions decide readability on the document as they are, returning + * at most one row past the document limit. + */ +export function directVisibleDocumentsQuery(conditions: (SQL | undefined)[]): SQL { + return sql` + SELECT ${document.id} AS id, ${document.connectorId} AS "connectorId", false AS saturated + FROM ${document} + WHERE ${and(...conditions)} + LIMIT ${VECTOR_PROBE_DOCUMENT_LIMIT + 1} + ` +} diff --git a/apps/sim/lib/knowledge/search/filter-conditions.ts b/apps/sim/lib/knowledge/search/filter-conditions.ts index b23a78bf099..fb9047c44a0 100644 --- a/apps/sim/lib/knowledge/search/filter-conditions.ts +++ b/apps/sim/lib/knowledge/search/filter-conditions.ts @@ -1,17 +1,26 @@ import { document, knowledgeConnector } from '@sim/db/schema' -import { eq, gte, inArray, isNull, lte, type SQL, sql } from 'drizzle-orm' +import { and, eq, gte, inArray, isNull, lte, type SQL, sql } from 'drizzle-orm' import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' +/** The date window a filter asks for, on the document row; nothing when none is asked. */ +export function searchDateFilterCondition(filters?: WorkspaceSearchFilters): SQL | undefined { + if (!filters?.modifiedAfter && !filters?.modifiedBefore) return undefined + return and( + filters.modifiedAfter + ? gte(document.sourceModifiedAt, new Date(filters.modifiedAfter)) + : undefined, + filters.modifiedBefore + ? lte(document.sourceModifiedAt, new Date(filters.modifiedBefore)) + : undefined + ) +} + /** Filters the document in every retrieval leg, alongside its current ACL. */ export function workspaceSearchFilterConditions(filters?: WorkspaceSearchFilters): SQL[] { const conditions: SQL[] = [] if (filters?.documentIds) conditions.push(inArray(document.id, filters.documentIds)) - if (filters?.modifiedAfter) { - conditions.push(gte(document.sourceModifiedAt, new Date(filters.modifiedAfter))) - } - if (filters?.modifiedBefore) { - conditions.push(lte(document.sourceModifiedAt, new Date(filters.modifiedBefore))) - } + const dateCondition = searchDateFilterCondition(filters) + if (dateCondition) conditions.push(dateCondition) if (filters?.source === 'upload') conditions.push(isNull(document.connectorId)) else if (filters?.source) { conditions.push( diff --git a/apps/sim/lib/knowledge/search/keyword-ranking.ts b/apps/sim/lib/knowledge/search/keyword-ranking.ts new file mode 100644 index 00000000000..ed9edf76869 --- /dev/null +++ b/apps/sim/lib/knowledge/search/keyword-ranking.ts @@ -0,0 +1,49 @@ +import { document, type embedding, type embeddingKeywordSearch } from '@sim/db/schema' +import { and, type SQL, sql } from 'drizzle-orm' + +/** + * A keyword ranking in three materialized steps: the chunks matching the query (identifiers + * only), the documents among them the reader may read, and the ranking of the readable matches. + * Deciding readability once per matched document rather than once per matched chunk, and ranking + * only readable matches, keeps a common term from paying the full access predicate and a + * text-search vector per match before the `LIMIT` applies. + * + * Restricting the documents with `document.id = ANY (ARRAY(...))` rather than a subquery keeps the + * narrowed lookup on a bitmap scan, which prefetches, where a plain `IN (SELECT ...)` plans as an + * index walk that does not. + */ +export function keywordCandidateRankingQuery(input: { + /** Selects `id` and `document_id` of every chunk matching the query. */ + matchedChunks: SQL + /** What a matched document must satisfy to be readable. */ + documentConditions: (SQL | undefined)[] + /** The table whose text-search vector `rank` reads, joined to the matches on `id`. */ + rankTable: typeof embedding | typeof embeddingKeywordSearch + rank: SQL + limit: number + offset: number +}): SQL { + return sql` + WITH matched_keyword_chunks AS MATERIALIZED (${input.matchedChunks} + ), visible_keyword_documents AS MATERIALIZED ( + SELECT ${document.id} AS id FROM ${document} + WHERE ${and( + sql`${document.id} = ANY (ARRAY(SELECT document_id FROM matched_keyword_chunks))`, + ...input.documentConditions + )} + ), ranked_keyword_candidates AS MATERIALIZED ( + SELECT matched_keyword_chunks.id, matched_keyword_chunks.document_id, + ${input.rank} AS keyword_rank + FROM matched_keyword_chunks INNER JOIN ${input.rankTable} + ON ${input.rankTable.id} = matched_keyword_chunks.id + WHERE matched_keyword_chunks.document_id IN (SELECT id FROM visible_keyword_documents) + ORDER BY keyword_rank DESC, matched_keyword_chunks.id + LIMIT ${input.limit} OFFSET ${input.offset} + ) + SELECT ranked_keyword_candidates.id, ${document.id} AS "documentId", + ${document.connectorId} AS "connectorId" + FROM ranked_keyword_candidates INNER JOIN ${document} + ON ${document.id} = ranked_keyword_candidates.document_id + ORDER BY ranked_keyword_candidates.keyword_rank DESC, ranked_keyword_candidates.id + ` +} diff --git a/apps/sim/lib/knowledge/search/queries.test.ts b/apps/sim/lib/knowledge/search/queries.test.ts index e33559bf3af..2bb9d17735b 100644 --- a/apps/sim/lib/knowledge/search/queries.test.ts +++ b/apps/sim/lib/knowledge/search/queries.test.ts @@ -5,47 +5,41 @@ import { resetDbChainMock, schemaMock, } from '@sim/testing' +import { sql } from 'drizzle-orm' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const { mockResolveTinKeywordQuery } = vi.hoisted(() => ({ - mockResolveTinKeywordQuery: vi.fn<() => Promise>(async () => null), -})) - -vi.mock('@/lib/knowledge/search/tin-keyword', () => ({ - resolveTinKeywordQuery: mockResolveTinKeywordQuery, -})) - +import { knowledgeAccessCondition } from '@/lib/knowledge/access/predicate' import { type KnowledgeAccessProvider, type UserAccessScope, WORKSPACE_ACCESS_TOKENS, } from '@/lib/knowledge/access/types' import { SearchBudget, type SearchExecutor } from '@/lib/knowledge/search/budget' +import { + type SearchParams, + type SearchResult, + VECTOR_PROBE_DOCUMENT_LIMIT, +} from '@/lib/knowledge/search/candidates' import type { SearchStage } from '@/lib/knowledge/search/diagnostics' import { - executeKeywordSearch, - forgetProjectionFilled, - forgetSearchReach, + forgetSaturatedReads, fuseByReciprocalRank, - getStructuredTagFilters, - handleTagAndVectorSearch, handleTagOnlySearch, - handleVectorOnlySearch, - PERMITTED_EXACT_DOCUMENT_LIMIT, - type PermittedDocuments, - resolvePermittedDocuments, - resolveReach, + handleVectorSearch, retrieveKnowledgeSearch, - type SearchParams, - type SearchResult, - VECTOR_PROBE_DOCUMENT_LIMIT, - vectorCandidatePoolLimit, - visibleDocumentsQuery, } from '@/lib/knowledge/search/queries' import { RRF_K } from '@/lib/knowledge/search/recency' -import { forgetIndexedVectorSources } from '@/lib/knowledge/search/source-vector-indexes' +import { getStructuredTagFilters } from '@/lib/knowledge/search/tag-filters' +import { vectorCandidatePoolLimit } from '@/lib/knowledge/search/vector-leg' import type { StructuredFilter } from '@/lib/knowledge/types' +/** A leg under the caller's ordinary document predicate: a search holding no live-proof source. */ +const readOf = (params: SearchParams) => ({ + rankCondition: knowledgeAccessCondition(params.access), + signedIn: params.access.kind === 'user', +}) +const vectorSearch = (params: SearchParams) => handleVectorSearch(params, readOf(params)) +const tagSearch = (params: SearchParams) => handleTagOnlySearch(params, readOf(params)) + /** * The builder only reads `embeddingTable[tagSlot]`, so a slot-to-name map stands * in for the real table and makes each rendered parameter readable. @@ -57,9 +51,6 @@ const embeddingTable = { boolean1: 'boolean1', } -/** The projection-fill memo outlives a test; every case starts without one. */ -beforeEach(() => forgetProjectionFilled()) - describe('retrieval leg budgets', () => { it.each([undefined, 3000])( 'applies vector budget %s without shortening keyword or tag retrieval', @@ -113,6 +104,69 @@ describe('retrieval leg budgets', () => { ) }) +describe('many-base tag and keyword legs rank globally', () => { + const knowledgeBaseIds = ['kb-1', 'kb-2', 'kb-3', 'kb-4', 'kb-5'] + const queryVector = { + vector: '[1,0]', + dimensions: 1536 as const, + model: 'text-embedding-3-small', + } + const tags: StructuredFilter[] = [ + { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }, + ] + const reader: UserAccessScope = { + kind: 'user', + userId: 'reader', + tokens: WORKSPACE_ACCESS_TOKENS, + } + const accessProvider: KnowledgeAccessProvider = { + get: async () => reader, + getForConnectors: async () => reader, + getForDocuments: async () => reader, + liveSourceConnectorCondition: async () => null, + } + beforeEach(() => { + resetDbChainMock() + }) + + it.each([ + [ + 'a keyword deadline', + new Error('Statement canceled', { cause: { code: '57014' } }), + 'partial', + ], + [ + 'any other keyword failure', + new Error('Connection lost', { cause: { code: '08006' } }), + 'fail', + ], + ] as const)('answers %s with %s', async (_label, failure, outcome) => { + const query = SearchBudget.prototype.query + vi.spyOn(SearchBudget.prototype, 'query').mockImplementation(function ( + this: SearchBudget, + stage: SearchStage, + run: (executor: SearchExecutor) => PromiseLike + ) { + if (stage === 'keyword.sql') return Promise.reject(failure) + return query.call(this, stage, run) as Promise + }) + const search = retrieveKnowledgeSearch({ + knowledgeBaseIds: ['kb-1'], + topK: 10, + access: reader, + accessProvider, + searchMode: 'hybrid', + query: 'release', + queryVector, + }) + if (outcome === 'fail') { + await expect(search).rejects.toBe(failure) + return + } + expect((await search).retrieval).toEqual({ status: 'partial', timedOutLegs: ['keyword'] }) + }) +}) + /** * The global `drizzle-orm` mock renders `sql` fragments to a `?`-placeholder * string via `toSQL()`, so we can assert the exact predicate each filter builds. @@ -121,15 +175,14 @@ function render(condition: unknown) { return (condition as { toSQL: () => { sql: string; params: unknown[] } }).toSQL() } -/** The permitted-document probe is the only statement that reports whether it saturated. */ -/** The permitted-documents probe: the reach count and the saturation sentinel, never a slice. */ +/** The readable-document probe is the only statement that reports whether it saturated. */ function isProbeStatement(sql: string) { - return sql.includes('AS saturated') && !sql.includes('readable_chunks') + return sql.includes('AS saturated') } -/** A graph walk: visibility joined per visited row, or decided on the row it visits. */ +/** A graph walk: visibility joined per visited row on its document. */ function isWalk(sql: string) { - return sql.includes('AS visible') || sql.includes('on-row visibility') + return sql.includes('AS visible') } /** `+ 0` is what keeps the exact ranking off the ANN index, so it also identifies the statement. */ @@ -137,13 +190,6 @@ function isExactRanking(sql: string) { return sql.includes(') + 0 LIMIT') } -/** The page read: a slice of the pool's identities, from the projection and its documents. */ -function isPageStatement(sql: string) { - return ( - sql.includes('AS "connectorId"') && sql.includes('= ANY(') && !sql.includes('ranked_tin_chunks') - ) -} - const statements = () => dbChainMockFns.execute.mock.calls.map(([query]) => render(query)) function renderOne(filters: StructuredFilter[]) { @@ -230,24 +276,18 @@ describe('getStructuredTagFilters', () => { describe('workspace-scoped vector retrieval', () => { const access = { kind: 'workspace' as const, tokens: WORKSPACE_ACCESS_TOKENS } - const getForConnectors = vi.fn() const params: SearchParams = { knowledgeBaseIds: ['kb-small'], topK: 2, access, - accessProvider: { - get: async () => access, - getForConnectors, - getForDocuments: async () => access, - liveSourceConnectorCondition: async () => null, - }, queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, distanceThreshold: 1, } const probe = Array.from({ length: 400 }, (_, index) => ({ id: `probe-${index}` })) const candidates = Array.from({ length: 400 }, (_, index) => ({ id: `candidate-${index}`, - initial_count: 400, + documentId: `candidate-doc-${index}`, + connectorId: null, })) const ranked = [ { @@ -264,32 +304,25 @@ describe('workspace-scoped vector retrieval', () => { }, ] let probeRows: Array<{ id: string }> - let traversedRows: Array<{ id: string; initial_count?: number }> + let traversedRows: Array<{ id: string }> let exactRows: Array<{ id: string }> - let failSettings: unknown let failCandidates: unknown beforeEach(() => { resetDbChainMock() - getForConnectors.mockReset() + forgetSaturatedReads() probeRows = probe - traversedRows = candidates + /** A full pool: the walk found as many readable neighbours as it was asked for. */ + traversedRows = [...ranked, ...candidates] exactRows = probe - failSettings = undefined failCandidates = undefined dbChainMockFns.execute.mockImplementation(async (query) => { const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('hnsw.iterative_scan')) { - if (failSettings) throw failSettings - return [] - } - if (statement.includes('AS visible')) { + if (statement.includes('hnsw.iterative_scan')) return [] + if (isWalk(statement)) { if (failCandidates) throw failCandidates return traversedRows } - if (isPageStatement(statement)) return ranked if (isExactRanking(statement)) return exactRows if (isProbeStatement(statement)) return probeRows return [] @@ -299,40 +332,37 @@ describe('workspace-scoped vector retrieval', () => { vi.useRealTimers() }) - it.each([handleVectorOnlySearch, handleTagAndVectorSearch])( - 'does not acquire a connection or start SQL after the KB retrieval deadline', - async (search) => { - const budget = new SearchBudget('vector', performance.now() - 1) - expect( - await search({ - ...params, - budget, - structuredFilters: [ - { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }, - ], - }) - ).toEqual([]) - expect(budget.timedOut).toBe(true) - expect(dbChainMockFns.transaction).not.toHaveBeenCalled() - expect(dbChainMockFns.select).not.toHaveBeenCalled() - expect(dbChainMockFns.execute).not.toHaveBeenCalled() - } - ) + it('does not acquire a connection or start SQL after the KB retrieval deadline', async () => { + const budget = new SearchBudget('vector', performance.now() - 1) + expect( + await vectorSearch({ + ...params, + budget, + structuredFilters: [ + { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }, + ], + }) + ).toEqual([]) + expect(budget.timedOut).toBe(true) + expect(dbChainMockFns.transaction).not.toHaveBeenCalled() + expect(dbChainMockFns.select).not.toHaveBeenCalled() + expect(dbChainMockFns.execute).not.toHaveBeenCalled() + }) - it('rescues an underfilled traversal by ranking the permitted set exactly', async () => { + it('rescues an underfilled traversal by ranking the readable set exactly', async () => { traversedRows = ranked probeRows = [{ id: 'near-doc' }, { id: 'far-doc' }] exactRows = ranked queueTableRows(schemaMock.embedding, [...ranked].reverse()) - expect((await handleVectorOnlySearch(params)).map((row) => row.id)).toEqual(['near', 'far']) + expect((await vectorSearch(params)).map((row) => row.id)).toEqual(['near', 'far']) const exact = statements().find((query) => isExactRanking(query.sql))! expect(exact.sql).not.toContain('CROSS JOIN LATERAL') expect(JSON.stringify(exact)).toContain('near-doc') const probeStatement = statements().find((query) => isProbeStatement(query.sql))! expect(probeStatement.sql).not.toContain('<=>') + expect(probeStatement.sql).not.toContain('WITH reach') expect(probeStatement.params).toContain(VECTOR_PROBE_DOCUMENT_LIMIT + 1) expect(JSON.stringify(probeStatement)).toContain('required_clause') - expect(getForConnectors).not.toHaveBeenCalled() }) it('counts only chunks the search can return when a tag filter decides a document', async () => { @@ -340,7 +370,7 @@ describe('workspace-scoped vector retrieval', () => { probeRows = [{ id: 'near-doc' }] exactRows = ranked queueTableRows(schemaMock.embedding, [...ranked].reverse()) - await handleTagAndVectorSearch({ + await vectorSearch({ ...params, structuredFilters: [{ tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'common' }], }) @@ -353,12 +383,19 @@ describe('workspace-scoped vector retrieval', () => { expect(probe).toContain(`"left":"${schemaMock.embedding.enabled}","right":true`) }) - it('keeps an underfilled traversal when the permitted set is too large to rank exactly', async () => { + it('keeps an underfilled traversal when the readable set is too large to rank exactly, and does not probe it again', async () => { traversedRows = ranked probeRows = new Array(VECTOR_PROBE_DOCUMENT_LIMIT + 1).fill({ id: 'doc' }) queueTableRows(schemaMock.embedding, [...ranked].reverse()) - expect((await handleVectorOnlySearch(params)).map((row) => row.id)).toEqual(['near', 'far']) + expect((await vectorSearch(params)).map((row) => row.id)).toEqual(['near', 'far']) + queueTableRows(schemaMock.embedding, [...ranked].reverse()) + expect((await vectorSearch(params)).map((row) => row.id)).toEqual(['near', 'far']) expect(statements().filter((query) => isExactRanking(query.sql))).toHaveLength(0) + expect(statements().filter((query) => isProbeStatement(query.sql))).toHaveLength(1) + /** A different reader's set is its own question. */ + queueTableRows(schemaMock.embedding, [...ranked].reverse()) + await vectorSearch({ ...params, knowledgeBaseIds: ['kb-other'] }) + expect(statements().filter((query) => isProbeStatement(query.sql))).toHaveLength(2) }) it('spends only its own share of the leg on a probe that runs long', async () => { @@ -367,17 +404,14 @@ describe('workspace-scoped vector retrieval', () => { exactRows = ranked const execute = dbChainMockFns.execute.getMockImplementation()! dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (isProbeStatement(statement)) + if (isProbeStatement(render(query).sql)) throw new Error('canceling statement due to statement timeout', { cause: { code: '57014' }, }) return execute(query) }) queueTableRows(schemaMock.embedding, [...ranked].reverse()) - expect((await handleVectorOnlySearch({ ...params, budget })).map((row) => row.id)).toEqual([ + expect((await vectorSearch({ ...params, budget })).map((row) => row.id)).toEqual([ 'near', 'far', ]) @@ -387,7 +421,7 @@ describe('workspace-scoped vector retrieval', () => { it('walks for a pool sized to the page, and scores the page on the original vectors', async () => { queueTableRows(schemaMock.embedding, [...ranked].reverse()) - await handleVectorOnlySearch(params) + await vectorSearch(params) const walk = statements().find((query) => isWalk(query.sql))! /** The page is the walk's order, so the walk ends at a page's worth of candidates, not a rerank's. */ expect(walk.params).toContain(200) @@ -405,21 +439,6 @@ describe('workspace-scoped vector retrieval', () => { expect(JSON.stringify(dbChainMockFns.leftJoin.mock.calls)).toContain('knowledgeConnector') }) - it('passes over a slice whose documents went away instead of ending the pool there', async () => { - const execute = dbChainMockFns.execute.getMockImplementation()! - let pages = 0 - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query) - /** The first slice's documents are gone; the next slice still has the readable rows. */ - if (isPageStatement(statement.sql)) return pages++ === 0 ? [] : ranked - return execute(query) - }) - queueTableRows(schemaMock.embedding, [...ranked].reverse()) - expect((await handleVectorOnlySearch(params)).map((row) => row.id)).toEqual(['near', 'far']) - expect(pages).toBe(2) - expect(statements().filter((query) => isWalk(query.sql))).toHaveLength(1) - }) - it('sizes the pool to the pages asked for, doubling a pool the pages outran', () => { expect(vectorCandidatePoolLimit(20, undefined)).toBe(200) expect(vectorCandidatePoolLimit(150, undefined)).toBe(300) @@ -427,9 +446,9 @@ describe('workspace-scoped vector retrieval', () => { expect(vectorCandidatePoolLimit(5000, 1600)).toBe(1600) }) - it('uses compact candidates for a large KB and applies full workspace access before its limit', async () => { + it('applies full workspace access on the document before the walk counts a candidate', async () => { queueTableRows(schemaMock.embedding, [...ranked].reverse()) - expect((await handleVectorOnlySearch(params)).map((row) => row.id)).toEqual(['near', 'far']) + await vectorSearch(params) const candidate = statements().find((query) => isWalk(query.sql))! expect(candidate.sql).toContain('CROSS JOIN LATERAL') expect(candidate.sql).toContain('LIMIT 1') @@ -442,40 +461,38 @@ describe('workspace-scoped vector retrieval', () => { expect(serialized).toContain('organizationSearchIntegration') expect(serialized).toContain(String(schemaMock.embeddingSearch.vector512)) expect(candidate.params).not.toContain(schemaMock.embedding.embedding) - /** The page is the ranking's own order; nothing rescores it against the original vectors. */ - const page = statements().find((query) => isPageStatement(query.sql))! - expect(page.sql).not.toContain('MATERIALIZED') - expect(JSON.stringify(page)).not.toContain(String(schemaMock.embedding.embedding)) - expect(JSON.stringify(page)).toContain('candidate-0') - expect(JSON.stringify(page)).not.toContain('probe-') - expect(getForConnectors).not.toHaveBeenCalled() + /** Readability is the document's alone: nothing asks the projector's marks. */ + expect(serialized).not.toContain('knowledgeProjectionDirty') + /** The walk carries each candidate's document, so its order is the page's with no page read. */ + expect( + statements().filter( + (query) => !isWalk(query.sql) && !JSON.stringify(query).includes('hnsw.iterative_scan') + ) + ).toHaveLength(0) + const hydrated = dbChainMockFns.where.mock.calls.at(-1)![0] + expect( + hasMockCondition( + hydrated, + (node) => + node.type === 'inArray' && + node.column === schemaMock.embedding.id && + Array.isArray(node.values) && + node.values[0] === 'near' + ) + ).toBe(true) }) - it('fills a single result from the same candidate page when its nearest row loses access', async () => { - const execute = dbChainMockFns.execute.getMockImplementation()! - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query) - if (statement.sql.includes('AS unfilled')) return [{ unfilled: true }] - if (isPageStatement(statement.sql)) { - queueTableRows( - schemaMock.embedding, - ranked.filter((row) => row.id !== 'near') - ) - return ranked - } - return execute(query) - }) - - const rows = await handleVectorOnlySearch({ ...params, topK: 1 }) - + it('fills a single result from the same pool when its nearest row loses access', async () => { + /** The nearest row no longer hydrates under the full predicate; the next one in the pool does. */ + queueTableRows(schemaMock.embedding, [ranked[1]]) + const rows = await vectorSearch({ ...params, topK: 1 }) expect(rows.map((row) => row.id)).toEqual(['far']) expect(statements().filter((query) => isWalk(query.sql))).toHaveLength(1) - expect(getForConnectors).not.toHaveBeenCalled() }) it('does not turn a broad tag filter into exhaustive full-vector ranking', async () => { queueTableRows(schemaMock.embedding, ranked) - const rows = await handleTagAndVectorSearch({ + const rows = await vectorSearch({ ...params, structuredFilters: [{ tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'common' }], }) @@ -485,7 +502,6 @@ describe('workspace-scoped vector retrieval', () => { expect(JSON.stringify(candidate)).toContain('common') expect(JSON.stringify(candidate)).toContain(String(schemaMock.embedding.tag1)) expect(JSON.stringify(dbChainMockFns.where.mock.calls.at(-1)![0])).toContain('common') - expect(getForConnectors).not.toHaveBeenCalled() }) it('ranks all selected KBs together instead of capping how many results one KB can contribute', async () => { @@ -494,7 +510,7 @@ describe('workspace-scoped vector retrieval', () => { schemaMock.embedding, ranked.map((row) => ({ ...row, knowledgeBaseId: 'kb-1' })) ) - const rows = await handleVectorOnlySearch({ ...params, knowledgeBaseIds }) + const rows = await vectorSearch({ ...params, knowledgeBaseIds }) expect(rows.map((row) => row.id)).toEqual(['near', 'far']) expect(rows.every((row) => row.knowledgeBaseId === 'kb-1')).toBe(true) const candidateQueries = statements().filter((query) => isWalk(query.sql)) @@ -503,7 +519,7 @@ describe('workspace-scoped vector retrieval', () => { expect(dbChainMockFns.transaction).toHaveBeenCalledOnce() }) - it.each(['vector.candidate_search', 'vector.page', 'vector.sql'] as const)( + it.each(['vector.candidate_search', 'vector.sql'] as const)( 'reports a %s timeout as partial, not a complete empty search', async (failedStage) => { const query = SearchBudget.prototype.query @@ -525,7 +541,7 @@ describe('workspace-scoped vector retrieval', () => { } ) - it('shares the remaining deadline across candidate selection, reranking and hydration', async () => { + it('shares the remaining deadline across candidate selection and hydration', async () => { vi.spyOn(performance, 'now').mockReturnValue(0) const query = SearchBudget.prototype.query vi.spyOn(SearchBudget.prototype, 'query').mockImplementation(async function ( @@ -535,16 +551,15 @@ describe('workspace-scoped vector retrieval', () => { ) { const result = await (query.bind(this) as SearchBudget['query'])(stage, run) if (stage === 'vector.candidate_search') vi.spyOn(performance, 'now').mockReturnValue(60) - if (stage === 'vector.page') vi.spyOn(performance, 'now').mockReturnValue(80) return result }) queueTableRows(schemaMock.embedding, ranked) - await handleVectorOnlySearch({ ...params, budget: new SearchBudget('vector', 100) }) + await vectorSearch({ ...params, budget: new SearchBudget('vector', 100) }) expect( statements() .filter((query) => query.sql.includes('statement_timeout')) .map((query) => query.params[0]) - ).toEqual(['100', '40', '20']) + ).toEqual(['100', '40']) }) it('does not convert an unexpected candidate failure into partial retrieval', async () => { @@ -605,7 +620,7 @@ describe('workspace search filters before ranking', () => { } } - it.each([handleVectorOnlySearch, handleTagOnlySearch, handleTagAndVectorSearch])( + it.each([vectorSearch, tagSearch])( 'applies the full document scope to vector and tag searches', async (search) => { queueTableRows(schemaMock.embedding, [{ id: 'candidate' }]) @@ -620,244 +635,105 @@ describe('workspace search filters before ranking', () => { ) }) -describe('hydration follows ranked candidates', () => { - const identity: UserAccessScope = { +describe('document-decided pages follow ranked candidates', () => { + const reader: UserAccessScope = { kind: 'user', userId: 'reader', - tokens: ['org', 's:github-repositories:-:42'], - } - const candidate = (id: string, connectorId: string) => ({ - id, - documentId: `doc-${id}`, - connectorId, - distance: 0.1, - }) - const provider: KnowledgeAccessProvider = { - get: async () => identity, - getForConnectors: async () => identity, - getForDocuments: async () => identity, - liveSourceConnectorCondition: async () => null, + tokens: ['pub', 'u:reader@example.com', 'ws'], } const params: SearchParams = { - knowledgeBaseIds: ['org-index'], - searchIndexOnly: true, + knowledgeBaseIds: ['kb'], topK: 1, - access: identity, - accessProvider: provider, + access: reader, queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, distanceThreshold: 0.8, - structuredFilters: [{ tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }], - } - - const probePages: Array> = [] - const exactPages: Array> = [] - const candidatePages: Array> = [] - const rerankPages: Array>> = [] - const keywordPages: Array>> = [] - function queueRerank(rows: Array>) { - rerankPages.push(rows) - } - function queueCandidates(rows: Array<{ id: string }>, initialCount = rows.length) { - candidatePages.push(rows.map(({ id }) => ({ id, initial_count: initialCount }))) } + const walked = Array.from({ length: 400 }, (_, index) => ({ + id: `candidate-${index}`, + documentId: `doc-${index}`, + connectorId: null, + })) + const hydrations = () => + dbChainMockFns.where.mock.calls.filter(([condition]) => + hasMockCondition( + condition, + (node) => node.type === 'inArray' && node.column === schemaMock.embedding.id + ) + ) beforeEach(() => { resetDbChainMock() - probePages.length = 0 - exactPages.length = 0 - candidatePages.length = 0 - rerankPages.length = 0 - keywordPages.length = 0 - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('AS visible')) return candidatePages.shift() ?? [] - if (isPageStatement(statement)) return rerankPages.shift() ?? [] - if (statement.includes('WITH matched_keyword_chunks')) return keywordPages.shift() ?? [] - if (isExactRanking(statement)) return exactPages.shift() ?? [] - if (isProbeStatement(statement)) return probePages.shift() ?? [] - return [] - }) - }) - - it('bounds broad vector ranking before metadata and reorders relaxed candidates before trimming', async () => { - probePages.push( - Array.from({ length: 400 }, (_, index) => candidate(`probe-${index}`, 'allowed-source')) + dbChainMockFns.execute.mockImplementation(async (query) => + isWalk(render(query).sql) ? walked : [] ) - queueCandidates(Array.from({ length: 400 }, (_, index) => ({ id: `candidate-${index}` }))) - queueRerank([ - { ...candidate('far', 'allowed-source'), distance: 0.3 }, - { ...candidate('near', 'allowed-source'), distance: 0.1 }, - ...Array.from({ length: 18 }, (_, index) => candidate(`other-${index}`, 'allowed-source')), - ]) - queueTableRows(schemaMock.embedding, [ - { id: 'far', content: 'Far authorized passage', distance: 0.3 }, - { id: 'near', content: 'Near authorized passage', distance: 0.1 }, - ]) - const rows = await handleVectorOnlySearch({ - ...params, - structuredFilters: undefined, - topK: 1, - }) - expect(rows.map((row) => row.id)).toEqual(['near']) - const candidateQuery = dbChainMockFns.execute.mock.calls.find(([query]) => - render(query).sql.includes('AS visible') - )![0] - expect(render(candidateQuery).sql).toContain('CROSS JOIN LATERAL') - expect(render(candidateQuery).sql).toContain('LIMIT 1') - expect(JSON.stringify(candidateQuery)).toContain('required_clause') - expect(JSON.stringify(candidateQuery)).toContain('subvector') - const pageQuery = dbChainMockFns.execute.mock.calls.find(([query]) => - isPageStatement(render(query).sql) - )![0] - expect(JSON.stringify(pageQuery)).not.toContain(String(schemaMock.embedding.embedding)) }) - it('advances past candidate pages that hydrate no current readable content', async () => { - const probe = Array.from({ length: 400 }, (_, index) => ({ id: `probe-${index}` })) - const identities = Array.from({ length: 400 }, (_, index) => ({ id: `candidate-${index}` })) - probePages.push(probe) - queueCandidates(identities) - queueRerank( - Array.from({ length: 20 }, (_, index) => candidate(`candidate-${index}`, 'allowed-source')) - ) + it('advances past candidate slices that hydrate no current readable content', async () => { queueTableRows(schemaMock.embedding, []) - probePages.push(probe) - queueCandidates(identities) - queueRerank([candidate('selected', 'allowed-source')]) queueTableRows(schemaMock.embedding, [ - { id: 'selected', content: 'Current readable content', distance: 0.1 }, + { id: 'candidate-20', content: 'Current readable content', distance: 0.1 }, ]) - const rows = await handleVectorOnlySearch({ ...params, structuredFilters: undefined }) - expect(rows.map((row) => row.id)).toEqual(['selected']) - expect( - dbChainMockFns.execute.mock.calls.filter(([query]) => isPageStatement(render(query).sql)) - ).toHaveLength(2) + const rows = await vectorSearch(params) + expect(rows.map((row) => row.id)).toEqual(['candidate-20']) + expect(statements().filter((query) => isWalk(query.sql))).toHaveLength(1) + expect(hydrations()).toHaveLength(2) }) - it('sorts hydrated candidates across pages by their original-vector distance', async () => { - const probe = Array.from({ length: 400 }, (_, index) => - candidate(`probe-${index}`, 'allowed-source') - ) - probePages.push(probe) - queueCandidates(Array.from({ length: 400 }, (_, index) => ({ id: `candidate-${index}` }))) - queueRerank([ - { ...candidate('far', 'allowed-source'), distance: 0.7 }, - ...Array.from({ length: 19 }, (_, index) => candidate(`hidden-${index}`, 'allowed-source')), - ]) - queueTableRows(schemaMock.embedding, [{ id: 'far', content: 'Far result', distance: 0.7 }]) - probePages.push(probe) - queueCandidates(Array.from({ length: 400 }, (_, index) => ({ id: `candidate-${index}` }))) - queueRerank([ - candidate('near', 'allowed-source'), - candidate('nearer', 'allowed-source'), - candidate('far', 'allowed-source'), - ]) + it('sorts hydrated candidates across slices by their original-vector distance', async () => { + queueTableRows(schemaMock.embedding, [{ id: 'candidate-0', content: 'Far', distance: 0.7 }]) queueTableRows(schemaMock.embedding, [ - { id: 'near', content: 'Near result', distance: 0.2 }, - { id: 'nearer', content: 'Nearest result', distance: 0.1 }, + { id: 'candidate-20', content: 'Near', distance: 0.2 }, + { id: 'candidate-21', content: 'Nearest', distance: 0.1 }, ]) - - const rows = await handleVectorOnlySearch({ - ...params, - topK: 2, - structuredFilters: undefined, - }) - - expect(rows.map((row) => row.id)).toEqual(['nearer', 'near']) - expect( - dbChainMockFns.execute.mock.calls.filter(([query]) => isPageStatement(render(query).sql)) - ).toHaveLength(2) + const rows = await vectorSearch({ ...params, topK: 2 }) + expect(rows.map((row) => row.id)).toEqual(['candidate-21', 'candidate-20']) + /** A later slice never re-reads a candidate an earlier one already hydrated. */ expect( hasMockCondition( - dbChainMockFns.where.mock.calls.at(-1)![0], + hydrations().at(-1)![0], (node) => node.type === 'inArray' && node.column === schemaMock.embedding.id && Array.isArray(node.values) && - node.values.length === 2 && - !node.values.includes('far') + !node.values.includes('candidate-0') ) ).toBe(true) }) - it.each(['vector', 'tag-vector', 'tags', 'keyword'] as const)( - '%s ranks identifiers before verification and loads content under the full predicate', - async (mode) => { - const candidates = [candidate('selected', 'allowed-source')] - if (mode === 'vector' || mode === 'tag-vector') { - probePages.push([{ id: 'doc-selected' }]) - exactPages.push([{ id: 'selected' }]) - queueRerank(candidates) - } - if (mode === 'keyword') keywordPages.push(candidates) - if (mode === 'tags') queueTableRows(schemaMock.embedding, candidates) - queueTableRows(schemaMock.embedding, [{ id: 'selected', content: 'verified result' }]) - const rows = - mode === 'vector' - ? await handleVectorOnlySearch(params) - : mode === 'tag-vector' - ? await handleTagAndVectorSearch(params) - : mode === 'tags' - ? await handleTagOnlySearch(params) - : await executeKeywordSearch({ - ...params, - query: 'release', - queryVector: params.queryVector!, - }) - expect(rows).toEqual([{ id: 'selected', content: 'verified result' }]) - if (mode === 'keyword') { - const ranking = render(dbChainMockFns.execute.mock.calls[0][0]).sql - expect(ranking).toContain('matched_keyword_chunks AS MATERIALIZED') - expect(ranking).toContain('ORDER BY keyword_rank DESC, matched_keyword_chunks.id') - expect(ranking).not.toContain('<=>') - expect(ranking).not.toContain('"content"') - } else if (mode === 'tags') { - expect(Object.keys(dbChainMockFns.select.mock.calls[0][0]).sort()).toEqual( - ['id', 'documentId', 'connectorId'].sort() - ) - } else { - const ranking = statements().find((query) => isExactRanking(query.sql))! - /** The identities are one nested fragment; the mock renders it into the parameters. */ - expect(JSON.stringify(ranking)).toContain('connectorId') - expect(ranking.sql).not.toContain('"content"') - } - const rankingOrder = - mode === 'tags' - ? dbChainMockFns.select.mock.invocationCallOrder[0] - : dbChainMockFns.execute.mock.invocationCallOrder[0] - expect(rankingOrder).toBeLessThan(dbChainMockFns.select.mock.invocationCallOrder.at(-1)!) - const fullPredicate = dbChainMockFns.where.mock.calls.at(-1)![0] - expect(JSON.stringify(fullPredicate)).toContain('acl') - expect( - hasMockCondition( - fullPredicate, - (node) => - node.type === 'inArray' && - node.column === schemaMock.embedding.id && - Array.isArray(node.values) && - node.values.length === 1 && - node.values[0] === 'selected' - ) - ).toBe(true) + it('asks a gated source for live proof only once one of its candidates is read', async () => { + const getForConnectors = vi.fn(async () => reader) + const accessProvider: KnowledgeAccessProvider = { + get: async () => reader, + getForConnectors, + getForDocuments: async () => reader, + liveSourceConnectorCondition: async () => sql`true`, } - ) + const search = () => + retrieveKnowledgeSearch({ + knowledgeBaseIds: ['kb'], + topK: 1, + access: reader, + accessProvider, + searchMode: 'vector', + query: 'release', + queryVector: params.queryVector, + }) + queueTableRows(schemaMock.knowledgeConnector, [{ id: 'gated-source' }]) + queueTableRows(schemaMock.embedding, [{ id: 'candidate-0', content: 'Ungated', distance: 0.1 }]) + await search() + expect(getForConnectors).not.toHaveBeenCalled() - it('matches keyword chunks before the visibility predicate and ranks only what survives it', async () => { - keywordPages.push([candidate('selected', 'allowed-source')]) - queueTableRows(schemaMock.embedding, [{ id: 'selected', content: 'verified result' }]) - await executeKeywordSearch({ ...params, query: 'release', queryVector: params.queryVector! }) - const ranking = render(dbChainMockFns.execute.mock.calls[0][0]).sql - const matched = ranking.indexOf('matched_keyword_chunks AS MATERIALIZED') - const visible = ranking.indexOf('visible_keyword_documents AS MATERIALIZED') - expect(matched).toBeGreaterThanOrEqual(0) - expect(visible).toBeGreaterThan(matched) - expect(ranking.slice(matched, visible)).not.toContain('keyword_rank') - expect(ranking.slice(visible)).toContain('FROM matched_keyword_chunks INNER JOIN') - /** The predicate fragments are parameterized, so the restriction is read off the query tree. */ - const fragments = JSON.stringify(dbChainMockFns.execute.mock.calls[0][0]) - expect(fragments).toContain('= ANY (ARRAY(SELECT document_id FROM matched_keyword_chunks))') + dbChainMockFns.execute.mockImplementation(async (query) => + isWalk(render(query).sql) + ? walked.map((candidate, index) => + index === 0 ? { ...candidate, connectorId: 'gated-source' } : candidate + ) + : [] + ) + queueTableRows(schemaMock.knowledgeConnector, [{ id: 'gated-source' }]) + queueTableRows(schemaMock.embedding, [{ id: 'candidate-0', content: 'Gated', distance: 0.1 }]) + await search() + expect(getForConnectors).toHaveBeenCalledExactlyOnceWith(['gated-source'], undefined) }) it.each([undefined, 'gmail'])( @@ -865,871 +741,23 @@ describe('hydration follows ranked candidates', () => { async (source) => { const cancellation = new AbortController() cancellation.abort(new Error('Search cancelled')) - queueTableRows(schemaMock.embedding, [candidate('selected', 'allowed-source')]) await expect( - handleTagOnlySearch({ ...params, filters: { source }, signal: cancellation.signal }) + tagSearch({ + ...params, + structuredFilters: [ + { tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }, + ], + filters: { source }, + signal: cancellation.signal, + }) ).rejects.toThrow('Search cancelled') expect(dbChainMockFns.select).not.toHaveBeenCalled() } ) }) -describe('permitted-document planner', () => { - const reader: UserAccessScope = { - kind: 'user', - userId: 'reader', - tokens: ['u:reader@example.com'], - } - const workspace = { kind: 'workspace' as const, tokens: WORKSPACE_ACCESS_TOKENS } - const provider: KnowledgeAccessProvider = { - get: async () => reader, - getForConnectors: async () => reader, - getForDocuments: async () => reader, - liveSourceConnectorCondition: async () => null, - } - const params: SearchParams = { - knowledgeBaseIds: ['org-index'], - searchIndexOnly: true, - topK: 1, - access: reader, - accessProvider: provider, - queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, - distanceThreshold: 1, - } - const hit = (id: string, connectorId: string | null) => ({ - id, - documentId: `doc-${id}`, - connectorId, - distance: 0.1, - }) - const bounded = ( - ...documents: Array<{ id: string; connectorId: string | null }> - ): PermittedDocuments => ({ kind: 'bounded', documents }) - - let probeRows: Array<{ id: string | null; connectorId: string | null; saturated: boolean }> - let exactRows: Array<{ id: string }> - let traversedRows: Array<{ id: string; distance?: number }> - let rerankRows: Array> - let indexedSourceRows: Array<{ name: string; connectorId: string }> - let sourceExactRows: Array<{ id: string; distance: number }> - - beforeEach(() => { - resetDbChainMock() - probeRows = [] - exactRows = [] - traversedRows = [] - rerankRows = [] - sourceExactRows = [] - indexedSourceRows = [] - forgetIndexedVectorSources() - forgetSearchReach() - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('pg_index')) return indexedSourceRows - if (isWalk(statement)) return traversedRows - if (isPageStatement(statement)) return rerankRows - if (statement.includes('WITH readable_chunks')) return sourceExactRows - if (isExactRanking(statement)) return exactRows - if (isProbeStatement(statement)) return probeRows - return [] - }) - }) - - it('ranks a bounded permitted set exactly without walking the graph', async () => { - exactRows = [{ id: 'a' }] - rerankRows = [hit('a', null)] - queueTableRows(schemaMock.embedding, [hit('a', null)]) - const results = await handleVectorOnlySearch({ - ...params, - permitted: bounded({ id: 'doc-a', connectorId: null }, { id: 'doc-b', connectorId: 'src' }), - }) - expect(results.map((row) => row.id)).toEqual(['a']) - const sqls = statements().map((query) => query.sql) - expect(sqls.some((sql) => sql.includes('hnsw.iterative_scan'))).toBe(false) - expect(sqls.some((sql) => sql.includes('AS visible'))).toBe(false) - expect(sqls.some(isProbeStatement)).toBe(false) - const exact = JSON.stringify(statements().find((query) => isExactRanking(query.sql))) - expect(exact).toContain('doc-a') - expect(exact).toContain('doc-b') - }) - - it('walks the whole graph once for a caller whose reach is broad', async () => { - const eligibility = { workspace: [], admin: ['other-src'], members: ['member-src'] } - indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] - /** A full pool: the walk found as many readable neighbours as it was asked for. */ - traversedRows = Array.from({ length: 400 }, (_, i) => ({ id: `walked-${i}`, distance: 0.2 })) - rerankRows = [hit('walked-0', 'member-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await handleVectorOnlySearch({ - ...params, - permitted: { kind: 'unbounded', broad: true }, - accessPlan: { - connectors: eligibility, - observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, - memberSources: ['member-src'], - connectorTypes: new Map(), - uploads: true, - }, - }) - /** One walk over every source, scoped to the bases alone — no source is singled out. */ - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(JSON.stringify(walks[0])).not.toContain('"right":"member-src"') - expect(statements().some((query) => query.sql.includes('WITH readable_chunks'))).toBe(false) - }) - - it('walks an indexed source a bounded caller is a member of instead of ranking it exactly', async () => { - const eligibility = { workspace: [], admin: ['small-src'], members: ['member-src'] } - indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] - sourceExactRows = [{ id: 'small-hit', distance: 0.3, saturated: false }] - traversedRows = [{ id: 'walked-hit', distance: 0.2 }] - rerankRows = [hit('walked-hit', 'member-src'), hit('small-hit', 'small-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await handleVectorOnlySearch({ - ...params, - topK: 2, - permitted: bounded( - { id: 'doc-a', connectorId: 'member-src' }, - { id: 'doc-b', connectorId: 'small-src' } - ), - accessPlan: { - connectors: eligibility, - observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, - memberSources: ['member-src'], - connectorTypes: new Map(), - uploads: true, - }, - }) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(JSON.stringify(walks[0])).toContain('"right":"member-src"') - expect(statements().some((q) => isExactRanking(q.sql))).toBe(false) - }) - - it('walks the sliced sources when more documents are readable than one ranking may enumerate', async () => { - const eligibility = { workspace: [], admin: ['sliced-src'], members: [] } - /** The slice enumerates in no order, so a saturated one would rank an arbitrary subset. */ - sourceExactRows = [{ id: 'arbitrary-hit', distance: 0.4, saturated: true }] - traversedRows = [{ id: 'walked-hit', distance: 0.2 }] - rerankRows = [hit('walked-hit', 'sliced-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await handleVectorOnlySearch({ - ...params, - permitted: { kind: 'unbounded', broad: false }, - accessPlan: { - connectors: eligibility, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(JSON.stringify(walks[0])).toContain('sliced-src') - const reranked = JSON.stringify(statements().find((query) => isPageStatement(query.sql))) - expect(reranked).toContain('walked-hit') - expect(reranked).not.toContain('arbitrary-hit') - }) - - it('ranks uploaded documents even when every connector source is walked', async () => { - const eligibility = { workspace: [], admin: [], members: ['member-src'] } - indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] - sourceExactRows = [{ id: 'upload-hit', distance: 0.05, saturated: false }] - traversedRows = [{ id: 'walked-hit', distance: 0.2 }] - rerankRows = [hit('upload-hit', null), hit('walked-hit', 'member-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await handleVectorOnlySearch({ - ...params, - topK: 2, - permitted: { kind: 'unbounded', broad: false }, - accessPlan: { - connectors: eligibility, - observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, - memberSources: ['member-src'], - connectorTypes: new Map(), - uploads: true, - }, - }) - /** Uploads carry no connector, so their slice runs even with no sliced source beside them. */ - const exact = statements().filter((query) => query.sql.includes('WITH readable_chunks')) - expect(exact).toHaveLength(1) - expect(JSON.stringify(statements().find((q) => isPageStatement(q.sql)))).toContain('upload-hit') - }) - - it('confines keyword matching to the bounded permitted set', async () => { - await executeKeywordSearch({ - ...params, - topK: 1, - query: 'release', - queryVector: params.queryVector!, - permitted: bounded({ id: 'doc-a', connectorId: null }), - }) - const keyword = statements().find((query) => query.sql.includes('WITH matched_keyword_chunks'))! - /** The mock renders the whole WHERE as one parameter, so the restriction shows up in it. */ - expect(JSON.stringify(keyword)).toContain('doc-a') - }) - - describe('Tin keyword ranking for an unbounded caller', () => { - const unbounded: PermittedDocuments = { kind: 'unbounded' } - const keyword = (overrides: Partial[0]> = {}) => - executeKeywordSearch({ - ...params, - topK: 1, - query: 'release', - queryVector: params.queryVector!, - permitted: unbounded, - searchIndexOnly: true, - ...overrides, - }) - const tinStatements = () => - statements().filter((query) => query.sql.includes('ranked_tin_chunks')) - const ginStatements = () => - statements().filter((query) => query.sql.includes('WITH matched_keyword_chunks')) - let tinPages: Array<{ ranked: number; candidates: ReturnType[] }> - - beforeEach(() => { - mockResolveTinKeywordQuery.mockReset() - mockResolveTinKeywordQuery.mockResolvedValue('"releas"') - tinPages = [] - dbChainMockFns.execute.mockImplementation(async (query) => - render(query).sql.includes('ranked_tin_chunks') - ? [tinPages.shift() ?? { ranked: 0, candidates: [] }] - : [] - ) - }) - - it('ranks with Tin and checks access only on the top of that ranking', async () => { - tinPages = [{ ranked: 1500, candidates: [hit('a', null)] }] - queueTableRows(schemaMock.embedding, [{ ...hit('a', null), content: 'release notes' }]) - const results = await keyword() - expect(results.map((row) => row.id)).toEqual(['a']) - expect(mockResolveTinKeywordQuery).toHaveBeenCalledWith( - true, - 'release', - 'english', - params.budget - ) - expect(ginStatements()).toHaveLength(0) - expect(JSON.stringify(tinStatements()[0])).toContain('2000') - /** `==>` binds tighter than `||`, so the concatenated query must be parenthesized. */ - expect(tinStatements()[0].sql).toContain('==> (?)') - }) - - it('decides a row the fill has not reached on its document while the fill runs', async () => { - tinPages.push({ - ranked: 1, - candidates: [{ id: 'a', documentId: 'doc-a', connectorId: 'src-a' }], - }) - const execute = dbChainMockFns.execute.getMockImplementation()! - dbChainMockFns.execute.mockImplementation(async (query) => - render(query).sql.includes('AS unfilled') ? [{ unfilled: true }] : execute(query) - ) - await keyword({ - accessPlan: { - connectors: { workspace: [], admin: ['src-a'], members: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - const statement = JSON.stringify(tinStatements()[0]) - /** A row the fill has not reached (`acl IS NULL`), or a marked document's row, is decided on its document. */ - expect(statement).toContain(' IS NULL OR ') - expect(statement).toContain('knowledgeProjectionDirty.documentId') - expect(statement).toContain('EXISTS (') - expect(statement).toContain('ranked_tin_chunks.document_id') - }) - - it('widens the window for a broad resolved scope whose first page came back short', async () => { - tinPages = [ - { ranked: 2000, candidates: [] }, - { ranked: 4000, candidates: [hit('b', 'src-a')] }, - ] - queueTableRows(schemaMock.embedding, [{ ...hit('b', 'src-a'), content: 'release notes' }]) - const results = await keyword({ - permitted: { kind: 'unbounded', broad: true }, - accessPlan: { - connectors: { workspace: [], admin: ['src-a'], members: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - expect(results.map((row) => row.id)).toEqual(['b']) - const windows = tinStatements().map((query) => JSON.stringify(query)) - expect(windows).toHaveLength(2) - expect(windows[0]).toContain('2000') - expect(windows[1]).toContain('10000') - }) - - it('hydrates an oversized keyword page in slices and stops at the results it needs', async () => { - const ranked = Array.from({ length: 1000 }, (_, i) => hit(`k-${i}`, 'src-a')) - tinPages = [{ ranked: 20_000, candidates: ranked }] - /** The first slice — as many candidates as results are wanted — fills the page of results. */ - queueTableRows( - schemaMock.embedding, - ranked.slice(0, 20).map((row) => ({ ...row, content: 'release notes' })) - ) - const results = await keyword({ - topK: 20, - permitted: { kind: 'unbounded', broad: false }, - accessPlan: { - connectors: { workspace: [], admin: ['src-a'], members: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - expect(results).toHaveLength(20) - expect(tinStatements()).toHaveLength(1) - /** One hydration, of one slice — never the whole page. */ - const hydrations = dbChainMockFns.where.mock.calls.filter(([condition]) => - hasMockCondition( - condition, - (node) => node.type === 'inArray' && node.column === schemaMock.embedding.id - ) - ) - expect(hydrations).toHaveLength(1) - expect( - hasMockCondition( - hydrations[0][0], - (node) => - node.type === 'inArray' && - node.column === schemaMock.embedding.id && - Array.isArray(node.values) && - node.values.length === 20 - ) - ).toBe(true) - }) - - describe('a bounded set past the exact-ranking size', () => { - const large = Array.from({ length: PERMITTED_EXACT_DOCUMENT_LIMIT }, (_, index) => ({ - id: `doc-${index}`, - connectorId: 'src-a', - })) - const accessPlan = { - connectors: { workspace: [], admin: ['src-a'], members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - } - - it('ranks with Tin as a narrow reader, decided on the row', async () => { - tinPages = [{ ranked: 1500, candidates: [hit('a', 'src-a')] }] - queueTableRows(schemaMock.embedding, [{ ...hit('a', 'src-a'), content: 'release notes' }]) - const results = await keyword({ - permitted: { kind: 'bounded', documents: large }, - accessPlan, - }) - expect(results.map((row) => row.id)).toEqual(['a']) - expect(mockResolveTinKeywordQuery).toHaveBeenCalledTimes(1) - expect(tinStatements()).toHaveLength(1) - expect(JSON.stringify(tinStatements()[0])).toContain('2000') - expect(JSON.stringify(tinStatements()[0])).not.toContain('doc-4999') - expect(ginStatements()).toHaveLength(0) - }) - - it('leaves a later page short rather than resuming a different ranking at its offset', async () => { - /** The first page fills from Tin; hydration keeps half, so a second page is asked for. */ - const first = Array.from({ length: 40 }, (_, index) => hit(`t-${index}`, 'src-a')) - tinPages = [ - { ranked: 2000, candidates: first }, - { ranked: 2000, candidates: [] }, - { ranked: 20_000, candidates: [] }, - ] - queueTableRows( - schemaMock.embedding, - first.slice(0, 20).map((row) => ({ ...row, content: 'release notes' })) - ) - const results = await keyword({ - topK: 40, - permitted: { kind: 'bounded', documents: large }, - accessPlan, - }) - expect(results).toHaveLength(20) - expect(tinStatements()).toHaveLength(3) - expect(ginStatements()).toHaveLength(0) - }) - }) - - it('keeps GIN ranking when Tin is not ready or cannot express the query', async () => { - mockResolveTinKeywordQuery.mockResolvedValue(null) - await keyword() - expect(tinStatements()).toHaveLength(0) - expect(ginStatements()).toHaveLength(1) - }) - }) - - it('skips keyword SQL entirely when nothing is permitted', async () => { - expect( - await executeKeywordSearch({ - ...params, - query: 'release', - queryVector: params.queryVector!, - permitted: bounded(), - }) - ).toEqual([]) - expect(dbChainMockFns.execute).not.toHaveBeenCalled() - }) - - it('reads a user scope through its reachable documents and reports saturation', () => { - const user = render(visibleDocumentsQuery(['org-index'], [], reader)) - const userSql = user.sql - expect(userSql).toContain('WITH reach AS MATERIALIZED') - expect(userSql).toContain('reachable AS MATERIALIZED') - expect(userSql).toContain('FROM reachable AS') - expect(userSql).toContain('AS saturated') - /** - * Baseline tokens reach every tenant's org-wide, public, and uploaded documents, so both the - * count and the rows are confined to the requested bases, outside the fence around the index. - */ - const [reach, reachable] = userSql.split('reachable AS MATERIALIZED') - for (const cte of [reach, reachable.split('FROM reachable AS')[0]]) { - expect(cte).toMatch(/OFFSET 0\s*\) AS \?\s*WHERE \?/) - } - expect(JSON.stringify(user.params)).toContain('org-index') - const workspaceSql = render(visibleDocumentsQuery(['org-index'], [], workspace)).sql - expect(workspaceSql).not.toContain('reachable') - expect(workspaceSql).toContain('AS saturated') - }) - - it.each([ - [[{ id: null, connectorId: null, saturated: true }], 'unbounded'], - [[{ id: 'doc-a', connectorId: null, saturated: false }], 'bounded'], - ] as const)('resolves %j as %s', async (rows, kind) => { - probeRows = [...rows] - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: { ...reader, tokens: [`u:resolves-${kind}@example.com`] }, - }) - expect(permitted.kind).toBe(kind) - if (permitted.kind === 'bounded') - expect(permitted.documents).toEqual([{ id: 'doc-a', connectorId: null }]) - }) - - describe('saturated reach', () => { - const scope = (name: string): UserAccessScope => ({ - ...reader, - tokens: [`u:${name}@example.com`], - }) - const resolve = (access: UserAccessScope, knowledgeBaseIds = ['org-index']) => - resolvePermittedDocuments({ knowledgeBaseIds, access }) - const probes = () => statements().filter((query) => isProbeStatement(query.sql)).length - - it('counts a saturated reach against the broad bound once, and remembers the answer', async () => { - /** The index holds a million documents; the bound is a quarter of them. */ - const counts = { index: 1_000_000, reached: 250_000 } - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (isProbeStatement(statement)) return [{ id: null, connectorId: null, saturated: true }] - if (statement.includes(') reached')) return [{ n: counts.reached }] - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': counts.index } }] }] - return [] - }) - const reachCounts = () => statements().filter((query) => query.sql.includes(') reached')) - const broad = await resolve(scope('broad-reach')) - expect(broad).toEqual({ kind: 'unbounded', broad: true }) - expect(reachCounts()).toHaveLength(1) - expect(JSON.stringify(reachCounts()[0])).toContain('250000') - await resolve(scope('broad-reach')) - expect(reachCounts()).toHaveLength(1) - counts.reached = 120_000 - const narrow = await resolve(scope('narrow-reach')) - expect(narrow).toEqual({ kind: 'unbounded', broad: false }) - expect(reachCounts()).toHaveLength(2) - }) - - it('does not remember a saturated reach whose count ran out of time', async () => { - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (isProbeStatement(statement)) return [{ id: null, connectorId: null, saturated: true }] - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] - if (statement.includes(') reached')) - throw Object.assign(new Error('canceling statement due to statement timeout'), { - code: '57014', - }) - return [] - }) - const reachCounts = () => statements().filter((query) => query.sql.includes(') reached')) - const budget = () => new SearchBudget('vector', performance.now() + 10_000) - expect( - await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: scope('timed-saturated'), - budget: budget(), - }) - ).toEqual({ kind: 'unbounded', broad: true }) - expect(reachCounts()).toHaveLength(1) - /** The next search probes and counts again rather than trusting a reach that was never measured. */ - await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: scope('timed-saturated'), - budget: budget(), - }) - expect(probes()).toBe(2) - expect(reachCounts()).toHaveLength(2) - }) - - it('does not read an unanalyzed index as a reach of nothing', async () => { - /** The planner knows no rows yet, so the bound is zero and the count looked at nothing. */ - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (statement.includes('EXPLAIN')) return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 0 } }] }] - if (statement.includes(') reached')) return [{ n: 0 }] - return [] - }) - await expect( - resolveReach( - ['org-index'], - scope('unanalyzed'), - new SearchBudget('vector', performance.now() + 10_000), - { - connectors: { workspace: [], admin: [], members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - } - ) - ).resolves.toEqual({ kind: 'unbounded', broad: true }) - }) - - it('is remembered per set of bases and tokens', async () => { - probeRows = [{ id: null, connectorId: null, saturated: true }] - await resolve(scope('per-key')) - probeRows = [{ id: 'doc-a', connectorId: null, saturated: false }] - expect((await resolve(scope('per-key'), ['other-index'])).kind).toBe('bounded') - expect((await resolve(scope('per-key-other'))).kind).toBe('bounded') - expect(probes()).toBe(3) - }) - }) - - it('reports an exhausted vector budget as unbounded instead of failing both legs', async () => { - const budget = new SearchBudget('vector', performance.now() - 1) - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: reader, - budget, - }) - expect(permitted.kind).toBe('unbounded') - expect(budget.timedOut).toBe(true) - }) - - const liveSearch = { - knowledgeBaseIds: ['org-index'], - searchIndexOnly: true, - topK: 1, - searchMode: 'hybrid' as const, - query: 'release', - queryVector: params.queryVector!, - } - - it('never asks a source for live grants when the scope reads none', async () => { - const getForConnectors = vi.fn(async () => reader) - await retrieveKnowledgeSearch({ - ...liveSearch, - access: reader, - accessProvider: { ...provider, getForConnectors }, - }) - expect(getForConnectors).not.toHaveBeenCalled() - }) - - it('rebuilds the pool without a gated source the caller turns out not to hold', async () => { - queueTableRows(schemaMock.knowledgeConnector, [ - { - id: 'gated-src', - accessMode: 'admin', - connectorType: 'confluence', - githubRepository: false, - }, - ]) - /** - * The first pool is filled by the gated source alone; only a pool built without it — the - * exclusion carries the source id into the walk — reaches the accessible candidate. The - * projection is filled, so the walk carries each candidate's source and no page is read. - */ - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (statement.includes('AS unfilled')) return [{ unfilled: false }] - const rebuilt = JSON.stringify(query).includes('/* excluded sources */') - if (isWalk(statement)) - return Array.from({ length: 400 }, (_, i) => - i === 0 - ? rebuilt - ? hit('b', 'other-src') - : hit('a', 'gated-src') - : hit(`w-${i}`, rebuilt ? 'other-src' : 'gated-src') - ) - return [] - }) - queueTableRows(schemaMock.embedding, []) - queueTableRows(schemaMock.embedding, [hit('b', 'other-src')]) - /** No grants come back, so the gated source is denied. */ - const getForConnectors = vi.fn(async () => reader) - const result = await retrieveKnowledgeSearch({ - ...liveSearch, - searchMode: 'vector', - access: reader, - accessProvider: { ...provider, getForConnectors }, - }) - expect(getForConnectors).toHaveBeenCalledOnce() - expect(result.rows.map((row) => row.id)).toEqual(['b']) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(2) - expect(JSON.stringify(walks[0])).not.toContain('/* excluded sources */') - expect(JSON.stringify(walks[1])).toContain('/* excluded sources */') - expect(JSON.stringify(walks[1])).toContain('OR NOT (') - expect(statements().some((query) => isPageStatement(query.sql))).toBe(false) - }) - - it.each([false, true])( - 'excludes a denied source through its documents (searchIndexOnly=%s)', - async (searchIndexOnly) => { - queueTableRows(schemaMock.knowledgeConnector, [ - { - id: 'gated-src', - accessMode: 'admin', - connectorType: 'confluence', - githubRepository: false, - }, - ]) - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fill has not reached every row, so a denied source cannot be read off the row. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - /** The mock renders nested fragments as parameters, so the marker is found in the whole query. */ - const rebuilt = JSON.stringify(query).includes('/* excluded sources */') - if (isPageStatement(statement)) - return JSON.stringify(render(query).params).includes('"b"') - ? [hit('b', 'other-src')] - : [hit('a', 'gated-src')] - if (isWalk(statement)) - return Array.from({ length: 400 }, (_, i) => ({ - id: i === 0 ? (rebuilt ? 'b' : 'a') : `w-${i}`, - distance: 0.1, - })) - return [] - }) - queueTableRows(schemaMock.embedding, []) - queueTableRows(schemaMock.embedding, [hit('b', 'other-src')]) - const getForConnectors = vi.fn( - async () => reader - ) - const result = await retrieveKnowledgeSearch({ - ...liveSearch, - searchIndexOnly, - searchMode: 'vector', - access: reader, - accessProvider: { ...provider, getForConnectors }, - }) - expect(result.rows.map((row) => row.id)).toEqual(['b']) - expect(statements().filter((query) => query.sql.includes('AS unfilled'))).toHaveLength( - searchIndexOnly ? 1 : 0 - ) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(2) - expect(JSON.stringify(walks[0])).not.toContain('/* excluded sources */') - expect(JSON.stringify(walks[1])).toContain('NOT EXISTS (SELECT 1 FROM') - expect(JSON.stringify(walks[1])).toContain('/* excluded sources */') - } - ) -}) - -describe('filters on a resolved scope', () => { - const reader: UserAccessScope = { - kind: 'user', - userId: 'reader', - tokens: ['u:reader@example.com'], - } - const provider: KnowledgeAccessProvider = { - get: async () => reader, - getForConnectors: async () => reader, - getForDocuments: async () => reader, - liveSourceConnectorCondition: async () => null, - } - const params: SearchParams = { - knowledgeBaseIds: ['org-index'], - searchIndexOnly: true, - topK: 1, - access: reader, - accessProvider: provider, - queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, - distanceThreshold: 1, - } - const plan = (sources: string[] = ['src-a']) => ({ - connectors: { workspace: [], admin: sources, members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(sources.map((id) => [id, 'slack'])), - uploads: true, - }) - const hit = (id: string, connectorId: string | null) => ({ - id, - documentId: `doc-${id}`, - connectorId, - distance: 0.1, - }) - let probeRows: Array<{ id: string | null; connectorId: string | null; saturated: boolean }> - let traversedRows: Array<{ id: string; distance?: number }> - let rerankRows: Array> - let exactRows: Array<{ id: string }> - let indexedSourceRows: Array<{ name: string; connectorId: string }> - - beforeEach(() => { - resetDbChainMock() - forgetIndexedVectorSources() - forgetSearchReach() - probeRows = [] - traversedRows = [] - rerankRows = [] - exactRows = [] - indexedSourceRows = [] - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] - if (statement.includes('pg_index')) return indexedSourceRows - if (isExactRanking(statement)) return exactRows - if (statement.includes(') reached')) return [{ n: 250_000 }] - if (isProbeStatement(statement)) return probeRows - if (isWalk(statement)) return traversedRows - if (isPageStatement(statement)) return rerankRows - if (statement.includes('ranked_tin_chunks')) return [{ ranked: 0, candidates: [] }] - return [] - }) - }) - - it('enumerates the documents a date filter admits even when the reach is remembered', async () => { - probeRows = [{ id: 'doc-recent', connectorId: 'src-a', saturated: false }] - const budget = () => new SearchBudget('vector', performance.now() + 10_000) - await resolveReach(['org-index'], reader, budget(), plan()) - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: reader, - filters: { modifiedAfter: '2026-09-13T00:00:00.000Z' }, - budget: budget(), - accessPlan: plan(), - }) - expect(permitted).toEqual({ - kind: 'bounded', - documents: [{ id: 'doc-recent', connectorId: 'src-a' }], - }) - const probes = statements().filter((query) => query.sql.includes('AS saturated')) - expect(probes).toHaveLength(1) - /** Filter first, over the date index: never the reach count that reports a broad reader saturated. */ - expect(probes[0].sql).not.toContain('WITH reach') - /** An index-driven probe earns its own budget: a window at the document limit fits inside it. */ - const deadlines = statements().filter((query) => query.sql.includes('statement_timeout')) - expect(deadlines.at(-1)?.params[0]).toBe('1500') - expect(JSON.stringify(probes[0])).toContain('"type":"gte"') - }) - - describe('a bounded set past the exact-ranking size', () => { - const large = Array.from({ length: PERMITTED_EXACT_DOCUMENT_LIMIT }, (_, index) => ({ - id: `doc-${index}`, - connectorId: 'src-a', - })) - const walked = Array.from({ length: 200 }, (_, index) => hit(`w-${index}`, 'src-a')) - const search = (documents: typeof large) => - handleVectorOnlySearch({ - ...params, - permitted: { kind: 'bounded', documents }, - accessPlan: plan(), - }) - beforeEach(() => { - const execute = dbChainMockFns.execute.getMockImplementation()! - dbChainMockFns.execute.mockImplementation(async (query) => { - /** The projection is filled, so a walk decides readability on the row. */ - if (render(query).sql.includes('AS unfilled')) return [{ unfilled: false }] - return execute(query) - }) - }) - - it('walks the graph on the row instead of ranking every chunk of the set', async () => { - traversedRows = walked - queueTableRows(schemaMock.embedding, [walked[0]]) - expect((await search(large)).map((row) => row.id)).toEqual(['w-0']) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(statements().filter((query) => isExactRanking(query.sql))).toHaveLength(0) - /** Readability rides on the row through the plan; the set's identifiers never cross the wire. */ - expect(JSON.stringify(walks[0])).not.toContain('doc-4999') - expect(JSON.stringify(walks[0])).toContain('src-a') - }) - }) - - it("estimates a filter under the leg's deadline and walks when the estimate runs out of time", async () => { - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('EXPLAIN') && JSON.stringify(render(query)).includes('"type":"gte"')) - throw Object.assign(new Error('canceling statement due to statement timeout'), { - code: '57014', - }) - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] - if (statement.includes(') reached')) return [{ n: 250_000 }] - if (isWalk(statement)) return traversedRows - if (isPageStatement(statement)) return rerankRows - return [] - }) - const result = await retrieveKnowledgeSearch({ - ...params, - accessProvider: provider, - searchMode: 'vector', - query: 'release', - filters: { modifiedAfter: '2026-09-13T00:00:00.000Z' }, - }) - /** The estimate ran inside a deadline statement; its own timeout chose the walk and cost the leg nothing. */ - const estimateAt = statements().findIndex((query) => - JSON.stringify(query).includes('"type":"gte"') - ) - expect(estimateAt).toBeGreaterThan(0) - expect(statements()[estimateAt - 1].sql).toContain('statement_timeout') - expect(statements().filter((query) => query.sql.includes('AS saturated'))).toHaveLength(0) - expect(statements().filter((query) => isWalk(query.sql))).toHaveLength(1) - expect(result.retrieval.status).toBe('complete') - }) - - it('keeps the default scan while the projection still holds unfilled rows', async () => { - traversedRows = [{ id: 'a' }] - rerankRows = [hit('a', 'src-a')] - queueTableRows(schemaMock.embedding, rerankRows) - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (isWalk(statement)) return traversedRows - if (isPageStatement(statement)) return rerankRows - return [] - }) - await handleVectorOnlySearch({ - ...params, - permitted: { kind: 'unbounded', broad: true }, - accessPlan: plan(), - }) - /** An unfilled row is decided through its document, so the walk keeps the cap sized for that. */ - const caps = statements() - .filter((query) => query.sql.includes('hnsw.max_scan_tuples')) - .map((query) => query.params.find((param) => param === '20000' || param === '100000')) - expect(caps.at(-1)).toBe('20000') - }) -}) - /** A retrieval row with only the fields fusion reads; the tag slots are irrelevant here. */ -function searchRow(id: string, distance: number): SearchResult { +function searchRow(id: string, distance = 0.1): SearchResult { return { id, content: id, @@ -1758,7 +786,7 @@ function searchRow(id: string, distance: number): SearchResult { } } -describe('fuseByReciprocalRank exposes the ordering key', () => { +describe('fuseByReciprocalRank', () => { it('stamps each row with the fused score it is ordered by and a 1-based rank, leaving similarity alone', () => { const lexical = [searchRow('a', 0.5), searchRow('b', 0.2)] const vector = [searchRow('b', 0.2), searchRow('c', 0.1)] @@ -1771,9 +799,103 @@ describe('fuseByReciprocalRank exposes the ordering key', () => { expect(fused[1].rankScore).toBeCloseTo(1 / (RRF_K + 1), 12) expect(fused[2].rankScore).toBeCloseTo(1 / (RRF_K + 2), 12) /** The order follows rankScore, which the cosine distance alone would not explain: c is the nearest chunk yet ranks last. */ - expect(fused.map((row) => row.rankScore)).toEqual( - [...fused.map((row) => row.rankScore)].sort((x, y) => (y ?? 0) - (x ?? 0)) - ) expect(fused.map((row) => row.distance)).toEqual([0.2, 0.5, 0.1]) }) + + it('ranks a row found by both legs above rows found by only one', () => { + const shared = searchRow('shared') + const vectorOnly = searchRow('vector-only') + const keywordOnly = searchRow('keyword-only') + + const fused = fuseByReciprocalRank( + [ + [vectorOnly, shared], + [keywordOnly, shared], + ], + 10 + ) + + /** `shared` is credited to both legs, so the following tie is even and resolves to the earliest list. */ + expect(fused.map((r) => r.id)).toEqual(['shared', 'vector-only', 'keyword-only']) + }) + + it('dedupes by chunk id, keeping the first occurrence', () => { + const fromVector = searchRow('chunk-1', 0.2) + const fromKeyword = { ...searchRow('chunk-1', 0.9), content: 'stale copy' } + + const fused = fuseByReciprocalRank([[fromVector], [fromKeyword]], 10) + + expect(fused).toHaveLength(1) + expect(fused[0].content).toBe('chunk-1') + expect(fused[0].distance).toBe(0.2) + }) + + it('scores by reciprocal rank so a deep double hit beats a shallow single hit', () => { + const deepShared = searchRow('deep-shared') + const topSingle = searchRow('top-single') + + /** `deep-shared` sits at rank 2 in both legs, `top-single` at rank 1 in one leg only. */ + const fused = fuseByReciprocalRank( + [ + [topSingle, deepShared], + [searchRow('other'), deepShared], + ], + 10 + ) + + expect(fused[0].id).toBe('deep-shared') + }) + + it('does not let the first leg starve the second at small topK', () => { + const lexicalOnly = searchRow('lexical-only') + const vectorOnly = searchRow('vector-only') + + /** + * Rank 1 in each leg scores identically. Ordering by score alone would + * always emit the first list's row, so a `topK: 1` hybrid search would + * return exactly what vector-only search already returned. + */ + expect(fuseByReciprocalRank([[lexicalOnly], [vectorOnly]], 1).map((r) => r.id)).toEqual([ + 'lexical-only', + ]) + expect(fuseByReciprocalRank([[lexicalOnly], [vectorOnly]], 2).map((r) => r.id)).toEqual([ + 'lexical-only', + 'vector-only', + ]) + }) + + it('interleaves tied ranks so neither leg monopolizes the head', () => { + const legA = [searchRow('a1'), searchRow('a2'), searchRow('a3')] + const legB = [searchRow('b1'), searchRow('b2'), searchRow('b3')] + + expect(fuseByReciprocalRank([legA, legB], 6).map((r) => r.id)).toEqual([ + 'a1', + 'b1', + 'a2', + 'b2', + 'a3', + 'b3', + ]) + }) + + it('does not let a shared top hit evict the lexical-only row at topK 2', () => { + const shared = searchRow('shared') + const lexicalOnly = searchRow('lexical-only') + const vectorOnly = searchRow('vector-only') + + /** + * `shared` is rank 1 in both legs. Crediting it to only one leg would + * leave the round-robin owing the other leg the remaining slot, evicting + * the row that only the shared hit's leg could produce. + */ + const fused = fuseByReciprocalRank( + [ + [shared, lexicalOnly], + [shared, vectorOnly], + ], + 2 + ) + + expect(fused.map((r) => r.id)).toEqual(['shared', 'lexical-only']) + }) }) diff --git a/apps/sim/lib/knowledge/search/queries.ts b/apps/sim/lib/knowledge/search/queries.ts index 5362d51d4cd..0002fd9bb6b 100644 --- a/apps/sim/lib/knowledge/search/queries.ts +++ b/apps/sim/lib/knowledge/search/queries.ts @@ -1,2524 +1,543 @@ import { db } from '@sim/db' -import { SOURCE_ACL_PROJECTIONS, type SourceAclProjection } from '@sim/db/knowledge-projection' -import { - document, - embedding, - embeddingKeywordSearch, - embeddingKeywordTin, - embeddingSearch, - knowledgeConnector, -} from '@sim/db/schema' -import { createLogger } from '@sim/logger' +import { document, embedding, embeddingSearch, knowledgeConnector } from '@sim/db/schema' import { sha256Hex } from '@sim/security/hash' -import { getErrorMessage, getPostgresErrorCode } from '@sim/utils/errors' -import { and, eq, gte, inArray, isNull, lte, type SQL, sql } from 'drizzle-orm' -import type { AnyPgColumn } from 'drizzle-orm/pg-core' +import { compareStrings } from '@sim/utils/string' +import { and, eq, inArray, or, type SQL, sql } from 'drizzle-orm' import { LRUCache } from 'lru-cache' -import { mapWithConcurrency } from '@/lib/core/utils/concurrency' -import { resolveSearchAccessPlan } from '@/lib/knowledge/access/connector-eligibility' import { knowledgeAccessCondition, - knowledgeAclOverlapCondition, - knowledgeCandidateAccessConditionForConnectors, knowledgeMetadataCandidateAccessCondition, - projectionCandidateAccessCondition, - projectionDecidedOnDocument, - projectionPending, - restrictSearchAccessPlan, - type SearchAccessPlan, - textArrayLiteral, } from '@/lib/knowledge/access/predicate' +import type { KnowledgeAccessProvider, KnowledgeAccessScope } from '@/lib/knowledge/access/types' import { - type ConfluenceSiteReadGrant, - type GitHubInstallationReadGrant, - type KnowledgeAccessProvider, - type KnowledgeAccessScope, - MAX_KNOWLEDGE_ACCESS_CANDIDATES, -} from '@/lib/knowledge/access/types' -import type { KbEmbeddingDimensions } from '@/lib/knowledge/embedding-models' -import { - type RetrievalLeg, runSearchQuery, SEARCH_RETRIEVAL_BUDGET_MS, SearchBudget, - SearchDeadlineError, - type SearchExecutor, - sessionSettingsStatement, } from '@/lib/knowledge/search/budget' import { - annotateSearchDiagnostics, - measureSearchStage, - recordSearchStageDuration, - type SearchStage, -} from '@/lib/knowledge/search/diagnostics' -import { workspaceSearchFilterConditions } from '@/lib/knowledge/search/filter-conditions' + candidateDocumentConditions, + directVisibleDocumentsQuery, + excludeSearchSources, + FTS_CONFIG, + getSearchResultFields, + getVisibilityConditions, + hydrateSearchCandidates, + type KeywordSearchParams, + type KnowledgeQueryVector, + type LiveSourceAccess, + liveSourceAccessForConnectors, + probeVisibleDocuments, + type RetrievalLegs, + type SearchParams, + type SearchReadCandidate, + type SearchResult, + selectAuthorizedSearchResults, +} from '@/lib/knowledge/search/candidates' +import { annotateSearchDiagnostics, measureSearchStage } from '@/lib/knowledge/search/diagnostics' import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' +import { keywordCandidateRankingQuery } from '@/lib/knowledge/search/keyword-ranking' import { applyRecencyBoost, RRF_K } from '@/lib/knowledge/search/recency' -import { indexedVectorSources } from '@/lib/knowledge/search/source-vector-indexes' -import { resolveTinKeywordQuery } from '@/lib/knowledge/search/tin-keyword' import { - coerceTagFilterValue, - escapeLikePattern, - uncompilableTagFilterError, -} from '@/lib/knowledge/tags/utils' -import type { StructuredFilter } from '@/lib/knowledge/types' + getStructuredTagFilters, + selectAuthorizedTagResults, +} from '@/lib/knowledge/search/tag-filters' import { - embeddingCandidateDimensions, - embeddingCandidateDistance, - embeddingDistance, -} from '@/lib/knowledge/vector-columns' - -const logger = createLogger('KnowledgeSearchQueries') - -/** SQLSTATE for an unrecognised configuration parameter — pgvector older than 0.8. */ -const UNDEFINED_OBJECT_SQLSTATE = '42704' -/** Bound candidate pages retained while live permissions are checked. */ -const MAX_AUTHORIZED_SEARCH_CANDIDATES = 20_000 -/** - * Approximate iterative-visit threshold for a permission-starved graph walk. It excludes - * pgvector's initial beam, so it only takes effect once `ResumeScanItems` starts widening. - * - * It must therefore stay roughly an order of magnitude above `ef_search`, or the first beam - * already exhausts the tuple budget and the scan stops before it can iterate at all — pgvector's - * maintainer says as much in pgvector#912. Measured on a production-shaped corpus, the previous - * pairing of a 1,000-wide beam against a 1,000-tuple budget returned fewer candidates than a - * narrower beam allowed to iterate, and spent longer inside the one uninterruptible beam. - */ -const CANDIDATE_HNSW_MAX_SCAN_TUPLES = '20000' -/** - * How far a walk that decides readability on the row may go before giving up: a cap, not a target, - * since the scan stops as soon as the limit is met. The default cap was sized for a walk that looked - * a document up per visited tuple; on the row a tuple costs a fraction of that, so a caller whose - * neighbourhood is mostly unreadable can be carried past it for tens of milliseconds rather than - * left with what the neighbourhood happened to hold. - */ -const ON_ROW_WALK_SCAN_TUPLES = 100_000 - -/** - * How far a walk may go when readability is on the row: the on-row cap, unless the walk still - * has to ask the document about tuples — a tag or date filter, or rows the source and ACL fill - * has not reached yet — in which case such a tuple costs what it did before the columns were mirrored, - * and the default cap keeps a walk through a mostly-excluded neighbourhood at a short answer - * rather than a missed deadline. - */ -function onRowWalkScanTuples( - documentCondition: SQL | undefined, - projectionFilled: boolean -): number { - return documentCondition === undefined && projectionFilled - ? ON_ROW_WALK_SCAN_TUPLES - : Number(CANDIDATE_HNSW_MAX_SCAN_TUPLES) -} - -/** How long a fully filled projection is taken on trust before its unfilled rows are looked for again. */ -const PROJECTION_FILLED_TTL_MS = 60_000 - -/** - * Whether the ranking projection still holds rows the source and ACL fill has not reached. Read off the - * unfilled-rows index in milliseconds and remembered briefly: the answer only ever changes once. - * - * The read asks for the last unfilled row by id, not whether one exists: an `EXISTS` drops its - * order and limit, and while most rows are unfilled the planner expects a sequential scan to - * meet one at once, then walks the whole projection when the unfilled rows sit past the filled - * ones. Ordered by id and capped at one row, the read can only be the partial index, whose - * last entry is the row the fill reaches last. - */ -const projectionFilled = new LRUCache< - SourceAclProjection, - boolean, - { budget: SearchBudget | undefined; stage: SearchStage } ->({ - max: SOURCE_ACL_PROJECTIONS.length, - ttl: PROJECTION_FILLED_TTL_MS, - /** - * The read that misses the cache is the search's own, under its budget like every other read - * of the leg, and the searches that miss together share it. A read that fails is not - * remembered: it answers unfilled, the slower and safe form, and the next search reads again. - */ - fetchMethod: async (projection, _stale, { context }) => { - const table = projection === 'embedding_search' ? embeddingSearch : embeddingKeywordTin - try { - const [row] = await runSearchQuery(context.budget, context.stage, (executor) => - executor.execute<{ unfilled: boolean }>(sql` - SELECT ( - SELECT ${table.id} FROM ${table} WHERE ${table.acl} IS NULL - ORDER BY ${table.id} DESC LIMIT 1 - ) IS NOT NULL AS unfilled`) - ) - return !row?.unfilled - } catch { - return undefined - } - }, -}) - -/** Whether every row of the projection carries its mirrored source and ACL; unknown counts as not yet. */ -async function isProjectionFilled( - projection: SourceAclProjection, - stage: SearchStage, - budget: SearchBudget | undefined -): Promise { - return (await projectionFilled.fetch(projection, { context: { budget, stage } })) ?? false -} - -/** Forgets whether the projections were filled; the memo is per process and otherwise expires on its own. */ -export function forgetProjectionFilled(): void { - projectionFilled.clear() -} - -/** - * Beam width per iteration. A beam is the granularity of cancellation: pgvector calls - * `CHECK_FOR_INTERRUPTS` only while building an index, never inside `hnswgettuple`, so neither - * `statement_timeout` nor a cancellation request can interrupt one. A narrower beam that iterates - * therefore bounds the leg's uninterruptible floor as well as widening its reach. - */ -const CANDIDATE_HNSW_EF_SEARCH = '200' -const CANDIDATE_HNSW_SCAN_MEM_MULTIPLIER = '2' -/** - * Candidates one walk gathers: the pages the search has asked for so far and as many again, in - * case the full read predicate refuses some. The walk ends as soon as it has them, so a pool the - * size of a page ends long before one sized for a rerank; a pool the pages outrun is walked again, - * wider. The ceiling bounds the widest walk. - */ -const VECTOR_CANDIDATE_POOL_MIN = 200 -const MAX_VECTOR_CANDIDATES = 1600 - -/** The pool a search needs to serve `needed` candidates, at least twice the last pool. */ -export function vectorCandidatePoolLimit(needed: number, previous: number | undefined): number { - return Math.min( - MAX_VECTOR_CANDIDATES, - Math.max(VECTOR_CANDIDATE_POOL_MIN, needed * 2, (previous ?? 0) * 2) - ) -} -/** - * The probe's share of the leg. It ranks nothing, so it must never be why the leg misses its - * own deadline. - * - * Its share comes out of what the rescue can claim from the narrowest live budget, - * `DIRECT_SEARCH_VECTOR_BUDGET_MS`, before live authorization, hydration and the exact rerank - * need the rest. - */ -const VECTOR_PROBE_BUDGET_MS = 600 -/** - * What one document costs the probe, measured on a corpus shaped like a search index under - * comparable cache pressure: the access predicate, evaluated once per document. - */ -const VECTOR_PROBE_MICROSECONDS_PER_DOCUMENT = 6 -/** - * What a filter-first probe may spend: it reads the filtered documents off their own index and - * tests each one's access, bounded by the same document limit, and measures around 2 µs per - * document to enumerate plus the access test — a window at the limit fits with room. Its result - * is ranked exactly, at a cost that is predictable where a walk through a mostly-excluded - * neighbourhood is not. - */ -const FILTERED_PROBE_BUDGET_MS = 1500 -/** - * Documents the probe enumerates before it concludes the permitted set is too large to rank - * exactly. Derived so that reaching it is what spends the probe's budget, rather than a separate - * number that a change to that budget could silently invalidate. - * - * It bounds the rescue's second step too: exact ranking of the `halfvec` projection measures at - * around half the probe's per-document cost, so a permitted set within this bound is affordable - * by construction. - */ -export const VECTOR_PROBE_DOCUMENT_LIMIT = Math.round( - (VECTOR_PROBE_BUDGET_MS * 1000) / VECTOR_PROBE_MICROSECONDS_PER_DOCUMENT -) - -/** - * Documents a bounded permitted set may hold before ranking it exactly costs more than walking - * the graph on the row. Exact ranking reads every chunk of the set, a few per document, where an - * on-row walk reads at most {@link CANDIDATE_HNSW_MAX_SCAN_TUPLES} tuples; at this size the two - * meet. A set past it is walked first and ranked exactly only if the walk cannot fill its pool, so - * its recall is never below the exact ranking's and its usual cost is the walk's. The same size - * turns the keyword leg from a read of the set's every chunk into a ranking decided on the row. - */ -export const PERMITTED_EXACT_DOCUMENT_LIMIT = 5_000 - -/** How long to stop trying the iterative-scan settings after the server rejected them. */ -const HNSW_SETTINGS_UNSUPPORTED_RETRY_MS = 10 * 60 * 1000 - -let hnswSettingsUnsupportedUntil = 0 - -/** - * Shared HNSW indexes can be selective on KB scope, access, or tags, even for - * small workspace searches. Iterative scans keep looking within a bounded - * tuple budget. A transaction keeps the settings local under pooled connections; - * older extensions retry without tuning until the compatibility cooldown expires. - */ -async function withVectorScanSettings( - run: (executor: SearchExecutor) => Promise, - budget: SearchBudget | undefined, - stage: SearchStage, - maxScanTuples: number = Number(CANDIDATE_HNSW_MAX_SCAN_TUPLES) -): Promise { - const untuned = () => runSearchQuery(budget, stage, run) - if (Date.now() < hnswSettingsUnsupportedUntil) return untuned() - const acquireStarted = performance.now() - const settings = [ - sql`set_config('hnsw.iterative_scan', 'relaxed_order', true)`, - sql`set_config('hnsw.max_scan_tuples', ${String(maxScanTuples)}, true)`, - sql`set_config('hnsw.ef_search', ${CANDIDATE_HNSW_EF_SEARCH}, true)`, - sql`set_config('hnsw.scan_mem_multiplier', ${CANDIDATE_HNSW_SCAN_MEM_MULTIPLIER}, true)`, - ] - /** Only a failure while the settings are being applied says the extension lacks them. */ - let applyingSettings = true - try { - /** Under a budget the settings ride in the deadline statement; alone they are one of their own. */ - if (budget) { - return await budget.query( - stage, - (tx) => { - applyingSettings = false - return run(tx) - }, - settings - ) - } - return await db.transaction(async (tx) => { - recordSearchStageDuration('vector.connection_acquire', performance.now() - acquireStarted) - await measureSearchStage('vector.settings', () => - tx.execute(sessionSettingsStatement(settings)) - ) - applyingSettings = false - return run(tx) - }) - } catch (error) { - if (!applyingSettings || getPostgresErrorCode(error) !== UNDEFINED_OBJECT_SQLSTATE) throw error - hnswSettingsUnsupportedUntil = Date.now() + HNSW_SETTINGS_UNSUPPORTED_RETRY_MS - logger.warn('pgvector iterative scan is unavailable; vector legs run without it', { - error: getErrorMessage(error), - }) - return untuned() - } -} - -export interface SearchResult { - id: string - content: string - documentId: string - chunkIndex: number - // Text tags - tag1: string | null - tag2: string | null - tag3: string | null - tag4: string | null - tag5: string | null - tag6: string | null - tag7: string | null - // Number tags (5 slots) - number1: number | null - number2: number | null - number3: number | null - number4: number | null - number5: number | null - // Date tags (2 slots) - date1: Date | null - date2: Date | null - // Boolean tags (3 slots) - boolean1: boolean | null - boolean2: boolean | null - boolean3: boolean | null - /** - * The score this row's position in the returned list comes from: the - * reciprocal-rank-fusion score in hybrid mode, the cosine similarity - * (`1 - distance`) in vector mode, and 1 for a tag-only search. Stamped by - * retrieval on every row it returns; absent on rows straight from a single - * retrieval leg. Recency may reorder rows without changing this score. - */ - rankScore?: number - /** 1-based position in the returned order, stamped alongside `rankScore`. */ - rank?: number - distance: number - knowledgeBaseId: string - /** When the source last changed the document; NULL for uploads and sources that do not say. */ - sourceModifiedAt: Date | null - filename: string - sourceUrl: string | null - /** The connector type behind the document; NULL for an upload. */ - connectorType: string | null -} - -/** - * A query embedding and the width it was produced at. The two travel together - * because the width selects both the pgvector column the comparison reads and - * the index form it has to be written in; a vector without it cannot be - * compared against anything. - */ -export interface KnowledgeQueryVector { - /** JSON array literal of the embedding, in pgvector's text input format. */ - vector: string - dimensions: KbEmbeddingDimensions - model: string -} - -export interface SearchParams { - knowledgeBaseIds: string[] - topK: number - /** What the caller may read; every leg applies it. Required so no leg can be written without it. */ - access: KnowledgeAccessScope - accessProvider?: KnowledgeAccessProvider - liveSourceAccess?: LiveSourceAccess - signal?: AbortSignal - budget?: SearchBudget - structuredFilters?: StructuredFilter[] - filters?: WorkspaceSearchFilters - queryVector?: KnowledgeQueryVector - distanceThreshold?: number - /** Resolved once per user-scoped search; absent for resolved scopes and explicit documents. */ - permitted?: PermittedDocuments - /** Connector state resolved once per search, so no candidate re-derives it. */ - accessPlan?: SearchAccessPlan - /** Every searched base is a Sim Search index; ordinary KBs use document-backed pages. */ - searchIndexOnly?: boolean -} - -/** All valid tag slot keys */ -const TAG_SLOT_KEYS = [ - // Text tags (7 slots) - 'tag1', - 'tag2', - 'tag3', - 'tag4', - 'tag5', - 'tag6', - 'tag7', - // Number tags (5 slots) - 'number1', - 'number2', - 'number3', - 'number4', - 'number5', - // Date tags (2 slots) - 'date1', - 'date2', - // Boolean tags (3 slots) - 'boolean1', - 'boolean2', - 'boolean3', -] as const - -type TagSlotKey = (typeof TAG_SLOT_KEYS)[number] - -function isTagSlotKey(key: string): key is TagSlotKey { - return TAG_SLOT_KEYS.includes(key as TagSlotKey) -} - -/** Common fields selected for search results */ -const getSearchResultFields = (distanceExpr: SQL | SQL.Aliased) => ({ - id: embedding.id, - content: embedding.content, - documentId: embedding.documentId, - chunkIndex: embedding.chunkIndex, - // Text tags - tag1: embedding.tag1, - tag2: embedding.tag2, - tag3: embedding.tag3, - tag4: embedding.tag4, - tag5: embedding.tag5, - tag6: embedding.tag6, - tag7: embedding.tag7, - // Number tags (5 slots) - number1: embedding.number1, - number2: embedding.number2, - number3: embedding.number3, - number4: embedding.number4, - number5: embedding.number5, - // Date tags (2 slots) - date1: embedding.date1, - date2: embedding.date2, - // Boolean tags (3 slots) - boolean1: embedding.boolean1, - boolean2: embedding.boolean2, - boolean3: embedding.boolean3, - distance: distanceExpr, - knowledgeBaseId: embedding.knowledgeBaseId, - sourceModifiedAt: document.sourceModifiedAt, - filename: document.filename, - sourceUrl: document.sourceUrl, - connectorType: knowledgeConnector.connectorType, -}) - -/** - * Build a single SQL condition for a filter - */ -function buildFilterCondition(filter: StructuredFilter, embeddingTable: any) { - const { tagSlot, fieldType, operator, value, valueTo } = filter - - if (!isTagSlotKey(tagSlot)) { - return null - } - - const column = embeddingTable[tagSlot] - if (!column) return null - - if (fieldType === 'text') { - const coerced = coerceTagFilterValue(value, 'text') - if (!coerced.ok) return null - const stringValue = coerced.value as string - const escaped = escapeLikePattern(stringValue) - switch (operator) { - case 'eq': - return sql`LOWER(${column}) = LOWER(${stringValue})` - case 'neq': - return sql`LOWER(${column}) != LOWER(${stringValue})` - case 'contains': - return sql`LOWER(${column}) LIKE LOWER(${`%${escaped}%`}) ESCAPE '\\'` - case 'not_contains': - return sql`LOWER(${column}) NOT LIKE LOWER(${`%${escaped}%`}) ESCAPE '\\'` - case 'starts_with': - return sql`LOWER(${column}) LIKE LOWER(${`${escaped}%`}) ESCAPE '\\'` - case 'ends_with': - return sql`LOWER(${column}) LIKE LOWER(${`%${escaped}`}) ESCAPE '\\'` - default: - return sql`LOWER(${column}) = LOWER(${stringValue})` - } - } - - if (fieldType === 'number') { - const coerced = coerceTagFilterValue(value, 'number') - if (!coerced.ok) return null - const numValue = coerced.value as number - - switch (operator) { - case 'eq': - return sql`${column} = ${numValue}` - case 'neq': - return sql`${column} != ${numValue}` - case 'gt': - return sql`${column} > ${numValue}` - case 'gte': - return sql`${column} >= ${numValue}` - case 'lt': - return sql`${column} < ${numValue}` - case 'lte': - return sql`${column} <= ${numValue}` - case 'between': - if (valueTo !== undefined) { - const coercedTo = coerceTagFilterValue(valueTo, 'number') - if (!coercedTo.ok) return sql`${column} = ${numValue}` - return sql`${column} >= ${numValue} AND ${column} <= ${coercedTo.value as number}` - } - return sql`${column} = ${numValue}` - default: - return sql`${column} = ${numValue}` - } - } - - // Date values arrive as YYYY-MM-DD strings from the frontend. - if (fieldType === 'date') { - const coerced = coerceTagFilterValue(value, 'date') - if (!coerced.ok) return null - const dateStr = coerced.value as string - - switch (operator) { - case 'eq': - return sql`${column}::date = ${dateStr}::date` - case 'neq': - return sql`${column}::date != ${dateStr}::date` - case 'gt': - return sql`${column}::date > ${dateStr}::date` - case 'gte': - return sql`${column}::date >= ${dateStr}::date` - case 'lt': - return sql`${column}::date < ${dateStr}::date` - case 'lte': - return sql`${column}::date <= ${dateStr}::date` - case 'between': - if (valueTo !== undefined) { - const coercedTo = coerceTagFilterValue(valueTo, 'date') - if (!coercedTo.ok) { - return sql`${column}::date = ${dateStr}::date` - } - const dateStrTo = coercedTo.value as string - return sql`${column}::date >= ${dateStr}::date AND ${column}::date <= ${dateStrTo}::date` - } - return sql`${column}::date = ${dateStr}::date` - default: - return sql`${column}::date = ${dateStr}::date` - } - } - - if (fieldType === 'boolean') { - const coerced = coerceTagFilterValue(value, 'boolean') - if (!coerced.ok) return null - const boolValue = coerced.value as boolean - switch (operator) { - case 'eq': - return sql`${column} = ${boolValue}` - case 'neq': - return sql`${column} != ${boolValue}` - default: - return sql`${column} = ${boolValue}` - } - } - - return sql`${column} = ${value}` -} - -/** - * Build SQL conditions from structured filters with operator support. Every - * filter is a conjunct, including two that name the same tag. - * - * Search used to group filters by slot and OR same-slot conditions together, - * which made the two surfaces over the same tag vocabulary answer different - * questions: the document list ANDs every filter, so `gte 9` plus `lte 2` on one - * number tag returned nothing there and a full page of results from search — - * a widening on the billed endpoint, the same failure mode as dropping a filter. - * OR also made a range on a single text tag (`contains A` and `contains B`) - * inexpressible, while the union it produced stays reachable as separate - * searches. Neither contract ever documented the OR, so no caller could have - * been relying on it deliberately. - * - * Every filter reaching here has already been validated, so one that fails to - * compile is a defect rather than a predicate to skip. Skipping it dropped the - * tag term from the WHERE clause entirely and answered a filtered search with - * the whole knowledge base under a 200 — and search is billed, so the caller - * paid for the widened scan. It is reported as a validation failure instead. - */ -export function getStructuredTagFilters(filters: StructuredFilter[], embeddingTable: any) { - return filters.map((filter) => { - const condition = buildFilterCondition(filter, embeddingTable) - if (condition === null) throw uncompilableTagFilterError(filter) - return condition - }) -} - -/** - * Match the normalization used by the stored text vectors so query terms and - * document terms resolve to the same lexemes in both keyword retrieval paths. - */ -const FTS_CONFIG = 'english' - -/** - * Chunks Tin ranks before access is checked, widening while too few are readable to fill a page. - * A caller past the permitted-set limit reads a large share of the index, so the first window - * almost always fills; the widest bounds the work before the GIN ranking takes over. - */ -const TIN_KEYWORD_WINDOWS = [2000, 10_000, 50_000] as const - -/** Readable rows one wide window returns for a narrow reader: several pages' worth, ranked once. */ -const NARROW_KEYWORD_PAGE = 1000 - -/** - * The widest window a narrow reader ranks: wide enough that a few percent of it fills their page - * several times over, and less than half the cost of the widest window the broad readers reach. - * It is tried only after the first window came back short: ranking costs grow with the window, - * and a term that is common where the reader can read fills the page from the narrowest one. - */ -const NARROW_KEYWORD_WINDOWS = [TIN_KEYWORD_WINDOWS[0], 20_000] as const - -/** - * Stamps each row with the score its position came from and its 1-based rank. - * - * Hybrid results are ordered by a fused score the caller never saw, while the - * `similarity` reported beside them is the vector leg's cosine value — so the - * two modes answered with byte-identical `similarity` for orderings that could - * differ. Exposing the ordering key makes the order explainable in either mode. - */ -function rankResults(rows: SearchResult[], scoreOf: (row: SearchResult) => number): SearchResult[] { - return rows.map((row, index) => ({ ...row, rankScore: scoreOf(row), rank: index + 1 })) -} - -/** - * Row visibility predicates shared by every search leg: a chunk is only - * retrievable when both it and its document are enabled, the document finished - * processing, it has not been excluded, archived, or soft-deleted, and its ACL - * overlaps the caller's tokens. Every leg spreads this helper rather than - * listing the predicates itself, so no leg can drift from the others. - */ -function getVisibilityConditions( - access: KnowledgeAccessScope, - filters?: WorkspaceSearchFilters, - accessCondition: SQL = knowledgeAccessCondition(access), - enabledColumn: typeof embedding.enabled | typeof embeddingSearch.enabled = embedding.enabled -) { - return [ - eq(enabledColumn, true), - ...getDocumentVisibilityConditions(access, filters, accessCondition), - ] -} - -function getDocumentVisibilityConditions( - access: KnowledgeAccessScope, - filters?: WorkspaceSearchFilters, - accessCondition: SQL = knowledgeAccessCondition(access) -) { - return [ - eq(document.enabled, true), - eq(document.processingStatus, 'completed'), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - accessCondition, - ...workspaceSearchFilterConditions(filters), - ] -} - -/** - * The candidate predicate a leg applies, with connector state resolved ahead of the query when the - * search resolved it. Both shapes admit exactly the same documents. - */ -function candidateAccessCondition( - access: KnowledgeAccessScope, - plan: SearchAccessPlan | undefined -): SQL { - return plan - ? knowledgeCandidateAccessConditionForConnectors(access, plan) - : knowledgeMetadataCandidateAccessCondition(access) -} - -/** - * The document-level candidate predicate every ranked leg applies. The permitted set is resolved - * with the same list, which is what lets a leg rank inside it without admitting anything more. - */ -function candidateDocumentConditions( - knowledgeBaseIds: string[], - access: KnowledgeAccessScope, - filters: WorkspaceSearchFilters | undefined, - accessCondition: SQL -) { - return [ - inArray(document.knowledgeBaseId, knowledgeBaseIds), - ...getDocumentVisibilityConditions(access, filters, accessCondition), - ] -} - -interface SearchReadCandidatePage { - candidates: SearchReadCandidate[] - nextOffset: number -} - -type SearchReadCandidate = { - id: string - documentId: string - connectorId: string | null -} - -/** - * A candidate's source: the row's, unless the row's document is marked for the projector, whose - * source may have moved since the row was written — then the document's, read for that row only. - */ -function projectionCandidateSource(projection: { - connectorId: AnyPgColumn | SQL - documentId: AnyPgColumn | SQL -}): SQL { - return sql`CASE WHEN ${projectionPending(projection.documentId)} - THEN (SELECT ${document.connectorId} FROM ${document} WHERE ${document.id} = ${projection.documentId}) - ELSE ${projection.connectorId} END` -} - -/** - * The same identities read off a projection row in raw SQL: the aliases are what - * `SearchReadCandidate` deserializes, so every walk reads them from one place. - */ -const PROJECTION_CANDIDATE_COLUMNS = sql`${embeddingSearch.id} AS id, ${embeddingSearch.documentId} AS "documentId", ${projectionCandidateSource(embeddingSearch)} AS "connectorId"` - -/** - * Keeps the rows of sources the caller turned out not to hold out of a ranking decided on the - * row. A row decided on its document — not yet filled, or its document marked for the projector, - * so its own source may be stale — asks the document instead. - */ -function excludeSearchSourcesOnRow( - projection: { - connectorId: AnyPgColumn | SQL - acl: AnyPgColumn | SQL - documentId: AnyPgColumn | SQL - }, - filled: boolean, - excludedSources: readonly string[] -): SQL | undefined { - if (!excludedSources.length) return undefined - const excluded = textArrayLiteral([...excludedSources]) - const decided = projectionDecidedOnDocument(projection, filled) - return sql`((${decided} AND NOT EXISTS (SELECT 1 FROM ${document} WHERE ${document.id} = ${projection.documentId} AND ${document.connectorId} = ANY(${excluded}))) - OR (NOT ${decided} AND (${projection.connectorId} IS NULL OR NOT (${projection.connectorId} = ANY(${excluded}))))) /* excluded sources */` -} - -/** Only opaque identifiers leave candidate ranking; content stays behind the full read predicate. */ -const SEARCH_READ_CANDIDATE_FIELDS = { - id: embedding.id, - documentId: document.id, - connectorId: document.connectorId, -} - -/** - * The caller's proof of reader access to the sources that require one, resolved at most once per - * search and only when a candidate from such a source is about to be read. - */ -export interface LiveSourceAccess { - gates: (connectorId: string) => boolean - /** The caller's scope with its grants, and the gated sources those grants do not cover. */ - resolve: () => Promise<{ access: KnowledgeAccessScope; denied: ReadonlySet }> -} - -/** Binds a search's gated sources to one memoized resolution of the caller's grants. */ -export function liveSourceAccessFor( - access: KnowledgeAccessScope, - plan: SearchAccessPlan | undefined, - accessProvider: KnowledgeAccessProvider | undefined, - signal?: AbortSignal -): LiveSourceAccess | undefined { - const gated = new Set(plan?.connectors.liveProofRequired ?? []) - if (!accessProvider || gated.size === 0) return undefined - let pending: Promise<{ access: KnowledgeAccessScope; denied: ReadonlySet }> | undefined - /** The provider authorizes connectors in bounded pages, so a wide scope resolves page by page. */ - const pages: string[][] = [] - for (const id of gated) { - const last = pages.at(-1) - if (!last || last.length === MAX_KNOWLEDGE_ACCESS_CANDIDATES) pages.push([id]) - else last.push(id) - } - return { - gates: (connectorId) => gated.has(connectorId), - resolve: () => { - pending ??= measureSearchStage('live_source_grants', async () => { - const scopes = await mapWithConcurrency(pages, SOURCE_RANKING_CONCURRENCY, (page) => - accessProvider.getForConnectors(page, signal) - ) - const [first] = scopes - const githubInstallationGrants: GitHubInstallationReadGrant[] = [] - const confluenceSiteGrants: ConfluenceSiteReadGrant[] = [] - for (const scope of scopes) { - if (scope.kind !== 'user') continue - if (scope.githubInstallationGrants) - githubInstallationGrants.push(...scope.githubInstallationGrants) - if (scope.confluenceSiteGrants) confluenceSiteGrants.push(...scope.confluenceSiteGrants) - } - const granted = new Set([ - ...githubInstallationGrants.map((grant) => grant.connectorId), - ...confluenceSiteGrants.map((grant) => grant.connectorId), - ]) - const denied = new Set([...gated].filter((id) => !granted.has(id))) - const merged = - first.kind === 'user' - ? { ...first, githubInstallationGrants, confluenceSiteGrants } - : first - return { access: merged, denied } - }) - return pending - }, - } -} - -const AUTHORIZED_SEARCH_PAGE_SIZE = 200 -const AUTHORIZED_SEARCH_BUDGET_MS = 8000 - -/** - * Verification follows ranked candidates, never the organization's source order. Denied - * sources are excluded on refill, so many matches from one revoked source cannot - * consume every result slot. Candidate pages and the shared deadline bound authorization work. - */ -async function selectAuthorizedSearchResults(input: { - leg: 'vector' | 'keyword' | 'tags' - access: KnowledgeAccessScope - filters?: WorkspaceSearchFilters - signal?: AbortSignal - budget?: SearchBudget - topK: number - selectPage: ( - limit: number, - offset: number, - excludedSources: readonly string[] - ) => Promise - compareResults?: (a: SearchResult, b: SearchResult) => number - hydrate: (ids: string[], access: KnowledgeAccessScope) => Promise - liveSourceAccess?: LiveSourceAccess -}): Promise { - const deadline = Date.now() + AUTHORIZED_SEARCH_BUDGET_MS - const pageSize = Math.min(AUTHORIZED_SEARCH_PAGE_SIZE, Math.max(input.topK, 20)) - const results = new Map() - const considered = new Set() - /** Gated sources the caller turned out not to hold: left out of every page once that is known. */ - let excluded: ReadonlySet = new Set() - let scanned = 0 - let offset = 0 - /** Candidates a ranking returned beyond the current hydration slice. */ - let pending: SearchReadCandidate[] = [] - let lastPageShort = false - try { - while ( - results.size < input.topK && - scanned < MAX_AUTHORIZED_SEARCH_CANDIDATES && - (input.budget !== undefined || Date.now() < deadline) - ) { - input.signal?.throwIfAborted() - input.budget?.remaining() - /** - * A ranking may hand back more candidates than one hydration should read — a narrow - * reader's keyword window is ranked once for several pages' worth — so a page is drained in - * hydration-sized slices, and what is left waits, unread, until the results still need it. - */ - if (!pending.length) { - const page = await measureSearchStage(`${input.leg}.candidates`, () => - input.selectPage(pageSize, offset, [...excluded]) - ) - if (!page.candidates.length) break - scanned += page.candidates.length - /** Short means the ranking had fewer to give, not that a read recovered fewer than it asked. */ - lastPageShort = page.nextOffset - offset < pageSize - offset = page.nextOffset - pending = page.candidates.filter((candidate) => !considered.has(candidate.id)) - if (!pending.length) { - if (lastPageShort) break - continue - } - } - /** - * A candidate counts as considered only once its slice is read: the slices a refill discards - * were never read, so the rebuilt pages may hand their readable candidates back. - */ - const candidates = pending - .slice(0, pageSize) - .filter((candidate) => !considered.has(candidate.id)) - pending = pending.slice(pageSize) - for (const candidate of candidates) considered.add(candidate.id) - if (!candidates.length) continue - /** - * A source that proves its reader live is asked for that proof only once a candidate of - * its own reaches this page, and then once for the whole search: a scope that ranks none - * of them — most scopes — never asks, and one that ranks many asks once. - */ - const proof = input.liveSourceAccess - const gatedOnPage = - proof !== undefined && - candidates.some((candidate) => candidate.connectorId && proof.gates(candidate.connectorId)) - let access = input.access - let refill = false - if (gatedOnPage && proof) { - const resolved = await proof.resolve() - access = resolved.access - /** - * A denied source's candidates cannot hydrate, yet they took the slots of sources the - * caller does hold. Once the denial is known the pages are rebuilt without that source, - * from the start; the ids already seen are not read twice. - */ - if (resolved.denied.size > excluded.size) { - excluded = resolved.denied - refill = true - } - } - const hydrated = await measureSearchStage(`${input.leg}.hydration`, () => - input.hydrate( - candidates.map((candidate) => candidate.id), - access - ) - ) - const byId = new Map(hydrated.map((row) => [row.id, row])) - for (const candidate of candidates) { - const row = byId.get(candidate.id) - if (row) results.set(row.id, row) - if (!input.compareResults && results.size === input.topK) break - } - if (refill) { - /** The rebuilt pages are a new stream of candidates, so the scan budget starts over. */ - offset = 0 - scanned = 0 - pending = [] - continue - } - /** A short page is the end of the candidates, whether or not they were reordered. */ - if (!pending.length && lastPageShort) break - } - } catch (error) { - if (!input.budget?.isTimeout(error)) throw error - } - input.signal?.throwIfAborted() - const rows = [...results.values()] - /** A reordered leg keeps every page's rows until the end: a later page cannot displace what an earlier one ranked. */ - return input.compareResults ? rows.sort(input.compareResults).slice(0, input.topK) : rows -} - -/** Keeps the candidates of sources the caller turned out not to hold out of a page. */ -function excludeSearchSources(sourceIds: readonly string[]): SQL | undefined { - return sourceIds.length - ? sql`(${document.connectorId} IS NULL OR NOT (${inArray(document.connectorId, [...sourceIds])}))` - : undefined -} - -/** - * Loads the content of candidates that survived ranking, under the read predicate. - * - * The resolved connector state bounds ranking, never this. A page of ranked identifiers is small, - * so its content is read under the full predicate, which re-reads each connector's own lifecycle - * and approval: a source deleted, archived or unapproved while the search was running stops - * answering here, at the gate that returns content. - */ -function hydrateSearchCandidates( - ids: string[], - access: KnowledgeAccessScope, - distance: SQL | SQL.Aliased, - filters: WorkspaceSearchFilters | undefined, - conditions: (SQL | undefined)[], - leg: RetrievalLeg, - budget?: SearchBudget, - /** Whether a condition reads the projection's stored halfvec, which only the vector leg's threshold does. */ - joinProjection = false -) { - const accessCondition = knowledgeAccessCondition(access) - /** - * The projection joins so a condition on its stored halfvec — the candidate threshold — can be - * tested here; the returned score is whatever the leg passes as `distance`. Both legs pass the - * original vector's cosine distance: one out-of-line read per hydrated row, the page's size, - * where scoring the whole candidate pool that way read one per candidate on every novel query. - */ - return runSearchQuery(budget, `${leg}.sql`, (executor) => { - const read = executor - .select(getSearchResultFields(distance)) - .from(embedding) - .innerJoin(document, eq(embedding.documentId, document.id)) - .leftJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) - return ( - joinProjection ? read.leftJoin(embeddingSearch, eq(embeddingSearch.id, embedding.id)) : read - ).where( - and( - inArray(embedding.id, ids), - ...getVisibilityConditions(access, filters, accessCondition), - ...conditions - ) - ) - }) -} - -/** Candidates each hybrid leg retrieves before the fused list is trimmed to `topK`. */ -const HYBRID_CANDIDATE_MIN = 50 -const HYBRID_CANDIDATE_MAX = 200 -export function hybridCandidateCount(topK: number): number { - return Math.min(Math.max(topK * 3, HYBRID_CANDIDATE_MIN), HYBRID_CANDIDATE_MAX) -} - -export function getQueryStrategy(kbCount: number, topK: number) { - return { - useParallel: kbCount > 4 || (kbCount > 2 && topK > 50), - distanceThreshold: kbCount > 3 ? 0.8 : 1.0, - } -} - -export async function handleTagOnlySearch(params: SearchParams): Promise { - const { knowledgeBaseIds, topK, structuredFilters, access } = params - - if (!structuredFilters || structuredFilters.length === 0) { - throw new Error('Tag filters are required for tag-only search') - } - - const strategy = getQueryStrategy(knowledgeBaseIds.length, topK) - const tagFilterConditions = getStructuredTagFilters(structuredFilters, embedding) - - if (params.accessProvider && access.kind === 'user') { - const conditions = [ - inArray(embedding.knowledgeBaseId, knowledgeBaseIds), - ...tagFilterConditions, - ] - return selectAuthorizedSearchResults({ - leg: 'tags', - access: params.access, - liveSourceAccess: params.liveSourceAccess, - filters: params.filters, - signal: params.signal, - budget: params.budget, - topK, - selectPage: async (limit, offset, excludedSources) => { - const candidates = await runSearchQuery(params.budget, 'tags.sql', (executor) => - executor - .select(SEARCH_READ_CANDIDATE_FIELDS) - .from(embedding) - .innerJoin(document, eq(embedding.documentId, document.id)) - .where( - and( - ...conditions, - ...getVisibilityConditions( - access, - params.filters, - candidateAccessCondition(access, params.accessPlan) - ), - excludeSearchSources(excludedSources) - ) - ) - .orderBy(embedding.id) - .limit(limit) - .offset(offset) - ) - return { candidates, nextOffset: offset + candidates.length } - }, - hydrate: (ids, authorized) => - hydrateSearchCandidates( - ids, - authorized, - sql`0`.as('distance'), - params.filters, - conditions, - 'tags', - params.budget - ), - }) - } - - if (strategy.useParallel) { - const parallelLimit = Math.ceil(topK / knowledgeBaseIds.length) + 5 - - const queryPromises = knowledgeBaseIds.map(async (kbId) => { - return await db - .select(getSearchResultFields(sql`0`.as('distance'))) - .from(embedding) - .innerJoin(document, eq(embedding.documentId, document.id)) - .leftJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) - .where( - and( - eq(embedding.knowledgeBaseId, kbId), - ...getVisibilityConditions(access, params.filters), - ...tagFilterConditions - ) - ) - .limit(parallelLimit) - }) - - const parallelResults = await Promise.all(queryPromises) - return parallelResults.flat().slice(0, topK) - } - // Single query for fewer KBs - return await db - .select(getSearchResultFields(sql`0`.as('distance'))) - .from(embedding) - .innerJoin(document, eq(embedding.documentId, document.id)) - .leftJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) - .where( - and( - inArray(embedding.knowledgeBaseId, knowledgeBaseIds), - ...getVisibilityConditions(access, params.filters), - ...tagFilterConditions - ) - ) - .limit(topK) -} - -export async function handleVectorOnlySearch(params: SearchParams): Promise { - const { queryVector, distanceThreshold } = params - if (!queryVector || !distanceThreshold) { - throw new Error('Query vector and distance threshold are required for vector-only search') - } - return selectVectorResults(params) -} - -type ProbeOutcome = - | { kind: 'documents'; documents: PermittedDocument[] } - /** The caller reads more documents than an exact ranking can afford. */ - | { kind: 'saturated' } - /** The probe spent its own deadline before finding out. */ - | { kind: 'timed_out' } - -/** - * Enumerate the documents the caller may read, stopping once there are more of them than an exact - * ranking can afford. The bound is documents examined, not chunks accumulated: the access - * predicate is evaluated once per document, and a search index holds only a few chunks per - * document, so a chunk-bounded enumeration walks many times more documents than its limit says. - * - * Neither saturation nor a timeout is a failure of the leg, which keeps the candidates it - * already has. - */ -async function probeVisibleDocuments( - knowledgeBaseIds: string[], - conditions: (SQL | undefined)[], - access: KnowledgeAccessScope, - budget: SearchBudget | undefined, - stage: 'vector.probe' | 'permitted_documents', - shape: 'reach-first' | 'direct' = 'reach-first' -): Promise { - const probeBudget = budget?.capped( - shape === 'direct' ? FILTERED_PROBE_BUDGET_MS : VECTOR_PROBE_BUDGET_MS - ) - try { - const probed = await runSearchQuery(probeBudget, stage, (executor) => - executor.execute( - visibleDocumentsQuery(knowledgeBaseIds, conditions, access, shape) - ) - ) - /** The saturation sentinel is only ever emitted alone. */ - if (probed.length > VECTOR_PROBE_DOCUMENT_LIMIT || probed[0]?.saturated) { - return { kind: 'saturated' } - } - return { - kind: 'documents', - documents: probed.map(({ id, connectorId }) => ({ id, connectorId })), - } - } catch (error) { - if (!budget || !probeBudget?.isTimeout(error)) throw error - /** Only the probe's share was spent; the leg's own deadline still governs. */ - budget.remaining() - return { kind: 'timed_out' } - } -} - -/** - * The probe's SQL, returning at most one row past the document limit. - * - * A user scope first materializes the documents its tokens reach in these bases, read through - * `doc_acl_gin_idx` alone, then applies the state and full access conditions to those rows in - * memory; the set is aliased as `document` so the shared conditions bind to it unchanged. Handed - * the combined predicate instead, PostgreSQL misjudges the token overlap as unselective and - * intersects it with base-wide indexes that read the whole search index. - * - * The index is global and every caller holds the baseline tokens every tenant's org-wide, public, - * and uploaded documents carry, so the reach must be counted inside these bases or those - * documents alone would saturate it. The base check is applied outside an `OFFSET 0` fence so it - * filters the index's rows instead of replacing the index with a base-wide scan. The reach is - * counted before any row is materialized, so a caller whose tokens reach past the limit pays only - * for the count, and a `saturated` sentinel row then reports the set as unbounded. Resolved scopes - * hold base-wide tokens, so they filter directly. - */ -export function visibleDocumentsQuery( - knowledgeBaseIds: string[], - conditions: (SQL | undefined)[], - access: KnowledgeAccessScope, - shape: 'reach-first' | 'direct' = 'reach-first' -): SQL { - const limit = VECTOR_PROBE_DOCUMENT_LIMIT + 1 - /** - * `direct` applies the conditions as they are: a date filter is selective on its own and has - * its own index, and a resolved scope's tokens are base-wide, so counting the reach first would - * only report a broad caller as saturated before the filter was consulted. - */ - if (access.kind !== 'user' || shape === 'direct') { - return sql` - SELECT ${document.id} AS id, ${document.connectorId} AS "connectorId", false AS saturated - FROM ${document} - WHERE ${and(...conditions)} - LIMIT ${limit} - ` - } - /** Exactly `doc_acl_gin_idx`'s predicate, so both the count and the rows read that index alone. */ - const reached = sql`${document.deletedAt} IS NULL AND ${knowledgeAclOverlapCondition(access)}` - const underLimit = sql`(SELECT n FROM reach) < ${limit}` - const inBases = inArray(document.knowledgeBaseId, knowledgeBaseIds) - return sql` - WITH reach AS MATERIALIZED ( - SELECT count(*) AS n FROM ( - SELECT 1 FROM ( - SELECT ${document.knowledgeBaseId} FROM ${document} WHERE ${reached} OFFSET 0 - ) AS ${document} - WHERE ${inBases} - LIMIT ${limit} - ) AS reached - ), reachable AS MATERIALIZED ( - SELECT * FROM ( - SELECT * FROM ${document} WHERE ${underLimit} AND ${reached} OFFSET 0 - ) AS ${document} - WHERE ${inBases} - ) - ( - SELECT ${document.id} AS id, ${document.connectorId} AS "connectorId", false AS saturated - FROM reachable AS ${document} - WHERE ${underLimit} AND ${and(...conditions)} - LIMIT ${limit} - ) - UNION ALL - SELECT NULL, NULL, true WHERE (SELECT n FROM reach) >= ${limit} - ` -} - -/** A document a caller may rank, with the source a live authorization pass may later exclude. */ -type PermittedDocument = { - id: string - connectorId: string | null -} - -/** - * The documents a user-scoped search may rank, resolved once before either leg runs. - * - * Organization search indexes grant most documents to a single mailbox, channel, or file owner, - * so a member typically reads a vanishing share of the index. Ranking the whole index and - * checking access afterwards then scans thousands of candidates to find none; ranking inside the - * permitted set finds every eligible chunk at a cost proportional to what the member can read. - * `unbounded` means the set exceeded the probe's limit, where post-filtered index search fills - * quickly because most candidates are readable. - */ -export type PermittedDocuments = - | { kind: 'bounded'; documents: readonly PermittedDocument[] } - | { kind: 'unbounded'; broad: boolean } - -/** - * The share of the index a caller must reach before the whole graph is walked for them. pgvector - * post-filters, so a walk returns a caller's own neighbours in proportion to their reach: above - * this share almost every neighbour the graph visits is theirs and one walk is the cheapest exact - * answer there is; below it the walk spends its budget on chunks they cannot read, and each - * readable source is searched on its own instead. - */ -export const BROAD_REACH_SHARE = 0.25 - -/** - * How long a caller's saturated reach is remembered. Reach counts the documents a caller's tokens - * touch in the bases, which moves slowly, and an unbounded set only means the legs search the - * index with the full access predicate, so a stale answer costs speed, never access. - */ -const SATURATED_REACH_TTL_MS = 5 * 60 * 1000 - -/** - * A counted reach: whether it is broad enough to walk the whole graph for, or empty, in which - * case the caller reads nothing in these bases and no leg has anything to rank. - */ -interface CountedReach { - broad: boolean - empty: boolean -} + annotateVectorPoolPlanned, + annotateVectorPoolSelected, + gatheredVectorCandidatePool, + hydrateVectorCandidates, + prepareVectorLeg, + rankVectorCandidatesExactly, + readVectorCandidatePool, + selectExactVectorPage, + sliceVectorCandidatePool, + type VectorCandidatePool, + withVectorScanSettings, +} from '@/lib/knowledge/search/vector-leg' +import type { StructuredFilter } from '@/lib/knowledge/types' +import { embeddingDistance } from '@/lib/knowledge/vector-columns' /** - * Only breadth is remembered. Emptiness decides completeness, not strategy, so it is counted on - * every search: the count of a reach of nothing finds nothing and costs almost nothing. + * How a search decides readability on each candidate's document. Ranking admits what the + * caller's stored grants and tokens already read and, from the sources a held reader credential + * could open, the candidates their mirrored permissions admit; the live proof those sources need + * is asked for only once one of their candidates reaches hydration, and then once per search. */ -const saturatedReach = new LRUCache({ - max: 10_000, - ttl: SATURATED_REACH_TTL_MS, -}) - -/** How many documents the bases hold: the denominator of a reach share, and it moves slowly. */ -const indexDocumentCounts = new LRUCache({ - max: 1000, - ttl: SATURATED_REACH_TTL_MS, +interface DocumentReadAccess { + /** The candidate predicate every leg ranks under. */ + rankCondition: SQL /** - * The planner's estimate of the bases' documents, from the statistics it already keeps: a share - * threshold needs the order of magnitude, and counting every row to learn it costs more than the - * search it serves. The read that misses is the search's own, under its deadline. - */ - fetchMethod: async (key, _stale, { context: budget }) => { - const [row] = await runSearchQuery(budget, 'permitted_documents', (executor) => - executor.execute<{ 'QUERY PLAN': Array<{ Plan: { 'Plan Rows': number } }> }>(sql` - EXPLAIN (FORMAT JSON) SELECT 1 FROM ${document} - WHERE ${document.knowledgeBaseId} = ANY(${textArrayLiteral(key.split(','))}) - AND ${document.deletedAt} IS NULL`) - ) - /** An empty answer is not remembered; the bases may simply not have been analyzed yet. */ - return Number(row?.['QUERY PLAN']?.[0]?.Plan?.['Plan Rows'] ?? 0) || undefined - }, -}) - -/** The date window a filter asks for, on the document row; nothing when none is asked. */ -function dateFilterCondition(filters: WorkspaceSearchFilters | undefined): SQL | undefined { - if (!filters?.modifiedAfter && !filters?.modifiedBefore) return undefined - return and( - filters.modifiedAfter - ? gte(document.sourceModifiedAt, new Date(filters.modifiedAfter)) - : undefined, - filters.modifiedBefore - ? lte(document.sourceModifiedAt, new Date(filters.modifiedBefore)) - : undefined - ) -} - -/** - * The planner's estimate of the documents a filter leaves in the bases — a date filter from the - * statistics on its index, a source filter from its connectors' — so whether the filtered set is - * worth enumerating is decided from its order of magnitude, without reading a row. - */ -async function estimateFilteredDocuments( - knowledgeBaseIds: string[], - filters: WorkspaceSearchFilters, - plan: SearchAccessPlan, - budget: SearchBudget | undefined -): Promise { - const [row] = await runSearchQuery(budget, 'permitted_documents', (executor) => - executor.execute<{ 'QUERY PLAN': Array<{ Plan: { 'Plan Rows': number } }> }>(sql` - EXPLAIN (FORMAT JSON) SELECT 1 FROM ${document} - WHERE ${and( - inArray(document.knowledgeBaseId, knowledgeBaseIds), - isNull(document.deletedAt), - dateFilterCondition(filters), - filters.source ? planSourceCondition(plan) : undefined - )}`) - ) - return Number(row?.['QUERY PLAN']?.[0]?.Plan?.['Plan Rows'] ?? 0) + * A signed-in reader: their tag and keyword legs run under the search deadline and rank in one + * statement. A caller with no person behind it runs those legs unbudgeted, as it always has. + */ + signedIn: boolean + /** Present only when a searched base holds a source whose reader must be proven live. */ + liveSourceAccess?: LiveSourceAccess } /** - * How far a caller reaches: broad when they reach at least {@link BROAD_REACH_SHARE} of the - * bases' documents, empty when they reach none. A reach of nothing is a bounded set of nothing: a - * caller who reads no document in these bases, such as a member with no source of their own yet, - * has nothing for any leg to rank, where an unbounded set would have each leg scan to its - * deadline for rows it cannot find. Breadth is counted once against the bound and remembered, so - * the first search after the window pays for it and the rest do not. A caller whose probe already - * saturated is known to reach past the probe's limit, so a bound inside that limit is met without - * counting. - * - * The count reads as many index entries as the caller reaches, so on a large index it can cost - * more than the leg it serves; it gets the probe's share of the deadline, never the whole leg's. - * A count that runs out of that share answers `null`: the leg keeps its time and its deadline - * intact, and the caller decides this search alone without remembering anything. + * Resolves a search's {@link DocumentReadAccess} without reaching any source: only the ids of the + * searched bases' sources a held reader credential could open are read, and a caller holding none + * ranks under the ordinary predicate alone. */ -async function countReach( +async function resolveDocumentReadAccess( knowledgeBaseIds: string[], access: KnowledgeAccessScope, - budget: SearchBudget | undefined, - plan: SearchAccessPlan | undefined, - saturated: boolean -): Promise { - if (access.kind !== 'user') return { broad: true, empty: false } - const countBudget = budget?.capped(VECTOR_PROBE_BUDGET_MS) - try { - const total = - (await indexDocumentCounts.fetch([...knowledgeBaseIds].sort().join(','), { - context: countBudget, - })) ?? 0 - const bound = Math.ceil(total * BROAD_REACH_SHARE) - if (saturated && bound <= VECTOR_PROBE_DOCUMENT_LIMIT) return { broad: true, empty: false } - const [row] = await runSearchQuery(countBudget, 'permitted_documents', (executor) => - executor.execute<{ n: number }>(sql` - SELECT count(*) AS n FROM ( - SELECT 1 FROM ${document} - WHERE ${and( - isNull(document.deletedAt), - knowledgeAclOverlapCondition(access), - inArray(document.knowledgeBaseId, knowledgeBaseIds), - planSourceCondition(plan) - )} - LIMIT ${bound} - ) reached`) - ) - const reached = Number(row?.n ?? 0) - /** A count that looked and found nothing: only a bound of zero looks at nothing. */ - return { broad: reached >= bound, empty: bound > 0 && reached === 0 } - } catch (error) { - if (!budget || !countBudget?.isTimeout(error)) throw error - /** Only the count's share was spent; the leg's own deadline still governs. */ - budget.remaining() - return null + accessProvider: KnowledgeAccessProvider | undefined, + signal: AbortSignal | undefined +): Promise { + const ordinary = knowledgeAccessCondition(access) + if (!accessProvider || access.kind !== 'user') { + return { rankCondition: ordinary, signedIn: false } + } + const liveSources = await accessProvider.liveSourceConnectorCondition() + if (!liveSources) return { rankCondition: ordinary, signedIn: true } + const gated = await measureSearchStage('access_batch.connectors', () => + db + .select({ id: knowledgeConnector.id }) + .from(knowledgeConnector) + .where(and(liveSources, inArray(knowledgeConnector.knowledgeBaseId, knowledgeBaseIds))) + ) + const gatedIds = gated.map((row) => row.id) + const liveSourceAccess = liveSourceAccessForConnectors(gatedIds, accessProvider, signal) + if (!liveSourceAccess) return { rankCondition: ordinary, signedIn: true } + return { + signedIn: true, + rankCondition: or( + ordinary, + and( + inArray(document.connectorId, gatedIds), + knowledgeMetadataCandidateAccessCondition(access) + ) + )!, + liveSourceAccess, } } /** - * Reach depends on the bases, the caller's tokens and, when the plan is confined to one kind of - * source, which sources those are; a date filter narrows the set, not the reach. + * Stamps each row with the score its position came from and its 1-based rank. + * + * Hybrid results are ordered by a fused score the caller never saw, while the + * `similarity` reported beside them is the vector leg's cosine value — so the + * two modes answered with byte-identical `similarity` for orderings that could + * differ. Exposing the ordering key makes the order explainable in either mode. */ -function reachKey( - knowledgeBaseIds: readonly string[], - access: KnowledgeAccessScope, - plan: SearchAccessPlan | undefined -): string | null { - if (access.kind !== 'user') return null - const sources = plan - ? `:${sha256Hex([...planSources(plan)].sort().join('\n'))}:${plan.uploads}` - : '' - return `${[...knowledgeBaseIds].sort().join(',')}:${sha256Hex([...access.tokens].sort().join('\n'))}${sources}` -} - -/** Every connector the plan admits, whatever its access mode. */ -function planSources(plan: SearchAccessPlan): readonly string[] { - return [...plan.connectors.workspace, ...plan.connectors.admin, ...plan.connectors.members] +function rankResults(rows: SearchResult[], scoreOf: (row: SearchResult) => number): SearchResult[] { + return rows.map((row, index) => ({ ...row, rankScore: scoreOf(row), rank: index + 1 })) } -/** The documents a plan's sources own, on the document row; every source when unconfined. */ -function planSourceCondition(plan: SearchAccessPlan | undefined): SQL | undefined { - if (!plan) return undefined - const owned = planSources(plan) - const inSources = owned.length - ? sql`${document.connectorId} = ANY(${textArrayLiteral([...owned])})` - : sql`false` - return plan.uploads ? sql`(${document.connectorId} IS NULL OR ${inSources})` : inSources +/** Candidates each hybrid leg retrieves before the fused list is trimmed to `topK`. */ +const HYBRID_CANDIDATE_MIN = 50 +const HYBRID_CANDIDATE_MAX = 200 +function hybridCandidateCount(topK: number): number { + return Math.min(Math.max(topK * 3, HYBRID_CANDIDATE_MIN), HYBRID_CANDIDATE_MAX) } -/** Forgets every remembered reach, after the bases' documents or a caller's tokens changed. */ -export function forgetSearchReach(): void { - saturatedReach.clear() - indexDocumentCounts.clear() +function getQueryStrategy(kbCount: number, topK: number) { + return { + useParallel: kbCount > 4 || (kbCount > 2 && topK > 50), + distanceThreshold: kbCount > 3 ? 0.8 : 1.0, + } } -/** A resolved scope's reach, remembered per bases and tokens, with no document enumerated. */ -export async function resolveReach( - knowledgeBaseIds: string[], - access: KnowledgeAccessScope, +/** + * A document-decided leg's statements run under the leg's deadline; a leg that reaches it before + * it ranked anything is short, not failed. + */ +async function shortOnDeadline( budget: SearchBudget | undefined, - plan: SearchAccessPlan -): Promise { - const key = reachKey(knowledgeBaseIds, access, plan) - const remembered = key ? saturatedReach.get(key) : undefined - if (remembered) return { kind: 'unbounded', broad: remembered.broad } + run: () => Promise +): Promise { try { - const reach = await countReach(knowledgeBaseIds, access, budget, plan, false) - /** A count that ran out of time decides this search only; the next one counts again. */ - if (reach === null) return { kind: 'unbounded', broad: true } - if (reach.empty) return { kind: 'bounded', documents: [] } - if (key) saturatedReach.set(key, { broad: reach.broad }) - return { kind: 'unbounded', broad: reach.broad } + return await run() } catch (error) { - /** The leg's own deadline passed during the count: the leg is short, the search is not failed. */ if (!budget?.isTimeout(error)) throw error - return { kind: 'unbounded', broad: true } + return [] } } /** - * Resolve the permitted set with the candidate predicate both legs apply, so restricting a leg - * to it never admits a document the leg would otherwise refuse. Tag filters stay chunk-level in - * each leg; the set is the document-level superset they narrow. - * - * It runs ahead of both legs on the vector leg's budget, so exhausting that budget here reports - * `unbounded` and marks the vector leg timed out rather than failing the keyword leg with it. + * Tag-only retrieval, decided on the document, in chunk-id order. A signed-in reader's tag leg + * runs under the search deadline in one statement; a caller with no person behind it runs + * unbudgeted, many bases in parallel. Each base is read up to the leg's whole `topK`, so the merged + * page is the same one a single statement over every base would return. A search holding a source + * whose reader must be proven live pages its candidates instead, so the proof is asked for only + * when one is read. */ -export async function resolvePermittedDocuments(params: { - knowledgeBaseIds: string[] - access: KnowledgeAccessScope - filters?: WorkspaceSearchFilters - budget?: SearchBudget - accessPlan?: SearchAccessPlan -}): Promise { - const key = reachKey(params.knowledgeBaseIds, params.access, params.accessPlan) - let probe: ProbeOutcome - let broad = true - /** - * A remembered reach says how much of the bases the caller reads, which a date filter does not - * change; the filtered set still has to be enumerated, so under one the probe always runs. - */ - /** A plan under a date or source filter enumerates the filtered set directly; reach cannot stand in for it. */ - const filteredDirectly = Boolean( - params.accessPlan && (dateFilterCondition(params.filters) || params.filters?.source) - ) - const remembered = key && !filteredDirectly ? saturatedReach.get(key) : undefined - if (remembered) { - probe = { kind: 'saturated' } - broad = remembered.broad - } else { - try { - probe = await probeVisibleDocuments( - params.knowledgeBaseIds, - candidateDocumentConditions( - params.knowledgeBaseIds, - params.access, - params.filters, - candidateAccessCondition(params.access, params.accessPlan) - ), - params.access, - params.budget, - 'permitted_documents', - filteredDirectly ? 'direct' : 'reach-first' +export async function handleTagOnlySearch( + params: SearchParams, + read: DocumentReadAccess +): Promise { + const { knowledgeBaseIds, topK, structuredFilters } = params + + if (!structuredFilters || structuredFilters.length === 0) { + throw new Error('Tag filters are required for tag-only search') + } + params.signal?.throwIfAborted() + if (read.liveSourceAccess) { + return selectAuthorizedTagResults(params, read.rankCondition, read.liveSourceAccess) + } + + const budget = read.signedIn ? params.budget : undefined + const tagFilterConditions = getStructuredTagFilters(structuredFilters, embedding) + const visibility = getVisibilityConditions(params.filters, read.rankCondition) + const selectTagged = (kbScope: SQL) => + runSearchQuery(budget, 'tags.sql', (executor) => + executor + .select(getSearchResultFields(sql`0`.as('distance'))) + .from(embedding) + .innerJoin(document, eq(embedding.documentId, document.id)) + .leftJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) + .where(and(kbScope, ...visibility, ...tagFilterConditions)) + .orderBy(embedding.id) + .limit(topK) + ) + + return shortOnDeadline(budget, async () => { + if (!read.signedIn && getQueryStrategy(knowledgeBaseIds.length, topK).useParallel) { + const perBase = await Promise.all( + knowledgeBaseIds.map((kbId) => selectTagged(eq(embedding.knowledgeBaseId, kbId))) ) - } catch (error) { - if (!params.budget?.isTimeout(error)) throw error - probe = { kind: 'timed_out' } - } - if (probe.kind === 'saturated') { - try { - const reach = await countReach( - params.knowledgeBaseIds, - params.access, - params.budget, - params.accessPlan, - true - ) - /** A count that ran out of time decides this search only; the next one counts again. */ - if (reach?.empty) probe = { kind: 'documents', documents: [] } - else if (reach !== null) { - broad = reach.broad - if (key) saturatedReach.set(key, { broad }) - } - } catch (error) { - /** The leg's own deadline passed during the count: the leg is short, the search is not failed. */ - if (!params.budget?.isTimeout(error)) throw error - } + return perBase + .flat() + .sort((a, b) => compareStrings(a.id, b.id)) + .slice(0, topK) } - } - const permitted: PermittedDocuments = - probe.kind === 'documents' - ? { kind: 'bounded', documents: probe.documents } - : { kind: 'unbounded', broad } - annotateSearchDiagnostics({ - permittedDocuments: permitted.kind, - ...(probe.kind === 'documents' ? { permittedDocumentCount: probe.documents.length } : {}), + return selectTagged(inArray(embedding.knowledgeBaseId, knowledgeBaseIds)) }) - return permitted } /** - * Tags live on chunks, so a row qualifies when a chunk it joins to carries them — and only a - * chunk the search can actually return counts, or a document whose sole match is disabled would - * be admitted by a check that ranking then discards. + * How long a probe's finding that a reader's set is too large to rank exactly is remembered. A + * readable set that large moves slowly, and the finding only skips the rescue, so a stale answer + * costs the rescue of a set that has since shrunk below the probe limit, never access. */ -function chunkTagCondition(join: SQL, tagConditions: SQL[]): SQL | undefined { - if (!tagConditions.length) return undefined - return sql`EXISTS ( - SELECT 1 FROM ${embedding} - WHERE ${and(join, eq(embedding.enabled, true), ...tagConditions)} - )` -} +const SATURATED_READ_TTL_MS = 60 * 1000 /** - * How many chunks the sliced sources contribute to exact ranking. A caller's slice of mirrored - * sources — their mail, their files, the spaces they belong to — sits below this, and ranking that - * many exactly, on the projection's half-precision vectors, measures in tens of milliseconds. + * Readable sets a probe found too large to rank exactly. An underfilled walk over a large base + * would otherwise probe it again on every search, though the probe can only confirm the same answer. */ -const SOURCE_EXACT_CHUNK_LIMIT = 150_000 - -/** Sources whose own index a caller's ranking walks, and whether anything is left to rank exactly. */ -interface SourceVectorPlan { - walked: readonly string[] - sliced: readonly string[] -} +const saturatedReads = new LRUCache({ max: 10_000, ttl: SATURATED_READ_TTL_MS }) -/** - * How each readable source contributes its nearest chunks. - * - * Membership decides it, not a count: a member of a source reads essentially all of it, so its own - * index is walked and the graph's neighbours are chunks they can read. Every other source is - * sliced — mirrored permissions give a caller their own mail, their own files — and those slices - * are ranked exactly together, which is cheaper than a walk and exact by construction. A source - * the caller is a member of but which has no index of its own is sliced too. - */ -function planSourceVectorCandidates(input: { - plan: SearchAccessPlan - indexedSources: ReadonlySet -}): SourceVectorPlan { - const eligible = [ - ...new Set([ - ...input.plan.connectors.workspace, - ...input.plan.connectors.admin, - ...input.plan.connectors.members, - ]), - ] - const walked = input.plan.memberSources.filter((id) => input.indexedSources.has(id)) - const walking = new Set(walked) - return { walked, sliced: eligible.filter((id) => !walking.has(id)) } +/** Clears the saturated-read cache, for tests. */ +export function forgetSaturatedReads(): void { + saturatedReads.clear() } /** - * The nearest readable chunks, gathered per source and merged by distance. - * - * Walking one source at a time is what keeps recall: pgvector post-filters, so a walk over every - * source spends its scan budget on the sources this caller cannot read and returns few of their - * true neighbours. Inside one source they read, almost every neighbour qualifies. - * - * Nothing but the merged identities crosses the wire — each source's readable documents are - * resolved inside its own statement. + * What decides a probe's answer: the bases, the reader's tokens, the filters, the tags and the + * sources a live proof excluded. Any change to them is a different set. */ -async function selectSourceVectorCandidates(input: { - access: KnowledgeAccessScope - knowledgeBaseIds: string[] - plan: SearchAccessPlan - tagCondition: SQL | undefined - documentCondition: SQL | undefined - /** Sources the caller turned out not to hold, kept out of every source's ranking. */ - exclusion: SQL | undefined - /** Whether every projection row carries its mirrored columns, so a walk needs no document. */ - projectionFilled: boolean - candidateDistance: SQL - candidateLimit: number - budget?: SearchBudget -}): Promise { - const sources = planSourceVectorCandidates({ - plan: input.plan, - indexedSources: await indexedVectorSources(input.budget), - }) - annotateSearchDiagnostics({ - vectorRanking: 'per-source', - vectorSourcesWalked: sources.walked.length, - vectorSourcesSliced: sources.sliced.length, - }) - const base = and( - inArray(embeddingSearch.knowledgeBaseId, input.knowledgeBaseIds), - eq(embeddingSearch.enabled, true), - input.tagCondition, - input.exclusion - ) - type RankedChunks = Promise> - /** - * Walks one source's own index, or the sliced sources together when their slice saturated. - * Readability is decided on the row the walk visits — the source and ACL are mirrored there — - * so the graph is not stalled by a document lookup per candidate; the tag filter, which lives on - * the chunk, still joins. - */ - const onRow = projectionCandidateAccessCondition(embeddingSearch, input.access, input.plan, { - filled: input.projectionFilled, - }) - const walk = - (scope: SQL): (() => RankedChunks) => - () => - withVectorScanSettings( - (executor) => - executor.execute(sql` - SELECT ${PROJECTION_CANDIDATE_COLUMNS}, ${input.candidateDistance} AS distance - FROM ${embeddingSearch} /* on-row visibility */ - WHERE ${and( - base, - scope, - onRow, - input.documentCondition === undefined - ? undefined - : sql`EXISTS ( - SELECT 1 FROM ${document} - WHERE ${and(eq(document.id, embeddingSearch.documentId), input.documentCondition)} - )` - )} - ORDER BY ${input.candidateDistance} LIMIT ${input.candidateLimit}`), - input.budget, - 'vector.source_walk', - onRowWalkScanTuples(input.documentCondition, input.projectionFilled) - ) - const walks: Array<() => RankedChunks> = sources.walked.map((connectorId) => - walk(eq(embeddingSearch.connectorId, connectorId)) - ) - const slicedScope = sql`(${embeddingSearch.connectorId} IS NULL - OR ${embeddingSearch.connectorId} = ANY(${textArrayLiteral([...sources.sliced])}))` - /** One statement for every sliced source: their readable chunks, ranked exactly on the row. */ - /** - * One statement for the sliced sources and, with them, every uploaded document: uploads carry no - * connector, so a caller who is a member of all the indexed sources would otherwise rank none. - */ - const slice: Array<() => RankedChunks> = [ - async () => { - /** - * The sliced sources' readable chunks, decided on the row, ranked exactly: the ACL index - * enumerates them and `+ 0` keeps the planner off the graph. The chunks are counted one past - * the bound in the same statement, so a set too large to rank exactly is known before it is. - */ - const readableChunks = and( - base, - slicedScope, - onRow, - input.documentCondition === undefined - ? undefined - : sql`EXISTS (SELECT 1 FROM ${document} WHERE ${and(eq(document.id, embeddingSearch.documentId), input.documentCondition)})` - ) - const rows = await runSearchQuery(input.budget, 'vector.source_exact', (executor) => - executor.execute(sql` - WITH readable_chunks AS MATERIALIZED ( - SELECT ${PROJECTION_CANDIDATE_COLUMNS}, ${input.candidateDistance} AS distance - FROM ${embeddingSearch} - WHERE ${readableChunks} - LIMIT ${SOURCE_EXACT_CHUNK_LIMIT + 1} - ) - SELECT id, "documentId", "connectorId", distance + 0 AS distance, - (SELECT count(*) FROM readable_chunks) > ${SOURCE_EXACT_CHUNK_LIMIT} AS saturated - FROM readable_chunks - ORDER BY distance LIMIT ${input.candidateLimit}`) - ) - /** - * The slice enumerates readable chunks in no particular order, so a set past its bound - * would rank an arbitrary subset and could miss the nearest chunks entirely. Walk those - * sources instead: approximate, but drawn from the whole of them. - */ - if (!rows.some((row) => row.saturated)) return rows - annotateSearchDiagnostics({ vectorSlicedSaturated: true }) - return walk(slicedScope)() - }, - ] - /** A source whose search runs out of budget marks the leg partial; the others' results stand. */ - const scored = await mapWithConcurrency( - [...walks, ...slice], - SOURCE_RANKING_CONCURRENCY, - async (run) => { - try { - return await run() - } catch (error) { - if (!input.budget?.isTimeout(error)) throw error - return [] - } - } +function saturatedReadKey(params: SearchParams, excludedKey: string): string { + return sha256Hex( + JSON.stringify([ + [...params.knowledgeBaseIds].sort(), + params.access.kind, + [...params.access.tokens].sort(), + params.filters ?? null, + params.structuredFilters ?? null, + excludedKey, + ]) ) - const ranked: Array = scored.flat() - return ranked - .sort((a, b) => Number(a.distance) - Number(b.distance)) - .slice(0, input.candidateLimit) } -/** Sources ranked at once; each holds a connection for its own statement. */ -const SOURCE_RANKING_CONCURRENCY = 3 - /** - * Select a bounded candidate pool and rerank it against the original vectors. - * - * A bounded ANN traversal fills that pool. When visibility leaves the traversal short of its - * limit, a bounded probe decides whether the permitted set is small enough to rank exactly - * instead, which recovers the candidates the traversal's post-filter discarded. + * Vector retrieval, decided on the document: a bounded candidate pool, then hydration of each page + * under the full read predicate, scored on the original vectors. * - * Live source authorization and content hydration still run after candidate ranking. - */ -async function selectVectorResults(params: SearchParams): Promise { - const queryVector = params.queryVector! - /** The walk and the candidate threshold use the projection's score, which stays in cache; the page is scored on the original vectors at hydration. */ - const distance = embeddingCandidateDistance( - queryVector.dimensions, - queryVector.vector, - queryVector.model - ) - const tagConditions = getStructuredTagFilters(params.structuredFilters ?? [], embedding) - const conditions = [ - inArray(embedding.knowledgeBaseId, params.knowledgeBaseIds), - ...tagConditions, - sql`${distance} < ${params.distanceThreshold!}`, - ] - /** Applied before the candidate limit, so the limit never counts rows the tags exclude. */ - const candidateTagCondition = chunkTagCondition( - eq(embedding.id, embeddingSearch.id), - tagConditions - ) - const accessProvider = params.access.kind === 'user' ? params.accessProvider : undefined - /** Only live-verified readers may defer source authorization until after candidate ranking. */ - const candidateAccess = accessProvider - ? candidateAccessCondition(params.access, params.accessPlan) - : knowledgeAccessCondition(params.access) - const candidateDistance = distance - const documentTagCondition = chunkTagCondition( - eq(embedding.documentId, document.id), - tagConditions - ) - /** - * What an on-row walk still has to ask the document: the tags, which live on chunks, and the - * date filter, which the row does not carry. A bounded set never walks, so this only runs when - * the filtered documents were too many to enumerate. - */ - const dateCondition = dateFilterCondition(params.filters) - const documentCondition = - documentTagCondition || dateCondition ? and(documentTagCondition, dateCondition) : undefined - /** - * Candidate selection ignores the page offset — only the rerank pages over the pool — so a - * refill reuses the pool it already has. Excluding another source is the only thing that - * changes which candidates belong in it, and that resets the offset to zero anyway. - */ - let candidatePool: - | { - excludedKey: string - ids: SearchReadCandidate[] - limit: number - exhausted: boolean - /** Whether the pool's rows carry their source, so a page needs no read of its own. */ - filled: boolean - } - | undefined + * A bounded ANN traversal fills the pool, joining each visited row's document laterally so + * readability — the caller's tokens included — is decided before the limit counts it. When + * visibility leaves the traversal short of its limit, a bounded probe decides whether the readable + * set is small enough to rank exactly instead, which recovers the candidates the traversal's + * post-filter discarded. The traversal and the rescue both carry each candidate's document, so a + * page is a slice of the pool. + */ +export async function handleVectorSearch( + params: SearchParams, + read: DocumentReadAccess +): Promise { + const setup = prepareVectorLeg(params) + let candidatePool: VectorCandidatePool | undefined return selectAuthorizedSearchResults({ leg: 'vector', access: params.access, - liveSourceAccess: params.liveSourceAccess, - filters: params.filters, + liveSourceAccess: read.liveSourceAccess, signal: params.signal, budget: params.budget, topK: params.topK, compareResults: (a, b) => a.distance - b.distance, selectPage: async (limit, offset, excludedSources) => { - const excludedKey = [...excludedSources].sort().join(',') - const visibility = [ - ...getVisibilityConditions(params.access, params.filters, candidateAccess), - excludeSearchSources(excludedSources), - ] - const candidateDocumentVisibility = [ - ...candidateDocumentConditions( - params.knowledgeBaseIds, - params.access, - params.filters, - candidateAccess - ), - excludeSearchSources(excludedSources), - ] - /** Explicit document IDs are already a bounded scope, and retain exhaustive ordering. */ - const exactPage = async () => { - annotateSearchDiagnostics({ vectorRanking: 'exact' }) - const candidates = await runSearchQuery(params.budget, 'vector.exact', (executor) => - executor - .select({ ...SEARCH_READ_CANDIDATE_FIELDS, distance: distance.as('distance') }) - .from(embedding) - .innerJoin(document, eq(embedding.documentId, document.id)) - .leftJoin(embeddingSearch, eq(embeddingSearch.id, embedding.id)) - .where(and(...conditions, ...visibility)) - .orderBy(sql`(${distance}) + 0`, embedding.id) - .limit(limit) - .offset(offset) + const exclusion = excludeSearchSources(excludedSources) + if (params.filters?.documentIds?.length) { + return selectExactVectorPage( + setup, + params.budget, + [...getVisibilityConditions(params.filters, read.rankCondition), exclusion], + limit, + offset ) - return { candidates, nextOffset: offset + candidates.length } } - if (params.filters?.documentIds?.length) return exactPage() - if (params.permitted?.kind === 'bounded' && params.permitted.documents.length === 0) - return { candidates: [], nextOffset: offset } - const needed = offset + limit - if ( - candidatePool?.excludedKey !== excludedKey || - (candidatePool.ids.length < needed && !candidatePool.exhausted) - ) { - const candidateLimit = vectorCandidatePoolLimit( - needed, - candidatePool?.excludedKey === excludedKey ? candidatePool.limit : undefined - ) - const plan = params.access.kind === 'user' ? params.accessPlan : undefined - /** Two remembered facts, read together when neither is remembered. */ - const [filled, plannedIndexedSources] = await Promise.all([ - params.searchIndexOnly === true - ? isProjectionFilled('embedding_search', 'vector.projection_filled', params.budget) - : false, - plan?.memberSources.length ? indexedVectorSources(params.budget) : undefined, - ]) - /** - * A source the caller turned out not to hold is left out where the pool is built: the - * pool is the page's order now, so a denied source's chunks would otherwise keep their - * slots. The row's mirrored source decides it, unless the row is decided on its document. - */ - const excludedOnRow = excludeSearchSourcesOnRow(embeddingSearch, filled, excludedSources) - annotateSearchDiagnostics({ - vectorRanking: 'projection-walk', - vectorCandidateLimit: candidateLimit, - vectorCandidateScan: 'planned', - vectorCandidateDimensions: embeddingCandidateDimensions( - queryVector.dimensions, - queryVector.model - ), - }) - /** - * `+ 0` keeps the planner off the ANN index, and the permitted identities keep the scan on - * `embedding_search_document_lookup_idx`, so this reads what the permitted set costs - * rather than re-deriving permission across the whole index. Exact ranking also honours - * `statement_timeout`, which a traversal cannot. - */ - /** Ranks the set's chunks exactly; `read` are chunks a pool already holds, ranked past. */ - const rankPermittedExactly = async (documentIds: string[], read?: readonly string[]) => { - annotateSearchDiagnostics({ vectorRanking: 'exact-candidates' }) - if (!documentIds.length) return [] - return runSearchQuery(params.budget, 'vector.exact_candidates', (executor) => - executor.execute(sql` - SELECT ${PROJECTION_CANDIDATE_COLUMNS} - FROM ${embeddingSearch} - WHERE ${and( - inArray(embeddingSearch.knowledgeBaseId, params.knowledgeBaseIds), - eq(embeddingSearch.enabled, true), - sql`${embeddingSearch.documentId} = ANY(${textArrayLiteral(documentIds)})`, - read?.length - ? sql`NOT (${embeddingSearch.id} = ANY(${textArrayLiteral([...read])}))` - : undefined, - candidateTagCondition, - excludedOnRow - )} - ORDER BY (${candidateDistance}) + 0 LIMIT ${candidateLimit} - `) - ) - } - let selected: SearchReadCandidate[] - /** Set where a pool's end is known better than by its length. */ - let exhausted: boolean | undefined - /** - * The bounded ANN traversal is the whole candidate set. LIMIT keeps document - * authorization downstream of the traversal, with a primary-key lookup per candidate. - */ - const scopeOfWalk = and( - inArray(embeddingSearch.knowledgeBaseId, params.knowledgeBaseIds), - eq(embeddingSearch.enabled, true), - excludedOnRow - ) - const walkGraph = () => - withVectorScanSettings( + candidatePool = await readVectorCandidatePool( + candidatePool, + excludedSources, + offset, + limit, + async ({ excludedKey, candidateLimit }) => { + annotateVectorPoolPlanned(setup, candidateLimit) + const candidateDocumentVisibility = [ + ...candidateDocumentConditions( + params.knowledgeBaseIds, + params.filters, + read.rankCondition + ), + exclusion, + ] + /** + * The bounded ANN traversal is the whole candidate set. The lateral read decides each + * visited row on its document, a primary-key lookup per candidate, before LIMIT counts it. + */ + let selected: SearchReadCandidate[] = await withVectorScanSettings( (executor) => - executor.execute( - plan - ? sql` - SELECT ${PROJECTION_CANDIDATE_COLUMNS} - FROM ${embeddingSearch} /* on-row visibility */ - WHERE ${and( - scopeOfWalk, - projectionCandidateAccessCondition(embeddingSearch, params.access, plan, { filled }), - documentCondition === undefined - ? undefined - : sql`EXISTS (SELECT 1 FROM ${document} WHERE ${and(eq(document.id, embeddingSearch.documentId), documentCondition)})` - )} - ORDER BY ${candidateDistance} LIMIT ${candidateLimit} - ` - : sql` + executor.execute(sql` SELECT ${embeddingSearch.id} AS id, ${embeddingSearch.documentId} AS "documentId", visible.connector_id AS "connectorId" FROM ${embeddingSearch} CROSS JOIN LATERAL ( SELECT ${document.connectorId} AS connector_id FROM ${document} - WHERE ${and(eq(document.id, embeddingSearch.documentId), ...candidateDocumentVisibility, candidateTagCondition)} + WHERE ${and(eq(document.id, embeddingSearch.documentId), ...candidateDocumentVisibility, setup.candidateTagCondition)} LIMIT 1 ) AS visible - WHERE ${scopeOfWalk} - ORDER BY ${candidateDistance} LIMIT ${candidateLimit} - ` - ), + WHERE ${and( + inArray(embeddingSearch.knowledgeBaseId, params.knowledgeBaseIds), + eq(embeddingSearch.enabled, true) + )} + ORDER BY ${setup.distance} LIMIT ${candidateLimit} + `), params.budget, - 'vector.candidate_search', - plan ? onRowWalkScanTuples(documentCondition, filled) : undefined + 'vector.candidate_search' ) - /** - * A source the caller is a member of that has its own index is walked on its own, which - * beats ranking it exactly once it is large enough to have earned that index. - */ - const walksASource = plan?.memberSources.some( - (id) => plannedIndexedSources?.has(id) ?? false - ) - if ( - params.permitted?.kind === 'bounded' && - plan && - filled && - params.permitted.documents.length >= PERMITTED_EXACT_DOCUMENT_LIMIT - ) { /** - * A set this large costs more to rank exactly than to walk: exact ranking reads every - * chunk of every document in it, while the walk decides readability on the rows it - * visits and stops at its tuple cap. The walk answers whenever the set is a fair share - * of the graph; where it is not, the walk underfills and the exact ranking that was - * always complete takes over, so nothing is lost but the walk's bounded cost. - * - * The walk decides readability on the projection row, which is broader than the - * document predicate hydration applies, so a pool it filled can still run short of - * readable rows. That shortfall is what refills a pool: the refill is the exact ranking, - * complete over the set, ranked past the rows already read and placed behind them, so - * the pages keep their offsets and every refill is a full window of fresh rows. + * A full traversal is already the nearest readable chunks. An underfilled one is the + * signal that visibility removed neighbours the graph had already chosen: pgvector's HNSW + * post-filters by construction, so a readable set that is a small share of the index is + * discarded after the graph has committed to its neighbours, and widening the traversal + * cannot recover them. Ranking the readable set exactly does, while that set is small + * enough to afford. A set a recent probe found too large is not probed again. */ - const permittedIds = params.permitted.documents.map((entry) => entry.id) - const previous = - candidatePool?.excludedKey === excludedKey ? candidatePool.ids : undefined - if (previous) { - const exact = await rankPermittedExactly( - permittedIds, - previous.map((candidate) => candidate.id) - ) - selected = [...previous, ...exact] - exhausted = exact.length < candidateLimit - } else { - selected = await walkGraph() - if (selected.length < candidateLimit) - selected = await rankPermittedExactly(permittedIds) - } - } else if ( - params.permitted?.kind === 'bounded' && - (!walksASource || dateFilterCondition(params.filters) || params.filters?.source) - ) { - /** - * A bounded permitted set is ranked exactly without walking the graph first: the walk - * post-filters, so when the caller reads a small share of the index it spends its whole - * uninterruptible tuple budget and still returns almost none of their neighbours. A - * member's indexed source is otherwise walked instead, but not under a filter: the walk - * cannot see the date, and a filtered set is small by construction. - */ - selected = await rankPermittedExactly(params.permitted.documents.map((entry) => entry.id)) - } else if (plan && !(params.permitted?.kind === 'unbounded' && params.permitted.broad)) { - /** - * Readability follows sources, so each readable source is searched in its own index and - * the results merged. A member reads a source whole or barely at all: walking one source - * spends its budget among chunks they can read, where a walk over every source spends it - * on the sources they cannot. A caller whose reach is broad skips this: for them the - * whole graph's neighbours are mostly theirs already, and one walk is the cheaper answer. - */ - selected = await selectSourceVectorCandidates({ - access: params.access, - knowledgeBaseIds: params.knowledgeBaseIds, - plan, - exclusion: excludedOnRow, - projectionFilled: filled, - tagCondition: candidateTagCondition, - documentCondition, - candidateDistance, - candidateLimit, - budget: params.budget, - }) - } else { - selected = await walkGraph() - /** - * A full traversal is already the nearest permitted chunks, so nothing else is worth - * running. An underfilled one is the signal that visibility removed neighbours the graph - * had already chosen: pgvector's HNSW post-filters by construction — it declares no scan - * strategies and never reads the scan keys — so a permitted set that is a small share of - * the index is discarded after the graph has committed to its neighbours, and widening - * the traversal cannot recover them. - * - * Ranking the permitted set exactly does recover them, while that set is small enough to - * afford. An `unbounded` permitted set already proved it is not, so the probe is skipped. - */ - /** - * A walk that decides readability on the row is not discarded that way: it keeps - * walking, up to its cap, until the limit is met, so an underfilled on-row walk means - * the caller's readable chunks near the query are simply that few. - */ - if (selected.length < candidateLimit && params.permitted?.kind !== 'unbounded') { + const saturationKey = + selected.length < candidateLimit ? saturatedReadKey(params, excludedKey) : undefined + if (saturationKey !== undefined && !saturatedReads.has(saturationKey)) { const probe = await probeVisibleDocuments( - params.knowledgeBaseIds, - [...candidateDocumentVisibility, documentTagCondition], - params.access, + directVisibleDocumentsQuery([ + ...candidateDocumentVisibility, + setup.documentTagCondition, + ]), params.budget, 'vector.probe' ) + if (probe.kind === 'saturated') saturatedReads.set(saturationKey, true) if (probe.kind === 'documents') { annotateSearchDiagnostics({ vectorProbeDocumentCount: probe.documents.length }) - selected = await rankPermittedExactly(probe.documents.map(({ id }) => id)) + const sources = new Map(probe.documents.map((entry) => [entry.id, entry.connectorId])) + /** The source comes from the probed document, which decided readability. */ + const ranked = await rankVectorCandidatesExactly({ + setup, + knowledgeBaseIds: params.knowledgeBaseIds, + documentIds: [...sources.keys()], + columns: sql`${embeddingSearch.id} AS id, ${embeddingSearch.documentId} AS "documentId", NULL AS "connectorId"`, + candidateLimit, + budget: params.budget, + }) + selected = ranked.map((row) => ({ + ...row, + connectorId: sources.get(row.documentId) ?? null, + })) } } + const pool = gatheredVectorCandidatePool(excludedKey, selected, candidateLimit) + annotateVectorPoolSelected(selected.length, candidateLimit) + return pool } - /** A pool the walk could not fill, or one at the ceiling, is all the pages will ever get. */ - candidatePool = { - excludedKey, - ids: selected, - limit: candidateLimit, - exhausted: - (exhausted ?? selected.length < candidateLimit) || - candidateLimit >= MAX_VECTOR_CANDIDATES, - filled, - } - annotateSearchDiagnostics({ - vectorCandidateCount: selected.length, - vectorCandidateScan: selected.length < candidateLimit ? 'underfilled' : 'planned', - }) - } - /** - * The walk's order is the page's order, and the walk carries each candidate's document and - * source, so a page is a slice of the pool. Rescoring the pool against the original vectors - * here read one out-of-line vector per candidate from storage no cache holds, seconds on a - * query nobody had run before; the page is scored at hydration instead. - */ - if (candidatePool.filled) { - const slice = candidatePool.ids.slice(offset, offset + limit) - return { candidates: slice, nextOffset: offset + slice.length } - } + ) /** - * While the source and ACL fill runs, a row it has not reached carries no source, so the - * page's identities are read off the documents; a slice whose documents all went away since - * the walk is passed over, not mistaken for the pool's end. This read, and the `vector.page` - * stage with it, can go once every deployment's projection is filled. + * Rescoring the pool against the original vectors here would read one out-of-line vector per + * candidate; the page is scored at hydration instead. */ - for (let start = offset; start < candidatePool.ids.length; start += limit) { - const slice = candidatePool.ids.slice(start, start + limit) - const ranked = new Map(slice.map((candidate, index) => [candidate.id, index])) - const identities = await runSearchQuery(params.budget, 'vector.page', (executor) => - executor.execute(sql` - SELECT ${embeddingSearch.id} AS id, ${document.id} AS "documentId", - ${document.connectorId} AS "connectorId" - FROM ${embeddingSearch} - INNER JOIN ${document} ON ${document.id} = ${embeddingSearch.documentId} - WHERE ${embeddingSearch.id} = ANY(${textArrayLiteral(slice.map((candidate) => candidate.id))}) - `) - ) - if (!identities.length) continue - const page = [...identities].sort( - (a, b) => (ranked.get(a.id) ?? 0) - (ranked.get(b.id) ?? 0) - ) - return { candidates: page, nextOffset: start + slice.length } - } - return { candidates: [], nextOffset: candidatePool.ids.length } + return sliceVectorCandidatePool(candidatePool, offset, limit) }, hydrate: (ids, authorized) => - hydrateSearchCandidates( - ids, - authorized, - embeddingDistance(queryVector.dimensions, queryVector.vector).as('distance'), - params.filters, - conditions, - 'vector', - params.budget, - true - ), + hydrateVectorCandidates(ids, knowledgeAccessCondition(authorized), setup, params), }) } -export interface KeywordSearchParams { - knowledgeBaseIds: string[] - topK: number - access: KnowledgeAccessScope - accessProvider?: KnowledgeAccessProvider - liveSourceAccess?: LiveSourceAccess - signal?: AbortSignal - budget?: SearchBudget - query: string - /** Query embedding, so keyword-only hits still carry a real cosine distance. */ - queryVector: KnowledgeQueryVector - structuredFilters?: StructuredFilter[] - filters?: WorkspaceSearchFilters - /** Resolved once per user-scoped search; absent for resolved scopes and explicit documents. */ - permitted?: PermittedDocuments - /** Connector state resolved once per search, so no candidate re-derives it. */ - accessPlan?: SearchAccessPlan - /** Every base is an organization search index; only those are projected for Tin ranking. */ - searchIndexOnly?: boolean -} - /** - * Lexical (full-text) retrieval leg. Matches chunks against the generated - * `content_tsv` column via `websearch_to_tsquery`, which tolerates arbitrary - * user input and supports quoted phrases and `-negation`. + * Lexical (full-text) retrieval, decided on the document. Matches chunks against the generated + * `embedding.content_tsv` column via `websearch_to_tsquery`, which tolerates arbitrary user input + * and supports quoted phrases and `-negation`. * - * Results carry the true cosine distance rather than a placeholder, so callers - * can report `similarity` for rows only the lexical leg found. Unlike the vector - * leg there is no distance threshold — surfacing exact-token matches that are + * Results carry the original vector's cosine distance rather than a placeholder, so callers can + * report `similarity` for rows only the lexical leg found, on the same scale the vector leg uses. + * Unlike the vector leg there is no distance threshold — surfacing exact-token matches that are * semantically distant is the entire point of this leg. * - * Candidate gathering mirrors the vector leg: resolved scopes use the same - * per-base strategy, and live user scopes verify bounded pages from the same - * global ranking pool before hydrating content. - * - * Ranking and hydration are two steps on purpose. Projecting the cosine - * distance in the ranking query makes Postgres detoast the chunk's vector and - * compute a distance for *every* full-text match before the `LIMIT` - * applies — work that scales with how common the query term is rather than - * with `topK` (measured at ~59x the buffer reads on a 20k-chunk base for a term - * matching every row). Ranking therefore touches no vectors, and only the rows - * that survive the limit are hydrated. - * - * The live-scope ranking query runs in three stages: match, authorize, rank. The - * visibility predicate carries correlated subqueries — one per connector, one per - * search-integration decision — so evaluating it across a base ahead of the query costs a table - * pass priced by how many documents the base holds rather than by how many the query matched. - * Matching first restricts that predicate to the documents the query actually matched. - * - * Two details keep that ordering from paying the saving back. Restricting the predicate with - * `document.id = ANY (...)` rather than a subquery keeps the narrowed lookup on a bitmap scan, - * which prefetches, where a plain `IN (SELECT ...)` plans as an index walk that does not. And - * the match stage carries identifiers only: ranking every match rather than every *visible* - * match would detoast one text-search vector per match, which on a mid-frequency term costs - * more than the pass it replaces. - */ -export async function executeKeywordSearch(params: KeywordSearchParams): Promise { - const { knowledgeBaseIds, topK, query, queryVector, structuredFilters, access } = params + * Ranking and hydration are two steps on purpose. Projecting the cosine distance in the ranking + * query makes Postgres detoast the chunk's vector and compute a distance for *every* full-text + * match before the `LIMIT` applies — work that scales with how common the query term is rather + * than with `topK`. Ranking therefore touches no vectors, and only the rows that survive the limit + * are hydrated. A signed-in reader ranks in one statement under the search deadline (see + * {@link rankSignedInKeywordCandidates}); a caller with no person behind it ranks unbudgeted, many + * bases in parallel, each up to the leg's whole `topK` so the merged ranking is the global one. A + * search holding a source whose reader must be proven live pages its ranking instead, so the + * proof is asked for only when one of its candidates is read. + */ +async function executeKeywordSearch( + params: KeywordSearchParams, + read: DocumentReadAccess +): Promise { + const { knowledgeBaseIds, topK, query, queryVector, structuredFilters } = params if (!query.trim()) { return [] } + params.signal?.throwIfAborted() const tsQuery = sql`websearch_to_tsquery(${FTS_CONFIG}, ${query})` - const rankExpr = sql`ts_rank_cd(${embedding.contentTsv}, ${tsQuery})` const tagFilterConditions = structuredFilters?.length ? getStructuredTagFilters(structuredFilters, embedding) : [] + const budget = read.signedIn ? params.budget : undefined + /** Hydration pass: full rows scored on the original vectors, bounded to the survivors — one out-of-line vector read per returned row. */ + const hydrate = (ids: string[], accessCondition: SQL) => + hydrateSearchCandidates( + ids, + accessCondition, + embeddingDistance(queryVector.dimensions, queryVector.vector).as('distance'), + params.filters, + [], + 'keyword', + budget + ) + const signedInRanking = { + knowledgeBaseIds, + tsQuery, + tagFilterConditions, + documentConditions: candidateDocumentConditions( + knowledgeBaseIds, + params.filters, + read.rankCondition + ), + budget, + } - if (params.accessProvider && access.kind === 'user') { - const conditions = [ - inArray(embedding.knowledgeBaseId, knowledgeBaseIds), - sql`${embedding.contentTsv} @@ ${tsQuery}`, - ...tagFilterConditions, - ] - const candidateRank = sql`ts_rank_cd(${embeddingKeywordSearch.contentTsv}, ${tsQuery})` - /** - * A caller reaching past the permitted-set limit reads much of the index, so ranking every - * match before checking access is the leg's whole cost for a common term. Where the Tin - * projection is complete, BM25 ranks inside the bases first and access is checked only on the - * top of that ranking. A bounded set past the exact-ranking size is read on the row like an - * unbounded one: the bounded read materializes every chunk of the set before it matches a - * term, where a ranking decided on the row costs what the term matches. - */ - const accessPlan = access.kind === 'user' ? params.accessPlan : undefined - const largePermittedSet = - accessPlan !== undefined && - params.permitted?.kind === 'bounded' && - params.permitted.documents.length >= PERMITTED_EXACT_DOCUMENT_LIMIT - const onRowReader = params.permitted?.kind === 'unbounded' || largePermittedSet - let tinQuery: Awaited> = null - if (onRowReader && tagFilterConditions.length === 0) { - try { - tinQuery = await resolveTinKeywordQuery( - params.searchIndexOnly === true, - query, - FTS_CONFIG, - params.budget - ) - } catch (error) { - /** A leg whose deadline passed before it ranked anything is short, not failed. */ - if (!params.budget?.isTimeout(error)) throw error - return [] - } - } - if (onRowReader) annotateSearchDiagnostics({ keywordRanking: tinQuery ? 'tin' : 'gin' }) - /** A filled projection decides readability on the ranked row alone; none of its rows needs the document. */ - const tinFilled = - accessPlan && tinQuery - ? await isProjectionFilled( - 'embedding_keyword_tin', - 'keyword.projection_filled', - params.budget - ) - : false - /** The ranked CTE's mirrored columns, which the on-row predicates read. */ - const rankedTinRow = { - connectorId: sql`ranked_tin_chunks.connector_id`, - acl: sql`ranked_tin_chunks.acl`, - documentId: sql`ranked_tin_chunks.document_id`, - } - /** The projection predicate over the ranked CTE's mirrored columns, plus any excluded source. */ - const onRowKeywordVisibility = (excludedSources: readonly string[]) => - and( - projectionCandidateAccessCondition(rankedTinRow, access, accessPlan!, { - filled: tinFilled, - }), - dateFilterCondition(params.filters) - ? sql`EXISTS (SELECT 1 FROM ${document} WHERE ${and(sql`${document.id} = ranked_tin_chunks.document_id`, dateFilterCondition(params.filters))})` - : undefined, - excludeSearchSourcesOnRow(rankedTinRow, tinFilled, excludedSources) - ) - const documentConditions = (excludedSources: readonly string[]) => - and( - ...candidateDocumentConditions( - knowledgeBaseIds, - access, - params.filters, - candidateAccessCondition(access, params.accessPlan) - ), - excludeSearchSources(excludedSources) - ) - /** - * A page read on the row takes each candidate's source from the row, which is what decides - * whether its live source proof is asked for. A row decided on its document — not yet filled, - * or its document marked for the projector — takes it from the document: one primary-key read - * per such row of the page, after its limit, never per ranked row. - */ - const onRowPage = (ranked: SQL) => sql` - SELECT paged.id, paged."documentId", - CASE WHEN paged.decided_on_document - THEN (SELECT ${document.connectorId} FROM ${document} WHERE ${document.id} = paged."documentId") - ELSE paged."connectorId" - END AS "connectorId", - paged.keyword_rank - FROM (${ranked}) AS paged` - /** - * One page from the top of Tin's ranking. The window of ranked chunks widens while too few of - * them are readable to fill the page; if the widest window still cannot, the page is left to - * the GIN ranking, which covers every match. - */ - const selectTinPage = async ( - scopedQuery: SQL, - limit: number, - offset: number, - excludedSources: readonly string[] - ): Promise => { - /** - * A resolved scope decides readability on the ranked row. The windows widen while the page - * is short, a narrow reader's to a wide one sooner and no further, and what the widest - * cannot fill is left short rather than handed to a ranking over every match. A large - * bounded set is the exception on its first page: its bounded read was exhaustive, so the - * widest window that still falls short hands that page to the GIN ranking, which covers - * every match. A later page stays with Tin: the two rankers order differently, so an offset - * advanced through one cannot resume the other. - */ - const narrow = - accessPlan !== undefined && - ((params.permitted?.kind === 'unbounded' && !params.permitted.broad) || largePermittedSet) - const windows: readonly number[] = narrow ? NARROW_KEYWORD_WINDOWS : TIN_KEYWORD_WINDOWS - /** - * A narrow reader's page is the readable remainder of a wide ranking, and that ranking is - * the cost: each page would rank the window again to find the next few readable rows, so one - * statement returns as many as several pages could ask for. - */ - const pageLimit = narrow ? Math.max(limit, NARROW_KEYWORD_PAGE) : limit - for (const window of windows) { - if (window < offset + limit) continue - const [page] = await runSearchQuery(params.budget, 'keyword.tin', (executor) => - executor.execute<{ ranked: number; candidates: SearchReadCandidate[] }>(sql` - WITH ranked_tin_chunks AS MATERIALIZED ( - SELECT ${embeddingKeywordTin.id} AS id, ${embeddingKeywordTin.documentId} AS document_id, - ${embeddingKeywordTin.enabled} AS enabled, ${embeddingKeywordTin.connectorId} AS connector_id, - ${embeddingKeywordTin.acl} AS acl, - tin.full_score(${embeddingKeywordTin}.ctid) AS keyword_rank - FROM ${embeddingKeywordTin} - WHERE ${embeddingKeywordTin.content} ==> (${scopedQuery}) - ORDER BY keyword_rank DESC - LIMIT ${window} - ), page AS ( - ${ - accessPlan - ? /** - * Readability decided on the ranked row: its source and ACL are mirrored there, - * so a window of mostly unreadable chunks costs an array test per row, not a - * document lookup. The full predicate follows at hydration. - */ - onRowPage( - sql` - SELECT ranked_tin_chunks.id, ranked_tin_chunks.document_id AS "documentId", - ranked_tin_chunks.connector_id AS "connectorId", ranked_tin_chunks.keyword_rank, - ${projectionDecidedOnDocument(rankedTinRow, tinFilled)} AS decided_on_document - FROM ranked_tin_chunks /* on-row visibility */ - WHERE ranked_tin_chunks.enabled AND ${onRowKeywordVisibility(excludedSources)} - ORDER BY ranked_tin_chunks.keyword_rank DESC, ranked_tin_chunks.id - LIMIT ${pageLimit} OFFSET ${offset}` - ) - : sql` - SELECT ranked_tin_chunks.id, ${document.id} AS "documentId", - ${document.connectorId} AS "connectorId", - ranked_tin_chunks.keyword_rank - FROM ranked_tin_chunks INNER JOIN ${document} - ON ${document.id} = ranked_tin_chunks.document_id - WHERE ranked_tin_chunks.enabled - AND ${and(...candidateDocumentConditions(knowledgeBaseIds, access, params.filters, knowledgeAccessCondition(access)), excludeSearchSources(excludedSources))} - ORDER BY ranked_tin_chunks.keyword_rank DESC, ranked_tin_chunks.id - LIMIT ${pageLimit} OFFSET ${offset}` - } - ) - SELECT (SELECT count(*)::int FROM ranked_tin_chunks) AS ranked, - coalesce(( - SELECT json_agg(json_build_object( - 'id', page.id, 'documentId', page."documentId", 'connectorId', page."connectorId" - ) ORDER BY page.keyword_rank DESC, page.id) - FROM page - ), '[]'::json) AS candidates - `) - ) - annotateSearchDiagnostics({ keywordTinWindow: window }) - if ( - page.candidates.length >= limit || - page.ranked < window || - (accessPlan !== undefined && - !(largePermittedSet && offset === 0) && - window === windows[windows.length - 1]) - ) { - return { candidates: page.candidates, nextOffset: offset + page.candidates.length } - } - } - return null - } - /** Parenthesized where used: `==>` binds tighter than `||`. */ - const tinScope = tinQuery - ? sql`'(' || ${sql.join( - knowledgeBaseIds.map((id) => sql`knowledge_tin_base_token(${id}) || '^0'`), - sql` || ' OR ' || ` - )} || ') AND (' || ${tinQuery} || ')'` - : undefined - /** - * Tin and GIN order candidates differently, so a search that once handed a page to GIN stays - * with GIN: an offset advanced through one ranking cannot resume the other. - */ - let handedToGin = false - /** Keep readable identities and rank scalars separate so sorts never carry full text-search vectors. */ + if (read.liveSourceAccess) { return selectAuthorizedSearchResults({ leg: 'keyword', access: params.access, - liveSourceAccess: params.liveSourceAccess, - filters: params.filters, + liveSourceAccess: read.liveSourceAccess, signal: params.signal, - budget: params.budget, + budget, topK, selectPage: async (limit, offset, excludedSources) => { - /** - * A bounded permitted set confines matching to the chunks the caller may read, so a term - * common across the index is ranked only where it can surface. The visibility CTE below - * still re-applies the candidate predicate, so the restriction can only narrow. - */ - const permittedIds = - params.permitted?.kind === 'bounded' && !largePermittedSet - ? params.permitted.documents.map((entry) => entry.id) - : undefined - if (permittedIds?.length === 0) return { candidates: [], nextOffset: offset } - if (tinScope && !handedToGin) { - const tinPage = await selectTinPage(tinScope, limit, offset, excludedSources) - if (tinPage) return tinPage - handedToGin = true - annotateSearchDiagnostics({ keywordRanking: 'gin' }) - } - const baseScope = and( - inArray(embeddingKeywordSearch.knowledgeBaseId, knowledgeBaseIds), - eq(embeddingKeywordSearch.enabled, true) - ) - const chunkMatch = and( - sql`${embeddingKeywordSearch.contentTsv} @@ ${tsQuery}`, - tagFilterConditions.length - ? sql`EXISTS ( - SELECT 1 FROM ${embedding} WHERE ${embedding.id} = ${embeddingKeywordSearch.id} - AND ${and(...tagFilterConditions)} - )` - : undefined - ) - /** - * A bounded permitted set is read through its documents alone and matched row by row, at a - * cost linear in the permitted chunks. Offered the text or base indexes alongside, - * PostgreSQL may intersect the permitted chunks with every chunk in the base that holds the - * term or sits in the base; measured on an organization index that plan cost several - * times the direct read, and the direct read is never materially slower. The permitted - * documents were resolved inside these bases; the base check still applies to the rows - * read, so the read can never widen the scope. `OFFSET 0` keeps the read from being - * flattened back into an intersection; the alias lets the shared conditions bind to it. - */ - const matchedChunks = permittedIds - ? sql` - SELECT ${embeddingKeywordSearch.id} AS id, ${embeddingKeywordSearch.documentId} AS document_id - FROM ( - SELECT * FROM ${embeddingKeywordSearch} - WHERE ${embeddingKeywordSearch.documentId} = ANY(${textArrayLiteral(permittedIds)}) - OFFSET 0 - ) AS ${embeddingKeywordSearch} - WHERE ${and(baseScope, chunkMatch)}` - : sql` - SELECT ${embeddingKeywordSearch.id} AS id, ${embeddingKeywordSearch.documentId} AS document_id - FROM ${embeddingKeywordSearch} - WHERE ${and(baseScope, chunkMatch)}` - const candidates = await runSearchQuery(params.budget, 'keyword.sql', (executor) => - executor.execute(sql` - WITH matched_keyword_chunks AS MATERIALIZED (${matchedChunks} - ), visible_keyword_documents AS MATERIALIZED ( - SELECT ${document.id} AS id FROM ${document} - WHERE ${and( - sql`${document.id} = ANY (ARRAY(SELECT document_id FROM matched_keyword_chunks))`, - documentConditions(excludedSources) - )} - ), ranked_keyword_candidates AS MATERIALIZED ( - SELECT matched_keyword_chunks.id, matched_keyword_chunks.document_id, - ${candidateRank} AS keyword_rank - FROM matched_keyword_chunks INNER JOIN ${embeddingKeywordSearch} - ON ${embeddingKeywordSearch.id} = matched_keyword_chunks.id - WHERE matched_keyword_chunks.document_id IN (SELECT id FROM visible_keyword_documents) - ORDER BY keyword_rank DESC, matched_keyword_chunks.id - LIMIT ${limit} OFFSET ${offset} - ) - SELECT ranked_keyword_candidates.id, ${document.id} AS "documentId", - ${document.connectorId} AS "connectorId" - FROM ranked_keyword_candidates INNER JOIN ${document} - ON ${document.id} = ranked_keyword_candidates.document_id - ORDER BY ranked_keyword_candidates.keyword_rank DESC, ranked_keyword_candidates.id - `) - ) + const candidates = await rankSignedInKeywordCandidates({ + ...signedInRanking, + exclusion: excludeSearchSources(excludedSources), + limit, + offset, + }) return { candidates, nextOffset: offset + candidates.length } }, - /** Every candidate already matched the query where it was ranked; matching it again here would detoast one text-search vector per result. */ - hydrate: (ids, authorized) => - hydrateSearchCandidates( - ids, - authorized, - embeddingDistance(queryVector.dimensions, queryVector.vector).as('distance'), - params.filters, - [inArray(embedding.knowledgeBaseId, knowledgeBaseIds), ...tagFilterConditions], - 'keyword', - params.budget - ), + hydrate: (ids, authorized) => hydrate(ids, knowledgeAccessCondition(authorized)), }) } - const rankConditions = (kbScope: SQL | undefined) => - and( - kbScope, - ...getVisibilityConditions(access, params.filters), - sql`${embedding.contentTsv} @@ ${tsQuery}`, - ...tagFilterConditions - ) - + const rankExpr = sql`ts_rank_cd(${embedding.contentTsv}, ${tsQuery})` + const visibility = getVisibilityConditions(params.filters, read.rankCondition) /** Ranking pass: ids and relevance only, so no vector is read. */ - const rankRows = (kbScope: SQL | undefined, limit: number) => + const rankRows = (kbScope: SQL) => db .select({ id: embedding.id, keywordRank: rankExpr.as('keyword_rank') }) .from(embedding) .innerJoin(document, eq(embedding.documentId, document.id)) - .where(rankConditions(kbScope)) + .where( + and( + kbScope, + ...visibility, + sql`${embedding.contentTsv} @@ ${tsQuery}`, + ...tagFilterConditions + ) + ) .orderBy(sql`${rankExpr} DESC`) - .limit(limit) - - const strategy = getQueryStrategy(knowledgeBaseIds.length, topK) + .limit(topK) + + return shortOnDeadline(budget, async () => { + let topIds: string[] + if (read.signedIn) { + const ranked = await rankSignedInKeywordCandidates({ ...signedInRanking, limit: topK }) + topIds = ranked.map((row) => row.id) + } else if (getQueryStrategy(knowledgeBaseIds.length, topK).useParallel) { + const perBase = await Promise.all( + knowledgeBaseIds.map((kbId) => rankRows(eq(embedding.knowledgeBaseId, kbId))) + ) + topIds = perBase + .flat() + .sort((a, b) => b.keywordRank - a.keywordRank) + .slice(0, topK) + .map((row) => row.id) + } else { + const ranked = await rankRows(inArray(embedding.knowledgeBaseId, knowledgeBaseIds)) + topIds = ranked.map((row) => row.id) + } + if (topIds.length === 0) { + return [] + } + const rowById = new Map((await hydrate(topIds, read.rankCondition)).map((row) => [row.id, row])) + return topIds + .map((id) => rowById.get(id)) + .filter((row): row is SearchResult => row !== undefined) + }) +} - let ranked: { id: string; keywordRank: number }[] - if (strategy.useParallel) { - const parallelLimit = Math.ceil(topK / knowledgeBaseIds.length) + 5 - const perBase = await Promise.all( - knowledgeBaseIds.map((kbId) => rankRows(eq(embedding.knowledgeBaseId, kbId), parallelLimit)) +/** A signed-in reader's keyword ranking over the source chunks, in one statement; see {@link keywordCandidateRankingQuery}. */ +function rankSignedInKeywordCandidates(input: { + knowledgeBaseIds: string[] + tsQuery: SQL + tagFilterConditions: SQL[] + documentConditions: (SQL | undefined)[] + exclusion?: SQL + limit: number + offset?: number + budget: SearchBudget | undefined +}): Promise { + return runSearchQuery(input.budget, 'keyword.sql', (executor) => + executor.execute( + keywordCandidateRankingQuery({ + matchedChunks: sql` + SELECT ${embedding.id} AS id, ${embedding.documentId} AS document_id + FROM ${embedding} + WHERE ${and( + inArray(embedding.knowledgeBaseId, input.knowledgeBaseIds), + eq(embedding.enabled, true), + sql`${embedding.contentTsv} @@ ${input.tsQuery}`, + ...input.tagFilterConditions + )}`, + documentConditions: [...input.documentConditions, input.exclusion], + rankTable: embedding, + rank: sql`ts_rank_cd(${embedding.contentTsv}, ${input.tsQuery})`, + limit: input.limit, + offset: input.offset ?? 0, + }) ) - ranked = perBase.flat().sort((a, b) => b.keywordRank - a.keywordRank) - } else { - ranked = await rankRows(inArray(embedding.knowledgeBaseId, knowledgeBaseIds), topK) - } + ) +} - const topIds = ranked.slice(0, topK).map((row) => row.id) - if (topIds.length === 0) { - return [] +/** The legs of a search whose readability is decided on each candidate's document. */ +function documentRetrievalLegs(read: DocumentReadAccess): RetrievalLegs { + return { + tags: (params) => handleTagOnlySearch(params, read), + vector: (params) => handleVectorSearch(params, read), + keyword: (params) => executeKeywordSearch(params, read), } - - /** Hydration pass: full rows plus the projection's distance, bounded to the survivors. */ - const hydrated = await db - .select( - getSearchResultFields( - embeddingCandidateDistance( - queryVector.dimensions, - queryVector.vector, - queryVector.model - ).as('distance') - ) - ) - .from(embedding) - .innerJoin(document, eq(embedding.documentId, document.id)) - .leftJoin(embeddingSearch, eq(embeddingSearch.id, embedding.id)) - .leftJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) - .where(and(inArray(embedding.id, topIds), ...getVisibilityConditions(access, params.filters))) - - const rowById = new Map(hydrated.map((row) => [row.id, row])) - return topIds.map((id) => rowById.get(id)).filter((row): row is SearchResult => row !== undefined) } /** @@ -2603,24 +622,13 @@ export function fuseByReciprocalRank(rankedLists: SearchResult[][], topK: number return rankResults(fused, (row) => scores.get(row.id) ?? 0) } -export async function handleTagAndVectorSearch(params: SearchParams): Promise { - const { structuredFilters, queryVector, distanceThreshold } = params - if (!structuredFilters || structuredFilters.length === 0) { - throw new Error('Tag filters are required for tag and vector search') - } - if (!queryVector || !distanceThreshold) { - throw new Error('Query vector and distance threshold are required for tag and vector search') - } - return selectVectorResults(params) -} - /** * `hybrid` fuses lexical and vector retrieval; `vector` is the legacy * semantic-only path, kept as an opt-out. */ export type KnowledgeSearchMode = 'hybrid' | 'vector' -export interface ExecuteKnowledgeSearchParams { +interface ExecuteKnowledgeSearchParams { /** Optional vector-leg budget; keyword and tag retrieval keep their default budgets. */ vectorBudgetMs?: number knowledgeBaseIds: string[] @@ -2629,7 +637,6 @@ export interface ExecuteKnowledgeSearchParams { /** What the caller may read; resolved from the principal by the use case, never from input. */ access: KnowledgeAccessScope accessProvider?: KnowledgeAccessProvider - liveSourceAccess?: LiveSourceAccess signal?: AbortSignal searchMode: KnowledgeSearchMode /** Lets a recently modified document edge past a stale one of similar relevance; off by default. */ @@ -2639,8 +646,8 @@ export interface ExecuteKnowledgeSearchParams { queryVector?: KnowledgeQueryVector structuredFilters?: StructuredFilter[] filters?: WorkspaceSearchFilters - /** Every base is an organization search index; only those are projected for Tin ranking. */ - searchIndexOnly?: boolean + /** Runs the search-index retrieval legs; see `usesIndexedRetrieval`. */ + indexedRetrieval?: boolean } export interface RetrievalStatus { @@ -2653,15 +660,6 @@ export interface KnowledgeRetrievalResult { retrieval: RetrievalStatus } -/** Retrieval for a surface that cannot present a partial result as complete. */ -export async function executeKnowledgeSearch( - params: ExecuteKnowledgeSearchParams -): Promise { - const result = await retrieveKnowledgeSearch(params) - if (result.retrieval.status === 'partial') throw new SearchDeadlineError() - return result.rows -} - /** Shared hybrid retrieval, with explicit completeness for surfaces that can represent it. */ export async function retrieveKnowledgeSearch( params: ExecuteKnowledgeSearchParams @@ -2674,6 +672,7 @@ export async function retrieveKnowledgeSearch( queryVector, structuredFilters, access, + accessProvider, boostRecency = false, } = params params.signal?.throwIfAborted() @@ -2686,6 +685,8 @@ export async function retrieveKnowledgeSearch( keyword: new SearchBudget('keyword', deadline, params.signal), tags: new SearchBudget('tags', deadline, params.signal), } + const hasQuery = Boolean(query?.trim()) + const hasFilters = Boolean(structuredFilters?.length) const finish = async (rows: SearchResult[]): Promise => { params.signal?.throwIfAborted() const timedOutLegs = Object.values(budgets) @@ -2699,122 +700,68 @@ export async function retrieveKnowledgeSearch( retrieval: { status: timedOutLegs.length ? 'partial' : 'complete', timedOutLegs }, } } + if (!hasQuery && !hasFilters) throw new Error('A search query or tag filters are required') + if (hasQuery && !queryVector) { + throw new Error('Query vector is required when searching with a query') + } /** - * Connector state is the same for every document a connector owns, so both legs read it from - * one resolution instead of proving it per candidate. + * The one seam between the two retrieval strategies. A signed-in reader's search over search + * indexes, while indexed organization search is on, binds the reader's resolved plan and ranks + * on the projection rows; the dormant module loads only then. Everything else decides + * readability on each candidate's document, under the caller's own tokens and any live source + * proof they hold. */ - const resolvedPlan = - access.kind === 'user' && params.accessProvider - ? await measureSearchStage('access_plan', () => - resolveSearchAccessPlan(knowledgeBaseIds, access) + const legs: RetrievalLegs = + access.kind === 'user' && accessProvider && params.indexedRetrieval === true + ? await (await import('@/lib/sim-search/indexed/retrieval')).prepareIndexedRetrieval({ + knowledgeBaseIds, + access, + accessProvider, + filters: params.filters, + signal: params.signal, + ranked: hasQuery, + budget: budgets.vector, + }) + : documentRetrievalLegs( + await resolveDocumentReadAccess(knowledgeBaseIds, access, accessProvider, params.signal) ) - : undefined - /** - * A source filter confines the plan rather than the rows: with only that kind of source - * eligible, every predicate the plan builds and every source the legs walk is that kind. - */ - const accessPlan = - resolvedPlan && params.filters?.source - ? restrictSearchAccessPlan(resolvedPlan, params.filters.source) - : resolvedPlan - const liveSourceAccess = liveSourceAccessFor( - access, - accessPlan, - params.accessProvider, - params.signal - ) const common = { knowledgeBaseIds, access, - accessProvider: params.accessProvider, signal: params.signal, filters: params.filters, structuredFilters, - liveSourceAccess, - searchIndexOnly: params.searchIndexOnly, } - const hasQuery = Boolean(query?.trim()) - const hasFilters = Boolean(structuredFilters?.length) if (!hasQuery) { - if (!hasFilters) throw new Error('A search query or tag filters are required') return finish( - await measureSearchStage('tags', () => - handleTagOnlySearch({ ...common, topK, budget: budgets.tags, accessPlan }) - ) + await measureSearchStage('tags', () => legs.tags({ ...common, topK, budget: budgets.tags })) ) } - if (!queryVector) throw new Error('Query vector is required when searching with a query') const { distanceThreshold } = getQueryStrategy(knowledgeBaseIds.length, topK) const legTopK = searchMode === 'hybrid' ? hybridCandidateCount(topK) : topK - /** - * Live user scopes resolve what they may read once, before either leg, so both rank inside it - * when it is small. Resolved scopes read whole bases, and explicit documents are already a - * bounded scope with their own exhaustive ordering. - */ - /** - * A filter that leaves few documents is enumerated and ranked exactly inside them, both legs: - * the row does not carry the document's date, and a keyword ranking of the whole base may hold - * few of a small source's matches. A filter that leaves many is ranked as the scope is — the - * source confined on the row, the date tested through the document — since a set that large - * holds most of the query's neighbours anyway. The planner's estimate decides which. - */ - /** Planning only, so a short cap of its own: running past it answers as the wide window it may be. */ - const estimateBudget = budgets.vector.capped(VECTOR_PROBE_BUDGET_MS) - const filters = params.filters - const enumerateFiltered = - accessPlan && filters && (dateFilterCondition(filters) || filters.source) - ? await estimateFilteredDocuments(knowledgeBaseIds, filters, accessPlan, estimateBudget) - .then((estimate) => estimate <= VECTOR_PROBE_DOCUMENT_LIMIT) - .catch((error) => { - if (!estimateBudget.isTimeout(error)) throw error - return false - }) - : false - const permitted = - access.kind === 'user' && params.accessProvider && !params.filters?.documentIds?.length - ? accessPlan && !enumerateFiltered - ? /** - * With readability decided on the projection row, a resolved scope never needs its - * readable documents enumerated ahead of ranking: its reach alone chooses between one - * walk over the whole graph and a search of each source. - */ - await resolveReach(knowledgeBaseIds, access, budgets.vector, accessPlan) - : await resolvePermittedDocuments({ - knowledgeBaseIds, - access, - filters: params.filters, - budget: budgets.vector, - accessPlan, - }) - : undefined - const vectorParams = { - ...common, - topK: legTopK, - queryVector, - distanceThreshold, - budget: budgets.vector, - permitted, - accessPlan, - } const vectorSearch = measureSearchStage('vector', () => - hasFilters ? handleTagAndVectorSearch(vectorParams) : handleVectorOnlySearch(vectorParams) + legs.vector({ + ...common, + topK: legTopK, + queryVector, + distanceThreshold, + budget: budgets.vector, + }) ) if (searchMode === 'vector') return finish(await vectorSearch) const keywordSearch = measureSearchStage('keyword', () => - executeKeywordSearch({ + legs.keyword({ ...common, topK: legTopK, query: query!, - queryVector, + queryVector: queryVector!, budget: budgets.keyword, - permitted, - accessPlan, }) ) - const legs = await Promise.allSettled([vectorSearch, keywordSearch]) + const settled = await Promise.allSettled([vectorSearch, keywordSearch]) /** Wait for both legs to release SQL resources; only deadline failures permit partial success. */ - for (const leg of legs) if (leg.status === 'rejected') throw leg.reason - const vectorResults = legs[0].status === 'fulfilled' ? legs[0].value : [] - const keywordResults = legs[1].status === 'fulfilled' ? legs[1].value : [] + for (const leg of settled) if (leg.status === 'rejected') throw leg.reason + const vectorResults = settled[0].status === 'fulfilled' ? settled[0].value : [] + const keywordResults = settled[1].status === 'fulfilled' ? settled[1].value : [] return finish(fuseByReciprocalRank([keywordResults, vectorResults], topK)) } diff --git a/apps/sim/lib/knowledge/search/search-index.ts b/apps/sim/lib/knowledge/search/search-index.ts index 25d4aea5998..b55cdb92730 100644 --- a/apps/sim/lib/knowledge/search/search-index.ts +++ b/apps/sim/lib/knowledge/search/search-index.ts @@ -29,7 +29,3 @@ export async function findSearchIndex( .limit(1) return index ? toActiveKnowledgeBaseReference(index) : null } - -export function findWorkspaceSearchIndex(workspaceId: string) { - return findSearchIndex({ kind: 'workspace', workspaceId }) -} diff --git a/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts b/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts index ca775a89f78..31c4e68ca10 100644 --- a/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts +++ b/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts @@ -1,72 +1,20 @@ -import { db } from '@sim/db' import { dbChainMockFns, resetDbChainMock } from '@sim/testing' import { beforeEach, describe, expect, it } from 'vitest' -import { - dropSourceVectorIndex, - ensureSourceVectorIndex, - indexedVectorSources, - SOURCE_INDEX_MIN_DOCUMENTS, -} from '@/lib/knowledge/search/source-vector-indexes' - -const CONNECTOR = '2bdd2c0c-1988-4a83-97b9-44562dd9e5f9' +import { dropSourceVectorIndex } from '@/lib/knowledge/search/source-vector-indexes' describe('source vector indexes', () => { - let indexed: Array<{ connectorId: string }> let statements: string[] - beforeEach(async () => { + beforeEach(() => { resetDbChainMock() - indexed = [] statements = [] dbChainMockFns.execute.mockImplementation(async (query: unknown) => { - const text = typeof query === 'string' ? query : JSON.stringify(query) - statements.push(text) - if (text.includes('pg_index')) return indexed - if (text.includes('sampled')) return [{ column: 'vector_512' }] + statements.push(typeof query === 'string' ? query : JSON.stringify(query)) return [] }) - /** The build reserves one connection; its statements are recorded with the rest. */ - Object.assign(db, { - $client: { - reserve: async () => ({ - unsafe: async (text: string) => { - statements.push(text) - /** The session lock is granted; no invalid leftover exists. */ - if (text.includes('pg_try_advisory_lock')) return [{ acquired: true }] - return [] - }, - release: () => undefined, - }), - }, - }) - /** Warms the catalog cache with this case's state, so a build decision is not a stale read. */ - expect((await indexedVectorSources()).size).toBe(indexed.length) - statements = [] - }) - - it('leaves the build to another sync that already holds the source', async () => { - dbChainMockFns.select.mockReturnValue({ - from: () => ({ where: async () => [{ documents: SOURCE_INDEX_MIN_DOCUMENTS }] }), - } as never) - Object.assign(db, { - $client: { - reserve: async () => ({ - unsafe: async (text: string) => { - statements.push(text) - if (text.includes('pg_try_advisory_lock')) return [{ acquired: false }] - return [] - }, - release: () => undefined, - }), - }, - }) - expect(await ensureSourceVectorIndex(CONNECTOR)).toBe(false) - expect(statements.some((text) => text.includes('CREATE INDEX'))).toBe(false) - expect(statements.some((text) => text.includes('DROP INDEX'))).toBe(false) }) it('never spells an unexpected identifier into DDL', async () => { - expect(await ensureSourceVectorIndex("x'; DROP TABLE document; --")).toBe(false) await dropSourceVectorIndex("x'; DROP TABLE document; --") expect(statements.some((text) => text.includes('DROP TABLE'))).toBe(false) }) diff --git a/apps/sim/lib/knowledge/search/source-vector-indexes.ts b/apps/sim/lib/knowledge/search/source-vector-indexes.ts index 9686c402ddd..4a12dea2038 100644 --- a/apps/sim/lib/knowledge/search/source-vector-indexes.ts +++ b/apps/sim/lib/knowledge/search/source-vector-indexes.ts @@ -1,20 +1,5 @@ import { db } from '@sim/db' -import { document, embeddingSearch } from '@sim/db/schema' -import { createLogger } from '@sim/logger' -import { getErrorMessage } from '@sim/utils/errors' -import { and, count, eq, isNull, sql } from 'drizzle-orm' -import { LRUCache } from 'lru-cache' -import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' - -const logger = createLogger('SourceVectorIndexes') - -/** - * Documents a source needs before it earns its own vector index. A walk of a source's own index - * costs a few milliseconds whatever its size once readability is decided on the row, where ranking - * a source exactly grows with it; below this a source is small enough that the difference does not - * matter, and its index would be maintenance without a return. - */ -export const SOURCE_INDEX_MIN_DOCUMENTS = 1_000 +import { sql } from 'drizzle-orm' /** * An index's name and predicate are spelled into DDL, which takes no parameters, so a connector id @@ -28,137 +13,12 @@ function indexName(connectorId: string): string { } /** - * The sources that have their own index, cached briefly: every unbounded ranking asks, and the - * answer changes only when a sync builds one or a deletion drops one. - */ -const indexedSources = new LRUCache<'sources', ReadonlySet>({ max: 1, ttl: 60 * 1000 }) - -/** Forgets the cached answer, after an index was built or dropped. */ -export function forgetIndexedVectorSources(): void { - indexedSources.clear() -} - -/** The sources with a graph of their own; a search that misses the memo reads under its own deadline. */ -export async function indexedVectorSources(budget?: SearchBudget): Promise> { - const cached = indexedSources.get('sources') - if (cached) return cached - const rows = await runSearchQuery(budget, 'vector.source_indexes', (executor) => - executor.execute<{ connectorId: string | null }>(sql` - SELECT substring(pg_get_expr(i.indpred, i.indrelid) from '''([0-9a-f-]+)''') AS "connectorId" - FROM pg_index i - JOIN pg_class c ON c.oid = i.indexrelid - WHERE i.indrelid = 'embedding_search'::regclass - AND c.relname LIKE 'embedding_search_src_%' AND i.indisvalid AND i.indisready`) - ) - const sources = new Set( - rows.map((row) => row.connectorId).filter((id): id is string => id !== null) - ) - indexedSources.set('sources', sources) - return sources -} - -/** The projection column a source's chunks rank on, or null where its base spans several widths. */ -async function projectionColumn(connectorId: string): Promise { - const [row] = await db.execute<{ column: string | null }>(sql` - SELECT CASE count(DISTINCT width) WHEN 1 THEN min(width) END AS column FROM ( - SELECT CASE - WHEN s.vector_512 IS NOT NULL THEN 'vector_512' - WHEN s.vector_384 IS NOT NULL THEN 'vector_384' - WHEN s.vector_768 IS NOT NULL THEN 'vector_768' - WHEN s.vector_1024 IS NOT NULL THEN 'vector_1024' - WHEN s.vector_3072 IS NOT NULL THEN 'vector_3072' - ELSE 'vector' - END AS width - FROM ${embeddingSearch} s - WHERE s.connector_id = ${connectorId} AND s.enabled - LIMIT 500 - ) sampled`) - return row?.column ?? null -} - -/** - * Drops an index only while it is unusable — a failed build leaves one behind, and the next build - * must clear it — on the session that goes on to build, so the check and the drop cannot straddle - * another session's build. - */ -async function dropInvalidIndex(session: ReservedSession, name: string): Promise { - const [row] = await session.unsafe>( - `SELECT NOT i.indisvalid OR NOT i.indisready AS invalid - FROM pg_class c JOIN pg_index i ON i.indexrelid = c.oid WHERE c.relname = $1`, - [name] - ) - if (row?.invalid) await session.unsafe(`DROP INDEX CONCURRENTLY IF EXISTS "${name}"`) -} - -type ReservedSession = Awaited> - -/** - * Gives a source its own vector index once it holds enough documents to need one, called after a - * sync that may have grown it. A member reads a source whole, so walking that source's index - * returns their own neighbours, where a walk over every source spends its scan budget on chunks - * they cannot read. - * - * Retrieval never waits on this: a source without an index is ranked exactly, which is what a - * smaller source affords anyway, so a build that is skipped, fails, or has not happened yet costs - * recall nothing. Builds run `CONCURRENTLY` and outside a transaction, so writers keep going; a - * failed build leaves an invalid index, which this drops before building again. + * Drops a deleted source's per-source vector index, where an earlier build left one. Nothing + * builds these any more; the dormant indexed search (`lib/sim-search/indexed/retrieval/`) still + * reads the ones that exist, and its brief cache of them only names sources a search can no + * longer reach once their connector is deleted. */ -export async function ensureSourceVectorIndex(connectorId: string): Promise { - if (!CONNECTOR_ID.test(connectorId)) return false - if ((await indexedVectorSources()).has(connectorId)) return false - const [{ documents }] = await db - .select({ documents: count() }) - .from(document) - .where(and(eq(document.connectorId, connectorId), isNull(document.deletedAt))) - if (documents < SOURCE_INDEX_MIN_DOCUMENTS) return false - const column = await projectionColumn(connectorId) - if (!column) return false - const name = indexName(connectorId) - const startedAt = Date.now() - /** - * The memory setting, the lock and the build must share a session, and `CONCURRENTLY` forbids a - * transaction, so one connection is reserved from the pool for the whole build. The session lock - * makes one build per source at a time: a sync that finds another build under way leaves it to - * that build, since the source is ranked exactly until an index exists either way. - */ - const session = await db.$client.reserve() - let locked = false - try { - const [lock] = await session.unsafe>( - 'SELECT pg_try_advisory_lock(hashtext($1)) AS acquired', - [`source_vector_index:${connectorId}`] - ) - locked = Boolean(lock?.acquired) - if (!locked) return false - await dropInvalidIndex(session, name) - await session.unsafe("SET maintenance_work_mem = '2GB'") - await session.unsafe(`CREATE INDEX CONCURRENTLY "${name}" ON embedding_search - USING hnsw (${column} halfvec_cosine_ops) WITH (m = 16, ef_construction = 64) - WHERE connector_id = '${connectorId}' AND enabled`) - logger.info('Built a source vector index', { - connectorId, - documents, - elapsedMs: Date.now() - startedAt, - }) - forgetIndexedVectorSources() - return true - } catch (error) { - logger.error('Source vector index build failed', { connectorId, error: getErrorMessage(error) }) - await dropInvalidIndex(session, name).catch(() => undefined) - return false - } finally { - await session.unsafe('RESET maintenance_work_mem').catch(() => undefined) - if (locked) - await session - .unsafe('SELECT pg_advisory_unlock(hashtext($1))', [`source_vector_index:${connectorId}`]) - .catch(() => undefined) - session.release() - } -} - -/** Drops a source's index when the source goes away; ranking falls back to the exact path. */ export async function dropSourceVectorIndex(connectorId: string): Promise { if (!CONNECTOR_ID.test(connectorId)) return await db.execute(sql.raw(`DROP INDEX CONCURRENTLY IF EXISTS "${indexName(connectorId)}"`)) - forgetIndexedVectorSources() } diff --git a/apps/sim/lib/knowledge/search/tag-filters.ts b/apps/sim/lib/knowledge/search/tag-filters.ts new file mode 100644 index 00000000000..d5025a2f50d --- /dev/null +++ b/apps/sim/lib/knowledge/search/tag-filters.ts @@ -0,0 +1,263 @@ +import { document, embedding } from '@sim/db/schema' +import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' +import { knowledgeAccessCondition } from '@/lib/knowledge/access/predicate' +import { runSearchQuery } from '@/lib/knowledge/search/budget' +import { + excludeSearchSources, + getVisibilityConditions, + hydrateSearchCandidates, + type LiveSourceAccess, + SEARCH_READ_CANDIDATE_FIELDS, + type SearchParams, + type SearchResult, + selectAuthorizedSearchResults, +} from '@/lib/knowledge/search/candidates' +import { + coerceTagFilterValue, + escapeLikePattern, + uncompilableTagFilterError, +} from '@/lib/knowledge/tags/utils' +import type { StructuredFilter } from '@/lib/knowledge/types' + +/** All valid tag slot keys */ +const TAG_SLOT_KEYS = [ + 'tag1', + 'tag2', + 'tag3', + 'tag4', + 'tag5', + 'tag6', + 'tag7', + 'number1', + 'number2', + 'number3', + 'number4', + 'number5', + 'date1', + 'date2', + 'boolean1', + 'boolean2', + 'boolean3', +] as const + +type TagSlotKey = (typeof TAG_SLOT_KEYS)[number] + +function isTagSlotKey(key: string): key is TagSlotKey { + return TAG_SLOT_KEYS.includes(key as TagSlotKey) +} + +/** The embedding columns a tag filter can compile against, one per tag slot. */ +type TagFilterTable = Pick + +/** + * Build a single SQL condition for a filter. Date values arrive as `YYYY-MM-DD` strings and + * compare as dates. + */ +function buildFilterCondition(filter: StructuredFilter, embeddingTable: TagFilterTable) { + const { tagSlot, fieldType, operator, value, valueTo } = filter + + if (!isTagSlotKey(tagSlot)) { + return null + } + + const column = embeddingTable[tagSlot] + if (!column) return null + + if (fieldType === 'text') { + const coerced = coerceTagFilterValue(value, 'text') + if (!coerced.ok) return null + const stringValue = coerced.value as string + const escaped = escapeLikePattern(stringValue) + switch (operator) { + case 'eq': + return sql`LOWER(${column}) = LOWER(${stringValue})` + case 'neq': + return sql`LOWER(${column}) != LOWER(${stringValue})` + case 'contains': + return sql`LOWER(${column}) LIKE LOWER(${`%${escaped}%`}) ESCAPE '\\'` + case 'not_contains': + return sql`LOWER(${column}) NOT LIKE LOWER(${`%${escaped}%`}) ESCAPE '\\'` + case 'starts_with': + return sql`LOWER(${column}) LIKE LOWER(${`${escaped}%`}) ESCAPE '\\'` + case 'ends_with': + return sql`LOWER(${column}) LIKE LOWER(${`%${escaped}`}) ESCAPE '\\'` + default: + return sql`LOWER(${column}) = LOWER(${stringValue})` + } + } + + if (fieldType === 'number') { + const coerced = coerceTagFilterValue(value, 'number') + if (!coerced.ok) return null + const numValue = coerced.value as number + + switch (operator) { + case 'eq': + return sql`${column} = ${numValue}` + case 'neq': + return sql`${column} != ${numValue}` + case 'gt': + return sql`${column} > ${numValue}` + case 'gte': + return sql`${column} >= ${numValue}` + case 'lt': + return sql`${column} < ${numValue}` + case 'lte': + return sql`${column} <= ${numValue}` + case 'between': + if (valueTo !== undefined) { + const coercedTo = coerceTagFilterValue(valueTo, 'number') + if (!coercedTo.ok) return sql`${column} = ${numValue}` + return sql`${column} >= ${numValue} AND ${column} <= ${coercedTo.value as number}` + } + return sql`${column} = ${numValue}` + default: + return sql`${column} = ${numValue}` + } + } + + if (fieldType === 'date') { + const coerced = coerceTagFilterValue(value, 'date') + if (!coerced.ok) return null + const dateStr = coerced.value as string + + switch (operator) { + case 'eq': + return sql`${column}::date = ${dateStr}::date` + case 'neq': + return sql`${column}::date != ${dateStr}::date` + case 'gt': + return sql`${column}::date > ${dateStr}::date` + case 'gte': + return sql`${column}::date >= ${dateStr}::date` + case 'lt': + return sql`${column}::date < ${dateStr}::date` + case 'lte': + return sql`${column}::date <= ${dateStr}::date` + case 'between': + if (valueTo !== undefined) { + const coercedTo = coerceTagFilterValue(valueTo, 'date') + if (!coercedTo.ok) { + return sql`${column}::date = ${dateStr}::date` + } + const dateStrTo = coercedTo.value as string + return sql`${column}::date >= ${dateStr}::date AND ${column}::date <= ${dateStrTo}::date` + } + return sql`${column}::date = ${dateStr}::date` + default: + return sql`${column}::date = ${dateStr}::date` + } + } + + if (fieldType === 'boolean') { + const coerced = coerceTagFilterValue(value, 'boolean') + if (!coerced.ok) return null + const boolValue = coerced.value as boolean + switch (operator) { + case 'eq': + return sql`${column} = ${boolValue}` + case 'neq': + return sql`${column} != ${boolValue}` + default: + return sql`${column} = ${boolValue}` + } + } + + return sql`${column} = ${value}` +} + +/** + * Build SQL conditions from structured filters with operator support. Every + * filter is a conjunct, including two that name the same tag. + * + * Search used to group filters by slot and OR same-slot conditions together, + * which made the two surfaces over the same tag vocabulary answer different + * questions: the document list ANDs every filter, so `gte 9` plus `lte 2` on one + * number tag returned nothing there and a full page of results from search — + * a widening on the billed endpoint, the same failure mode as dropping a filter. + * OR also made a range on a single text tag (`contains A` and `contains B`) + * inexpressible, while the union it produced stays reachable as separate + * searches. Neither contract ever documented the OR, so no caller could have + * been relying on it deliberately. + * + * Every filter reaching here has already been validated, so one that fails to + * compile is a defect rather than a predicate to skip. Skipping it dropped the + * tag term from the WHERE clause entirely and answered a filtered search with + * the whole knowledge base under a 200 — and search is billed, so the caller + * paid for the widened scan. It is reported as a validation failure instead. + */ +export function getStructuredTagFilters( + filters: StructuredFilter[], + embeddingTable: TagFilterTable +) { + return filters.map((filter) => { + const condition = buildFilterCondition(filter, embeddingTable) + if (condition === null) throw uncompilableTagFilterError(filter) + return condition + }) +} + +/** + * Tags live on chunks, so a row qualifies when a chunk it joins to carries them — and only a + * chunk the search can actually return counts, or a document whose sole match is disabled would + * be admitted by a check that ranking then discards. + */ +export function chunkTagCondition(join: SQL, tagConditions: SQL[]): SQL | undefined { + if (!tagConditions.length) return undefined + return sql`EXISTS ( + SELECT 1 FROM ${embedding} + WHERE ${and(join, eq(embedding.enabled, true), ...tagConditions)} + )` +} + +/** + * Tag-only candidates in id order under a candidate predicate, hydrated under the full read + * predicate once any live source proof a page needs is known. + */ +export function selectAuthorizedTagResults( + params: SearchParams, + candidateAccess: SQL, + liveSourceAccess: LiveSourceAccess | undefined +): Promise { + const conditions = [ + inArray(embedding.knowledgeBaseId, params.knowledgeBaseIds), + ...getStructuredTagFilters(params.structuredFilters ?? [], embedding), + ] + return selectAuthorizedSearchResults({ + leg: 'tags', + access: params.access, + liveSourceAccess, + signal: params.signal, + budget: params.budget, + topK: params.topK, + selectPage: async (limit, offset, excludedSources) => { + const candidates = await runSearchQuery(params.budget, 'tags.sql', (executor) => + executor + .select(SEARCH_READ_CANDIDATE_FIELDS) + .from(embedding) + .innerJoin(document, eq(embedding.documentId, document.id)) + .where( + and( + ...conditions, + ...getVisibilityConditions(params.filters, candidateAccess), + excludeSearchSources(excludedSources) + ) + ) + .orderBy(embedding.id) + .limit(limit) + .offset(offset) + ) + return { candidates, nextOffset: offset + candidates.length } + }, + hydrate: (ids, authorized) => + hydrateSearchCandidates( + ids, + knowledgeAccessCondition(authorized), + sql`0`.as('distance'), + params.filters, + conditions, + 'tags', + params.budget + ), + }) +} diff --git a/apps/sim/lib/knowledge/search/vector-leg.ts b/apps/sim/lib/knowledge/search/vector-leg.ts new file mode 100644 index 00000000000..2113961b7f9 --- /dev/null +++ b/apps/sim/lib/knowledge/search/vector-leg.ts @@ -0,0 +1,340 @@ +import { db } from '@sim/db' +import { document, embedding, embeddingSearch } from '@sim/db/schema' +import { createLogger } from '@sim/logger' +import { getErrorMessage, getPostgresErrorCode } from '@sim/utils/errors' +import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' +import { textArrayLiteral } from '@/lib/knowledge/access/predicate' +import { + runSearchQuery, + type SearchBudget, + type SearchExecutor, + sessionSettingsStatement, +} from '@/lib/knowledge/search/budget' +import { + hydrateSearchCandidates, + type KnowledgeQueryVector, + SEARCH_READ_CANDIDATE_FIELDS, + type SearchParams, + type SearchReadCandidate, + type SearchReadCandidatePage, +} from '@/lib/knowledge/search/candidates' +import { + annotateSearchDiagnostics, + measureSearchStage, + recordSearchStageDuration, + type SearchStage, +} from '@/lib/knowledge/search/diagnostics' +import { chunkTagCondition, getStructuredTagFilters } from '@/lib/knowledge/search/tag-filters' +import { + embeddingCandidateDimensions, + embeddingCandidateDistance, + embeddingDistance, +} from '@/lib/knowledge/vector-columns' + +const logger = createLogger('KnowledgeSearchCandidates') + +/** SQLSTATE for an unrecognised configuration parameter — pgvector older than 0.8. */ +const UNDEFINED_OBJECT_SQLSTATE = '42704' +/** + * Approximate iterative-visit threshold for a permission-starved graph walk. It excludes + * pgvector's initial beam, so it only takes effect once `ResumeScanItems` starts widening. + * + * It must therefore stay roughly an order of magnitude above `ef_search`, or the first beam + * already exhausts the tuple budget and the scan stops before it can iterate at all — pgvector's + * maintainer says as much in pgvector#912. Measured on a large corpus, pairing a 1,000-wide beam + * with a 1,000-tuple budget returned fewer candidates than a narrower beam allowed to iterate, + * and spent longer inside the one uninterruptible beam. + */ +export const CANDIDATE_HNSW_MAX_SCAN_TUPLES = 20_000 + +/** + * Beam width per iteration. A beam is the granularity of cancellation: pgvector calls + * `CHECK_FOR_INTERRUPTS` only while building an index, never inside `hnswgettuple`, so neither + * `statement_timeout` nor a cancellation request can interrupt one. A narrower beam that iterates + * therefore bounds the leg's uninterruptible floor as well as widening its reach. + */ +const CANDIDATE_HNSW_EF_SEARCH = '200' +const CANDIDATE_HNSW_SCAN_MEM_MULTIPLIER = '2' +/** + * Candidates one walk gathers: the pages the search has asked for so far and as many again, in + * case the full read predicate refuses some. The walk ends as soon as it has them, so a pool the + * size of a page ends long before one sized for a rerank; a pool the pages outrun is walked again, + * wider. The ceiling bounds the widest walk. + */ +const VECTOR_CANDIDATE_POOL_MIN = 200 +export const MAX_VECTOR_CANDIDATES = 1600 + +/** The pool a search needs to serve `needed` candidates, at least twice the last pool. */ +export function vectorCandidatePoolLimit(needed: number, previous: number | undefined): number { + return Math.min( + MAX_VECTOR_CANDIDATES, + Math.max(VECTOR_CANDIDATE_POOL_MIN, needed * 2, (previous ?? 0) * 2) + ) +} + +/** + * A vector leg's candidate pool. Candidate selection ignores the page offset — only hydration + * pages over the pool — so a page the hydrated rows outran reuses the pool it already has, and a + * pool the pages outran is gathered again, wider. Excluding a denied source changes which + * candidates belong in it. + */ +export interface VectorCandidatePool { + excludedKey: string + ids: SearchReadCandidate[] + limit: number + exhausted: boolean +} + +/** What a pool rebuild is given: the pool's key, its new limit, and the pool it widens, if any. */ +export interface VectorCandidatePoolRebuild

{ + excludedKey: string + candidateLimit: number + previous: P | undefined +} + +/** + * The pool a page reads: the current one while it still serves the page under the same excluded + * sources, otherwise the one `rebuild` gathers. + */ +export async function readVectorCandidatePool

( + pool: P | undefined, + excludedSources: readonly string[], + offset: number, + limit: number, + rebuild: (plan: VectorCandidatePoolRebuild

) => Promise

+): Promise

{ + const excludedKey = [...excludedSources].sort().join(',') + const needed = offset + limit + if (pool?.excludedKey === excludedKey && (pool.exhausted || pool.ids.length >= needed)) { + return pool + } + const previous = pool?.excludedKey === excludedKey ? pool : undefined + return rebuild({ + excludedKey, + candidateLimit: vectorCandidatePoolLimit(needed, previous?.limit), + previous, + }) +} + +/** + * A gathered pool. One the walk could not fill, or one at the ceiling, is all the pages will ever + * get; `knownExhausted` overrides the fill test where a pool's end is known better than by its + * length. + */ +export function gatheredVectorCandidatePool( + excludedKey: string, + selected: SearchReadCandidate[], + candidateLimit: number, + knownExhausted?: boolean +): VectorCandidatePool { + return { + excludedKey, + ids: selected, + limit: candidateLimit, + exhausted: + (knownExhausted ?? selected.length < candidateLimit) || + candidateLimit >= MAX_VECTOR_CANDIDATES, + } +} + +/** The pool's order is the page's order, so a page is a slice of it. */ +export function sliceVectorCandidatePool( + pool: VectorCandidatePool, + offset: number, + limit: number +): SearchReadCandidatePage { + const slice = pool.ids.slice(offset, offset + limit) + return { candidates: slice, nextOffset: offset + slice.length } +} + +/** How long to stop trying the iterative-scan settings after the server rejected them. */ +const HNSW_SETTINGS_UNSUPPORTED_RETRY_MS = 10 * 60 * 1000 + +let hnswSettingsUnsupportedUntil = 0 + +/** + * Shared HNSW indexes can be selective on KB scope, access, or tags, even for + * small workspace searches. Iterative scans keep looking within a bounded + * tuple budget. A transaction keeps the settings local under pooled connections; + * older extensions retry without tuning until the compatibility cooldown expires. + */ +export async function withVectorScanSettings( + run: (executor: SearchExecutor) => Promise, + budget: SearchBudget | undefined, + stage: SearchStage, + maxScanTuples: number = CANDIDATE_HNSW_MAX_SCAN_TUPLES +): Promise { + const untuned = () => runSearchQuery(budget, stage, run) + if (Date.now() < hnswSettingsUnsupportedUntil) return untuned() + const acquireStarted = performance.now() + const settings = [ + sql`set_config('hnsw.iterative_scan', 'relaxed_order', true)`, + sql`set_config('hnsw.max_scan_tuples', ${String(maxScanTuples)}, true)`, + sql`set_config('hnsw.ef_search', ${CANDIDATE_HNSW_EF_SEARCH}, true)`, + sql`set_config('hnsw.scan_mem_multiplier', ${CANDIDATE_HNSW_SCAN_MEM_MULTIPLIER}, true)`, + ] + /** Only a failure while the settings are being applied says the extension lacks them. */ + let applyingSettings = true + try { + /** Under a budget the settings ride in the deadline statement; alone they are one of their own. */ + if (budget) { + return await budget.query( + stage, + (tx) => { + applyingSettings = false + return run(tx) + }, + settings + ) + } + return await db.transaction(async (tx) => { + recordSearchStageDuration('vector.connection_acquire', performance.now() - acquireStarted) + await measureSearchStage('vector.settings', () => + tx.execute(sessionSettingsStatement(settings)) + ) + applyingSettings = false + return run(tx) + }) + } catch (error) { + if (!applyingSettings || getPostgresErrorCode(error) !== UNDEFINED_OBJECT_SQLSTATE) throw error + hnswSettingsUnsupportedUntil = Date.now() + HNSW_SETTINGS_UNSUPPORTED_RETRY_MS + logger.warn('pgvector iterative scan is unavailable; vector legs run without it', { + error: getErrorMessage(error), + }) + return untuned() + } +} + +/** What every vector leg derives from its query before it ranks a candidate. */ +export interface VectorLegSetup { + queryVector: KnowledgeQueryVector + /** The projection's score: the walk and the candidate threshold stay in cache; a page is scored on the original vectors at hydration. */ + distance: SQL + /** The bases, the tag filters and the threshold: what every hydrated row still meets. */ + conditions: SQL[] + /** Applied on the projection row before the candidate limit, so the limit never counts rows the tags exclude. */ + candidateTagCondition: SQL | undefined + /** The same tags, asked of a document: whether any chunk it can return carries them. */ + documentTagCondition: SQL | undefined +} + +/** The shared setup of a vector leg; a leg without a query vector and threshold is a caller defect. */ +export function prepareVectorLeg(params: SearchParams): VectorLegSetup { + const { queryVector, distanceThreshold } = params + if (!queryVector || !distanceThreshold) { + throw new Error('Query vector and distance threshold are required for vector search') + } + const distance = embeddingCandidateDistance( + queryVector.dimensions, + queryVector.vector, + queryVector.model + ) + const tagConditions = getStructuredTagFilters(params.structuredFilters ?? [], embedding) + return { + queryVector, + distance, + conditions: [ + inArray(embedding.knowledgeBaseId, params.knowledgeBaseIds), + ...tagConditions, + sql`${distance} < ${distanceThreshold}`, + ], + candidateTagCondition: chunkTagCondition(eq(embedding.id, embeddingSearch.id), tagConditions), + documentTagCondition: chunkTagCondition(eq(embedding.documentId, document.id), tagConditions), + } +} + +/** Explicit document IDs are already a bounded scope, and keep exhaustive ordering: one exact page. */ +export async function selectExactVectorPage( + setup: VectorLegSetup, + budget: SearchBudget | undefined, + visibility: (SQL | undefined)[], + limit: number, + offset: number +): Promise { + annotateSearchDiagnostics({ vectorRanking: 'exact' }) + const candidates = await runSearchQuery(budget, 'vector.exact', (executor) => + executor + .select({ ...SEARCH_READ_CANDIDATE_FIELDS, distance: setup.distance.as('distance') }) + .from(embedding) + .innerJoin(document, eq(embedding.documentId, document.id)) + .leftJoin(embeddingSearch, eq(embeddingSearch.id, embedding.id)) + .where(and(...setup.conditions, ...visibility)) + .orderBy(sql`(${setup.distance}) + 0`, embedding.id) + .limit(limit) + .offset(offset) + ) + return { candidates, nextOffset: offset + candidates.length } +} + +/** + * Ranks the chunks of a known set of documents exactly: `+ 0` keeps the planner off the ANN + * index, and the document identities keep the scan on the projection's document lookup, so this + * reads what the set costs. `columns` names the candidate identities each strategy reads off the + * row. + */ +export function rankVectorCandidatesExactly(input: { + setup: VectorLegSetup + knowledgeBaseIds: string[] + documentIds: readonly string[] + columns: SQL + conditions?: (SQL | undefined)[] + candidateLimit: number + budget: SearchBudget | undefined +}): Promise { + annotateSearchDiagnostics({ vectorRanking: 'exact-candidates' }) + if (!input.documentIds.length) return Promise.resolve([]) + return runSearchQuery(input.budget, 'vector.exact_candidates', (executor) => + executor.execute(sql` + SELECT ${input.columns} + FROM ${embeddingSearch} + WHERE ${and( + inArray(embeddingSearch.knowledgeBaseId, input.knowledgeBaseIds), + eq(embeddingSearch.enabled, true), + sql`${embeddingSearch.documentId} = ANY(${textArrayLiteral([...input.documentIds])})`, + input.setup.candidateTagCondition, + ...(input.conditions ?? []) + )} + ORDER BY (${input.setup.distance}) + 0 LIMIT ${input.candidateLimit} + `) + ) +} + +/** Records the pool a vector leg is about to gather. */ +export function annotateVectorPoolPlanned(setup: VectorLegSetup, candidateLimit: number): void { + annotateSearchDiagnostics({ + vectorRanking: 'projection-walk', + vectorCandidateLimit: candidateLimit, + vectorCandidateScan: 'planned', + vectorCandidateDimensions: embeddingCandidateDimensions( + setup.queryVector.dimensions, + setup.queryVector.model + ), + }) +} + +/** Records how much of its pool a vector leg gathered. */ +export function annotateVectorPoolSelected(selected: number, candidateLimit: number): void { + annotateSearchDiagnostics({ + vectorCandidateCount: selected, + vectorCandidateScan: selected < candidateLimit ? 'underfilled' : 'planned', + }) +} + +/** Loads a vector page under the read predicate, scored on the original vectors. */ +export function hydrateVectorCandidates( + ids: string[], + accessCondition: SQL, + setup: VectorLegSetup, + params: SearchParams +) { + return hydrateSearchCandidates( + ids, + accessCondition, + embeddingDistance(setup.queryVector.dimensions, setup.queryVector.vector).as('distance'), + params.filters, + setup.conditions, + 'vector', + params.budget, + true + ) +} diff --git a/apps/sim/lib/knowledge/transfer/bundle.test.ts b/apps/sim/lib/knowledge/transfer/bundle.test.ts index 6c75c8fda3c..1d1bca7fe4e 100644 --- a/apps/sim/lib/knowledge/transfer/bundle.test.ts +++ b/apps/sim/lib/knowledge/transfer/bundle.test.ts @@ -1,11 +1,6 @@ import { describe, expect, it } from 'vitest' import { MAX_KNOWLEDGE_BUNDLE_DOCUMENTS } from '@/lib/knowledge/constants' -import { - decodeVectorBase64, - encodeVectorBase64, - KnowledgeBundleVectorError, - knowledgeBundleManifestSchema, -} from '@/lib/knowledge/transfer/bundle' +import { knowledgeBundleManifestSchema } from '@/lib/knowledge/transfer/bundle' const DOCUMENT_ID = 'a2f1c3d4-1111-4222-8333-444455556666' @@ -123,20 +118,3 @@ describe('knowledgeBundleManifestSchema', () => { expect(knowledgeBundleManifestSchema.safeParse(manifest({ documents })).success).toBe(false) }) }) - -describe('vector codec', () => { - it('refuses a payload whose width differs from the declared dimension', () => { - expect(() => decodeVectorBase64(encodeVectorBase64([1, 2, 3]), 4)).toThrow( - KnowledgeBundleVectorError - ) - }) - - it('refuses non-finite values', () => { - expect(() => decodeVectorBase64(encodeVectorBase64([1, Number.NaN]), 2)).toThrow( - KnowledgeBundleVectorError - ) - expect(() => decodeVectorBase64(encodeVectorBase64([Number.POSITIVE_INFINITY, 1]), 2)).toThrow( - KnowledgeBundleVectorError - ) - }) -}) diff --git a/apps/sim/lib/knowledge/transfer/bundle.ts b/apps/sim/lib/knowledge/transfer/bundle.ts index cf03fa43c4a..4b47ae4a4ca 100644 --- a/apps/sim/lib/knowledge/transfer/bundle.ts +++ b/apps/sim/lib/knowledge/transfer/bundle.ts @@ -280,27 +280,3 @@ export function toManifestDocument( export function encodeVectorBase64(vector: readonly number[]): string { return Buffer.from(Float32Array.from(vector).buffer).toString('base64') } - -export class KnowledgeBundleVectorError extends Error { - constructor(message: string) { - super(message) - this.name = 'KnowledgeBundleVectorError' - } -} - -/** Reverses {@link encodeVectorBase64}, refusing any width or value pgvector could not store. */ -export function decodeVectorBase64(encoded: string, dimension: number): number[] { - const bytes = Buffer.from(encoded, 'base64') - if (bytes.byteLength !== dimension * Float32Array.BYTES_PER_ELEMENT) { - throw new KnowledgeBundleVectorError( - `Vector holds ${bytes.byteLength} bytes; expected ${dimension} float32 values` - ) - } - const vector = Array.from( - new Float32Array(bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength)) - ) - if (!vector.every(Number.isFinite)) { - throw new KnowledgeBundleVectorError('Vector contains a non-finite value') - } - return vector -} diff --git a/apps/sim/lib/knowledge/types.ts b/apps/sim/lib/knowledge/types.ts index 1aab5c59142..dfca6d3e5be 100644 --- a/apps/sim/lib/knowledge/types.ts +++ b/apps/sim/lib/knowledge/types.ts @@ -222,7 +222,3 @@ interface DocumentsPagination { /** The member engine's states, as stored on `knowledge_connector.member_sync_status`. */ export const MEMBER_SYNC_STATUSES = ['idle', 'pending', 'running', 'error', 'disabled'] as const export type MemberSyncStatus = (typeof MEMBER_SYNC_STATUSES)[number] - -export function isMemberSyncStatus(value: string): value is MemberSyncStatus { - return (MEMBER_SYNC_STATUSES as readonly string[]).includes(value) -} diff --git a/apps/sim/lib/mothership/application/load-search-integrations.test.ts b/apps/sim/lib/mothership/application/load-search-integrations.test.ts index e23c2eebe10..e1d38de0db7 100644 --- a/apps/sim/lib/mothership/application/load-search-integrations.test.ts +++ b/apps/sim/lib/mothership/application/load-search-integrations.test.ts @@ -42,6 +42,8 @@ const emptyPage: InventoryPage = { describe('loadCopilotSearchIntegrations', () => { beforeEach(() => { resetEnvFlagsMock() + /** The paged inventory below is the indexed arm; the live test opts back in. */ + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) authorizeChat.mockResolvedValue(undefined) listIntegrations.mockResolvedValue(emptyPage) }) @@ -162,7 +164,7 @@ describe('loadCopilotSearchIntegrations', () => { ...emptyPage, nextCursor: `page-${listIntegrations.mock.calls.length + 1}`, })) - await expect(loadCopilotSearchIntegrations(context)).rejects.toThrow('pagination limit') + await expect(loadCopilotSearchIntegrations(context)).rejects.toThrow('exceeded 100 pages') expect(listIntegrations).toHaveBeenCalledTimes(100) }) diff --git a/apps/sim/lib/mothership/application/load-search-integrations.ts b/apps/sim/lib/mothership/application/load-search-integrations.ts index 0baa56dceca..68c85837496 100644 --- a/apps/sim/lib/mothership/application/load-search-integrations.ts +++ b/apps/sim/lib/mothership/application/load-search-integrations.ts @@ -1,4 +1,3 @@ -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { knowledgeDelegationPolicy } from '@/lib/knowledge/application/authorization' import { listPersonalSearchIntegrations } from '@/lib/knowledge/application/personal-search-integrations' import { @@ -6,9 +5,10 @@ import { createTrustedOrganizationCopilotPrincipal, } from '@/lib/mothership/auth/application-delegation' import { authorizeOrganizationChatDelegation } from '@/lib/mothership/chat/organization-chats' +import { loadIndexedSearchIntegrationInventory } from '@/lib/sim-search/indexed' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { listLiveSearchAccounts } from '@/lib/sim-search/live/application' -const MAX_INVENTORY_PAGES = 100 const MAX_INVENTORY_BYTES = 256 * 1024 interface SearchIntegrationsContext { @@ -19,8 +19,6 @@ interface SearchIntegrationsContext { signal?: AbortSignal } -type IntegrationInventory = Awaited> - /** Loads the complete current person's Search inventory for one authenticated chat turn. */ export async function loadCopilotSearchIntegrations( context: SearchIntegrationsContext @@ -35,55 +33,33 @@ export async function loadCopilotSearchIntegrations( ) await authorizeOrganizationChatDelegation.execute({ principal }) - if (isLiveEnterpriseSearchEnabled) { - const [inventory, connections] = await Promise.all([ - listLiveSearchAccounts.execute({ - principal, - input: { organizationId: context.organizationId }, - }), - listPersonalSearchIntegrations.execute({ - principal, - input: { organizationId: context.organizationId }, - }), - ]) - context.signal?.throwIfAborted() - const result = JSON.stringify({ - ...inventory, - connections: connections.connections, - available: connections.available, - connectionGuidance: - 'Offer the exact available target or account action in a terminal tag to connect or reconnect in chat. Search uses the provider APIs directly. Account inventory alone does not establish permission to search; a provider search must succeed.', + if (isIndexedOrgSearchEnabled()) + return loadIndexedSearchIntegrationInventory({ + principal, + organizationId: context.organizationId, + signal: context.signal, + maxBytes: MAX_INVENTORY_BYTES, }) - if (Buffer.byteLength(result) > MAX_INVENTORY_BYTES) - throw new Error('Search integration inventory exceeds the prompt size limit') - return result - } - const connections: Array = [] - const available = new Map() - const cursors = new Set() - let cursor: string | undefined - for (let pageNumber = 0; pageNumber < MAX_INVENTORY_PAGES; pageNumber++) { - context.signal?.throwIfAborted() - const page = await listPersonalSearchIntegrations.execute({ + const [inventory, connections] = await Promise.all([ + listLiveSearchAccounts.execute({ principal, - input: { organizationId: context.organizationId, ...(cursor ? { cursor } : {}) }, - }) - context.signal?.throwIfAborted() - connections.push(...page.connections) - for (const entry of page.available) { - available.set(JSON.stringify(entry.target), entry) - } - const inventory = JSON.stringify({ connections, available: [...available.values()] }) - if (Buffer.byteLength(inventory) > MAX_INVENTORY_BYTES) { - throw new Error('Search integration inventory exceeds the prompt size limit') - } - if (page.nextCursor === null) return inventory - if (cursors.has(page.nextCursor)) { - throw new Error('Search integration inventory pagination did not advance') - } - cursors.add(page.nextCursor) - cursor = page.nextCursor - } - throw new Error('Search integration inventory exceeds the pagination limit') + input: { organizationId: context.organizationId }, + }), + listPersonalSearchIntegrations.execute({ + principal, + input: { organizationId: context.organizationId }, + }), + ]) + context.signal?.throwIfAborted() + const result = JSON.stringify({ + ...inventory, + connections: connections.connections, + available: connections.available, + connectionGuidance: + 'Offer the exact available target or account action in a terminal tag to connect or reconnect in chat. Search uses the provider APIs directly. Account inventory alone does not establish permission to search; a provider search must succeed.', + }) + if (Buffer.byteLength(result) > MAX_INVENTORY_BYTES) + throw new Error('Search integration inventory exceeds the prompt size limit') + return result } diff --git a/apps/sim/lib/mothership/chat/payload.test.ts b/apps/sim/lib/mothership/chat/payload.test.ts index 26651a0a420..2e510ec0453 100644 --- a/apps/sim/lib/mothership/chat/payload.test.ts +++ b/apps/sim/lib/mothership/chat/payload.test.ts @@ -1,4 +1,4 @@ -import { envFlagsMockFns, resetEnvFlagsMock, setEnvFlags, workflowsUtilsMock } from '@sim/testing' +import { envFlagsMockFns, resetEnvFlagsMock, workflowsUtilsMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' import { billingPlanHelpersMock } from '@sim/testing/mocks/billing-plan-helpers.mock' import { @@ -613,7 +613,6 @@ describe('Assistant payload', () => { }) it('discovers the existing GitHub PR-count tool with a personal credential in live Search', async () => { clearIntegrationToolSchemaCacheForTests() - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) mockSearchApprovals.mockResolvedValue(new Map([['github', true]])) vi.mocked(getExposedIntegrationTools).mockReturnValueOnce([ { diff --git a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts index c4cc5c05c3a..aea763d0c6c 100644 --- a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts +++ b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts @@ -1,16 +1,17 @@ +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { getMockLogger } from '@sim/testing/mocks/logger.mock' import { mothershipOrganizationChatsMock, mothershipOrganizationChatsMockFns, } from '@sim/testing/mocks/mothership-organization-chats.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const hoisted = vi.hoisted(() => ({ search: vi.fn(), read: vi.fn(), })) vi.mock('@/lib/mothership/chat/organization-chats', () => mothershipOrganizationChatsMock) -vi.mock('@/lib/knowledge/application/workspace-search', () => ({ +vi.mock('@/lib/sim-search/indexed', () => ({ searchOrganizationKnowledge: { get operation() { return knowledgeOperations.search @@ -23,8 +24,6 @@ vi.mock('@/lib/knowledge/application/workspace-search', () => ({ }, execute: hoisted.search, }, -})) -vi.mock('@/lib/knowledge/application/read-search-document', () => ({ readSearchDocument: { get operation() { return knowledgeOperations.readDocument @@ -60,7 +59,10 @@ const context = { }), } describe('Assistant retrieval tools', () => { + afterEach(resetEnvFlagsMock) beforeEach(() => { + /** These cover the indexed arm of Sim's search and read tools. */ + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) mocks.search.mockResolvedValue({ retrieval: { status: 'complete', timedOutLegs: [] }, knowledgeBases: [{ id: 'index', name: 'Enterprise Search' }], diff --git a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts index 1d178aba563..f1e205bc93a 100644 --- a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts +++ b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts @@ -4,14 +4,8 @@ import { readDocumentInputSchema, searchWorkspaceInputSchema, } from '@/lib/api/contracts/mothership-assistant-tools' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { getBaseUrl } from '@/lib/core/utils/urls' import { EmbeddingConfigurationError } from '@/lib/embeddings/configuration-error' -import { readSearchDocument } from '@/lib/knowledge/application/read-search-document' -import { - searchOrganizationKnowledge, - searchWorkspaceKnowledge, -} from '@/lib/knowledge/application/workspace-search' import { sourceAuthor } from '@/lib/knowledge/search/author' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' import { createKnowledgeDocumentCitation, liveCitationId } from '@/lib/knowledge/search/citation' @@ -31,6 +25,12 @@ import { } from '@/lib/mothership/application/execute-knowledge-use-case' import type { BaseServerTool, ServerToolContext } from '@/lib/mothership/tools/server/base-tool' import { connectorDisplayName } from '@/lib/sim-search/connectors' +import { + readSearchDocument, + searchOrganizationKnowledge, + searchWorkspaceKnowledge, +} from '@/lib/sim-search/indexed' +import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { readLiveDocument, searchLiveKnowledge } from '@/lib/sim-search/live/application' import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' @@ -38,9 +38,9 @@ const logger = createLogger('WorkspaceSearchTool') const CITATION_INSTRUCTION = 'Cite the evidence you use as {"id":""}. Use only IDs returned by these tools.' + - (isLiveEnterpriseSearchEnabled - ? ' When referring to a Slack conversation, link the returned sourceContainerName to its sourceContainerUrl when available.' - : '') + (isIndexedOrgSearchEnabled() + ? '' + : ' When referring to a Slack conversation, link the returned sourceContainerName to its sourceContainerUrl when available.') export const searchWorkspaceServerTool: BaseServerTool = { name: 'search_workspace', @@ -78,7 +78,7 @@ export const searchWorkspaceServerTool: BaseServerTool = { resultSecretRegistry: registry, signal: context?.abortSignal, } as const - if (isLiveEnterpriseSearchEnabled) { + if (!isIndexedOrgSearchEnabled()) { const nativeProjection = projectResolvedSecretModelContent( nativeQueries ?? [], registry @@ -237,7 +237,7 @@ export const readDocumentServerTool: BaseServerTool = { const input = readDocumentInputSchema.parse(raw) const registry = context?.resolvedSecretTraceRegistry if (!registry) throw new Error('Knowledge result provenance is unavailable') - if (isLiveEnterpriseSearchEnabled) { + if (!isIndexedOrgSearchEnabled()) { const liveInput = { ...input, filters: intersectWorkspaceSearchFilters( diff --git a/apps/sim/lib/sim-search/connectors.ts b/apps/sim/lib/sim-search/connectors.ts index fac1541618d..b8e892ebf8c 100644 --- a/apps/sim/lib/sim-search/connectors.ts +++ b/apps/sim/lib/sim-search/connectors.ts @@ -176,15 +176,6 @@ export interface OAuthServiceAvailabilityContext { isIntegrationAvailabilityReady: boolean } -export interface SearchConnectorAvailabilityContext extends OAuthServiceAvailabilityContext { - /** Whether per-member access is on for the workspace. */ - memberAccessAvailable: boolean - /** Whether someone already connected this source in the workspace. */ - hasConnection: boolean - /** Whether the viewer may turn a source on for the workspace; the first connect needs an admin. */ - canCreate: boolean -} - type SearchIntegrationAvailability = Pick & Partial> @@ -238,23 +229,6 @@ export function getConnectorAccessAvailability( } } -/** Why a source cannot be connected on this surface right now; null when it can. */ -export function searchConnectorUnavailableReason( - connector: SearchConnector, - integrationAvailability: ReadonlyMap, - context: SearchConnectorAvailabilityContext -): string | null { - if (!context.isIntegrationAvailabilityReady) return 'Source availability is not loaded yet' - if (!isSearchConnectorAvailable(connector, integrationAvailability, context)) { - return `${connector.meta.name} is unavailable in this deployment` - } - if (!context.memberAccessAvailable) return 'Per-member access is not available in this workspace' - if (!context.hasConnection && !context.canCreate) { - return `Ask a workspace admin to connect ${connector.meta.name} first` - } - return null -} - /** * Slack members authorize through the workspace's custom app. Other sources * use the deployment's OAuth client. The custom-app path remains available on diff --git a/apps/sim/lib/sim-search/indexed/README.md b/apps/sim/lib/sim-search/indexed/README.md new file mode 100644 index 00000000000..de0358f05cc --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/README.md @@ -0,0 +1,38 @@ +# Indexed organization search (dormant) + +The indexed backend for Sim Search: retrieval over `is_search_index` knowledge bases that organization and workspace connectors crawl into, ranked from the embedding projections. Live Search (`../live/`) replaced it. **This code is dormant**: it stays in the tree so it can be switched back on, but no request reaches it in a default deployment. + +Ordinary workspace knowledge bases, the Knowledge block, the embedding projections, the projector, and the document access predicate (`lib/knowledge/access/predicate.ts`) live outside this directory and behave the same whichever backend is selected. + +## The gate + +`isIndexedOrgSearchEnabled()` in `gate.ts` is the only switch. It is the inverse of `SIM_SEARCH_LIVE`, which defaults to `true`, so indexed search is on only where a deployment sets `SIM_SEARCH_LIVE=false`. Every use case in this directory calls `assertIndexedOrgSearchEnabled()` itself, so dormancy holds even for a caller that skipped the gate. + +While the gate is off: + +- Search, the MCP tools, and Sim's `search_workspace` and `read_document` tools serve Live Search, and personal Search integrations are read from live accounts. +- Indexed-only surfaces refuse with `SearchIndexDormantError` (a `409`): the Stats report and connecting a source that crawls into a search index. The indexed document page is not found. +- A knowledge search that names a search-index knowledge base (the Knowledge block, v1, v2, Sim's knowledge tool) still answers from the documents it already holds, decided on each document exactly as a workspace knowledge base is. +- Nothing crawls into search indexes: content syncs, member syncs, and processing recovery skip them (`lib/knowledge/connectors/indexing-policy.ts`). +- The projector owes search-index documents nothing: their marks are released with the rest, and it writes no Tin keyword rows. + +## Layout + +- `gate.ts`: the switch, `SearchIndexDormantError`, and the search-index helpers. Anything may import it. +- `index.ts`: the use-case barrel. Its callers branch on the gate first. +- `search/`, `documents/`, `mcp/`, `integrations/`: indexed search, document reads, the indexed MCP tools, and the indexed arms of the personal Search integration inventory. +- `retrieval/`: the search-index retrieval legs behind one entry, `prepareIndexedRetrieval`, which `lib/knowledge/search/queries.ts` loads with a dynamic import only for a signed-in reader whose every base is a search index while the gate is on. Every other search, including every workspace knowledge base search, decides readability on the document and reads none of it. + +The dormant UI sits in `indexed/` folders next to the component that picks it from `features.liveEnterpriseSearch` (`useDeploymentShape()`), so each can be deleted in one step: `app/o/[organizationId]/integrations/indexed/`, `app/o/[organizationId]/settings/components/integrations/indexed/`, and `app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/`. + +## Re-enabling + +1. Set `SIM_SEARCH_LIVE=false` in both the app and the Trigger.dev environment, and deploy. The container entrypoint (`apps/sim/bootstrap.ts`) mirrors it to `NEXT_PUBLIC_SIM_SEARCH_LIVE` for the client; crawling, processing, and projection read it in whichever process runs them. +2. Confirm the Tin objects exist (`0019_tin_keyword_projection`, `0024_knowledge_projection_async`), backfill `embedding_keyword_tin` for every search-index knowledge base, and build its index. +3. Resume and fully resync the connectors of search-index knowledge bases, so content that went stale while dormant is indexed again. + +Projection rows written before projections carried their document's source and ACL are decided on their document until they are rewritten. + +## Database objects it depends on + +`knowledge_base.is_search_index`, `document.acl`, `document.connector_id`, `knowledge_connector`, `embedding_search`, `embedding_keyword_search`, `embedding_keyword_tin`, `knowledge_projection_dirty`, and the Tin extension objects. All of them are owned by `packages/db`; no schema or migration belongs to this directory. diff --git a/apps/sim/lib/knowledge/application/read-indexed-document.ts b/apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts similarity index 98% rename from apps/sim/lib/knowledge/application/read-indexed-document.ts rename to apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts index 35ad2c41182..30b2fa5e947 100644 --- a/apps/sim/lib/knowledge/application/read-indexed-document.ts +++ b/apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts @@ -20,6 +20,7 @@ import { createKnowledgeDocumentSourceValue, importKnowledgePersistedResponseSecretProvenance, } from '@/lib/knowledge/secret-provenance' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import type { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' type IndexedKnowledgeDocumentTarget = @@ -88,6 +89,7 @@ function validateReadInput(input: ReadIndexedKnowledgeDocumentInput) { export const readIndexedKnowledgeDocument = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.readDocument, resolveContext: ({ input }: { input: ReadIndexedKnowledgeDocumentInput }) => { + assertIndexedOrgSearchEnabled() input.signal?.throwIfAborted() validateReadInput(input) return resolveKnowledgeOrganizationContext({ organizationId: input.organizationId }) diff --git a/apps/sim/lib/knowledge/application/read-search-document.test.ts b/apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts similarity index 95% rename from apps/sim/lib/knowledge/application/read-search-document.test.ts rename to apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts index cae1b3f6e3f..ce6a0052598 100644 --- a/apps/sim/lib/knowledge/application/read-search-document.test.ts +++ b/apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts @@ -1,13 +1,14 @@ import { member } from '@sim/db/schema' import { queueTableRows, resetDbChainMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeContextsMock, knowledgeContextsMockFns, } from '@sim/testing/mocks/knowledge-contexts.mock' import { permissionGroupsResolveMock } from '@sim/testing/mocks/permission-groups-resolve.mock' import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' vi.mock('@/lib/knowledge/search/search-index', () => ({ findSearchIndex: async () => ({ id: 'index' }), @@ -28,7 +29,7 @@ vi.mock('@/lib/execution/durable-secret-provenance', () => ({ importDurableSecretProvenance: mocks.importProvenance, })) -import { readSearchDocument } from '@/lib/knowledge/application/read-search-document' +import { readSearchDocument } from '@/lib/sim-search/indexed/documents/read-search-document' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' workspaceAuthzMockFns.mockPermissionSatisfies.mockImplementation( @@ -63,6 +64,11 @@ const input = { workspaceId: 'workspace', }), } + +/** These use cases run only while indexed organization search is on. */ +beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) +afterEach(resetEnvFlagsMock) + describe('Assistant document read', () => { beforeEach(() => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') diff --git a/apps/sim/lib/knowledge/application/read-search-document.ts b/apps/sim/lib/sim-search/indexed/documents/read-search-document.ts similarity index 98% rename from apps/sim/lib/knowledge/application/read-search-document.ts rename to apps/sim/lib/sim-search/indexed/documents/read-search-document.ts index 94aef40c21f..309754a0482 100644 --- a/apps/sim/lib/knowledge/application/read-search-document.ts +++ b/apps/sim/lib/sim-search/indexed/documents/read-search-document.ts @@ -13,6 +13,7 @@ import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' import { findSearchIndex } from '@/lib/knowledge/search/search-index' import { passageWindow } from '@/lib/knowledge/search/snippet' import { importKnowledgeSearchResultSecretProvenance } from '@/lib/knowledge/secret-provenance' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' import type { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' @@ -44,6 +45,7 @@ export const readSearchDocument = defineAuthorizedKnowledgeUseCase({ principal: Principal input: ReadSearchDocumentInput }) => { + assertIndexedOrgSearchEnabled() if (Boolean(input.assertedWorkspaceId) === Boolean(input.assertedOrganizationId)) throw new OrchestrationError('validation', 'Document reads require exactly one search owner') const index = await findSearchIndex( diff --git a/apps/sim/lib/sim-search/indexed/gate.ts b/apps/sim/lib/sim-search/indexed/gate.ts new file mode 100644 index 00000000000..9b5771ad870 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/gate.ts @@ -0,0 +1,42 @@ +import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' + +/** + * The single switch for indexed organization search: retrieval over `is_search_index` knowledge + * bases and the crawling that fills them. It is the inverse of the Live Search backend selector + * (`SIM_SEARCH_LIVE`, on by default), so indexed search is dormant unless a deployment sets + * `SIM_SEARCH_LIVE=false`. The selector is read once at startup, so the answer is constant for the + * life of the process. + */ +export function isIndexedOrgSearchEnabled(): boolean { + return !isLiveEnterpriseSearchEnabled +} + +/** An indexed-only surface was reached while indexed organization search is dormant. */ +export class SearchIndexDormantError extends Error { + constructor() { + super('This search index is inactive; use Sim Search.') + this.name = 'SearchIndexDormantError' + } +} + +/** + * Refuses entry to dormant indexed organization search. Every indexed use case and entry calls it + * itself, so dormancy holds even for a caller that forgot to ask the gate. + */ +export function assertIndexedOrgSearchEnabled(): void { + if (!isIndexedOrgSearchEnabled()) throw new SearchIndexDormantError() +} + +/** + * Whether a search runs the search-index retrieval legs: indexed organization search is on and + * every knowledge base it names is a search index. Every other search decides readability on + * each candidate's document. + */ +export function usesIndexedRetrieval( + knowledgeBases: ReadonlyArray<{ isSearchIndex?: boolean | null }> +): boolean { + return ( + isIndexedOrgSearchEnabled() && + knowledgeBases.every((knowledgeBase) => knowledgeBase.isSearchIndex === true) + ) +} diff --git a/apps/sim/lib/sim-search/indexed/index.ts b/apps/sim/lib/sim-search/indexed/index.ts new file mode 100644 index 00000000000..65d2f4e08a9 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/index.ts @@ -0,0 +1,18 @@ +/** + * Dormant indexed organization search: the use cases that search and read `is_search_index` + * knowledge bases, and the indexed arms of the personal Search integration inventory. Callers + * check `isIndexedOrgSearchEnabled()` before reaching these, and each refuses on its own while the + * gate is off; see this directory's README. + */ + +export { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' +export { readSearchDocument } from '@/lib/sim-search/indexed/documents/read-search-document' +export { ownsIndexedPersonalSearchAccount } from '@/lib/sim-search/indexed/integrations/personal-account-ownership' +export { loadIndexedSearchIntegrationInventory } from '@/lib/sim-search/indexed/integrations/personal-inventory' +export { listIndexedPersonalSearchIntegrations } from '@/lib/sim-search/indexed/integrations/personal-search-integrations' +export { registerIndexedKnowledgeMcpTools } from '@/lib/sim-search/indexed/mcp/register-tools' +export { + searchOrganizationKnowledge, + searchScopedKnowledge, + searchWorkspaceKnowledge, +} from '@/lib/sim-search/indexed/search/scoped-search' diff --git a/apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts b/apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts new file mode 100644 index 00000000000..561e7cbb039 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts @@ -0,0 +1,29 @@ +import type { Principal } from '@sim/auth/principal' +import { personalSearchIntegrationPages } from '@/lib/knowledge/application/personal-search-integration-pages' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' + +/** + * The indexed arm of organization personal-token ownership: whether the viewer's personal Search + * inventory lists `credentialId` as a connected account on this connector type. Indexed search + * answers ownership from its per-source inventory, where Live Search reads the live accounts. + */ +export async function ownsIndexedPersonalSearchAccount( + principal: Principal, + input: { organizationId: string; connectorType: string; credentialId: string } +): Promise { + assertIndexedOrgSearchEnabled() + for await (const page of personalSearchIntegrationPages({ + principal, + input: { organizationId: input.organizationId, connectorType: input.connectorType }, + })) { + if ( + page.connections.some((connection) => + connection.accounts.some( + (account) => account.credentialId === input.credentialId && account.status === 'connected' + ) + ) + ) + return true + } + return false +} diff --git a/apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts b/apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts new file mode 100644 index 00000000000..6b7309b7055 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts @@ -0,0 +1,43 @@ +import type { Principal } from '@sim/auth/principal' +import { + type PersonalSearchIntegrationsPage, + personalSearchIntegrationPages, +} from '@/lib/knowledge/application/personal-search-integration-pages' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' + +/** + * The indexed arm of Sim's Search inventory for one chat turn: every page of the viewer's + * personal connections on the indexed sources, merged and serialized for the prompt. Indexed + * inventory pages by source, where Live Search returns a single page. + */ +export async function loadIndexedSearchIntegrationInventory({ + principal, + organizationId, + signal, + maxBytes, +}: { + principal: Principal + organizationId: string + signal?: AbortSignal + maxBytes: number +}): Promise { + assertIndexedOrgSearchEnabled() + const connections: Array = [] + const available = new Map() + let inventory = JSON.stringify({ connections, available: [] }) + for await (const page of personalSearchIntegrationPages({ + principal, + input: { organizationId }, + signal, + })) { + connections.push(...page.connections) + for (const entry of page.available) { + available.set(JSON.stringify(entry.target), entry) + } + inventory = JSON.stringify({ connections, available: [...available.values()] }) + if (Buffer.byteLength(inventory) > maxBytes) { + throw new Error('Search integration inventory exceeds the prompt size limit') + } + } + return inventory +} diff --git a/apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts b/apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts new file mode 100644 index 00000000000..8762e38f8e9 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts @@ -0,0 +1,164 @@ +import type { Principal } from '@sim/auth/principal' +import { readSearchConnectionCompletion } from '@/lib/credential-groups/search-connection-completion' +import { + getIntegrationAvailability, + isOAuthServiceDeploymentAvailable, +} from '@/lib/integrations/availability.server' +import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' +import type { KnowledgeOrganizationContext } from '@/lib/knowledge/application/contexts' +import type { ListPersonalSearchIntegrationsInput } from '@/lib/knowledge/application/personal-search-integrations' +import { listConfiguredSearchProviderTypes } from '@/lib/knowledge/application/search-source-overview' +import { listSearchSources } from '@/lib/knowledge/application/search-sources' +import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' +import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' +import { getConnectorAccessAvailability, SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' +import { findSharedSlackSearchInstallation } from '@/lib/slack-search/shared-app' + +/** + * The indexed arm of `listPersonalSearchIntegrations`: the viewer's accounts on the connectors + * that crawl the organization search index, with each source's indexing state. The caller has + * authorized the principal and loaded the viewer; it reaches this only while + * `isIndexedOrgSearchEnabled()` is on. + */ +export async function listIndexedPersonalSearchIntegrations({ + principal, + input, + context, + userId, + viewer, +}: { + principal: Principal + input: ListPersonalSearchIntegrationsInput + context: KnowledgeOrganizationContext + userId: string + viewer: { emailVerified: boolean } +}) { + assertIndexedOrgSearchEnabled() + const [page, configuredTypes, approvals, access, sharedSlack] = await Promise.all([ + listSearchSources.execute({ principal, input }), + listConfiguredSearchProviderTypes({ organizationId: context.organizationId }), + listOrganizationSearchApprovals(context.organizationId), + resolveKnowledgeAccessAvailability(context), + findSharedSlackSearchInstallation(context.organizationId), + ]) + const deployment = new Map( + getIntegrationAvailability().map((entry) => [entry.type.toLowerCase(), entry]) + ) + const oauth = new Map( + SEARCH_CONNECTORS.map((entry) => [ + entry.providerId, + isOAuthServiceDeploymentAvailable(entry.providerId), + ]) + ) + const configured = new Set(configuredTypes) + const eligible = (connectorType: string) => { + const connector = SEARCH_CONNECTORS.find((entry) => entry.type === connectorType) + return Boolean( + viewer.emailVerified && + connector && + approvals.get(connectorType) && + getConnectorAccessAvailability(connector.meta, deployment, { + memberAccessAvailable: access.memberScoped, + mirroredAccessAvailable: access.sourceMirrored, + oauthServiceAvailability: oauth, + isIntegrationAvailabilityReady: true, + }).members + ) + } + const projected = page.sources.flatMap((source) => { + const connector = SEARCH_CONNECTORS.find((entry) => entry.type === source.connectorType) + if (!connector) return [] + const target: SearchConnectionTarget = { + type: 'link', + provider: connector.providerId, + connectorType: source.connectorType, + connectorId: source.connectorId, + } + const canConnect = + eligible(source.connectorType) && + source.enabled && + source.availability === 'available' && + source.viewerEmailVerified && + source.connectionRequired && + source.viewerMembership !== null && + !['revoked', 'unverified_email'].includes(source.viewerMembership) + const accounts = source.viewerAccounts.map((account) => { + if (!account.status) throw new Error('Personal Search account status is missing') + return { + credentialId: account.credentialId, + displayName: account.displayName, + status: + account.status === 'active' ? ('connected' as const) : ('reconnect_needed' as const), + action: + canConnect && account.status === 'needs_reauth' + ? { ...target, credentialId: account.credentialId } + : null, + } + }) + return [ + { + name: connector.meta.name, + providerId: connector.providerId, + connectorType: connector.type, + connectorId: source.connectorId, + knowledgeBaseId: source.knowledgeBaseId, + description: source.sourceDescription, + accounts, + connectionStatus: accounts.some((account) => account.status === 'reconnect_needed') + ? ('reconnect_needed' as const) + : accounts.length + ? ('connected' as const) + : canConnect + ? ('not_connected' as const) + : ('unavailable' as const), + indexingStatus: + !source.enabled || source.availability !== 'available' || source.approved === false + ? ('paused' as const) + : source.isSyncing + ? ('indexing' as const) + : source.hasSyncError || source.viewerFailedDocumentCount > 0 + ? ('sync_failed' as const) + : source.hasViewerDocuments + ? ('indexed' as const) + : ('not_indexed' as const), + action: canConnect && !accounts.length ? target : null, + }, + ] + }) + const available: Array<{ name: string; description: string; target: SearchConnectionTarget }> = [ + ...projected.flatMap((entry) => + entry.action + ? [{ name: entry.name, description: entry.description, target: entry.action }] + : [] + ), + ...SEARCH_CONNECTORS.filter( + (connector) => + !input.connectorId && + (!input.connectorType || connector.type === input.connectorType) && + (connector.type !== 'slack' || sharedSlack !== null) && + (!configured.has(connector.type) || connector.setupFields.length > 0) && + eligible(connector.type) + ).map((connector) => ({ + name: connector.meta.name, + description: '', + target: { + type: 'link' as const, + provider: connector.providerId, + connectorType: connector.type, + }, + })), + ] + return { + completedCredentialId: input.completionId + ? await readSearchConnectionCompletion({ + organizationId: context.organizationId, + userId, + completionId: input.completionId, + }) + : null, + connections: projected.filter((entry) => entry.accounts.length > 0), + available, + nextCursor: page.nextCursor, + } +} diff --git a/apps/sim/lib/sim-search/indexed/mcp/register-tools.ts b/apps/sim/lib/sim-search/indexed/mcp/register-tools.ts new file mode 100644 index 00000000000..174defef89b --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/mcp/register-tools.ts @@ -0,0 +1,148 @@ +import type { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js' +import type { Principal } from '@sim/auth/principal' +import type { NextRequest } from 'next/server' +import { readDocumentMcpSchema, searchMcpSchema } from '@/lib/api/contracts/knowledge/mcp' +import type { ResourceScope } from '@/lib/core/resource-scope' +import { getBaseUrl } from '@/lib/core/utils/urls' +import { knowledgeOperations } from '@/lib/knowledge/application/operations' +import { searchKnowledge } from '@/lib/knowledge/application/search' +import { + KNOWLEDGE_MCP_READ_ONLY, + type KnowledgeMcpToolRunner, + projectResult, +} from '@/lib/knowledge/mcp/tool-runner' +import { createKnowledgeDocumentCitation } from '@/lib/knowledge/search/citation' +import { toolError } from '@/lib/mcp/tool-result' +import { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' + +interface IndexedKnowledgeMcpToolsContext { + server: McpServer + principal: Principal + request: NextRequest + organizationId: string + /** The organization's search index, or null when no source is connected yet. */ + searchIndexId: string | null + execute: KnowledgeMcpToolRunner +} + +/** + * Registers the indexed `search` and `read_document` Search MCP tools: passages ranked from the + * organization's search index, and indexed documents read by id or source URL. Called only while + * indexed organization search is on, and refuses otherwise; both tools run through use cases that + * refuse a dormant search index on their own. + */ +export function registerIndexedKnowledgeMcpTools(context: IndexedKnowledgeMcpToolsContext): void { + assertIndexedOrgSearchEnabled() + const { server, principal, request, organizationId, searchIndexId, execute } = context + const scope: ResourceScope = { kind: 'organization', organizationId } + + server.registerTool( + 'search', + { + title: 'Search', + description: + 'Search accessible passages in this organization’s Search index. Use source (for example, jira), modifiedAfter (an ISO timestamp), or documentIds to narrow results. Results are candidates; score is similarity, not answer confidence. Use read_document for context and cite citationUrl.', + inputSchema: searchMcpSchema, + annotations: KNOWLEDGE_MCP_READ_ONLY, + }, + async (input: unknown, extra: { signal: AbortSignal }) => + execute('search', knowledgeOperations.search, extra.signal, async (registry, signal) => { + const { query, topK, ...filters } = searchMcpSchema.parse(input) + if (!searchIndexId) { + return projectResult( + { + results: [], + message: 'No Search index is configured. Ask an admin to connect a source.', + }, + registry + ) + } + const result = await searchKnowledge.execute({ + principal, + input: { + organizationId, + knowledgeBaseIds: [searchIndexId], + query, + topK, + filters, + resultSecretRegistry: registry, + surface: 'mcp', + signal, + }, + request, + }) + return projectResult( + { + results: result.results.map((row) => ({ + documentId: row.documentId, + title: row.documentName, + sourceUrl: row.sourceUrl, + ...createKnowledgeDocumentCitation({ + scope, + knowledgeBaseId: row.knowledgeBaseId, + documentId: row.documentId, + sourceUrl: row.sourceUrl, + baseUrl: getBaseUrl(), + }), + sourceModifiedAt: row.sourceModifiedAt?.toISOString() ?? null, + connectorType: row.connectorType, + content: row.content, + chunkIndex: row.chunkIndex, + score: row.similarity, + })), + }, + result.resultSecretRegistry ?? registry + ) + }) + ) + server.registerTool( + 'read_document', + { + title: 'Read document', + description: + 'Read an indexed document by documentId from search or its original URL. URLs must match an accessible indexed source; this tool does not browse the web. Set aroundChunkIndex to a search hit’s chunkIndex for nearby context, or use offset for sequential pages. When pagination.hasMore is true, continue with pagination.offset + pagination.limit. Cite citationUrl. Documents still indexing return metadata only.', + inputSchema: readDocumentMcpSchema, + annotations: KNOWLEDGE_MCP_READ_ONLY, + }, + async (raw: unknown, extra: { signal: AbortSignal }) => + execute( + 'read_document', + knowledgeOperations.readDocument, + extra.signal, + async (registry, signal) => { + const input = readDocumentMcpSchema.parse(raw) + if (!input.url && !input.documentId) return toolError('Document not found') + const result = await readIndexedKnowledgeDocument.execute({ + principal, + input: { + organizationId, + target: input.url + ? { kind: 'url', url: input.url } + : { kind: 'id', documentId: input.documentId! }, + limit: input.limit, + offset: input.offset, + aroundChunkIndex: input.aroundChunkIndex, + resultSecretRegistry: registry, + signal, + }, + request, + }) + const { knowledgeBaseId, ...document } = result + return projectResult( + { + ...document, + ...createKnowledgeDocumentCitation({ + scope, + knowledgeBaseId, + documentId: result.documentId, + sourceUrl: result.sourceUrl, + baseUrl: getBaseUrl(), + }), + }, + registry + ) + } + ) + ) +} diff --git a/apps/sim/lib/knowledge/access/connector-eligibility.ts b/apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts similarity index 64% rename from apps/sim/lib/knowledge/access/connector-eligibility.ts rename to apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts index fdf4cf2f992..eba0bc5aa4f 100644 --- a/apps/sim/lib/knowledge/access/connector-eligibility.ts +++ b/apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts @@ -2,16 +2,79 @@ import { db } from '@sim/db' import { knowledgeConnector, knowledgeConnectorMember } from '@sim/db/schema' import { and, eq, inArray, isNull, sql } from 'drizzle-orm' import { SOURCE_ACL_MAX_AGE_MS } from '@/lib/knowledge/access/freshness' -import { - type KnowledgeConnectorEligibility, - type KnowledgeMemberObserver, - type KnowledgeMemberObservers, - type SearchAccessPlan, - textArrayLiteral, -} from '@/lib/knowledge/access/predicate' +import { textArrayLiteral } from '@/lib/knowledge/access/predicate' import type { KnowledgeAccessScope } from '@/lib/knowledge/access/types' import { searchIntegrationAccessCondition } from '@/lib/knowledge/search/integration-policy' +/** + * The connectors a search may read from, resolved once per query: their ids grouped by the shape + * their documents' ACLs take, and separately those whose reader access is proven live per request. + */ +export interface KnowledgeConnectorEligibility { + /** Documents carry the workspace ACL. */ + workspace: readonly string[] + /** Documents carry mirrored source permissions verified as a whole. */ + admin: readonly string[] + /** Documents carry the subject tokens of the members who observe them. */ + members: readonly string[] + /** Of the above, those that additionally require this request's live source proof. */ + liveProofRequired: readonly string[] +} + +/** One of the caller's member identities and the connector it belongs to. */ +export interface KnowledgeMemberObserver { + id: string + connectorId: string +} + +/** + * The caller's active member identities on the connectors a search reads, by what makes their + * observations current: `confirmed` members drained their change feed inside the freshness window, + * so every observation they hold stands; `observed` members are trusted only where the observation + * itself is recent. + */ +export interface KnowledgeMemberObservers { + confirmed: readonly KnowledgeMemberObserver[] + observed: readonly KnowledgeMemberObserver[] +} + +/** What a search resolves once about its sources and the caller's standing in them. */ +export interface SearchAccessPlan { + connectors: KnowledgeConnectorEligibility + observers: KnowledgeMemberObservers + /** Connectors the caller is an active member of, whose documents they read broadly. */ + memberSources: readonly string[] + /** Each eligible connector's type, so a search may be confined to one kind of source. */ + connectorTypes: ReadonlyMap + /** Whether documents without a source — uploads — are in scope. */ + uploads: boolean +} + +/** + * The plan confined to one kind of source: the connectors of that type keep their eligibility and + * the rest lose it, so every predicate built from the plan — on the row and on the document — and + * every source the legs walk or rank are that kind alone. `upload` keeps only source-less documents. + */ +export function restrictSearchAccessPlan(plan: SearchAccessPlan, source: string): SearchAccessPlan { + const keep = (id: string) => source !== 'upload' && plan.connectorTypes.get(id) === source + const kept = (ids: readonly string[]) => ids.filter(keep) + return { + connectors: { + workspace: kept(plan.connectors.workspace), + admin: kept(plan.connectors.admin), + members: kept(plan.connectors.members), + liveProofRequired: kept(plan.connectors.liveProofRequired), + }, + observers: { + confirmed: plan.observers.confirmed.filter((observer) => keep(observer.connectorId)), + observed: plan.observers.observed.filter((observer) => keep(observer.connectorId)), + }, + memberSources: kept(plan.memberSources), + connectorTypes: plan.connectorTypes, + uploads: source === 'upload', + } +} + /** * The connectors a search may read from, grouped by access mode, with the ones whose reader access * must be proven live marked. diff --git a/apps/sim/lib/sim-search/indexed/retrieval/index.ts b/apps/sim/lib/sim-search/indexed/retrieval/index.ts new file mode 100644 index 00000000000..79691075482 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/index.ts @@ -0,0 +1,10 @@ +/** + * The search-index retrieval legs shared retrieval (`lib/knowledge/search/queries.ts`) runs for a + * user-scoped search over search indexes while indexed organization search is on. Everything that + * decides readability on the projection row lives behind this barrel: the resolved access plan + * and the projection-row predicates built from it, reach and permitted sets, per-source vector + * walks, the projection-fill probe, live source proof, and keyword ranking over the GIN and Tin + * projections. Kept apart from the use-case barrel because the use cases depend on that retrieval + * layer, which depends on these. + */ +export { prepareIndexedRetrieval } from '@/lib/sim-search/indexed/retrieval/legs' diff --git a/apps/sim/lib/sim-search/indexed/retrieval/keyword.ts b/apps/sim/lib/sim-search/indexed/retrieval/keyword.ts new file mode 100644 index 00000000000..c0016e77518 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/keyword.ts @@ -0,0 +1,325 @@ +import { document, embedding, embeddingKeywordSearch, embeddingKeywordTin } from '@sim/db/schema' +import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' +import { knowledgeAccessCondition, textArrayLiteral } from '@/lib/knowledge/access/predicate' +import { runSearchQuery } from '@/lib/knowledge/search/budget' +import { + candidateDocumentConditions, + excludeSearchSources, + FTS_CONFIG, + hydrateSearchCandidates, + type KeywordSearchParams, + type SearchReadCandidate, + type SearchReadCandidatePage, + type SearchResult, + selectAuthorizedSearchResults, +} from '@/lib/knowledge/search/candidates' +import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' +import { searchDateFilterCondition } from '@/lib/knowledge/search/filter-conditions' +import { keywordCandidateRankingQuery } from '@/lib/knowledge/search/keyword-ranking' +import { getStructuredTagFilters } from '@/lib/knowledge/search/tag-filters' +import { embeddingDistance } from '@/lib/knowledge/vector-columns' +import { + documentSatisfies, + type IndexedRetrievalContext, + PERMITTED_EXACT_DOCUMENT_LIMIT, +} from '@/lib/sim-search/indexed/retrieval/permitted' +import { + excludeSearchSourcesOnRow, + knowledgeCandidateAccessConditionForConnectors, + projectionCandidateAccessCondition, + projectionDecidedOnDocument, +} from '@/lib/sim-search/indexed/retrieval/projection-access' +import { isProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' +import { resolveTinKeywordQuery } from '@/lib/sim-search/indexed/retrieval/tin-keyword' + +/** + * Chunks Tin ranks before access is checked, widening while too few are readable to fill a page. + * A caller past the permitted-set limit reads a large share of the index, so the first window + * almost always fills; the widest bounds the work before the GIN ranking takes over. + */ +const TIN_KEYWORD_WINDOWS = [2000, 10_000, 50_000] as const + +/** Readable rows one wide window returns for a narrow reader: several pages' worth, ranked once. */ +const NARROW_KEYWORD_PAGE = 1000 + +/** + * The widest window a narrow reader ranks: wide enough that a few percent of it fills their page + * several times over, and less than half the cost of the widest window the broad readers reach. + * It is tried only after the first window came back short: ranking costs grow with the window, + * and a term that is common where the reader can read fills the page from the narrowest one. + */ +const NARROW_KEYWORD_WINDOWS = [TIN_KEYWORD_WINDOWS[0], 20_000] as const + +/** + * The keyword leg of a user-scoped search-index search. + * + * A bounded permitted set confines matching to the chunks the caller may read. A caller reaching + * past the permitted-set limit reads much of the index, so ranking every match before checking + * access is the leg's whole cost for a common term: where the Tin projection is complete, BM25 + * ranks inside the bases first and access is checked, on the row, only on the top of that + * ranking. Otherwise the GIN projection (`embedding_keyword_search`) ranks in three stages: + * match, authorize, rank. + * + * The visibility predicate carries correlated subqueries — one per connector, one per + * search-integration decision — so evaluating it across a base ahead of the query costs a table + * pass priced by how many documents the base holds rather than by how many the query matched. + * Matching first restricts that predicate to the documents the query actually matched. + * + * Two details keep that ordering from paying the saving back. Restricting the predicate with + * `document.id = ANY (...)` rather than a subquery keeps the narrowed lookup on a bitmap scan, + * which prefetches, where a plain `IN (SELECT ...)` plans as an index walk that does not. And + * the match stage carries identifiers only: ranking every match rather than every *visible* + * match would detoast one text-search vector per match, which on a mid-frequency term costs + * more than the pass it replaces. + */ +export async function executeIndexedKeywordSearch( + params: KeywordSearchParams, + context: IndexedRetrievalContext +): Promise { + const { knowledgeBaseIds, topK, query, queryVector, structuredFilters } = params + if (!query.trim()) return [] + const { access, accessPlan, permitted } = context + const tsQuery = sql`websearch_to_tsquery(${FTS_CONFIG}, ${query})` + const tagFilterConditions = structuredFilters?.length + ? getStructuredTagFilters(structuredFilters, embedding) + : [] + const candidateRank = sql`ts_rank_cd(${embeddingKeywordSearch.contentTsv}, ${tsQuery})` + /** + * A bounded set past the exact-ranking size is read on the row like an unbounded one: the + * bounded read materializes every chunk of the set before it matches a term, where a ranking + * decided on the row costs what the term matches. + */ + const largePermittedSet = + permitted?.kind === 'bounded' && permitted.documents.length >= PERMITTED_EXACT_DOCUMENT_LIMIT + const onRowReader = permitted?.kind === 'unbounded' || largePermittedSet + let tinQuery: Awaited> = null + if (onRowReader && tagFilterConditions.length === 0) { + try { + tinQuery = await resolveTinKeywordQuery(query, FTS_CONFIG, params.budget) + } catch (error) { + /** A leg whose deadline passed before it ranked anything is short, not failed. */ + if (!params.budget?.isTimeout(error)) throw error + return [] + } + } + if (onRowReader) annotateSearchDiagnostics({ keywordRanking: tinQuery ? 'tin' : 'gin' }) + /** A filled projection decides readability on the ranked row alone; none of its rows needs the document. */ + const tinFilled = tinQuery + ? await isProjectionFilled('embedding_keyword_tin', 'keyword.projection_filled', params.budget) + : false + /** The ranked CTE's mirrored columns, which the on-row predicates read. */ + const rankedTinRow = { + connectorId: sql`ranked_tin_chunks.connector_id`, + acl: sql`ranked_tin_chunks.acl`, + documentId: sql`ranked_tin_chunks.document_id`, + } + /** The projection predicate over the ranked CTE's mirrored columns, plus any excluded source. */ + const onRowKeywordVisibility = (excludedSources: readonly string[]) => + and( + projectionCandidateAccessCondition(rankedTinRow, access, accessPlan, { + filled: tinFilled, + }), + documentSatisfies( + sql`ranked_tin_chunks.document_id`, + searchDateFilterCondition(params.filters) + ), + excludeSearchSourcesOnRow(rankedTinRow, tinFilled, excludedSources) + ) + const documentConditions = (excludedSources: readonly string[]) => + and( + ...candidateDocumentConditions( + knowledgeBaseIds, + params.filters, + knowledgeCandidateAccessConditionForConnectors(access, accessPlan) + ), + excludeSearchSources(excludedSources) + ) + /** + * A page read on the row takes each candidate's source from the row, which is what decides + * whether its live source proof is asked for. A row decided on its document — not yet filled, + * or its document marked for the projector — takes it from the document: one primary-key read + * per such row of the page, after its limit, never per ranked row. + */ + const onRowPage = (ranked: SQL) => sql` + SELECT paged.id, paged."documentId", + CASE WHEN paged.decided_on_document + THEN (SELECT ${document.connectorId} FROM ${document} WHERE ${document.id} = paged."documentId") + ELSE paged."connectorId" + END AS "connectorId", + paged.keyword_rank + FROM (${ranked}) AS paged` + /** + * One page from the top of Tin's ranking. Readability is decided on the ranked row. The windows + * widen while the page is short, a narrow reader's to a wide one sooner and no further, and what + * the widest cannot fill is left short rather than handed to a ranking over every match. A large + * bounded set is the exception on its first page: its bounded read was exhaustive, so the widest + * window that still falls short hands that page to the GIN ranking, which covers every match. A + * later page stays with Tin: the two rankers order differently, so an offset advanced through + * one cannot resume the other. + */ + const selectTinPage = async ( + scopedQuery: SQL, + limit: number, + offset: number, + excludedSources: readonly string[] + ): Promise => { + const narrow = (permitted?.kind === 'unbounded' && !permitted.broad) || largePermittedSet + const windows: readonly number[] = narrow ? NARROW_KEYWORD_WINDOWS : TIN_KEYWORD_WINDOWS + /** + * A narrow reader's page is the readable remainder of a wide ranking, and that ranking is + * the cost: each page would rank the window again to find the next few readable rows, so one + * statement returns as many as several pages could ask for. + */ + const pageLimit = narrow ? Math.max(limit, NARROW_KEYWORD_PAGE) : limit + for (const window of windows) { + if (window < offset + limit) continue + const [page] = await runSearchQuery(params.budget, 'keyword.tin', (executor) => + executor.execute<{ ranked: number; candidates: SearchReadCandidate[] }>(sql` + WITH ranked_tin_chunks AS MATERIALIZED ( + SELECT ${embeddingKeywordTin.id} AS id, ${embeddingKeywordTin.documentId} AS document_id, + ${embeddingKeywordTin.enabled} AS enabled, ${embeddingKeywordTin.connectorId} AS connector_id, + ${embeddingKeywordTin.acl} AS acl, + tin.full_score(${embeddingKeywordTin}.ctid) AS keyword_rank + FROM ${embeddingKeywordTin} + WHERE ${embeddingKeywordTin.content} ==> (${scopedQuery}) + ORDER BY keyword_rank DESC + LIMIT ${window} + ), page AS ( + ${ + /** + * Readability decided on the ranked row: its source and ACL are mirrored there, so a + * window of mostly unreadable chunks costs an array test per row, not a document + * lookup. The full predicate follows at hydration. + */ + onRowPage( + sql` + SELECT ranked_tin_chunks.id, ranked_tin_chunks.document_id AS "documentId", + ranked_tin_chunks.connector_id AS "connectorId", ranked_tin_chunks.keyword_rank, + ${projectionDecidedOnDocument(rankedTinRow, tinFilled)} AS decided_on_document + FROM ranked_tin_chunks /* on-row visibility */ + WHERE ranked_tin_chunks.enabled AND ${onRowKeywordVisibility(excludedSources)} + ORDER BY ranked_tin_chunks.keyword_rank DESC, ranked_tin_chunks.id + LIMIT ${pageLimit} OFFSET ${offset}` + ) + } + ) + SELECT (SELECT count(*)::int FROM ranked_tin_chunks) AS ranked, + coalesce(( + SELECT json_agg(json_build_object( + 'id', page.id, 'documentId', page."documentId", 'connectorId', page."connectorId" + ) ORDER BY page.keyword_rank DESC, page.id) + FROM page + ), '[]'::json) AS candidates + `) + ) + annotateSearchDiagnostics({ keywordTinWindow: window }) + if ( + page.candidates.length >= limit || + page.ranked < window || + (!(largePermittedSet && offset === 0) && window === windows[windows.length - 1]) + ) { + return { candidates: page.candidates, nextOffset: offset + page.candidates.length } + } + } + return null + } + /** Parenthesized where used: `==>` binds tighter than `||`. */ + const tinScope = tinQuery + ? sql`'(' || ${sql.join( + knowledgeBaseIds.map((id) => sql`knowledge_tin_base_token(${id}) || '^0'`), + sql` || ' OR ' || ` + )} || ') AND (' || ${tinQuery} || ')'` + : undefined + /** + * Tin and GIN order candidates differently, so a search that once handed a page to GIN stays + * with GIN: an offset advanced through one ranking cannot resume the other. + */ + let handedToGin = false + /** Keep readable identities and rank scalars separate so sorts never carry full text-search vectors. */ + return selectAuthorizedSearchResults({ + leg: 'keyword', + access, + liveSourceAccess: context.liveSourceAccess, + signal: params.signal, + budget: params.budget, + topK, + selectPage: async (limit, offset, excludedSources) => { + /** + * A bounded permitted set confines matching to the chunks the caller may read, so a term + * common across the index is ranked only where it can surface. The visibility CTE below + * still re-applies the candidate predicate, so the restriction can only narrow. + */ + const permittedIds = + permitted?.kind === 'bounded' && !largePermittedSet + ? permitted.documents.map((entry) => entry.id) + : undefined + if (permittedIds?.length === 0) return { candidates: [], nextOffset: offset } + if (tinScope && !handedToGin) { + const tinPage = await selectTinPage(tinScope, limit, offset, excludedSources) + if (tinPage) return tinPage + handedToGin = true + annotateSearchDiagnostics({ keywordRanking: 'gin' }) + } + const baseScope = and( + inArray(embeddingKeywordSearch.knowledgeBaseId, knowledgeBaseIds), + eq(embeddingKeywordSearch.enabled, true) + ) + const chunkMatch = and( + sql`${embeddingKeywordSearch.contentTsv} @@ ${tsQuery}`, + tagFilterConditions.length + ? sql`EXISTS ( + SELECT 1 FROM ${embedding} WHERE ${embedding.id} = ${embeddingKeywordSearch.id} + AND ${and(...tagFilterConditions)} + )` + : undefined + ) + /** + * A bounded permitted set is read through its documents alone and matched row by row, at a + * cost linear in the permitted chunks. Offered the text or base indexes alongside, + * PostgreSQL may intersect the permitted chunks with every chunk in the base that holds the + * term or sits in the base; measured on an organization index that plan cost several + * times the direct read, and the direct read is never materially slower. The permitted + * documents were resolved inside these bases; the base check still applies to the rows + * read, so the read can never widen the scope. `OFFSET 0` keeps the read from being + * flattened back into an intersection; the alias lets the shared conditions bind to it. + */ + const matchedChunks = permittedIds + ? sql` + SELECT ${embeddingKeywordSearch.id} AS id, ${embeddingKeywordSearch.documentId} AS document_id + FROM ( + SELECT * FROM ${embeddingKeywordSearch} + WHERE ${embeddingKeywordSearch.documentId} = ANY(${textArrayLiteral(permittedIds)}) + OFFSET 0 + ) AS ${embeddingKeywordSearch} + WHERE ${and(baseScope, chunkMatch)}` + : sql` + SELECT ${embeddingKeywordSearch.id} AS id, ${embeddingKeywordSearch.documentId} AS document_id + FROM ${embeddingKeywordSearch} + WHERE ${and(baseScope, chunkMatch)}` + const candidates = await runSearchQuery(params.budget, 'keyword.sql', (executor) => + executor.execute( + keywordCandidateRankingQuery({ + matchedChunks, + documentConditions: [documentConditions(excludedSources)], + rankTable: embeddingKeywordSearch, + rank: candidateRank, + limit, + offset, + }) + ) + ) + return { candidates, nextOffset: offset + candidates.length } + }, + /** Every candidate already matched the query where it was ranked; matching it again here would detoast one text-search vector per result. */ + hydrate: (ids, authorized) => + hydrateSearchCandidates( + ids, + knowledgeAccessCondition(authorized), + embeddingDistance(queryVector.dimensions, queryVector.vector).as('distance'), + params.filters, + [inArray(embedding.knowledgeBaseId, knowledgeBaseIds), ...tagFilterConditions], + 'keyword', + params.budget + ), + }) +} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts new file mode 100644 index 00000000000..ed313c45f8c --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts @@ -0,0 +1,995 @@ +import { + dbChainMockFns, + hasMockCondition, + queueTableRows, + resetDbChainMock, + resetEnvFlagsMock, + schemaMock, + setEnvFlags, +} from '@sim/testing' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { mockResolveTinKeywordQuery } = vi.hoisted(() => ({ + mockResolveTinKeywordQuery: vi.fn<() => Promise>(async () => null), +})) + +vi.mock('@/lib/sim-search/indexed/retrieval/tin-keyword', () => ({ + resolveTinKeywordQuery: mockResolveTinKeywordQuery, +})) + +import type { KnowledgeAccessProvider, UserAccessScope } from '@/lib/knowledge/access/types' +import { SearchBudget } from '@/lib/knowledge/search/budget' +import type { KeywordSearchParams, SearchParams } from '@/lib/knowledge/search/candidates' +import { retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' +import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' +import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' +import { selectIndexedTagResults } from '@/lib/sim-search/indexed/retrieval/legs' +import { + forgetSearchReach, + type IndexedRetrievalContext, + isSearchFiltered, + PERMITTED_EXACT_DOCUMENT_LIMIT, + type PermittedDocuments, + resolvePermittedDocuments, + resolveReach, +} from '@/lib/sim-search/indexed/retrieval/permitted' +import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' +import { forgetIndexedVectorSources } from '@/lib/sim-search/indexed/retrieval/source-vector-indexes' +import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' + +/** + * The projection-fill memo outlives a test; every case starts without one. Shared retrieval runs + * these legs only while indexed organization search is on. + */ +beforeEach(() => { + forgetProjectionFilled() + setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) +}) +afterEach(resetEnvFlagsMock) + +/** A plan that admits every source and resolves no membership: what a test leaves unsaid. */ +const openPlan = (): SearchAccessPlan => ({ + connectors: { workspace: [], admin: [], members: [], liveProofRequired: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, +}) + +type Resolved = Partial> & { + accessPlan?: Omit & { + connectors: Omit & { + liveProofRequired?: readonly string[] + } + } +} + +/** The caller's resolved state, split from the leg's own parameters. */ +function context( + { accessPlan, permitted, liveSourceAccess }: Resolved, + params: SearchParams +): IndexedRetrievalContext { + return { + access: params.access as UserAccessScope, + filtered: isSearchFiltered(params.filters), + accessPlan: accessPlan + ? { + ...accessPlan, + connectors: { liveProofRequired: [], ...accessPlan.connectors }, + } + : openPlan(), + permitted, + liveSourceAccess, + } +} + +const vectorSearch = ({ + accessPlan, + permitted, + liveSourceAccess, + ...params +}: SearchParams & Resolved) => + selectIndexedVectorResults(params, context({ accessPlan, permitted, liveSourceAccess }, params)) + +const tagSearch = ({ + accessPlan, + permitted, + liveSourceAccess, + ...params +}: SearchParams & Resolved) => + selectIndexedTagResults(params, context({ accessPlan, permitted, liveSourceAccess }, params)) + +const keywordSearch = ({ + accessPlan, + permitted, + liveSourceAccess, + ...params +}: KeywordSearchParams & Resolved) => + executeIndexedKeywordSearch(params, context({ accessPlan, permitted, liveSourceAccess }, params)) + +/** + * The global `drizzle-orm` mock renders `sql` fragments to a `?`-placeholder + * string via `toSQL()`, so we can assert the exact predicate each statement builds. + */ +function render(condition: unknown) { + return (condition as { toSQL: () => { sql: string; params: unknown[] } }).toSQL() +} + +/** The permitted-documents probe: the reach count and the saturation sentinel, never a slice. */ +function isProbeStatement(sql: string) { + return sql.includes('AS saturated') && !sql.includes('readable_chunks') +} + +/** A graph walk decided on the row it visits. */ +function isWalk(sql: string) { + return sql.includes('on-row visibility') && !sql.includes('ranked_tin_chunks') +} + +/** `+ 0` is what keeps the exact ranking off the ANN index, so it also identifies the statement. */ +function isExactRanking(sql: string) { + return sql.includes(') + 0 LIMIT') +} + +/** The page read: a slice of the pool's identities, from the projection and its documents. */ +function isPageStatement(sql: string) { + return ( + sql.includes('AS "connectorId"') && sql.includes('= ANY(') && !sql.includes('ranked_tin_chunks') + ) +} + +const statements = () => dbChainMockFns.execute.mock.calls.map(([query]) => render(query)) + +const bounded = ( + ...documents: Array<{ id: string; connectorId: string | null }> +): PermittedDocuments => ({ kind: 'bounded', documents }) + +describe('search-index legs rank identifiers before verification', () => { + const identity: UserAccessScope = { + kind: 'user', + userId: 'reader', + tokens: ['org', 's:github-repositories:-:42'], + } + const candidate = (id: string, connectorId: string) => ({ + id, + documentId: `doc-${id}`, + connectorId, + distance: 0.1, + }) + const provider: KnowledgeAccessProvider = { + get: async () => identity, + getForConnectors: async () => identity, + getForDocuments: async () => identity, + liveSourceConnectorCondition: async () => null, + } + const params: SearchParams = { + knowledgeBaseIds: ['org-index'], + topK: 1, + access: identity, + queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, + distanceThreshold: 0.8, + structuredFilters: [{ tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }], + } + + const probePages: Array> = [] + const exactPages: Array> = [] + const candidatePages: Array> = [] + const rerankPages: Array>> = [] + const keywordPages: Array>> = [] + function queueRerank(rows: Array>) { + rerankPages.push(rows) + } + function queueCandidates(rows: Array<{ id: string }>, initialCount = rows.length) { + candidatePages.push(rows.map(({ id }) => ({ id, initial_count: initialCount }))) + } + + beforeEach(() => { + resetDbChainMock() + probePages.length = 0 + exactPages.length = 0 + candidatePages.length = 0 + rerankPages.length = 0 + keywordPages.length = 0 + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + /** The fixtures model the page read, which only an unfilled projection makes. */ + if (statement.includes('AS unfilled')) return [{ unfilled: true }] + if (statement.includes('AS visible')) return candidatePages.shift() ?? [] + if (isPageStatement(statement)) return rerankPages.shift() ?? [] + if (statement.includes('WITH matched_keyword_chunks')) return keywordPages.shift() ?? [] + if (isExactRanking(statement)) return exactPages.shift() ?? [] + if (isProbeStatement(statement)) return probePages.shift() ?? [] + return [] + }) + }) + + it.each(['vector', 'tag-vector', 'tags', 'keyword'] as const)( + '%s ranks identifiers before verification and loads content under the full predicate', + async (mode) => { + const candidates = [candidate('selected', 'allowed-source')] + if (mode === 'vector' || mode === 'tag-vector') { + exactPages.push([{ id: 'selected' }]) + queueRerank(candidates) + } + if (mode === 'keyword') keywordPages.push(candidates) + if (mode === 'tags') queueTableRows(schemaMock.embedding, candidates) + queueTableRows(schemaMock.embedding, [{ id: 'selected', content: 'verified result' }]) + const permitted = bounded({ id: 'doc-selected', connectorId: 'allowed-source' }) + const rows = + mode === 'vector' + ? await vectorSearch({ ...params, structuredFilters: undefined, permitted }) + : mode === 'tag-vector' + ? await vectorSearch({ ...params, permitted }) + : mode === 'tags' + ? await tagSearch(params) + : await keywordSearch({ + ...params, + query: 'release', + queryVector: params.queryVector!, + }) + expect(rows).toEqual([{ id: 'selected', content: 'verified result' }]) + if (mode === 'keyword') { + const ranking = render(dbChainMockFns.execute.mock.calls[0][0]).sql + expect(ranking).toContain('matched_keyword_chunks AS MATERIALIZED') + expect(ranking).toContain('ORDER BY keyword_rank DESC, matched_keyword_chunks.id') + expect(ranking).not.toContain('<=>') + expect(ranking).not.toContain('"content"') + } else if (mode === 'tags') { + expect(Object.keys(dbChainMockFns.select.mock.calls[0][0]).sort()).toEqual( + ['id', 'documentId', 'connectorId'].sort() + ) + } else { + const ranking = statements().find((query) => isExactRanking(query.sql))! + /** The identities are one nested fragment; the mock renders it into the parameters. */ + expect(JSON.stringify(ranking)).toContain('connectorId') + expect(ranking.sql).not.toContain('"content"') + } + const rankingOrder = + mode === 'tags' + ? dbChainMockFns.select.mock.invocationCallOrder[0] + : dbChainMockFns.execute.mock.invocationCallOrder[0] + expect(rankingOrder).toBeLessThan(dbChainMockFns.select.mock.invocationCallOrder.at(-1)!) + const fullPredicate = dbChainMockFns.where.mock.calls.at(-1)![0] + expect(JSON.stringify(fullPredicate)).toContain('acl') + expect( + hasMockCondition( + fullPredicate, + (node) => + node.type === 'inArray' && + node.column === schemaMock.embedding.id && + Array.isArray(node.values) && + node.values.length === 1 && + node.values[0] === 'selected' + ) + ).toBe(true) + } + ) + + it('matches keyword chunks before the visibility predicate and ranks only what survives it', async () => { + keywordPages.push([candidate('selected', 'allowed-source')]) + queueTableRows(schemaMock.embedding, [{ id: 'selected', content: 'verified result' }]) + await keywordSearch({ ...params, query: 'release', queryVector: params.queryVector! }) + const ranking = render(dbChainMockFns.execute.mock.calls[0][0]).sql + const matched = ranking.indexOf('matched_keyword_chunks AS MATERIALIZED') + const visible = ranking.indexOf('visible_keyword_documents AS MATERIALIZED') + expect(matched).toBeGreaterThanOrEqual(0) + expect(visible).toBeGreaterThan(matched) + expect(ranking.slice(matched, visible)).not.toContain('keyword_rank') + expect(ranking.slice(visible)).toContain('FROM matched_keyword_chunks INNER JOIN') + /** The predicate fragments are parameterized, so the restriction is read off the query tree. */ + const fragments = JSON.stringify(dbChainMockFns.execute.mock.calls[0][0]) + expect(fragments).toContain('= ANY (ARRAY(SELECT document_id FROM matched_keyword_chunks))') + }) +}) + +describe('permitted-document planner', () => { + const reader: UserAccessScope = { + kind: 'user', + userId: 'reader', + tokens: ['u:reader@example.com'], + } + const provider: KnowledgeAccessProvider = { + get: async () => reader, + getForConnectors: async () => reader, + getForDocuments: async () => reader, + liveSourceConnectorCondition: async () => null, + } + const params: SearchParams = { + knowledgeBaseIds: ['org-index'], + topK: 1, + access: reader, + queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, + distanceThreshold: 1, + } + const hit = (id: string, connectorId: string | null) => ({ + id, + documentId: `doc-${id}`, + connectorId, + distance: 0.1, + }) + let probeRows: Array<{ id: string | null; connectorId: string | null; saturated: boolean }> + let exactRows: Array<{ id: string }> + let traversedRows: Array<{ id: string; distance?: number }> + let rerankRows: Array> + let indexedSourceRows: Array<{ name: string; connectorId: string }> + let sourceExactRows: Array<{ id: string; distance: number }> + + beforeEach(() => { + resetDbChainMock() + probeRows = [] + exactRows = [] + traversedRows = [] + rerankRows = [] + sourceExactRows = [] + indexedSourceRows = [] + forgetIndexedVectorSources() + forgetSearchReach() + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + /** The fixtures model the page read, which only an unfilled projection makes. */ + if (statement.includes('AS unfilled')) return [{ unfilled: true }] + if (statement.includes('pg_index')) return indexedSourceRows + if (isWalk(statement)) return traversedRows + if (isPageStatement(statement)) return rerankRows + if (statement.includes('WITH readable_chunks')) return sourceExactRows + if (isExactRanking(statement)) return exactRows + if (isProbeStatement(statement)) return probeRows + return [] + }) + }) + + it('ranks a bounded permitted set exactly without walking the graph', async () => { + exactRows = [{ id: 'a' }] + rerankRows = [hit('a', null)] + queueTableRows(schemaMock.embedding, [hit('a', null)]) + const results = await vectorSearch({ + ...params, + permitted: bounded({ id: 'doc-a', connectorId: null }, { id: 'doc-b', connectorId: 'src' }), + }) + expect(results.map((row) => row.id)).toEqual(['a']) + const sqls = statements().map((query) => query.sql) + expect(sqls.some((sql) => sql.includes('hnsw.iterative_scan'))).toBe(false) + expect(sqls.some((sql) => sql.includes('AS visible'))).toBe(false) + expect(sqls.some(isProbeStatement)).toBe(false) + const exact = JSON.stringify(statements().find((query) => isExactRanking(query.sql))) + expect(exact).toContain('doc-a') + expect(exact).toContain('doc-b') + }) + + it('walks the whole graph once for a caller whose reach is broad', async () => { + const eligibility = { workspace: [], admin: ['other-src'], members: ['member-src'] } + indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] + /** A full pool: the walk found as many readable neighbours as it was asked for. */ + traversedRows = Array.from({ length: 400 }, (_, i) => ({ id: `walked-${i}`, distance: 0.2 })) + rerankRows = [hit('walked-0', 'member-src')] + queueTableRows(schemaMock.embedding, rerankRows) + await vectorSearch({ + ...params, + permitted: { kind: 'unbounded', broad: true }, + accessPlan: { + connectors: eligibility, + observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, + memberSources: ['member-src'], + connectorTypes: new Map(), + uploads: true, + }, + }) + /** One walk over every source, scoped to the bases alone — no source is singled out. */ + const walks = statements().filter((query) => isWalk(query.sql)) + expect(walks).toHaveLength(1) + expect(JSON.stringify(walks[0])).not.toContain('"right":"member-src"') + expect(statements().some((query) => query.sql.includes('WITH readable_chunks'))).toBe(false) + }) + + it('walks an indexed source a bounded caller is a member of instead of ranking it exactly', async () => { + const eligibility = { workspace: [], admin: ['small-src'], members: ['member-src'] } + indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] + sourceExactRows = [{ id: 'small-hit', distance: 0.3, saturated: false }] + traversedRows = [{ id: 'walked-hit', distance: 0.2 }] + rerankRows = [hit('walked-hit', 'member-src'), hit('small-hit', 'small-src')] + queueTableRows(schemaMock.embedding, rerankRows) + await vectorSearch({ + ...params, + topK: 2, + permitted: bounded( + { id: 'doc-a', connectorId: 'member-src' }, + { id: 'doc-b', connectorId: 'small-src' } + ), + accessPlan: { + connectors: eligibility, + observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, + memberSources: ['member-src'], + connectorTypes: new Map(), + uploads: true, + }, + }) + const walks = statements().filter((query) => isWalk(query.sql)) + expect(walks).toHaveLength(1) + expect(JSON.stringify(walks[0])).toContain('"right":"member-src"') + expect(statements().some((q) => isExactRanking(q.sql))).toBe(false) + }) + + it('walks the sliced sources when more documents are readable than one ranking may enumerate', async () => { + const eligibility = { workspace: [], admin: ['sliced-src'], members: [] } + /** The slice enumerates in no order, so a saturated one would rank an arbitrary subset. */ + sourceExactRows = [{ id: 'arbitrary-hit', distance: 0.4, saturated: true }] + traversedRows = [{ id: 'walked-hit', distance: 0.2 }] + rerankRows = [hit('walked-hit', 'sliced-src')] + queueTableRows(schemaMock.embedding, rerankRows) + await vectorSearch({ + ...params, + permitted: { kind: 'unbounded', broad: false }, + accessPlan: { + connectors: eligibility, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, + }, + }) + const walks = statements().filter((query) => isWalk(query.sql)) + expect(walks).toHaveLength(1) + expect(JSON.stringify(walks[0])).toContain('sliced-src') + const reranked = JSON.stringify(statements().find((query) => isPageStatement(query.sql))) + expect(reranked).toContain('walked-hit') + expect(reranked).not.toContain('arbitrary-hit') + }) + + it('ranks uploaded documents even when every connector source is walked', async () => { + const eligibility = { workspace: [], admin: [], members: ['member-src'] } + indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] + sourceExactRows = [{ id: 'upload-hit', distance: 0.05, saturated: false }] + traversedRows = [{ id: 'walked-hit', distance: 0.2 }] + rerankRows = [hit('upload-hit', null), hit('walked-hit', 'member-src')] + queueTableRows(schemaMock.embedding, rerankRows) + await vectorSearch({ + ...params, + topK: 2, + permitted: { kind: 'unbounded', broad: false }, + accessPlan: { + connectors: eligibility, + observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, + memberSources: ['member-src'], + connectorTypes: new Map(), + uploads: true, + }, + }) + /** Uploads carry no connector, so their slice runs even with no sliced source beside them. */ + const exact = statements().filter((query) => query.sql.includes('WITH readable_chunks')) + expect(exact).toHaveLength(1) + expect(JSON.stringify(statements().find((q) => isPageStatement(q.sql)))).toContain('upload-hit') + }) + + it('confines keyword matching to the bounded permitted set', async () => { + await keywordSearch({ + ...params, + topK: 1, + query: 'release', + queryVector: params.queryVector!, + permitted: bounded({ id: 'doc-a', connectorId: null }), + }) + const keyword = statements().find((query) => query.sql.includes('WITH matched_keyword_chunks'))! + /** The mock renders the whole WHERE as one parameter, so the restriction shows up in it. */ + expect(JSON.stringify(keyword)).toContain('doc-a') + }) + + describe('Tin keyword ranking for an unbounded caller', () => { + const unbounded: PermittedDocuments = { kind: 'unbounded' } + const keyword = (overrides: Partial[0]> = {}) => + keywordSearch({ + ...params, + topK: 1, + query: 'release', + queryVector: params.queryVector!, + permitted: unbounded, + ...overrides, + }) + const tinStatements = () => + statements().filter((query) => query.sql.includes('ranked_tin_chunks')) + const ginStatements = () => + statements().filter((query) => query.sql.includes('WITH matched_keyword_chunks')) + let tinPages: Array<{ ranked: number; candidates: ReturnType[] }> + + beforeEach(() => { + mockResolveTinKeywordQuery.mockReset() + mockResolveTinKeywordQuery.mockResolvedValue('"releas"') + tinPages = [] + dbChainMockFns.execute.mockImplementation(async (query) => + render(query).sql.includes('ranked_tin_chunks') + ? [tinPages.shift() ?? { ranked: 0, candidates: [] }] + : [] + ) + }) + + it('ranks with Tin and checks access only on the top of that ranking', async () => { + tinPages = [{ ranked: 1500, candidates: [hit('a', null)] }] + queueTableRows(schemaMock.embedding, [{ ...hit('a', null), content: 'release notes' }]) + const results = await keyword() + expect(results.map((row) => row.id)).toEqual(['a']) + expect(mockResolveTinKeywordQuery).toHaveBeenCalledWith('release', 'english', undefined) + expect(ginStatements()).toHaveLength(0) + expect(JSON.stringify(tinStatements()[0])).toContain('2000') + /** `==>` binds tighter than `||`, so the concatenated query must be parenthesized. */ + expect(tinStatements()[0].sql).toContain('==> (?)') + }) + + it('decides a row the fill has not reached on its document while the fill runs', async () => { + tinPages.push({ + ranked: 1, + candidates: [{ id: 'a', documentId: 'doc-a', connectorId: 'src-a' }], + }) + const execute = dbChainMockFns.execute.getMockImplementation()! + dbChainMockFns.execute.mockImplementation(async (query) => + render(query).sql.includes('AS unfilled') ? [{ unfilled: true }] : execute(query) + ) + await keyword({ + accessPlan: { + connectors: { workspace: [], admin: ['src-a'], members: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, + }, + }) + const statement = JSON.stringify(tinStatements()[0]) + /** A row the fill has not reached (`acl IS NULL`), or a marked document's row, is decided on its document. */ + expect(statement).toContain(' IS NULL OR ') + expect(statement).toContain('knowledgeProjectionDirty.documentId') + expect(statement).toContain('EXISTS (') + expect(statement).toContain('ranked_tin_chunks.document_id') + }) + + it('widens the window for a broad resolved scope whose first page came back short', async () => { + tinPages = [ + { ranked: 2000, candidates: [] }, + { ranked: 4000, candidates: [hit('b', 'src-a')] }, + ] + queueTableRows(schemaMock.embedding, [{ ...hit('b', 'src-a'), content: 'release notes' }]) + const results = await keyword({ + permitted: { kind: 'unbounded', broad: true }, + accessPlan: { + connectors: { workspace: [], admin: ['src-a'], members: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, + }, + }) + expect(results.map((row) => row.id)).toEqual(['b']) + const windows = tinStatements().map((query) => JSON.stringify(query)) + expect(windows).toHaveLength(2) + expect(windows[0]).toContain('2000') + expect(windows[1]).toContain('10000') + }) + + it('hydrates an oversized keyword page in slices and stops at the results it needs', async () => { + const ranked = Array.from({ length: 1000 }, (_, i) => hit(`k-${i}`, 'src-a')) + tinPages = [{ ranked: 20_000, candidates: ranked }] + /** The first slice — as many candidates as results are wanted — fills the page of results. */ + queueTableRows( + schemaMock.embedding, + ranked.slice(0, 20).map((row) => ({ ...row, content: 'release notes' })) + ) + const results = await keyword({ + topK: 20, + permitted: { kind: 'unbounded', broad: false }, + accessPlan: { + connectors: { workspace: [], admin: ['src-a'], members: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, + }, + }) + expect(results).toHaveLength(20) + expect(tinStatements()).toHaveLength(1) + /** One hydration, of one slice — never the whole page. */ + const hydrations = dbChainMockFns.where.mock.calls.filter(([condition]) => + hasMockCondition( + condition, + (node) => node.type === 'inArray' && node.column === schemaMock.embedding.id + ) + ) + expect(hydrations).toHaveLength(1) + expect( + hasMockCondition( + hydrations[0][0], + (node) => + node.type === 'inArray' && + node.column === schemaMock.embedding.id && + Array.isArray(node.values) && + node.values.length === 20 + ) + ).toBe(true) + }) + + describe('a bounded set past the exact-ranking size', () => { + const large = Array.from({ length: PERMITTED_EXACT_DOCUMENT_LIMIT }, (_, index) => ({ + id: `doc-${index}`, + connectorId: 'src-a', + })) + const accessPlan = { + connectors: { workspace: [], admin: ['src-a'], members: [], liveProofRequired: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, + } + + it('ranks with Tin as a narrow reader, decided on the row', async () => { + tinPages = [{ ranked: 1500, candidates: [hit('a', 'src-a')] }] + queueTableRows(schemaMock.embedding, [{ ...hit('a', 'src-a'), content: 'release notes' }]) + const results = await keyword({ + permitted: { kind: 'bounded', documents: large }, + accessPlan, + }) + expect(results.map((row) => row.id)).toEqual(['a']) + expect(mockResolveTinKeywordQuery).toHaveBeenCalledTimes(1) + expect(tinStatements()).toHaveLength(1) + expect(JSON.stringify(tinStatements()[0])).toContain('2000') + expect(JSON.stringify(tinStatements()[0])).not.toContain('doc-4999') + expect(ginStatements()).toHaveLength(0) + }) + + it('leaves a later page short rather than resuming a different ranking at its offset', async () => { + /** The first page fills from Tin; hydration keeps half, so a second page is asked for. */ + const first = Array.from({ length: 40 }, (_, index) => hit(`t-${index}`, 'src-a')) + tinPages = [ + { ranked: 2000, candidates: first }, + { ranked: 2000, candidates: [] }, + { ranked: 20_000, candidates: [] }, + ] + queueTableRows( + schemaMock.embedding, + first.slice(0, 20).map((row) => ({ ...row, content: 'release notes' })) + ) + const results = await keyword({ + topK: 40, + permitted: { kind: 'bounded', documents: large }, + accessPlan, + }) + expect(results).toHaveLength(20) + expect(tinStatements()).toHaveLength(3) + expect(ginStatements()).toHaveLength(0) + }) + }) + + it('keeps GIN ranking when Tin is not ready or cannot express the query', async () => { + mockResolveTinKeywordQuery.mockResolvedValue(null) + await keyword() + expect(tinStatements()).toHaveLength(0) + expect(ginStatements()).toHaveLength(1) + }) + }) + + it('skips keyword SQL entirely when nothing is permitted', async () => { + expect( + await keywordSearch({ + ...params, + query: 'release', + queryVector: params.queryVector!, + permitted: bounded(), + }) + ).toEqual([]) + expect(dbChainMockFns.execute).not.toHaveBeenCalled() + }) + + it('reads a user scope through its reachable documents and reports saturation', async () => { + probeRows = [{ id: 'doc-a', connectorId: null, saturated: false }] + await resolvePermittedDocuments({ + knowledgeBaseIds: ['org-index'], + access: { ...reader, tokens: ['u:reachable-documents@example.com'] }, + accessPlan: openPlan(), + filtered: false, + }) + const user = statements().find((query) => isProbeStatement(query.sql))! + const userSql = user.sql + expect(userSql).toContain('WITH reach AS MATERIALIZED') + expect(userSql).toContain('reachable AS MATERIALIZED') + expect(userSql).toContain('FROM reachable AS') + expect(userSql).toContain('AS saturated') + /** + * Baseline tokens reach every tenant's org-wide, public, and uploaded documents, so both the + * count and the rows are confined to the requested bases, outside the fence around the index. + */ + const [reach, reachable] = userSql.split('reachable AS MATERIALIZED') + for (const cte of [reach, reachable.split('FROM reachable AS')[0]]) { + expect(cte).toMatch(/OFFSET 0\s*\) AS \?\s*WHERE \?/) + } + expect(JSON.stringify(user.params)).toContain('org-index') + }) + + it.each([ + [[{ id: null, connectorId: null, saturated: true }], 'unbounded'], + [[{ id: 'doc-a', connectorId: null, saturated: false }], 'bounded'], + ] as const)('resolves %j as %s', async (rows, kind) => { + probeRows = [...rows] + const permitted = await resolvePermittedDocuments({ + knowledgeBaseIds: ['org-index'], + access: { ...reader, tokens: [`u:resolves-${kind}@example.com`] }, + accessPlan: openPlan(), + filtered: false, + }) + expect(permitted.kind).toBe(kind) + if (permitted.kind === 'bounded') + expect(permitted.documents).toEqual([{ id: 'doc-a', connectorId: null }]) + }) + + describe('saturated reach', () => { + const scope = (name: string): UserAccessScope => ({ + ...reader, + tokens: [`u:${name}@example.com`], + }) + const resolve = (access: UserAccessScope, knowledgeBaseIds = ['org-index']) => + resolvePermittedDocuments({ + knowledgeBaseIds, + access, + accessPlan: openPlan(), + filtered: false, + }) + const probes = () => statements().filter((query) => isProbeStatement(query.sql)).length + + it('counts a saturated reach against the broad bound once, and remembers the answer', async () => { + /** The index holds a million documents; the bound is a quarter of them. */ + const counts = { index: 1_000_000, reached: 250_000 } + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + if (isProbeStatement(statement)) return [{ id: null, connectorId: null, saturated: true }] + if (statement.includes(') reached')) return [{ n: counts.reached }] + if (statement.includes('EXPLAIN')) + return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': counts.index } }] }] + return [] + }) + const reachCounts = () => statements().filter((query) => query.sql.includes(') reached')) + const broad = await resolve(scope('broad-reach')) + expect(broad).toEqual({ kind: 'unbounded', broad: true }) + expect(reachCounts()).toHaveLength(1) + expect(JSON.stringify(reachCounts()[0])).toContain('250000') + await resolve(scope('broad-reach')) + expect(reachCounts()).toHaveLength(1) + counts.reached = 120_000 + const narrow = await resolve(scope('narrow-reach')) + expect(narrow).toEqual({ kind: 'unbounded', broad: false }) + expect(reachCounts()).toHaveLength(2) + }) + + it('does not remember a saturated reach whose count ran out of time', async () => { + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + if (isProbeStatement(statement)) return [{ id: null, connectorId: null, saturated: true }] + if (statement.includes('EXPLAIN')) + return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] + if (statement.includes(') reached')) + throw Object.assign(new Error('canceling statement due to statement timeout'), { + code: '57014', + }) + return [] + }) + const reachCounts = () => statements().filter((query) => query.sql.includes(') reached')) + const budget = () => new SearchBudget('vector', performance.now() + 10_000) + expect( + await resolvePermittedDocuments({ + knowledgeBaseIds: ['org-index'], + access: scope('timed-saturated'), + budget: budget(), + accessPlan: openPlan(), + filtered: false, + }) + ).toEqual({ kind: 'unbounded', broad: true }) + expect(reachCounts()).toHaveLength(1) + /** The next search probes and counts again rather than trusting a reach that was never measured. */ + await resolvePermittedDocuments({ + knowledgeBaseIds: ['org-index'], + access: scope('timed-saturated'), + budget: budget(), + accessPlan: openPlan(), + filtered: false, + }) + expect(probes()).toBe(2) + expect(reachCounts()).toHaveLength(2) + }) + + it('does not read an unanalyzed index as a reach of nothing', async () => { + /** The planner knows no rows yet, so the bound is zero and the count looked at nothing. */ + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + if (statement.includes('EXPLAIN')) return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 0 } }] }] + if (statement.includes(') reached')) return [{ n: 0 }] + return [] + }) + await expect( + resolveReach( + ['org-index'], + scope('unanalyzed'), + new SearchBudget('vector', performance.now() + 10_000), + { + connectors: { workspace: [], admin: [], members: [], liveProofRequired: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(), + uploads: true, + } + ) + ).resolves.toEqual({ kind: 'unbounded', broad: true }) + }) + + it('is remembered per set of bases and tokens', async () => { + probeRows = [{ id: null, connectorId: null, saturated: true }] + await resolve(scope('per-key')) + probeRows = [{ id: 'doc-a', connectorId: null, saturated: false }] + expect((await resolve(scope('per-key'), ['other-index'])).kind).toBe('bounded') + expect((await resolve(scope('per-key-other'))).kind).toBe('bounded') + expect(probes()).toBe(3) + }) + }) + + it('reports an exhausted vector budget as unbounded instead of failing both legs', async () => { + const budget = new SearchBudget('vector', performance.now() - 1) + const permitted = await resolvePermittedDocuments({ + knowledgeBaseIds: ['org-index'], + access: reader, + budget, + accessPlan: openPlan(), + filtered: false, + }) + expect(permitted.kind).toBe('unbounded') + expect(budget.timedOut).toBe(true) + }) + + const liveSearch = { + knowledgeBaseIds: ['org-index'], + indexedRetrieval: true, + topK: 1, + searchMode: 'hybrid' as const, + query: 'release', + queryVector: params.queryVector!, + } + + it('never asks a source for live grants when the scope reads none', async () => { + const getForConnectors = vi.fn(async () => reader) + await retrieveKnowledgeSearch({ + ...liveSearch, + access: reader, + accessProvider: { ...provider, getForConnectors }, + }) + expect(getForConnectors).not.toHaveBeenCalled() + }) + + it('rebuilds the pool without a gated source the caller turns out not to hold', async () => { + queueTableRows(schemaMock.knowledgeConnector, [ + { + id: 'gated-src', + accessMode: 'admin', + connectorType: 'confluence', + githubRepository: false, + }, + ]) + /** + * The first pool is filled by the gated source alone; only a pool built without it — the + * exclusion carries the source id into the walk — reaches the accessible candidate. The + * projection is filled, so the walk carries each candidate's source and no page is read. + */ + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + if (statement.includes('AS unfilled')) return [{ unfilled: false }] + const rebuilt = JSON.stringify(query).includes('/* excluded sources */') + if (isWalk(statement)) + return Array.from({ length: 400 }, (_, i) => + i === 0 + ? rebuilt + ? hit('b', 'other-src') + : hit('a', 'gated-src') + : hit(`w-${i}`, rebuilt ? 'other-src' : 'gated-src') + ) + return [] + }) + queueTableRows(schemaMock.embedding, []) + queueTableRows(schemaMock.embedding, [hit('b', 'other-src')]) + /** No grants come back, so the gated source is denied. */ + const getForConnectors = vi.fn(async () => reader) + const result = await retrieveKnowledgeSearch({ + ...liveSearch, + searchMode: 'vector', + access: reader, + accessProvider: { ...provider, getForConnectors }, + }) + expect(getForConnectors).toHaveBeenCalledOnce() + expect(result.rows.map((row) => row.id)).toEqual(['b']) + const walks = statements().filter((query) => isWalk(query.sql)) + expect(walks).toHaveLength(2) + expect(JSON.stringify(walks[0])).not.toContain('/* excluded sources */') + expect(JSON.stringify(walks[1])).toContain('/* excluded sources */') + expect(JSON.stringify(walks[1])).toContain('OR NOT (') + expect(statements().some((query) => isPageStatement(query.sql))).toBe(false) + }) +}) + +describe('filters on a resolved scope', () => { + const reader: UserAccessScope = { + kind: 'user', + userId: 'reader', + tokens: ['u:reader@example.com'], + } + const provider: KnowledgeAccessProvider = { + get: async () => reader, + getForConnectors: async () => reader, + getForDocuments: async () => reader, + liveSourceConnectorCondition: async () => null, + } + const params: SearchParams = { + knowledgeBaseIds: ['org-index'], + topK: 1, + access: reader, + queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, + distanceThreshold: 1, + } + const plan = (sources: string[] = ['src-a']) => ({ + connectors: { workspace: [], admin: sources, members: [], liveProofRequired: [] }, + observers: { confirmed: [], observed: [] }, + memberSources: [], + connectorTypes: new Map(sources.map((id) => [id, 'slack'])), + uploads: true, + }) + const hit = (id: string, connectorId: string | null) => ({ + id, + documentId: `doc-${id}`, + connectorId, + distance: 0.1, + }) + let probeRows: Array<{ id: string | null; connectorId: string | null; saturated: boolean }> + let traversedRows: Array<{ id: string; distance?: number }> + let rerankRows: Array> + let exactRows: Array<{ id: string }> + let indexedSourceRows: Array<{ name: string; connectorId: string }> + + beforeEach(() => { + resetDbChainMock() + forgetIndexedVectorSources() + forgetSearchReach() + probeRows = [] + traversedRows = [] + rerankRows = [] + exactRows = [] + indexedSourceRows = [] + dbChainMockFns.execute.mockImplementation(async (query) => { + const statement = render(query).sql + /** The fixtures model the page read, which only an unfilled projection makes. */ + if (statement.includes('AS unfilled')) return [{ unfilled: true }] + if (statement.includes('EXPLAIN')) + return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] + if (statement.includes('pg_index')) return indexedSourceRows + if (isExactRanking(statement)) return exactRows + if (statement.includes(') reached')) return [{ n: 250_000 }] + if (isProbeStatement(statement)) return probeRows + if (isWalk(statement)) return traversedRows + if (isPageStatement(statement)) return rerankRows + if (statement.includes('ranked_tin_chunks')) return [{ ranked: 0, candidates: [] }] + return [] + }) + }) + + it('enumerates the documents a date filter admits even when the reach is remembered', async () => { + probeRows = [{ id: 'doc-recent', connectorId: 'src-a', saturated: false }] + const budget = () => new SearchBudget('vector', performance.now() + 10_000) + await resolveReach(['org-index'], reader, budget(), plan()) + const permitted = await resolvePermittedDocuments({ + knowledgeBaseIds: ['org-index'], + access: reader, + filters: { modifiedAfter: '2026-09-13T00:00:00.000Z' }, + budget: budget(), + accessPlan: plan(), + filtered: true, + }) + expect(permitted).toEqual({ + kind: 'bounded', + documents: [{ id: 'doc-recent', connectorId: 'src-a' }], + }) + const probes = statements().filter((query) => query.sql.includes('AS saturated')) + expect(probes).toHaveLength(1) + /** Filter first, over the date index: never the reach count that reports a broad reader saturated. */ + expect(probes[0].sql).not.toContain('WITH reach') + /** An index-driven probe earns its own budget: a window at the document limit fits inside it. */ + const deadlines = statements().filter((query) => query.sql.includes('statement_timeout')) + expect(deadlines.at(-1)?.params[0]).toBe('1500') + expect(JSON.stringify(probes[0])).toContain('"type":"gte"') + }) +}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/legs.ts b/apps/sim/lib/sim-search/indexed/retrieval/legs.ts new file mode 100644 index 00000000000..2eb4a9aa2c6 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/legs.ts @@ -0,0 +1,128 @@ +import type { KnowledgeAccessProvider, UserAccessScope } from '@/lib/knowledge/access/types' +import type { SearchBudget } from '@/lib/knowledge/search/budget' +import { + liveSourceAccessForConnectors, + type RetrievalLegs, + type SearchParams, + type SearchResult, + VECTOR_PROBE_BUDGET_MS, + VECTOR_PROBE_DOCUMENT_LIMIT, +} from '@/lib/knowledge/search/candidates' +import { measureSearchStage } from '@/lib/knowledge/search/diagnostics' +import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' +import { selectAuthorizedTagResults } from '@/lib/knowledge/search/tag-filters' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' +import { + resolveSearchAccessPlan, + restrictSearchAccessPlan, +} from '@/lib/sim-search/indexed/retrieval/access-plan' +import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' +import { + estimateFilteredDocuments, + type IndexedRetrievalContext, + isSearchFiltered, + type PermittedDocuments, + resolvePermittedDocuments, + resolveReach, +} from '@/lib/sim-search/indexed/retrieval/permitted' +import { knowledgeCandidateAccessConditionForConnectors } from '@/lib/sim-search/indexed/retrieval/projection-access' +import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' + +/** + * The tag-only leg of a user-scoped search-index search: candidate identities under the caller's + * resolved plan, in id order, hydrated under the full read predicate once live source proof is + * known. + */ +export function selectIndexedTagResults( + params: SearchParams, + context: IndexedRetrievalContext +): Promise { + if (!params.structuredFilters || params.structuredFilters.length === 0) { + throw new Error('Tag filters are required for tag-only search') + } + return selectAuthorizedTagResults( + params, + knowledgeCandidateAccessConditionForConnectors(context.access, context.accessPlan), + context.liveSourceAccess + ) +} + +/** + * Resolves what a user-scoped search over search indexes needs before any leg ranks, and binds it + * to the search-index legs. Connector state is the same for every document a connector owns, so + * every leg reads it from one resolution instead of proving it per candidate. A source filter + * confines the plan rather than the rows: with only that kind of source eligible, every predicate + * the plan builds and every source the legs walk is that kind. + * + * A ranked search also resolves, once and on the vector leg's budget, what the caller may read. + * A filter that leaves few documents is enumerated and ranked exactly inside them, both legs: the + * row does not carry the document's date, and a keyword ranking of the whole base may hold few of + * a small source's matches. A filter that leaves many is ranked as the scope is — the source + * confined on the row, the date tested through the document — since a set that large holds most + * of the query's neighbours anyway. The planner's estimate decides which. Otherwise readability is + * decided on the projection row, so the caller's reach alone chooses between one walk over the + * whole graph and a search of each source. Explicit documents are already a bounded scope with + * their own exhaustive ordering. + */ +export async function prepareIndexedRetrieval(input: { + knowledgeBaseIds: string[] + access: UserAccessScope + accessProvider: KnowledgeAccessProvider + filters?: WorkspaceSearchFilters + signal?: AbortSignal + /** Whether the search ranks a query; a tag-only search needs no permitted set. */ + ranked: boolean + /** The vector leg's budget, which resolving the permitted set spends. */ + budget: SearchBudget +}): Promise { + assertIndexedOrgSearchEnabled() + const { knowledgeBaseIds, access, filters } = input + const resolvedPlan = await measureSearchStage('access_plan', () => + resolveSearchAccessPlan(knowledgeBaseIds, access) + ) + const accessPlan = filters?.source + ? restrictSearchAccessPlan(resolvedPlan, filters.source) + : resolvedPlan + const liveSourceAccess = liveSourceAccessForConnectors( + accessPlan.connectors.liveProofRequired, + input.accessProvider, + input.signal + ) + const filtered = isSearchFiltered(filters) + let permitted: PermittedDocuments | undefined + if (input.ranked && !filters?.documentIds?.length) { + /** Planning only, so a short cap of its own: running past it answers as the wide window it may be. */ + const estimateBudget = input.budget.capped(VECTOR_PROBE_BUDGET_MS) + const enumerateFiltered = + filters && filtered + ? await estimateFilteredDocuments(knowledgeBaseIds, filters, accessPlan, estimateBudget) + .then((estimate) => estimate <= VECTOR_PROBE_DOCUMENT_LIMIT) + .catch((error) => { + if (!estimateBudget.isTimeout(error)) throw error + return false + }) + : false + permitted = enumerateFiltered + ? await resolvePermittedDocuments({ + knowledgeBaseIds, + access, + filters, + budget: input.budget, + accessPlan, + filtered, + }) + : await resolveReach(knowledgeBaseIds, access, input.budget, accessPlan) + } + const context: IndexedRetrievalContext = { + access, + accessPlan, + filtered, + permitted, + liveSourceAccess, + } + return { + tags: (params) => selectIndexedTagResults(params, context), + vector: (params) => selectIndexedVectorResults(params, context), + keyword: (params) => executeIndexedKeywordSearch(params, context), + } +} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/permitted.ts b/apps/sim/lib/sim-search/indexed/retrieval/permitted.ts new file mode 100644 index 00000000000..70518416745 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/permitted.ts @@ -0,0 +1,424 @@ +import { document } from '@sim/db/schema' +import { sha256Hex } from '@sim/security/hash' +import { and, inArray, isNull, type SQL, sql } from 'drizzle-orm' +import type { AnyPgColumn } from 'drizzle-orm/pg-core' +import { LRUCache } from 'lru-cache' +import { textArrayLiteral } from '@/lib/knowledge/access/predicate' +import type { UserAccessScope } from '@/lib/knowledge/access/types' +import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' +import { + candidateDocumentConditions, + directVisibleDocumentsQuery, + type LiveSourceAccess, + type PermittedDocument, + type ProbeOutcome, + probeVisibleDocuments, + VECTOR_PROBE_BUDGET_MS, + VECTOR_PROBE_DOCUMENT_LIMIT, +} from '@/lib/knowledge/search/candidates' +import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' +import { searchDateFilterCondition } from '@/lib/knowledge/search/filter-conditions' +import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' +import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' +import { + knowledgeAclOverlapCondition, + knowledgeCandidateAccessConditionForConnectors, +} from '@/lib/sim-search/indexed/retrieval/projection-access' + +/** + * What a filter-first probe may spend: it reads the filtered documents off their own index and + * tests each one's access, bounded by the same document limit, and measures around 2 µs per + * document to enumerate plus the access test — a window at the limit fits with room. Its result + * is ranked exactly, at a cost that is predictable where a walk through a mostly-excluded + * neighbourhood is not. + */ +const FILTERED_PROBE_BUDGET_MS = 1500 + +/** + * Whether a filter narrows the documents in a way the projection row cannot see — a date window, + * or a source kind, which confines the plan — so the filtered set is worth estimating and, when + * small, enumerating directly. + */ +export function isSearchFiltered(filters: WorkspaceSearchFilters | undefined): boolean { + return Boolean(searchDateFilterCondition(filters) || filters?.source) +} + +/** A row whose document satisfies `condition`; nothing when there is nothing to ask the document. */ +export function documentSatisfies( + documentId: AnyPgColumn | SQL, + condition: SQL | undefined +): SQL | undefined { + if (condition === undefined) return undefined + return sql`EXISTS (SELECT 1 FROM ${document} WHERE ${and(sql`${document.id} = ${documentId}`, condition)})` +} + +/** + * Documents a bounded permitted set may hold before ranking it exactly costs more than walking + * the graph on the row. Exact ranking reads every chunk of the set, a few per document, where an + * on-row walk reads at most a capped number of tuples; at this size the two meet. A set past it + * is walked first and ranked exactly only if the walk cannot fill its pool, so its recall is never + * below the exact ranking's and its usual cost is the walk's. The same size turns the keyword leg + * from a read of the set's every chunk into a ranking decided on the row. + */ +export const PERMITTED_EXACT_DOCUMENT_LIMIT = 5_000 + +/** + * The documents a user-scoped search-index search may rank, resolved once before either leg runs. + * + * Organization search indexes grant most documents to a single mailbox, channel, or file owner, + * so a member typically reads a vanishing share of the index. Ranking the whole index and + * checking access afterwards then scans thousands of candidates to find none; ranking inside the + * permitted set finds every eligible chunk at a cost proportional to what the member can read. + * `unbounded` means the set exceeded the probe's limit, where post-filtered index search fills + * quickly because most candidates are readable. + */ +export type PermittedDocuments = + | { kind: 'bounded'; documents: readonly PermittedDocument[] } + | { kind: 'unbounded'; broad: boolean } + +/** + * The share of the index a caller must reach before the whole graph is walked for them. pgvector + * post-filters, so a walk returns a caller's own neighbours in proportion to their reach: above + * this share almost every neighbour the graph visits is theirs and one walk is the cheapest exact + * answer there is; below it the walk spends its budget on chunks they cannot read, and each + * readable source is searched on its own instead. + */ +const BROAD_REACH_SHARE = 0.25 + +/** + * How long a caller's saturated reach is remembered. Reach counts the documents a caller's tokens + * touch in the bases, which moves slowly, and an unbounded set only means the legs search the + * index with the full access predicate, so a stale answer costs speed, never access. + */ +const SATURATED_REACH_TTL_MS = 5 * 60 * 1000 + +/** + * A counted reach: whether it is broad enough to walk the whole graph for, or empty, in which + * case the caller reads nothing in these bases and no leg has anything to rank. + */ +interface CountedReach { + broad: boolean + empty: boolean +} + +/** + * Only breadth is remembered. Emptiness decides completeness, not strategy, so it is counted on + * every search: the count of a reach of nothing finds nothing and costs almost nothing. + */ +const saturatedReach = new LRUCache({ + max: 10_000, + ttl: SATURATED_REACH_TTL_MS, +}) + +/** How many documents the bases hold: the denominator of a reach share, and it moves slowly. */ +const indexDocumentCounts = new LRUCache({ + max: 1000, + ttl: SATURATED_REACH_TTL_MS, + /** + * The planner's estimate of the bases' documents, from the statistics it already keeps: a share + * threshold needs the order of magnitude, and counting every row to learn it costs more than the + * search it serves. The read that misses is the search's own, under its deadline. + */ + fetchMethod: async (key, _stale, { context: budget }) => { + const [row] = await runSearchQuery(budget, 'permitted_documents', (executor) => + executor.execute<{ 'QUERY PLAN': Array<{ Plan: { 'Plan Rows': number } }> }>(sql` + EXPLAIN (FORMAT JSON) SELECT 1 FROM ${document} + WHERE ${document.knowledgeBaseId} = ANY(${textArrayLiteral(key.split(','))}) + AND ${document.deletedAt} IS NULL`) + ) + /** An empty answer is not remembered; the bases may simply not have been analyzed yet. */ + return Number(row?.['QUERY PLAN']?.[0]?.Plan?.['Plan Rows'] ?? 0) || undefined + }, +}) + +/** + * The permitted-set probe's SQL, returning at most one row past the document limit. + * + * A user scope first materializes the documents its tokens reach in these bases, read through + * `doc_acl_gin_idx` alone, then applies the state and full access conditions to those rows in + * memory; the set is aliased as `document` so the shared conditions bind to it unchanged. Handed + * the combined predicate instead, PostgreSQL misjudges the token overlap as unselective and + * intersects it with base-wide indexes that read the whole search index. + * + * The index is global and every caller holds the baseline tokens every tenant's org-wide, public, + * and uploaded documents carry, so the reach must be counted inside these bases or those + * documents alone would saturate it. The base check is applied outside an `OFFSET 0` fence so it + * filters the index's rows instead of replacing the index with a base-wide scan. The reach is + * counted before any row is materialized, so a caller whose tokens reach past the limit pays only + * for the count, and a `saturated` sentinel row then reports the set as unbounded. Resolved scopes + * hold base-wide tokens, so they filter directly. + */ +function visibleDocumentsQuery( + knowledgeBaseIds: string[], + conditions: (SQL | undefined)[], + access: UserAccessScope, + shape: 'reach-first' | 'direct' = 'reach-first' +): SQL { + const limit = VECTOR_PROBE_DOCUMENT_LIMIT + 1 + /** + * `direct` applies the conditions as they are: a date filter is selective on its own and has + * its own index, so counting the reach first would only report a broad caller as saturated + * before the filter was consulted. + */ + if (shape === 'direct') return directVisibleDocumentsQuery(conditions) + /** Exactly `doc_acl_gin_idx`'s predicate, so both the count and the rows read that index alone. */ + const reached = sql`${document.deletedAt} IS NULL AND ${knowledgeAclOverlapCondition(access)}` + const underLimit = sql`(SELECT n FROM reach) < ${limit}` + const inBases = inArray(document.knowledgeBaseId, knowledgeBaseIds) + return sql` + WITH reach AS MATERIALIZED ( + SELECT count(*) AS n FROM ( + SELECT 1 FROM ( + SELECT ${document.knowledgeBaseId} FROM ${document} WHERE ${reached} OFFSET 0 + ) AS ${document} + WHERE ${inBases} + LIMIT ${limit} + ) AS reached + ), reachable AS MATERIALIZED ( + SELECT * FROM ( + SELECT * FROM ${document} WHERE ${underLimit} AND ${reached} OFFSET 0 + ) AS ${document} + WHERE ${inBases} + ) + ( + SELECT ${document.id} AS id, ${document.connectorId} AS "connectorId", false AS saturated + FROM reachable AS ${document} + WHERE ${underLimit} AND ${and(...conditions)} + LIMIT ${limit} + ) + UNION ALL + SELECT NULL, NULL, true WHERE (SELECT n FROM reach) >= ${limit} + ` +} + +/** + * The planner's estimate of the documents a filter leaves in the bases — a date filter from the + * statistics on its index, a source filter from its connectors' — so whether the filtered set is + * worth enumerating is decided from its order of magnitude, without reading a row. + */ +export async function estimateFilteredDocuments( + knowledgeBaseIds: string[], + filters: WorkspaceSearchFilters, + plan: SearchAccessPlan, + budget: SearchBudget | undefined +): Promise { + const [row] = await runSearchQuery(budget, 'permitted_documents', (executor) => + executor.execute<{ 'QUERY PLAN': Array<{ Plan: { 'Plan Rows': number } }> }>(sql` + EXPLAIN (FORMAT JSON) SELECT 1 FROM ${document} + WHERE ${and( + inArray(document.knowledgeBaseId, knowledgeBaseIds), + isNull(document.deletedAt), + searchDateFilterCondition(filters), + filters.source ? planSourceCondition(plan) : undefined + )}`) + ) + return Number(row?.['QUERY PLAN']?.[0]?.Plan?.['Plan Rows'] ?? 0) +} + +/** + * How far a caller reaches: broad when they reach at least {@link BROAD_REACH_SHARE} of the + * bases' documents, empty when they reach none. A reach of nothing is a bounded set of nothing: a + * caller who reads no document in these bases, such as a member with no source of their own yet, + * has nothing for any leg to rank, where an unbounded set would have each leg scan to its + * deadline for rows it cannot find. Breadth is counted once against the bound and remembered, so + * the first search after the window pays for it and the rest do not. A caller whose probe already + * saturated is known to reach past the probe's limit, so a bound inside that limit is met without + * counting. + * + * The count reads as many index entries as the caller reaches, so on a large index it can cost + * more than the leg it serves; it gets the probe's share of the deadline, never the whole leg's. + * A count that runs out of that share answers `null`: the leg keeps its time and its deadline + * intact, and the caller decides this search alone without remembering anything. + */ +async function countReach( + knowledgeBaseIds: string[], + access: UserAccessScope, + budget: SearchBudget | undefined, + plan: SearchAccessPlan, + saturated: boolean +): Promise { + const countBudget = budget?.capped(VECTOR_PROBE_BUDGET_MS) + try { + const total = + (await indexDocumentCounts.fetch([...knowledgeBaseIds].sort().join(','), { + context: countBudget, + })) ?? 0 + const bound = Math.ceil(total * BROAD_REACH_SHARE) + if (saturated && bound <= VECTOR_PROBE_DOCUMENT_LIMIT) return { broad: true, empty: false } + const [row] = await runSearchQuery(countBudget, 'permitted_documents', (executor) => + executor.execute<{ n: number }>(sql` + SELECT count(*) AS n FROM ( + SELECT 1 FROM ${document} + WHERE ${and( + isNull(document.deletedAt), + knowledgeAclOverlapCondition(access), + inArray(document.knowledgeBaseId, knowledgeBaseIds), + planSourceCondition(plan) + )} + LIMIT ${bound} + ) reached`) + ) + const reached = Number(row?.n ?? 0) + /** A count that looked and found nothing: only a bound of zero looks at nothing. */ + return { broad: reached >= bound, empty: bound > 0 && reached === 0 } + } catch (error) { + if (!budget || !countBudget?.isTimeout(error)) throw error + /** Only the count's share was spent; the leg's own deadline still governs. */ + budget.remaining() + return null + } +} + +/** + * Reach depends on the bases, the caller's tokens and, when the plan is confined to one kind of + * source, which sources those are; a date filter narrows the set, not the reach. + */ +function reachKey( + knowledgeBaseIds: readonly string[], + access: UserAccessScope, + plan: SearchAccessPlan +): string { + const sources = `:${sha256Hex([...planSources(plan)].sort().join('\n'))}:${plan.uploads}` + return `${[...knowledgeBaseIds].sort().join(',')}:${sha256Hex([...access.tokens].sort().join('\n'))}${sources}` +} + +/** Every connector the plan admits, whatever its access mode. */ +function planSources(plan: SearchAccessPlan): readonly string[] { + return [...plan.connectors.workspace, ...plan.connectors.admin, ...plan.connectors.members] +} + +/** The documents a plan's sources own, on the document row; every source when unconfined. */ +function planSourceCondition(plan: SearchAccessPlan): SQL { + const owned = planSources(plan) + const inSources = owned.length + ? sql`${document.connectorId} = ANY(${textArrayLiteral([...owned])})` + : sql`false` + return plan.uploads ? sql`(${document.connectorId} IS NULL OR ${inSources})` : inSources +} + +/** Forgets every remembered reach, after the bases' documents or a caller's tokens changed. */ +export function forgetSearchReach(): void { + saturatedReach.clear() + indexDocumentCounts.clear() +} + +/** A resolved scope's reach, remembered per bases and tokens, with no document enumerated. */ +export async function resolveReach( + knowledgeBaseIds: string[], + access: UserAccessScope, + budget: SearchBudget | undefined, + plan: SearchAccessPlan +): Promise { + const key = reachKey(knowledgeBaseIds, access, plan) + const remembered = saturatedReach.get(key) + if (remembered) return { kind: 'unbounded', broad: remembered.broad } + try { + const reach = await countReach(knowledgeBaseIds, access, budget, plan, false) + /** A count that ran out of time decides this search only; the next one counts again. */ + if (reach === null) return { kind: 'unbounded', broad: true } + if (reach.empty) return { kind: 'bounded', documents: [] } + saturatedReach.set(key, { broad: reach.broad }) + return { kind: 'unbounded', broad: reach.broad } + } catch (error) { + /** The leg's own deadline passed during the count: the leg is short, the search is not failed. */ + if (!budget?.isTimeout(error)) throw error + return { kind: 'unbounded', broad: true } + } +} + +/** + * Resolve the permitted set with the candidate predicate both legs apply, so restricting a leg + * to it never admits a document the leg would otherwise refuse. Tag filters stay chunk-level in + * each leg; the set is the document-level superset they narrow. + * + * It runs ahead of both legs on the vector leg's budget, so exhausting that budget here reports + * `unbounded` and marks the vector leg timed out rather than failing the keyword leg with it. + */ +export async function resolvePermittedDocuments(params: { + knowledgeBaseIds: string[] + access: UserAccessScope + filters?: WorkspaceSearchFilters + budget?: SearchBudget + accessPlan: SearchAccessPlan + /** Whether a date or source filter narrows the set, which then is enumerated directly. */ + filtered: boolean +}): Promise { + const key = reachKey(params.knowledgeBaseIds, params.access, params.accessPlan) + let probe: ProbeOutcome + let broad = true + /** + * A remembered reach says how much of the bases the caller reads, which a date filter does not + * change; the filtered set still has to be enumerated, so under one the probe always runs. A plan + * under a date or source filter enumerates the filtered set directly; reach cannot stand in for it. + */ + const remembered = params.filtered ? undefined : saturatedReach.get(key) + if (remembered) { + probe = { kind: 'saturated' } + broad = remembered.broad + } else { + try { + probe = await probeVisibleDocuments( + visibleDocumentsQuery( + params.knowledgeBaseIds, + candidateDocumentConditions( + params.knowledgeBaseIds, + params.filters, + knowledgeCandidateAccessConditionForConnectors(params.access, params.accessPlan) + ), + params.access, + params.filtered ? 'direct' : 'reach-first' + ), + params.budget, + 'permitted_documents', + params.filtered ? FILTERED_PROBE_BUDGET_MS : VECTOR_PROBE_BUDGET_MS + ) + } catch (error) { + if (!params.budget?.isTimeout(error)) throw error + probe = { kind: 'timed_out' } + } + if (probe.kind === 'saturated') { + try { + const reach = await countReach( + params.knowledgeBaseIds, + params.access, + params.budget, + params.accessPlan, + true + ) + /** A count that ran out of time decides this search only; the next one counts again. */ + if (reach?.empty) probe = { kind: 'documents', documents: [] } + else if (reach !== null) { + broad = reach.broad + saturatedReach.set(key, { broad }) + } + } catch (error) { + /** The leg's own deadline passed during the count: the leg is short, the search is not failed. */ + if (!params.budget?.isTimeout(error)) throw error + } + } + } + const permitted: PermittedDocuments = + probe.kind === 'documents' + ? { kind: 'bounded', documents: probe.documents } + : { kind: 'unbounded', broad } + annotateSearchDiagnostics({ + permittedDocuments: permitted.kind, + ...(probe.kind === 'documents' ? { permittedDocumentCount: probe.documents.length } : {}), + }) + return permitted +} + +/** + * What a user-scoped search-index search resolves about the caller once, before any leg ranks: + * the connectors it may read and the caller's standing in them, the permitted set or reach, and + * the proof the gated sources still need. + */ +export interface IndexedRetrievalContext { + access: UserAccessScope + accessPlan: SearchAccessPlan + /** Whether a date or source filter narrows the documents; see {@link isSearchFiltered}. */ + filtered: boolean + /** Absent for a tag-only search and for explicit documents, which are already a bounded scope. */ + permitted?: PermittedDocuments + liveSourceAccess?: LiveSourceAccess +} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts b/apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts new file mode 100644 index 00000000000..284b08e3f61 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts @@ -0,0 +1,234 @@ +import { + document, + knowledgeConnector, + knowledgeDocumentObservation, + knowledgeProjectionDirty, +} from '@sim/db/schema' +import { type SQL, sql } from 'drizzle-orm' +import type { AnyPgColumn } from 'drizzle-orm/pg-core' +import { + aclOverlap, + aclRequirementsSatisfied, + documentHasMirroredAcl, + documentHasWorkspaceAcl, + sourceAclFreshnessCutoff, + textArrayLiteral, +} from '@/lib/knowledge/access/predicate' +import type { UserAccessScope } from '@/lib/knowledge/access/types' +import type { + KnowledgeMemberObserver, + KnowledgeMemberObservers, + SearchAccessPlan, +} from '@/lib/sim-search/indexed/retrieval/access-plan' + +/** + * The candidate predicate with connector state resolved ahead of the query instead of per row. + * + * Deletion, archival, a pending access rewrite, the organization's integration approval and the + * access mode are facts about a connector, not a document, so checking them once per query leaves + * each candidate an id comparison plus its own columns. + * + * `liveSourceAccess` is the caller's live source proof, and defaults to admitting everything: + * candidate ranking defers that proof until after ranking, exactly as + * `knowledgeMetadataCandidateAccessCondition` does, and only a reader that already holds the + * grants — content hydration — passes it. The connectors it would gate are listed separately so + * that clause is applied to those alone. + * + * Either way it narrows exactly as the predicate it stands in for: the eligible ids are the + * connectors that predicate's `EXISTS` would admit, and every document-level clause is carried + * over unchanged. + */ +export function knowledgeCandidateAccessConditionForConnectors( + scope: UserAccessScope, + plan: SearchAccessPlan, + liveSourceAccess: SQL = sql`true` +): SQL { + const eligibility = plan.connectors + if (scope.tokens.length === 0) return sql`false` + const tokens = textArrayLiteral(scope.tokens) + const cutoff = sourceAclFreshnessCutoff() + const liveProof = new Set(eligibility.liveProofRequired) + const inConnectors = (ids: readonly string[]): SQL => + ids.length === 0 + ? sql`false` + : sql`${document.connectorId} = ANY(${textArrayLiteral([...ids])})` + const mirrored = (ids: readonly string[], current: SQL): SQL => { + const direct = ids.filter((id) => !liveProof.has(id)) + const gated = ids.filter((id) => liveProof.has(id)) + const currentAndMirrored = sql`${documentHasMirroredAcl()} AND ${current}` + return sql`( + (${inConnectors(direct)} AND ${currentAndMirrored}) + OR (${inConnectors(gated)} AND ${currentAndMirrored} AND EXISTS ( + SELECT 1 FROM ${knowledgeConnector} + WHERE ${knowledgeConnector.id} = ${document.connectorId} + AND ${liveSourceAccess} + )) + )` + } + const workspaceOwned = plan.uploads + ? sql`(${document.connectorId} IS NULL OR ${inConnectors(eligibility.workspace)})` + : inConnectors(eligibility.workspace) + return sql`( + ${aclOverlap(tokens)} + AND ${aclRequirementsSatisfied(tokens)} + AND ( + (${workspaceOwned} AND ${documentHasWorkspaceAcl()}) + OR ${mirrored(eligibility.admin, sql`${document.aclVerifiedAt} > ${cutoff}`)} + OR ${mirrored(eligibility.members, resolvedObservationCondition(plan.observers, cutoff))} + ) + )` +} + +/** + * Whether a projection row belongs to a document marked for the knowledge projector: its source, + * ACL, or chunks changed and its rows may not show it yet. A probe of the marks' primary key: the + * planner may instead hash the whole set once per statement, which is as cheap while the marks are + * few, and an `IN` would risk re-reading them per row once they outgrow the hash. + */ +export function projectionPending(documentId: AnyPgColumn | SQL): SQL { + return sql`(EXISTS (SELECT 1 FROM ${knowledgeProjectionDirty} WHERE ${knowledgeProjectionDirty.documentId} = ${documentId}))` +} + +/** + * Whether a projection row is decided on its document rather than on its own columns: its + * document is marked for the projector, or, while the source and ACL fill runs, the row has not + * been filled. + */ +export function projectionDecidedOnDocument( + projection: { acl: AnyPgColumn | SQL; documentId: AnyPgColumn | SQL }, + filled: boolean +): SQL { + const pending = projectionPending(projection.documentId) + return filled ? pending : sql`(${projection.acl} IS NULL OR ${pending})` +} + +/** + * The candidate predicate on a ranking projection's own row, for a scope whose connectors were + * resolved: `connectorId` and `acl` are mirrored there from the document, so a walk or a keyword + * window decides readability on the row it scores instead of joining `document` per candidate. + * + * It admits a superset of the document predicate, never a subset: a mirrored ACL names the members + * who observe a document, so overlap with the caller's tokens is the per-row test without the + * observation's freshness, and requirement clauses live on the document. Both are refused there, + * under the full predicate, before content is returned — this predicate only decides what is worth + * ranking. + * + * A row whose columns may be behind its document is decided on the document instead, under + * {@link knowledgeCandidateAccessConditionForConnectors} — the join per candidate that every row + * paid before the columns existed: a row the source and ACL fill has not reached (`acl IS NULL`), + * and every row of a document marked for the knowledge projector. A revoked grant still on such a + * row never admits it, and a new grant not yet on it never hides it from a statement that reaches + * the row. A source-scoped walk or slice reaches rows by the source on the row, though, so a + * document that moved to another source joins that source's ranking once the projector has + * rewritten its rows; until then it can be missing there, never shown where it is not readable. + * The projector and the fill run in the background, so search never waits on either. + */ +export function projectionCandidateAccessCondition( + projection: { + connectorId: AnyPgColumn | SQL + acl: AnyPgColumn | SQL + documentId: AnyPgColumn | SQL + }, + scope: UserAccessScope, + plan: SearchAccessPlan, + options: { + /** + * Whether every row of the projection carries its mirrored source and ACL. While the fill + * is under way, a row it has not reached is decided on its document; once it is complete only + * a marked document's rows are. + */ + filled?: boolean + } = {} +): SQL { + if (scope.tokens.length === 0) return sql`false` + const tokens = textArrayLiteral(scope.tokens) + const inSources = (ids: readonly string[]): SQL => + ids.length === 0 + ? sql`false` + : sql`${projection.connectorId} = ANY(${textArrayLiteral([...ids])})` + const mirrored = [ + ...plan.connectors.workspace, + ...plan.connectors.admin, + ...plan.connectors.members, + ] + const owned = plan.uploads + ? sql`(${projection.connectorId} IS NULL OR ${inSources(mirrored)})` + : inSources(mirrored) + const onRow = sql`(${projection.acl} && ${tokens} AND ${owned})` + /** + * A scalar subquery rather than `EXISTS`: the planner may turn an `EXISTS` into one hash of every + * readable document, a sequential scan of `document` for a statement that only needs a few rows + * decided. A scalar subquery is only ever a primary-key probe per row that needs it. + */ + const onDocument = sql`(SELECT ${document.id} FROM ${document} + WHERE ${document.id} = ${projection.documentId} + AND ${knowledgeCandidateAccessConditionForConnectors(scope, plan)} + LIMIT 1) IS NOT NULL` + return sql`((${projectionDecidedOnDocument(projection, options.filled ?? false)} AND ${onDocument}) + OR (${onRow} AND NOT ${projectionPending(projection.documentId)}))` +} + +/** + * The same membership, resolved ahead of the query: each candidate costs one lookup on the + * observation key instead of a join to the member behind it. Equivalent by construction — the ids + * are the members that join would have matched, and each one's freshness rule is carried over. + */ +function resolvedObservationCondition(observers: KnowledgeMemberObservers, cutoff: SQL): SQL { + if (observers.confirmed.length === 0 && observers.observed.length === 0) return sql`false` + /** + * An observation vouches for a document only from a member of the document's own connector: a + * document that changed hands keeps its old observations, which must not carry it. + */ + const byMember = (members: readonly KnowledgeMemberObserver[]): SQL => + sql`(${knowledgeDocumentObservation.memberId}, ${document.connectorId}) IN (${sql.join( + members.map((member) => sql`(${member.id}, ${member.connectorId})`), + sql`, ` + )})` + const current = + observers.confirmed.length === 0 + ? sql`${byMember(observers.observed)} AND ${knowledgeDocumentObservation.lastSeenAt} > ${cutoff}` + : observers.observed.length === 0 + ? byMember(observers.confirmed) + : sql`(${byMember(observers.confirmed)} + OR (${byMember(observers.observed)} AND ${knowledgeDocumentObservation.lastSeenAt} > ${cutoff}))` + return sql`EXISTS ( + SELECT 1 FROM ${knowledgeDocumentObservation} + WHERE ${knowledgeDocumentObservation.documentId} = ${document.id} + AND ${current} + )` +} + +/** + * The token half of the stored access predicate: the documents a caller's tokens reach before + * any source, freshness, or requirement check narrows them. It is a necessary condition of + * `knowledgeAccessCondition`, never a substitute for it. + * + * Paired with `deleted_at IS NULL` it matches `doc_acl_gin_idx` exactly, so a query can enumerate + * a member's reachable documents from that index alone. PostgreSQL cannot estimate array-overlap + * selectivity, so left to itself it intersects this highly selective bitmap with base-wide ones. + */ +export function knowledgeAclOverlapCondition(scope: UserAccessScope): SQL { + if (scope.tokens.length === 0) return sql`false` + return aclOverlap(textArrayLiteral(scope.tokens)) +} + +/** + * Keeps the rows of sources the caller turned out not to hold out of a ranking decided on the + * row. A row decided on its document — not yet filled, or its document marked for the projector, + * so its own source may be stale — asks the document instead. + */ +export function excludeSearchSourcesOnRow( + projection: { + connectorId: AnyPgColumn | SQL + acl: AnyPgColumn | SQL + documentId: AnyPgColumn | SQL + }, + filled: boolean, + excludedSources: readonly string[] +): SQL | undefined { + if (!excludedSources.length) return undefined + const excluded = textArrayLiteral([...excludedSources]) + const decided = projectionDecidedOnDocument(projection, filled) + return sql`((${decided} AND NOT EXISTS (SELECT 1 FROM ${document} WHERE ${document.id} = ${projection.documentId} AND ${document.connectorId} = ANY(${excluded}))) + OR (NOT ${decided} AND (${projection.connectorId} IS NULL OR NOT (${projection.connectorId} = ANY(${excluded}))))) /* excluded sources */` +} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts new file mode 100644 index 00000000000..98f83e7d799 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts @@ -0,0 +1,78 @@ +/** + * @vitest-environment node + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { SearchBudget } from '@/lib/knowledge/search/budget' +import { + forgetProjectionFilled, + isProjectionFilled, +} from '@/lib/sim-search/indexed/retrieval/projection-fill' + +const LEG_BUDGET_MS = 8000 + +describe('projection fill probe', () => { + beforeEach(() => forgetProjectionFilled()) + + it('remembers a failed probe as unfilled rather than probing again on every search', async () => { + const query = vi + .spyOn(SearchBudget.prototype, 'query') + .mockRejectedValue(new Error('connection reset')) + const budget = new SearchBudget('keyword', performance.now() + LEG_BUDGET_MS) + await expect( + isProjectionFilled('embedding_keyword_tin', 'keyword.projection_filled', budget) + ).resolves.toBe(false) + await expect( + isProjectionFilled('embedding_keyword_tin', 'keyword.projection_filled', budget) + ).resolves.toBe(false) + expect(query).toHaveBeenCalledOnce() + }) + + it('does not hold a later search past its own share while another search probes', async () => { + vi.useFakeTimers() + try { + let answerFirst: (rows: Array<{ unfilled: boolean }>) => void = () => {} + vi.spyOn(SearchBudget.prototype, 'query').mockImplementation( + () => + new Promise((resolve) => { + answerFirst = resolve as typeof answerFirst + }) as ReturnType + ) + const first = isProjectionFilled( + 'embedding_search', + 'vector.projection_filled', + new SearchBudget('vector', performance.now() + LEG_BUDGET_MS) + ) + let secondSettled = false + const second = isProjectionFilled( + 'embedding_search', + 'vector.projection_filled', + new SearchBudget('vector', performance.now() + 20) + ).finally(() => { + secondSettled = true + }) + await vi.advanceTimersByTimeAsync(20) + expect(secondSettled).toBe(true) + await expect(second).resolves.toBe(false) + answerFirst([{ unfilled: false }]) + await expect(first).resolves.toBe(true) + } finally { + vi.useRealTimers() + } + }) + + it('does not remember a probe its own search cancelled', async () => { + const query = vi + .spyOn(SearchBudget.prototype, 'query') + .mockRejectedValue(new DOMException('aborted', 'AbortError')) + const controller = new AbortController() + controller.abort() + const budget = new SearchBudget('vector', performance.now() + LEG_BUDGET_MS, controller.signal) + await expect( + isProjectionFilled('embedding_search', 'vector.projection_filled', budget) + ).resolves.toBe(false) + await expect( + isProjectionFilled('embedding_search', 'vector.projection_filled', budget) + ).resolves.toBe(false) + expect(query).toHaveBeenCalledTimes(2) + }) +}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts b/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts new file mode 100644 index 00000000000..04b8a1592ff --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts @@ -0,0 +1,102 @@ +import { SOURCE_ACL_PROJECTIONS, type SourceAclProjection } from '@sim/db/knowledge-projection' +import { embeddingKeywordTin, embeddingSearch } from '@sim/db/schema' +import { sleep } from '@sim/utils/helpers' +import { sql } from 'drizzle-orm' +import { LRUCache } from 'lru-cache' +import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' +import type { SearchStage } from '@/lib/knowledge/search/diagnostics' + +/** How long a fully filled projection is taken on trust before its unfilled rows are looked for again. */ +const PROJECTION_FILLED_TTL_MS = 60_000 + +/** + * The most of a leg's budget the probe may spend. The probe is one index read that answers in + * milliseconds when the partial index serves it; a read slower than this is fighting a cold cache + * or a busy database, and waiting longer would spend the leg's ranking time on an optimization. + * An unanswered probe costs only the slower plan, so a small cap loses nothing. + */ +export const PROJECTION_FILLED_PROBE_BUDGET_MS = 250 + +/** + * How long an unanswered probe is remembered as unfilled. Short enough that a recovered database + * is asked again within seconds, long enough that searches arriving during an outage do not each + * spend their own probe budget rediscovering it. + */ +const PROJECTION_FILLED_UNKNOWN_TTL_MS = 5_000 + +/** + * Whether the ranking projection still holds rows the source and ACL fill has not reached. Read off the + * unfilled-rows index in milliseconds and remembered briefly: the answer only ever changes once. + * + * The read asks for the last unfilled row by id, not whether one exists: an `EXISTS` drops its + * order and limit, and while most rows are unfilled the planner expects a sequential scan to + * meet one at once, then walks the whole projection when the unfilled rows sit past the filled + * ones. Ordered by id and capped at one row, the read can only be the partial index, whose + * last entry is the row the fill reaches last. + */ +const projectionFilled = new LRUCache< + SourceAclProjection, + boolean, + { budget: SearchBudget | undefined; stage: SearchStage } +>({ + max: SOURCE_ACL_PROJECTIONS.length, + ttl: PROJECTION_FILLED_TTL_MS, + /** + * The read that misses the cache is the search's own, capped to a small share of its budget, + * and the searches that miss together share it. A read that fails or runs out of that share + * answers unfilled, the slower and safe form, and that answer is remembered briefly so the + * searches behind it do not each pay for the same failure. A read cut short by its own + * search's cancellation learned nothing about the projection and is not remembered. + */ + fetchMethod: async (projection, _stale, { context, options }) => { + const table = projection === 'embedding_search' ? embeddingSearch : embeddingKeywordTin + try { + const [row] = await runSearchQuery( + context.budget?.capped(PROJECTION_FILLED_PROBE_BUDGET_MS), + context.stage, + (executor) => + executor.execute<{ unfilled: boolean }>(sql` + SELECT ( + SELECT ${table.id} FROM ${table} WHERE ${table.acl} IS NULL + ORDER BY ${table.id} DESC LIMIT 1 + ) IS NOT NULL AS unfilled`) + ) + return !row?.unfilled + } catch { + if (context.budget?.signal?.aborted) return undefined + options.ttl = PROJECTION_FILLED_UNKNOWN_TTL_MS + return false + } + }, +}) + +/** + * Whether every row of the projection carries its mirrored source and ACL; unknown counts as not yet. + * + * Searches that miss the cache together share the first one's read, which is capped to that + * search's share. Each caller still waits no longer than its own share, or its own deadline if + * nearer, and reads an unanswered probe as unfilled: a caller that joined late, with less of its + * leg left, never waits on another search's timetable. A remembered answer is returned at once, + * so only a search that missed the memo starts a wait. + */ +export async function isProjectionFilled( + projection: SourceAclProjection, + stage: SearchStage, + budget: SearchBudget | undefined +): Promise { + const remembered = projectionFilled.get(projection) + if (remembered !== undefined) return remembered + const answer = projectionFilled.fetch(projection, { context: { budget, stage } }) + if (!budget) return (await answer) ?? false + const waitMs = Math.max( + 0, + Math.min(PROJECTION_FILLED_PROBE_BUDGET_MS, budget.deadline - performance.now()) + ) + const unanswered = sleep(waitMs).then(() => undefined) + return (await Promise.race([answer.catch(() => undefined), unanswered])) ?? false +} + +/** Forgets whether the projections were filled; the memo is per process and otherwise expires on its own. */ +export function forgetProjectionFilled(): void { + projectionFilled.clear() +} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts b/apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts new file mode 100644 index 00000000000..165757779fe --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts @@ -0,0 +1,33 @@ +import { sql } from 'drizzle-orm' +import { LRUCache } from 'lru-cache' +import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' + +/** + * The sources that have their own index, cached briefly: every unbounded ranking asks, and the + * answer changes only when a connector deletion drops one. + */ +const indexedSources = new LRUCache<'sources', ReadonlySet>({ max: 1, ttl: 60 * 1000 }) + +/** Forgets the cached answer. */ +export function forgetIndexedVectorSources(): void { + indexedSources.clear() +} + +/** The sources with a graph of their own; a search that misses the memo reads under its own deadline. */ +export async function indexedVectorSources(budget?: SearchBudget): Promise> { + const cached = indexedSources.get('sources') + if (cached) return cached + const rows = await runSearchQuery(budget, 'vector.source_indexes', (executor) => + executor.execute<{ connectorId: string | null }>(sql` + SELECT substring(pg_get_expr(i.indpred, i.indrelid) from '''([0-9a-f-]+)''') AS "connectorId" + FROM pg_index i + JOIN pg_class c ON c.oid = i.indexrelid + WHERE i.indrelid = 'embedding_search'::regclass + AND c.relname LIKE 'embedding_search_src_%' AND i.indisvalid AND i.indisready`) + ) + const sources = new Set( + rows.map((row) => row.connectorId).filter((id): id is string => id !== null) + ) + indexedSources.set('sources', sources) + return sources +} diff --git a/apps/sim/lib/knowledge/search/tin-keyword-readiness.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts similarity index 58% rename from apps/sim/lib/knowledge/search/tin-keyword-readiness.test.ts rename to apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts index 8e5c4531ec2..d6b1fbece87 100644 --- a/apps/sim/lib/knowledge/search/tin-keyword-readiness.test.ts +++ b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts @@ -1,12 +1,6 @@ import { dbChainMockFns, resetDbChainMock } from '@sim/testing' -import { featureFlagsMock, featureFlagsMockFns } from '@sim/testing/mocks/feature-flags.mock' -import { expect, it, vi } from 'vitest' - -vi.mock('@/lib/core/config/feature-flags', () => featureFlagsMock) - -import { resolveTinKeywordQuery } from '@/lib/knowledge/search/tin-keyword' - -featureFlagsMockFns.mockIsFeatureEnabled.mockResolvedValue(true) +import { expect, it } from 'vitest' +import { resolveTinKeywordQuery } from '@/lib/sim-search/indexed/retrieval/tin-keyword' /** Its own file, so the process-wide readiness cache starts empty. */ it('stays on the GIN projection while the Tin index is incomplete, and remembers that', async () => { @@ -18,9 +12,9 @@ it('stays on the GIN projection while the Tin index is incomplete, and remembers if (text.includes('websearch_to_tsquery')) return [{ rendered: "'releas'" }] return [] }) - expect(await resolveTinKeywordQuery(true, 'release', 'english', undefined)).toBeNull() + expect(await resolveTinKeywordQuery('release', 'english', undefined)).toBeNull() indexValid = true - expect(await resolveTinKeywordQuery(true, 'release', 'english', undefined)).toBeNull() + expect(await resolveTinKeywordQuery('release', 'english', undefined)).toBeNull() const readinessReads = dbChainMockFns.execute.mock.calls.filter(([query]) => JSON.stringify(query).includes('indisvalid') ) diff --git a/apps/sim/lib/knowledge/search/tin-keyword.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts similarity index 54% rename from apps/sim/lib/knowledge/search/tin-keyword.test.ts rename to apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts index d35664b52d0..4d6558042c4 100644 --- a/apps/sim/lib/knowledge/search/tin-keyword.test.ts +++ b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts @@ -1,13 +1,7 @@ import { dbChainMockFns, resetDbChainMock } from '@sim/testing' -import { featureFlagsMock, featureFlagsMockFns } from '@sim/testing/mocks/feature-flags.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -vi.mock('@/lib/core/config/feature-flags', () => featureFlagsMock) - +import { beforeEach, describe, expect, it } from 'vitest' import { SearchBudget, SearchDeadlineError } from '@/lib/knowledge/search/budget' -import { resolveTinKeywordQuery } from '@/lib/knowledge/search/tin-keyword' - -const mockIsFeatureEnabled = featureFlagsMockFns.mockIsFeatureEnabled +import { resolveTinKeywordQuery } from '@/lib/sim-search/indexed/retrieval/tin-keyword' /** Readiness is cached per process; the incomplete-index case lives in its own file, where the cache starts empty. */ describe('resolveTinKeywordQuery', () => { @@ -16,7 +10,6 @@ describe('resolveTinKeywordQuery', () => { beforeEach(() => { resetDbChainMock() - mockIsFeatureEnabled.mockResolvedValue(true) indexValid = true rendered = "'releas' & 'note'" dbChainMockFns.execute.mockImplementation(async (query) => { @@ -27,8 +20,8 @@ describe('resolveTinKeywordQuery', () => { }) }) - it('translates the analyzed query when every base is a search index', async () => { - expect(await resolveTinKeywordQuery(true, 'release notes', 'english', undefined)).toBe( + it('translates the analyzed query', async () => { + expect(await resolveTinKeywordQuery('release notes', 'english', undefined)).toBe( '("releas" AND "note")' ) }) @@ -36,13 +29,13 @@ describe('resolveTinKeywordQuery', () => { it('reads under the keyword budget, so an expired deadline ends the leg instead of querying', async () => { const expired = new SearchBudget('keyword', performance.now() - 1) await expect( - resolveTinKeywordQuery(true, 'release notes', 'english', expired) + resolveTinKeywordQuery('release notes', 'english', expired) ).rejects.toBeInstanceOf(SearchDeadlineError) expect(dbChainMockFns.execute).not.toHaveBeenCalled() }) - it('falls back to GIN instead of failing the search when readiness cannot be read', async () => { - mockIsFeatureEnabled.mockRejectedValue(new Error('config unavailable')) - expect(await resolveTinKeywordQuery(true, 'release notes', 'english', undefined)).toBeNull() + it('falls back to GIN instead of failing the search when the query cannot be analyzed', async () => { + dbChainMockFns.execute.mockRejectedValue(new Error('connection reset')) + expect(await resolveTinKeywordQuery('release notes', 'english', undefined)).toBeNull() }) }) diff --git a/apps/sim/lib/knowledge/search/tin-keyword.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts similarity index 71% rename from apps/sim/lib/knowledge/search/tin-keyword.ts rename to apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts index 534cb42f094..37360aae927 100644 --- a/apps/sim/lib/knowledge/search/tin-keyword.ts +++ b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts @@ -3,15 +3,14 @@ import { createLogger } from '@sim/logger' import { getErrorMessage } from '@sim/utils/errors' import { sql } from 'drizzle-orm' import { LRUCache } from 'lru-cache' -import { isFeatureEnabled } from '@/lib/core/config/feature-flags' import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' -import { tinQueryFromTsquery } from '@/lib/knowledge/search/tin-query' +import { tinQueryFromTsquery } from '@/lib/sim-search/indexed/retrieval/tin-query' const logger = createLogger('TinKeywordSearch') /** * How long a readiness answer holds. The index only becomes valid once the projection is fully - * backfilled, and flipping the flag off takes effect within this window. + * backfilled, and a newly valid or dropped index is noticed within this window. */ const READINESS_TTL_MS = 60 * 1000 @@ -33,11 +32,10 @@ const indexReadiness = new LRUCache<'index', boolean, SearchBudget | undefined>( }) /** - * The TINQL query that ranks `query` inside the search's bases, or null when keyword search must - * keep the GIN projection: a base is not an organization search index (only those are - * projected), the rollout flag is off, the database has no complete Tin index, or the query uses - * a shape TINQL cannot express. The text is analyzed by the same `websearch_to_tsquery` the GIN - * path uses, so both engines match the same stemmed terms. + * The TINQL query that ranks `query` inside a search index's bases, or null when keyword search + * must keep the GIN projection: the database has no complete Tin index, or the query uses a shape + * TINQL cannot express. The text is analyzed by the same `websearch_to_tsquery` the GIN path + * uses, so both engines match the same stemmed terms. * * Every read runs under the keyword leg's `budget`, so deciding the engine cannot outlast the leg's * deadline. A read that fails for another reason, including a shared cache read cut short by @@ -45,15 +43,11 @@ const indexReadiness = new LRUCache<'index', boolean, SearchBudget | undefined>( * cancellation propagates like any other keyword query's. */ export async function resolveTinKeywordQuery( - searchIndexOnly: boolean, query: string, ftsConfig: string, budget: SearchBudget | undefined ): Promise { - if (!searchIndexOnly) return null try { - /** The flag is read from memory and decides whether the index is worth asking about at all. */ - if (!(await isFeatureEnabled('knowledge-tin-keyword'))) return null if (!(await indexReadiness.fetch('index', { context: budget }))) return null const [{ rendered }] = await runSearchQuery(budget, 'keyword.tin_query', (executor) => executor.execute<{ rendered: string }>( diff --git a/apps/sim/lib/knowledge/search/tin-query.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts similarity index 91% rename from apps/sim/lib/knowledge/search/tin-query.test.ts rename to apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts index 17c0193ca95..472a52d9665 100644 --- a/apps/sim/lib/knowledge/search/tin-query.test.ts +++ b/apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { tinQueryFromTsquery } from '@/lib/knowledge/search/tin-query' +import { tinQueryFromTsquery } from '@/lib/sim-search/indexed/retrieval/tin-query' /** Inputs are `websearch_to_tsquery('english', …)::text` exactly as PostgreSQL renders them. */ describe('tinQueryFromTsquery', () => { diff --git a/apps/sim/lib/knowledge/search/tin-query.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-query.ts similarity index 100% rename from apps/sim/lib/knowledge/search/tin-query.ts rename to apps/sim/lib/sim-search/indexed/retrieval/tin-query.ts diff --git a/apps/sim/lib/sim-search/indexed/retrieval/vector.ts b/apps/sim/lib/sim-search/indexed/retrieval/vector.ts new file mode 100644 index 00000000000..9ccb72b1fd9 --- /dev/null +++ b/apps/sim/lib/sim-search/indexed/retrieval/vector.ts @@ -0,0 +1,491 @@ +import { document, embeddingSearch } from '@sim/db/schema' +import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' +import type { AnyPgColumn } from 'drizzle-orm/pg-core' +import { mapWithConcurrency } from '@/lib/core/utils/concurrency' +import { knowledgeAccessCondition, textArrayLiteral } from '@/lib/knowledge/access/predicate' +import type { UserAccessScope } from '@/lib/knowledge/access/types' +import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' +import { + excludeSearchSources, + getVisibilityConditions, + type SearchParams, + type SearchReadCandidate, + type SearchResult, + SOURCE_RANKING_CONCURRENCY, + selectAuthorizedSearchResults, +} from '@/lib/knowledge/search/candidates' +import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' +import { searchDateFilterCondition } from '@/lib/knowledge/search/filter-conditions' +import { + annotateVectorPoolPlanned, + annotateVectorPoolSelected, + CANDIDATE_HNSW_MAX_SCAN_TUPLES, + gatheredVectorCandidatePool, + hydrateVectorCandidates, + prepareVectorLeg, + rankVectorCandidatesExactly, + readVectorCandidatePool, + selectExactVectorPage, + sliceVectorCandidatePool, + type VectorCandidatePool, + withVectorScanSettings, +} from '@/lib/knowledge/search/vector-leg' +import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' +import { + documentSatisfies, + type IndexedRetrievalContext, + PERMITTED_EXACT_DOCUMENT_LIMIT, +} from '@/lib/sim-search/indexed/retrieval/permitted' +import { + excludeSearchSourcesOnRow, + knowledgeCandidateAccessConditionForConnectors, + projectionCandidateAccessCondition, + projectionPending, +} from '@/lib/sim-search/indexed/retrieval/projection-access' +import { isProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' +import { indexedVectorSources } from '@/lib/sim-search/indexed/retrieval/source-vector-indexes' + +/** + * How far a walk that decides readability on the row may go before giving up: a cap, not a target, + * since the scan stops as soon as the limit is met. The default cap was sized for a walk that looked + * a document up per visited tuple; on the row a tuple costs a fraction of that, so a caller whose + * neighbourhood is mostly unreadable can be carried past it for tens of milliseconds rather than + * left with what the neighbourhood happened to hold. + */ +const ON_ROW_WALK_SCAN_TUPLES = 100_000 + +/** + * How far a walk may go when readability is on the row: the on-row cap, unless the walk still + * has to ask the document about tuples — a tag or date filter, or rows the source and ACL fill + * has not reached yet — in which case such a tuple costs what it did before the columns were mirrored, + * and the default cap keeps a walk through a mostly-excluded neighbourhood at a short answer + * rather than a missed deadline. + */ +function onRowWalkScanTuples( + documentCondition: SQL | undefined, + projectionFilled: boolean +): number { + return documentCondition === undefined && projectionFilled + ? ON_ROW_WALK_SCAN_TUPLES + : CANDIDATE_HNSW_MAX_SCAN_TUPLES +} + +/** + * A candidate's source: the row's, unless the row's document is marked for the projector, whose + * source may have moved since the row was written — then the document's, read for that row only. + */ +function projectionCandidateSource(projection: { + connectorId: AnyPgColumn | SQL + documentId: AnyPgColumn | SQL +}): SQL { + return sql`CASE WHEN ${projectionPending(projection.documentId)} + THEN (SELECT ${document.connectorId} FROM ${document} WHERE ${document.id} = ${projection.documentId}) + ELSE ${projection.connectorId} END` +} + +/** + * The same identities read off a projection row in raw SQL: the aliases are what + * `SearchReadCandidate` deserializes, so every walk reads them from one place. + */ +const PROJECTION_CANDIDATE_COLUMNS = sql`${embeddingSearch.id} AS id, ${embeddingSearch.documentId} AS "documentId", ${projectionCandidateSource(embeddingSearch)} AS "connectorId"` + +/** + * How many chunks the sliced sources contribute to exact ranking. A caller's slice of mirrored + * sources — their mail, their files, the spaces they belong to — sits below this, and ranking that + * many exactly, on the projection's half-precision vectors, measures in tens of milliseconds. + */ +const SOURCE_EXACT_CHUNK_LIMIT = 150_000 + +/** Sources whose own index a caller's ranking walks, and whether anything is left to rank exactly. */ +interface SourceVectorPlan { + walked: readonly string[] + sliced: readonly string[] +} + +/** + * How each readable source contributes its nearest chunks. + * + * Membership decides it, not a count: a member of a source reads essentially all of it, so its own + * index is walked and the graph's neighbours are chunks they can read. Every other source is + * sliced — mirrored permissions give a caller their own mail, their own files — and those slices + * are ranked exactly together, which is cheaper than a walk and exact by construction. A source + * the caller is a member of but which has no index of its own is sliced too. + */ +function planSourceVectorCandidates(input: { + plan: SearchAccessPlan + indexedSources: ReadonlySet +}): SourceVectorPlan { + const eligible = [ + ...new Set([ + ...input.plan.connectors.workspace, + ...input.plan.connectors.admin, + ...input.plan.connectors.members, + ]), + ] + const walked = input.plan.memberSources.filter((id) => input.indexedSources.has(id)) + const walking = new Set(walked) + return { walked, sliced: eligible.filter((id) => !walking.has(id)) } +} + +/** + * The nearest readable chunks, gathered per source and merged by distance. + * + * Walking one source at a time is what keeps recall: pgvector post-filters, so a walk over every + * source spends its scan budget on the sources this caller cannot read and returns few of their + * true neighbours. Inside one source they read, almost every neighbour qualifies. + * + * Nothing but the merged identities crosses the wire — each source's readable documents are + * resolved inside its own statement. + */ +async function selectSourceVectorCandidates(input: { + access: UserAccessScope + knowledgeBaseIds: string[] + plan: SearchAccessPlan + tagCondition: SQL | undefined + documentCondition: SQL | undefined + /** Sources the caller turned out not to hold, kept out of every source's ranking. */ + exclusion: SQL | undefined + /** Whether every projection row carries its mirrored columns, so a walk needs no document. */ + projectionFilled: boolean + candidateDistance: SQL + candidateLimit: number + budget?: SearchBudget +}): Promise { + const sources = planSourceVectorCandidates({ + plan: input.plan, + indexedSources: await indexedVectorSources(input.budget), + }) + annotateSearchDiagnostics({ + vectorRanking: 'per-source', + vectorSourcesWalked: sources.walked.length, + vectorSourcesSliced: sources.sliced.length, + }) + const base = and( + inArray(embeddingSearch.knowledgeBaseId, input.knowledgeBaseIds), + eq(embeddingSearch.enabled, true), + input.tagCondition, + input.exclusion + ) + type RankedChunks = Promise> + /** + * Walks one source's own index, or the sliced sources together when their slice saturated. + * Readability is decided on the row the walk visits — the source and ACL are mirrored there — + * so the graph is not stalled by a document lookup per candidate; the tag filter, which lives on + * the chunk, still joins. + */ + const onRow = projectionCandidateAccessCondition(embeddingSearch, input.access, input.plan, { + filled: input.projectionFilled, + }) + const walk = + (scope: SQL): (() => RankedChunks) => + () => + withVectorScanSettings( + (executor) => + executor.execute(sql` + SELECT ${PROJECTION_CANDIDATE_COLUMNS}, ${input.candidateDistance} AS distance + FROM ${embeddingSearch} /* on-row visibility */ + WHERE ${and( + base, + scope, + onRow, + documentSatisfies(embeddingSearch.documentId, input.documentCondition) + )} + ORDER BY ${input.candidateDistance} LIMIT ${input.candidateLimit}`), + input.budget, + 'vector.source_walk', + onRowWalkScanTuples(input.documentCondition, input.projectionFilled) + ) + const walks: Array<() => RankedChunks> = sources.walked.map((connectorId) => + walk(eq(embeddingSearch.connectorId, connectorId)) + ) + const slicedScope = sql`(${embeddingSearch.connectorId} IS NULL + OR ${embeddingSearch.connectorId} = ANY(${textArrayLiteral([...sources.sliced])}))` + /** + * One statement for the sliced sources and, with them, every uploaded document: uploads carry no + * connector, so a caller who is a member of all the indexed sources would otherwise rank none. + */ + const slice: Array<() => RankedChunks> = [ + async () => { + /** + * The sliced sources' readable chunks, decided on the row, ranked exactly: the ACL index + * enumerates them and `+ 0` keeps the planner off the graph. The chunks are counted one past + * the bound in the same statement, so a set too large to rank exactly is known before it is. + */ + const readableChunks = and( + base, + slicedScope, + onRow, + documentSatisfies(embeddingSearch.documentId, input.documentCondition) + ) + const rows = await runSearchQuery(input.budget, 'vector.source_exact', (executor) => + executor.execute(sql` + WITH readable_chunks AS MATERIALIZED ( + SELECT ${PROJECTION_CANDIDATE_COLUMNS}, ${input.candidateDistance} AS distance + FROM ${embeddingSearch} + WHERE ${readableChunks} + LIMIT ${SOURCE_EXACT_CHUNK_LIMIT + 1} + ) + SELECT id, "documentId", "connectorId", distance + 0 AS distance, + (SELECT count(*) FROM readable_chunks) > ${SOURCE_EXACT_CHUNK_LIMIT} AS saturated + FROM readable_chunks + ORDER BY distance LIMIT ${input.candidateLimit}`) + ) + /** + * The slice enumerates readable chunks in no particular order, so a set past its bound + * would rank an arbitrary subset and could miss the nearest chunks entirely. Walk those + * sources instead: approximate, but drawn from the whole of them. + */ + if (!rows.some((row) => row.saturated)) return rows + annotateSearchDiagnostics({ vectorSlicedSaturated: true }) + return walk(slicedScope)() + }, + ] + /** A source whose search runs out of budget marks the leg partial; the others' results stand. */ + const scored = await mapWithConcurrency( + [...walks, ...slice], + SOURCE_RANKING_CONCURRENCY, + async (run) => { + try { + return await run() + } catch (error) { + if (!input.budget?.isTimeout(error)) throw error + return [] + } + } + ) + const ranked: Array = scored.flat() + return ranked + .sort((a, b) => Number(a.distance) - Number(b.distance)) + .slice(0, input.candidateLimit) +} + +/** + * The vector leg of a user-scoped search-index search: a bounded candidate pool decided on the + * projection row through the caller's resolved plan, then hydrated under the full read predicate. + * + * A bounded permitted set is ranked exactly; a reader who is a member of indexed sources has each + * walked on its own; a broad reader walks the whole graph once. Live source authorization and + * content hydration still run after candidate ranking. + */ +export async function selectIndexedVectorResults( + params: SearchParams, + context: IndexedRetrievalContext +): Promise { + const setup = prepareVectorLeg(params) + const { access, accessPlan: plan, permitted } = context + /** Only live-verified readers may defer source authorization until after candidate ranking. */ + const candidateAccess = knowledgeCandidateAccessConditionForConnectors(access, plan) + /** + * What an on-row walk still has to ask the document: the tags, which live on chunks, and the + * date filter, which the row does not carry. A bounded set never walks, so this only runs when + * the filtered documents were too many to enumerate. + */ + const dateCondition = searchDateFilterCondition(params.filters) + const documentCondition = + setup.documentTagCondition || dateCondition + ? and(setup.documentTagCondition, dateCondition) + : undefined + /** `filled`: whether the pool's rows carry their source, so a page needs no read of its own. */ + let candidatePool: (VectorCandidatePool & { filled: boolean }) | undefined + return selectAuthorizedSearchResults({ + leg: 'vector', + access: params.access, + liveSourceAccess: context.liveSourceAccess, + signal: params.signal, + budget: params.budget, + topK: params.topK, + compareResults: (a, b) => a.distance - b.distance, + selectPage: async (limit, offset, excludedSources) => { + if (params.filters?.documentIds?.length) { + return selectExactVectorPage( + setup, + params.budget, + [ + ...getVisibilityConditions(params.filters, candidateAccess), + excludeSearchSources(excludedSources), + ], + limit, + offset + ) + } + if (permitted?.kind === 'bounded' && permitted.documents.length === 0) + return { candidates: [], nextOffset: offset } + candidatePool = await readVectorCandidatePool( + candidatePool, + excludedSources, + offset, + limit, + async ({ excludedKey, candidateLimit, previous: previousPool }) => { + /** Two remembered facts, read together when neither is remembered. */ + const [filled, plannedIndexedSources] = await Promise.all([ + isProjectionFilled('embedding_search', 'vector.projection_filled', params.budget), + plan.memberSources.length ? indexedVectorSources(params.budget) : undefined, + ]) + /** + * A source the caller turned out not to hold is left out where the pool is built: the + * pool is the page's order now, so a denied source's chunks would otherwise keep their + * slots. The row's mirrored source decides it, unless the row is decided on its document. + */ + const excludedOnRow = excludeSearchSourcesOnRow(embeddingSearch, filled, excludedSources) + annotateVectorPoolPlanned(setup, candidateLimit) + /** + * Exact ranking reads what the permitted set costs rather than re-deriving permission + * across the whole index, and honours `statement_timeout`, which a traversal cannot. + * `read` are chunks a pool already holds, ranked past. + */ + const rankPermittedExactly = (documentIds: string[], read?: readonly string[]) => + rankVectorCandidatesExactly({ + setup, + knowledgeBaseIds: params.knowledgeBaseIds, + documentIds, + columns: PROJECTION_CANDIDATE_COLUMNS, + conditions: [ + read?.length + ? sql`NOT (${embeddingSearch.id} = ANY(${textArrayLiteral([...read])}))` + : undefined, + excludedOnRow, + ], + candidateLimit, + budget: params.budget, + }) + let selected: SearchReadCandidate[] + /** Set where a pool's end is known better than by its length. */ + let exhausted: boolean | undefined + const walkGraph = () => + withVectorScanSettings( + (executor) => + executor.execute(sql` + SELECT ${PROJECTION_CANDIDATE_COLUMNS} + FROM ${embeddingSearch} /* on-row visibility */ + WHERE ${and( + inArray(embeddingSearch.knowledgeBaseId, params.knowledgeBaseIds), + eq(embeddingSearch.enabled, true), + excludedOnRow, + projectionCandidateAccessCondition(embeddingSearch, access, plan, { filled }), + documentSatisfies(embeddingSearch.documentId, documentCondition) + )} + ORDER BY ${setup.distance} LIMIT ${candidateLimit} + `), + params.budget, + 'vector.candidate_search', + onRowWalkScanTuples(documentCondition, filled) + ) + /** + * A source the caller is a member of that has its own index is walked on its own, which + * beats ranking it exactly once it is large enough to have earned that index. + */ + const walksASource = plan.memberSources.some( + (id) => plannedIndexedSources?.has(id) ?? false + ) + if ( + permitted?.kind === 'bounded' && + filled && + permitted.documents.length >= PERMITTED_EXACT_DOCUMENT_LIMIT + ) { + /** + * A set this large costs more to rank exactly than to walk: exact ranking reads every + * chunk of every document in it, while the walk decides readability on the rows it + * visits and stops at its tuple cap. The walk answers whenever the set is a fair share + * of the graph; where it is not, the walk underfills and the exact ranking that was + * always complete takes over, so nothing is lost but the walk's bounded cost. + * + * The walk decides readability on the projection row, which is broader than the + * document predicate hydration applies, so a pool it filled can still run short of + * readable rows. That shortfall is what refills a pool: the refill is the exact ranking, + * complete over the set, ranked past the rows already read and placed behind them, so + * the pages keep their offsets and every refill is a full window of fresh rows. + */ + const permittedIds = permitted.documents.map((entry) => entry.id) + const previous = previousPool?.ids + if (previous) { + const exact = await rankPermittedExactly( + permittedIds, + previous.map((candidate) => candidate.id) + ) + selected = [...previous, ...exact] + exhausted = exact.length < candidateLimit + } else { + selected = await walkGraph() + if (selected.length < candidateLimit) + selected = await rankPermittedExactly(permittedIds) + } + } else if (permitted?.kind === 'bounded' && (!walksASource || context.filtered)) { + /** + * A bounded permitted set is ranked exactly without walking the graph first: the walk + * post-filters, so when the caller reads a small share of the index it spends its whole + * uninterruptible tuple budget and still returns almost none of their neighbours. A + * member's indexed source is otherwise walked instead, but not under a filter: the walk + * cannot see the date, and a filtered set is small by construction. + */ + selected = await rankPermittedExactly(permitted.documents.map((entry) => entry.id)) + } else if (!(permitted?.kind === 'unbounded' && permitted.broad)) { + /** + * Readability follows sources, so each readable source is searched in its own index and + * the results merged. A member reads a source whole or barely at all: walking one source + * spends its budget among chunks they can read, where a walk over every source spends it + * on the sources they cannot. A caller whose reach is broad skips this: for them the + * whole graph's neighbours are mostly theirs already, and one walk is the cheaper answer. + */ + selected = await selectSourceVectorCandidates({ + access, + knowledgeBaseIds: params.knowledgeBaseIds, + plan, + exclusion: excludedOnRow, + projectionFilled: filled, + tagCondition: setup.candidateTagCondition, + documentCondition, + candidateDistance: setup.distance, + candidateLimit, + budget: params.budget, + }) + } else { + /** + * A broad reader's nearest chunks are mostly theirs, so one walk over the whole graph, + * deciding readability on the row, is the cheapest exact answer there is. + */ + selected = await walkGraph() + } + const pool = { + ...gatheredVectorCandidatePool(excludedKey, selected, candidateLimit, exhausted), + filled, + } + annotateVectorPoolSelected(selected.length, candidateLimit) + return pool + } + ) + /** + * The walk carries each candidate's document and source, so a page is a slice of the pool. + * Rescoring the pool against the original vectors here read one out-of-line vector per + * candidate from storage no cache holds, seconds on a query nobody had run before; the page + * is scored at hydration instead. + */ + if (candidatePool.filled) return sliceVectorCandidatePool(candidatePool, offset, limit) + /** + * While the source and ACL fill runs, a row it has not reached carries no source, so the + * page's identities are read off the documents; a slice whose documents all went away since + * the walk is passed over, not mistaken for the pool's end. + */ + for (let start = offset; start < candidatePool.ids.length; start += limit) { + const slice = candidatePool.ids.slice(start, start + limit) + const ranked = new Map(slice.map((candidate, index) => [candidate.id, index])) + const identities = await runSearchQuery(params.budget, 'vector.page', (executor) => + executor.execute(sql` + SELECT ${embeddingSearch.id} AS id, ${document.id} AS "documentId", + ${document.connectorId} AS "connectorId" + FROM ${embeddingSearch} + INNER JOIN ${document} ON ${document.id} = ${embeddingSearch.documentId} + WHERE ${embeddingSearch.id} = ANY(${textArrayLiteral(slice.map((candidate) => candidate.id))}) + `) + ) + if (!identities.length) continue + const page = [...identities].sort( + (a, b) => (ranked.get(a.id) ?? 0) - (ranked.get(b.id) ?? 0) + ) + return { candidates: page, nextOffset: start + slice.length } + } + return { candidates: [], nextOffset: candidatePool.ids.length } + }, + hydrate: (ids, authorized) => + hydrateVectorCandidates(ids, knowledgeAccessCondition(authorized), setup, params), + }) +} diff --git a/apps/sim/lib/knowledge/application/workspace-search.activity.test.ts b/apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts similarity index 92% rename from apps/sim/lib/knowledge/application/workspace-search.activity.test.ts rename to apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts index 14016b456cf..612f5617545 100644 --- a/apps/sim/lib/knowledge/application/workspace-search.activity.test.ts +++ b/apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts @@ -1,6 +1,7 @@ import { member } from '@sim/db/schema' import { queueTableRows, resetDbChainMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -18,7 +19,7 @@ import { permissionGroupsResolveMockFns, } from '@sim/testing/mocks/permission-groups-resolve.mock' import { workspaceAuthzMock } from '@sim/testing/mocks/workspace-authz.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const hoisted = vi.hoisted(() => ({ findIndex: vi.fn(), @@ -29,7 +30,6 @@ vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveM vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) vi.mock('@/lib/knowledge/search/search-index', () => ({ findSearchIndex: hoisted.findIndex, - findWorkspaceSearchIndex: hoisted.findIndex, })) vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) vi.mock('@/lib/knowledge/search/activity', () => ({ @@ -40,7 +40,7 @@ vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) import { searchOrganizationKnowledge, searchScopedKnowledge, -} from '@/lib/knowledge/application/workspace-search' +} from '@/lib/sim-search/indexed/search/scoped-search' const mocks = { ...hoisted, @@ -59,6 +59,10 @@ knowledgeContextsMockFns.mockResolveKnowledgeWorkspaceContext.mockImplementation const principal = createSessionPrincipal({ userId: 'reader', sessionId: 'session' }) const input = { organizationId: 'org', query: 'policy', topK: 20, surface: 'slack' } as const +/** These use cases run only while indexed organization search is on. */ +beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) +afterEach(resetEnvFlagsMock) + beforeEach(() => { resetDbChainMock() knowledgeContextsMockFns.mockResolveKnowledgeOrganizationContext.mockResolvedValue({ diff --git a/apps/sim/lib/knowledge/application/workspace-search.test.ts b/apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts similarity index 77% rename from apps/sim/lib/knowledge/application/workspace-search.test.ts rename to apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts index 88d30b834f7..70891390be9 100644 --- a/apps/sim/lib/knowledge/application/workspace-search.test.ts +++ b/apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts @@ -6,6 +6,7 @@ import { schemaMock, } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' +import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeContextsMock, knowledgeContextsMockFns, @@ -15,13 +16,14 @@ import { knowledgeSearchUseCaseMockFns, } from '@sim/testing/mocks/knowledge-search-use-case.mock' import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) -import { searchWorkspaceKnowledge } from '@/lib/knowledge/application/workspace-search' +import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' +import { searchWorkspaceKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' const mocks = { afterSearch: knowledgeSearchUseCaseMockFns.mockAfterKnowledgeSearch, @@ -35,6 +37,11 @@ workspaceAuthzMockFns.mockPermissionSatisfies.mockImplementation( const principal = createSessionPrincipal({ userId: 'reader', sessionId: 'session' }) const input = { workspaceId: 'workspace', query: 'orion', topK: 20, filters: { source: 'slack' } } + +/** These use cases run only while indexed organization search is on. */ +beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) +afterEach(resetEnvFlagsMock) + describe('canonical workspace search', () => { beforeEach(() => { resetDbChainMock() @@ -75,6 +82,14 @@ describe('canonical workspace search', () => { ).toBe(true) expect(dbChainMockFns.limit).toHaveBeenCalledWith(1) }) + it('refuses on its own while indexed organization search is dormant', async () => { + setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) + await expect(searchWorkspaceKnowledge.execute({ principal, input })).rejects.toBeInstanceOf( + SearchIndexDormantError + ) + expect(knowledgeContextsMockFns.mockResolveKnowledgeWorkspaceContext).not.toHaveBeenCalled() + expect(dbChainMockFns.where).not.toHaveBeenCalled() + }) it('refuses a nonmember before querying the protected index', async () => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(null) await expect(searchWorkspaceKnowledge.execute({ principal, input })).rejects.toThrow( diff --git a/apps/sim/lib/knowledge/application/workspace-search.ts b/apps/sim/lib/sim-search/indexed/search/scoped-search.ts similarity index 94% rename from apps/sim/lib/knowledge/application/workspace-search.ts rename to apps/sim/lib/sim-search/indexed/search/scoped-search.ts index ea22ab38bfb..63c038f4c75 100644 --- a/apps/sim/lib/knowledge/application/workspace-search.ts +++ b/apps/sim/lib/sim-search/indexed/search/scoped-search.ts @@ -23,7 +23,8 @@ import { instrumentSearchUseCase } from '@/lib/knowledge/application/search-diag import type { ActiveKnowledgeBaseReference } from '@/lib/knowledge/knowledge-base-reference' import { recordOrganizationSearchActivity } from '@/lib/knowledge/search/activity' import { measureSearchStage } from '@/lib/knowledge/search/diagnostics' -import { findSearchIndex, findWorkspaceSearchIndex } from '@/lib/knowledge/search/search-index' +import { findSearchIndex } from '@/lib/knowledge/search/search-index' +import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' export type SearchWorkspaceKnowledgeInput = Omit< SearchKnowledgeInput, @@ -103,8 +104,10 @@ function defineScopedSearchUseCase< }) { return defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.search, - resolveContext: ({ input }: { input: I }) => - measureSearchStage('scope_resolution', () => surface.resolveContext(input)), + resolveContext: ({ input }: { input: I }) => { + assertIndexedOrgSearchEnabled() + return measureSearchStage('scope_resolution', () => surface.resolveContext(input)) + }, async execute({ principal, input, context }): Promise { input.signal?.throwIfAborted() if ( @@ -149,7 +152,8 @@ export const searchWorkspaceKnowledge = instrumentSearchUseCase( 'workspace_application', defineScopedSearchUseCase({ resolveContext: (input) => resolveKnowledgeWorkspaceContext(input), - findIndex: (context) => findWorkspaceSearchIndex(context.workspaceId!), + findIndex: (context) => + findSearchIndex({ kind: 'workspace', workspaceId: context.workspaceId! }), searchInput: (input, context) => ({ ...input, workspaceId: context.workspaceId }), }) ) diff --git a/apps/sim/lib/sim-search/live/README.md b/apps/sim/lib/sim-search/live/README.md index db1f20d01d1..c67b634c5d9 100644 --- a/apps/sim/lib/sim-search/live/README.md +++ b/apps/sim/lib/sim-search/live/README.md @@ -1,6 +1,6 @@ # Federated Search access and connector behavior -This describes the live enterprise-search path. Credential Groups and ordinary knowledge-base indexing retain their existing behavior. Live Search is enabled by default; only an explicit `SIM_SEARCH_LIVE=false` selects the legacy indexed backend. Live Search sources do not create content-indexing jobs, and queued content or persisted-directory jobs stop before crawling, embedding, or building ACL snapshots. Ordinary KB jobs remain enabled. Administrators can still maintain GitLab CSV grants, and request-time source permission checks remain required. +This describes the live enterprise-search path. Credential Groups and ordinary knowledge-base indexing retain their existing behavior. Live Search is enabled by default; only an explicit `SIM_SEARCH_LIVE=false` selects the legacy indexed backend, which is kept dormant in `../indexed/` (see its README). Live Search sources do not create content-indexing jobs, and queued content or persisted-directory jobs stop before crawling, embedding, or building ACL snapshots. Ordinary KB jobs remain enabled. Administrators can still maintain GitLab CSV grants, and request-time source permission checks remain required. ## Admin and member surfaces diff --git a/apps/sim/lib/sim-search/live/application.test.ts b/apps/sim/lib/sim-search/live/application.test.ts index 9559227692c..0108c2fa6ac 100644 --- a/apps/sim/lib/sim-search/live/application.test.ts +++ b/apps/sim/lib/sim-search/live/application.test.ts @@ -1,6 +1,5 @@ import { resetDbChainMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' -import { setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { inputValidationMock, inputValidationMockFns, @@ -103,7 +102,6 @@ const document = { describe('authorized live retrieval', () => { beforeEach(() => { resetDbChainMock() - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) mocks.service.mockResolvedValue(undefined) knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue({ workspaceId: 'workspace', diff --git a/apps/sim/lib/table/application/operations.ts b/apps/sim/lib/table/application/operations.ts index d54e6d5ca36..556294ab263 100644 --- a/apps/sim/lib/table/application/operations.ts +++ b/apps/sim/lib/table/application/operations.ts @@ -139,27 +139,6 @@ export const tableOperations = { restore: writeOperation('tables.restore'), bulkMove: writeOperation('tables.bulk_move'), bulkDelete: writeOperation('tables.bulk_delete'), - renameByVfsPath: defineWorkspaceOperation({ - id: 'tables.vfs.rename', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'tables.use', - ...COPILOT_PRINCIPAL_POLICY, - }), - moveByVfsPath: defineWorkspaceOperation({ - id: 'tables.vfs.move', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'tables.use', - ...COPILOT_PRINCIPAL_POLICY, - }), - deleteByVfsPath: defineWorkspaceOperation({ - id: 'tables.vfs.delete', - minimumRole: 'write', - workspaceApiKey: 'deny', - capability: 'tables.use', - ...COPILOT_PRINCIPAL_POLICY, - }), listFolders: readOperation('tables.folders.list'), createFolder: writeOperation('tables.folders.create'), updateFolder: writeOperation('tables.folders.update'), diff --git a/apps/sim/lib/table/application/table-vfs.ts b/apps/sim/lib/table/application/table-vfs.ts index f4a50dcbc20..a32e5510457 100644 --- a/apps/sim/lib/table/application/table-vfs.ts +++ b/apps/sim/lib/table/application/table-vfs.ts @@ -1,75 +1,55 @@ -import { AuditAction, AuditResourceType } from '@sim/audit' -import { resolvePrincipalAttribution } from '@sim/auth/principal' import { OrchestrationError } from '@/lib/core/orchestration/types' -import { generateRequestId } from '@/lib/core/utils/request' -import { - createResourceVfsFolders, - deleteResourceVfsFolders, - type FolderedResourceAdapter, - resolveResourceRowBySegments, - transferResourceVfsItems, -} from '@/lib/folders/application/resource-vfs' -import { notifyWorkspaceTablesChanged } from '@/lib/realtime/notify' -import { defineAuthorizedTableUseCase } from '@/lib/table/application/authorized-table-use-case' -import { resolveTableWorkspaceContext } from '@/lib/table/application/context' -import { tableOperations } from '@/lib/table/application/operations' -import { - deleteTable, - findActiveTablesByExactName, - listTables, - moveTableToFolder, - renameTable, -} from '@/lib/table/service' +import { listFoldersForWorkspace } from '@/lib/folders/queries' +import { findActiveTablesByExactName, listTables } from '@/lib/table/service' import type { TableDefinition } from '@/lib/table/types' -interface TableVfsReferenceInput { - workspaceId: string - sourceName: string - /** Folder segments + leaf name; when present the nested-aware resolver is used. */ - sourceSegments?: string[] -} - -const tableVfsAdapter: FolderedResourceAdapter = { - resourceType: 'table', - rootSegment: 'tables', - label: 'table', - async listRows(workspaceId) { - const tables = await listTables(workspaceId) - return tables.map((t) => ({ id: t.id, name: t.name, folderId: t.folderId ?? null })) - }, - async moveRow(row, folderId, workspaceId) { - await moveTableToFolder(row.id, workspaceId, folderId, generateRequestId()) - }, - async renameRow(row, newName, workspaceId) { - const renamed = await renameTable(row.id, newName, generateRequestId(), { - expectedWorkspaceId: workspaceId, - skipNotify: true, - }) - return { id: renamed.id, name: renamed.name } - }, -} - -export interface RenameTableByVfsPathInput extends TableVfsReferenceInput { - newName: string +type TableFolder = Awaited>[number] + +/** + * Resolves the folder a path of folder names leads to from the root: `null` for the root itself, + * `undefined` when a segment names no folder. + */ +function resolveTableFolderId( + folders: readonly TableFolder[], + segments: readonly string[] +): string | null | undefined { + let current: string | null = null + for (const segment of segments) { + const next = folders.find((folder) => folder.parentId === current && folder.name === segment) + if (!next) return undefined + current = next.id + } + return current } -export type DeleteTableByVfsPathInput = TableVfsReferenceInput - +/** + * Resolves a table from its VFS path, `tables/{...folders}/{name}`. A bare name matches the one + * active table of that name anywhere in the tree. Authorization is the caller's: every entry + * point is a per-resource authorized use case. + */ export async function resolveTableByVfsName( workspaceId: string, sourceName: string, sourceSegments?: string[] ): Promise { if (sourceSegments && sourceSegments.length > 1) { - const row = await resolveResourceRowBySegments(tableVfsAdapter, workspaceId, sourceSegments) - const matches = await findActiveTablesByExactName(workspaceId, row.name) - const table = matches.find((t) => t.id === row.id) - if (!table) - throw new OrchestrationError( - 'not_found', - `Table not found at tables/${sourceSegments.join('/')}` + const path = `tables/${sourceSegments.join('/')}` + const leaf = sourceSegments[sourceSegments.length - 1] + const folders = await listFoldersForWorkspace(workspaceId, 'active', 'table') + const parentId = resolveTableFolderId(folders, sourceSegments.slice(0, -1)) + if (parentId !== undefined) { + const inParent = (await listTables(workspaceId)).filter( + (table) => table.name === leaf && (table.folderId ?? null) === parentId ) - return table + if (inParent.length === 1) return inParent[0] + if (resolveTableFolderId(folders, sourceSegments)) { + throw new OrchestrationError( + 'validation', + `${path} is a folder; this operation takes a table.` + ) + } + } + throw new OrchestrationError('not_found', `No table or folder found at ${path}`) } const matches = await findActiveTablesByExactName(workspaceId, sourceName) if (matches.length > 1) { @@ -79,162 +59,3 @@ export async function resolveTableByVfsName( if (!table) throw new OrchestrationError('not_found', `Table not found at tables/${sourceName}`) return table } - -export const renameTableByVfsPath = defineAuthorizedTableUseCase({ - operation: tableOperations.renameByVfsPath, - resolveContext: ({ input }: { input: RenameTableByVfsPathInput }) => - resolveTableWorkspaceContext(input.workspaceId), - async execute({ input, context }) { - const table = await resolveTableByVfsName( - context.workspaceId, - input.sourceName, - input.sourceSegments - ) - const renamed = await renameTable(table.id, input.newName, generateRequestId(), { - expectedWorkspaceId: context.workspaceId, - skipNotify: true, - }) - return { - id: renamed.id, - name: renamed.name, - previousName: table.name, - workspaceId: context.workspaceId, - } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.TABLE_UPDATED, - resourceType: AuditResourceType.TABLE, - resourceId: result.id, - resourceName: result.name, - description: `Renamed table to "${result.name}"`, - metadata: { op: 'rename', previousName: result.previousName, source: 'copilot_vfs' }, - }), - afterSuccess: ({ context }) => notifyWorkspaceTablesChanged(context.workspaceId), -}) - -export const deleteTableByVfsPath = defineAuthorizedTableUseCase({ - operation: tableOperations.deleteByVfsPath, - resolveContext: ({ input }: { input: DeleteTableByVfsPathInput }) => - resolveTableWorkspaceContext(input.workspaceId), - async execute({ input, context }) { - const table = await resolveTableByVfsName( - context.workspaceId, - input.sourceName, - input.sourceSegments - ) - const { archived } = await deleteTable(table.id, generateRequestId(), { - expectedWorkspaceId: context.workspaceId, - skipNotify: true, - }) - if (!archived) - throw new OrchestrationError('not_found', `Table not found at tables/${input.sourceName}`) - return { - id: table.id, - name: archived.name, - workspaceId: context.workspaceId, - deleted: true as const, - } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.TABLE_DELETED, - resourceType: AuditResourceType.TABLE, - resourceId: result.id, - resourceName: result.name, - description: `Archived table "${result.name}"`, - metadata: { source: 'copilot_vfs' }, - }), - afterSuccess: ({ context }) => notifyWorkspaceTablesChanged(context.workspaceId), -}) - -export interface TableVfsPathsInput { - workspaceId: string - paths: Array<{ source: string; segments: string[] }> -} - -export interface TransferTableVfsItemsInput { - workspaceId: string - sources: Array<{ source: string; segments: string[] }> - destination: { segments: string[]; trailingSlash: boolean } -} - -/** mkdir -p under tables/ — folder invariants live in lib/folders. */ -export const createTableVfsFolders = defineAuthorizedTableUseCase({ - operation: tableOperations.createFolder, - resolveContext: ({ input }: { input: TableVfsPathsInput }) => - resolveTableWorkspaceContext(input.workspaceId), - async execute({ principal, input, context }) { - const userId = resolvePrincipalAttribution(principal, { - workspaceBillingOwnerUserId: context.billedAccountUserId, - }).attributedUserId - const outcomes = await createResourceVfsFolders(tableVfsAdapter, { - workspaceId: context.workspaceId, - userId, - paths: input.paths, - }) - return { outcomes, workspaceId: context.workspaceId } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.TABLE_UPDATED, - resourceType: AuditResourceType.TABLE, - resourceId: result.workspaceId, - resourceName: 'tables', - description: 'Created table folders', - metadata: { op: 'vfs_mkdir', count: result.outcomes.length, source: 'copilot_vfs' }, - }), - afterSuccess: ({ context }) => notifyWorkspaceTablesChanged(context.workspaceId), -}) - -/** mv under tables/: rows into folders, folder moves/renames, leaf renames. */ -export const transferTableVfsItems = defineAuthorizedTableUseCase({ - operation: tableOperations.moveByVfsPath, - resolveContext: ({ input }: { input: TransferTableVfsItemsInput }) => - resolveTableWorkspaceContext(input.workspaceId), - async execute({ principal, input, context }) { - const userId = resolvePrincipalAttribution(principal, { - workspaceBillingOwnerUserId: context.billedAccountUserId, - }).attributedUserId - const outcomes = await transferResourceVfsItems(tableVfsAdapter, { - workspaceId: context.workspaceId, - userId, - sources: input.sources, - destination: input.destination, - }) - return { outcomes, workspaceId: context.workspaceId } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.TABLE_UPDATED, - resourceType: AuditResourceType.TABLE, - resourceId: result.workspaceId, - resourceName: 'tables', - description: 'Moved table VFS items', - metadata: { op: 'vfs_mv', count: result.outcomes.length, source: 'copilot_vfs' }, - }), - afterSuccess: ({ context }) => notifyWorkspaceTablesChanged(context.workspaceId), -}) - -/** rm of tables/ folder paths — recursive via the shared cascade. */ -export const deleteTableVfsFolders = defineAuthorizedTableUseCase({ - operation: tableOperations.deleteFolder, - resolveContext: ({ input }: { input: TableVfsPathsInput }) => - resolveTableWorkspaceContext(input.workspaceId), - async execute({ principal, input, context }) { - const userId = resolvePrincipalAttribution(principal, { - workspaceBillingOwnerUserId: context.billedAccountUserId, - }).attributedUserId - const outcomes = await deleteResourceVfsFolders(tableVfsAdapter, { - workspaceId: context.workspaceId, - userId, - paths: input.paths, - }) - return { outcomes, workspaceId: context.workspaceId } - }, - projectAudit: ({ result }) => ({ - action: AuditAction.TABLE_DELETED, - resourceType: AuditResourceType.TABLE, - resourceId: result.workspaceId, - resourceName: 'tables', - description: 'Deleted table folders', - metadata: { op: 'vfs_rm_folder', count: result.outcomes.length, source: 'copilot_vfs' }, - }), - afterSuccess: ({ context }) => notifyWorkspaceTablesChanged(context.workspaceId), -}) diff --git a/packages/db/knowledge-projection.test.ts b/packages/db/knowledge-projection.test.ts index def4d6a65f8..22e9d4b86f6 100644 --- a/packages/db/knowledge-projection.test.ts +++ b/packages/db/knowledge-projection.test.ts @@ -1,8 +1,4 @@ -import { - FILL_MARK_CEILING, - markUnfilledProjectionDocuments, - runKnowledgeProjection, -} from '@sim/db/knowledge-projection' +import { runKnowledgeProjection } from '@sim/db/knowledge-projection' import type { Sql } from 'postgres' import { describe, expect, it, vi } from 'vitest' @@ -128,7 +124,7 @@ describe('runKnowledgeProjection', () => { ]), }) const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql) + const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) expect(progress).toMatchObject({ settled: 2, deferred: 0, remaining: false }) expect(state.marks.size).toBe(0) expect(trace.locks).toEqual(['doc-a', 'doc-b']) @@ -149,7 +145,7 @@ describe('runKnowledgeProjection', () => { const state = database({ marks: new Map([['doc', { generation: 1, content: false }]]) }) state.beforeSettle = () => state.marks.set('doc', { generation: 2, content: false }) const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql) + const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) expect(progress).toMatchObject({ settled: 0, deferred: 1, remaining: true }) expect(state.marks.get('doc')?.generation).toBe(2) /** A document given up is not claimed again by the same pass. */ @@ -165,7 +161,7 @@ describe('runKnowledgeProjection', () => { lockedElsewhere: new Set(['held']), }) const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql) + const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) expect(progress).toMatchObject({ settled: 1, deferred: 1, remaining: true }) expect(state.marks.has('held')).toBe(true) expect(trace.pages.some((page) => page.documentId === 'held')).toBe(false) @@ -188,7 +184,7 @@ describe('runKnowledgeProjection', () => { if (documentId === 'slow') throw failure } const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql) + const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) expect(progress).toMatchObject({ settled: 1, deferred: 1, remaining: true }) expect(state.marks.has('slow')).toBe(true) expect(trace.unlocks).toEqual(['slow', 'fine']) @@ -201,7 +197,7 @@ describe('runKnowledgeProjection', () => { throw postgresError('23503', 'insert violates foreign key') } const { sql } = fakeSql(state) - await expect(runKnowledgeProjection(sql)).resolves.toMatchObject({ + await expect(runKnowledgeProjection(sql, { searchIndexes: true })).resolves.toMatchObject({ settled: 0, deferred: 0, remaining: false, @@ -214,7 +210,9 @@ describe('runKnowledgeProjection', () => { throw postgresError('42P01', 'relation does not exist') } const { sql, trace } = fakeSql(state) - await expect(runKnowledgeProjection(sql)).rejects.toThrow('relation does not exist') + await expect(runKnowledgeProjection(sql, { searchIndexes: true })).rejects.toThrow( + 'relation does not exist' + ) expect(trace.unlocks).toEqual(['doc']) expect(state.marks.has('doc')).toBe(true) }) @@ -231,7 +229,11 @@ describe('runKnowledgeProjection', () => { vi.advanceTimersByTime(40) } const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql, { budgetMs: 100, pageSize: 2 }) + const progress = await runKnowledgeProjection(sql, { + budgetMs: 100, + pageSize: 2, + searchIndexes: true, + }) expect(trace.pages).toHaveLength(3) expect(progress).toMatchObject({ settled: 0, deferred: 1, remaining: true }) expect(state.marks.has('long')).toBe(true) @@ -241,43 +243,3 @@ describe('runKnowledgeProjection', () => { } }) }) - -describe('markUnfilledProjectionDocuments', () => { - /** A session answering the fill's two statements from `outstanding` and a queue of scan results. */ - function fillSql(outstanding: number, scans: Array<{ marked: number; last_id: string | null }>) { - const statements: Array<{ text: string; values: unknown[] }> = [] - const sql = Object.assign( - async (strings: TemplateStringsArray) => { - if (strings.join('').includes('count(*)')) return [{ outstanding }] - return [] - }, - { - unsafe: async (text: string, values: unknown[]) => { - statements.push({ text, values }) - return [scans.shift() ?? { marked: 0, last_id: null }] - }, - } - ) as unknown as Sql - return { sql, statements } - } - - it('marks nothing while the outstanding marks are at the ceiling', async () => { - const { sql, statements } = fillSql(FILL_MARK_CEILING, []) - await expect(markUnfilledProjectionDocuments(sql)).resolves.toEqual({ - marked: 0, - cursor: { projection: 0, afterId: '' }, - }) - expect(statements).toEqual([]) - }) - - it('offers the room left and continues from the row the statement stopped before', async () => { - const { sql, statements } = fillSql(FILL_MARK_CEILING - 3, [{ marked: 3, last_id: 'row-9' }]) - await expect(markUnfilledProjectionDocuments(sql)).resolves.toEqual({ - marked: 3, - cursor: { projection: 0, afterId: 'row-9' }, - }) - expect(statements[0]?.text).toContain('FROM embedding_search WHERE acl IS NULL') - expect(statements[0]?.text).toContain('u.id < held.first_id') - expect(statements[0]?.values).toEqual(['', 3]) - }) -}) diff --git a/packages/db/knowledge-projection.ts b/packages/db/knowledge-projection.ts index 802874bd3ee..37a837294a4 100644 --- a/packages/db/knowledge-projection.ts +++ b/packages/db/knowledge-projection.ts @@ -5,26 +5,27 @@ import type { Sql, TransactionSql } from 'postgres' const logger = createLogger('KnowledgeProjection') /** - * The transaction setting a writer declares to leave its search projection rows to the projector. - * Unset, which is every writer that predates it, keeps the synchronous triggers; either way the + * The transaction setting that skips the synchronous projection triggers. No application writer + * sets it: every writer's projection rows are written by those triggers in its own transaction. + * Releases that carried the `knowledge-async-projection` flag set it to leave a chunk write's rows + * to the projector, which is why a mark can still carry content to project. Either way the * triggers mark the document in `knowledge_projection_dirty`. */ const KNOWLEDGE_PROJECTION_MODE_SETTING = 'sim.projection_mode' /** - * The expression a knowledge writer selects in its transaction to leave its search projection rows - * to the projector: `SELECT ${DEFER_KNOWLEDGE_PROJECTION}`. Transaction-local, so a pooled - * connection never carries it into the next transaction. Writers resolve whether to run it from the - * `knowledge-async-projection` flag before their transaction begins. + * Selected by every projector transaction. The projector writes each projection row's source and + * ACL itself, so the projection tables' own source and ACL triggers, which would re-read the + * document for every row, are skipped. */ -export const DEFER_KNOWLEDGE_PROJECTION = `set_config('${KNOWLEDGE_PROJECTION_MODE_SETTING}', 'async', true)` +const SKIP_SYNCHRONOUS_PROJECTION = `set_config('${KNOWLEDGE_PROJECTION_MODE_SETTING}', 'async', true)` -/** Whether the current transaction ran {@link DEFER_KNOWLEDGE_PROJECTION}, as trigger SQL reads it. */ +/** Whether the current transaction skips the synchronous projection triggers, as trigger SQL reads it. */ export const KNOWLEDGE_PROJECTION_DEFERRED = `current_setting('${KNOWLEDGE_PROJECTION_MODE_SETTING}', true) IS NOT DISTINCT FROM 'async'` /** * The `WHEN` clause of every trigger that writes projection rows in the writer's transaction: it - * fires unless the transaction deferred them. + * fires unless the transaction skips them. */ export const SYNCHRONOUS_PROJECTION_WHEN = `NOT (${KNOWLEDGE_PROJECTION_DEFERRED})` @@ -122,8 +123,9 @@ const PAGE_RESULT = `SELECT (SELECT count(*)::int FROM page) AS scanned, (SELECT max(chunk_index) FROM page) AS last_chunk` /** - * Rewrites a page's projection rows from their chunks and the document, writing only the rows - * that differ, so a document whose rows are already current costs reads and no index writes. The + * Rewrites a page's projection rows from their chunks and the document, for a mark whose chunk + * write skipped the synchronous triggers (only releases that deferred projection wrote those), + * writing only the rows that differ, so a document whose rows are already current costs reads and no index writes. The * source and ACL are the document's as this statement reads it; a change that commits after it * marks the document again, so the projector's settle leaves the mark for the next pass. */ @@ -196,7 +198,7 @@ function contentPageStatement(projection: KnowledgeProjection): string { /** * Copies the document's source and ACL onto a page's existing projection rows that differ, - * including rows the source and ACL fill has not reached. Chunks are untouched, so their vectors + * including rows written before projections carried a source and ACL. Chunks are untouched, so their vectors * are never read; a row a chunk change has not projected yet is left to that change's own mark. */ function sourceAclPageStatement(projection: KnowledgeProjection): string { @@ -226,7 +228,7 @@ async function enterProjectorTransaction(tx: TransactionSql, lockTimeoutMs: numb await tx.unsafe( `SELECT set_config('lock_timeout', '${lockTimeoutMs}ms', true), set_config('statement_timeout', '${PROJECTION_PAGE_STATEMENT_TIMEOUT_MS}ms', true), - ${DEFER_KNOWLEDGE_PROJECTION}` + ${SKIP_SYNCHRONOUS_PROJECTION}` ) } @@ -244,6 +246,13 @@ export interface KnowledgeProjectionOptions { */ budgetMs?: number pageSize?: number + /** + * Whether search-index rows are read, that is whether indexed organization search is on (see + * {@link MarkScope}). Only then is the Tin keyword projection written, since it holds only + * search-index rows; the other projections are written either way, and Tin is still skipped + * where it is not installed. + */ + searchIndexes: boolean /** Called after each page commits, for tests that interleave writes with a run. */ onPage?: (page: { documentId: string @@ -435,13 +444,14 @@ async function projectMarkedDocument( */ export async function runKnowledgeProjection( sql: Sql, - options: KnowledgeProjectionOptions = {} + options: KnowledgeProjectionOptions ): Promise { const deadline = options.budgetMs === undefined ? Number.POSITIVE_INFINITY : Date.now() + options.budgetMs - const projections = (await tinInstalled(sql)) - ? KNOWLEDGE_PROJECTIONS - : KNOWLEDGE_PROJECTIONS.filter((projection) => projection !== 'embedding_keyword_tin') + const projections = + options.searchIndexes && (await tinInstalled(sql)) + ? KNOWLEDGE_PROJECTIONS + : KNOWLEDGE_PROJECTIONS.filter((projection) => projection !== 'embedding_keyword_tin') const totals = { pages: 0, written: 0 } const skipped: string[] = [] let settled = 0 @@ -467,67 +477,112 @@ export async function runKnowledgeProjection( return { settled, deferred: skipped.length, ...totals, remaining: true } } -/** Unfilled rows the fill reads per round, from each projection's unfilled-rows index. */ -const FILL_SCAN_ROWS = 2_000 +/** + * How long one release of settled marks runs, wherever it is called from. A release is cleanup + * ahead of the real work, so it gets a short budget of its own rather than the caller's deadline; + * whatever it leaves, the next sweep releases. + */ +export const MARK_RELEASE_BUDGET_MS = 10_000 + +/** Marks one release statement removes; a release repeats it while statements come back full. */ +const RELEASE_BATCH_SIZE = 1_000 /** - * Marks at most this many documents may be outstanding before the fill adds more. Marks are - * projected oldest first, so this bounds how much fill work a fresh write can wait behind, and - * search, which decides a marked document's rows on the document, keeps doing so for few of them. + * Which marks a pass is owed besides content: search-index documents, whose rows mirror their + * source and ACL, while indexed organization search reads them. With it off, a mark with no content + * to project is owed nothing, whatever its knowledge base. */ -export const FILL_MARK_CEILING = 100 +export interface MarkScope { + searchIndexes: boolean +} /** - * Marks the documents of projection rows the source and ACL fill has not reached, so the projector - * fills them as it converges any other document. Rows are read off each projection's unfilled - * index after `afterId`, and only while fewer than {@link FILL_MARK_CEILING} marks are outstanding; - * a row whose document is gone is passed over. Documents are taken in the order of their first - * row read, up to the room left under the ceiling, and the cursor moves only past rows of the - * documents taken: when the room runs out, it stops before the first row of the first document - * left out, so the next call starts there. A taken document is marked only once its row is locked - * `FOR KEY SHARE`, so a deletion committing meanwhile can never fail the mark's foreign key; one - * whose row a deletion or another writer holds is skipped rather than waited on, since that writer - * removes or rewrites its rows itself, and the next pass reads whatever is still unfilled. Returns - * how many documents were marked and the id to continue after, or `null` once every projection's - * unfilled rows have been read. + * Whether any mark needs a projector pass: one carrying content to project, or, when `scope` owes + * search-index documents a pass, one on a search-index document. Every other mark is released by + * {@link releaseSettledMarks} without a pass. + * + * Asked right after a release. One that drained its marks left only the marks a pass is owed + * (and the few a writer held), so the join that tells a search-index mark apart walks a handful of + * rows. One cut short left a backlog the join would walk in full, so only the content marks are + * asked about: a search-index mark behind that backlog waits for the sweep whose release drains it. */ -export async function markUnfilledProjectionDocuments( +export async function hasKnowledgeProjectionWork( + sql: Sql | TransactionSql, + release: { drained: boolean }, + scope: MarkScope +): Promise { + const [row] = + release.drained && scope.searchIndexes + ? await sql>` + SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty WHERE content) + OR EXISTS ( + SELECT 1 FROM knowledge_projection_dirty d + JOIN document doc ON doc.id = d.document_id + JOIN knowledge_base k ON k.id = doc.knowledge_base_id + WHERE k.is_search_index + ) AS pending` + : await sql>` + SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty WHERE content) AS pending` + return Boolean(row?.pending) +} + +export interface SettledMarkRelease { + /** Marks removed. */ + released: number + /** Whether the release ran out of marks to remove rather than out of time. */ + drained: boolean + /** Whether no mark at all was found, so nothing is left for a pass either. */ + empty: boolean +} + +/** + * Transitional: removes the marks that carry no content to project and that `scope` owes no pass, + * until a follow-up migration scopes the mark triggers to search-index knowledge bases. Their + * synchronous writers already wrote every projection row a workspace search reads, and those + * searches decide nothing on a projection row's source, ACL, or mark, so a pass over them would + * only re-read rows it then leaves as they are. While indexed organization search is on, the marks + * of search-index documents, whose rows it reads by source and ACL, stay for a pass. + * + * An empty mark table costs one probe, and reports `empty` so its caller asks nothing further. Marks a writer holds are skipped rather than waited on, + * and a mark whose writer skipped the synchronous triggers carries content and stays for a pass. + * Stops once a statement comes back short or `deadline` passes. + */ +export async function releaseSettledMarks( sql: Sql, - cursor: { projection: number; afterId: string } = { projection: 0, afterId: '' } -): Promise<{ marked: number; cursor: { projection: number; afterId: string } | null }> { - const [{ outstanding }] = await sql>` - SELECT count(*)::int AS outstanding FROM knowledge_projection_dirty` - if (outstanding >= FILL_MARK_CEILING) return { marked: 0, cursor } - for (let index = cursor.projection; index < SOURCE_ACL_PROJECTIONS.length; index++) { - const projection = SOURCE_ACL_PROJECTIONS[index] - const afterId = index === cursor.projection ? cursor.afterId : '' - const [row] = await sql.unsafe>( - `WITH unfilled AS MATERIALIZED ( - SELECT id, document_id FROM ${projection} WHERE acl IS NULL AND id > $1 - ORDER BY id LIMIT ${FILL_SCAN_ROWS} - ), documents AS MATERIALIZED ( - SELECT u.document_id, min(u.id) AS first_id FROM unfilled u - WHERE EXISTS (SELECT 1 FROM document d WHERE d.id = u.document_id) - GROUP BY u.document_id - ), chosen AS MATERIALIZED ( - SELECT document_id, first_id FROM documents ORDER BY first_id LIMIT $2 - ), locked AS MATERIALIZED ( - SELECT d.id FROM document d WHERE d.id IN (SELECT document_id FROM chosen) - FOR KEY SHARE SKIP LOCKED - ), held AS ( - SELECT min(first_id) AS first_id FROM documents - WHERE document_id NOT IN (SELECT document_id FROM chosen) - ), marked AS ( - INSERT INTO knowledge_projection_dirty (document_id) - SELECT id FROM locked ORDER BY id - ON CONFLICT (document_id) DO NOTHING RETURNING document_id - ) SELECT (SELECT count(*)::int FROM marked) AS marked, - (SELECT max(u.id) FROM unfilled u, held - WHERE held.first_id IS NULL OR u.id < held.first_id) AS last_id`, - [afterId, FILL_MARK_CEILING - outstanding] - ) - if (row?.last_id) - return { marked: row.marked, cursor: { projection: index, afterId: row.last_id } } + deadline: number, + scope: MarkScope +): Promise { + const [marked] = await sql>` + SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty) AS any` + if (!marked?.any) return { released: 0, drained: true, empty: true } + let released = 0 + while (Date.now() < deadline) { + const count = await sql.begin(async (tx) => { + await enterProjectorTransaction(tx, SETTLE_LOCK_TIMEOUT_MS) + const settled = scope.searchIndexes + ? tx` + SELECT d.document_id FROM knowledge_projection_dirty d + JOIN document doc ON doc.id = d.document_id + JOIN knowledge_base k ON k.id = doc.knowledge_base_id + WHERE NOT d.content AND NOT k.is_search_index + LIMIT ${RELEASE_BATCH_SIZE} + FOR UPDATE OF d SKIP LOCKED` + : tx` + SELECT d.document_id FROM knowledge_projection_dirty d + WHERE NOT d.content + LIMIT ${RELEASE_BATCH_SIZE} + FOR UPDATE SKIP LOCKED` + const [row] = await tx>` + WITH released AS ( + DELETE FROM knowledge_projection_dirty m + WHERE m.document_id IN (${settled}) AND NOT m.content + RETURNING 1 + ) + SELECT count(*)::int AS released FROM released` + return row?.released ?? 0 + }) + released += count + if (count < RELEASE_BATCH_SIZE) return { released, drained: true, empty: false } } - return { marked: 0, cursor: null } + return { released, drained: false, empty: false } } diff --git a/packages/db/schema.ts b/packages/db/schema.ts index f5091327c1a..330068231c5 100644 --- a/packages/db/schema.ts +++ b/packages/db/schema.ts @@ -3746,8 +3746,9 @@ export const embeddingKeywordTin = pgTable( /** * Candidate projection. Keeping identities and half-precision vectors apart from content prevents * candidate scans from fetching full-precision TOAST values. Application writers only change - * `embedding`: its trigger writes this projection in the writer's transaction, or, for a writer that - * deferred it, the knowledge projector writes it after the commit (see {@link knowledgeProjectionDirty}). + * `embedding`: its trigger writes this projection in the writer's transaction. A chunk write that + * skipped the trigger, as releases that deferred projection did, is written by the knowledge + * projector after the commit (see {@link knowledgeProjectionDirty}). */ export const embeddingSearch = pgTable( 'embedding_search', @@ -3816,8 +3817,10 @@ export const embeddingSearch = pgTable( * Documents whose search projection rows may lag their source rows. The `document` and `embedding` * triggers mark a document here whenever they change what its projection rows carry, in the * writer's transaction; the projector rewrites the rows and then removes the mark, but only on the - * generation it read, so a change made while it ran leaves the mark in place. Search decides a - * marked document's rows on the document itself, so a mark never widens what a reader sees. + * generation it read, so a change made while it ran leaves the mark in place. A workspace + * document's mark with no content to project is removed without a pass: its writer already wrote + * the rows its searches read. Search-index search decides a marked document's rows on the + * document itself, so a mark never widens what a reader sees. * * A side table rather than a column on `document`: a mark is written by the writer that already * holds the document row, but clearing it would otherwise take that row again, and readers probe diff --git a/packages/db/script-migrations/0022_projection_source_acl_backfill.ts b/packages/db/script-migrations/0022_projection_source_acl_backfill.ts index 2d0aa70e3e8..28783495757 100644 --- a/packages/db/script-migrations/0022_projection_source_acl_backfill.ts +++ b/packages/db/script-migrations/0022_projection_source_acl_backfill.ts @@ -8,10 +8,9 @@ import type { ScriptMigration } from '@sim/db/script-migrations/types' * Installs the projection source and ACL triggers and builds their indexes; both are idempotent, * so a run that was cut short completes on the next deploy. The columns are not filled here: on * `embedding_search` every filled row is re-inserted into each HNSW index, which puts the full - * projection far beyond what a deploy job can wait for. The knowledge projector fills them after - * the deploy, marking the documents of unfilled rows in bounded batches and converging them as it - * converges any other document. Until then, an unfilled row is decided on its document, the join - * per candidate every row paid before the columns existed. + * projection far beyond what a deploy job can wait for. A row the triggers have not written since + * stays unfilled and is decided on its document, the join per candidate every row paid before the + * columns existed. * * Supersedes `0021_embedding_search_connector`, whose synchronous backfill this replaces: a * database that recorded it still gains the unfilled indexes, and one where it was cut short diff --git a/packages/testing/src/mocks/deployment-shape.mock.ts b/packages/testing/src/mocks/deployment-shape.mock.ts index 3ec37734903..443f6ba8082 100644 --- a/packages/testing/src/mocks/deployment-shape.mock.ts +++ b/packages/testing/src/mocks/deployment-shape.mock.ts @@ -26,7 +26,8 @@ export interface MockDeploymentShape { /** * Builds a deployment shape from the `env-flags` mock defaults (self-hosted, billing off, Chat on, - * inbox/whitelabeling/session policies on, everything else off), with shallow `features` overrides. + * Live Search, inbox/whitelabeling/session policies on, everything else off), with shallow + * `features` overrides. */ export function createMockDeploymentShape( overrides: Partial> & { @@ -42,7 +43,7 @@ export function createMockDeploymentShape( cohereConfigured: false, ...rest, features: { - liveEnterpriseSearch: false, + liveEnterpriseSearch: true, accessControl: false, auditLogs: false, customBlocks: false, diff --git a/packages/testing/src/mocks/env-flags.mock.ts b/packages/testing/src/mocks/env-flags.mock.ts index fb72be956ae..98666e661ec 100644 --- a/packages/testing/src/mocks/env-flags.mock.ts +++ b/packages/testing/src/mocks/env-flags.mock.ts @@ -86,10 +86,11 @@ const defaultEnvFlagsState: EnvFlagsMockState = { isUsageMonitoringEnabled: false, isAccessControlEnabled: false, isOrganizationsEnabled: false, + /** Live Search is the default Sim Search backend; indexed search is dormant. */ + isLiveEnterpriseSearchEnabled: true, // True with billing off and no flags set — these carry a legacy default of // `true` so upgrades do not remove a feature. See // ENTERPRISE_FEATURE_LEGACY_DEFAULTS. - isLiveEnterpriseSearchEnabled: false, isInboxEnabled: true, isSandboxDeploymentEntitled: false, isSandboxesEnabled: false, diff --git a/packages/testing/src/mocks/indexed-org-search.mock.ts b/packages/testing/src/mocks/indexed-org-search.mock.ts new file mode 100644 index 00000000000..d13b8db4a63 --- /dev/null +++ b/packages/testing/src/mocks/indexed-org-search.mock.ts @@ -0,0 +1,22 @@ +/** + * The `vi.mock` factory a real-infrastructure suite of indexed organization search passes for + * `@/lib/core/config/env-flags`: the real module, with Live Search off so the dormant indexed + * backend is the one selected. Unit suites use the shared env-flags mock's `setEnvFlags` instead. + * + * Loaded inside the factory, which runs before the test file's own imports are initialized. + * + * @example + * ```ts + * vi.mock('@/lib/core/config/env-flags', async (importOriginal) => + * (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags(importOriginal) + * ) + * ``` + */ +export async function indexedOrgSearchEnvFlags( + importOriginal: () => Promise +): Promise> { + return { + ...(await importOriginal>()), + isLiveEnterpriseSearchEnabled: false, + } +} diff --git a/packages/testing/src/mocks/knowledge-access-scope.mock.ts b/packages/testing/src/mocks/knowledge-access-scope.mock.ts index 4d5f133f524..a2d41973b2e 100644 --- a/packages/testing/src/mocks/knowledge-access-scope.mock.ts +++ b/packages/testing/src/mocks/knowledge-access-scope.mock.ts @@ -13,7 +13,6 @@ import { vi } from 'vitest' */ export const knowledgeAccessScopeMockFns = { mockResolveKnowledgeAccessScope: vi.fn(), - mockResolveUserKnowledgeAccessScope: vi.fn(), mockCreateKnowledgeAccessProvider: vi.fn(), mockCreateUserKnowledgeAccessProvider: vi.fn(), } @@ -33,7 +32,6 @@ export const knowledgeAccessScopeMock = { tokens: ['pub', 'ws'] as const, }), resolveKnowledgeAccessScope: knowledgeAccessScopeMockFns.mockResolveKnowledgeAccessScope, - resolveUserKnowledgeAccessScope: knowledgeAccessScopeMockFns.mockResolveUserKnowledgeAccessScope, createKnowledgeAccessProvider: knowledgeAccessScopeMockFns.mockCreateKnowledgeAccessProvider, createUserKnowledgeAccessProvider: knowledgeAccessScopeMockFns.mockCreateUserKnowledgeAccessProvider, diff --git a/packages/testing/src/mocks/knowledge-api-utils.mock.ts b/packages/testing/src/mocks/knowledge-api-utils.mock.ts index b0cb3ed816a..8e9013a64af 100644 --- a/packages/testing/src/mocks/knowledge-api-utils.mock.ts +++ b/packages/testing/src/mocks/knowledge-api-utils.mock.ts @@ -17,7 +17,6 @@ import { vi } from 'vitest' */ export const knowledgeApiUtilsMockFns = { mockCheckKnowledgeBaseAccess: vi.fn(), - mockCheckKnowledgeBaseWriteAccess: vi.fn(), mockCheckDocumentWriteAccess: vi.fn(), mockCheckDocumentAccess: vi.fn(), mockCheckChunkAccess: vi.fn(), @@ -33,7 +32,6 @@ export const knowledgeApiUtilsMockFns = { */ export const knowledgeApiUtilsMock = { checkKnowledgeBaseAccess: knowledgeApiUtilsMockFns.mockCheckKnowledgeBaseAccess, - checkKnowledgeBaseWriteAccess: knowledgeApiUtilsMockFns.mockCheckKnowledgeBaseWriteAccess, checkDocumentWriteAccess: knowledgeApiUtilsMockFns.mockCheckDocumentWriteAccess, checkDocumentAccess: knowledgeApiUtilsMockFns.mockCheckDocumentAccess, checkChunkAccess: knowledgeApiUtilsMockFns.mockCheckChunkAccess, diff --git a/packages/testing/src/mocks/knowledge-base-use-cases.mock.ts b/packages/testing/src/mocks/knowledge-base-use-cases.mock.ts index 8c64acc26f1..b228366d600 100644 --- a/packages/testing/src/mocks/knowledge-base-use-cases.mock.ts +++ b/packages/testing/src/mocks/knowledge-base-use-cases.mock.ts @@ -15,8 +15,6 @@ import { vi } from 'vitest' export const knowledgeBaseUseCasesMockFns = { mockListKnowledgeBasesAuthorize: vi.fn(), mockListKnowledgeBasesExecute: vi.fn(), - mockListKnowledgeBaseCatalogAuthorize: vi.fn(), - mockListKnowledgeBaseCatalogExecute: vi.fn(), mockRestoreKnowledgeBaseAuthorize: vi.fn(), mockRestoreKnowledgeBaseExecute: vi.fn(), mockCreateKnowledgeBaseAuthorize: vi.fn(), @@ -53,11 +51,6 @@ export const knowledgeBaseUseCasesMock = { authorize: fns.mockListKnowledgeBasesAuthorize, execute: fns.mockListKnowledgeBasesExecute, }, - listKnowledgeBaseCatalog: { - operation: { id: 'knowledge.list' }, - authorize: fns.mockListKnowledgeBaseCatalogAuthorize, - execute: fns.mockListKnowledgeBaseCatalogExecute, - }, restoreKnowledgeBase: { operation: { id: 'knowledge.restore' }, authorize: fns.mockRestoreKnowledgeBaseAuthorize, diff --git a/packages/testing/src/mocks/sim-search-connectors.mock.ts b/packages/testing/src/mocks/sim-search-connectors.mock.ts index cff26f3411a..f5870f53c72 100644 --- a/packages/testing/src/mocks/sim-search-connectors.mock.ts +++ b/packages/testing/src/mocks/sim-search-connectors.mock.ts @@ -45,8 +45,7 @@ function personalSetupFields(meta: MockConnectorMeta): MockConnectorConfigField[ * - `mockConnectorDisplayName` returns the connector type unchanged. * - `mockSearchMemberAccountProvider` returns `null`. * - * `mockGetConnectorAccessAvailability`, `mockSearchConnectorUnavailableReason`, and - * `mockIsSearchConnectorAvailable` are bare. + * `mockGetConnectorAccessAvailability` and `mockIsSearchConnectorAvailable` are bare. * * @example * ```ts @@ -92,7 +91,6 @@ export const simSearchConnectorsMockFns = { ), mockConnectorDisplayName: vi.fn((connectorType: string): string => connectorType), mockGetConnectorAccessAvailability: vi.fn(), - mockSearchConnectorUnavailableReason: vi.fn(), mockIsSearchConnectorAvailable: vi.fn(), } @@ -123,6 +121,5 @@ export const simSearchConnectorsMock = { missingSetupFields: simSearchConnectorsMockFns.mockMissingSetupFields, connectorDisplayName: simSearchConnectorsMockFns.mockConnectorDisplayName, getConnectorAccessAvailability: simSearchConnectorsMockFns.mockGetConnectorAccessAvailability, - searchConnectorUnavailableReason: simSearchConnectorsMockFns.mockSearchConnectorUnavailableReason, isSearchConnectorAvailable: simSearchConnectorsMockFns.mockIsSearchConnectorAvailable, }