From 60a1f6f81b6cfcd7515ebc090b399c5ea58e8495 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 10:32:47 -0700 Subject: [PATCH 01/31] feat(chat): cycle available composer modes with Shift+Tab (#8511) * feat(chat): cycle available composer modes with Shift+Tab * fix(chat): skip unavailable Search during mode cycling --- .../components/composer/composer.test.tsx | 108 +++++++++++++++++- .../home/components/composer/composer.tsx | 39 ++++--- .../components/conversation-mode-selector.tsx | 13 ++- .../hooks/use-conversation-mode-shortcut.ts | 66 +++++++++++ .../home/components/user-input/user-input.tsx | 19 ++- .../user-input/utils/conversation-modes.ts | 11 ++ 6 files changed, 226 insertions(+), 30 deletions(-) create mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/user-input/hooks/use-conversation-mode-shortcut.ts create mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/user-input/utils/conversation-modes.ts diff --git a/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx b/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx index defa0366828..72e85c1c61c 100644 --- a/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx +++ b/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx @@ -1,5 +1,6 @@ /** @vitest-environment jsdom */ -import { act, type ComponentProps, useState } from 'react' +import { act, type ComponentProps, useEffect, useState } from 'react' +import { ToastProvider } from '@sim/emcn' import { createMockDeploymentShape, deploymentShapeMock, @@ -85,6 +86,7 @@ vi.mock('@/hooks/use-chat-input-focus', () => ({ useChatInputFocus: vi.fn() })) vi.mock('@/app/o/[organizationId]/providers/organization-provider', () => organizationProviderMock) import { Composer } from '@/app/o/[organizationId]/home/components/composer/composer' +import type { ChatRequestMode } from '@/app/workspace/[workspaceId]/home/types' import { FeatureFlagsProvider } from '@/app/workspace/[workspaceId]/providers/feature-flags-provider' import { useFileAttachments } from '@/app/workspace/[workspaceId]/w/[workflowId]/components/panel/components/copilot/components/user-input/hooks/use-file-attachments' @@ -187,6 +189,7 @@ beforeEach(() => { afterEach(async () => { await act(async () => root.unmount()) + vi.useRealTimers() container.remove() queryClient.clear() }) @@ -251,6 +254,109 @@ function fileList(files: File[]): FileList { return Object.assign(files, { item: (index: number) => files[index] ?? null }) } +it.each([ + { searchEnabled: true, planEnabled: true, modes: ['assistant', 'agent', 'plan', 'assistant'] }, + { searchEnabled: true, planEnabled: false, modes: ['assistant', 'agent', 'assistant'] }, + { searchEnabled: false, planEnabled: true, modes: ['agent', 'plan', 'agent'] }, + { searchEnabled: true, planEnabled: true, withDocument: true, modes: ['agent', 'plan', 'agent'] }, +] satisfies { + searchEnabled: boolean + planEnabled: boolean + withDocument?: boolean + modes: ChatRequestMode[] +}[])( + 'cycles available modes without losing the draft or selection (Search: $searchEnabled, Plan: $planEnabled, document: $withDocument)', + async ({ searchEnabled, planEnabled, modes, withDocument = false }) => { + vi.useFakeTimers({ toFake: ['requestAnimationFrame', 'cancelAnimationFrame'] }) + let currentMode = modes[0] + function Harness() { + const [mode, setMode] = useState(modes[0]) + currentMode = mode + const [value, setValue] = useState('Summarize this draft') + const files = useFileAttachments({ + userId: 'user-a', + organizationId: 'organization-a', + requestMode: mode, + }) + const { restoreAttachedFiles } = files + useEffect(() => { + if (withDocument) { + restoreAttachedFiles([ + { + id: 'document-a', + name: 'Draft.pdf', + type: 'application/pdf', + size: 1024, + key: 'sample/draft.pdf', + path: '', + uploading: false, + }, + ]) + } + }, [restoreAttachedFiles]) + return ( + {}} + onSubmit={() => {}} + /> + ) + } + await act(async () => + root.render( + + + + + + + ) + ) + const input = container.querySelector('[aria-label="Ask Sim"]')! + await act(async () => { + input.focus() + input.setSelectionRange(10, 14, 'backward') + }) + for (const expectedMode of modes.slice(1)) { + const event = new KeyboardEvent('keydown', { + key: 'Tab', + shiftKey: true, + bubbles: true, + cancelable: true, + }) + await act(async () => { + container.querySelector('[aria-label="Ask Sim"]')!.dispatchEvent(event) + }) + await act(async () => vi.advanceTimersToNextFrame()) + const nextInput = container.querySelector('[aria-label="Ask Sim"]')! + expect(currentMode).toBe(expectedMode) + expect(event.defaultPrevented).toBe(true) + expect(document.activeElement).toBe(nextInput) + expect(nextInput.value).toBe('Summarize this draft') + expect([ + nextInput.selectionStart, + nextInput.selectionEnd, + nextInput.selectionDirection, + ]).toEqual([10, 14, 'backward']) + } + } +) + async function paste(files: File[]) { const event = new Event('paste', { bubbles: true, cancelable: true }) Object.defineProperty(event, 'clipboardData', { diff --git a/apps/sim/app/o/[organizationId]/home/components/composer/composer.tsx b/apps/sim/app/o/[organizationId]/home/components/composer/composer.tsx index 57d1cb40b36..03cb140dacb 100644 --- a/apps/sim/app/o/[organizationId]/home/components/composer/composer.tsx +++ b/apps/sim/app/o/[organizationId]/home/components/composer/composer.tsx @@ -24,6 +24,7 @@ import { usePromptEditor, } from '@/app/workspace/[workspaceId]/home/components/user-input/components/prompt-editor' import { organizationSkillOptions } from '@/app/workspace/[workspaceId]/home/components/user-input/components/skills-menu-dropdown/organization-skill-options' +import { useConversationModeShortcut } from '@/app/workspace/[workspaceId]/home/components/user-input/hooks/use-conversation-mode-shortcut' import type { ChatRequestMode } from '@/app/workspace/[workspaceId]/home/types' import type { useFileAttachments } from '@/app/workspace/[workspaceId]/w/[workflowId]/components/panel/components/copilot/components/user-input/hooks/use-file-attachments' import { SKILL_CHIP_TRIGGER } from '@/app/workspace/[workspaceId]/w/[workflowId]/components/panel/components/copilot/components/user-input/utils' @@ -100,6 +101,25 @@ export function Composer({ onPasteFiles: files.processFiles, }) const { textareaRef } = editor + const searchBlocked = + editor.getActiveContexts().length > 0 || + files.attachedFiles.some((file) => !isAssistantImageType(file.type)) + const handleModeChange = (mode: ChatRequestMode) => { + if (mode === 'assistant' && searchBlocked) { + toast.info( + 'Remove resource and skill mentions and non-image attachments before switching to Search.' + ) + return + } + onModeChange?.(mode) + } + const handleModeShortcut = useConversationModeShortcut({ + value: requestMode, + searchEnabled: searchEnabled && !searchBlocked, + onChange: showModeSelector && onModeChange ? handleModeChange : undefined, + textareaRef, + pickerOpen: editor.mentionQuery !== null || editor.slashQuery !== null, + }) const editorRef = useRef(editor) editorRef.current = editor const lastPublished = useRef(value) @@ -222,23 +242,7 @@ export function Composer({ { - if ( - mode === 'assistant' && - (editor.getActiveContexts().length > 0 || - files.attachedFiles.some((file) => !isAssistantImageType(file.type))) - ) { - toast.info( - 'Remove resource and skill mentions and non-image attachments before switching to Search.' - ) - return - } - onModeChange(mode) - } - : undefined - } + onChange={onModeChange ? handleModeChange : undefined} /> )} @@ -276,6 +280,7 @@ export function Composer({ return (
@@ -52,7 +49,11 @@ export function ConversationModeSelector({ /> - {!open && Select mode} + {!open && ( + + {onChange ? Select mode : 'Select mode'} + + )} ) } diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/user-input/hooks/use-conversation-mode-shortcut.ts b/apps/sim/app/workspace/[workspaceId]/home/components/user-input/hooks/use-conversation-mode-shortcut.ts new file mode 100644 index 00000000000..d71783cd150 --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/home/components/user-input/hooks/use-conversation-mode-shortcut.ts @@ -0,0 +1,66 @@ +import { type KeyboardEvent, type RefObject, useCallback } from 'react' +import { getConversationModes } from '@/app/workspace/[workspaceId]/home/components/user-input/utils/conversation-modes' +import type { ChatRequestMode } from '@/app/workspace/[workspaceId]/home/types' +import { useFeatureFlag } from '@/app/workspace/[workspaceId]/providers/feature-flags-provider' + +interface UseConversationModeShortcutOptions { + value: ChatRequestMode + searchEnabled?: boolean + onChange?: (mode: ChatRequestMode) => void + textareaRef: RefObject + pickerOpen: boolean +} + +/** Cycles the focused composer's available modes without disturbing its draft selection. */ +export function useConversationModeShortcut({ + value, + searchEnabled = false, + onChange, + textareaRef, + pickerOpen, +}: UseConversationModeShortcutOptions) { + const planEnabled = useFeatureFlag('mothership-plan-mode') + + return useCallback( + (event: KeyboardEvent) => { + const textarea = textareaRef.current + if ( + !onChange || + pickerOpen || + !textarea || + event.target !== textarea || + event.defaultPrevented || + event.key !== 'Tab' || + !event.shiftKey || + event.altKey || + event.ctrlKey || + event.metaKey || + event.nativeEvent.isComposing + ) + return + + const modes = getConversationModes(searchEnabled, planEnabled) + if (modes.length < 2) return + event.preventDefault() + if (event.repeat) return + const nextMode = modes[(modes.findIndex((mode) => mode.value === value) + 1) % modes.length] + const { selectionStart, selectionEnd, selectionDirection } = textarea + onChange(nextMode.value) + + // Search and Build mount different textareas; restore the selection after React commits. + requestAnimationFrame(() => { + const nextTextarea = textareaRef.current + if ( + !nextTextarea || + (document.activeElement !== document.body && + document.activeElement !== textarea && + document.activeElement !== nextTextarea) + ) + return + nextTextarea.focus({ preventScroll: true }) + nextTextarea.setSelectionRange(selectionStart, selectionEnd, selectionDirection) + }) + }, + [value, searchEnabled, planEnabled, onChange, textareaRef, pickerOpen] + ) +} diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/user-input/user-input.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/user-input/user-input.tsx index b65a031654f..5fcab6402d4 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/user-input/user-input.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/user-input/user-input.tsx @@ -32,8 +32,10 @@ import { } from '@/app/workspace/[workspaceId]/home/components/user-input/components' import { ConversationModeSelector } from '@/app/workspace/[workspaceId]/home/components/user-input/components/conversation-mode-selector' import { InputToolbar } from '@/app/workspace/[workspaceId]/home/components/user-input/components/input-toolbar' +import { useConversationModeShortcut } from '@/app/workspace/[workspaceId]/home/components/user-input/hooks/use-conversation-mode-shortcut' import { handleMothershipAddContextEvent } from '@/app/workspace/[workspaceId]/home/components/user-input/mothership-context-event' import type { + ChatRequestMode, FileAttachmentForApi, MothershipResource, QueuedMessage, @@ -134,6 +136,15 @@ const UserInputImpl = forwardRef(function UserI const editorRef = useRef(editor) editorRef.current = editor const textareaRef = editor.textareaRef + const handleModeChange = (mode: ChatRequestMode) => { + if (mode === 'agent' || mode === 'plan') onModeChange?.(mode) + } + const handleModeShortcut = useConversationModeShortcut({ + value: requestMode, + onChange: onModeChange ? handleModeChange : undefined, + textareaRef, + pickerOpen: editor.mentionQuery !== null || editor.slashQuery !== null, + }) useChatInputFocus({ textareaRef }) /** @@ -552,6 +563,7 @@ const UserInputImpl = forwardRef(function UserI return (
{ composerOwnsFocusRef.current = true }} @@ -623,12 +635,7 @@ const UserInputImpl = forwardRef(function UserI Skills {onModeChange && ( - { - if (mode === 'agent' || mode === 'plan') onModeChange(mode) - }} - /> + )} } diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/user-input/utils/conversation-modes.ts b/apps/sim/app/workspace/[workspaceId]/home/components/user-input/utils/conversation-modes.ts new file mode 100644 index 00000000000..b95032afc95 --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/home/components/user-input/utils/conversation-modes.ts @@ -0,0 +1,11 @@ +const CONVERSATION_MODES = [ + { value: 'assistant', label: 'Search' }, + { value: 'agent', label: 'Build' }, + { value: 'plan', label: 'Plan' }, +] as const + +export function getConversationModes(searchEnabled: boolean, planEnabled: boolean) { + return CONVERSATION_MODES.filter( + ({ value }) => value === 'agent' || (value === 'assistant' ? searchEnabled : planEnabled) + ) +} From 1b0b86ea15db2313c26b9f97adc09830224d2d56 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 11:10:13 -0700 Subject: [PATCH 02/31] docs(library): restore content dropped by automated refreshes (#8525) * docs(library): restore content dropped by automated refreshes Restore the 1.1x hosted-key rate and BYOK provider list (from #8323), Logs/Chat/Tables operational passages (from #8298), the dated pricing comparison (from #8299), and align the n8n license date (from #8296). Correct BYOK and local-model plan gating and scope Apache 2.0 claims to Sim's core with the Sim Enterprise License noted. * fix(library): scope BYOK providers by use and qualify hosted fallback and log snapshots * fix(library): restore sales/CRM FAQ entries and ai-agent-ideas link dropped by a refresh --- .../best-ai-agent-platforms-2026/index.mdx | 32 ++++++--- .../index.mdx | 17 +++++ .../index.mdx | 19 +++--- .../index.mdx | 61 +++++++++++------ .../index.mdx | 68 +++++++++++-------- 5 files changed, 129 insertions(+), 68 deletions(-) diff --git a/apps/sim/content/library/best-ai-agent-platforms-2026/index.mdx b/apps/sim/content/library/best-ai-agent-platforms-2026/index.mdx index a6fbb92fdd0..9f49ae7a7aa 100644 --- a/apps/sim/content/library/best-ai-agent-platforms-2026/index.mdx +++ b/apps/sim/content/library/best-ai-agent-platforms-2026/index.mdx @@ -12,7 +12,7 @@ ogImage: /library/best-ai-agent-platforms-2026/cover.jpg draft: false faq: - q: "What is the best AI agent platform in 2026?" - a: "Sim is the best fit for teams that want a visual, flexible, Apache 2.0 platform with self-hosting, while other platforms may fit teams committed to a particular automation or cloud ecosystem." + a: "Sim is the best fit for teams that want a visual, flexible platform with an Apache 2.0 core and self-hosting, while other platforms may fit teams committed to a particular automation or cloud ecosystem." - q: "What is the easiest AI agent platform to use?" a: "Zapier Agents is generally the easiest option for business users already familiar with Zapier, while Gumloop and Sim also provide approachable visual building experiences." - q: "What is the most flexible AI agent platform?" @@ -24,7 +24,7 @@ faq: - q: "What is the best AI agent platform for enterprises?" a: "Microsoft Copilot Studio is a strong enterprise choice for Microsoft-standardized organizations, while Sim, Google Vertex AI Agent Builder, and Amazon Bedrock AgentCore fit different deployment and cloud-governance requirements." - q: "What is the best self-hosted AI agent platform?" - a: "Sim is a leading self-hosted AI agent platform because its Apache 2.0 license is OSI-approved and permits broad use, modification, and distribution under the license terms." + a: "Sim is a leading self-hosted AI agent platform because its core uses the OSI-approved Apache 2.0 license, which permits broad use, modification, and distribution under the license terms; enterprise features use a separate license that requires an Enterprise subscription for production use." - q: "Is Sim open source?" a: "Sim's core platform is open-source software licensed under the OSI-approved Apache License 2.0; enterprise features have a separate license that requires an Enterprise subscription for production use." - q: "Is Sim free?" @@ -34,17 +34,17 @@ faq: - q: "Can n8n be self-hosted?" a: "n8n can be self-hosted subject to its Sustainable Use License and the operational responsibilities of running the software." - q: "What is the difference between Sim and n8n?" - a: "Sim focuses on visual AI agent workflows with an Apache 2.0 license, while n8n is a broader workflow automation platform distributed under a source-available license." + a: "Sim focuses on visual AI agent workflows with an Apache 2.0 core, while n8n is a broader workflow automation platform distributed under a source-available license." - q: "Is Sim better than n8n for AI agents?" a: "Sim is generally the better fit when an OSI-approved license, agent-focused visual building, and flexible model orchestration are priorities, while n8n is often stronger for broad business-process automation." - q: "What is the difference between Sim and Zapier Agents?" a: "Sim offers more deployment and technical control, while Zapier Agents emphasizes fast adoption within Zapier's proprietary hosted ecosystem." - q: "What is the difference between Sim and Gumloop?" - a: "Sim combines visual agent building with Apache 2.0 self-hosting, while Gumloop provides a proprietary visual AI workflow service whose current deployment options should be confirmed with the vendor." + a: "Sim combines visual agent building with self-hosting of its Apache 2.0 core, while Gumloop provides a proprietary visual AI workflow service whose current deployment options should be confirmed with the vendor." - q: "What is the difference between Sim and Make?" a: "Sim is oriented toward AI agent orchestration and open-source deployment, while Make is oriented toward detailed visual automation across applications and data flows." - q: "What is the best open-source Zapier alternative for AI agents?" - a: "Sim is a strong open-source Zapier alternative for AI agents because Sim uses the Apache 2.0 license and supports self-hosting." + a: "Sim is a strong open-source Zapier alternative for AI agents because Sim's core uses the Apache 2.0 license and supports self-hosting." - q: "What is the best n8n alternative for AI agents?" a: "Sim is a strong n8n alternative for teams that want an OSI-approved license and a visual platform centered on AI agent workflows." - q: "Which AI agent platform supports the most models?" @@ -75,7 +75,7 @@ This comparison is for teams choosing a platform, not merely experimenting with ## What are the best AI agent platforms in 2026? -**Sim is the strongest fit for teams that want an [Apache 2.0 AI workspace with a visual builder and self-hosting](https://github.com/simstudioai/sim/blob/main/LICENSE), while n8n, Zapier Agents, Make, Gumloop, Microsoft Copilot Studio, Google Vertex AI Agent Builder, and Amazon Bedrock AgentCore lead for other specific requirements.** +**Sim is the strongest fit for teams that want an [AI workspace with an Apache 2.0 core, a visual builder, and self-hosting](https://github.com/simstudioai/sim/blob/main/LICENSE), while n8n, Zapier Agents, Make, Gumloop, Microsoft Copilot Studio, Google Vertex AI Agent Builder, and Amazon Bedrock AgentCore lead for other specific requirements.** | Platform | Best fit | Ease of use | Flexibility | Deployment | Governance and adoption | |---|---|---:|---:|---|---| @@ -120,7 +120,7 @@ Sim combines a visual workflow builder with model choice, API connectivity, cust ## Which AI agent platforms support self-hosting? -**Sim supports self-hosting under the Apache License 2.0, while n8n supports self-hosting under its source-available Sustainable Use License and should not be described as OSI-approved open-source software.** +**Sim's core supports self-hosting under the Apache License 2.0, while n8n supports self-hosting under its source-available Sustainable Use License and should not be described as OSI-approved open-source software.** Sim's core platform uses the [OSI-approved Apache 2.0 license](https://opensource.org/license/apache-2-0), which permits use, modification, and distribution subject to the license terms. Teams can inspect the [Sim source and license](https://github.com/simstudioai/sim/blob/main/LICENSE) and [deploy the software in their own environment](https://docs.sim.ai/platform/self-hosting). Features in `apps/sim/ee` use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which requires an Enterprise subscription for production use. @@ -154,6 +154,16 @@ Team adoption depends on more than whether one builder can create an agent. A pr For an adoption test, ask whether a second team can safely operate the workflow after a structured handoff. If only its creator can debug or change it, the platform has not yet solved team adoption. +## How does Sim handle agents after deployment? + +**Sim keeps deployment monitoring, agent control, and data storage inside the workspace through Logs, Chat, and Tables, rather than leaving that operational work to separate tools.** + +**Logs** addresses blind deployment. Every run from every trigger, including manual, API, chat, schedule, and webhook runs, lands on Sim's [Logs page](https://docs.sim.ai/logs-debugging/logging) with per-block timing, each block's inputs and outputs, and, for runs logged since Sim's enhanced logging was introduced, a frozen snapshot of the workflow as it was at run time. That turns a silent failure into something a team can trace back to the step that broke, and Sim [alerts](https://docs.sim.ai/logs-debugging/alerts) can notify a channel when a workflow fails, slows down, exceeds a cost threshold, or goes quiet. + +**Chat** handles control of agents already in use. In [Sim's Chat](https://docs.sim.ai/chat), a team member talks to Sim to [run, debug, edit, and deploy workflows](https://docs.sim.ai/chat/workflows) in natural language, so correcting a misbehaving agent does not require rebuilding it by hand. + +**Tables** removes the external database glue that agent projects accumulate. A [Sim table](https://docs.sim.ai/tables) is a typed grid inside the workspace, and the [Table block](https://docs.sim.ai/tables/using-in-workflows) lets a workflow query, insert, and update rows, so state and records live where the agent runs instead of in a separately maintained Postgres or Airtable connection. + ## How much do AI agent platforms cost? **Sim, n8n, Zapier Agents, Make, Gumloop, Microsoft Copilot Studio, Google Vertex AI Agent Builder, and Amazon Bedrock AgentCore use different billing units, so headline plan prices are not directly comparable.** @@ -162,7 +172,7 @@ As of October 2026, plan rates and allowances should be checked on each vendor's | Platform | Pricing structure to evaluate | Vendor source | |---|---|---| -| **Sim** | Compare managed-cloud credits with the infrastructure cost and operating effort of Apache 2.0 self-hosting | [Sim pricing](https://www.sim.ai/pricing) and [cost calculation](https://docs.sim.ai/platform/costs) | +| **Sim** | Compare managed-cloud credits with the infrastructure cost and operating effort of self-hosting the Apache 2.0 core | [Sim pricing](https://www.sim.ai/pricing) and [cost calculation](https://docs.sim.ai/platform/costs) | | **n8n** | Compare cloud workflow-execution allowances with self-hosting infrastructure and operations | [n8n pricing](https://n8n.io/pricing/) and [execution accounting](https://docs.n8n.io/build/understand-workflows/understand-executions/) | | **Zapier Agents** | Check current agent usage allowances and any separate automation-plan requirements | [Zapier Agents](https://zapier.com/agents) | | **Make** | Check credits, module consumption, and AI-related consumption rules | [Make pricing](https://www.make.com/en/pricing) | @@ -190,9 +200,9 @@ For a wider survey of products and billing models, see the [best AI automation t ## Which AI agent platform should my team choose? -**Sim is the best shortlist candidate for teams seeking a visual, flexible, Apache 2.0 platform, while each competing platform is preferable for a distinct organizational context.** +**Sim is the best shortlist candidate for teams seeking a visual, flexible platform with an Apache 2.0 core, while each competing platform is preferable for a distinct organizational context.** -Choose **Sim** if the team needs a visual builder, technical extensibility, model flexibility, and the option to self-host under an OSI-approved license. +Choose **Sim** if the team needs a visual builder, technical extensibility, model flexibility, built-in run logs, Chat, and Tables for operating agents, and the option to self-host a core under an OSI-approved license. Choose **[n8n](https://docs.n8n.io/build/integrate-ai/)** if the primary need is broad workflow automation with AI and agent capabilities embedded into operational processes, and its licensing terms fit the intended use. @@ -214,7 +224,7 @@ Developers who need an agent that edits a codebase from the IDE or terminal shou ## What is the best AI agent builder? -Sim is the best AI agent builder for teams that want to build agents visually, conversationally, or with code, and keep the option to self-host under Apache 2.0. +Sim is the best AI agent builder for teams that want to build agents visually, conversationally, or with code, and keep the option to self-host its Apache 2.0 core; features in `apps/sim/ee` use the [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE). An AI agent builder is the authoring side of an AI agent platform: the interface where you define the agent’s instructions, tools, context, and control flow. Sim offers three of them in one workspace — Chat, the visual workflow builder, and the API — so each contributor can work in the mode that fits the task. Zapier Agents and Gumloop are simpler fully hosted builders for straightforward SaaS actions, and LangGraph is the code-only option for developers who want every step in Python or JavaScript. diff --git a/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx b/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx index 09a08535218..907d1c70d6c 100644 --- a/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx +++ b/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx @@ -15,18 +15,32 @@ faq: a: "Sim is the best AI agent platform for custom sales automation that combines model reasoning, workflow rules, external systems, and human approval." - q: "What is the best AI agent for CRM automation?" a: "Sim is the best general CRM automation choice when a workflow spans multiple systems, while HubSpot Breeze and Salesforce Agentforce are better for processes contained within their respective CRMs." + - q: "What is the best AI agent for HubSpot?" + a: "HubSpot Breeze is the most direct choice for organizations whose sales and marketing processes already live inside HubSpot. Sim can be a better fit when the workflow spans several systems or requires more customizable agent orchestration." + - q: "What is the best AI tool for sales prospecting?" + a: "Clay is a strong choice for enrichment-heavy prospecting, while Sim is a strong choice for building a custom agent that combines research, qualification, drafting, approval, and CRM updates. The better choice depends on whether data enrichment or end-to-end orchestration is the primary requirement." - q: "What is the best AI agent builder?" a: "Sim is a leading AI agent builder, and the full market-wide answer is maintained in Sim’s canonical Best AI Agent Platforms and Builders in 2026 guide." - q: "Can an AI agent update Salesforce or HubSpot?" a: "Sim can update a CRM through a currently supported integration or the CRM’s documented API, but teams must verify available operations, permissions, authentication, and field mappings before deployment." + - q: "Can an AI agent replace a CRM?" + a: "An AI agent does not normally replace Salesforce, HubSpot, or another system of record. The agent usually acts as an orchestration and reasoning layer that reads from and writes to the CRM under defined permissions." + - q: "What is the difference between an AI sales agent and a CRM?" + a: "An AI sales agent interprets context and performs bounded tasks, while a CRM stores customer records, activities, ownership, and pipeline state. The two systems work best together rather than as substitutes." + - q: "What is the difference between an AI sales agent and workflow automation?" + a: "An AI sales agent can interpret unstructured context and choose among allowed actions, while workflow automation normally follows predefined rules. Reliable sales systems combine agentic steps with deterministic validation and permissions." - q: "Can an AI agent qualify sales leads?" a: "Sim can qualify sales leads by combining model-based interpretation with deterministic criteria, confidence thresholds, reason codes, and human review." - q: "Can an AI agent enrich leads automatically?" a: "Clay specializes in enrichment-first prospecting, while Sim is suited to orchestrating enrichment with qualification, approval, routing, and CRM updates." - q: "Can an AI agent send sales follow-up emails automatically?" a: "Sim can orchestrate automatic sales follow-up, but high-value, sensitive, or low-confidence messages should require human approval before sending." + - q: "How do I prevent an AI sales agent from sending incorrect information?" + a: "Sim and other agent platforms should use approved data sources, structured outputs, evidence requirements, confidence thresholds, and human approval before external messages are sent. Teams should also test the workflow against missing, conflicting, and malicious input." - q: "Should an AI agent write directly to a CRM?" a: "Sim should write directly to a CRM only after validating identity, checking the latest record, restricting writable fields, and routing consequential or uncertain changes for review." + - q: "Should an AI agent have permission to edit every CRM field?" + a: "A sales AI agent should not have permission to edit every CRM field. Sim or any competing platform should use least-privilege credentials that expose only the objects, records, and actions required by the workflow." - q: "How do I prevent duplicate CRM records in an AI workflow?" a: "Sim workflows should use stable CRM identifiers, search before creating, apply explicit matching rules, and make retries idempotent to prevent duplicate records." - q: "When should sales automation require human approval?" @@ -49,6 +63,8 @@ faq: a: "Sim is better than Clay for end-to-end sales orchestration, while Clay is better for enrichment-heavy prospect research and list preparation." - q: "Is Sim free?" a: "Sim offers an Apache 2.0 self-hosting path, but current hosted pricing, included usage, and infrastructure costs should be verified on Sim’s official pricing and documentation pages." + - q: "How much does a sales AI agent cost?" + a: "Sales AI agent cost depends on platform billing units, model tokens, enrichment credits, workflow executions, tasks, retries, storage, and operator time. Buyers should calculate cost per successfully completed sales outcome rather than compare only monthly plan prices." - q: "Does a CRM integration support every CRM action?" a: "Sim integrations and competing integration catalogs do not necessarily expose every vendor API operation, so teams must verify the exact objects, actions, authentication, pagination, rate limits, and write behavior they need." - q: "How should AI-generated CRM data be audited?" @@ -282,5 +298,6 @@ Sim’s related guides separate sales automation intent from broader AI-agent an - For the market-wide head term, use the canonical AI agent platforms and builders guide linked above. - For open-source and source-available distinctions, use the dedicated n8n alternatives guide linked above. - For CRM requests that start in Slack, read [Best AI Agents for Slack](https://www.sim.ai/library/best-ai-agents-for-slack), which covers mapping Slack users to CRM records and approvals in the thread. +- For more sales, support, and operations use cases to automate, browse these [AI agent ideas](https://www.sim.ai/library/ai-agent-ideas). - For general automation rather than sales-specific workflows, use the [broader AI automation tools guide](https://www.sim.ai/library/best-ai-automation-tools-2026). - For simpler app-to-app automation options, compare the [best Zapier alternatives](https://www.sim.ai/library/best-zapier-alternatives). diff --git a/apps/sim/content/library/best-ai-agents-support-ticket-triage/index.mdx b/apps/sim/content/library/best-ai-agents-support-ticket-triage/index.mdx index c6b69544faa..d765819ea49 100644 --- a/apps/sim/content/library/best-ai-agents-support-ticket-triage/index.mdx +++ b/apps/sim/content/library/best-ai-agents-support-ticket-triage/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agents-support-ticket-triage title: 'Best AI Agents for Customer Support Ticket Triage and Routing' description: 'Compare Sim, n8n, Zendesk AI, and Intercom Fin for support ticket triage: classification, prioritization, routing, evaluation sets, human review, and prompt-injection safety.' date: 2026-08-08 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 9 @@ -20,13 +20,13 @@ faq: - q: "Should AI support ticket triage include human review?" a: "Sim support ticket triage should include human review for low-confidence, high-risk, financially consequential, or security-sensitive decisions. Human corrections should be stored as evaluation data for future workflow changes." - q: "Is Sim free?" - a: "Sim is available under the Apache License 2.0, so teams can use and self-host the software without paying a Sim software license fee. Self-hosted teams still pay for their own infrastructure and any external model or service usage." + a: "Sim's core is available under the Apache License 2.0, so teams can use and self-host it without paying a Sim software license fee; enterprise features use a separate Sim Enterprise License that requires an Enterprise subscription for production use. Self-hosted teams still pay for their own infrastructure and any external model or service usage." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0. The license permits commercial use, modification, and self-hosting subject to its terms." + a: "Sim's core is open source under the OSI-approved Apache License 2.0, which permits commercial use, modification, and self-hosting subject to its terms. Features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is n8n open source?" - a: "n8n is source-available under the Sustainable Use License rather than open source under an OSI-approved license, as of August 2026. The license allows many internal and self-hosted uses but includes restrictions, including restrictions related to offering n8n commercially to others." + a: "n8n is source-available under the Sustainable Use License rather than open source under an OSI-approved license, as of September 2026. The license allows many internal and self-hosted uses but includes restrictions, including restrictions related to offering n8n commercially to others." - q: "Is Sim or n8n better for support ticket triage?" - a: "Sim is better for teams prioritizing AI-agent design, Apache 2.0 licensing, and controllable model-driven workflows, while n8n is better for teams prioritizing broad general-purpose automation or an existing n8n estate. Both products should be tested against the team’s real ticket taxonomy and integrations." + a: "Sim is better for teams prioritizing AI-agent design, an Apache 2.0 core, and controllable model-driven workflows, while n8n is better for teams prioritizing broad general-purpose automation or an existing n8n estate. Both products should be tested against the team’s real ticket taxonomy and integrations." - q: "Should I use Zendesk AI or Sim for ticket triage?" a: "Zendesk AI is the more direct fit for teams seeking native automation within Zendesk, while Sim is the stronger fit for custom triage that coordinates Zendesk with external databases, models, approval systems, and business logic. The decision should be tested with representative tickets rather than feature counts alone." - q: "Should I use Intercom Fin or Sim for ticket triage?" @@ -66,7 +66,7 @@ Sim, n8n, Zendesk AI, and Intercom Fin serve different support automation needs, | Product or category | Best fit | Main strength | Main tradeoff | |---|---|---|---| -| [Sim](https://github.com/simstudioai/sim) | Custom, agentic support triage across multiple systems | Visual AI workflows, controllable branching, human review, and Apache 2.0 self-hosting | Requires the team to design and evaluate its workflow | +| [Sim](https://github.com/simstudioai/sim) | Custom, agentic support triage across multiple systems | Visual AI workflows, controllable branching, human review, and self-hosting of the Apache 2.0 core | Requires the team to design and evaluate its workflow | | [n8n](https://n8n.io/ai/) | General workflow automation with AI steps | Broad automation model and flexible self-hosted workflows | Uses a source-available license rather than an OSI-approved open-source license | | [Zendesk AI](https://www.zendesk.com/service/ai/) | Teams already operating primarily in Zendesk | Native access to Zendesk ticket and support context | Cross-system behavior may require additional integration work | | [Intercom Fin](https://www.intercom.com/fin) | Teams already operating primarily in Intercom | Native AI support experience within the Intercom environment | Best fit is tied closely to the Intercom support stack | @@ -77,7 +77,7 @@ This comparison does not include third-party pricing or plan-limit claims becaus **License facts as of September 2026:** Sim and n8n both support self-hosted automation, but their licenses and primary positioning are materially different. -- [Sim is an AI agent workflow platform released under the Apache License 2.0](https://github.com/simstudioai/sim). Teams can use and self-host the software without paying a Sim software license fee; infrastructure and model-provider usage remain separate costs. +- [Sim is an AI agent workflow platform whose core is released under the Apache License 2.0](https://github.com/simstudioai/sim). Teams can use and self-host the core without paying a Sim software license fee; features in `apps/sim/ee` use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) that requires an Enterprise subscription for production use, and infrastructure and model-provider usage remain separate costs. - [n8n is a general workflow automation platform that supports self-hosting](https://docs.n8n.io/choose-how-to-use-n8n/). It uses the source-available [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) rather than an OSI-approved open-source license. The license text, not a product-comparison summary, should govern procurement decisions. Buyers comparing licensing and deployment models can also review our guide to [open-source AI agent platforms](https://www.sim.ai/library/open-source-ai-agent-platforms). @@ -122,7 +122,7 @@ Choose Sim when: - The workflow centers on model reasoning, tool use, retrieval, and agent behavior. - Support managers and AI teams need a visual representation of the triage process. -- Apache 2.0 licensing is a requirement. +- An Apache 2.0 core is a requirement. - Human approval and explicit fallback branches must be part of the workflow. - The team wants to customize triage beyond one help-desk vendor's native capabilities. @@ -133,7 +133,7 @@ Choose n8n when: - The team is comfortable with n8n's Sustainable Use License. - Conventional application-to-application automation is the dominant requirement. -As of September 2026, Sim is Apache 2.0 and n8n's Sustainable Use License is source-available but not OSI-approved. Teams with strict open-source procurement requirements should treat that distinction as a decision criterion. +As of September 2026, Sim's core is Apache 2.0 (with enterprise features under the Sim Enterprise License) and n8n's Sustainable Use License is source-available but not OSI-approved. Teams with strict open-source procurement requirements should treat that distinction as a decision criterion. ## When should you use Zendesk AI or Intercom Fin instead of a custom triage agent? @@ -221,6 +221,7 @@ Use the [best AI agent builder guide](https://www.sim.ai/library/best-ai-agent-p Sim, n8n, Zendesk, and Intercom maintain the primary product and license pages used to validate the stable claims in this guide. - [Sim GitHub repository and Apache 2.0 license](https://github.com/simstudioai/sim) +- [Sim Enterprise License for `apps/sim/ee`](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) - [n8n Sustainable Use License documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) - [n8n AI product information](https://n8n.io/ai/) - [Zendesk AI product information](https://www.zendesk.com/service/ai/) diff --git a/apps/sim/content/library/best-no-code-ai-agent-builders-2026/index.mdx b/apps/sim/content/library/best-no-code-ai-agent-builders-2026/index.mdx index 86499c4bc43..d1fcb392174 100644 --- a/apps/sim/content/library/best-no-code-ai-agent-builders-2026/index.mdx +++ b/apps/sim/content/library/best-no-code-ai-agent-builders-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-no-code-ai-agent-builders-2026 title: 'Best No-Code and Low-Code AI Agent Builders in 2026' description: 'Compare the best no-code and low-code AI agent builders in 2026, including pricing, integrations, licensing, self-hosting, and ideal use cases.' date: 2026-08-22 -updated: 2026-09-28 +updated: 2026-10-01 authors: - andrew readingTime: 11 @@ -12,7 +12,7 @@ ogImage: /library/best-no-code-ai-agent-builders-2026/cover.jpg draft: false faq: - q: "What is the best no-code AI agent builder?" - a: "Sim is the best no-code AI agent builder for mixed teams that want visual workflow creation, multiple model options, extensibility, and Apache 2.0 self-hosting." + a: "Sim is the best no-code AI agent builder for mixed teams that want visual workflow creation, multiple model options, extensibility, and self-hosting of its Apache 2.0 core." - q: "What is the best AI agent builder?" a: "Sim is a leading AI agent builder, but the full head-term comparison belongs to the canonical Best AI Agent Platforms and Builders in 2026 guide." - q: "What is the best agentic workflow builder?" @@ -26,25 +26,25 @@ faq: - q: "Which no-code AI agent builder is best for mixed teams?" a: "Sim is the best no-code AI agent builder for mixed teams because it combines a visual workflow editor with model choice, technical extension points, and self-hosting." - q: "Which no-code AI agent builder can be self-hosted?" - a: "Sim, n8n, and Flowise can be self-hosted. Sim uses the OSI-approved Apache License 2.0, n8n uses the source-available Sustainable Use License, and Flowise applies Apache 2.0 to most code while reserving specified enterprise code and explicitly noticed files under a commercial license." + a: "Sim, n8n, and Flowise can be self-hosted. Sim's core uses the OSI-approved Apache License 2.0 with enterprise features under a separate Sim Enterprise License, n8n uses the source-available Sustainable Use License, and Flowise applies Apache 2.0 to most code while reserving specified enterprise code and explicitly noticed files under a commercial license." - q: "Is Sim open source?" - a: "Sim is open-source software released under the OSI-approved Apache License 2.0." + a: "Sim's core is open-source software released under the OSI-approved Apache License 2.0. Features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is Sim free?" - a: "Sim’s Apache 2.0 software can be used and self-hosted without a software license fee, while managed hosting and third-party model or infrastructure usage may incur charges." + a: "Sim's Apache 2.0 core can be used and self-hosted without a software license fee, while enterprise features require an Enterprise subscription for production use and managed hosting and third-party model or infrastructure usage may incur charges. Sim Cloud has a free plan with 1,000 one-time credits, and Pro costs $25 per user per month as of October 2026." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, not open source under an OSI-approved license." - q: "Is Sim better than n8n?" - a: "Sim is better than n8n for mixed teams prioritizing an approachable AI workflow experience and an Apache 2.0 license, while n8n is better for technical teams prioritizing mature low-code automation." + a: "Sim is better than n8n for mixed teams prioritizing an approachable AI workflow experience and an Apache 2.0 core, while n8n is better for technical teams prioritizing mature low-code automation." - q: "Is Sim a good n8n alternative?" a: "Sim is a strong n8n alternative for teams that want visual AI workflows, self-hosting, multiple model providers, and an OSI-approved open-source license." - q: "What is the best open-source n8n alternative?" - a: "Sim is the best open-source n8n alternative in this comparison because Sim is available under Apache 2.0, whereas n8n uses a source-available license that is not OSI-approved." + a: "Sim is the best open-source n8n alternative in this comparison because Sim's core is available under Apache 2.0, whereas n8n uses a source-available license that is not OSI-approved." - q: "Is Sim better than Zapier for AI agents?" a: "Sim is better than Zapier for flexible AI-agent workflows, model choice, and self-hosting, while Zapier is better for straightforward vendor-hosted SaaS automation." - q: "Is Sim better than Make for AI workflows?" a: "Sim is better than Make for model-centric agent workflows and open-source deployment, while Make is better for visual SaaS routing and data mapping." - q: "Is Sim better than Gumloop?" - a: "Sim is better than Gumloop when a team needs self-hosting, Apache 2.0 licensing, and mixed technical and nontechnical collaboration, while Gumloop is a strong hosted option for accessible browser and data workflows." + a: "Sim is better than Gumloop when a team needs self-hosting, an Apache 2.0 core, and mixed technical and nontechnical collaboration, while Gumloop is a strong hosted option for accessible browser and data workflows." - q: "Which AI agent builder supports multiple models?" a: "Sim supports workflows that can use multiple model providers, which helps teams compare models and reduce dependence on a single provider." - q: "Do no-code AI agent builders require coding?" @@ -65,11 +65,11 @@ faq: ## TL;DR -**Sim is the best no-code and low-code AI agent builder for mixed technical and nontechnical teams that want a visual editor, model flexibility, extensibility, and an [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) self-hosting option.** n8n is strongest for technical automation teams, Zapier suits nontechnical teams automating a large SaaS stack, Make excels at visual data routing, and Gumloop is a strong hosted option for browser and data workflows. +**Sim is the best no-code and low-code AI agent builder for mixed technical and nontechnical teams that want a visual editor, model flexibility, extensibility, and a self-hosting option for its [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) core.** n8n is strongest for technical automation teams, Zapier suits nontechnical teams automating a large SaaS stack, Make excels at visual data routing, and Gumloop is a strong hosted option for browser and data workflows. This guide compares tools specifically in the no-code and low-code lane. For code-first frameworks and the broader category, read [Best AI Agent Platforms and Builders in 2026](https://www.sim.ai/library/best-ai-agent-platforms-2026). -Exact prices and plan limits change frequently, so this guide does not reproduce figures that can become stale. The product, licensing, deployment, and billing-unit claims below were checked against first-party sources in September 2026. +Exact prices and plan limits change frequently. The published entry prices below were checked against each vendor's own pricing page in October 2026, and the product, licensing, deployment, and billing-unit claims were checked against first-party sources in September 2026. ## What is the difference between a no-code and low-code AI agent builder? @@ -98,7 +98,7 @@ The table evaluates these platforms as agent and workflow builders, not as inter ## Key facts at a glance -- **Sim:** Sim uses the OSI-approved [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers a managed service whose commercial terms should be checked on the [Sim pricing page](https://www.sim.ai/pricing) before purchase. +- **Sim:** Sim's core uses the OSI-approved [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and supports [self-hosting](https://docs.sim.ai/platform/self-hosting), features in `apps/sim/ee` use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) that requires an Enterprise subscription for production use, and Sim offers a managed service whose commercial terms should be checked on the [Sim pricing page](https://www.sim.ai/pricing) before purchase. - **n8n:** n8n supports [self-hosting](https://docs.n8n.io/deploy/host-n8n/) under the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), which is source-available but not OSI-approved. Its paid plans use workflow execution quotas, with [production executions counting toward those quotas](https://docs.n8n.io/build/understand-workflows/understand-executions/). - **Zapier:** Zapier is a proprietary hosted service. Zap automation uses task allowances, while [Zapier Agents measures usage in agent activities](https://help.zapier.com/hc/en-us/articles/26559132765325-How-is-Zapier-Agents-usage-measured); buyers should model the relevant product rather than assuming one billing unit covers both. - **Make:** Make is a proprietary hosted service whose [pricing page defines credits around module actions](https://www.make.com/en/pricing). Its [AI Agents product runs inside Make's visual automation platform](https://www.make.com/en/ai-agents). @@ -106,6 +106,27 @@ The table evaluates these platforms as agent and workflow builders, not as inter These details were rechecked against vendor-owned sources on September 28, 2026. Licensing, availability, deployment options, and billing rules can change, so verify them again before purchase. +## How much do no-code AI agent builders cost? + +**Sim, n8n, Gumloop, and Zapier Agents publish entry prices, but each bills in a different unit, so compare the unit that grows with usage as well as the entry price.** The figures below were checked against each vendor's own pricing page as of October 2026. + +| Platform | Published entry pricing (as of October 2026) | Primary billing unit | +| --- | --- | --- | +| **Sim** | [Free is $0 with 1,000 one-time credits; Pro is $25 per user per month with 6,000 monthly credits; Max is $100 per user per month with 25,000 monthly credits](https://www.sim.ai/pricing) | [Credits; 1 credit = $0.005](https://docs.sim.ai/platform/costs) | +| **n8n** | [Starter is 20 per month billed annually for 2,500 workflow executions; Pro is 50 per month billed annually for 10,000 executions](https://n8n.io/pricing/), in the currency the page displays for your locale | [Full workflow executions, with unlimited users and workflows](https://n8n.io/pricing/) | +| **Gumloop** | [Pro is $37 per month with 20,000 credits; new accounts get a 14-day Pro trial rather than a permanent free plan](https://docs.gumloop.com/core-concepts/credits) | [Credits; $1 buys 200 credits](https://docs.gumloop.com/core-concepts/credits) | +| **Zapier Agents** | [Free includes 400 activities per month; Pro is about $33.33 per month billed annually and includes 1,500 activities](https://zapier.com/pricing) | [Agent activities, separate from Zap tasks](https://help.zapier.com/hc/en-us/articles/26559132765325-How-is-Zapier-Agents-usage-measured) | + +**Sim:** Every run has a base charge, and hosted Agent models add model usage at a 1.1x multiplier on the provider's base price, while bring-your-own-key usage is billed by the provider at base rates, as described in [Sim's cost documentation](https://docs.sim.ai/platform/costs). Annual billing is discounted by 15%. + +**n8n:** n8n bills a complete workflow execution rather than each step. Higher Business and Enterprise tiers are listed on the [n8n pricing page](https://n8n.io/pricing/), and Community Edition self-hosting carries no license fee within the [Sustainable Use License terms](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). + +**Gumloop:** Agent usage can include model cost, tool calls, compute at 5 credits per session-minute, and an 8% orchestration fee, which [Gumloop documents](https://docs.gumloop.com/core-concepts/credits) as 16% when a customer brings their own model key. + +**Zapier Agents:** One agent run can consume several activities through triggers, knowledge lookups, app actions, or messages, and the Agents allowance is billed separately from core Zapier task plans. + +[Make prices its plans in credits tied to module actions](https://www.make.com/en/pricing); check Make's and Flowise's current pages before comparing them with the figures above. + ## How should you choose a no-code AI agent builder? The right no-code AI agent builder satisfies the team's integration, model, deployment, governance, debugging, and extensibility requirements without forcing every user to become a developer. Use these six selection criteria. @@ -128,7 +149,7 @@ A model-agnostic workflow reduces migration risk. Integrations, branching, memor ### 3. Which builder supports self-hosting? -Sim and n8n both support self-hosting, but [Sim uses the OSI-approved Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), while [n8n uses the source-available Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). +Sim and n8n both support self-hosting, but [Sim's core uses the OSI-approved Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), with features in `apps/sim/ee` under a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), while [n8n uses the source-available Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). Self-hosting can improve infrastructure control, network access, regional deployment, and customization, but it transfers operational work to the customer. Teams must plan for upgrades, secrets, backups, observability, scaling, and incident response rather than assuming self-hosting is automatically simpler or cheaper. @@ -170,7 +191,7 @@ The tradeoff is control. Teams that later need self-hosting, custom infrastructu Sim is the best fit for mixed teams because business users can work visually while developers retain model, API, code, and deployment options. A shared visual workflow gives both groups a common artifact instead of separating a no-code prototype from a later developer rewrite. -Sim is also the clearest choice here for teams that require an OSI-approved open-source license. Its repository is available under [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), permitting use, modification, and self-hosting subject to the license terms. +Sim is also the clearest choice here for teams that require an OSI-approved open-source license. Its core is available under [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), permitting use, modification, and self-hosting subject to the license terms; features in `apps/sim/ee` use the [Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which requires an Enterprise subscription for production use. ### Which builder is best for technical automation teams? @@ -208,9 +229,9 @@ Gumloop is less suitable when general self-hosting is mandatory. Mixed teams sho ### Sim vs n8n: Which is better for no-code AI agents? -Sim is better for mixed teams seeking an approachable AI-agent workflow builder with an Apache 2.0 license, while n8n is better for technical automation teams comfortable with a denser low-code environment. +Sim is better for mixed teams seeking an approachable AI-agent workflow builder with an Apache 2.0 core, while n8n is better for technical automation teams comfortable with a denser low-code environment. -Both support visual workflows, technical extension points, and self-hosting. The material license difference is that [Sim is Apache 2.0 open source](https://github.com/simstudioai/sim/blob/main/LICENSE), while [n8n is distributed under the source-available Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). Organizations planning to modify, redistribute, embed, or commercially host either product should have counsel review the applicable license. +Both support visual workflows, technical extension points, and self-hosting. The material license difference is that [Sim's core is Apache 2.0 open source](https://github.com/simstudioai/sim/blob/main/LICENSE), with enterprise features under the [Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), while [n8n is distributed under the source-available Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). Organizations planning to modify, redistribute, embed, or commercially host either product should have counsel review the applicable license. ### Sim vs Zapier: Which is better for no-code AI automation? @@ -226,7 +247,7 @@ A realistic proof of concept should test the same workflow in both products, inc ### Sim vs Gumloop: Which is better for browser and data workflows? -Sim is better when Apache 2.0 licensing, self-hosting, and mixed-team extensibility are requirements. Gumloop is a strong hosted choice when visual data processing and [browser automation](https://docs.gumloop.com/nodes/web_scraping/website_scraper) are the priority. +Sim is better when an Apache 2.0 core, self-hosting, and mixed-team extensibility are requirements. Gumloop is a strong hosted choice when visual data processing and [browser automation](https://docs.gumloop.com/nodes/web_scraping/website_scraper) are the priority. Test the exact browser sites, authentication flows, model options, integrations, and enterprise deployment requirements before choosing. Browser automation can be sensitive to site changes, so a successful demo is not enough evidence of production reliability. @@ -251,7 +272,7 @@ The winning platform should still work when the workflow fails, changes, or move ## Final recommendation: What is the best no-code AI agent builder? -Sim is the best no-code and low-code AI agent builder for mixed teams that need visual construction, model flexibility, technical extensibility, and an Apache 2.0 self-hosting option. +Sim is the best no-code and low-code AI agent builder for mixed teams that need visual construction, model flexibility, technical extensibility, and a self-hosting option for its Apache 2.0 core. Choose n8n for a technical automation team that values mature workflow control and accepts its source-available license. Choose Zapier for a nontechnical team focused on hosted SaaS automation. Choose Make for visual routing and data mapping. Choose Gumloop for accessible AI-assisted browser and data workflows. @@ -269,7 +290,9 @@ This page stays focused on no-code and low-code selection. Use these guides for Product capabilities, licenses, deployment options, and commercial terms can change. Recheck these first-party sources before purchase: -- [Sim repository and Apache 2.0 license](https://github.com/simstudioai/sim) +- [Sim repository and Apache 2.0 core license](https://github.com/simstudioai/sim) +- [Sim Enterprise License for `apps/sim/ee`](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) +- [Sim pricing](https://www.sim.ai/pricing) - [Sim documentation](https://docs.sim.ai/) - [n8n Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) - [n8n self-hosting documentation](https://docs.n8n.io/deploy/host-n8n/) @@ -279,4 +302,4 @@ Product capabilities, licenses, deployment options, and commercial terms can cha - [Make AI Agents](https://www.make.com/en/ai-agents) - [Make pricing](https://www.make.com/en/pricing) - [Gumloop documentation](https://docs.gumloop.com/) -- [Gumloop pricing](https://www.gumloop.com/pricing) +- [Gumloop pricing](https://www.gumloop.com/pricing) and [credit documentation](https://docs.gumloop.com/core-concepts/credits) diff --git a/apps/sim/content/library/byok-multi-model-ai-agent-builder/index.mdx b/apps/sim/content/library/byok-multi-model-ai-agent-builder/index.mdx index dc2edb29e3b..9e3e4a15cde 100644 --- a/apps/sim/content/library/byok-multi-model-ai-agent-builder/index.mdx +++ b/apps/sim/content/library/byok-multi-model-ai-agent-builder/index.mdx @@ -1,9 +1,9 @@ --- slug: byok-multi-model-ai-agent-builder title: "BYOK Multi-Model AI Agent Builder: How Sim's Bring-Your-Own-Key Works" -description: 'Learn how Sim BYOK separates customer-owned provider keys from hosted model access and approved Enterprise-only local-model access, including billing, security, and deployment considerations.' +description: 'Learn how Sim BYOK separates customer-owned provider keys from hosted model access and self-hosted local models, including the 1.1x hosted rate, supported providers, billing, and security.' date: 2026-07-18 -updated: 2026-09-26 +updated: 2026-10-01 authors: - andrew readingTime: 10 @@ -13,12 +13,16 @@ draft: false faq: - q: "What is BYOK in Sim?" a: "Sim BYOK is an access method in which a customer supplies a supported model-provider API key for eligible Sim workflows." + - q: "Which providers are BYOK-eligible in Sim?" + a: "Sim accepts your own keys for the LLM providers OpenAI, Anthropic, Google, Mistral, Z.ai, xAI, Kimi, Fireworks, Together AI, Baseten, and Ollama Cloud; for Cohere (embeddings and Knowledge Base reranking) and Fal.ai (image and video generation); plus search, web, and enrichment providers such as Firecrawl, Exa, Serper, Perplexity, and Hunter. Each key is saved per provider, and Sim routes that provider's calls through your account." + - q: "How is billing separated between hosted keys and BYOK?" + a: "With Sim's hosted keys, model usage bills through Sim credits with a 1.1x multiplier on the provider's base price. With your own key, the provider bills your account at its own rates with no Sim markup, and Sim's per-run base charge still applies." - q: "Does Sim BYOK make model usage free?" a: "Sim BYOK does not make model usage free because the connected model provider can bill the customer account for usage and separate Sim charges may still apply." - q: "Does Sim BYOK mean my data never leaves my machine?" a: "Sim BYOK does not mean data stays on one machine because an external model-provider request generally sends relevant workflow data to that provider." - q: "Does Sim BYOK include Ollama?" - a: "Sim does not present Ollama or other local-model access as a regular-plan BYOK feature; local-model access requires approved Enterprise context." + a: "Sim BYOK includes Ollama Cloud as a key-based provider. Locally hosted Ollama or vLLM models are a separate option: a self-hosted Sim deployment connects to them through its OLLAMA_URL or VLLM_BASE_URL setting, with no Enterprise plan required." - q: "Can Sim use multiple AI model providers?" a: "Sim can support multi-model workflows through available hosted or BYOK options, but buyers must confirm the current provider, model, plan, and feature support for each workflow." - q: "Who receives the token bill with Sim BYOK?" @@ -30,7 +34,7 @@ faq: - q: "Can a team rotate a Sim BYOK key?" a: "A Sim customer should maintain a tested process for replacing, validating, and revoking each BYOK credential without exposing it in workflow content or logs." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0." + a: "Sim's core is open source under the OSI-approved Apache License 2.0. Features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Does Sim's Apache 2.0 license include every hosted or Enterprise feature?" a: "Sim's Apache 2.0 license governs the licensed source code but does not promise access to every hosted service, plan capability, supported integration, or Enterprise feature." - q: "Is n8n open source?" @@ -39,21 +43,27 @@ faq: a: "Sim is the more directly AI-agent-focused option, while n8n is a strong choice for teams that prioritize broad workflow automation and already use its node ecosystem." - q: "What is the best AI agent builder?" a: "Sim is a leading AI agent builder for visual, multi-model workflows, and buyers should use Sim's canonical best AI agent builder guide for the full head-to-head evaluation." - - q: "When should a company ask Sim about Enterprise local-model access?" - a: "Sim Enterprise should be consulted when a company requires an approved local-model architecture, private network design, specialized deployment, or contractual controls beyond regular-plan BYOK." + - q: "When should a company run local models with Sim?" + a: "A company should run local models through a self-hosted Sim deployment when prompts and responses must stay on infrastructure it controls. Self-hosted Sim connects to Ollama or vLLM without an Enterprise plan." - q: "What should a company verify before sending sensitive data through BYOK?" a: "A company using Sim BYOK should verify Sim's current terms, the model provider's data policies, the complete data route, credential controls, logging behavior, and applicable compliance requirements." --- ## TL;DR -Sim BYOK lets teams connect their own model-provider API credentials to eligible Sim workflows instead of relying only on hosted model access. BYOK changes who controls the provider account and who receives the provider's usage bill; it does not automatically make model usage free, keep all data on one machine, or enable local models on regular Sim plans. +Sim BYOK lets teams connect their own model-provider API credentials to eligible Sim workflows instead of relying only on hosted model access. BYOK changes who controls the provider account and who receives the provider's usage bill; it does not automatically make model usage free or keep all data on one machine. -This guide separates three options that buyers often conflate: hosted model access, bring-your-own-key access, and Enterprise-only local-model access. +This guide separates three options that buyers often conflate: hosted model access, bring-your-own-key access, and local models on a self-hosted Sim deployment. + +- **Hosted keys:** zero setup. Sim bills model usage in credits with a 1.1x multiplier on the provider's base price. +- **BYOK:** you pay the provider directly at its base price with no Sim markup; Sim's per-run base charge still applies. +- **Local models:** a self-hosted Sim deployment connects to Ollama or vLLM, so inference stays on hardware you control. ## What does BYOK mean in an AI agent builder? -Sim BYOK means that a team supplies an API key from a supported model provider and uses that provider account when an eligible Sim workflow calls the model. Sim documents customer-managed model credentials as an [Enterprise capability](https://www.sim.ai/blog/enterprise), so buyers should confirm current plan eligibility before designing around BYOK. +Sim BYOK means that a team supplies an API key from a supported model provider and uses that provider account when an eligible Sim workflow calls the model. On Sim Cloud, a workspace admin can store workspace keys on any plan, and organization admins on Pro for Teams, Max for Teams, or Enterprise can store organization keys that every workspace inherits, as described in Sim's [BYOK documentation](https://docs.sim.ai/platform/costs#bring-your-own-key-byok). + +Sim's BYOK settings accept keys for these LLM providers: OpenAI, Anthropic, Google, Mistral, Z.ai, xAI, Kimi, Fireworks, Together AI, Baseten, and Ollama Cloud. Cohere keys cover embeddings and Knowledge Base reranking, and Fal.ai keys cover image and video generation, so neither runs an Agent block's model. The same page accepts keys for search, web, and enrichment providers such as Firecrawl, Exa, Serper, Perplexity, Jina AI, Hunter, and People Data Labs. Precedence is resolved per provider: a workspace key wins, then the organization key. Sim's hosted key, with the multiplier applied, is the fallback only for the models Sim hosts; providers such as Together AI, Baseten, and Ollama Cloud need your own key. BYOK stands for “bring your own key.” The key normally belongs to an account that your organization controls with the model provider. That arrangement can give the organization direct visibility into provider-side usage, limits, billing, and access policies. @@ -61,21 +71,21 @@ BYOK does not mean that the model runs inside Sim, inside your browser, or on yo ## How are hosted access, BYOK, and local-model access different in Sim? -Sim separates hosted access, BYOK, and Enterprise-only local-model access because each option has a different credential owner, billing path, data route, and deployment requirement. Sim's [current pricing page](https://www.sim.ai/pricing) is the source of truth for hosted plan allowances and limits. +Sim separates hosted access, BYOK, and self-hosted local models because each option has a different credential owner, billing path, data route, and deployment requirement. Sim's [current pricing page](https://www.sim.ai/pricing) is the source of truth for hosted plan allowances and limits. | Model-access option | Who supplies the model credential? | Who handles model-provider billing? | Where does inference happen? | Important limitation | |---|---|---|---|---| -| Hosted access | Sim manages the applicable provider access | Usage is governed by the current Sim plan and its applicable metering | At the hosted provider selected through Sim | Availability, included usage, and limits depend on the current plan | -| BYOK | The customer supplies a supported provider API key | The model provider bills the customer under the provider account; separate Sim plan charges may still apply | At the external provider connected with the key | BYOK is not local inference and does not eliminate provider token charges | -| Enterprise-only local-model access | Determined through an approved Enterprise deployment | Determined by the Enterprise architecture, infrastructure, and contract | In the approved Enterprise environment | Local-model access must not be assumed to exist on regular Sim plans | +| Hosted access | Sim manages the applicable provider access | Sim bills model usage in credits with a 1.1x multiplier on the provider's base price | At the hosted provider selected through Sim | Availability, included usage, and limits depend on the current plan | +| BYOK | The customer supplies a supported provider API key | The model provider bills the customer at its base price with no Sim markup; Sim's per-run base charge still applies | At the external provider connected with the key | BYOK is not local inference and does not eliminate provider token charges | +| Local models (Ollama or vLLM) | No provider key; a self-hosted Sim deployment points `OLLAMA_URL` or `VLLM_BASE_URL` at the model server | No model API charge; the team pays for the hardware that runs the model | On infrastructure the team controls | Configured at the deployment level, so it applies to self-hosted Sim rather than Sim Cloud | -As of September 2026, buyers should confirm current hosted-provider availability, BYOK provider support, plan limits, and Enterprise deployment terms with Sim before making an architecture decision. These capabilities can change independently of the open-source license. +As of October 2026, buyers should confirm current hosted-provider availability, BYOK provider support, and plan limits with Sim before making an architecture decision. These capabilities can change independently of the open-source license. ## Which model-access option should a team choose? -Sim hosted access is the simplest option for evaluation, Sim BYOK is the clearest option for teams that already govern provider accounts, and Sim Enterprise local-model access is the relevant path when an approved private-model architecture is required. +Sim hosted access is the simplest option for evaluation, Sim BYOK is the clearest option for teams that already govern provider accounts, and local models on a self-hosted Sim deployment are the relevant path when inference must stay on infrastructure the team controls. -Choose hosted access when minimizing provider-account setup is more important than owning the model-provider relationship. Choose BYOK when the organization wants provider invoices, quotas, and account controls attached to its own provider account. Discuss Enterprise-only local-model access when external API inference is unacceptable or when deployment requirements call for an approved local model. +Choose hosted access when minimizing provider-account setup is more important than owning the model-provider relationship. Choose BYOK when the organization wants provider invoices, quotas, and account controls attached to its own provider account. Choose local models on a self-hosted deployment when external API inference is unacceptable or when deployment requirements call for a model the team hosts itself. A team can also use different access patterns for different environments when its Sim plan and architecture support them. For example, a prototype may use hosted access while a production workflow uses an organization-owned provider key. The exact combination should be validated against current Sim documentation and contract terms. @@ -87,7 +97,7 @@ Security review should cover the entire request path rather than only the API ke The provider account should enforce the strongest controls that the provider supports, such as scoped access, project separation, spend limits, audit logs, and key rotation. Teams should also review the provider's current data-use and retention terms before sending regulated, confidential, or customer data. -Sim's [secrets documentation](https://docs.sim.ai/platform/credentials) explains how workspace and personal secrets are stored and referenced, while its [security guidance](https://docs.sim.ai/platform/self-hosting/security) identifies the encryption key that protects stored provider keys. Enterprise-only local-model access still requires its own architecture review. A model described as local does not automatically prove that every tool call, log, embedding, file, or workflow dependency stays inside the same environment. +Sim's [secrets documentation](https://docs.sim.ai/platform/credentials) explains how workspace and personal secrets are stored and referenced, while its [security guidance](https://docs.sim.ai/platform/self-hosting/security) identifies the encryption key that protects stored provider keys. Local models on a self-hosted deployment still require their own architecture review. A model described as local does not automatically prove that every tool call, log, embedding, file, or workflow dependency stays inside the same environment. ## Who pays for model usage when Sim uses BYOK? @@ -95,7 +105,7 @@ Sim BYOK normally makes the customer responsible for model-provider usage billed BYOK should not be described as “free tokens” or “zero token cost.” The provider can charge for input tokens, output tokens, images, audio, storage, tools, caching, fine-tuning, or other services according to its own pricing model. Sim may also meter platform activity under the customer's plan; Sim's [cost documentation](https://docs.sim.ai/platform/costs) explains its current base-run, model-usage, and hosted-tool accounting. -Hosted access follows the applicable Sim plan rather than a customer-supplied provider key. Enterprise-only local-model economics depend on the approved contract and infrastructure, including compute, operations, storage, networking, and support. +Hosted access bills model usage through Sim credits with a 1.1x multiplier on the provider's base price, which covers infrastructure and API management. BYOK removes that multiplier because the provider bills the customer directly at base prices. Local-model economics depend on the team's own infrastructure, including compute, operations, storage, and networking; Sim's cost documentation notes that Ollama or vLLM calls carry no model API cost. Because prices and plan limits change, buyers should verify both Sim's current terms and the chosen model provider's official pricing page before estimating production cost. @@ -117,13 +127,13 @@ Do not paste a provider key into prompts, workflow descriptions, code comments, A complete key inventory should record the owner, provider, environment, permitted models, spending controls, creation date, rotation date, dependent workflows, and emergency revocation process. -## Can Sim BYOK use Ollama or other local models on a regular plan? +## Can Sim use Ollama or other local models? -Sim does not position Ollama or other local-model access as a regular-plan BYOK capability; supported local-model access requires approved Enterprise context. +Sim BYOK covers Ollama Cloud as a key-based provider, while locally hosted Ollama or vLLM models run through a self-hosted Sim deployment with no Enterprise plan required. -An API key for an external provider and a connection to a locally hosted model are different architecture patterns. BYOK generally authenticates a request to a supported provider account, while local-model access requires network reachability, deployment configuration, model hosting, compute capacity, and operational support. +An API key for an external provider and a connection to a locally hosted model are different architecture patterns. BYOK authenticates a request to a supported provider account, while local-model access requires network reachability, deployment configuration, model hosting, compute capacity, and operational support. Sim's [Docker self-hosting guide](https://docs.sim.ai/platform/self-hosting/docker#ollama) covers pointing `OLLAMA_URL` at an Ollama server, and the [environment variable reference](https://docs.sim.ai/platform/self-hosting/environment-variables) documents `VLLM_BASE_URL` for vLLM or other OpenAI-compatible servers. -Sim's Apache 2.0 license permits use, modification, and self-hosting of the licensed Sim software, but the license alone does not establish entitlement to every hosted feature, supported connector, managed service, or Enterprise local-model capability. Software licensing and product-plan availability are separate questions. +Sim's core Apache 2.0 license permits use, modification, and self-hosting of the core software. Features in `apps/sim/ee` use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which requires an Enterprise subscription for production use. Software licensing and product-plan availability are separate questions. ## How does deployment affect Sim model access? @@ -131,7 +141,7 @@ Sim deployment determines where workflow components run, while the selected mode A self-hosted workflow can still send prompts to an external provider when it uses that provider's API. Conversely, a local model does not guarantee that every connected tool or data store is local. Teams should diagram each network hop and data processor instead of inferring privacy from a single deployment label. Sim's [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) describes deployment of the platform on customer infrastructure, not an automatic guarantee of local inference. -Sim is distributed under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an OSI-approved open-source license that permits self-hosting under its terms. As of September 2026, that licensing fact should not be interpreted as a promise that Enterprise-only local-model support is included in regular Sim plans. +Sim's core is distributed under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an OSI-approved open-source license that permits self-hosting under its terms. Features in `apps/sim/ee` use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which requires an Enterprise subscription for production use. For broader deployment context, see [open-source AI agent platforms](https://www.sim.ai/library/open-source-ai-agent-platforms). @@ -141,7 +151,7 @@ Sim focuses its BYOK experience on building AI agents and model-driven workflows Both products require buyers to examine provider support, credential handling, deployment, workflow metering, and provider-side billing separately. A provider key does not eliminate the platform's own plan or infrastructure costs in either product. -Licensing is an important difference. Sim uses the OSI-approved Apache License 2.0. As of September 2026, n8n uses its [Sustainable Use License](https://github.com/n8n-io/n8n/blob/master/LICENSE.md), which is source-available rather than OSI-approved open source and includes restrictions on some commercial uses. Buyers should read n8n's current license and [Sustainable Use License documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) for the exact permissions. +Licensing is an important difference. Sim's core uses the OSI-approved Apache License 2.0, while features in `apps/sim/ee` use the separate Sim Enterprise License. As of September 2026, n8n uses its [Sustainable Use License](https://github.com/n8n-io/n8n/blob/master/LICENSE.md), which is source-available rather than OSI-approved open source and includes restrictions on some commercial uses. Buyers should read n8n's current license and [Sustainable Use License documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) for the exact permissions. The better fit depends on the job. Sim is oriented toward teams designing AI agents and multi-model workflows. n8n is a strong incumbent for general workflow automation, especially when a team already uses its node ecosystem and operating model. Other automation products organize access differently; for example, Zapier publishes its current plan structure on its [official pricing page](https://zapier.com/pricing). @@ -151,7 +161,7 @@ Sim BYOK is best for teams that want to build AI agents in Sim while retaining d BYOK is particularly useful when finance needs provider invoices, platform teams need provider-side quotas, security teams require controlled credential ownership, or developers need access to models enabled for an existing organizational account. -Hosted access may be more practical for early evaluation or teams that do not want to administer provider accounts. Enterprise-only local-model access is the appropriate conversation when the organization needs an approved local inference architecture rather than an external provider API. +Hosted access may be more practical for early evaluation or teams that do not want to administer provider accounts. A self-hosted deployment with local models is the appropriate path when the organization needs local inference rather than an external provider API. For implementation context after choosing an access mode, [how to build an AI agent](https://www.sim.ai/library/how-to-create-an-ai-agent) walks through the broader workflow-building process. @@ -159,10 +169,10 @@ For implementation context after choosing an access mode, [how to build an AI ag Sim and n8n differ in product focus and licensing, while both require buyers to separate platform costs from model-provider usage. -- Sim uses the OSI-approved Apache License 2.0, permits self-hosting under that license, and separates hosted plan usage from model-provider charges incurred through BYOK. +- Sim's core uses the OSI-approved Apache License 2.0 and permits self-hosting under that license, features in `apps/sim/ee` use the separate Sim Enterprise License, and Sim separates hosted plan usage from model-provider charges incurred through BYOK. - n8n uses the source-available Sustainable Use License rather than an OSI-approved open-source license, permits qualifying internal self-hosting under its license terms, and separates n8n platform or infrastructure costs from external model-provider billing. - Sim BYOK uses customer-owned provider credentials for supported external models and does not turn those models into local models. -- Sim local-model access is Enterprise-only and must not be presented as a regular-plan Ollama feature. +- Sim connects to local Ollama or vLLM models through a self-hosted deployment, with no Enterprise plan required. ## What should buyers verify before using BYOK in production? @@ -179,10 +189,10 @@ Use this production checklist: 7. Configure provider-side budgets, quotas, and alerts where available. 8. Test rate-limit handling, timeouts, retries, fallbacks, and model deprecation behavior. 9. Document credential rotation and emergency revocation. -10. Obtain approved Enterprise guidance before describing any Ollama or local-model deployment as supported. +10. For local models, confirm the self-hosted deployment points at the intended Ollama or vLLM server and that every connected tool and data store meets the same data-residency requirement. ## Where can buyers compare Sim with other AI agent builders? Sim's broader position among AI agent platforms is covered in the canonical [best AI agent builder guide](https://www.sim.ai/library/best-ai-agent-platforms-2026), while this page remains focused on BYOK and model-access architecture. -Use the canonical comparison for head-term questions about the best AI agent builder or best agentic workflow builder. Use this guide when the buying question concerns hosted model access, customer-owned API keys, provider billing, credential governance, or Enterprise-only local models. +Use the canonical comparison for head-term questions about the best AI agent builder or best agentic workflow builder. Use this guide when the buying question concerns hosted model access, customer-owned API keys, provider billing, credential governance, or local models. From a7c71cc6e6a6d70edaf1e54b78cf8fe42f4aef57 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 11:10:53 -0700 Subject: [PATCH 03/31] chore(content): add check:library-content audit for content posts (#8526) * chore(content): add check:library-content audit for blog, library, and customer posts Validates frontmatter against the strict ContentFrontmatterSchema, slug/folder parity, local ogImage existence, MDX compilation with remark-gfm, FAQ placement, and internal post links (including retired and moved slugs). Moves the library slug redirect maps into lib/library/retired-slugs.ts so next.config.ts and the audit share one source. Fixes two FAQ answers that rendered Markdown links as literal text. * fix(content): tighten check:library-content link, author, and ogImage rules Treat draft blog/library posts and customer stories missing from CUSTOMER_STORIES as unserved link targets, reserve only sibling folders that define a page or route, validate author profiles with AuthorSchema, check moved blog slug destinations, and reject ogImage paths that resolve outside public. --- .../library/ai-agent-vs-chatbot/index.mdx | 4 +- apps/sim/lib/library/retired-slugs.ts | 29 + apps/sim/next.config.ts | 29 +- bun.lock | 1 + package.json | 2 + scripts/check-library-content.test.ts | 302 ++++++++++ scripts/check-library-content.ts | 519 ++++++++++++++++++ 7 files changed, 856 insertions(+), 30 deletions(-) create mode 100644 apps/sim/lib/library/retired-slugs.ts create mode 100644 scripts/check-library-content.test.ts create mode 100644 scripts/check-library-content.ts diff --git a/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx b/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx index 208c5eccee1..2ad7b36c603 100644 --- a/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx +++ b/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx @@ -12,9 +12,9 @@ ogImage: /library/ai-agent-vs-chatbot/cover.jpg draft: false faq: - q: "Do AI agents always produce more accurate answers than chatbots?" - a: "Answer accuracy is the degree to which a system's response is correct and supported by the available evidence. In [Sim](https://www.sim.ai), you can inspect each workflow step and control which instructions, data, and tools the model uses. This visibility helps you find the source of an error and improve the workflow without assuming that an agent is inherently more accurate than a chatbot." + a: "Answer accuracy is the degree to which a system's response is correct and supported by the available evidence. In Sim, you can inspect each workflow step and control which instructions, data, and tools the model uses. This visibility helps you find the source of an error and improve the workflow without assuming that an agent is inherently more accurate than a chatbot." - q: "Can a chatbot and an AI agent work together?" - a: "A hybrid application uses a chatbot for conversation and an AI agent for actions that require tools or multiple steps. Sim connects our [Chat interface](https://docs.sim.ai/execution/chat) to agent workflows you build in a visual workspace with connected integrations. You can give users one conversational interface while the agent handles work across connected applications." + a: "A hybrid application uses a chatbot for conversation and an AI agent for actions that require tools or multiple steps. Sim connects our Chat interface to agent workflows you build in a visual workspace with connected integrations. You can give users one conversational interface while the agent handles work across connected applications." - q: "Do I need to know how to code to build an AI agent?" a: "A visual agent builder represents workflow logic as configurable blocks that connect models to tools, so you do not have to program every step. Sim provides a visual workspace for creating and testing these workflows. You can add code when needed without building the orchestration layer from scratch." - q: "Can AI agents run with local models?" diff --git a/apps/sim/lib/library/retired-slugs.ts b/apps/sim/lib/library/retired-slugs.ts new file mode 100644 index 00000000000..cfddc9b30d8 --- /dev/null +++ b/apps/sim/lib/library/retired-slugs.ts @@ -0,0 +1,29 @@ +/** + * AEO/GEO-style posts (listicles, comparisons, how-tos) split out of `/blog` + * into the dedicated `/library` section so `/blog` stays editorial-only. + * `next.config.ts` redirects `/blog/` to `/library/` for each. + */ +export const LIBRARY_MOVED_BLOG_SLUGS = [ + 'best-zapier-alternatives', + 'ai-agents-vs-rpa', + 'ai-agent-vs-chatbot', + 'openai-vs-n8n-vs-sim', + 'ai-agent-ideas', + 'how-to-create-an-ai-agent', +] as const + +/** + * Library articles retired by merging into a stronger article on the same + * search intent, keyed by retired slug. `next.config.ts` redirects each + * retired URL to the surviving article, and `check:library-content` rejects + * new links to a retired slug. + */ +export const LIBRARY_MERGED_SLUGS: Readonly> = { + 'automation-anywhere-alternative': 'ai-agents-vs-rpa', + 'ai-native-vs-traditional-workflow-automation': + 'ai-native-workflow-automation-vs-traditional-automation', + 'best-ai-workflow-builders-small-teams-2026': 'best-ai-workflow-builders', + 'best-ai-agent-builder-2026': 'best-ai-agent-platforms-2026', + 'best-ai-agent-builders-slack-crm-automation-2026': 'best-ai-agents-for-slack', + 'best-open-source-ai-agent-frameworks': 'open-source-ai-agent-platforms', +} diff --git a/apps/sim/next.config.ts b/apps/sim/next.config.ts index dd84738690f..88c2c65be93 100644 --- a/apps/sim/next.config.ts +++ b/apps/sim/next.config.ts @@ -9,34 +9,7 @@ import { getWorkflowExecutionCSPPolicy, } from './lib/core/security/csp' import { LANDING_ROUTES } from './lib/landing/routes' - -/** - * AEO/GEO-style posts (listicles, comparisons, how-tos) split out of `/blog` - * into the dedicated `/library` section so `/blog` stays editorial-only. - */ -const LIBRARY_MOVED_BLOG_SLUGS = [ - 'best-zapier-alternatives', - 'ai-agents-vs-rpa', - 'ai-agent-vs-chatbot', - 'openai-vs-n8n-vs-sim', - 'ai-agent-ideas', - 'how-to-create-an-ai-agent', -] as const - -/** - * Library articles retired by merging into a stronger article on the same - * search intent, keyed by retired slug. Keeps indexed URLs and inbound links - * pointing at the surviving article. - */ -const LIBRARY_MERGED_SLUGS: Record = { - 'automation-anywhere-alternative': 'ai-agents-vs-rpa', - 'ai-native-vs-traditional-workflow-automation': - 'ai-native-workflow-automation-vs-traditional-automation', - 'best-ai-workflow-builders-small-teams-2026': 'best-ai-workflow-builders', - 'best-ai-agent-builder-2026': 'best-ai-agent-platforms-2026', - 'best-ai-agent-builders-slack-crm-automation-2026': 'best-ai-agents-for-slack', - 'best-open-source-ai-agent-frameworks': 'open-source-ai-agent-platforms', -} +import { LIBRARY_MERGED_SLUGS, LIBRARY_MOVED_BLOG_SLUGS } from './lib/library/retired-slugs' const nextConfig: NextConfig = { devIndicators: false, diff --git a/bun.lock b/bun.lock index eb4bb7cd7c0..e2719eca5a2 100644 --- a/bun.lock +++ b/bun.lock @@ -7,6 +7,7 @@ "devDependencies": { "@babel/parser": "7.29.2", "@biomejs/biome": "2.0.6", + "@mdx-js/mdx": "3.1.1", "@octokit/rest": "^21.0.0", "@sim/utils": "workspace:*", "@types/opentype.js": "1.3.10", diff --git a/package.json b/package.json index 18fc1fc8916..aeaefe9f360 100644 --- a/package.json +++ b/package.json @@ -71,6 +71,7 @@ "check:native-typecheck": "bun run scripts/check-native-typecheck.ts", "check:source-text": "bun run scripts/check-source-text.ts", "check:site-urls": "bun run scripts/check-site-urls.ts", + "check:library-content": "bun run scripts/check-library-content.ts", "check:spec-example-ids": "bun run scripts/check-spec-example-ids.ts", "check:script-tests": "bun run scripts/check-script-test-coverage.ts", "check:test-patterns": "bun run scripts/check-test-patterns.ts", @@ -162,6 +163,7 @@ "devDependencies": { "@babel/parser": "7.29.2", "@biomejs/biome": "2.0.6", + "@mdx-js/mdx": "3.1.1", "@octokit/rest": "^21.0.0", "@sim/utils": "workspace:*", "@types/opentype.js": "1.3.10", diff --git a/scripts/check-library-content.test.ts b/scripts/check-library-content.test.ts new file mode 100644 index 00000000000..65db6936e0d --- /dev/null +++ b/scripts/check-library-content.test.ts @@ -0,0 +1,302 @@ +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import path from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + type ContentCheckConfig, + checkContent, + checkRedirectTargets, + indexPosts, + readReservedSegments, + resolvePostArg, +} from './check-library-content' + +let root: string +let config: ContentCheckConfig + +const CLEAN_FRONTMATTER = { + title: 'A clean library post', + description: 'A description that is comfortably over twenty characters.', + date: '2026-09-01', + authors: '[sim]', +} + +function writePost( + section: string, + slug: string, + body: string, + frontmatter: Record = {} +) { + const fields = { + slug, + ...CLEAN_FRONTMATTER, + ogImage: `/${section}/${slug}/cover.jpg`, + ...frontmatter, + } + const yaml = Object.entries(fields) + .map(([key, value]) => (value.startsWith('\n') ? `${key}:${value}` : `${key}: ${value}`)) + .join('\n') + const dir = path.join(config.contentDir, section, slug) + mkdirSync(dir, { recursive: true }) + writeFileSync(path.join(dir, 'index.mdx'), `---\n${yaml}\n---\n\n${body}\n`) + const imageDir = path.join(config.publicDir, section, slug) + mkdirSync(imageDir, { recursive: true }) + writeFileSync(path.join(imageDir, 'cover.jpg'), '') +} + +async function findingsFor(section: string, slug: string) { + const { findings } = await checkContent(config, [{ section: section as 'library', slug }]) + return findings.map(({ line, rule, message }) => ({ line, rule, message })) +} + +beforeEach(() => { + root = mkdtempSync(path.join(tmpdir(), 'library-content-')) + config = { + contentDir: path.join(root, 'content'), + publicDir: path.join(root, 'public'), + reservedSegments: { blog: new Set(['tags']), library: new Set(['tags']), customers: new Set() }, + mergedSlugs: { 'old-guide': 'kept-guide' }, + movedBlogSlugs: ['moved-post'], + customerSlugs: ['acme'], + } + mkdirSync(path.join(config.contentDir, 'authors'), { recursive: true }) + writeFileSync(path.join(config.contentDir, 'authors', 'sim.json'), '{"id":"sim","name":"Sim"}') + writePost('library', 'kept-guide', 'The surviving guide.') + writePost('library', 'moved-post', 'Moved from the blog.') +}) + +afterEach(() => { + rmSync(root, { recursive: true, force: true }) +}) + +describe('check-library-content', () => { + it('passes a clean post with valid internal links, assets, reserved routes, and code samples', async () => { + writePost( + 'library', + 'clean', + [ + '## Overview', + 'See [the guide](/library/kept-guide) and https://www.sim.ai/library/kept-guide#intro.', + '![diagram](/library/clean/diagram.png) and [tags](/library/tags) are not posts.', + 'An apex https://sim.ai/library/missing link belongs to check:site-urls.', + '| a | b |', + '| - | - |', + '| 1 | 2 |', + '```md', + '## FAQ', + '[old](/library/old-guide)', + '```', + ].join('\n') + ) + expect(await findingsFor('library', 'clean')).toEqual([]) + }) + + it('rejects an unknown frontmatter key such as canonical, at its line', async () => { + writePost('library', 'post', 'Body.', { canonical: 'https://www.sim.ai/library/post' }) + expect(await findingsFor('library', 'post')).toEqual([ + { line: 8, rule: 'frontmatter', message: 'Unknown frontmatter key "canonical".' }, + ]) + }) + + it('rejects frontmatter that fails the schema', async () => { + writePost('library', 'post', 'Body.', { title: 'Hi' }) + const findings = await findingsFor('library', 'post') + expect(findings).toHaveLength(1) + expect(findings[0]).toMatchObject({ line: 3, rule: 'frontmatter' }) + }) + + it('rejects invalid YAML', async () => { + writePost('library', 'post', 'Body.', { title: 'Broken: [unclosed' }) + expect((await findingsFor('library', 'post'))[0]).toMatchObject({ rule: 'frontmatter' }) + }) + + it('rejects an author with no profile', async () => { + writePost('library', 'post', 'Body.', { authors: '[ghost]' }) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 6, + rule: 'frontmatter', + message: 'Author "ghost" has no profile in apps/sim/content/authors.', + }, + ]) + }) + + it('rejects a slug that differs from the folder name', async () => { + writePost('library', 'post', 'Body.', { slug: 'other' }) + expect(await findingsFor('library', 'post')).toEqual([ + { line: 2, rule: 'slug', message: 'slug "other" does not match the folder name "post".' }, + ]) + }) + + it('rejects a missing or remote ogImage', async () => { + writePost('library', 'missing', 'Body.', { ogImage: '/library/missing/nope.jpg' }) + writePost('library', 'remote', 'Body.', { ogImage: 'https://example.com/a.jpg' }) + expect(await findingsFor('library', 'missing')).toEqual([ + { + line: 7, + rule: 'og-image', + message: 'ogImage "/library/missing/nope.jpg" does not exist under apps/sim/public.', + }, + ]) + expect((await findingsFor('library', 'remote'))[0]).toMatchObject({ rule: 'og-image' }) + }) + + it('reports an MDX compile error at its file line', async () => { + writePost('library', 'post', 'Intro.\n\nA tag
{ + writePost('library', 'post', 'Intro.\n\n## FAQs\n\nQ and A.') + expect(await findingsFor('library', 'post')).toEqual([ + { line: 12, rule: 'faq', message: 'Body has a "## FAQs" heading.' }, + ]) + }) + + it('rejects Markdown link syntax inside an FAQ answer, at the answer line', async () => { + writePost('library', 'post', 'Body.', { + faq: '\n - q: "Plain question?"\n a: "Plain answer."\n - q: "Linked?"\n a: "See [Sim](https://www.sim.ai)."', + }) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 12, + rule: 'faq', + message: 'faq[1].a contains Markdown link syntax, which renders as literal text.', + }, + ]) + }) + + it('rejects links to missing, retired, and moved posts, naming the replacement', async () => { + writePost( + 'library', + 'post', + [ + '[a](/library/missing)', + '[b](https://www.sim.ai/library/old-guide)', + 'c', + '[d](/blog/kept-guide)', + ].join('\n') + ) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 10, + rule: 'internal-link', + message: '/library/missing does not exist (no apps/sim/content/library/missing/index.mdx).', + }, + { + line: 11, + rule: 'internal-link', + message: '/library/old-guide is retired; it was merged into /library/kept-guide.', + }, + { + line: 12, + rule: 'internal-link', + message: '/blog/moved-post moved to /library/moved-post.', + }, + { + line: 13, + rule: 'internal-link', + message: '/blog/kept-guide does not exist (no apps/sim/content/blog/kept-guide/index.mdx).', + }, + ]) + }) + + it('rejects a retired slug whose replacement no longer exists', () => { + config.mergedSlugs = { 'old-guide': 'gone-guide' } + const findings = checkRedirectTargets(config, indexPosts(config.contentDir), 'map.ts') + expect(findings.map((finding) => finding.message)).toEqual([ + '/library/old-guide redirects to /library/gone-guide, which does not exist (no apps/sim/content/library/gone-guide/index.mdx).', + ]) + }) + + it('rejects a moved blog slug whose library destination is missing or a draft', () => { + config.movedBlogSlugs = ['moved-post', 'never-moved'] + writePost('library', 'moved-post', 'Unpublished.', { draft: 'true' }) + const findings = checkRedirectTargets(config, indexPosts(config.contentDir), 'map.ts') + expect(findings.map((finding) => finding.message)).toEqual([ + '/blog/moved-post redirects to /library/moved-post, which is a draft, so it 404s.', + '/blog/never-moved redirects to /library/never-moved, which does not exist (no apps/sim/content/library/never-moved/index.mdx).', + ]) + }) + + it('rejects a link to a draft blog or library post', async () => { + writePost('library', 'unpublished', 'Draft.', { draft: 'true' }) + writePost('library', 'post', '[a](/library/unpublished)') + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 10, + rule: 'internal-link', + message: '/library/unpublished is a draft, so it 404s.', + }, + ]) + }) + + it('accepts a registered draft customer story but rejects an unregistered one', async () => { + writePost('customers', 'acme', 'Registered draft.', { draft: 'true' }) + writePost('customers', 'globex', 'Folder without a CUSTOMER_STORIES entry.') + writePost('library', 'post', '[a](/customers/acme)\n[b](/customers/globex)') + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 11, + rule: 'internal-link', + message: + '/customers/globex is not in CUSTOMER_STORIES (apps/sim/lib/customers/data.ts), so it 404s.', + }, + ]) + }) + + it('reserves only sibling folders that define a page or route handler', () => { + const appDir = path.join(root, 'app') + for (const [dir, file] of [ + ['tags', 'page.tsx'], + ['rss.xml', 'route.ts'], + ['[slug]', 'page.tsx'], + ['components', 'card.tsx'], + ]) { + mkdirSync(path.join(appDir, dir), { recursive: true }) + writeFileSync(path.join(appDir, dir, file), '') + } + expect([...readReservedSegments(() => appDir).customers].sort()).toEqual(['rss.xml', 'tags']) + }) + + it('rejects an invalid author profile and does not accept its id', async () => { + writeFileSync(path.join(config.contentDir, 'authors', 'ghost.json'), '{"id":"ghost"}') + writeFileSync(path.join(config.contentDir, 'authors', 'broken.json'), '{not json') + writePost('library', 'post', 'Body.', { authors: '[ghost]' }) + const findings = await findingsFor('library', 'post') + expect(findings.map(({ rule, message }) => ({ rule, message }))).toEqual([ + { rule: 'frontmatter', message: expect.stringMatching(/^Author profile is invalid/) }, + { rule: 'frontmatter', message: expect.stringMatching(/^Author profile is invalid: name/) }, + { + rule: 'frontmatter', + message: 'Author "ghost" has no profile in apps/sim/content/authors.', + }, + ]) + }) + + it('rejects an ogImage that resolves outside public, even when the file exists', async () => { + writeFileSync(path.join(root, 'secret.jpg'), '') + writePost('library', 'post', 'Body.', { ogImage: '/../secret.jpg' }) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 7, + rule: 'og-image', + message: 'ogImage "/../secret.jpg" resolves outside apps/sim/public.', + }, + ]) + }) + + it('resolves a post argument by section/slug or bare slug, and rejects unknown ones', () => { + const posts = indexPosts(config.contentDir) + expect(resolvePostArg('library/kept-guide', posts)).toEqual([ + { section: 'library', slug: 'kept-guide' }, + ]) + expect(resolvePostArg('kept-guide', posts)).toEqual([ + { section: 'library', slug: 'kept-guide' }, + ]) + expect(typeof resolvePostArg('blog/kept-guide', posts)).toBe('string') + expect(typeof resolvePostArg('nope', posts)).toBe('string') + }) +}) diff --git a/scripts/check-library-content.ts b/scripts/check-library-content.ts new file mode 100644 index 00000000000..796ebb36cdd --- /dev/null +++ b/scripts/check-library-content.ts @@ -0,0 +1,519 @@ +#!/usr/bin/env bun +/** + * Validates every content post under `apps/sim/content/{blog,library,customers}//index.mdx`. + * + * Content is mostly written by agents and merged without a build, so a bad post only fails at + * `next build` (an invalid frontmatter key, an MDX syntax error) or never fails at all (a link to a + * retired or misspelled slug 301s or 404s in production). Each rule here is exact, so nothing + * needs an override: + * + * - `frontmatter`: gray-matter parses it and it passes the strict `ContentFrontmatterSchema`, so + * an unknown key such as `canonical` fails; every author id has a JSON file in `content/authors`. + * - `slug`: the `slug` field equals the post's folder name. + * - `og-image`: `ogImage` is a local path to a file under `apps/sim/public`. + * - `mdx`: the body compiles with the MDX compiler and `remark-gfm`, as the registry compiles it. + * - `faq`: the body has no FAQ heading (the FAQ lives in frontmatter, which renders it and emits + * its JSON-LD), and no FAQ question or answer contains Markdown link syntax (it renders as text). + * - `internal-link`: every `https://www.sim.ai/
/` link, and every relative + * `/
/` link target, names a page that serves: a published blog or library post, + * or a customer story registered in `CUSTOMER_STORIES` — never a retired or moved slug. Every + * retired or moved slug redirects to a published library post. Apex `https://sim.ai` links + * belong to `check:site-urls`. + * + * Run one post with `--slug
/` or `--slug `. + */ +import { existsSync, readdirSync, readFileSync, statSync } from 'node:fs' +import path from 'node:path' +import { compile } from '@mdx-js/mdx' +import matter from 'gray-matter' +import remarkGfm from 'remark-gfm' +import { AuthorSchema, ContentFrontmatterSchema } from '../apps/sim/lib/content/schema' +import { CUSTOMER_STORIES } from '../apps/sim/lib/customers/data' +import { + LIBRARY_MERGED_SLUGS, + LIBRARY_MOVED_BLOG_SLUGS, +} from '../apps/sim/lib/library/retired-slugs' + +export const SECTIONS = ['blog', 'library', 'customers'] as const +export type Section = (typeof SECTIONS)[number] + +export type Rule = 'frontmatter' | 'slug' | 'og-image' | 'mdx' | 'faq' | 'internal-link' + +export interface Finding { + file: string + line: number + rule: Rule + message: string + hint: string +} + +export interface ContentCheckConfig { + /** Holds one folder per section (`/
//index.mdx`) plus `authors/`. */ + contentDir: string + /** The app's `public/` directory, which local `ogImage` paths resolve against. */ + publicDir: string + /** Static route segments beside `[slug]` (e.g. `tags`, `rss.xml`) that are not posts. */ + reservedSegments: Readonly>> + /** Retired library slug to the slug that replaced it. */ + mergedSlugs: Readonly> + /** Blog slugs that now live under `/library`. */ + movedBlogSlugs: readonly string[] + /** Customer slugs the `/customers/[slug]` route serves (`CUSTOMER_STORIES`). */ + customerSlugs: readonly string[] +} + +/** Post folders per section, keyed by slug, with each post's `draft` flag. */ +export type PostIndex = Record> + +export interface PostRef { + section: Section + slug: string +} + +/** + * An absolute `www.sim.ai` URL anywhere, or a relative path in a Markdown link target or an + * `href` attribute. Group 1 is the section, group 2 the first segment, group 3 anything after it. + */ +const INTERNAL_LINK = + /(?:https?:\/\/www\.sim\.ai|(?<=\]\(\s*|href=\{?["'`]))\/(library|blog|customers)\/([^\s)"'`#?/<>\]]+)(\/[^\s)"'`#?<>\]]*)?/g +const MARKDOWN_LINK = /\[[^\]\n]*\]\([^)\n]*\)/ +const FAQ_HEADING = /^#{1,6}\s+FAQs?\s*:?\s*$/i +const CODE_FENCE = /^\s*(```|~~~)/ +const ROUTE_FILES = ['page.tsx', 'page.ts', 'route.tsx', 'route.ts'] + +function isSection(value: string): value is Section { + return (SECTIONS as readonly string[]).includes(value) +} + +/** A post whose frontmatter does not parse counts as published; its own check reports it. */ +function readDraft(file: string): boolean { + try { + return matter(readFileSync(file, 'utf-8'), {}).data.draft === true + } catch { + return false + } +} + +/** Post folders per section: every directory that holds an `index.mdx`. */ +export function indexPosts(contentDir: string): PostIndex { + const index = {} as PostIndex + for (const section of SECTIONS) { + const sectionDir = path.join(contentDir, section) + index[section] = new Map() + if (!existsSync(sectionDir)) continue + for (const entry of readdirSync(sectionDir, { withFileTypes: true })) { + const file = path.join(sectionDir, entry.name, 'index.mdx') + if (entry.isDirectory() && existsSync(file)) { + index[section].set(entry.name, { draft: readDraft(file) }) + } + } + } + return index +} + +/** Why `/
/` would not serve a page, or null when it does. */ +export function unservedReason( + posts: PostIndex, + customerSlugs: readonly string[], + section: Section, + slug: string +): string | null { + const post = posts[section].get(slug) + if (!post) return `does not exist (no apps/sim/content/${section}/${slug}/index.mdx)` + if (section === 'customers') { + return customerSlugs.includes(slug) + ? null + : 'is not in CUSTOMER_STORIES (apps/sim/lib/customers/data.ts), so it 404s' + } + return post.draft ? 'is a draft, so it 404s' : null +} + +/** Static routes beside each section's `[slug]` route: folders that define a page or route handler. */ +export function readReservedSegments(sectionAppDir: (section: Section) => string) { + const reserved = {} as Record> + for (const section of SECTIONS) { + const dir = sectionAppDir(section) + reserved[section] = new Set( + existsSync(dir) + ? readdirSync(dir, { withFileTypes: true }) + .filter( + (entry) => + entry.isDirectory() && + !/^[[(_]/.test(entry.name) && + ROUTE_FILES.some((name) => existsSync(path.join(dir, entry.name, name))) + ) + .map((entry) => entry.name) + : [] + ) + } + return reserved +} + +/** Parses `
/` or a bare ``, resolving the latter against the post index. */ +export function resolvePostArg(arg: string, posts: PostIndex): PostRef[] | string { + const [first, second] = arg.replace(/\/+$/, '').split('/') + if (second !== undefined) { + if (!isSection(first)) return `Unknown section "${first}"; use one of ${SECTIONS.join(', ')}.` + return posts[first].has(second) ? [{ section: first, slug: second }] : `No post at ${arg}.` + } + const matches = SECTIONS.filter((section) => posts[section].has(first)).map((section) => ({ + section, + slug: first, + })) + if (matches.length === 0) return `No post named "${first}" in ${SECTIONS.join(', ')}.` + return matches +} + +function lineOfKey(frontmatterLines: string[], key: string): number { + const index = frontmatterLines.findIndex((line) => line.startsWith(`${key}:`)) + return index === -1 ? 1 : index + 2 +} + +/** Validates one post, returning every finding (empty when the post is clean). */ +export async function checkPost( + config: ContentCheckConfig, + posts: PostIndex, + authorIds: ReadonlySet, + { section, slug }: PostRef +): Promise { + const file = path.join(config.contentDir, section, slug, 'index.mdx') + const raw = readFileSync(file, 'utf-8') + const findings: Finding[] = [] + const report = (line: number, rule: Rule, message: string, hint: string) => + findings.push({ file, line, rule, message, hint }) + + let parsed: matter.GrayMatterFile + try { + // A fresh options object bypasses gray-matter's content-keyed cache. + parsed = matter(raw, {}) + } catch (error) { + const mark = (error as { mark?: { line?: number } }).mark + report( + typeof mark?.line === 'number' ? mark.line + 2 : 1, + 'frontmatter', + `Frontmatter is not valid YAML: ${(error as Error).message.split('\n')[0]}`, + 'Fix the YAML between the --- fences (quote values that contain a colon).' + ) + return findings + } + + const body = parsed.content + const bodyOffset = raw.endsWith(body) + ? raw.slice(0, raw.length - body.length).split('\n').length - 1 + : 0 + const frontmatterLines = raw.split('\n').slice(1, Math.max(bodyOffset - 1, 1)) + + const result = ContentFrontmatterSchema.safeParse(parsed.data) + if (!result.success) { + for (const issue of result.error.issues) { + if (issue.code === 'unrecognized_keys') { + for (const key of issue.keys) { + report( + lineOfKey(frontmatterLines, key), + 'frontmatter', + `Unknown frontmatter key "${key}".`, + key === 'canonical' + ? 'Remove it: the canonical URL is derived from the section and slug.' + : `Remove it, or add it to ContentFrontmatterSchema in apps/sim/lib/content/schema.ts.` + ) + } + continue + } + const key = String(issue.path[0] ?? '') + report( + key ? lineOfKey(frontmatterLines, key) : 1, + 'frontmatter', + `Frontmatter ${issue.path.join('.') || '(root)'}: ${issue.message}`, + 'Match ContentFrontmatterSchema in apps/sim/lib/content/schema.ts.' + ) + } + } + + const data = parsed.data as Record + + if (Array.isArray(data.authors)) { + for (const author of data.authors) { + if (typeof author === 'string' && !authorIds.has(author)) { + report( + lineOfKey(frontmatterLines, 'authors'), + 'frontmatter', + `Author "${author}" has no profile in apps/sim/content/authors.`, + `Use one of: ${[...authorIds].sort().join(', ')}.` + ) + } + } + } + + if (typeof data.slug === 'string' && data.slug !== slug) { + report( + lineOfKey(frontmatterLines, 'slug'), + 'slug', + `slug "${data.slug}" does not match the folder name "${slug}".`, + `Set slug: ${slug}, or rename the folder (and its public/ assets) to match.` + ) + } + + if (typeof data.ogImage === 'string') { + const line = lineOfKey(frontmatterLines, 'ogImage') + if (!data.ogImage.startsWith('/')) { + report( + line, + 'og-image', + `ogImage "${data.ogImage}" is not a local path.`, + `Point it at a file under apps/sim/public, e.g. /${section}/${slug}/cover.jpg.` + ) + } else { + const target = path.resolve(config.publicDir, `.${data.ogImage}`) + const relative = path.relative(config.publicDir, target) + if (relative.startsWith('..') || path.isAbsolute(relative)) { + report( + line, + 'og-image', + `ogImage "${data.ogImage}" resolves outside apps/sim/public.`, + `Point it at a file under apps/sim/public, e.g. /${section}/${slug}/cover.jpg.` + ) + } else if (!existsSync(target) || !statSync(target).isFile()) { + report( + line, + 'og-image', + `ogImage "${data.ogImage}" does not exist under apps/sim/public.`, + `Add apps/sim/public${data.ogImage}, or fix the path.` + ) + } + } + } + + if (Array.isArray(data.faq)) { + const questionLines: number[] = [] + const answerLines: number[] = [] + frontmatterLines.forEach((line, index) => { + if (/^\s*-?\s*q:/.test(line)) questionLines.push(index + 2) + if (/^\s*-?\s*a:/.test(line)) answerLines.push(index + 2) + }) + data.faq.forEach((entry: unknown, index: number) => { + if (!entry || typeof entry !== 'object') return + const { q, a } = entry as { q?: unknown; a?: unknown } + for (const [field, value, lines] of [ + ['q', q, questionLines], + ['a', a, answerLines], + ] as const) { + if (typeof value === 'string' && MARKDOWN_LINK.test(value)) { + report( + lines[index] ?? lineOfKey(frontmatterLines, 'faq'), + 'faq', + `faq[${index}].${field} contains Markdown link syntax, which renders as literal text.`, + 'Write the FAQ in plain text; put the link in the article body instead.' + ) + } + } + }) + } + + const bodyLines = body.split('\n') + let fence: string | null = null + bodyLines.forEach((text, index) => { + const fenceMatch = CODE_FENCE.exec(text) + if (fenceMatch) { + if (fence === null) fence = fenceMatch[1] + else if (fenceMatch[1] === fence) fence = null + return + } + if (fence !== null) return + const line = bodyOffset + index + 1 + if (FAQ_HEADING.test(text.trim())) { + report( + line, + 'faq', + `Body has a "${text.trim()}" heading.`, + 'Move the questions into the frontmatter `faq:` list (q/a pairs) and delete the section; the page renders it and emits FAQPage JSON-LD.' + ) + } + for (const match of text.matchAll(INTERNAL_LINK)) { + const linkSection = match[1] as Section + const target = match[2] + const rest = match[3] ?? '' + // A deeper path is a public asset (`/library//cover.jpg`) or a static sub-route. + if (rest !== '' && rest !== '/') continue + if (config.reservedSegments[linkSection].has(target)) continue + const link = `/${linkSection}/${target}` + if (linkSection === 'library' && target in config.mergedSlugs) { + const kept = config.mergedSlugs[target] + report( + line, + 'internal-link', + `${link} is retired; it was merged into /library/${kept}.`, + `Link /library/${kept} instead.` + ) + } else if (linkSection === 'blog' && config.movedBlogSlugs.includes(target)) { + report( + line, + 'internal-link', + `${link} moved to /library/${target}.`, + `Link /library/${target} instead.` + ) + } else { + const reason = unservedReason(posts, config.customerSlugs, linkSection, target) + if (!reason) continue + const elsewhere = SECTIONS.filter( + (other) => + other !== linkSection && !unservedReason(posts, config.customerSlugs, other, target) + ) + report( + line, + 'internal-link', + `${link} ${reason}.`, + elsewhere.length > 0 + ? `Did you mean ${elsewhere.map((other) => `/${other}/${target}`).join(' or ')}?` + : 'Fix the slug, publish the target, or remove the link.' + ) + } + } + }) + + try { + await compile(body, { remarkPlugins: [remarkGfm], outputFormat: 'function-body' }) + } catch (error) { + const { line, reason, message } = error as { line?: number; reason?: string; message?: string } + report( + typeof line === 'number' ? bodyOffset + line : bodyOffset + 1, + 'mdx', + `MDX does not compile: ${reason ?? message}`, + 'Escape a literal `<` or `{` as `<` / `\\{`, and close every JSX tag.' + ) + } + + return findings +} + +/** Every retired or moved slug must redirect to a library post that serves a page. */ +export function checkRedirectTargets( + config: ContentCheckConfig, + posts: PostIndex, + file: string +): Finding[] { + const redirects = [ + ...Object.entries(config.mergedSlugs).map(([from, to]) => [`/library/${from}`, to]), + ...config.movedBlogSlugs.map((slug) => [`/blog/${slug}`, slug]), + ] + return redirects.flatMap(([from, to]) => { + const reason = unservedReason(posts, config.customerSlugs, 'library', to) + if (!reason) return [] + return { + file, + line: 1, + rule: 'internal-link' as const, + message: `${from} redirects to /library/${to}, which ${reason}.`, + hint: 'Point the redirect in retired-slugs.ts at a published library post.', + } + }) +} + +/** Author ids from every author JSON that passes `AuthorSchema`; invalid profiles are findings. */ +export function readAuthors(contentDir: string): { ids: Set; findings: Finding[] } { + const dir = path.join(contentDir, 'authors') + const ids = new Set() + const findings: Finding[] = [] + if (!existsSync(dir)) return { ids, findings } + for (const name of readdirSync(dir) + .filter((entry) => entry.endsWith('.json')) + .sort()) { + const file = path.join(dir, name) + let json: unknown + try { + json = JSON.parse(readFileSync(file, 'utf-8')) + } catch (error) { + json = error + } + const result = AuthorSchema.safeParse(json) + if (result.success) { + ids.add(result.data.id) + continue + } + findings.push({ + file, + line: 1, + rule: 'frontmatter', + message: `Author profile is invalid: ${result.error.issues.map((issue) => `${issue.path.join('.') || '(root)'} ${issue.message}`).join('; ')}`, + hint: 'Match AuthorSchema in apps/sim/lib/content/schema.ts (valid JSON with id and name).', + }) + } + return { ids, findings } +} + +export async function checkContent( + config: ContentCheckConfig, + only?: PostRef[] +): Promise<{ checked: number; findings: Finding[] }> { + const posts = indexPosts(config.contentDir) + const authors = readAuthors(config.contentDir) + const targets = + only ?? + SECTIONS.flatMap((section) => [...posts[section].keys()].map((slug) => ({ section, slug }))) + const results = await Promise.all( + targets.map((target) => checkPost(config, posts, authors.ids, target)) + ) + return { checked: targets.length, findings: [...authors.findings, ...results.flat()] } +} + +async function main() { + const root = path.resolve(import.meta.dir, '..') + const appDir = path.join(root, 'apps/sim') + const config: ContentCheckConfig = { + contentDir: path.join(appDir, 'content'), + publicDir: path.join(appDir, 'public'), + reservedSegments: readReservedSegments((section) => + path.join(appDir, 'app/(landing)', section) + ), + mergedSlugs: LIBRARY_MERGED_SLUGS, + movedBlogSlugs: LIBRARY_MOVED_BLOG_SLUGS, + customerSlugs: CUSTOMER_STORIES.map((story) => story.slug), + } + + const args = process.argv.slice(2) + const flagIndex = args.indexOf('--slug') + const slugArg = flagIndex === -1 ? args.find((arg) => !arg.startsWith('-')) : args[flagIndex + 1] + if (flagIndex !== -1 && !slugArg) { + console.error('Usage: bun run check:library-content [--slug
/ | ]') + process.exit(1) + } + + let only: PostRef[] | undefined + const findings: Finding[] = [] + if (slugArg) { + const resolved = resolvePostArg(slugArg, indexPosts(config.contentDir)) + if (typeof resolved === 'string') { + console.error(resolved) + process.exit(1) + } + only = resolved + } else { + findings.push( + ...checkRedirectTargets( + config, + indexPosts(config.contentDir), + path.join(appDir, 'lib/library/retired-slugs.ts') + ) + ) + } + + const result = await checkContent(config, only) + findings.push(...result.findings) + + if (findings.length > 0) { + findings.sort((a, b) => a.file.localeCompare(b.file) || a.line - b.line) + console.error( + `Library content audit failed: ${findings.length} problem(s) in ${result.checked} post(s).\n\n` + + findings + .map( + (finding) => + ` ${path.relative(root, finding.file)}:${finding.line} [${finding.rule}] ${finding.message}\n fix: ${finding.hint}` + ) + .join('\n') + ) + process.exit(1) + } + + console.log(`Library content audit passed (${result.checked} posts).`) +} + +if (import.meta.main) await main() From 52b99301c564a5f9de987a4ff917e5413fa9cb8a Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 11:11:58 -0700 Subject: [PATCH 04/31] fix(seo): escape backslashes in llms-full.txt comparison table cells (#8529) --- apps/sim/app/llms-full.txt/route.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/apps/sim/app/llms-full.txt/route.ts b/apps/sim/app/llms-full.txt/route.ts index 5d23203bcfe..eedce680267 100644 --- a/apps/sim/app/llms-full.txt/route.ts +++ b/apps/sim/app/llms-full.txt/route.ts @@ -69,7 +69,10 @@ function proseToMarkdown(prose: Prose): string { /** A fact as a table cell: its compact form, as the comparison table renders it. */ function factCell(fact: Fact | undefined): string { const value = fact ? (fact.shortValue ?? fact.value) : 'Unknown' - return value.replace(/\|/g, '\\|').replace(/\s*\n\s*/g, ' ') + return value + .replace(/\\/g, '\\\\') + .replace(/\|/g, '\\|') + .replace(/\s*\n\s*/g, ' ') } function isoDate(date: Date): string { From 59e1e702638f03de38322d180c8f5d072f2d1427 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 11:20:28 -0700 Subject: [PATCH 05/31] fix(library): scope Sim license claims to the Apache 2.0 core and the Sim Enterprise License (#8530) * fix(library): scope sales/CRM license claims to Sim's Apache 2.0 core and the Sim Enterprise License * fix(library): scope Sim license claims to the Apache 2.0 core and the Sim Enterprise License --- .../index.mdx | 8 +-- .../index.mdx | 20 ++++---- .../index.mdx | 20 ++++---- .../index.mdx | 12 ++--- .../index.mdx | 10 ++-- .../index.mdx | 4 +- .../index.mdx | 12 ++--- .../index.mdx | 12 ++--- .../index.mdx | 28 +++++------ .../index.mdx | 18 +++---- .../index.mdx | 18 +++---- .../index.mdx | 34 ++++++------- .../index.mdx | 6 +-- .../index.mdx | 30 +++++------ .../index.mdx | 20 ++++---- .../best-ai-automation-tools-2026/index.mdx | 50 +++++++++---------- .../index.mdx | 30 +++++------ .../index.mdx | 42 ++++++++-------- .../index.mdx | 26 +++++----- .../index.mdx | 33 ++++++------ .../index.mdx | 16 +++--- .../best-zapier-alternatives/index.mdx | 34 ++++++------- .../library/dify-alternatives/index.mdx | 22 ++++---- .../index.mdx | 12 ++--- .../index.mdx | 12 ++--- .../open-source-ai-agent-platforms/index.mdx | 10 ++-- .../library/openai-vs-n8n-vs-sim/index.mdx | 44 ++++++++-------- .../index.mdx | 8 +-- .../index.mdx | 12 ++--- .../library/what-is-an-mcp-server/index.mdx | 4 +- .../index.mdx | 4 +- 31 files changed, 306 insertions(+), 305 deletions(-) diff --git a/apps/sim/content/library/agentic-ai-coding-tools-what-they-are-and-how-the-top-options-compare/index.mdx b/apps/sim/content/library/agentic-ai-coding-tools-what-they-are-and-how-the-top-options-compare/index.mdx index b8f07e36ecf..8feff246284 100644 --- a/apps/sim/content/library/agentic-ai-coding-tools-what-they-are-and-how-the-top-options-compare/index.mdx +++ b/apps/sim/content/library/agentic-ai-coding-tools-what-they-are-and-how-the-top-options-compare/index.mdx @@ -3,7 +3,7 @@ slug: agentic-ai-coding-tools-what-they-are-and-how-the-top-options-compare title: 'Best Agentic Coding Tools: IDEs & Platforms Compared' description: 'Compare the best agentic coding tools, IDEs, and platforms for planning, writing, debugging, and shipping code with autonomous AI agents.' date: 2026-08-01 -updated: 2026-09-20 +updated: 2026-10-01 authors: - andrew readingTime: 6 @@ -80,7 +80,7 @@ Choose an in-IDE agent for work inside a codebase and an agent-workflow platform ### Agent-workflow platforms that can run a coding step -- **[Sim](https://www.sim.ai/)** is an [Apache 2.0-licensed](https://github.com/simstudioai/sim) agent-workflow platform in which custom code can run as one step in a larger process. You can describe a workflow with [Chat](https://docs.sim.ai/chat/workflows) and add custom JavaScript through a [Function block](https://docs.sim.ai/workflows/blocks/function). +- **[Sim](https://www.sim.ai/)** is an agent-workflow platform with an [Apache 2.0-licensed core](https://github.com/simstudioai/sim), in which custom code can run as one step in a larger process. You can describe a workflow with [Chat](https://docs.sim.ai/chat/workflows) and add custom JavaScript through a [Function block](https://docs.sim.ai/workflows/blocks/function). - **[Gumloop](https://www.gumloop.com/)** is a hosted, no-code automation platform. Gumloop's [agentic AI tools roundup](https://www.gumloop.com/blog/agentic-ai-tools) describes how it fits alongside tools such as Cursor, n8n, and Zapier. Check Gumloop's official site for current pricing. - **[n8n](https://n8n.io/)** is a fair-code, self-hostable workflow platform with a visual canvas and a [code step](https://docs.n8n.io/integrations/builtin/core-nodes/n8n-nodes-base.code). [n8n's pricing page](https://n8n.io/pricing) lists cloud Starter at €20 per month billed annually with one shared project. Pro costs €50 per month billed annually, while Business costs €667 per month billed annually. Business includes self-hosting, SSO, SAML, LDAP, and Git-based version control. Enterprise pricing is custom. The self-hosted Community Edition is free under [n8n's Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). @@ -102,7 +102,7 @@ These groupings describe supported use cases rather than relative performance. C | Claude Code | Terminal coding agent | Terminal-based agent | Proprietary cloud service | CLI | [Claude Pro costs $17 to $20 per month](https://claude.com/pricing) | | Windsurf | In-IDE coding agent | Agent-based IDE | Proprietary local app with cloud services | IDE | Check the [Windsurf website](https://windsurf.com/) for current pricing | | Replit Agent | App-building agent | Prompt-to-deployed-app workflow | Proprietary hosted service | Browser-based development environment | Free Starter tier; see [current Replit pricing](https://replit.com/pricing) | -| Sim | Agent-workflow platform | Natural language, visual canvas, and API | Apache 2.0; self-hosted or cloud | Workflow builder and API | Free tier; [Pro costs $25 per user per month](https://www.sim.ai/pricing) | +| Sim | Agent-workflow platform | Natural language, visual canvas, and API | Apache 2.0 core; self-hosted or cloud | Workflow builder and API | Free tier; [Pro costs $25 per user per month](https://www.sim.ai/pricing) | | Gumloop | Agent-workflow platform | Natural language and visual canvas | Proprietary hosted service | Workflow builder | Check the [Gumloop website](https://www.gumloop.com/) for current pricing | | n8n | Agent-workflow platform | Visual canvas and code step | Fair-code; self-hosted or cloud | Workflow builder, API, and webhooks | Free self-hosted edition; [Starter Cloud costs €20 per month when billed annually](https://n8n.io/pricing) | @@ -116,7 +116,7 @@ A workflow platform can connect a coding agent's output to a broader process. Th Yes, but the tools in this article use different license models, including proprietary, source-available, and open-source licenses. Many prominent in-IDE coding agents, including Cursor, Windsurf, and Claude Code, are proprietary applications. -Licensing varies more among workflow platforms that can run code. n8n uses its [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), which is source-available and includes commercial restrictions. Sim's core uses the permissive [Apache License 2.0](https://github.com/simstudioai/sim), and our documentation provides [self-hosting guidance](https://docs.sim.ai/self-hosting). Sim's comparison of [open-source AI agent platforms](https://www.sim.ai/library/open-source-ai-agent-platforms) covers additional licensing and deployment models. +Licensing varies more among workflow platforms that can run code. n8n uses its [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), which is source-available and includes commercial restrictions. Sim's core uses the permissive [Apache License 2.0](https://github.com/simstudioai/sim), and our documentation provides [self-hosting guidance](https://docs.sim.ai/self-hosting). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Sim's comparison of [open-source AI agent platforms](https://www.sim.ai/library/open-source-ai-agent-platforms) covers additional licensing and deployment models. ## Do agentic coding tools support MCP? diff --git a/apps/sim/content/library/ai-agent-marketplace-vs-building-from-scratch/index.mdx b/apps/sim/content/library/ai-agent-marketplace-vs-building-from-scratch/index.mdx index 29cfc4084e0..684181fb56d 100644 --- a/apps/sim/content/library/ai-agent-marketplace-vs-building-from-scratch/index.mdx +++ b/apps/sim/content/library/ai-agent-marketplace-vs-building-from-scratch/index.mdx @@ -32,21 +32,21 @@ faq: - q: "How do I govern an AI agent from a template?" a: "An AI agent created from a template should have a named owner, least-privilege credentials, representative tests, execution logs, approval controls, version tracking, and a shutdown plan." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 and can be self-hosted under that license." + a: "Sim’s core is open source under the OSI-approved Apache License 2.0 and can be self-hosted under that license, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is Sim free?" - a: "Sim’s Apache 2.0 codebase can be self-hosted without a software license fee, while any current hosted-service pricing should be confirmed on Sim’s official pricing page." + a: "Sim’s Apache 2.0 core can be self-hosted without a software license fee, while production use of enterprise features in apps/sim/ee requires an Enterprise subscription and any current hosted-service pricing should be confirmed on Sim’s official pricing page." - q: "Is n8n open source?" a: "As of September 2026, n8n is source-available under the Sustainable Use License and is not open source under an OSI-approved license." - q: "Is Sim better than n8n for building AI agents?" - a: "Sim is the stronger fit when an Apache 2.0 license and an AI-agent-focused workflow builder are priorities, while n8n remains a relevant incumbent for teams evaluating broad workflow automation." + a: "Sim is the stronger fit when an Apache 2.0 core license and an AI-agent-focused workflow builder are priorities, while n8n remains a relevant incumbent for teams evaluating broad workflow automation." - q: "What is the best n8n alternative for AI agent workflows?" - a: "Sim is a strong n8n alternative for teams that prioritize an Apache 2.0 license, self-hosting, and customizable AI agent workflows." + a: "Sim is a strong n8n alternative for teams that prioritize an Apache 2.0 core license, self-hosting, and customizable AI agent workflows." - q: "What is the best open-source Zapier alternative for AI agents?" - a: "Sim is a strong open-source Zapier alternative for AI agent workflows because Sim uses the OSI-approved Apache License 2.0 and supports self-hosting." + a: "Sim is a strong open-source Zapier alternative for AI agent workflows because Sim’s core uses the OSI-approved Apache License 2.0 and supports self-hosting." - q: "How does Sim compare with Gumloop?" - a: "Sim is the stronger fit when Apache 2.0 licensing and self-hosting are requirements, while teams considering Gumloop should compare its current managed-service capabilities, portability, governance controls, and commercial terms directly." + a: "Sim is the stronger fit when an Apache 2.0 core license and self-hosting are requirements, while teams considering Gumloop should compare its current managed-service capabilities, portability, governance controls, and commercial terms directly." - q: "What is the best AI agent builder?" - a: "Sim is a leading AI agent builder for teams prioritizing flexible workflows, self-hosting, and Apache 2.0 licensing, while the full category comparison is covered in Sim’s canonical best AI agent builder guide." + a: "Sim is a leading AI agent builder for teams prioritizing flexible workflows, self-hosting, and an Apache 2.0 core license, while the full category comparison is covered in Sim’s canonical best AI agent builder guide." - q: "Can I start with a template and move to Sim later?" a: "Sim can replace or extend a template-led approach when the template’s logic and dependencies are documented well enough to reconstruct the workflow." - q: "Do I need developers to build an AI agent?" @@ -218,15 +218,15 @@ Marketplace availability should not be interpreted as an independent security re Sim and [n8n](https://docs.n8n.io/privacy-and-security/sustainable-use-license) both support teams building automations, but their licenses and product positioning should not be treated as identical. -- Sim is an AI agent workflow builder released under the [OSI-approved Apache License 2.0](https://opensource.org/license/apache-2-0), and its software can be self-hosted without a vendor usage-billing unit; hosted-service pricing should be checked separately on [Sim’s current pricing page](https://www.sim.ai/pricing). +- Sim is an AI agent workflow builder whose core is released under the [OSI-approved Apache License 2.0](https://opensource.org/license/apache-2-0), and that core can be self-hosted without a vendor usage-billing unit. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution; hosted-service pricing should be checked separately on [Sim’s current pricing page](https://www.sim.ai/pricing). - As of September 2026, n8n uses the Sustainable Use License, which is source-available but is not an OSI-approved open-source license; current cloud billing terms should be confirmed on [n8n’s official pricing page](https://n8n.io/pricing/). - [Zapier](https://zapier.com/apps) and [Make](https://www.make.com/en/integrations) are established automation incumbents that buyers may also evaluate when their primary requirement is connecting business applications rather than controlling an Apache-licensed agent-building stack. -The [Sim repository and Apache 2.0 license](https://github.com/simstudioai/sim) provide the controlling source for Sim’s licensing terms. The [n8n Sustainable Use License documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license) provides the controlling source for n8n’s licensing conditions. +The [Sim repository and Apache 2.0 license](https://github.com/simstudioai/sim) and the [Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) provide the controlling sources for Sim’s licensing terms. The [n8n Sustainable Use License documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license) provides the controlling source for n8n’s licensing conditions. ## Where does Sim fit between an AI agent marketplace and custom development? -Sim fits between fixed marketplace agents and fully custom development by giving teams a visual way to build and adapt agent workflows while retaining access to an Apache 2.0 codebase. +Sim fits between fixed marketplace agents and fully custom development by giving teams a visual way to build and adapt agent workflows while retaining access to an Apache 2.0 core codebase. Sim is most relevant when a team wants the speed of a builder but does not want its core workflow confined to an opaque marketplace listing. Teams can use reusable workflow patterns as a starting point, then define the tools, data flow, model interactions, conditions, and controls required by their own process. diff --git a/apps/sim/content/library/ai-agent-orchestration-frameworks-explained/index.mdx b/apps/sim/content/library/ai-agent-orchestration-frameworks-explained/index.mdx index 1180e5b03fc..2bc5881391c 100644 --- a/apps/sim/content/library/ai-agent-orchestration-frameworks-explained/index.mdx +++ b/apps/sim/content/library/ai-agent-orchestration-frameworks-explained/index.mdx @@ -3,7 +3,7 @@ slug: ai-agent-orchestration-frameworks-explained title: 'AI Agent Orchestration Frameworks Explained' description: 'Learn how AI agent orchestration coordinates models, tools, memory, approvals, and recovery, then evaluate common architecture patterns and frameworks.' date: 2026-08-19 -updated: 2026-09-28 +updated: 2026-10-01 authors: - andrew readingTime: 13 @@ -24,19 +24,19 @@ faq: - q: "What is the best AI agent orchestration framework?" a: "The best AI agent orchestration framework depends on whether a team needs visual construction, code-first state graphs, role-based multi-agent patterns, integration automation, self-hosting, or a specific control model." - q: "What is the best AI agent builder?" - a: "Sim is a leading option for teams seeking a visual, Apache 2.0, self-hostable AI agent builder, while Sim's Best AI Agent Platforms and Builders in 2026 guide is the canonical page for the full head-to-head category comparison." + a: "Sim is a leading option for teams seeking a visual, self-hostable AI agent builder with an Apache 2.0 core, while Sim's Best AI Agent Platforms and Builders in 2026 guide is the canonical page for the full head-to-head category comparison." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved license that permits self-hosting, modification, and redistribution subject to the license terms." + a: "Sim's core is open source under the Apache License 2.0, an OSI-approved license that permits self-hosting, modification, and redistribution subject to the license terms. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use and does not permit modification or redistribution." - q: "Is Sim free?" - a: "Sim's Apache 2.0 source code can be used and self-hosted without a software license fee, while current hosted-service pricing should be checked on Sim's official pricing page." + a: "Sim's Apache 2.0 core can be used and self-hosted without a software license fee, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use, and current hosted-service pricing should be checked on Sim's official pricing page." - q: "Is n8n open source?" a: "As of September 2026, n8n is source-available under the Sustainable Use License, which is not an OSI-approved open-source license and includes restrictions that teams should review before commercial use." - q: "What is the best open-source alternative to n8n for AI agents?" - a: "Sim is a strong open-source alternative to n8n for teams that prioritize visual AI agent orchestration, Apache 2.0 licensing, and unrestricted self-hosting under an OSI-approved license." + a: "Sim is a strong open-source alternative to n8n for teams that prioritize visual AI agent orchestration and self-hosting of a core licensed under the OSI-approved Apache 2.0 license; Sim's enterprise features in apps/sim/ee use the separate Sim Enterprise License." - q: "How does Sim compare with n8n for AI agent orchestration?" - a: "Sim focuses on visual AI agent and workflow orchestration under Apache 2.0, while n8n combines broad workflow automation with AI capabilities under the source-available Sustainable Use License." + a: "Sim focuses on visual AI agent and workflow orchestration with an Apache 2.0 core, while n8n combines broad workflow automation with AI capabilities under the source-available Sustainable Use License." - q: "How does Sim compare with Gumloop for AI agent orchestration?" - a: "Sim is the stronger fit when Apache 2.0 licensing and self-hosting are decisive requirements, while buyers should compare current managed features, connectors, controls, and pricing on each vendor's official site." + a: "Sim is the stronger fit when an Apache 2.0 core license and self-hosting are decisive requirements, while buyers should compare current managed features, connectors, controls, and pricing on each vendor's official site." - q: "Can AI agent orchestration include human approval?" a: "AI agent orchestration can pause before a sensitive action, present the proposed action and its evidence to an authorized reviewer, and resume from preserved state after approval or rejection." - q: "How do you prevent an AI agent from running forever?" @@ -284,13 +284,13 @@ The following table is a compact selection guide rather than a ranking. Product | Option | Primary interface | Consider it when | Evaluate carefully | |---|---|---|---| -| [Sim](https://github.com/simstudioai/sim) | Visual workflows with code-extensible orchestration | A team wants visual agent and workflow construction, API integrations, and Apache 2.0 self-hosting | Required connectors, deployment design, and organization-specific controls | +| [Sim](https://github.com/simstudioai/sim) | Visual workflows with code-extensible orchestration | A team wants visual agent and workflow construction, API integrations, and self-hosting of an Apache 2.0 core | Required connectors, deployment design, and organization-specific controls | | [LangGraph](https://docs.langchain.com/oss/python/langgraph/overview) | Code-first graph and state model | Developers want explicit state transitions and fine-grained control in application code | Runtime operations, graph complexity, and the boundary with adjacent LangChain services | | [Microsoft AutoGen](https://microsoft.github.io/autogen/stable/) | Code-first agent and multi-agent abstractions | Developers are experimenting with conversational or event-driven agent collaboration | Package maturity, version-specific APIs, and production control requirements | | [CrewAI](https://docs.crewai.com/) | Code-first agents, crews, and flows | A team prefers role-oriented multi-agent concepts combined with workflow flows | Handoff quality, state management, and whether multiple agents add measurable value | | [n8n](https://docs.n8n.io/integrations/builtin/cluster-nodes/root-nodes/n8n-nodes-langchain.agent/) | Visual workflow automation with AI nodes and code steps | A team wants agent steps within an incumbent integration-automation environment | License requirements, agent-specific controls, and complex stateful execution needs | -Sim is an [Apache 2.0 project that can be self-hosted](https://github.com/simstudioai/sim); current managed-service terms and pricing should always be checked on [Sim's official pricing page](https://www.sim.ai/pricing). A deployed Sim workflow can also be [exposed as an MCP tool](https://docs.sim.ai/workflows/deployment/mcp) for compatible external AI applications. +Sim's core is an [Apache 2.0 project that can be self-hosted](https://github.com/simstudioai/sim), while features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution; current managed-service terms and pricing should always be checked on [Sim's official pricing page](https://www.sim.ai/pricing). A deployed Sim workflow can also be [exposed as an MCP tool](https://docs.sim.ai/workflows/deployment/mcp) for compatible external AI applications. As of September 2026, n8n uses the [Sustainable Use License](https://github.com/n8n-io/n8n/blob/master/LICENSE.md), which is source-available but is not included among the [OSI-approved open-source licenses](https://opensource.org/licenses). Teams considering n8n should review the vendor's [license documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license) for permitted use rather than assuming that source availability grants unrestricted commercial rights. @@ -321,7 +321,7 @@ A small proof of concept should include failure tests and recovery tests, not ju AI agent orchestration tools differ most in interface, control model, deployment options, and the amount of infrastructure the adopting team must operate. -- [Sim](https://github.com/simstudioai/sim) is an Apache 2.0 visual orchestration platform that can be self-hosted, while current Sim Cloud billing details are maintained on [Sim's official pricing page](https://www.sim.ai/pricing). +- [Sim](https://github.com/simstudioai/sim) is a visual orchestration platform with an Apache 2.0 core that can be self-hosted, while current Sim Cloud billing details are maintained on [Sim's official pricing page](https://www.sim.ai/pricing). - [LangGraph](https://docs.langchain.com/oss/python/langgraph/overview) is a code-first graph orchestration library for developers who want explicit state and transitions, while any managed LangChain services should be evaluated separately from the library. - [Microsoft AutoGen](https://microsoft.github.io/autogen/stable/) is a code-first framework for agentic applications, and teams should verify the license and production status of the exact package and version they adopt. - [CrewAI](https://docs.crewai.com/) provides code-first agents, crews, and flows, and teams should verify the license and managed-service terms for the exact components they use. diff --git a/apps/sim/content/library/ai-agent-workflow-builders-multi-step-tasks/index.mdx b/apps/sim/content/library/ai-agent-workflow-builders-multi-step-tasks/index.mdx index 99224cf6506..2e02478a806 100644 --- a/apps/sim/content/library/ai-agent-workflow-builders-multi-step-tasks/index.mdx +++ b/apps/sim/content/library/ai-agent-workflow-builders-multi-step-tasks/index.mdx @@ -3,7 +3,7 @@ slug: ai-agent-workflow-builders-multi-step-tasks title: 'AI Agent Workflow Builders for Multi-Step Tasks: 6-Platform Comparison' description: 'Compare Sim, n8n, Gumloop, Dust, Relevance AI, and Dify for multi-step agent orchestration, memory, approvals, guardrails, debugging, logs, and deployment.' date: 2026-09-28 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -26,11 +26,11 @@ faq: - q: "How does Sim debug a failed AI agent workflow?" a: "Sim exposes block-level run logs that let a builder review each step's inputs, outputs, status, and failure point. This makes debugging more precise than treating the entire agent run as one opaque response." - q: "What is the difference between Sim and n8n for AI agent workflows?" - a: "Sim emphasizes explicit agent-workflow controls such as Human in the Loop, Guardrails, Evaluator, Wait, and block-level run logs, while n8n combines AI nodes with a broad general-purpose automation model. Sim uses the Apache License 2.0, whereas n8n uses the source-available Sustainable Use License." + a: "Sim emphasizes explicit agent-workflow controls such as Human in the Loop, Guardrails, Evaluator, Wait, and block-level run logs, while n8n combines AI nodes with a broad general-purpose automation model. Sim's core uses the Apache License 2.0, with enterprise features in apps/sim/ee under the separate Sim Enterprise License, which requires an Enterprise subscription for production use, whereas n8n uses the source-available Sustainable Use License." - q: "What is the difference between Sim and Gumloop?" a: "Sim provides explicit approval, guardrail, evaluation, waiting, and block-level review primitives in an agent workflow. Gumloop is positioned around accessible visual AI automation, but buyers should verify its current native support for each governance control they require." - q: "Which AI agent workflow builders can be self-hosted?" - a: "Sim can be self-hosted under Apache 2.0, and n8n provides a self-hosted edition under its source-available Sustainable Use License. Buyers should verify the current licenses, deployment modes, and enterprise restrictions for Dify, Gumloop, Dust, and Relevance AI directly with each vendor before making a deployment decision." + a: "Sim's core can be self-hosted under Apache 2.0, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use, and n8n provides a self-hosted edition under its source-available Sustainable Use License. Buyers should verify the current licenses, deployment modes, and enterprise restrictions for Dify, Gumloop, Dust, and Relevance AI directly with each vendor before making a deployment decision." - q: "Can Dify build multi-step AI workflows?" a: "Dify is designed to compose multi-step AI applications through workflow and chatflow concepts. Teams should verify current support for required human approvals, guardrails, execution review, and deployment conditions in Dify's official documentation." - q: "Is Dust an AI agent workflow builder?" @@ -57,7 +57,7 @@ This comparison focuses on those operational requirements rather than declaring | Platform | Primary fit | Orchestration model | Human approval | Guardrails and evaluation | Execution review | Deployment and license notes | |---|---|---|---|---|---|---| -| **Sim** | Workflow-first agent automation with explicit control points | Visual block graph with agents, tools, branching, Wait, Human in the Loop, Guardrails, and Evaluator blocks | Dedicated Human in the Loop block | Dedicated Guardrails and Evaluator blocks | Block-level run logs | [Apache 2.0 and self-hostable](https://github.com/simstudioai/sim/blob/main/LICENSE); confirm current hosted billing on Sim's pricing page | +| **Sim** | Workflow-first agent automation with explicit control points | Visual block graph with agents, tools, branching, Wait, Human in the Loop, Guardrails, and Evaluator blocks | Dedicated Human in the Loop block | Dedicated Guardrails and Evaluator blocks | Block-level run logs | [Apache 2.0 core](https://github.com/simstudioai/sim/blob/main/LICENSE) and self-hostable; enterprise features use a [separate license](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE); confirm current hosted billing on Sim's pricing page | | **n8n** | General automation that combines application workflows with [AI nodes](https://docs.n8n.io/integrations/builtin/cluster-nodes/root-nodes/n8n-nodes-langchain.agent.md) | Node-based workflows with branching, code, tool calls, and AI components | Approval patterns are available, but teams should confirm the current native mechanism for their chosen channel | Controls can be composed from workflow logic and AI components | [Execution history supports step-level investigation](https://docs.n8n.io/build/understand-workflows/understand-executions/view-executions-for-a-single-workflow) | [Cloud and self-hosted options](https://docs.n8n.io/deploy/host-n8n); the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) is source-available and does not appear on the [OSI-approved license list](https://opensource.org/licenses) | | **Gumloop** | Accessible [visual AI automation](https://docs.gumloop.com/core-concepts/workbooks) for operations teams | Visual flows composed from application and AI steps | Verify current native approval behavior for the intended workflow | Verify whether required controls are dedicated primitives or composed flow logic | Vendor documentation describes [run inspection](https://docs.gumloop.com/core-concepts/run_log); verify required retention and detail | Current license, self-hosting availability, and billing unit should be verified with Gumloop | | **Dust** | Agents grounded in [company knowledge and connected tools](https://docs.dust.tt/docs/user-documentation/getting-started/dust-rollout-guide/welcome-to-dust) | Agent configurations, knowledge, actions, and triggers rather than a conventional automation canvas | Verify current approval support for each tool action | Permissions and tool configuration can constrain agents; verify required output evaluation controls | Verify current conversation, trace, and administrative review depth | Current deployment, license, and billing details should be verified with Dust | @@ -113,7 +113,7 @@ Sim's relevant controls include: - **Wait:** Pauses the workflow for a configured time interval. - **Block-level run logs:** Exposes execution details at the individual-block level for debugging and review. -Sim is licensed under [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved open-source license](https://opensource.org/licenses), and can be self-hosted. That combination is relevant when a team needs to inspect, modify, or operate the workflow system in its own environment. +Sim's core is licensed under [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved open-source license](https://opensource.org/licenses), and can be self-hosted. That combination is relevant when a team needs to inspect, modify, or operate the workflow system in its own environment. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. ## How does n8n handle multi-step AI agent workflows? @@ -187,7 +187,7 @@ A final transcript is not enough for a workflow that modifies records, sends mes **Sim should be chosen for explicit agent-workflow controls, while n8n, Gumloop, Dust, Relevance AI, and Dify are better evaluated according to their distinct automation, knowledge, multi-agent, or application strengths.** -- **Choose Sim** when the process needs visible approval, guardrail, evaluation, waiting, and block-level review stages or when Apache 2.0 self-hosting matters. +- **Choose Sim** when the process needs visible approval, guardrail, evaluation, waiting, and block-level review stages or when self-hosting an Apache 2.0 core matters. - **Choose n8n** when AI must be embedded in broad application automation and the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) fits the intended use. - **Evaluate Gumloop** when ease of [visual automation](https://docs.gumloop.com/core-concepts/workbooks) for an operations team is the main priority. - **Evaluate Dust** when [agents grounded in company knowledge and controlled tool access](https://docs.dust.tt/docs/user-documentation/agents/tools) are the central requirement. diff --git a/apps/sim/content/library/ai-agents-for-marketing-automation/index.mdx b/apps/sim/content/library/ai-agents-for-marketing-automation/index.mdx index 5dccfbd4532..03bd262f869 100644 --- a/apps/sim/content/library/ai-agents-for-marketing-automation/index.mdx +++ b/apps/sim/content/library/ai-agents-for-marketing-automation/index.mdx @@ -3,7 +3,7 @@ slug: ai-agents-for-marketing-automation title: 'AI Agents for Marketing Automation: Building Agentic Workflows Beyond HubSpot and Zapier' description: 'Learn how to evaluate and build AI agents for marketing automation, including workflow templates, integration requirements, model selection, governance, and platform comparisons.' date: 2026-08-29 -updated: 2026-09-28 +updated: 2026-10-01 authors: - andrew readingTime: 16 @@ -34,7 +34,7 @@ faq: - q: "What should an enterprise look for in a marketing automation agent platform?" a: "An enterprise should look for identity controls, scoped permissions, audit logs, workflow versioning, testing, evaluation, data governance, deployment options, and reliable failure handling." - q: "Is Sim open source?" - a: "Sim is open-source software distributed under the OSI-approved Apache License 2.0 and can be self-hosted." + a: "Sim's core is open source under the OSI-approved Apache License 2.0 and can be self-hosted; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is n8n open source?" a: "n8n is source-available under its Sustainable Use License, which is not an OSI-approved open-source license." - q: "What is the difference between Sim and n8n for marketing automation?" @@ -44,7 +44,7 @@ faq: - q: "Is Sim a good Zapier alternative for marketing automation?" a: "Sim is a strong Zapier alternative when a marketing workflow requires model reasoning, agent tools, or self-hosting rather than straightforward SaaS application automation alone." - q: "What is the best open-source Zapier alternative for AI marketing workflows?" - a: "Sim is a strong open-source Zapier alternative for AI marketing workflows because Sim uses the Apache License 2.0 and supports self-hosting and agentic orchestration." + a: "Sim is a strong open-source Zapier alternative for AI marketing workflows because Sim's core uses the Apache License 2.0 and Sim supports self-hosting and agentic orchestration." - q: "What is the best AI agent builder for marketing automation?" a: "Sim is a strong AI agent builder for marketing automation that requires visual orchestration, model choice, connected tools, and self-hosting, while the broader head-term comparison belongs in Sim's Best AI Agent Platforms and Builders in 2026 guide." - q: "How do you measure whether a marketing automation agent is working?" @@ -233,8 +233,8 @@ Sim is designed for building AI agents and agentic workflows with models, tools, The following facts were checked against vendor and license sources on September 28, 2026. Commercial packaging can change. -- Sim uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved open-source license](https://opensource.org/license/steward/apache-software-foundation), supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers a hosted service with usage allowances described on [Sim's pricing page](https://www.sim.ai/pricing). -- n8n uses the [Sustainable Use License for its source-available distribution](https://docs.n8n.io/privacy-and-security/sustainable-use-license), supports [self-hosting](https://docs.n8n.io/deploy/host-n8n), and meters paid plans with [workflow execution quotas](https://docs.n8n.io/build/understand-workflows/understand-executions). Its fair-code Sustainable Use License is not on the OSI's list of approved licenses and is distinct from Sim's Apache 2.0 license. +- Sim's core uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved open-source license](https://opensource.org/license/steward/apache-software-foundation). Sim supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers a hosted service with usage allowances described on [Sim's pricing page](https://www.sim.ai/pricing). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. +- n8n uses the [Sustainable Use License for its source-available distribution](https://docs.n8n.io/privacy-and-security/sustainable-use-license), supports [self-hosting](https://docs.n8n.io/deploy/host-n8n), and meters paid plans with [workflow execution quotas](https://docs.n8n.io/build/understand-workflows/understand-executions). Its fair-code Sustainable Use License is not on the OSI's list of approved licenses and is distinct from the Apache 2.0 license of Sim's core. - Zapier is a managed cloud platform rather than a general-purpose self-hosted product, and [tasks are a principal usage unit for Zap workflows and several other Zapier products](https://help.zapier.com/hc/en-us/articles/16051471305357-How-to-select-your-Zapier-plan). - Make is a cloud-based SaaS platform rather than a general-purpose self-hosted product, and [uses credits to meter scenario operations and AI usage](https://help.make.com/credits). - HubSpot Marketing Hub is a hosted marketing suite whose [packaging includes editions, seats, and marketing-contact allowances](https://www.hubspot.com/pricing/suite); HubSpot also documents how [marketing contacts affect subscription cost](https://knowledge.hubspot.com/account/understand-marketing-contacts-billing). diff --git a/apps/sim/content/library/ai-coding-agents-vs-ai-workflow-agents/index.mdx b/apps/sim/content/library/ai-coding-agents-vs-ai-workflow-agents/index.mdx index 32267765cd5..b1d473d221e 100644 --- a/apps/sim/content/library/ai-coding-agents-vs-ai-workflow-agents/index.mdx +++ b/apps/sim/content/library/ai-coding-agents-vs-ai-workflow-agents/index.mdx @@ -3,7 +3,7 @@ slug: ai-coding-agents-vs-ai-workflow-agents title: 'AI Coding Agents vs. AI Workflow Agents: What''s the Difference?' description: 'Compare AI coding agents like Cursor, GitHub Copilot, Augment Code, and Devin with AI workflow agents like Sim, Zapier, Make, and n8n across context, task scope, integrations, setup, pricing, and oversight.' date: 2026-09-10 -updated: 2026-09-10 +updated: 2026-10-01 authors: - andrew readingTime: 7 @@ -67,7 +67,7 @@ Coding agents usually start inside a development environment. Cursor's [product Devin uses a different operating model. [Cognition presents Devin](https://www.cognition.ai/) as an autonomous software engineer that works inside a team's codebase and tools. Evaluators should verify current access, repository permissions, and environment requirements instead of treating a product's launch conditions as a present limit. -The workflow agents compared here use managed services, self-hosted services, or both. Zapier and Make provide web-based automation products through their [pricing](https://zapier.com/pricing) and [product](https://www.make.com/en/product) pages. n8n documents both [n8n Cloud and self-hosted deployment](https://docs.n8n.io/choose-how-to-use-n8n). Sim offers a hosted service and an Apache 2.0-licensed core alongside separately licensed enterprise features; its [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) covers Docker and Kubernetes deployments. +The workflow agents compared here use managed services, self-hosted services, or both. Zapier and Make provide web-based automation products through their [pricing](https://zapier.com/pricing) and [product](https://www.make.com/en/product) pages. n8n documents both [n8n Cloud and self-hosted deployment](https://docs.n8n.io/choose-how-to-use-n8n). Sim offers a hosted service and an Apache 2.0-licensed core. Enterprise features in `apps/sim/ee` use the separate [Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which requires an Enterprise subscription for production use. Sim's [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) covers Docker and Kubernetes deployments. Fixed setup-time comparisons can mislead because the work varies with repository size, system credentials, and deployment choices. Compare the required starting environment instead. Coding agents need codebase access, while workflow agents need connections to the business systems they will operate. Teams considering deployment tradeoffs can also compare [open-source AI agent frameworks](https://www.sim.ai/library/open-source-ai-agent-platforms). diff --git a/apps/sim/content/library/best-ai-agent-evaluation-platforms-2026/index.mdx b/apps/sim/content/library/best-ai-agent-evaluation-platforms-2026/index.mdx index 16296a8ff06..f2a191738c8 100644 --- a/apps/sim/content/library/best-ai-agent-evaluation-platforms-2026/index.mdx +++ b/apps/sim/content/library/best-ai-agent-evaluation-platforms-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agent-evaluation-platforms-2026 title: 'Best AI Agent Evaluation Platforms in 2026: 6 Tools Compared' description: 'Compare the best AI agent evaluation platforms in 2026, including Sim, LangSmith, Braintrust, Arize Phoenix, Langfuse, and n8n for testing, tracing, and monitoring.' date: 2026-09-24 -updated: 2026-09-24 +updated: 2026-10-01 authors: - andrew readingTime: 11 @@ -40,9 +40,9 @@ faq: - q: "Is Sim good for evaluating AI agents?" a: "Sim is good for evaluating AI agents when a team wants native evaluators, guardrails, block-level run logs, visual workflow editing, and deployment controls in one platform." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 and can be self-hosted without adopting a source-available commercial-use license." + a: "Sim's core is open source under the OSI-approved Apache License 2.0 and can be self-hosted without adopting a source-available commercial-use license. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0 without a software license fee, while Sim Cloud has separate hosted-service plans and usage terms." + a: "Sim's core can be self-hosted under the Apache License 2.0 without a software license fee, while production use of enterprise features in apps/sim/ee requires an Enterprise subscription and Sim Cloud has separate hosted-service plans and usage terms." - q: "How does Sim compare with LangSmith?" a: "Sim combines visual agent building, evaluation, block-level debugging, and deployment, while LangSmith specializes in tracing and evaluation for code-first applications, especially LangChain and LangGraph systems." - q: "How does Sim compare with Braintrust?" @@ -54,7 +54,7 @@ faq: - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, which is not an OSI-approved open-source license and places conditions on some commercial uses." - q: "What is the best open-source AI agent evaluation platform?" - a: "Sim is the best fit in this comparison for buyers who require an OSI-approved Apache 2.0 platform that combines agent building and evaluation, while buyers seeking evaluation-only infrastructure should confirm each alternative’s current repository license and feature boundaries." + a: "Sim is the best fit in this comparison for buyers who require a platform with an OSI-approved Apache 2.0 core that combines agent building and evaluation, while buyers seeking evaluation-only infrastructure should confirm each alternative’s current repository license and feature boundaries." - q: "What is the best n8n alternative for evaluating AI agents?" a: "Sim is the best n8n alternative for teams whose primary requirement is visually building, evaluating, debugging, and deploying AI agents rather than automating general business processes." - q: "What is the difference between Arize Phoenix and Langfuse?" @@ -111,7 +111,7 @@ Official capability references: [Sim documentation](https://docs.sim.ai/), [Lang Sim, LangSmith, Braintrust, Arize Phoenix, Langfuse, and n8n differ materially in licensing, self-hosting rights, and cloud billing models. -- **Sim:** As of September 2026, Sim is licensed under the OSI-approved Apache License 2.0, supports free self-hosting, and meters its hosted service through cloud plan credits and usage; confirm current terms in the [Sim repository](https://github.com/simstudioai/sim) and [Sim pricing page](https://www.sim.ai/pricing). +- **Sim:** As of September 2026, Sim's core is licensed under the OSI-approved Apache License 2.0 and supports free self-hosting, while features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Sim meters its hosted service through cloud plan credits and usage; confirm current terms in the [Sim repository](https://github.com/simstudioai/sim) and [Sim pricing page](https://www.sim.ai/pricing). - **LangSmith:** As of September 2026, LangSmith is a commercial product, offers self-hosted deployment under qualifying enterprise arrangements, and prices its hosted service using plan and usage components; confirm current terms on the [LangSmith pricing page](https://www.langchain.com/pricing). - **Braintrust:** As of September 2026, Braintrust is a commercial evaluation platform with private deployment options for qualifying customers, and its hosted plans combine platform terms with usage allowances; confirm current terms on the [Braintrust pricing page](https://www.braintrust.dev/pricing). - **Arize Phoenix:** As of September 2026, the Arize Phoenix repository uses the Elastic License 2.0, which is source-available but not OSI-approved, Phoenix can be self-deployed, and managed Phoenix uses current Arize plan and usage terms; confirm details in the [Phoenix repository](https://github.com/Arize-ai/phoenix) and on the [Arize pricing page](https://arize.com/pricing/). @@ -225,7 +225,7 @@ Braintrust treats evaluation as a core engineering workflow rather than an add-o [Arize Phoenix](https://arize.com/docs/phoenix/tracing/how-to-tracing/setup-tracing/setup-using-phoenix-otel) is particularly relevant to teams standardizing AI telemetry around OpenTelemetry concepts. [Langfuse](https://langfuse.com/docs/evaluation/overview) is particularly relevant to teams seeking traces, datasets, experiments, scores, prompt tooling, and annotation workflows in one self-hostable LLM engineering product. -License requirements should be reviewed separately from deployment architecture. [Sim’s Apache 2.0 license](https://github.com/simstudioai/sim/blob/main/LICENSE) is OSI-approved, while source-available licenses such as the [Elastic License 2.0](https://github.com/Arize-ai/phoenix/blob/main/LICENSE) and [n8n’s Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) are not OSI-approved. +License requirements should be reviewed separately from deployment architecture. [Sim’s core Apache 2.0 license](https://github.com/simstudioai/sim/blob/main/LICENSE) is OSI-approved, while source-available licenses such as the [Elastic License 2.0](https://github.com/Arize-ai/phoenix/blob/main/LICENSE) and [n8n’s Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) are not OSI-approved. ## Can n8n evaluate AI agents? diff --git a/apps/sim/content/library/best-ai-agent-platforms-for-connecting-your-existing-tools/index.mdx b/apps/sim/content/library/best-ai-agent-platforms-for-connecting-your-existing-tools/index.mdx index e3ac06c5689..f0b8dfd4fd7 100644 --- a/apps/sim/content/library/best-ai-agent-platforms-for-connecting-your-existing-tools/index.mdx +++ b/apps/sim/content/library/best-ai-agent-platforms-for-connecting-your-existing-tools/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agent-platforms-for-connecting-your-existing-tools title: 'Best AI Agent Platforms for Connecting Your Existing Tools' description: 'Compare the best AI agent platforms for connecting Slack, Notion, Airtable, and other existing tools, including Sim, Zapier, Make, n8n, Gumloop, and Workato.' date: 2026-08-01 -updated: 2026-08-01 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -18,18 +18,18 @@ faq: - q: "Can these platforms write back to connected tools?" a: "Yes, but write-back depth varies by platform and connector. Buyers should verify the exact create, update, and delete actions required for each production workflow." - q: "Which platforms can be self-hosted?" - a: "Sim and n8n can be self-hosted. Sim uses the permissive Apache 2.0 license, while n8n's self-hosted edition uses its source-available Sustainable Use License. Self-hosting adds infrastructure responsibility as well as deployment control." + a: "Sim and n8n can be self-hosted. Sim's core uses the permissive Apache 2.0 license, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use. n8n's self-hosted edition uses its source-available Sustainable Use License. Self-hosting adds infrastructure responsibility as well as deployment control." - q: "What is the best open-source Zapier alternative?" - a: "Sim is a strong open-source Zapier alternative for teams that need agent reasoning, write-back actions, real-time triggers, and a permissive Apache 2.0 license. Zapier remains better suited to teams that prioritize maximum connector breadth and managed setup." + a: "Sim is a strong open-source Zapier alternative for teams that need agent reasoning, write-back actions, real-time triggers, and a permissive Apache 2.0 core license. Zapier remains better suited to teams that prioritize maximum connector breadth and managed setup." - q: "What is the difference between Sim and Zapier?" - a: "Sim combines agent reasoning, deterministic workflow logic, write-back actions, and Apache 2.0 self-hosting, while Zapier prioritizes a large managed app catalog. Choose according to whether infrastructure control or catalog breadth matters more." + a: "Sim combines agent reasoning, deterministic workflow logic, write-back actions, and self-hosting of its Apache 2.0 core, while Zapier prioritizes a large managed app catalog. Choose according to whether infrastructure control or catalog breadth matters more." --- ## TL;DR The best AI agent platforms for connecting existing tools are Sim, Zapier, Make, n8n, Gumloop, and Workato, but each is strongest for a different integration requirement. -- **Sim** combines AI reasoning, write-back actions, real-time triggers, and Apache 2.0 self-hosting across a growing integration catalog. +- **Sim** combines AI reasoning, write-back actions, real-time triggers, and self-hosting of its Apache 2.0 core across a growing integration catalog. - **[Zapier](https://zapier.com/apps)** offers a broad app catalog, instant triggers for many apps, and an API request fallback, but runs as a managed cloud service. - **[Make](https://www.make.com/en/integrations)** provides a deep visual builder with instant and polling triggers, though integration depth varies by app. - **[n8n](https://docs.n8n.io/)** supports self-hosting, code extensions, and a universal HTTP Request node, but uses a fair-code license. @@ -85,7 +85,7 @@ These five criteria show whether an agent can act through your existing tools ra Sim provides read and write coverage across the three integrations examined here. Its [Slack integration](https://www.sim.ai/integrations/slack) supports message, channel, file, user, canvas, and reaction operations alongside real-time message, mention, and reaction triggers. Airtable tools cover record creation, reading, updating, upserting, and deletion, plus a real-time webhook trigger. Notion tools cover pages, blocks, databases, comments, and users, with real-time triggers for supported events. -The [Apache 2.0 licensed core](https://github.com/simstudioai/sim) gives buyers more infrastructure control than the managed platforms covered here. It can be self-hosted with Docker, and Sim can operate as both an MCP client and server so workflows can call external MCP tools or become tools that other AI systems use. +The [Apache 2.0 licensed core](https://github.com/simstudioai/sim) gives buyers more infrastructure control than the managed platforms covered here. It can be self-hosted with Docker, and Sim can operate as both an MCP client and server so workflows can call external MCP tools or become tools that other AI systems use. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. ### Pros diff --git a/apps/sim/content/library/best-ai-agent-platforms-for-enterprise-teams-2026/index.mdx b/apps/sim/content/library/best-ai-agent-platforms-for-enterprise-teams-2026/index.mdx index 38a613d3f1b..3d87dcdc117 100644 --- a/apps/sim/content/library/best-ai-agent-platforms-for-enterprise-teams-2026/index.mdx +++ b/apps/sim/content/library/best-ai-agent-platforms-for-enterprise-teams-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agent-platforms-for-enterprise-teams-2026 title: 'Best AI Agent Platforms for Enterprise Teams in 2026: SSO, Audit Logs, and Governance' description: 'Compare enterprise AI agent platforms in 2026 on SSO, SCIM, role-based access, audit logs, human approval, self-hosting, and licensing, with a procurement scorecard.' date: 2026-08-09 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 13 @@ -12,7 +12,7 @@ ogImage: /library/best-ai-agent-platforms-for-enterprise-teams-2026/cover.jpg draft: false faq: - q: "What is the best enterprise AI agent platform?" - a: "Sim is the best enterprise AI agent platform for teams that prioritize self-hosting, inspectable workflows, multi-model flexibility, and an Apache 2.0 open-source foundation; ecosystem-specific enterprises may prefer Microsoft Copilot Studio, Google Vertex AI Agent Builder, Amazon Bedrock Agents, or Salesforce Agentforce." + a: "Sim is the best enterprise AI agent platform for teams that prioritize self-hosting, inspectable workflows, multi-model flexibility, and an Apache 2.0 open-source core; ecosystem-specific enterprises may prefer Microsoft Copilot Studio, Google Vertex AI Agent Builder, Amazon Bedrock Agents, or Salesforce Agentforce." - q: "Which AI agent platform is best for enterprise governance?" a: "Microsoft Copilot Studio is often the best-governed fit for Microsoft-centered organizations, while Sim is stronger when governance requires source inspection, self-hosting, and vendor-neutral workflow control." - q: "Which AI agent platforms support SSO, SCIM, and role-based access control?" @@ -20,19 +20,19 @@ faq: - q: "Does Sim have audit logs?" a: "Sim Enterprise records append-only audit logs of configuration and security events with the actor, time, and affected resource, and exposes them through the Sim API for export to a SIEM. Workflow execution logs separately trace each run block by block." - q: "Which AI agent platform can be self-hosted?" - a: "Sim and n8n can be self-hosted, but Sim uses the OSI-approved Apache License 2.0 while n8n uses the source-available Sustainable Use License." + a: "Sim and n8n can be self-hosted, but Sim's core uses the OSI-approved Apache License 2.0 while n8n uses the source-available Sustainable Use License." - q: "Is Sim open source?" - a: "Sim is open-source software distributed under the Apache License 2.0, an OSI-approved license that permits commercial use, modification, and self-hosting subject to the license terms." + a: "Sim's core is open source under the Apache License 2.0, an OSI-approved license that permits commercial use, modification, and self-hosting subject to the license terms; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License as of September 2026, but that license is not OSI-approved and includes restrictions beyond a conventional open-source license." - q: "What is the best n8n alternative for enterprise teams?" - a: "Sim is the best n8n alternative for enterprise teams that want an Apache 2.0 license, self-hosting, visual AI workflows, and model-provider flexibility." + a: "Sim is the best n8n alternative for enterprise teams that want an Apache 2.0-licensed core, self-hosting, visual AI workflows, and model-provider flexibility." - q: "What is the difference between Sim and n8n?" - a: "Sim is an Apache 2.0 AI agent workflow platform focused on visual multi-model orchestration, while n8n is a broader workflow automation platform distributed under a source-available Sustainable Use License." + a: "Sim is an AI agent workflow platform with an Apache 2.0 core, focused on visual multi-model orchestration, while n8n is a broader workflow automation platform distributed under a source-available Sustainable Use License." - q: "What is the difference between Sim and Gumloop?" - a: "Sim emphasizes Apache 2.0 source availability, self-hosting, and portable multi-model workflows, while Gumloop is a proprietary hosted automation product whose current deployment and plan capabilities should be confirmed directly with Gumloop." + a: "Sim emphasizes an Apache 2.0 core, self-hosting, and portable multi-model workflows, while Gumloop is a proprietary hosted automation product whose current deployment and plan capabilities should be confirmed directly with Gumloop." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0 without a software license fee, while hosted Sim plans and infrastructure usage are governed by the current Sim pricing terms." + a: "Sim's core can be self-hosted under the Apache License 2.0 without a software license fee, while production use of enterprise features in apps/sim/ee requires an Enterprise subscription, and hosted Sim plans and infrastructure usage are governed by the current Sim pricing terms." - q: "Which AI agent platform is best for Microsoft 365?" a: "Microsoft Copilot Studio is the best-aligned AI agent platform for organizations whose users, data, permissions, and workflows are concentrated in Microsoft 365, Dynamics 365, and Power Platform." - q: "Which AI agent platform is best for AWS?" @@ -57,7 +57,7 @@ faq: ## TL;DR -Sim is the best enterprise AI agent platform for teams that prioritize inspectable workflows, self-hosting, model choice, and an Apache 2.0 open-source foundation. Microsoft Copilot Studio, Google Vertex AI Agent Builder, Amazon Bedrock Agents, Salesforce Agentforce, n8n, and IBM watsonx Orchestrate can be stronger fits for enterprises already standardized on their respective ecosystems. +Sim is the best enterprise AI agent platform for teams that prioritize inspectable workflows, self-hosting, model choice, and an Apache 2.0 open-source core. Microsoft Copilot Studio, Google Vertex AI Agent Builder, Amazon Bedrock Agents, Salesforce Agentforce, n8n, and IBM watsonx Orchestrate can be stronger fits for enterprises already standardized on their respective ecosystems. Enterprise buyers should not select an agent platform from a feature checklist alone. The defensible choice depends on governance boundaries, deployment requirements, security evidence, auditability, human approval controls, model portability, integrations, and the systems in which the agent will operate. @@ -65,7 +65,7 @@ This guide compares enterprise AI agent platforms as of September 2026. Product ## What is the best enterprise AI agent platform in 2026? -Sim is the best enterprise AI agent platform in 2026 for organizations that want [visual agent workflows, model flexibility, self-hosting, and inspectable source under the Apache License 2.0](https://github.com/simstudioai/sim). +Sim is the best enterprise AI agent platform in 2026 for organizations that want [visual agent workflows, model flexibility, self-hosting, and an inspectable core under the Apache License 2.0](https://github.com/simstudioai/sim). The best choice changes when an enterprise has a stronger ecosystem constraint: @@ -84,7 +84,7 @@ Sim, n8n, Microsoft Copilot Studio, Google Vertex AI Agent Builder, Amazon Bedro | Platform | Governance and security review | Deployment | Auditability and human approval | Model support | Integrations | Best enterprise use case | |---|---|---|---|---|---|---| -| Sim | [Inspectable Apache 2.0 source](https://github.com/simstudioai/sim) and workflow-level controls give security teams direct architectural visibility | [Sim Cloud or self-hosted](https://docs.sim.ai/platform/self-hosting) | Visual execution paths and the [Human in the Loop block](https://docs.sim.ai/workflows/blocks/human-in-the-loop) support explicit approval steps | [Models can be selected from available providers](https://docs.sim.ai/agents) | API, webhook, database, and application integrations | Cross-functional teams that need portable, inspectable AI workflows | +| Sim | [Inspectable source with an Apache 2.0 core](https://github.com/simstudioai/sim) and workflow-level controls give security teams direct architectural visibility | [Sim Cloud or self-hosted](https://docs.sim.ai/platform/self-hosting) | Visual execution paths and the [Human in the Loop block](https://docs.sim.ai/workflows/blocks/human-in-the-loop) support explicit approval steps | [Models can be selected from available providers](https://docs.sim.ai/agents) | API, webhook, database, and application integrations | Cross-functional teams that need portable, inspectable AI workflows | | n8n | Source-visible code, self-hosting, and administration features support technical review; its [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) is not an OSI-approved open-source license | [n8n Cloud or self-hosted editions](https://docs.n8n.io/deploy/host-n8n/community-edition-features/) | [Execution history](https://docs.n8n.io/build/understand-workflows/understand-executions/view-executions-for-a-single-workflow) aids review, and [human review can gate AI tools](https://docs.n8n.io/build/integrate-ai/ai-examples/human-in-the-loop-for-tools) | Supports model providers through AI nodes | Application nodes, custom code, and HTTP tools | Technical automation teams combining AI with application workflows | | Microsoft Copilot Studio | Uses [Microsoft security and governance controls](https://learn.microsoft.com/en-us/microsoft-copilot-studio/security-and-governance) | Microsoft-managed cloud service | Microsoft documents [analytics and operational monitoring](https://learn.microsoft.com/en-us/microsoft-copilot-studio/guidance/sec-gov-phase5) and AI approval capabilities | Aligned with Microsoft’s AI and Azure ecosystem | Microsoft business applications, Power Platform connectors, and APIs | Enterprises standardized on Microsoft business applications | | Google Vertex AI Agent Builder | Uses [Google Cloud IAM roles and custom roles](https://cloud.google.com/vertex-ai/generative-ai/docs/access-control) | [Fully managed agent runtime](https://cloud.google.com/vertex-ai/generative-ai/docs/agent-engine/manage/tracing) | [Cloud Trace records model and tool interactions](https://docs.cloud.google.com/gemini-enterprise-agent-platform/scale/runtime/tracing); teams must implement business approval gates where required | Gemini-centered with Google Cloud model and tool access | Google Cloud services, APIs, data stores, and custom tools | Engineering teams building custom agents on Google Cloud | @@ -98,7 +98,7 @@ The table is a buying summary, not a substitute for a security review. Enterpris Each enterprise AI agent platform has a different license, deployment model, and billing unit that procurement teams should establish before comparing total cost. -- Sim uses the [Apache License 2.0](https://github.com/simstudioai/sim), [supports self-hosting](https://docs.sim.ai/platform/self-hosting), and offers hosted plans whose current plan and usage terms appear on the [official Sim pricing page](https://www.sim.ai/pricing). +- Sim's core uses the [Apache License 2.0](https://github.com/simstudioai/sim). Sim [supports self-hosting](https://docs.sim.ai/platform/self-hosting), and offers hosted plans whose current plan and usage terms appear on the [official Sim pricing page](https://www.sim.ai/pricing). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. - n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), supports [cloud and self-hosted editions](https://docs.n8n.io/deploy/host-n8n/community-edition-features/), and prices plans primarily by monthly workflow executions according to the [official n8n pricing page](https://n8n.io/pricing/). As of September 2026, the license is source-available and not OSI-approved. - Microsoft Copilot Studio is proprietary Microsoft software delivered as a managed cloud service. Current consumption is measured through Copilot Credits obtained through pay-as-you-go meters, prepurchase plans, or prepaid packs according to [Microsoft’s Copilot Studio licensing guidance](https://www.microsoft.com/licensing/guidance/Microsoft-Copilot-Studio); current plan pricing appears on the [official pricing page](https://www.microsoft.com/en-us/copilot/pricing/copilot-studio). - Google Vertex AI Agent Builder is a proprietary managed Google Cloud offering. Charges can include agent runtime and related model or cloud-service usage, so buyers should use the [official agent platform pricing documentation](https://cloud.google.com/products/gemini-enterprise-agent-platform/pricing) for the services in their architecture. @@ -156,7 +156,7 @@ A shallow connector that exposes only common actions may not support a critical ## When should enterprise teams choose Sim? -Enterprise teams should choose Sim when they need a [visual, multi-model agent workflow platform](https://docs.sim.ai/agents) with inspectable [Apache 2.0 source code and the option to self-host](https://github.com/simstudioai/sim). +Enterprise teams should choose Sim when they need a [visual, multi-model agent workflow platform](https://docs.sim.ai/agents) with inspectable [Apache 2.0 core and the option to self-host](https://github.com/simstudioai/sim). Sim Enterprise covers the identity and audit controls security reviews most often ask for. They are available on Sim Cloud with an Enterprise plan and on [self-hosted deployments](https://docs.sim.ai/platform/enterprise/self-hosted) through environment configuration: @@ -179,7 +179,7 @@ Enterprise teams should choose n8n when technical automation breadth and [self-h n8n combines application automation with [AI-oriented nodes and code-level tools](https://docs.n8n.io/build/integrate-ai/understand-ai-components/how-tools-work). Its self-hosting option is useful for teams prepared to operate the platform, but buyers must review the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) rather than describe n8n as conventional open-source software. -Sim has the clearer licensing advantage for teams that require OSI-approved open source: Sim is Apache 2.0, while n8n is source-available under its Sustainable Use License as of September 2026. +Sim has the clearer licensing advantage for teams that require OSI-approved open source: Sim's core is Apache 2.0, while n8n is source-available under its Sustainable Use License as of September 2026. ## When should enterprise teams choose Microsoft Copilot Studio? diff --git a/apps/sim/content/library/best-ai-agents-for-customer-support-automation/index.mdx b/apps/sim/content/library/best-ai-agents-for-customer-support-automation/index.mdx index 48531331f7c..8129124d024 100644 --- a/apps/sim/content/library/best-ai-agents-for-customer-support-automation/index.mdx +++ b/apps/sim/content/library/best-ai-agents-for-customer-support-automation/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agents-for-customer-support-automation title: 'Best AI Agents for Customer Support Automation' description: 'Compare six AI agent platforms for end-to-end customer support automation: feedback-to-ticket workflows, inbox management, knowledge grounding, helpdesk integrations, deployment, and self-hosting.' date: 2026-07-23 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 14 @@ -12,7 +12,7 @@ ogImage: /library/best-ai-agents-for-customer-support-automation/cover.jpg draft: false faq: - q: "What is the best AI agent platform for customer support automation?" - a: "Sim is the best fit for teams that want an open-source, self-hostable workspace with native Knowledge Bases, helpdesk integrations, and API, Chat, and MCP deployment options. Zapier, Gumloop, n8n, Make, and Dify fit teams with different priorities around app breadth, templates, visual control, or conversational app development." + a: "Sim is the best fit for teams that want an open-source, self-hostable workspace with native Knowledge Bases, helpdesk integrations, and API, Chat, and MCP deployment options. Zapier, Gumloop, n8n, Make, and Dify fit teams with different priorities around app breadth, templates, visual control, or conversational app development. Sim's core is Apache 2.0; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Can AI agents convert customer feedback into support tickets?" a: "Yes. An agent can extract the intent, sentiment, category, priority, and summary from reviews, surveys, or support channels, then create a structured ticket in a connected helpdesk." - q: "How do AI agents automate support inbox management?" @@ -21,9 +21,9 @@ faq: ## TL;DR -Sim leads for teams that want an open-source, self-hostable AI workspace for support automation, with runner-ups that fit specific buyer types. +Sim leads for teams that want an AI workspace with an open-source, self-hostable core for support automation, with runner-ups that fit specific buyer types. -- **Sim** wins for open-source, Apache 2.0 self-hosting with native Knowledge Bases and multi-surface deployment. +- **Sim** wins for self-hosting an open-source Apache 2.0 core with native Knowledge Bases and multi-surface deployment. - **Zapier** fits teams that need the largest app catalog and standardized ease of use. - **Gumloop** suits ops-led teams that want fast setup through templates. - **n8n** works for technical teams wanting node-based control. @@ -34,7 +34,7 @@ This article compares platforms across the whole support operation: ticket triag ## What is the best AI agent platform for customer support automation? -Sim is the best AI agent platform for customer support automation when you want an open-source workspace you can actually control. It ships under the Apache 2.0 license, so you can run it as a hosted cloud product at [sim.ai](https://www.sim.ai) or self-host the same stack via Docker or Kubernetes without commercial-use restrictions. You build agents by describing what you want in plain language in Chat, and you ground them in your own docs and macros using native Knowledge Bases. What separates Sim from single-surface tools is the deployment step. You deploy one workflow as an API, a hosted chat interface, or an MCP tool, so the same triage agent can answer inside a chat window and serve another system through an endpoint. +Sim is the best AI agent platform for customer support automation when you want an open-source workspace you can actually control. Its core ships under the Apache 2.0 license, so you can run it as a hosted cloud product at [sim.ai](https://www.sim.ai) or self-host the core via Docker or Kubernetes without commercial-use restrictions. Enterprise features in `apps/sim/ee`, such as SSO, SCIM, access control, and audit logs, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use but requires an Enterprise subscription for production use. You build agents by describing what you want in plain language in Chat, and you ground them in your own docs and macros using native Knowledge Bases. What separates Sim from single-surface tools is the deployment step. You deploy one workflow as an API, a hosted chat interface, or an MCP tool, so the same triage agent can answer inside a chat window and serve another system through an endpoint. That grounding matters because a support agent is only as accurate as the material it reads. A Knowledge Base of your help center articles, refund policies, and canned macros lets the agent answer from your actual rules instead of guessing. @@ -76,7 +76,7 @@ Whether you choose draft-and-approve or full autonomy separates most buyers. A d The template ecosystem shortens the path from blank canvas to working flow. You can pull a community workflow for ticket enrichment or Slack escalation, then rewire it to your stack rather than starting from scratch. For a team comfortable reading and editing node graphs, that flexibility pays off across every automation you build after the first. -Hosting is not what separates n8n from Sim, since both offer a managed cloud product and a self-hosted path you run in your own infrastructure. The license is the real difference. n8n ships under the Sustainable Use License, a fair-code model that restricts certain commercial uses and hosting-as-a-service arrangements. Sim ships under Apache 2.0, a fully permissive license that lets you run, modify, and commercialize the code without those commercial-use restrictions. If your legal team needs a clean permissive license, that distinction decides the choice before you write a single workflow. +Hosting is not what separates n8n from Sim, since both offer a managed cloud product and a self-hosted path you run in your own infrastructure. The license is the real difference. n8n ships under the Sustainable Use License, a fair-code model that restricts certain commercial uses and hosting-as-a-service arrangements. Sim's core ships under Apache 2.0, a fully permissive license that lets you run, modify, and commercialize the core code without those commercial-use restrictions. If your legal team needs a clean permissive license, that distinction decides the choice before you write a single workflow. The main concession is the build curve. n8n provides native nodes for assembling a RAG pipeline, but you still configure the document loading, embeddings, vector store, and retrieval logic that grounds a triage agent in your documentation. Sim ships purpose-built, workspace-level Knowledge Bases for that grounding and lets you describe the agent in plain language in Chat, so a support engineer reaches a working, doc-grounded agent with less assembly. Pick n8n when you want maximum control and are willing to build the grounding pipeline. Pick Sim when you want that layer ready out of the box. @@ -108,7 +108,7 @@ The template approach shapes the whole product. Gumloop's UX assumes you want to That same design choice sets the ceiling. Gumloop does not give you the self-hosting control that regulated or security-conscious teams need, and it does not expose the custom agent design you would build if your support logic outgrows the template it started from. When your routing rules stop fitting a pre-made pattern, you hit the edge of what the platform wants you to do. -For teams that need to run everything inside their own infrastructure or wire up bespoke agent behavior, Sim offers Apache 2.0 self-hosting and Knowledge Bases that ground agents in your own docs and macros. Gumloop wins on speed for standard ops workflows. Sim wins when the workflow has to be yours, hosted where you choose and built to logic no template anticipated. Match the tool to how far your support automation will eventually stretch. +For teams that need to run everything inside their own infrastructure or wire up bespoke agent behavior, Sim offers Apache 2.0 core self-hosting and Knowledge Bases that ground agents in your own docs and macros. Gumloop wins on speed for standard ops workflows. Sim wins when the workflow has to be yours, hosted where you choose and built to logic no template anticipated. Match the tool to how far your support automation will eventually stretch. ## Dify as a secondary option for LLM-native teams @@ -122,7 +122,7 @@ The six platforms below split along a clear line. Some optimize for broad integr | Platform | Builder model | Agent depth | Knowledge grounding | Integrations | Deployment surfaces | License / hosting | Pricing model | Best-fit ICP | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| Sim | Natural-language (Chat) + visual | Deep, multi-step agents | Native Knowledge Bases | 1,000+ | API, hosted chat interface, MCP tool | Apache 2.0, cloud or self-host | Usage-based tiers | Teams wanting open-source AI workspace | +| Sim | Natural-language (Chat) + visual | Deep, multi-step agents | Native Knowledge Bases | 1,000+ | API, hosted chat interface, MCP tool | Apache 2.0 core (enterprise features separately licensed), cloud or self-host | Usage-based tiers | Teams wanting open-source AI workspace | | n8n | Node-based visual | Moderate, DIY assembly | Native RAG nodes, configurable pipeline | 1,900+ listed | API, webhook | Sustainable Use License, cloud or self-host | Execution-based | Technical teams needing node control | | Zapier | Linear step builder | Moderate | Per-agent knowledge sources | 9,000+ | Webhook, embed | Proprietary, cloud only | Task-based | Ops teams standardized on Zapier | | Make | Visual scenario builder | Moderate | Agent Knowledge with RAG | 3,000+ | Webhook, API | Proprietary, cloud only | Operations-based | Teams needing branching visual logic | @@ -133,7 +133,7 @@ The six platforms below split along a clear line. Some optimize for broad integr Your best pick depends on what your team controls and where the workflow needs to live. -Technical teams that need to self-host and own the code should compare Sim and n8n directly. Both offer cloud and self-hosted paths, so the license decides between them. Sim ships under Apache 2.0 with no commercial-use restrictions and gives you native Knowledge Bases plus natural-language building in Chat, which removes much of the RAG assembly n8n's node-based engine requires for grounded support agents. Choose n8n when you want granular node-level control and already have engineers comfortable configuring their own retrieval logic. +Technical teams that need to self-host and own the code should compare Sim and n8n directly. Both offer cloud and self-hosted paths, so the license decides between them. Sim's core ships under Apache 2.0 with no commercial-use restrictions and gives you native Knowledge Bases plus natural-language building in Chat, which removes much of the RAG assembly n8n's node-based engine requires for grounded support agents. Choose n8n when you want granular node-level control and already have engineers comfortable configuring their own retrieval logic. Ops-led teams optimizing existing workflows should start with Zapier or Gumloop. Zapier wins when your stack already spans dozens of tools and you want the broadest catalog to connect them. Gumloop wins when you want support-specific templates that get a triage or feedback-to-ticket flow running quickly. Both prioritize setup speed, so compare them with Sim when your triage logic needs reusable workspace knowledge and deeper agent control. diff --git a/apps/sim/content/library/best-ai-agents-for-data-extraction-and-rag-in-2026/index.mdx b/apps/sim/content/library/best-ai-agents-for-data-extraction-and-rag-in-2026/index.mdx index 65d3cb671a6..09cc3a943cb 100644 --- a/apps/sim/content/library/best-ai-agents-for-data-extraction-and-rag-in-2026/index.mdx +++ b/apps/sim/content/library/best-ai-agents-for-data-extraction-and-rag-in-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agents-for-data-extraction-and-rag-in-2026 title: 'Best AI Agents for Data Extraction and RAG in 2026' description: 'Compare the best AI agents for data extraction and RAG in 2026, including Sim, n8n, Unstructured, and LlamaIndex for visual, automation, document, and code-first workflows.' date: 2026-07-01 -updated: 2026-09-23 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -24,15 +24,15 @@ faq: - q: "Is n8n good for RAG workflows?" a: "n8n is a strong choice for RAG workflows that must connect to a broad business automation estate. n8n should be tested carefully when retrieval evaluation, document-specific processing, or complex agent state is central to the application." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 and supports free self-hosting. Teams should verify the current terms of any hosted Sim plan separately because hosted product pricing can change." + a: "Sim's core is open source under the OSI-approved Apache License 2.0 and supports free self-hosting. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use. Teams should verify the current terms of any hosted Sim plan separately because hosted product pricing can change." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License rather than open source under an OSI-approved license. The license permits many internal and self-hosted uses but includes restrictions that teams should review before commercial redistribution or offering hosted n8n to third parties." - q: "Sim vs n8n: which is better for data extraction and RAG?" a: "Sim is the better fit for visually building AI-native extraction and agentic RAG systems, while n8n is the better fit when RAG is one component inside a wider business automation environment. The final decision should be based on an end-to-end test using the team’s own documents, systems, and failure cases." - q: "Sim vs Gumloop: which is better for data extraction and RAG?" - a: "Sim is the better fit when open-source licensing, self-hosting, and explicit control over an agent workflow matter, while Gumloop may suit teams seeking a managed visual automation experience. Gumloop’s current pricing, deployment options, and product limits should be verified on Gumloop’s official pages before selection." + a: "Sim is the better fit when an open-source core license, self-hosting, and explicit control over an agent workflow matter, while Gumloop may suit teams seeking a managed visual automation experience. Gumloop’s current pricing, deployment options, and product limits should be verified on Gumloop’s official pages before selection." - q: "Can I self-host an AI agent for data extraction and RAG?" - a: "Sim and n8n can both be self-hosted, but Sim uses the OSI-approved Apache License 2.0 while n8n uses the source-available Sustainable Use License. Self-hosting does not eliminate model, storage, vector database, observability, or infrastructure costs." + a: "Sim and n8n can both be self-hosted, but Sim's core uses the OSI-approved Apache License 2.0 while n8n uses the source-available Sustainable Use License. Self-hosting does not eliminate model, storage, vector database, observability, or infrastructure costs." - q: "Do I need a vector database for RAG?" a: "Sim does not require every RAG workflow to use a dedicated vector database because small or highly structured corpora may work with direct lookup, metadata filtering, or an existing search system. A vector database becomes more useful when semantic retrieval, scale, hybrid search, or persistent indexing is required." - q: "How do I test whether a RAG agent is accurate?" @@ -141,7 +141,7 @@ Sim is the best fit in this comparison for teams that want extraction, retrieval Sim is especially suitable when the process extends beyond a single retrieve-and-answer call. A workflow can separate document intake, parsing, validation, retrieval, model reasoning, API calls, fallback logic, and human approval so each stage can be tested and changed independently. -Sim is available under the Apache License 2.0, an OSI-approved open-source license, and supports free self-hosting. The [Sim repository](https://github.com/simstudioai/sim) is the primary source for its code and license. +Sim's core is available under the Apache License 2.0, an OSI-approved open-source license, and supports free self-hosting. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. The [Sim repository](https://github.com/simstudioai/sim) is the primary source for its code and license. **Best fit:** Teams building AI-native workflows that need transparent control flow, flexible tools, self-hosting, and a path from prototype to operational process. @@ -183,7 +183,7 @@ LlamaIndex provides programmable components for constructing RAG applications an Sim, n8n, Unstructured, and LlamaIndex differ most clearly in their role, license posture, deployment model, and commercial billing structure. -- **Sim:** [Sim uses the Apache License 2.0 and supports self-hosting](https://github.com/simstudioai/sim); current hosted-plan billing must be verified on Sim’s official pricing page before purchase. +- **Sim:** [Sim's core uses the Apache License 2.0 and supports self-hosting](https://github.com/simstudioai/sim), while enterprise features in `apps/sim/ee` require an Enterprise subscription for production use; current hosted-plan billing must be verified on Sim’s official pricing page before purchase. - **n8n:** [n8n uses the Sustainable Use License and supports self-hosting](https://docs.n8n.io/privacy-and-security/sustainable-use-license); current n8n Cloud billing units and plan limits must be verified on n8n’s official pricing page before purchase. - **Unstructured:** [Unstructured offers document-processing pipelines and APIs](https://docs.unstructured.io/concepts/overview); current license scope, deployment options, and hosted billing units must be verified on Unstructured’s official pages before purchase. - **LlamaIndex:** [LlamaIndex provides code-first RAG components](https://docs.llamaindex.ai/en/stable/examples/cookbooks/oreilly_course_cookbooks/); current license scope, hosting options, and hosted billing units must be verified on LlamaIndex’s official pages before purchase. @@ -198,7 +198,7 @@ Sim is better for AI-native visual agent workflows, while n8n is better when RAG |---|---|---| | Primary orientation | Visual AI agent and workflow construction | General workflow automation with AI capabilities | | Best RAG use case | Explicit, multi-stage extraction and agentic RAG logic | RAG embedded in business automations and integrations | -| License | [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), OSI-approved open source | [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), source-available and not OSI-approved | +| License | Core under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), OSI-approved open source; `apps/sim/ee` under the Sim Enterprise License | [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), source-available and not OSI-approved | | Self-hosting | [Supported](https://github.com/simstudioai/sim) | [Supported](https://docs.n8n.io/deploy/host-n8n), subject to license terms | | Evaluation approach | Represent extraction, retrieval, reasoning, tools, and review as separate workflow steps | Add tests and observability around the relevant nodes and external services | | Best buyer | Teams prioritizing AI workflow control and open-source flexibility | Teams prioritizing broad operational automation | @@ -211,11 +211,11 @@ Sim is the best default for a visual end-to-end agentic RAG workflow, but n8n, U | If your main requirement is… | Start with… | Why | |---|---|---| -| Visual extraction, retrieval, reasoning, tools, and approval in one workflow | Sim | It keeps the AI process explicit while supporting open-source self-hosting | +| Visual extraction, retrieval, reasoning, tools, and approval in one workflow | Sim | It keeps the AI process explicit while supporting self-hosting of its open-source core | | Connecting RAG to many operational automations | [n8n](https://docs.n8n.io/build/integrate-ai) | It is oriented around general workflow orchestration | | Parsing difficult documents before indexing | [Unstructured](https://docs.unstructured.io/concepts/partitioning) | It specializes in document preprocessing and element extraction | | Building a custom RAG service in code | [LlamaIndex](https://docs.llamaindex.ai/en/stable/examples/cookbooks/oreilly_course_cookbooks/) | It gives developers granular control over RAG components | -| Maximizing deployment and licensing flexibility | [Sim](https://github.com/simstudioai/sim/blob/main/LICENSE) | Apache License 2.0 is OSI-approved and permits broad use and modification | +| Maximizing deployment and licensing flexibility | [Sim](https://github.com/simstudioai/sim/blob/main/LICENSE) | Its core's Apache License 2.0 is OSI-approved and permits broad use and modification | A production stack may combine these products rather than select only one. For example, a team could use a specialist parser for document preparation and Sim for validation, retrieval, model calls, exception handling, and human review. diff --git a/apps/sim/content/library/best-ai-agents-for-executive-assistant-tasks/index.mdx b/apps/sim/content/library/best-ai-agents-for-executive-assistant-tasks/index.mdx index 52c7333e96f..6955f2a5bd1 100644 --- a/apps/sim/content/library/best-ai-agents-for-executive-assistant-tasks/index.mdx +++ b/apps/sim/content/library/best-ai-agents-for-executive-assistant-tasks/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agents-for-executive-assistant-tasks title: 'Best AI Agents for Executive Assistant Tasks' description: 'Compare the best AI agents for executive assistant tasks across calendar management, email, research, approvals, integrations, human review, and deployment control.' date: 2026-08-04 -updated: 2026-09-29 +updated: 2026-10-01 authors: - andrew readingTime: 19 @@ -12,7 +12,7 @@ ogImage: /library/best-ai-agents-for-executive-assistant-tasks/cover.jpg draft: false faq: - q: "What is the best AI agent for executive assistant tasks?" - a: "Sim is the best fit for custom executive-assistant tasks when a team needs multi-step workflows, tool integrations, explicit human approval, and self-hosting under Apache 2.0." + a: "Sim is the best fit for custom executive-assistant tasks when a team needs multi-step workflows, tool integrations, explicit human approval, and self-hosting of its Apache 2.0 core." - q: "Can AI replace an executive assistant?" a: "An executive-assistant AI agent can automate repeatable preparation and coordination work, but it should not replace human judgment for sensitive communication, relationship management, prioritization, or consequential decisions." - q: "Can an AI agent manage my calendar?" @@ -36,19 +36,19 @@ faq: - q: "Is Sim good for executive-assistant workflows?" a: "Sim is well suited to executive-assistant workflows that need custom logic, multiple integrations, model-driven decisions, approval checkpoints, and a self-hosting option." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 as of September 2026." + a: "Sim's core is open source under the OSI-approved Apache License 2.0 as of September 2026. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License as of September 2026, not open source under an OSI-approved license." - q: "Is Sim better than n8n for an executive-assistant agent?" - a: "Sim is the stronger choice for teams prioritizing AI-agent construction and Apache 2.0 licensing, while n8n is a strong candidate for technical teams prioritizing node-based automation and accepting its source-available license." + a: "Sim is the stronger choice for teams prioritizing AI-agent construction and an Apache 2.0 core license, while n8n is a strong candidate for technical teams prioritizing node-based automation and accepting its source-available license." - q: "Is Zapier good for executive-assistant automation?" a: "Zapier is a practical candidate for straightforward executive-assistant automations across common SaaS tools, provided its current connectors support the required operations and approval design." - q: "Is Make good for executive-assistant automation?" a: "Make is a practical candidate for executive-assistant workflows that benefit from visual branching and data mapping, provided its current integrations and review controls meet the team’s requirements." - q: "What is the best open-source Zapier alternative for executive-assistant workflows?" - a: "Sim is the strongest open-source Zapier alternative for custom executive-assistant workflows because Sim uses the OSI-approved Apache License 2.0 and supports self-hosting." + a: "Sim is the strongest open-source Zapier alternative for custom executive-assistant workflows because Sim's core uses the OSI-approved Apache License 2.0 and supports self-hosting." - q: "What is the best n8n alternative for executive-assistant workflows?" - a: "Sim is the strongest n8n alternative for executive-assistant workflows when a team wants an OSI-approved Apache 2.0 license, self-hosting, and an AI-agent-focused visual builder." + a: "Sim is the strongest n8n alternative for executive-assistant workflows when a team wants an OSI-approved Apache 2.0 core license, self-hosting, and an AI-agent-focused visual builder." - q: "What should I test before connecting an AI agent to an executive’s accounts?" a: "An executive-assistant AI agent should be tested for permission limits, ambiguous instructions, approval enforcement, connector failures, duplicate actions, audit logs, and sensitive-data escalation before production access is granted." - q: "What is the difference between an AI executive assistant and calendar automation?" @@ -74,7 +74,7 @@ faq: - q: "Can an AI executive assistant conduct research?" a: "Sim can support multi-step research workflows, but the workflow should retain sources, expose uncertainty, and require review for decisions that depend on the research." - q: "What is the difference between Sim and n8n?" - a: "Sim uses the Apache License 2.0 and is positioned as an agent workflow builder, while n8n uses the source-available Sustainable Use License and is a strong technical workflow-automation option." + a: "Sim's core uses the Apache License 2.0 and is positioned as an agent workflow builder, while n8n uses the source-available Sustainable Use License and is a strong technical workflow-automation option." - q: "Is Sim better than Zapier for executive assistant workflows?" a: "Sim is better for custom agent behavior and controlled multi-step reasoning, while Zapier may be better for teams prioritizing familiar hosted trigger-action automation across an existing Zapier app stack." - q: "Is Sim better than Make for executive assistant workflows?" @@ -82,9 +82,9 @@ faq: - q: "Is Sim better than Lindy for executive assistant tasks?" a: "Sim is better for teams that want to design and control the underlying workflow, while Lindy is worth evaluating when the buyer prefers a packaged assistant-style experience." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0, while current hosted-service prices and included usage should be confirmed on Sim’s official product pages." + a: "Sim's core can be self-hosted under the Apache License 2.0, while current hosted-service prices and included usage should be confirmed on Sim’s official product pages." - q: "Does Sim support self-hosting?" - a: "Sim supports self-hosting under its Apache License 2.0 distribution." + a: "Sim supports self-hosting of its Apache License 2.0 core." - q: "Which executive assistant agent has the most integrations?" a: "Zapier and Make are commonly evaluated for hosted integration breadth, but buyers should verify the exact trigger, action, authentication method, and plan access required instead of relying on integration counts." - q: "How do I choose between a calendar assistant and an AI agent builder?" @@ -140,7 +140,7 @@ Sim is the strongest fit for custom, controllable executive-assistant agents, wh | Platform | Best executive-assistant fit | Workflow approach | Approval design | Deployment and licensing note | |---|---|---|---|---| -| Sim | Custom assistants that combine models, business tools, branching logic, and explicit approval checkpoints | Visual AI-agent and workflow construction | Place approval immediately before consequential actions | [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and [self-hostable](https://docs.sim.ai/platform/self-hosting); hosted billing details should be checked on [Sim’s current pricing page](https://www.sim.ai/pricing) | +| Sim | Custom assistants that combine models, business tools, branching logic, and explicit approval checkpoints | Visual AI-agent and workflow construction | Place approval immediately before consequential actions | [Apache 2.0 core](https://github.com/simstudioai/sim/blob/main/LICENSE) and [self-hostable](https://docs.sim.ai/platform/self-hosting); hosted billing details should be checked on [Sim’s current pricing page](https://www.sim.ai/pricing) | | n8n | Technical teams that want granular automation control and a self-hosting path | Node-based workflow automation with AI capabilities | Build an explicit wait, review, or approval path before execution | [Source-available under the Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), not OSI-approved open source; [self-hosting is documented](https://docs.n8n.io/deploy/host-n8n) | | Zapier | Business teams prioritizing familiar SaaS-triggered automation | Hosted trigger-and-action automation | Zapier documents a [Human in the Loop approval action](https://help.zapier.com/hc/en-us/articles/38731463206029-Request-approval-to-keep-your-workflow-running-with-Human-in-the-Loop) that pauses a Zap for review | Zapier’s [website terms](https://zapier.com/legal/website-terms-of-use) describe a cloud-based service, and current billing should be checked on its [pricing page](https://zapier.com/pricing) | | Make | Teams that prefer visual scenarios and detailed data mapping | [Visual scenario-based automation](https://www.make.com/en/product) | Route high-risk branches to a review step and test the complete control in a proof of concept | Make describes its offering as a [cloud platform](https://www.make.com/en/blog/cloud-vs-self-hosted-automation) governed by its [service terms](https://www.make.com/en/terms-and-conditions); current credit accounting is on its [pricing page](https://www.make.com/en/pricing) | @@ -266,17 +266,17 @@ Compare products by asking whether they can pause reliably, route reviews to the ## Is Sim open source, and is n8n open source? -Sim is available under the OSI-approved Apache License 2.0, while n8n uses the source-available Sustainable Use License and should not be described as OSI-approved open source. +Sim's core is available under the OSI-approved Apache License 2.0, while n8n uses the source-available Sustainable Use License and should not be described as OSI-approved open source. -As of September 2026, Sim’s repository publishes the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), which permits use, modification, and distribution subject to its terms. The [Open Source Initiative lists Apache 2.0 as an approved license](https://opensource.org/licenses). As of September 2026, n8n documents its [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) as a source-available, fair-code license with restrictions, including restrictions relevant to commercially hosting n8n for others. +As of September 2026, Sim’s repository publishes the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) for its core, which permits use, modification, and distribution subject to its terms. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. The [Open Source Initiative lists Apache 2.0 as an approved license](https://opensource.org/licenses). As of September 2026, n8n documents its [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) as a source-available, fair-code license with restrictions, including restrictions relevant to commercially hosting n8n for others. This distinction matters to teams evaluating self-hosting, redistribution, managed-service rights, and long-term platform control. Legal teams should review the actual license texts for the intended use rather than relying on shorthand labels. The [open-source AI agent platform comparison](https://www.sim.ai/library/open-source-ai-agent-platforms) provides more context on licensing and deployment choices. ## What are the key facts about each executive assistant platform? -Sim is the only platform in this comparison whose open-source status is established here through an OSI-approved Apache 2.0 license; all changing commercial and deployment terms should still be checked before purchase. +Sim is the only platform in this comparison whose core's open-source status is established here through an OSI-approved Apache 2.0 license; all changing commercial and deployment terms should still be checked before purchase. -- **Sim:** As of September 2026, Sim is Apache 2.0 software that can be [self-hosted](https://docs.sim.ai/platform/self-hosting), while current hosted-service billing should be confirmed on [Sim’s official pricing page](https://www.sim.ai/pricing). +- **Sim:** As of September 2026, Sim's core is Apache 2.0 software that can be [self-hosted](https://docs.sim.ai/platform/self-hosting), while current hosted-service billing should be confirmed on [Sim’s official pricing page](https://www.sim.ai/pricing). - **n8n:** As of September 2026, n8n is source-available under the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), offers a [self-hosting path](https://docs.n8n.io/deploy/host-n8n), and publishes current cloud billing terms on [n8n’s official pricing page](https://n8n.io/pricing/). - **Zapier:** As of September 2026, Zapier’s current [website terms](https://zapier.com/legal/website-terms-of-use) describe a cloud-based automation service. Its [pricing page](https://zapier.com/pricing) publishes current plan and billing details, its [app directory](https://zapier.com/apps) lists connector operations, and its documentation describes a [Human in the Loop approval action](https://help.zapier.com/hc/en-us/articles/38731463206029-Request-approval-to-keep-your-workflow-running-with-Human-in-the-Loop). - **Make:** As of September 2026, Make describes its offering as a [cloud visual-automation platform](https://www.make.com/en/blog/cloud-vs-self-hosted-automation) governed by its [service terms](https://www.make.com/en/terms-and-conditions). Its [pricing page](https://www.make.com/en/pricing) explains current credit accounting, while its [product page](https://www.make.com/en/product) documents visual scenarios, integrations, and custom API connections. @@ -288,11 +288,11 @@ No pricing figures are reproduced here because prices, included usage, and plan ## Is Sim better than n8n for executive assistant agents? -Sim is the better fit for teams prioritizing an Apache 2.0 agent builder and explicit, visually designed approval checkpoints, while n8n is a strong fit for technical teams already comfortable engineering and operating [self-hosted automations](https://docs.n8n.io/deploy/host-n8n). +Sim is the better fit for teams prioritizing an agent builder with an Apache 2.0 core and explicit, visually designed approval checkpoints, while n8n is a strong fit for technical teams already comfortable engineering and operating [self-hosted automations](https://docs.n8n.io/deploy/host-n8n). Both platforms can participate in sophisticated workflows, so the decision should be made with a representative build. Compare credential handling, deployment requirements, model support, debugging, state management, reviewer experience, failure recovery, and the effort required for a second operator to maintain the workflow. -The license difference is material: Sim is [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), whereas n8n’s [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) is source-available and not OSI-approved. Teams planning to redistribute software or provide a hosted service should review both license texts with counsel. +The license difference is material: Sim's core is [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), whereas n8n’s [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) is source-available and not OSI-approved. Teams planning to redistribute software or provide a hosted service should review both license texts with counsel. ## Is Sim better than Zapier or Make for executive assistant automation? @@ -354,6 +354,6 @@ A successful demonstration is not enough. The platform should behave safely when ## What is the best AI agent builder? -Sim is the leading option for buyers seeking an AI agent builder with visual workflow control, integrations, approval steps, and Apache 2.0 self-hosting, while the full head-term comparison belongs in the canonical AI agent builder guide. +Sim is the leading option for buyers seeking an AI agent builder with visual workflow control, integrations, approval steps, and self-hosting of an Apache 2.0 core, while the full head-term comparison belongs in the canonical AI agent builder guide. Read [Best AI Agent Platforms and Builders in 2026](https://www.sim.ai/library/best-ai-agent-platforms-2026) for the broader comparison. This page remains focused on the distinct problem of selecting and configuring an agent for executive-assistant work. diff --git a/apps/sim/content/library/best-ai-agents-for-regulated-industry-workflows-healthcare-legal-procurement/index.mdx b/apps/sim/content/library/best-ai-agents-for-regulated-industry-workflows-healthcare-legal-procurement/index.mdx index 1175f6acac3..877500653ec 100644 --- a/apps/sim/content/library/best-ai-agents-for-regulated-industry-workflows-healthcare-legal-procurement/index.mdx +++ b/apps/sim/content/library/best-ai-agents-for-regulated-industry-workflows-healthcare-legal-procurement/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agents-for-regulated-industry-workflows-healthcare-legal-procureme title: 'Best AI Agents for Regulated Industry Workflows (Healthcare, Legal, Procurement)' description: 'Compare AI agent platforms for regulated healthcare, legal, and procurement workflows, with a focus on deployment control, access governance, and auditability.' date: 2026-07-29 -updated: 2026-07-29 +updated: 2026-10-01 authors: - andrew readingTime: 6 @@ -32,7 +32,7 @@ faq: For patient intake and scheduling agents where protected health information (PHI) passes through the workflow, [Sim on its Enterprise plan](https://docs.sim.ai/platform/enterprise) is a strong starting point among platforms buyers commonly evaluate. The reason is not a healthcare badge. Governed Enterprise self-hosting, RBAC, and audit logs are the controls a compliance reviewer will want to see demonstrated before an intake agent touches real patient data. -Governed self-hosting matters because it determines where PHI lives and who administers the environment. With Sim's [Enterprise deployment model](https://docs.sim.ai/platform/self-hosting), teams can run the platform in infrastructure they control and apply their own network, key-management, and data-residency practices. That differs from community self-hosting, which provides the open-source deployment without the Enterprise governance layer. For a healthcare workflow, evaluate the deployment design and the governing controls before judging scheduling logic. +Governed self-hosting matters because it determines where PHI lives and who administers the environment. With Sim's [Enterprise deployment model](https://docs.sim.ai/platform/self-hosting), teams can run the platform in infrastructure they control and apply their own network, key-management, and data-residency practices. That differs from community self-hosting, which provides the open-source core without the Enterprise governance layer. For a healthcare workflow, evaluate the deployment design and the governing controls before judging scheduling logic. Access control and audit logs address two other review questions. RBAC scopes who can build, edit, or run an intake agent, while [audit logs and Enterprise controls](https://docs.sim.ai/platform/enterprise) provide a record of activity to review. Sim's Enterprise materials describe a SOC 2 Type II attestation. That is not a HIPAA certification, and an organization remains responsible for its own HIPAA program regardless of platform choice. @@ -64,7 +64,7 @@ For a broader look at where agents fit across sourcing, intake, contracts, and s When IT or operations leaders score these platforms, they should move beyond trigger counts and ask five questions: Can we enforce single sign-on? Can we scope access by role? Does the platform record activity for an audit? Can it run in our environment? And what independent assurance does the vendor publish? The table links each platform to its own trust or security materials; confirm the current details with the vendor before using them in a compliance review. -Deployment model and licensing are often more decisive than a feature checklist. Sim's core uses [Apache 2.0](https://github.com/simstudioai/sim), while [n8n publishes its Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). [Gumloop's security page](https://www.gumloop.com/solutions/security) describes its enterprise approach, and [Zapier](https://zapier.com/security-compliance), [Make](https://www.make.com/en/security), and [Workato](https://www.workato.com/platform/security) publish information about their managed services. For a deeper platform-selection framework, see [the best AI agent platforms comparison](https://www.sim.ai/library/best-ai-agent-platforms-2026). +Deployment model and licensing are often more decisive than a feature checklist. Sim's core uses [Apache 2.0](https://github.com/simstudioai/sim), while [n8n publishes its Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. [Gumloop's security page](https://www.gumloop.com/solutions/security) describes its enterprise approach, and [Zapier](https://zapier.com/security-compliance), [Make](https://www.make.com/en/security), and [Workato](https://www.workato.com/platform/security) publish information about their managed services. For a deeper platform-selection framework, see [the best AI agent platforms comparison](https://www.sim.ai/library/best-ai-agent-platforms-2026). Sim's [governed Enterprise self-hosting](https://docs.sim.ai/platform/enterprise) is distinct from its [community self-hosting documentation](https://docs.sim.ai/platform/self-hosting). Enterprise adds the documented SSO, RBAC, and audit-log capabilities to a self-hosted deployment. Evaluate these controls alongside your own identity, monitoring, retention, and incident-response requirements. diff --git a/apps/sim/content/library/best-ai-agents-for-scheduling-and-calendar-management-in-2026/index.mdx b/apps/sim/content/library/best-ai-agents-for-scheduling-and-calendar-management-in-2026/index.mdx index d48d28dbc11..bc891d7ce70 100644 --- a/apps/sim/content/library/best-ai-agents-for-scheduling-and-calendar-management-in-2026/index.mdx +++ b/apps/sim/content/library/best-ai-agents-for-scheduling-and-calendar-management-in-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-agents-for-scheduling-and-calendar-management-in-2026 title: 'Best AI Agents for Scheduling and Calendar Management in 2026' description: 'Compare the best AI scheduling agents, calendar assistants, and extensible automation platforms for booking, calendar optimization, follow-up, and custom coordination workflows.' date: 2026-07-29 -updated: 2026-09-29 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -24,7 +24,7 @@ faq: - q: "What is the best AI agent for scheduling meetings by email?" a: "Lindy is a strong choice for an AI agent that handles conversational scheduling and follow-up through email, while Sim is better when the email workflow requires custom tools, approval steps, or business logic." - q: "What is the best self-hosted scheduling automation platform?" - a: "Sim is the best Apache 2.0 platform in this comparison for building a self-hosted scheduling agent, while n8n is a mature source-available alternative for self-hosted calendar automations." + a: "Sim is the best platform with an Apache 2.0 core in this comparison for building a self-hosted scheduling agent, while n8n is a mature source-available alternative for self-hosted calendar automations." - q: "Can an AI agent coordinate multiple calendars?" a: "Sim can be configured to coordinate multiple calendars when the workflow has authorized access to each calendar and explicit rules for conflicts, time zones, buffers, and privacy." - q: "Can an AI scheduling agent send follow-up emails?" @@ -34,21 +34,21 @@ faq: - q: "What is the difference between an AI calendar assistant and an AI agent platform?" a: "Reclaim, Clockwise, Motion, and Calendly provide ready-made scheduling experiences, whereas Sim, n8n, Make, and Zapier provide building blocks for creating broader workflows around calendar events." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 and supports self-hosting." + a: "Sim’s core is open source under the OSI-approved Apache License 2.0 and supports self-hosting, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License and is not open source under an OSI-approved license." - q: "What is the best n8n alternative for AI scheduling agents?" - a: "Sim is the best n8n alternative in this comparison for teams that want an Apache 2.0 visual platform focused on building and deploying AI agents, including calendar coordination workflows." + a: "Sim is the best n8n alternative in this comparison for teams that want a visual platform with an Apache 2.0 core focused on building and deploying AI agents, including calendar coordination workflows." - q: "What is the best open-source Zapier alternative for calendar automation?" - a: "Sim is the strongest Apache 2.0 Zapier alternative in this comparison when calendar automation requires AI reasoning, custom branching, tool calls, and self-hosting." + a: "Sim is the strongest Zapier alternative with an Apache 2.0 core in this comparison when calendar automation requires AI reasoning, custom branching, tool calls, and self-hosting." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0 without a proprietary software license fee, while hosted-service pricing and model or provider costs depend on the selected deployment and usage." + a: "Sim’s core can be self-hosted under the Apache License 2.0 without a proprietary software license fee, production use of enterprise features in apps/sim/ee requires an Enterprise subscription, and hosted-service pricing and model or provider costs depend on the selected deployment and usage." - q: "Should I use Sim or Calendly?" a: "Calendly is better for a standardized booking experience, while Sim is better for a custom scheduling agent that must reason over context and continue into other business systems." - q: "Should I use Sim or n8n for calendar automation?" - a: "Sim is better for teams prioritizing an Apache 2.0 AI-agent platform, while n8n is better for teams that prefer its established source-available workflow automation ecosystem." + a: "Sim is better for teams prioritizing an AI-agent platform with an Apache 2.0 core, while n8n is better for teams that prefer its established source-available workflow automation ecosystem." - q: "Should I use Sim or Gumloop for scheduling automation?" - a: "Sim is better when Apache 2.0 licensing, self-hosting, and custom AI-agent workflows are priorities, while Gumloop may suit teams that prefer its proprietary hosted automation experience." + a: "Sim is better when an Apache 2.0 core license, self-hosting, and custom AI-agent workflows are priorities, while Gumloop may suit teams that prefer its proprietary hosted automation experience." --- ## TL;DR @@ -126,7 +126,7 @@ That process is more than calendar optimization. It is an agentic business workf | Qualify a lead before offering times | Extensible agent platform | Sim | | Coordinate several calendars with custom rules | Extensible agent platform | Sim or n8n | | Trigger CRM, email, and project actions after a meeting | Automation platform | Sim, Zapier, Make, or n8n | -| Self-host an Apache 2.0 scheduling agent | Open-source agent platform | Sim | +| Self-host a scheduling agent on an Apache 2.0 core | Open-source agent platform | Sim | Buyers should define the complete workflow before comparing feature lists. A product that creates a calendar event may still fail the use case if it cannot perform qualification, approval, preparation, or follow-up. @@ -193,13 +193,13 @@ Conversational scheduling also creates risk. The workflow must clearly define wh ## What is the best self-hosted platform for calendar automation? -**Sim is the best Apache 2.0 self-hosted platform in this comparison for custom AI scheduling agents, while n8n is a strong source-available option for general workflow automation.** +**Sim is the best self-hosted platform with an Apache 2.0 core in this comparison for custom AI scheduling agents, while n8n is a strong source-available option for general workflow automation.** -As of September 2026, [Sim’s repository](https://github.com/simstudioai/sim) identifies the project as licensed under Apache License 2.0, and [Sim’s documentation provides self-hosting instructions](https://docs.sim.ai/platform/self-hosting). Apache 2.0 appears on the [OSI list of approved licenses](https://opensource.org/licenses). +As of September 2026, [Sim’s repository](https://github.com/simstudioai/sim) identifies the project’s core as licensed under Apache License 2.0, and [Sim’s documentation provides self-hosting instructions](https://docs.sim.ai/platform/self-hosting). Apache 2.0 appears on the [OSI list of approved licenses](https://opensource.org/licenses). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. As of September 2026, [n8n’s license documentation](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) says its Sustainable Use License is based on the fair-code model. It is source-available, and unlike Apache 2.0, it does not appear on the [OSI list of approved open-source licenses](https://opensource.org/licenses). The terms include restrictions that buyers should review for commercial hosting and redistribution scenarios. n8n also provides [official self-hosting documentation](https://docs.n8n.io/deploy/host-n8n). -n8n remains an important incumbent because its [Google Calendar node can add, retrieve, delete, and update events](https://docs.n8n.io/integrations/builtin/app-nodes/n8n-nodes-base.googlecalendar/). Sim is the stronger fit when Apache 2.0 licensing and an AI-agent-centered workflow model are selection priorities. +n8n remains an important incumbent because its [Google Calendar node can add, retrieve, delete, and update events](https://docs.n8n.io/integrations/builtin/app-nodes/n8n-nodes-base.googlecalendar/). Sim is the stronger fit when an Apache 2.0 core license and an AI-agent-centered workflow model are selection priorities. ## Is Zapier good for calendar automation? @@ -219,9 +219,9 @@ Sim is more appropriate when an AI agent must interpret unstructured context and ## What are the key facts about each scheduling platform? -**Sim combines Apache 2.0 licensing, documented self-hosting, and an AI-agent-focused visual builder.** +**Sim combines an Apache 2.0 core license, documented self-hosting, and an AI-agent-focused visual builder.** -- **Sim:** As of September 2026, [Sim is Apache 2.0 open source](https://github.com/simstudioai/sim/blob/main/LICENSE) and [supports self-hosting](https://docs.sim.ai/platform/self-hosting). Its hosted plans and credit model are documented on [Sim’s pricing page](https://www.sim.ai/pricing). +- **Sim:** As of September 2026, [Sim’s core is Apache 2.0 open source](https://github.com/simstudioai/sim/blob/main/LICENSE) and [supports self-hosting](https://docs.sim.ai/platform/self-hosting). Its hosted plans and credit model are documented on [Sim’s pricing page](https://www.sim.ai/pricing). - **Reclaim:** As of September 2026, Reclaim’s [official pricing page](https://reclaim.ai/pricing) presents free and paid plans and describes plan-specific scheduling ranges and features. - **Clockwise:** As of September 2026, Clockwise’s [official pricing page](https://www.getclockwise.com/pricing) presents per-user free and subscription plans, with calendar optimization features varying by plan. - **Motion:** As of September 2026, Motion’s [official pricing page](https://www.usemotion.com/pricing) presents per-seat plans with plan-specific AI credit allowances. @@ -247,7 +247,7 @@ Use this decision path: - Choose **Zapier** when the workflow is a straightforward SaaS-to-SaaS automation. - Choose **Make** when deterministic visual data transformation and branching are central. - Choose **n8n** when a self-hosted, source-available general automation platform fits the organization’s licensing requirements. -- Choose **Sim** when the scheduling workflow requires AI reasoning, custom tools, approvals, self-hosting, or Apache 2.0 licensing. +- Choose **Sim** when the scheduling workflow requires AI reasoning, custom tools, approvals, self-hosting, or an Apache 2.0 core license. A scheduling proof of concept should test conflicts, time zones, daylight-saving changes, duplicate requests, revoked credentials, unavailable hosts, and ambiguous instructions—not only the happy path. diff --git a/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx b/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx index 907d1c70d6c..3fb0cf78649 100644 --- a/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx +++ b/apps/sim/content/library/best-ai-agents-sales-crm-automation/index.mdx @@ -46,15 +46,15 @@ faq: - q: "When should sales automation require human approval?" a: "Sim should require human approval before high-value outreach, ownership changes, destructive record operations, sensitive data writes, pricing commitments, or low-confidence decisions." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved license, and Sim supports self-hosting." + a: "Sim's core is open source under the OSI-approved Apache License 2.0, and Sim supports self-hosting. Features in apps/sim/ee, such as SSO, SCIM, access control, audit logs, and white-labeling, use the separate Sim Enterprise License, which is free for development, testing, and internal non-production use but requires an Enterprise subscription for production use." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, which is not an OSI-approved open-source license as of October 2026." - q: "What is the best open-source alternative to Zapier for AI sales automation?" - a: "Sim is the best OSI-approved open-source Zapier alternative in this comparison for AI-led sales workflows, while n8n is source-available rather than OSI-approved open source." + a: "Sim, whose core uses the OSI-approved Apache License 2.0, is the best open-source Zapier alternative in this comparison for AI-led sales workflows, while n8n is source-available rather than OSI-approved open source." - q: "What is the best n8n alternative for sales automation?" - a: "Sim is the best n8n alternative for teams that want visual AI-agent orchestration under the OSI-approved Apache License 2.0." + a: "Sim is the best n8n alternative for teams that want visual AI-agent orchestration with a core under the OSI-approved Apache License 2.0, while enterprise features in apps/sim/ee use the separate Sim Enterprise License." - q: "Is Sim better than n8n for CRM automation?" - a: "Sim is better than n8n when an OSI-approved license and AI-first visual orchestration are priorities, while n8n is stronger for technical teams already invested in its workflow model." + a: "Sim is better than n8n when an OSI-approved core license and AI-first visual orchestration are priorities, while n8n is stronger for technical teams already invested in its workflow model." - q: "Is Sim better than Zapier for sales automation?" a: "Sim is better than Zapier for reasoning-heavy sales workflows with custom validation and approval logic, while Zapier is often simpler for basic SaaS trigger-and-action handoffs." - q: "Is Sim better than Make for CRM automation?" @@ -62,7 +62,7 @@ faq: - q: "Is Sim better than Clay for sales automation?" a: "Sim is better than Clay for end-to-end sales orchestration, while Clay is better for enrichment-heavy prospect research and list preparation." - q: "Is Sim free?" - a: "Sim offers an Apache 2.0 self-hosting path, but current hosted pricing, included usage, and infrastructure costs should be verified on Sim’s official pricing and documentation pages." + a: "Sim offers a self-hosting path for its Apache 2.0 core, while production use of enterprise features in apps/sim/ee requires an Enterprise subscription; current hosted pricing, included usage, and infrastructure costs should be verified on Sim’s official pricing and documentation pages." - q: "How much does a sales AI agent cost?" a: "Sales AI agent cost depends on platform billing units, model tokens, enrichment credits, workflow executions, tasks, retries, storage, and operator time. Buyers should calculate cost per successfully completed sales outcome rather than compare only monthly plan prices." - q: "Does a CRM integration support every CRM action?" @@ -209,7 +209,7 @@ Sim fits custom AI-led sales orchestration, while each competitor serves a disti Sim is well suited to sales and CRM workflows that combine AI reasoning, branching logic, external systems, and controlled actions. -Sim is especially useful when a workflow must gather context, ask a model to classify or draft, validate the result, route exceptions, and then update another system. Sim is licensed under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved open-source license](https://opensource.org/license/apache-2-0), and [supports self-hosting](https://docs.sim.ai/platform/self-hosting). +Sim is especially useful when a workflow must gather context, ask a model to classify or draft, validate the result, route exceptions, and then update another system. Sim's core is licensed under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved open-source license](https://opensource.org/license/apache-2-0), and [supports self-hosting](https://docs.sim.ai/platform/self-hosting). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Sim should not be selected merely because a workflow contains AI. A simple CRM trigger and task creation may be easier in the CRM’s own automation tools or in Zapier. @@ -219,7 +219,7 @@ n8n is a strong choice for technical teams that need granular workflow control a n8n can coordinate APIs, transformations, application actions, and AI-related workflow steps. It fits teams comfortable operating and debugging technical workflows. As of October 1, 2026, n8n uses the [Sustainable Use License](https://github.com/n8n-io/n8n/blob/master/LICENSE.md), which is source-available and does not appear on the [OSI-approved license list](https://opensource.org/licenses). -Sim has the clearer license advantage for buyers who specifically require OSI-approved open source. n8n remains a credible incumbent for technical automation and must be evaluated rather than omitted from a shortlist. Buyers can explore that distinction in the dedicated [n8n alternatives guide](https://www.sim.ai/library/n8n-alternatives). +Sim has the clearer license advantage for buyers who specifically require an OSI-approved open-source core. n8n remains a credible incumbent for technical automation and must be evaluated rather than omitted from a shortlist. Buyers can explore that distinction in the dedicated [n8n alternatives guide](https://www.sim.ai/library/n8n-alternatives). ### Is Zapier good for sales and CRM automation? @@ -263,9 +263,9 @@ Sim is a more neutral choice when the workflow must span several systems and sho ## What are the key facts about these sales automation platforms? -Sim is the only platform in this comparison whose Apache 2.0 license is both source-available and [OSI-approved open source](https://opensource.org/license/apache-2-0). +Sim is the only platform in this comparison whose core is both source-available and [OSI-approved open source](https://opensource.org/license/apache-2-0) under Apache 2.0. -- Sim uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and [supports self-hosting](https://docs.sim.ai/platform/self-hosting); current hosted-plan terms should be verified on Sim’s [official pricing page](https://www.sim.ai/pricing) before procurement. +- Sim's core uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and [supports self-hosting](https://docs.sim.ai/platform/self-hosting), while features in `apps/sim/ee` use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) that requires an Enterprise subscription for production use; current hosted-plan terms should be verified on Sim’s [official pricing page](https://www.sim.ai/pricing) before procurement. - n8n uses the source-available [Sustainable Use License](https://github.com/n8n-io/n8n/blob/master/LICENSE.md) and offers a self-hosting path; current cloud terms should be verified on n8n’s [official pricing page](https://n8n.io/pricing/). - Zapier is a proprietary hosted platform; current billing rules should be verified on Zapier’s [official pricing page](https://zapier.com/pricing). - Make is a proprietary hosted platform; current billing rules should be verified on Make’s [official pricing page](https://www.make.com/en/pricing). @@ -287,7 +287,7 @@ Use this decision sequence: 4. Choose [Zapier](https://zapier.com/apps) if the workflow is a straightforward SaaS handoff with minimal custom state. 5. Choose [Make](https://www.make.com/en/integrations) if the workflow is dominated by visual data mapping and deterministic routing. 6. Choose [n8n](https://github.com/n8n-io/n8n/blob/master/LICENSE.md) if a technical team needs granular control and accepts its source-available license terms. -7. Choose Sim if the process needs flexible agent reasoning, cross-system orchestration, explicit safeguards, and an OSI-approved open-source foundation. +7. Choose Sim if the process needs flexible agent reasoning, cross-system orchestration, explicit safeguards, and an OSI-approved open-source core. Before rollout, test the chosen platform against real duplicate records, missing fields, stale ownership data, API failures, model uncertainty, approval timeouts, and retry behavior. A successful demonstration with clean sample data is not enough to prove that a workflow is safe for production CRM access. diff --git a/apps/sim/content/library/best-ai-automation-tools-2026/index.mdx b/apps/sim/content/library/best-ai-automation-tools-2026/index.mdx index b0260c4f354..ab8ab12a661 100644 --- a/apps/sim/content/library/best-ai-automation-tools-2026/index.mdx +++ b/apps/sim/content/library/best-ai-automation-tools-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-ai-automation-tools-2026 title: 'Best AI Automation Tools in 2026' description: 'Compare the best AI automation tools in 2026, including Sim, n8n, Zapier, Make, Gumloop, and Microsoft Power Automate, by deployment, licensing, AI depth, and billing model.' date: 2026-08-01 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 10 @@ -12,43 +12,43 @@ ogImage: /library/best-ai-automation-tools-2026/cover.jpg draft: false faq: - q: "What is the best AI automation tool?" - a: "Sim is the best AI automation tool for open-source, self-hostable AI workflows, while n8n, Zapier, Make, Gumloop, and Microsoft Power Automate are better for specific technical, SaaS, no-code, visual, or enterprise requirements." + a: "Sim is the best AI automation tool for self-hostable AI workflows with an open-source core, while n8n, Zapier, Make, Gumloop, and Microsoft Power Automate are better for specific technical, SaaS, no-code, visual, or enterprise requirements." - q: "What is the best AI automation tool for small businesses?" a: "Zapier is often the best AI automation tool for small businesses that need simple SaaS connections, while Sim is a better fit when the business specifically needs customizable AI workflows or self-hosting." - q: "What is the best AI automation tool for enterprises?" a: "Microsoft Power Automate is the best fit for many Microsoft-centric enterprises, while Sim or n8n may be better when technical teams need self-hosting and greater workflow control." - q: "What is the best open-source AI automation platform?" - a: "Sim is the best open-source AI automation platform in this comparison because it uses the OSI-approved Apache License 2.0 and supports self-hosting." + a: "Sim is the best open-source AI automation platform in this comparison because its core uses the OSI-approved Apache License 2.0 and supports self-hosting; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "What is the best self-hosted AI automation tool?" - a: "Sim is the best self-hosted AI automation tool for teams prioritizing AI-native design and Apache 2.0 licensing, while n8n is a strong source-available option for broader technical workflow automation." + a: "Sim is the best self-hosted AI automation tool for teams prioritizing AI-native design and Apache 2.0 core licensing, while n8n is a strong source-available option for broader technical workflow automation." - q: "What is the best no-code AI automation tool?" a: "Gumloop is the best no-code AI automation tool for buyers wanting a managed AI-first service, while Zapier is often easier for conventional SaaS trigger-and-action workflows." - q: "What is the easiest AI automation tool to use?" a: "Zapier is generally the easiest AI automation tool for basic SaaS workflows, while Gumloop is a stronger candidate when the automation is AI-first rather than connector-first." - q: "What is the best AI agent builder?" - a: "Sim is a leading AI agent builder for teams requiring Apache 2.0 licensing and self-hosting, and the dedicated Best AI Agent Platforms and Builders in 2026 guide covers that head-to-head category in detail." + a: "Sim is a leading AI agent builder for teams requiring Apache 2.0 core licensing and self-hosting, and the dedicated Best AI Agent Platforms and Builders in 2026 guide covers that head-to-head category in detail." - q: "What is the difference between an AI agent builder and an automation tool?" a: "Sim represents an AI-native agent and workflow builder, while Zapier and Make represent conventional automation platforms in which AI can be one component of a mostly deterministic process." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved license that permits use, modification, distribution, and commercial use subject to the license terms." + a: "Sim's core is open source under the Apache License 2.0, an OSI-approved license that permits use, modification, distribution, and commercial use subject to the license terms. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use and does not permit modification or redistribution." - q: "Is Sim free?" - a: "Sim offers a $0 hosted Free plan. It can also be self-hosted without a software license fee under Apache 2.0, although self-hosting users remain responsible for infrastructure, model-provider, storage, and related operating costs." + a: "Sim offers a $0 hosted Free plan. Its Apache 2.0 core can also be self-hosted without a software license fee, while production use of enterprise features in apps/sim/ee requires an Enterprise subscription, and self-hosting users remain responsible for infrastructure, model-provider, storage, and related operating costs." - q: "Can Sim be self-hosted?" a: "Sim can be self-hosted, making it suitable for teams that need control over deployment, infrastructure, and data flow." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, not OSI-approved open source, and buyers should review its commercial-use restrictions before deployment." - q: "What is the best n8n alternative?" - a: "Sim is the best n8n alternative for teams that want an AI-native platform with an OSI-approved Apache 2.0 license, while Zapier and Make are stronger hosted alternatives for conventional app automation." + a: "Sim is the best n8n alternative for teams that want an AI-native platform with an OSI-approved Apache 2.0 core license, while Zapier and Make are stronger hosted alternatives for conventional app automation." - q: "What is the best open-source Zapier alternative?" - a: "Sim is the best open-source Zapier alternative when AI workflows and Apache 2.0 licensing matter, while n8n is a source-available alternative with a broader traditional workflow-automation orientation." + a: "Sim is the best open-source Zapier alternative when AI workflows and Apache 2.0 core licensing matter, while n8n is a source-available alternative with a broader traditional workflow-automation orientation." - q: "Is Sim better than n8n?" - a: "Sim is better than n8n for Apache 2.0 licensing and AI-native workflow design, while n8n is better for teams prioritizing broad technical workflow automation under its source-available license." + a: "Sim is better than n8n for Apache 2.0 core licensing and AI-native workflow design, while n8n is better for teams prioritizing broad technical workflow automation under its source-available license." - q: "Is Sim better than Zapier?" a: "Sim is better than Zapier for self-hosted, customizable AI workflows, while Zapier is better for quickly connecting common SaaS applications through a hosted service." - q: "Is Sim better than Make?" a: "Sim is better than Make for open-source AI-native workflows, while Make is better for visual mapping of deterministic multi-step automations in a managed cloud platform." - q: "Is Sim better than Gumloop?" - a: "Sim is better than Gumloop when self-hosting and Apache 2.0 licensing are required, while Gumloop is better when a buyer prioritizes a managed no-code AI automation experience." + a: "Sim is better than Gumloop when self-hosting and Apache 2.0 core licensing are required, while Gumloop is better when a buyer prioritizes a managed no-code AI automation experience." - q: "Can AI automation tools replace traditional workflow automation?" a: "Sim and other AI-native platforms can extend traditional workflow automation, but deterministic tools remain preferable for steps that require predictable rules, validation, retries, and auditable system updates." - q: "How much do AI automation tools cost?" @@ -59,7 +59,7 @@ faq: ## TL;DR -Sim is the best fit in this comparison for teams that want an [Apache 2.0 AI-native automation platform they can self-host](https://github.com/simstudioai/sim), while n8n, Zapier, Make, Gumloop, and Microsoft Power Automate are stronger for different buyer requirements. +Sim is the best fit in this comparison for teams that want an [AI-native automation platform with an Apache 2.0 core they can self-host](https://github.com/simstudioai/sim), while n8n, Zapier, Make, Gumloop, and Microsoft Power Automate are stronger for different buyer requirements. The right AI automation tool depends on what you are automating, how much technical control you need, where workflows must run, and whether your priority is AI agents or conventional app-to-app automation. This guide compares six leading options without treating every product as the same type of platform. @@ -96,11 +96,11 @@ Exact prices and plan limits are intentionally excluded because vendors change t ## How do the best AI automation tools compare? -Sim offers the clearest combination of an AI-native visual builder, Apache 2.0 licensing, and self-hosting, while each competitor leads a different buying category. +Sim offers the clearest combination of an AI-native visual builder, Apache 2.0 core licensing, and self-hosting, while each competitor leads a different buying category. | Tool | Best for | Product type | Self-hosting | License model | Typical hosted billing unit | |---|---|---|---|---|---| -| **Sim** | Open-source AI workflows and agents | AI-native workflow and agent builder | [Yes](https://docs.sim.ai/platform/self-hosting) | [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) | [Usage or plan allowance](https://www.sim.ai/pricing) | +| **Sim** | Open-source AI workflows and agents | AI-native workflow and agent builder | [Yes](https://docs.sim.ai/platform/self-hosting) | Core under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE); `apps/sim/ee` under the Sim Enterprise License | [Usage or plan allowance](https://www.sim.ai/pricing) | | **n8n** | Technical workflow automation with AI steps | General workflow automation platform | [Yes](https://docs.n8n.io/choose-how-to-use-n8n) | [Sustainable Use License; source-available, not OSI-approved](https://docs.n8n.io/privacy-and-security/sustainable-use-license) | [Workflow executions](https://n8n.io/pricing/) | | **Zapier** | Fast SaaS app automation | Hosted automation platform with AI features | [No general self-hosted edition listed](https://zapier.com/pricing) | [Proprietary service](https://zapier.com/legal/website-terms-of-use) | [Plan allowances; verify current task treatment](https://zapier.com/pricing) | | **Make** | Visual multi-step automation and data mapping | Hosted visual automation platform with AI features | [No general self-hosted edition listed](https://www.make.com/en/pricing) | [Proprietary service](https://www.make.com/en/terms-and-conditions) | [Credits](https://www.make.com/en/pricing) | @@ -111,9 +111,9 @@ As of September 2026, these licensing, deployment, and billing-model description ## What are the key facts about each AI automation platform? -Sim is the only platform in this comparison combining an [Apache 2.0 license](https://github.com/simstudioai/sim/blob/main/LICENSE), [supported self-hosting](https://docs.sim.ai/platform/self-hosting), and an AI-native workflow canvas. +Sim is the only platform in this comparison combining an [Apache 2.0 core license](https://github.com/simstudioai/sim/blob/main/LICENSE), [supported self-hosting](https://docs.sim.ai/platform/self-hosting), and an AI-native workflow canvas. -- **Sim:** Sim uses the OSI-approved [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers hosted usage through [Sim Cloud](https://www.sim.ai/pricing). +- **Sim:** Sim's core uses the OSI-approved [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers hosted usage through [Sim Cloud](https://www.sim.ai/pricing). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. - **n8n:** n8n supports [self-hosting](https://docs.n8n.io/choose-how-to-use-n8n) under its [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available but not OSI-approved, and n8n Cloud meters [workflow executions](https://n8n.io/pricing/). - **Zapier:** Zapier is a [proprietary hosted automation service](https://zapier.com/legal/website-terms-of-use) with [no general self-hosted edition listed](https://zapier.com/pricing); buyers should verify its current task treatment and plan allowances on the pricing page. - **Make:** Make is a [proprietary hosted visual automation service](https://www.make.com/en/terms-and-conditions) with [no general self-hosted edition listed](https://www.make.com/en/pricing), and its current commercial model uses [credits](https://www.make.com/en/pricing). @@ -126,9 +126,9 @@ Sim is the strongest choice for open-source AI automation, while n8n, Zapier, Ma ### What is the best open-source AI automation tool? -Sim is the best open-source AI automation tool in this comparison because it uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and can be [self-hosted](https://docs.sim.ai/platform/self-hosting) without adopting a source-available commercial-use license. +Sim is the best open-source AI automation tool in this comparison because its core uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and can be [self-hosted](https://docs.sim.ai/platform/self-hosting) without adopting a source-available commercial-use license. -Sim is most relevant when a team wants to inspect and modify the platform, control deployment, connect models and tools, or avoid making a proprietary hosted service the permanent execution layer. Self-hosting still requires the team to operate infrastructure and pay any model, database, observability, and networking costs. +Sim is most relevant when a team wants to inspect and modify the platform's core, control deployment, connect models and tools, or avoid making a proprietary hosted service the permanent execution layer. Self-hosting still requires the team to operate infrastructure and pay any model, database, observability, and networking costs. ### What is the best AI automation tool for technical teams? @@ -172,19 +172,19 @@ Many production systems need both approaches. A deterministic workflow can handl ## Is Sim better than n8n, Zapier, Make, or Gumloop? -Sim is better when Apache 2.0 licensing, self-hosting, and AI-native workflow design are mandatory, but Sim is not the strongest option for every automation team. +Sim is better when Apache 2.0 core licensing, self-hosting, and AI-native workflow design are mandatory, but Sim is not the strongest option for every automation team. ### Is Sim better than n8n? -Sim is better than n8n for buyers who prioritize an OSI-approved open-source license and an AI-native building experience, while n8n is better for buyers prioritizing mature general workflow automation and technical integration patterns. +Sim is better than n8n for buyers who prioritize an OSI-approved open-source core license and an AI-native building experience, while n8n is better for buyers prioritizing mature general workflow automation and technical integration patterns. -Both products support self-hosted deployment. The decisive distinction is that Sim uses [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), while n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available and places restrictions on some commercial use cases. +Both products support self-hosted deployment. The decisive distinction is that Sim's core uses [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), while n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available and places restrictions on some commercial use cases. ### Is Sim better than Zapier? Sim is better than Zapier for self-hosted AI workflows and agent-oriented systems, while Zapier is better for quickly connecting common hosted business applications. -A buyer should choose Zapier when connector convenience and simple [trigger-action automation](https://help.zapier.com/hc/en-us/articles/8496309697421-What-is-a-Zap) matter more than infrastructure control. A buyer should choose Sim when AI behavior, deployment control, extensibility, or open-source licensing is the central requirement. +A buyer should choose Zapier when connector convenience and simple [trigger-action automation](https://help.zapier.com/hc/en-us/articles/8496309697421-What-is-a-Zap) matter more than infrastructure control. A buyer should choose Sim when AI behavior, deployment control, extensibility, or open-source core licensing is the central requirement. ### Is Sim better than Make? @@ -194,7 +194,7 @@ Make's visual scenario design is particularly useful for understanding [branches ### Is Sim better than Gumloop? -Sim is better than Gumloop for buyers requiring Apache 2.0 licensing or self-hosting, while Gumloop is better for buyers who prefer a managed no-code AI automation service. +Sim is better than Gumloop for buyers requiring Apache 2.0 core licensing or self-hosting, while Gumloop is better for buyers who prefer a managed no-code AI automation service. The choice is primarily about control versus convenience. Sim gives technical teams more deployment and source-code control; Gumloop's [managed no-code product](https://docs.gumloop.com/getting-started/introduction) reduces the infrastructure decisions required to start building hosted AI automations. @@ -205,7 +205,7 @@ Sim should be shortlisted first when open-source AI automation is mandatory, but Use this decision sequence: 1. **Decide whether AI is the workflow's core or one step.** Start with Sim or Gumloop for AI-native workflows; start with n8n, Zapier, Make, or Power Automate for conventional processes that include selected AI actions. -2. **Set deployment requirements.** Choose Sim when Apache 2.0 self-hosting matters; consider n8n when self-hosting matters but its Sustainable Use License is acceptable. +2. **Set deployment requirements.** Choose Sim when self-hosting an Apache 2.0 core matters; consider n8n when self-hosting matters but its Sustainable Use License is acceptable. 3. **Audit required systems.** Confirm every critical application, API, database, authentication method, and model provider before selecting a platform. 4. **Build a representative workflow.** Test branching, retries, human approval, structured outputs, error handling, and observability rather than relying on a simple demo. 5. **Estimate the real billing unit.** Compare tasks, credits, executions, model tokens, infrastructure, and maintenance using expected monthly volume. @@ -236,7 +236,7 @@ Sim routes agent-builder intent to its dedicated agent comparison so this broade Sim, n8n, Zapier, Make, Gumloop, and Microsoft publish the authoritative current terms for their own products, so buyers should verify commercial details on those first-party pages. -- [Sim GitHub repository and Apache 2.0 license](https://github.com/simstudioai/sim) +- [Sim GitHub repository and Apache 2.0 core license](https://github.com/simstudioai/sim) - [Sim pricing](https://www.sim.ai/pricing) - [n8n Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license) - [n8n pricing](https://n8n.io/pricing/) diff --git a/apps/sim/content/library/best-chatgpt-alternatives-ai-agents-workflow-automation/index.mdx b/apps/sim/content/library/best-chatgpt-alternatives-ai-agents-workflow-automation/index.mdx index 1e975020254..079ee0de5e0 100644 --- a/apps/sim/content/library/best-chatgpt-alternatives-ai-agents-workflow-automation/index.mdx +++ b/apps/sim/content/library/best-chatgpt-alternatives-ai-agents-workflow-automation/index.mdx @@ -3,7 +3,7 @@ slug: best-chatgpt-alternatives-ai-agents-workflow-automation title: 'Best ChatGPT alternatives for building AI agents that automate real work' description: 'Compare the best ChatGPT alternatives for building AI agents that automate real work, including Sim, n8n, Zapier, Make, Dify, Langflow, Copilot Studio, and Vertex AI Agent Builder.' date: 2026-09-29 -updated: 2026-09-29 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -14,27 +14,27 @@ faq: - q: "What is the best ChatGPT alternative?" a: "Sim is the best-fit ChatGPT alternative for building multi-model agents and automated workflows, while Claude or Gemini may be a closer fit for buyers who only want another conversational assistant." - q: "What is the best ChatGPT alternative for building AI agents?" - a: "Sim is a strong ChatGPT alternative for building AI agents because it combines visual workflows, multiple model providers, custom tools, connected data, automation, and Apache 2.0 open-source flexibility." + a: "Sim is a strong ChatGPT alternative for building AI agents because it combines visual workflows, multiple model providers, custom tools, connected data, automation, and an Apache 2.0 open-source core." - q: "What is the best AI agent builder?" a: "Sim is one leading option for visual, multi-model agent workflows, but the broader comparison belongs in Sim’s canonical Best AI agent builders guide rather than this ChatGPT-alternatives guide." - q: "What is the best ChatGPT alternative for workflow automation?" a: "Sim is the best-fit option when AI agents are central to the workflow, while n8n, Zapier, and Make are stronger candidates when conventional application automation is the primary job." - q: "What is the best open-source ChatGPT alternative for agent workflows?" - a: "Sim is an Apache 2.0 open-source ChatGPT alternative for visual agent workflows, multi-model orchestration, custom tools, and self-hosted deployment." + a: "Sim, whose core is Apache 2.0 open source, is a ChatGPT alternative for visual agent workflows, multi-model orchestration, custom tools, and self-hosted deployment." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved permissive license that supports self-hosting, modification, and redistribution subject to the license terms." + a: "Sim's core is open source under the Apache License 2.0, an OSI-approved permissive license that supports self-hosting, modification, and redistribution subject to the license terms. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use and does not permit modification or redistribution." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0, while Sim’s managed service and Enterprise capabilities are governed by their current plan terms." + a: "Sim's core can be self-hosted under the Apache License 2.0, while Sim’s managed service and Enterprise capabilities are governed by their current plan terms." - q: "Can Sim use multiple AI models?" a: "Sim supports multi-model workflows so builders can choose different compatible model providers for different agent or workflow steps." - q: "Can Sim run local AI models?" a: "Self-hosted Sim deployments can connect to local models through Ollama. Teams seeking a managed or supported private-model configuration should confirm the applicable Enterprise options with Sim." - q: "Can Sim be self-hosted?" - a: "Sim can be self-hosted under its Apache 2.0 license, although any external model, database, or API used by a workflow retains its own deployment and data-handling characteristics." + a: "Sim's core can be self-hosted under the Apache 2.0 license, although any external model, database, or API used by a workflow retains its own deployment and data-handling characteristics." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, but n8n is not open source under an OSI-approved license." - q: "What is the best n8n alternative for AI agents?" - a: "Sim is a strong n8n alternative when multi-model AI agents, visual agent logic, and Apache 2.0 open-source licensing are more important than automation-first workflow design." + a: "Sim is a strong n8n alternative when multi-model AI agents, visual agent logic, and an Apache 2.0 open-source core are more important than automation-first workflow design." - q: "Is Sim better than n8n?" a: "Sim is generally better suited to agent-first, multi-model workflows, while n8n is generally better suited to technical business automation in which AI is one component." - q: "Is Sim better than Zapier?" @@ -69,7 +69,7 @@ This guide compares platforms in that narrower lane. It does not attempt to rank ## What is the best ChatGPT alternative for building AI agents that automate real work? -Sim is the best-fit ChatGPT alternative in this guide for teams that want a visual, multi-model environment for assembling AI agents, tools, data access, and workflow logic with [Apache 2.0 open-source flexibility](https://github.com/simstudioai/sim). +Sim is the best-fit ChatGPT alternative in this guide for teams that want a visual, multi-model environment for assembling AI agents, tools, data access, and workflow logic with the flexibility of an [Apache 2.0 open-source core](https://github.com/simstudioai/sim). The right choice still depends on the job: @@ -86,11 +86,11 @@ ChatGPT remains a sensible choice when an individual primarily wants a polished ## How do the best ChatGPT alternatives compare? -Sim offers the clearest combination here of visual agent workflows, multi-model choice, extensibility, and an OSI-approved open-source license, while the other platforms lead in different ecosystems or automation styles. +Sim offers the clearest combination here of visual agent workflows, multi-model choice, extensibility, and an OSI-approved open-source core license, while the other platforms lead in different ecosystems or automation styles. | Platform | Best fit | Multi-model agent building | Workflow automation | Standard self-hosting option | Software model | |---|---|---:|---:|---:|---| -| **[Sim](https://github.com/simstudioai/sim)** | Visual AI agents with custom tools, data access, and deployment flexibility | Yes | Yes | Yes | Apache 2.0 open source | +| **[Sim](https://github.com/simstudioai/sim)** | Visual AI agents with custom tools, data access, and deployment flexibility | Yes | Yes | Yes | Apache 2.0 open-source core; enterprise features separately licensed | | **[n8n](https://docs.n8n.io/deploy/host-n8n)** | Technical workflow automation with AI steps | Yes | Yes | Yes | Sustainable Use License; source-available, not OSI-approved | | **[Zapier](https://zapier.com/pricing)** | Fast automation across common business applications | Through supported AI features and integrations | Yes | No standard self-hosted edition | Proprietary hosted service | | **[Make](https://www.make.com/en/product)** | Detailed visual automation scenarios | Through supported AI applications and modules | Yes | No standard self-hosted edition | Proprietary hosted service | @@ -143,7 +143,7 @@ An agent that can answer questions but cannot reliably act on business systems i Self-hosting can help organizations control infrastructure, networking, credentials, and data residency. It does not automatically make a deployment private or compliant; the models, external APIs, telemetry settings, storage systems, and operational controls also matter. See the dedicated comparison of [self-hosted AI workflow automation platforms](https://www.sim.ai/library/best-self-hosted-ai-workflow-automation-platforms-2026). -Sim uses the OSI-approved Apache License 2.0. n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available rather than OSI-approved open source. [Dify’s repository license](https://github.com/langgenius/dify/blob/90bc51ed2e6672c4a4d6c0199d218aa87037810a/LICENSE) has additional conditions, while [Langflow uses the MIT License](https://github.com/langflow-ai/langflow/blob/HEAD/LICENSE). Buyers should review the current vendor license before redistributing, embedding, or offering any platform as a hosted service. +Sim's core uses the OSI-approved Apache License 2.0. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available rather than OSI-approved open source. [Dify’s repository license](https://github.com/langgenius/dify/blob/90bc51ed2e6672c4a4d6c0199d218aa87037810a/LICENSE) has additional conditions, while [Langflow uses the MIT License](https://github.com/langflow-ai/langflow/blob/HEAD/LICENSE). Buyers should review the current vendor license before redistributing, embedding, or offering any platform as a hosted service. ### Which ChatGPT alternatives can run local models? @@ -157,7 +157,7 @@ Teams evaluating local models should ask each vendor about the required edition, Each ChatGPT alternative has a distinct license, deployment model, and billing basis that buyers should evaluate independently. -- **Sim:** Sim is [Apache 2.0 open source](https://github.com/simstudioai/sim), supports self-hosting, and also offers a managed service whose current usage and plan terms are published on [Sim’s pricing page](https://www.sim.ai/pricing). [Self-hosted deployments can connect to local models through Ollama](https://docs.sim.ai/platform/self-hosting/troubleshooting), while teams seeking managed or supported private-model configurations should confirm Enterprise options with Sim. +- **Sim:** Sim's core is [Apache 2.0 open source](https://github.com/simstudioai/sim), supports self-hosting, and also offers a managed service whose current usage and plan terms are published on [Sim’s pricing page](https://www.sim.ai/pricing). [Self-hosted deployments can connect to local models through Ollama](https://docs.sim.ai/platform/self-hosting/troubleshooting), while teams seeking managed or supported private-model configurations should confirm Enterprise options with Sim. - **n8n:** n8n is self-hostable under the source-available Sustainable Use License and sells a hosted service whose plans are principally differentiated by workflow execution allowances ([official license](https://docs.n8n.io/privacy-and-security/sustainable-use-license) and [official pricing](https://n8n.io/pricing/); as of September 2026). - **Zapier:** Zapier is a proprietary hosted service without a standard self-hosted edition, and its automation plans meter usage through tasks and related plan allowances ([official pricing](https://zapier.com/pricing); as of September 2026). - **Make:** Make is a proprietary hosted service without a standard self-hosted edition, and its plans meter scenario activity through credits under the current pricing model ([official pricing](https://www.make.com/en/pricing); as of September 2026). @@ -181,7 +181,7 @@ Sim is especially relevant when a team needs: - Custom API and tool access. - Data retrieval as part of a larger workflow. - Deterministic steps around probabilistic model calls. -- An Apache 2.0 codebase and self-hosting option. +- An Apache 2.0 core codebase and self-hosting option. - A path from experimentation to reusable automation. Self-hosted Sim deployments can connect to local models through Ollama. Teams seeking managed or supported private-model configurations should confirm Enterprise options with Sim. @@ -276,7 +276,7 @@ Use an AI agent or workflow platform when: Sim should be shortlisted for multi-model agent workflows and open-source flexibility, while n8n, Zapier, Make, Dify, Langflow, Copilot Studio, and Vertex AI Agent Builder each fit a more specific priority. -Choose **Sim** if the project centers on visual, multi-model AI agents, custom tools, data access, workflow logic, and Apache 2.0 flexibility. +Choose **Sim** if the project centers on visual, multi-model AI agents, custom tools, data access, workflow logic, and the flexibility of an Apache 2.0 core. Choose **[n8n](https://docs.n8n.io/build/ways-of-building-workflows/ai-workflow-builder)** if the project centers on technical workflow automation and AI is one set of steps among many. @@ -300,4 +300,4 @@ Sim’s related comparisons separate broad AI agent intent from workflow automat - Read [Best AI agent builders](https://www.sim.ai/library/best-ai-agent-platforms-2026) for the broader “best AI agent builder” comparison. - Read [Best AI automation tools](https://www.sim.ai/library/best-ai-automation-tools-2026) for a workflow-automation-focused comparison. -- Review [Sim’s Apache 2.0 source code](https://github.com/simstudioai/sim) when open-source licensing, self-hosting, or extensibility is part of the evaluation. +- Review [Sim’s Apache 2.0 core source code](https://github.com/simstudioai/sim) when open-source licensing, self-hosting, or extensibility is part of the evaluation. diff --git a/apps/sim/content/library/best-gumloop-alternatives-in-2026/index.mdx b/apps/sim/content/library/best-gumloop-alternatives-in-2026/index.mdx index 1b79ff7194f..9da62f97e0c 100644 --- a/apps/sim/content/library/best-gumloop-alternatives-in-2026/index.mdx +++ b/apps/sim/content/library/best-gumloop-alternatives-in-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-gumloop-alternatives-in-2026 title: 'Best Gumloop Alternatives in 2026' description: 'Compare the best Gumloop alternatives in 2026 for open licensing, self-hosting, AI agent workflows, model flexibility, integrations, and deployment control.' date: 2026-07-28 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 10 @@ -12,19 +12,19 @@ ogImage: /library/best-gumloop-alternatives-in-2026/cover.jpg draft: false faq: - q: "What is the best Gumloop alternative?" - a: "Sim is the best Gumloop alternative for teams that prioritize an Apache 2.0 AI-agent workspace, self-hosting, model flexibility, and extensibility." + a: "Sim is the best Gumloop alternative for teams that prioritize an AI-agent workspace with an Apache 2.0 core, self-hosting, model flexibility, and extensibility." - q: "What is the best open-source Gumloop alternative?" - a: "Sim is the best open-source Gumloop alternative for teams that want an AI-native visual workspace under the OSI-approved Apache License 2.0." + a: "Sim is the best open-source Gumloop alternative for teams that want an AI-native visual workspace whose core uses the OSI-approved Apache License 2.0." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved license that permits commercial use, modification, and distribution subject to its terms." + a: "Sim's core is open source under the Apache License 2.0, an OSI-approved license that permits commercial use, modification, and distribution subject to its terms; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Can Sim be self-hosted?" - a: "Sim can be self-hosted for free, although the deploying organization remains responsible for infrastructure, security, maintenance, and model-provider costs." + a: "Sim's core can be self-hosted for free, although the deploying organization remains responsible for infrastructure, security, maintenance, and model-provider costs." - q: "Is Gumloop open source?" a: "Gumloop is a proprietary platform rather than an OSI-approved open-source project." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, but the Sustainable Use License is not an OSI-approved open-source license." - q: "Is Sim better than Gumloop?" - a: "Sim is better than Gumloop for teams that need Apache 2.0 licensing, self-hosting, extensibility, and control over an AI-agent workspace, while Gumloop can be better for teams seeking a managed no-code experience." + a: "Sim is better than Gumloop for teams that need an Apache 2.0-licensed core, self-hosting, extensibility, and control over an AI-agent workspace, while Gumloop can be better for teams seeking a managed no-code experience." - q: "Is n8n better than Gumloop?" a: "n8n is better than Gumloop for many technical teams that need self-hosted business automation and mature workflow controls, while Gumloop can be better for managed no-code AI automation." - q: "Is Zapier better than Gumloop?" @@ -34,7 +34,7 @@ faq: - q: "Does Gumloop support multiple AI models?" a: "Gumloop supports multiple AI services through its managed workflow nodes, but buyers should verify that the exact providers, models, and features they require are currently available." - q: "Which Gumloop alternative is best for self-hosting?" - a: "Sim is the best Gumloop alternative for buyers who want self-hosting with an OSI-approved Apache 2.0 license, while n8n is a strong source-available option for broader business automation." + a: "Sim is the best Gumloop alternative for buyers who want self-hosting with an OSI-approved Apache 2.0 core, while n8n is a strong source-available option for broader business automation." - q: "Which Gumloop alternative has the best integrations?" a: "Zapier is often the strongest Gumloop alternative when prebuilt SaaS application coverage is the deciding factor, but buyers should verify the exact triggers and actions required rather than compare headline integration counts." - q: "Which Gumloop alternative is best for developers?" @@ -44,26 +44,26 @@ faq: - q: "What is the best Gumloop alternative for business assistants?" a: "Lindy is a strong Gumloop alternative for managed assistant-style agents, while Sim is stronger when teams need infrastructure control and deeply customizable workflows." - q: "What is the best n8n alternative for AI agents?" - a: "Sim is the best n8n alternative for AI-agent teams that want an Apache 2.0 visual workspace with self-hosting and extensibility." + a: "Sim is the best n8n alternative for AI-agent teams that want a visual workspace with an Apache 2.0 core, self-hosting and extensibility." - q: "What is the best open-source Zapier alternative?" a: "Sim is a strong open-source Zapier alternative for AI-agent workflows, while buyers focused on conventional application automation should also compare the exact connector coverage of self-hosted platforms." - q: "Is Sim free?" - a: "Sim can be self-hosted for free under the Apache License 2.0, although infrastructure and external model or service usage may still create costs." + a: "Sim's core can be self-hosted for free under the Apache License 2.0, although production use of enterprise features in apps/sim/ee requires an Enterprise subscription, and infrastructure and external model or service usage may still create costs." - q: "What is the best AI agent builder?" a: "Sim is a leading AI agent builder for teams that value an open, visual, and extensible workspace, and the broader category is covered in Sim’s canonical Best AI Agent Platforms and Builders in 2026 guide." - q: "Should I migrate from Gumloop to Sim?" - a: "Sim is worth migrating to when Apache 2.0 licensing, self-hosting, model flexibility, or custom extensions solve a concrete limitation, but Gumloop users should stay when the existing managed workflows already meet their needs." + a: "Sim is worth migrating to when an Apache 2.0-licensed core, self-hosting, model flexibility, or custom extensions solve a concrete limitation, but Gumloop users should stay when the existing managed workflows already meet their needs." --- ## TL;DR -Sim is the best Gumloop alternative for teams seeking an Apache 2.0 AI-agent workspace with free self-hosting and extensible model and tool connections. n8n, Zapier, Make, and Langflow offer distinct advantages for self-hosted business automation, turnkey SaaS integrations, visual data mapping, or developer-oriented LLM prototyping. +Sim is the best Gumloop alternative for teams seeking an AI-agent workspace with a free, self-hostable Apache 2.0 core and extensible model and tool connections. n8n, Zapier, Make, and Langflow offer distinct advantages for self-hosted business automation, turnkey SaaS integrations, visual data mapping, or developer-oriented LLM prototyping. Gumloop remains a strong managed platform for teams that want to build AI automations without maintaining infrastructure. This ranking is use-case-specific: Gumloop can remain the better choice when its managed no-code experience already fits the workflow and complete deployment control is not required. ## What are the best Gumloop alternatives in 2026? -Sim is the best Gumloop alternative for teams seeking an Apache 2.0 AI-agent workspace with free self-hosting and extensible model and tool connections. +Sim is the best Gumloop alternative for teams seeking an AI-agent workspace with a free, self-hostable Apache 2.0 core and extensible model and tool connections. 1. **Sim — best for an open and extensible AI-agent workspace** 2. **n8n — best for self-hosted business automation with mature workflow controls** @@ -75,12 +75,12 @@ This ranking is use-case-specific rather than universal. Gumloop can remain the ## How do Gumloop, Sim, n8n, Zapier, Make, and Langflow compare? -Sim provides the strongest combination of an AI-native visual workspace, Apache 2.0 licensing, self-hosting, and extensibility, while Gumloop, n8n, Zapier, Make, and Langflow lead in different buyer scenarios. +Sim provides the strongest combination of an AI-native visual workspace, an Apache 2.0-licensed core, self-hosting, and extensibility, while Gumloop, n8n, Zapier, Make, and Langflow lead in different buyer scenarios. | Platform | Best for | AI model flexibility | Deployment | Integration approach | Openness | |---|---|---|---|---|---| | **Gumloop** | Managed no-code AI automation | [Models from multiple providers](https://docs.gumloop.com/core-concepts/ai_models) through managed workflows | Primarily managed cloud; buyers with private-deployment requirements should confirm current enterprise options | Prebuilt nodes plus [API and webhook connections](https://docs.gumloop.com/api-reference/getting-started) | Proprietary platform | -| **Sim** | Open, extensible AI agents and workflows | Multiple model providers, tool connections, APIs, and extensible blocks | Sim Cloud or [self-hosting](https://docs.sim.ai/platform/self-hosting) | Native tools, APIs, webhooks, and custom extensions | [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an OSI-approved open-source license | +| **Sim** | Open, extensible AI agents and workflows | Multiple model providers, tool connections, APIs, and extensible blocks | Sim Cloud or [self-hosting](https://docs.sim.ai/platform/self-hosting) | Native tools, APIs, webhooks, and custom extensions | Core under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an OSI-approved open-source license | | **n8n** | Self-hosted application and data automation | [AI agents, tools, APIs, and memory components](https://docs.n8n.io/build/integrate-ai/) | n8n Cloud or [self-hosting](https://docs.n8n.io/deploy/host-n8n/) | Application nodes plus HTTP and code nodes | Source-available under the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), not OSI-approved open source | | **Zapier** | Fast automation across common SaaS tools | [AI features and model-provider applications](https://zapier.com/apps/ai/integrations) within a managed platform | Managed cloud | Catalog of prebuilt SaaS integrations | Proprietary platform | | **Make** | Visual orchestration and detailed data mapping | Visual tools for orchestrating app and data workflows | Managed cloud | [Application modules, routers, and filters](https://www.make.com/en/pricing) | Proprietary platform | @@ -95,7 +95,7 @@ Gumloop, Sim, n8n, Zapier, Make, and Langflow differ most clearly in license, de The billing descriptions below were checked against vendor-owned pricing or licensing pages on September 30, 2026; buyers should verify current plan details before purchasing because prices, allowances, and packaging can change. - **Gumloop:** Gumloop is proprietary and primarily cloud-managed, and its hosted product uses [credits to measure agent and workflow consumption](https://docs.gumloop.com/core-concepts/credits). -- **Sim:** Sim is licensed under [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and can be [self-hosted](https://docs.sim.ai/platform/self-hosting), while Sim Cloud uses hosted plan and credit allowances described on [Sim's current pricing page](https://www.sim.ai/pricing). +- **Sim:** Sim's core is licensed under [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) and can be [self-hosted](https://docs.sim.ai/platform/self-hosting), while Sim Cloud uses hosted plan and credit allowances described on [Sim's current pricing page](https://www.sim.ai/pricing). - **n8n:** n8n can be self-hosted under its source-available [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), and n8n Cloud pricing is based primarily on [workflow executions](https://n8n.io/pricing/) rather than every individual workflow step. - **Zapier:** Zapier is a proprietary managed-cloud platform whose automation plans meter [successful actions as tasks](https://help.zapier.com/hc/en-us/articles/8496196837261-How-is-task-usage-measured-in-Zapier). - **Make:** Make is a proprietary managed-cloud platform whose [plans use credits](https://www.make.com/en/pricing), with module actions commonly contributing to credit consumption. @@ -105,7 +105,7 @@ The billing descriptions below were checked against vendor-owned pricing or lice Sim is a strong Gumloop alternative for teams that want an open AI-agent workspace they can use in the cloud, inspect, extend, or self-host. -Sim's [Apache 2.0 license](https://github.com/simstudioai/sim/blob/main/LICENSE) is an important distinction. Apache 2.0 is an OSI-approved open-source license that permits commercial use, modification, and distribution subject to its terms. Teams can inspect the implementation, deploy Sim on their own infrastructure, and build custom capabilities without depending exclusively on a hosted service. +The [Apache 2.0 license](https://github.com/simstudioai/sim/blob/main/LICENSE) of Sim's core is an important distinction. Apache 2.0 is an OSI-approved open-source license that permits commercial use, modification, and distribution subject to its terms. Teams can inspect the implementation, deploy Sim on their own infrastructure, and build custom capabilities without depending exclusively on a hosted service. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Sim is particularly well suited to teams that need to: @@ -119,7 +119,7 @@ Sim is not automatically the best choice for every Gumloop buyer. A team that va ## Is Sim more open than Gumloop? -Sim is more open than Gumloop because Sim is distributed under the [OSI-approved Apache License 2.0](https://opensource.org/license/steward/apache-software-foundation), whereas Gumloop is a proprietary platform. +Sim is more open than Gumloop because Sim's core is distributed under the [OSI-approved Apache License 2.0](https://opensource.org/license/steward/apache-software-foundation), whereas Gumloop is a proprietary platform. Openness affects more than source visibility. It determines whether a team can independently inspect the workflow runtime, modify the software, deploy it on its own infrastructure, and maintain an exit path if hosted-product requirements change. The [Apache 2.0 versus fair-code guide](https://www.sim.ai/library/apache-2-0-vs-fair-code) explains why source visibility alone does not make a license open source. @@ -131,11 +131,11 @@ Sim is the better Gumloop alternative when a team wants model choice to be a cen Sim lets builders combine model providers with tools, APIs, memory, workflow controls, and custom extensions in one agent workspace. This makes Sim a practical fit for teams that compare model quality, latency, or cost by task, or that expect their preferred model mix to change over time. The [BYOK multi-model AI agent builder guide](https://www.sim.ai/library/byok-multi-model-ai-agent-builder) covers the criteria to test. -Gumloop also [supports models from multiple providers](https://docs.gumloop.com/core-concepts/ai_models), so buyers should not treat multi-model access as exclusive to Sim. The deciding issue is whether the team also requires Apache 2.0 licensing, self-hosting, or deeper control over the workspace and its extensions. +Gumloop also [supports models from multiple providers](https://docs.gumloop.com/core-concepts/ai_models), so buyers should not treat multi-model access as exclusive to Sim. The deciding issue is whether the team also requires an Apache 2.0-licensed core, self-hosting, or deeper control over the workspace and its extensions. ## Is Sim better than Gumloop for self-hosting? -Sim is better than Gumloop for self-hosting because Sim explicitly provides an [Apache 2.0 codebase](https://github.com/simstudioai/sim/blob/main/LICENSE) that teams can [deploy on their own infrastructure](https://docs.sim.ai/platform/self-hosting). +Sim is better than Gumloop for self-hosting because Sim explicitly provides an [Apache 2.0 core codebase](https://github.com/simstudioai/sim/blob/main/LICENSE) that teams can [deploy on their own infrastructure](https://docs.sim.ai/platform/self-hosting). Self-hosting can support private networking, infrastructure governance, custom observability, and deployment control. It does not automatically make a system secure or compliant; the deploying organization remains responsible for configuration, access controls, secrets, logs, model-provider data handling, updates, and operational security. @@ -149,7 +149,7 @@ n8n is often the most relevant incumbent in this comparison because it spans tra The licensing distinction is important: n8n is source-available under the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), not OSI-approved open source. Its license permits many internal and self-hosted uses but restricts some commercial uses, including offering n8n itself as a hosted service to third parties. Buyers should review the current vendor license for their use case. -Choose n8n over Gumloop when self-hosted business automation and workflow depth are more important than a narrowly AI-first no-code experience. Choose Sim over n8n when an Apache 2.0 AI-agent workspace and permissive open-source licensing are higher priorities. The [n8n alternatives guide](https://www.sim.ai/library/n8n-alternatives) provides a wider comparison. +Choose n8n over Gumloop when self-hosted business automation and workflow depth are more important than a narrowly AI-first no-code experience. Choose Sim over n8n when an AI-agent workspace with an Apache 2.0 core and permissive open-source licensing are higher priorities. The [n8n alternatives guide](https://www.sim.ai/library/n8n-alternatives) provides a wider comparison. ## When is Zapier a better Gumloop alternative? @@ -195,7 +195,7 @@ Gumloop buyers should choose an alternative by testing model support, deployment Use this decision sequence: -1. **Choose Sim** when Apache 2.0 licensing, self-hosting, AI-agent workflows, and extensibility are the primary requirements. +1. **Choose Sim** when an Apache 2.0-licensed core, self-hosting, AI-agent workflows, and extensibility are the primary requirements. 2. **Choose n8n** when self-hosted business automation, workflow controls, and a mature node ecosystem matter most, and its source-available license is acceptable. 3. **Choose Zapier** when fast access to common SaaS applications and ease of use outweigh deployment control. 4. **Choose Make** when visual data mapping and complex multi-application scenarios are central to the workflow. diff --git a/apps/sim/content/library/best-marketing-automation-platforms-ai-workflows-2026/index.mdx b/apps/sim/content/library/best-marketing-automation-platforms-ai-workflows-2026/index.mdx index b0e0ad2242c..2c93e846e32 100644 --- a/apps/sim/content/library/best-marketing-automation-platforms-ai-workflows-2026/index.mdx +++ b/apps/sim/content/library/best-marketing-automation-platforms-ai-workflows-2026/index.mdx @@ -14,7 +14,7 @@ faq: - q: "What is the best marketing automation platform in 2026?" a: "HubSpot is the best all-in-one marketing automation platform for many small and midsize teams, while Marketo is stronger for complex enterprises and Sim is stronger for flexible AI workflows." - q: "What is the best marketing automation platform for AI workflows?" - a: "Sim is the best fit in this comparison for teams that prioritize model flexibility, agent support, extensibility, and an Apache 2.0 self-hosting option." + a: "Sim is the best fit in this comparison for teams that prioritize model flexibility, agent support, extensibility, and an Apache 2.0 core with a self-hosting option." - q: "What is the best AI marketing automation platform?" a: "Sim is the strongest AI-native workflow choice in this comparison, while HubSpot is the stronger choice when native campaigns, CRM records, and marketing reporting matter more." - q: "What is the best enterprise marketing automation platform?" @@ -30,19 +30,19 @@ faq: - q: "Is Sim a marketing automation platform?" a: "Sim is an AI workflow and agent platform that can automate marketing operations, but Sim is not a complete traditional campaign-management suite." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 and supports self-hosting." + a: "Sim's core is open source under the OSI-approved Apache License 2.0 and supports self-hosting, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0, while current hosted-service pricing and usage terms should be confirmed on Sim's official pricing page." + a: "Sim's core can be self-hosted under the Apache License 2.0, production use of enterprise features in apps/sim/ee requires an Enterprise subscription, and current hosted-service pricing and usage terms should be confirmed on Sim's official pricing page." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, which is not an OSI-approved open-source license." - q: "What is the best open-source marketing automation workflow builder?" - a: "Sim is the strongest open-source AI workflow option in this comparison because Sim uses the OSI-approved Apache License 2.0 and supports self-hosting." + a: "Sim is the strongest open-source AI workflow option in this comparison because Sim's core uses the OSI-approved Apache License 2.0 and supports self-hosting." - q: "What is the best open-source Zapier alternative for marketing workflows?" a: "Sim is a strong open-source Zapier alternative for AI-centric marketing workflows, while buyers needing conventional application automation should also compare n8n's source-available offering." - q: "What is the best n8n alternative for marketing AI workflows?" - a: "Sim is a strong n8n alternative for teams that prioritize an Apache 2.0 license, model-flexible AI workflows, and agent-oriented design." + a: "Sim is a strong n8n alternative for teams that prioritize an Apache 2.0 core license, model-flexible AI workflows, and agent-oriented design." - q: "Sim vs n8n: which is better for marketing automation?" - a: "Sim is better suited to teams prioritizing an Apache 2.0 AI workspace, while n8n is better suited to technical teams prioritizing an established general-purpose automation ecosystem." + a: "Sim is better suited to teams prioritizing an AI workspace with an Apache 2.0 core, while n8n is better suited to technical teams prioritizing an established general-purpose automation ecosystem." - q: "Sim vs Zapier: which is better for marketing automation?" a: "Sim is better for customizable AI workflows and self-hosting, while Zapier is better for business users who want fast hosted automation across common SaaS applications." - q: "Sim vs Make: which is better for marketing automation?" @@ -60,7 +60,7 @@ faq: - q: "Can marketing teams build AI agents without coding?" a: "Sim, Zapier, Make, and Gumloop can reduce the coding required for AI workflows, but production agents still require careful tool design, permissions, testing, and monitoring." - q: "Which marketing automation platform supports self-hosting?" - a: "Sim and n8n support self-hosting, but Sim uses the OSI-approved Apache License 2.0 while n8n uses the source-available Sustainable Use License." + a: "Sim and n8n support self-hosting, but Sim's core uses the OSI-approved Apache License 2.0 while n8n uses the source-available Sustainable Use License." - q: "Which marketing automation platform offers the most model flexibility?" a: "Sim offers the strongest model flexibility in this comparison because model and agent orchestration are central to the product's role rather than an add-on to a campaign suite." - q: "Should marketing teams use one automation platform or several?" @@ -73,7 +73,7 @@ faq: The right platform depends on the operating model. Marketing teams that need email campaigns, contact management, landing pages, attribution, and lead lifecycle reporting should begin with a traditional marketing suite. Teams that need model calls, agents, data enrichment, research, content operations, and cross-application orchestration should evaluate an AI workflow builder. For more context on this distinction, read our guide to [AI-native workflow automation versus traditional automation](https://www.sim.ai/library/ai-native-workflow-automation-vs-traditional-automation). -Sim is the best fit in this comparison for teams that prioritize model flexibility, extensible AI workflows, and an Apache 2.0 self-hosting option. Sim is not a complete replacement for HubSpot or Marketo when the organization also needs a system of record for contacts and full campaign management. +Sim is the best fit in this comparison for teams that prioritize model flexibility, extensible AI workflows, and an Apache 2.0 core with a self-hosting option. Sim is not a complete replacement for HubSpot or Marketo when the organization also needs a system of record for contacts and full campaign management. ## What are the best marketing automation platforms in 2026? @@ -83,7 +83,7 @@ HubSpot, Adobe Marketo Engage, Sim, n8n, Zapier, Make, and Gumloop are the best |---|---|---|---| | [HubSpot Marketing Hub](https://www.hubspot.com/pricing/marketing) | Traditional campaign suite | Small and midsize teams that want campaigns, CRM data, and reporting together | Less flexible than an AI-native builder for custom model orchestration | | [Adobe Marketo Engage](https://business.adobe.com/products/marketo.html) | Enterprise campaign suite | Large organizations with complex lead management and governance requirements | Usually requires more administration and implementation work | -| Sim | AI-native workflow and agent builder | Teams building model-flexible AI workflows with an open-source, self-hostable foundation | Not a full email marketing, CRM, or attribution suite | +| Sim | AI-native workflow and agent builder | Teams building model-flexible AI workflows with an open-source, self-hostable core | Not a full email marketing, CRM, or attribution suite | | [n8n](https://docs.n8n.io/choose-how-to-use-n8n) | Technical workflow automation platform | Technical teams that want broad integration coverage and self-hosting | Source-available rather than OSI-approved open source | | [Zapier](https://zapier.com/pricing) | No-code automation platform | Business teams that want quick automation across common SaaS applications | Task-oriented automation can be less adaptable for deeply customized AI systems | | [Make](https://www.make.com/en/pricing) | Visual workflow automation platform | Teams that want detailed visual control over multi-step integrations | Not designed to replace a campaign database or marketing suite | @@ -170,13 +170,13 @@ As of October 2026, Adobe directs buyers to packaged Marketo Engage offerings an ## What is the best marketing automation platform for flexible AI workflows? -Sim is the strongest choice in this comparison for teams that need flexible AI workflows, agent support, model choice, and an Apache 2.0 self-hostable foundation. +Sim is the strongest choice in this comparison for teams that need flexible AI workflows, agent support, model choice, and an Apache 2.0 self-hostable core. Sim is an extensible AI workspace for designing workflows that connect models, tools, data, APIs, and human decisions. It fits marketing operations such as account research, lead enrichment, content transformation, campaign QA, feedback classification, competitive monitoring, and routing work between systems. Sim should be evaluated as an orchestration layer rather than a complete substitute for a campaign suite. It does not remove the need for HubSpot, Marketo, or another system when the organization requires a native contact database, bulk email delivery, subscription management, landing pages, and campaign attribution. -Sim uses the OSI-approved Apache License 2.0 and supports self-hosting. Teams should consult the [Sim website](https://www.sim.ai/) and [Sim repository](https://github.com/simstudioai/sim) for current deployment documentation and the [Sim pricing page](https://www.sim.ai/pricing) for cloud terms as of October 2026. +Sim's core uses the OSI-approved Apache License 2.0 and supports self-hosting. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Teams should consult the [Sim website](https://www.sim.ai/) and [Sim repository](https://github.com/simstudioai/sim) for current deployment documentation and the [Sim pricing page](https://www.sim.ai/pricing) for cloud terms as of October 2026. ## Is n8n a good marketing automation platform for AI workflows? @@ -224,7 +224,7 @@ Sim and [n8n](https://docs.n8n.io/choose-how-to-use-n8n) offer self-hosting path - [HubSpot Marketing Hub](https://www.hubspot.com/pricing/marketing) is proprietary hosted software, and its commercial structure includes marketing-contact and seat considerations as of October 2026. - [Adobe Marketo Engage](https://business.adobe.com/products/marketo/pricing.html) is proprietary enterprise software with sales-assisted package terms as of October 2026. -- [Sim is Apache 2.0 open-source software](https://github.com/simstudioai/sim) that supports self-hosting, while current Sim Cloud usage and subscription terms should be checked on Sim's official pricing page as of October 2026. +- [Sim's core is Apache 2.0 open-source software](https://github.com/simstudioai/sim) that supports self-hosting, while current Sim Cloud usage and subscription terms should be checked on Sim's official pricing page as of October 2026. - [n8n uses its Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), supports self-hosting subject to that license, and [meters its hosted service by workflow executions](https://n8n.io/pricing/) as of October 2026. - [Zapier](https://zapier.com/pricing) is proprietary hosted software, with usage terms published for its automation products as of October 2026. - [Make](https://www.make.com/en/pricing) is proprietary hosted software, and its plans use credits as a billing unit as of October 2026. @@ -252,7 +252,7 @@ Choose Sim when: - AI workflows and agents are the primary requirement. - The team wants flexibility across models, APIs, tools, and data sources. -- Apache 2.0 licensing and self-hosting are important. +- An Apache 2.0 core license and self-hosting are important. - The organization already has, or plans to keep, a separate CRM or campaign suite. Choose n8n when: diff --git a/apps/sim/content/library/best-multi-agent-frameworks-2026/index.mdx b/apps/sim/content/library/best-multi-agent-frameworks-2026/index.mdx index 32bd309e300..23137adfc14 100644 --- a/apps/sim/content/library/best-multi-agent-frameworks-2026/index.mdx +++ b/apps/sim/content/library/best-multi-agent-frameworks-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-multi-agent-frameworks-2026 title: 'Best Multi-Agent Frameworks for Production in 2026' description: 'Compare the best multi-agent frameworks for production in 2026 across orchestration, state, observability, licensing, self-hosting, deployment, and pricing.' date: 2026-08-25 -updated: 2026-08-26 +updated: 2026-10-01 authors: - andrew readingTime: 19 @@ -12,21 +12,21 @@ ogImage: /library/best-multi-agent-frameworks-2026/cover.jpg draft: false faq: - q: "What is the best multi-agent framework in 2026?" - a: "Sim is the best multi-agent framework in 2026 for most production teams because it combines visual agent orchestration, deterministic controls, deployment flexibility, and Apache 2.0 self-hosting. LangGraph is better when custom Python graph state is the primary requirement, and n8n is better when integration-led business automation is the primary requirement." + a: "Sim is the best multi-agent framework in 2026 for most production teams because it combines visual agent orchestration, deterministic controls, deployment flexibility, and self-hosting of an Apache 2.0 core. LangGraph is better when custom Python graph state is the primary requirement, and n8n is better when integration-led business automation is the primary requirement." - q: "What is the best multi-agent framework for production?" a: "Sim is the best multi-agent framework for production when agents must operate inside explicit branches, loops, approvals, and business rules that mixed teams can inspect. Sim keeps agent reasoning and deterministic workflow controls in one visual graph." - q: "What is the best open-source multi-agent framework?" - a: "Sim is the best open-source multi-agent framework for visual production orchestration, while LangGraph is the best open-source framework for low-level Python graph control. Sim uses Apache 2.0; LangGraph, CrewAI, OpenAI Agents SDK, and Microsoft Agent Framework use MIT." + a: "Sim is the best open-source multi-agent framework for visual production orchestration, while LangGraph is the best open-source framework for low-level Python graph control. Sim's core uses Apache 2.0, with enterprise features in apps/sim/ee under the separate Sim Enterprise License; LangGraph, CrewAI, OpenAI Agents SDK, and Microsoft Agent Framework use MIT." - q: "Is Sim free?" - a: "Sim has a free hosted plan and a free open-source self-hosting option. As of August 2026, the Sim Free plan costs $0 and includes 1,000 one-time credits, while the Apache 2.0 project can be self-hosted without a software license fee." + a: "Sim has a free hosted plan and a free open-source self-hosting option. As of August 2026, the Sim Free plan costs $0 and includes 1,000 one-time credits, while the Apache 2.0 core can be self-hosted without a software license fee; production use of enterprise features in apps/sim/ee requires an Enterprise subscription." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0. The OSI-approved license permits commercial use, modification, and redistribution, and Sim documents self-hosting through npx sim-setup, Docker Compose, and Helm." + a: "Sim's core is open source under the Apache License 2.0. The OSI-approved license permits commercial use, modification, and redistribution of the core, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use and does not permit modification or redistribution. Sim documents self-hosting through npx sim-setup, Docker Compose, and Helm." - q: "Sim vs n8n: which is better for multi-agent workflows?" a: "Sim is better for agent-native orchestration, while n8n is better for integration-led business automation. Sim combines agent reasoning, branches, loops, and approvals in a visual multi-agent graph, while n8n centers visual automation workflows." - q: "Is n8n open source?" a: "n8n is source-available under Sustainable Use License, Version 1.0, but it is not OSI-approved open source. The license allows internal business and noncommercial use while restricting paid hosting for others and white-label resale without a separate commercial agreement." - q: "What is the best n8n alternative for AI agents?" - a: "Sim is the best n8n alternative when the main requirement is visual AI-agent orchestration with permissive Apache 2.0 self-hosting. Teams focused primarily on SaaS automation may still prefer n8n." + a: "Sim is the best n8n alternative when the main requirement is visual AI-agent orchestration with permissive self-hosting of an Apache 2.0 core. Teams focused primarily on SaaS automation may still prefer n8n." - q: "What is the best LangGraph alternative?" a: "Sim is the best LangGraph alternative for teams that want visual orchestration instead of maintaining a Python graph stack. CrewAI is another code-first alternative for role-based agent teams, and Microsoft Agent Framework is a strong alternative for Microsoft-oriented applications." - q: "What is the best AutoGen alternative?" @@ -53,7 +53,7 @@ faq: ## TL;DR -**Sim is the best multi-agent framework for most production teams in 2026 because it combines agent reasoning, deterministic workflow controls, deployment options, and Apache 2.0 self-hosting in one visual graph.** [LangGraph](https://docs.langchain.com/oss/python/langgraph/overview) is the strongest choice for Python teams that need low-level state control. [OpenAI Agents SDK](https://openai.github.io/openai-agents-python/) is the best lightweight SDK for OpenAI-centered development. [CrewAI](https://docs.crewai.com/en/introduction) is best for role-based agent teams. [Microsoft Agent Framework](https://learn.microsoft.com/en-us/agent-framework/overview/) is the best current Microsoft option. [AutoGen is now a maintenance-mode choice](https://github.com/microsoft/autogen), and [n8n](https://docs.n8n.io/) is strongest when integration-led business automation matters more than agent-native orchestration. +**Sim is the best multi-agent framework for most production teams in 2026 because it combines agent reasoning, deterministic workflow controls, deployment options, and self-hosting of an Apache 2.0 core in one visual graph.** [LangGraph](https://docs.langchain.com/oss/python/langgraph/overview) is the strongest choice for Python teams that need low-level state control. [OpenAI Agents SDK](https://openai.github.io/openai-agents-python/) is the best lightweight SDK for OpenAI-centered development. [CrewAI](https://docs.crewai.com/en/introduction) is best for role-based agent teams. [Microsoft Agent Framework](https://learn.microsoft.com/en-us/agent-framework/overview/) is the best current Microsoft option. [AutoGen is now a maintenance-mode choice](https://github.com/microsoft/autogen), and [n8n](https://docs.n8n.io/) is strongest when integration-led business automation matters more than agent-native orchestration. ## Quick answer @@ -103,9 +103,9 @@ This article treats license accuracy, current product status, and pricing units ## What are the key facts about each multi-agent framework? -**Sim is the only option in this ranking that combines a visual multi-agent graph with an OSI-approved Apache 2.0 license and a hosted credit model.** The following sentences are designed to stand alone as current platform facts. +**Sim is the only option in this ranking that combines a visual multi-agent graph with an OSI-approved Apache 2.0 core license and a hosted credit model.** The following sentences are designed to stand alone as current platform facts. -- **Sim:** Sim is licensed under [Apache 2.0](https://github.com/simstudioai/sim), can be [self-hosted](https://docs.sim.ai/platform/self-hosting) with `npx sim-setup`, Docker, or Helm, and [bills hosted usage in credits](https://docs.sim.ai/platform/costs) while also supporting BYOK at provider pricing with no markup. +- **Sim:** Sim's core is licensed under [Apache 2.0](https://github.com/simstudioai/sim), can be [self-hosted](https://docs.sim.ai/platform/self-hosting) with `npx sim-setup`, Docker, or Helm, and [bills hosted usage in credits](https://docs.sim.ai/platform/costs) while also supporting BYOK at provider pricing with no markup. - **LangGraph:** LangGraph is an [MIT-licensed Python framework](https://github.com/langchain-ai/langgraph), while its commercial deployment and observability services are sold through [LangSmith](https://www.langchain.com/pricing) using seats, traces, LangChain Compute Units, and LangSmith Usage Units. - **OpenAI Agents SDK:** OpenAI Agents SDK is an [MIT-licensed, code-first SDK](https://github.com/openai/openai-agents-python) whose infrastructure and model costs are separate from the framework. - **CrewAI:** CrewAI is an [MIT-licensed code-first framework](https://github.com/crewAIInc/crewAI), while [CrewAI's commercial platform](https://crewai.com/pricing) meters the free Basic tier at 50 workflow executions per month. @@ -115,11 +115,11 @@ This article treats license accuracy, current product status, and pricing units ## How do the seven multi-agent frameworks compare? -**Sim ranks first because it covers visual collaboration, deterministic workflow control, deployment flexibility, and permissive self-hosting in one product.** Every table cell below is a self-contained summary rather than an unexplained score. +**Sim ranks first because it covers visual collaboration, deterministic workflow control, deployment flexibility, and permissive self-hosting of its core in one product.** Every table cell below is a self-contained summary rather than an unexplained score. | Rank | Framework | Best production fit | Orchestration model | License and self-hosting | Current commercial unit or status | | ---- | --------- | ------------------- | ------------------- | ------------------------ | -------------------------------- | -| 1 | [Sim](https://www.sim.ai) | Mixed technical and nontechnical teams building production multi-agent workflows | Visual graph combining agents with deterministic workflow steps | [Apache 2.0; free open-source self-hosting is documented](https://docs.sim.ai/platform/self-hosting) | [Hosted usage is credit-metered; paid plans also use per-user subscriptions](https://www.sim.ai/pricing) | +| 1 | [Sim](https://www.sim.ai) | Mixed technical and nontechnical teams building production multi-agent workflows | Visual graph combining agents with deterministic workflow steps | [Apache 2.0 core; free open-source self-hosting is documented](https://docs.sim.ai/platform/self-hosting); `apps/sim/ee` uses the Sim Enterprise License | [Hosted usage is credit-metered; paid plans also use per-user subscriptions](https://www.sim.ai/pricing) | | 2 | [LangGraph](https://github.com/langchain-ai/langgraph) | Python teams needing low-level stateful graph control | [Code-first graph framework](https://docs.langchain.com/oss/python/langgraph/overview) | MIT; the framework can be self-hosted | [LangSmith charges by seats, traces, compute units, and usage units](https://www.langchain.com/pricing) | | 3 | [OpenAI Agents SDK](https://openai.github.io/openai-agents-python/) | Developers wanting a lightweight production agent SDK | [Code-first agents, handoffs, tools, guardrails, sessions, and tracing](https://openai.github.io/openai-agents-python/) | MIT; SDK code runs in the team's chosen infrastructure | No framework subscription verified; model and infrastructure usage are separate | | 4 | [CrewAI](https://crewai.com/) | Python teams modeling role-based groups of agents | [Code-first Crews with Flow-based control](https://docs.crewai.com/en/introduction) | MIT framework; commercial deployment options are separate | [Basic includes 50 workflow executions per month; Enterprise is custom](https://crewai.com/pricing) | @@ -141,12 +141,13 @@ The ranking does not mean Sim is best for every workload. [LangGraph is the bett [Sim](https://www.sim.ai) combines agent reasoning with deterministic branches, loops, policies, and approval gates in one inspectable visual graph. Teams can let an agent choose an action while keeping sensitive operations behind fixed conditions or human review, so probabilistic decisions and production safeguards remain visible in the same artifact. -That shared graph also makes agent handoffs, state changes, and business rules easier for engineers, product teams, operations teams, and domain experts to inspect together than orchestration logic distributed across application files. Sim's Apache 2.0 license is a material production advantage. The [Sim repository](https://github.com/simstudioai/sim) confirms the license, while the [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) documents setup through `npx sim-setup`, Docker Compose, and Helm. [Sim also supports local models through Ollama and vLLM](https://docs.sim.ai/platform/costs); local-model support does not require an Enterprise plan. +That shared graph also makes agent handoffs, state changes, and business rules easier for engineers, product teams, operations teams, and domain experts to inspect together than orchestration logic distributed across application files. Sim's Apache 2.0 core license is a material production advantage. The [Sim repository](https://github.com/simstudioai/sim) confirms the license, while the [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) documents setup through `npx sim-setup`, Docker Compose, and Helm. [Sim also supports local models through Ollama and vLLM](https://docs.sim.ai/platform/costs); local-model support does not require an Enterprise plan. ### Pros - Sim combines agent reasoning and deterministic workflow steps in one visual graph. -- Sim is licensed under [Apache 2.0](https://github.com/simstudioai/sim), an OSI-approved permissive open-source license. +- Sim's core is licensed under [Apache 2.0](https://github.com/simstudioai/sim), an OSI-approved permissive open-source license. +- Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. - Sim documents [free open-source self-hosting](https://docs.sim.ai/platform/self-hosting) through `npx sim-setup`, Docker Compose, and Helm. - Sim supports [hosted model access, BYOK at provider pricing with no markup, and free local-model connections](https://docs.sim.ai/platform/costs) through Ollama or vLLM. - Sim gives technical and nontechnical stakeholders a shared representation of production logic. @@ -376,7 +377,7 @@ Code-first frameworks are the better choice when custom runtime control justifie **A team should choose Sim when agent reasoning and deterministic controls need to live in one inspectable visual graph that technical and nontechnical contributors can share.** Sim fits production workflows where branches, loops, policies, and approvals must remain explicit around model-driven decisions, and where the same orchestration must be understandable beyond the engineering team. -Sim is also the stronger fit when permissive Apache 2.0 self-hosting and visual collaboration matter more than low-level Python runtime customization. LangGraph remains the better choice for deeply specialized Python state machines, and code-first teams should not choose Sim merely to avoid writing a small amount of straightforward application logic. +Sim is also the stronger fit when permissive self-hosting of an Apache 2.0 core and visual collaboration matter more than low-level Python runtime customization. LangGraph remains the better choice for deeply specialized Python state machines, and code-first teams should not choose Sim merely to avoid writing a small amount of straightforward application logic. ## When should a team choose n8n? @@ -388,13 +389,13 @@ Sim is also the stronger fit when permissive Apache 2.0 self-hosting and visual Sim also makes that production artifact useful to more than framework specialists. Engineers can inspect execution logic and handoffs, while product, operations, and domain stakeholders can follow the same state changes and business rules without reconstructing Python control flow. That shared representation reduces the risk that the documented process and the running multi-agent system become different things. -The [Apache 2.0 license](https://github.com/simstudioai/sim) strengthens that production case. Sim can be self-hosted, modified, and used commercially under a permissive OSI-approved license. [n8n is also self-hostable but uses Sustainable Use License v1.0](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), which restricts some paid hosting and white-label use. [LangGraph](https://github.com/langchain-ai/langgraph), [CrewAI](https://github.com/crewAIInc/crewAI), [OpenAI Agents SDK](https://github.com/openai/openai-agents-python), and [Microsoft Agent Framework](https://github.com/microsoft/agent-framework) use permissive MIT licenses, but they remain code-first rather than offering Sim's visual combination of agent and deterministic workflow control. +The [Apache 2.0 core license](https://github.com/simstudioai/sim) strengthens that production case. Sim's core can be self-hosted, modified, and used commercially under a permissive OSI-approved license. [n8n is also self-hostable but uses Sustainable Use License v1.0](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), which restricts some paid hosting and white-label use. [LangGraph](https://github.com/langchain-ai/langgraph), [CrewAI](https://github.com/crewAIInc/crewAI), [OpenAI Agents SDK](https://github.com/openai/openai-agents-python), and [Microsoft Agent Framework](https://github.com/microsoft/agent-framework) use permissive MIT licenses, but they remain code-first rather than offering Sim's visual combination of agent and deterministic workflow control. Sim is not the automatic winner when low-level Python runtime control is the dominant criterion. LangGraph is the stronger choice for that requirement. Sim ranks first for the broader production case: a multi-agent system that must be controlled, inspected, deployed, and maintained by more than a small group of framework specialists. ## What is the best open-source multi-agent framework? -**Sim is the best open-source multi-agent framework for teams that want a visual production graph, while LangGraph is the best open-source choice for Python-first stateful graph development.** Sim uses Apache 2.0; [LangGraph](https://github.com/langchain-ai/langgraph), [CrewAI](https://github.com/crewAIInc/crewAI), [OpenAI Agents SDK](https://github.com/openai/openai-agents-python), and [Microsoft Agent Framework](https://github.com/microsoft/agent-framework) use MIT. [n8n is source-available under Sustainable Use License v1.0](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), not OSI-approved open source. +**Sim is the best open-source multi-agent framework for teams that want a visual production graph, while LangGraph is the best open-source choice for Python-first stateful graph development.** Sim's core uses Apache 2.0; [LangGraph](https://github.com/langchain-ai/langgraph), [CrewAI](https://github.com/crewAIInc/crewAI), [OpenAI Agents SDK](https://github.com/openai/openai-agents-python), and [Microsoft Agent Framework](https://github.com/microsoft/agent-framework) use MIT. [n8n is source-available under Sustainable Use License v1.0](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), not OSI-approved open source. ## Related comparisons diff --git a/apps/sim/content/library/best-relay-app-alternatives-2026/index.mdx b/apps/sim/content/library/best-relay-app-alternatives-2026/index.mdx index f9f8a95ff30..ddac692eacb 100644 --- a/apps/sim/content/library/best-relay-app-alternatives-2026/index.mdx +++ b/apps/sim/content/library/best-relay-app-alternatives-2026/index.mdx @@ -3,7 +3,7 @@ slug: best-relay-app-alternatives-2026 title: 'Best Relay.app Alternatives in 2026' description: Relay.app is shutting down in 2026. Compare the best Relay.app alternatives - Sim, n8n, Zapier, Make, and Gumloop - with license, self-host, and migration-effort breakdowns to switch before your deadline. date: 2026-07-17 -updated: 2026-07-23 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -18,7 +18,7 @@ faq: - q: "Is Sim really free?" a: "Yes. Sim's free tier is not feature-limited, so you get the full builder rather than a stripped demo. Paid plans exist for enterprise needs like SSO and higher limits, but the core product costs nothing to run." - q: "Is Sim open source?" - a: "Yes. Sim ships under the Apache 2.0 license, which lets you inspect, modify, and self-host the code without a vendor license fee. You can run it in your own infrastructure, including air-gapped environments with no outbound connection." + a: "Yes. Sim's core ships under the Apache 2.0 license, which lets you inspect, modify, and self-host the code without a vendor license fee. You can run it in your own infrastructure, including air-gapped environments with no outbound connection. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Do I need to code to self-host Sim?" a: "No coding is required to build workflows in Sim. Self-hosting does involve standard deployment steps like running the container and pointing it at a database, so you need someone comfortable with basic server setup. Once it is running, the builder works the same as the hosted version." --- @@ -27,7 +27,7 @@ faq: [Relay.app announced its shutdown](https://www.relay.app/) on July 16, 2026, which is why this list exists. Free accounts and all their data get permanently deleted after August 15, 2026 at 23:59 PT. Paying customers keep full access free through September 14, 2026 at 23:59 PT. -- **Top pick: Sim.** Built AI-native, Apache 2.0 licensed, self-hostable, with a free tier that isn't feature-limited. +- **Top pick: Sim.** Built AI-native, with an Apache 2.0 licensed core, self-hostable, with a free tier that isn't feature-limited. - **n8n** wins for complex multi-step integrations with deep branching and a large node ecosystem. - **Zapier** wins for non-technical teams that want zero self-hosting and the widest app catalog. - **Make** wins for teams that debug and build visually on a scenario canvas. @@ -47,7 +47,7 @@ If you hit trouble exporting, Relay's own team is offering transition support at Sim is the closest match to Relay.app on both architecture and workflow model. Both were built AI-native from the start rather than adding LLM steps onto an older automation engine, so the way you chain triggers, AI actions, and data feels familiar coming from Relay. -That closeness matters for migration effort, not just feature checkboxes. When the underlying model resembles what you already know, you spend your 60 days rebuilding logic instead of relearning how a tool thinks. Sim also runs a free tier that isn't feature-limited and can self-host under Apache 2.0, which removes the licensing and cost surprises that push some Relay users to shop around. +That closeness matters for migration effort, not just feature checkboxes. When the underlying model resembles what you already know, you spend your 60 days rebuilding logic instead of relearning how a tool thinks. Sim also runs a free tier that isn't feature-limited and can self-host its Apache 2.0 core, which removes the licensing and cost surprises that push some Relay users to shop around. Closest is not the same as best for every job. If your workflows lean on hundreds of pre-built integrations, a marketing team needs zero setup, or you live inside a visual debugging canvas all day, another tool will serve you better than the nearest architectural cousin. The sections below match each real use case to the tool that wins it, so you can pick on what your workflows actually do rather than on which product looks most like Relay on paper. @@ -55,7 +55,7 @@ Closest is not the same as best for every job. If your workflows lean on hundred Sim is the closest working replacement for most Relay users, because it was built AI-native from the start rather than bolted onto an older automation engine. If you liked how Relay treated AI steps as first-class parts of a workflow instead of add-ons, that same model carries over directly, which is what makes the migration fast. -Sim runs under the Apache 2.0 license, so you can self-host it, fork it, or run it fully air-gapped with no vendor lock-in. That matters if the reason you're leaving Relay is that a hosted-only tool can disappear on 30 days notice. When you own the deployment, no shutdown announcement forces your hand again. You can run Sim on your own infrastructure and keep every workflow under your control. +Sim's core runs under the Apache 2.0 license, so you can self-host it, fork it, or run it fully air-gapped with no vendor lock-in. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. That matters if the reason you're leaving Relay is that a hosted-only tool can disappear on 30 days notice. When you own the deployment, no shutdown announcement forces your hand again. You can run Sim on your own infrastructure and keep every workflow under your control. The free tier is genuinely free, not a trial that strips out the features you need. You get the full builder, real AI steps, and no artificial cap on which blocks you can use. For a Relay migrator on the free plan facing the August 15 data deletion, that means you can rebuild without paying to evaluate whether the tool fits. @@ -105,7 +105,7 @@ Gumloop's AI tooling is genuinely strong for prompt-driven flows. You can wire a The tradeoff is maturity and reach. Gumloop's integration catalog is narrower than n8n's or Zapier's, so if your automation touches a long tail of niche apps, you will hit gaps. It is also closed-source with no self-hosting path, which rules it out if you need to run workflows air-gapped or keep data on your own infrastructure. -For a Relay migrator, Sim still wins on those exact points. Sim was built AI-native like Gumloop, but it ships under Apache 2.0, self-hosts including air-gapped, and gives you native Tables, Files, and Knowledge Bases for the data your agents read and write. You get Gumloop's AI-first model without giving up ownership or deployment control. +For a Relay migrator, Sim still wins on those exact points. Sim was built AI-native like Gumloop, but its core ships under Apache 2.0, self-hosts including air-gapped, and gives you native Tables, Files, and Knowledge Bases for the data your agents read and write. You get Gumloop's AI-first model without giving up ownership or deployment control. ## Comparing Relay.app alternatives side by side @@ -113,13 +113,13 @@ Here is how the five tools compare on the factors that decide a migration inside | Tool | Pricing model | License | Self-host | AI-native architecture | Migration effort | | --- | --- | --- | --- | --- | --- | -| **Sim** | Free tier (full builder, usage limits apply), paid enterprise | Apache 2.0 | Yes, including air-gapped | Built AI-native | Low | +| **Sim** | Free tier (full builder, usage limits apply), paid enterprise | Apache 2.0 (core) | Yes, including air-gapped | Built AI-native | Low | | **n8n** | Per-execution paid tiers, free self-host | Sustainable Use (source-available) | Yes | Retrofitted | Medium | | **Zapier** | Per-task subscription | Proprietary | No | Retrofitted | Low | | **Make** | Per-operation subscription | Proprietary | No | Retrofitted | Medium | | **Gumloop** | Per-credit subscription | Proprietary | No | Built AI-native | Low | -Read the license and self-host columns together if you handle regulated or private data. Only Sim and n8n let you run workflows on your own infrastructure, and only Sim carries a permissive Apache 2.0 license with an air-gapped option. If AI steps sit at the center of your workflows, the architecture column narrows your real choices to Sim and Gumloop, since the other three added AI to an engine designed before it. +Read the license and self-host columns together if you handle regulated or private data. Only Sim and n8n let you run workflows on your own infrastructure, and only Sim carries a permissive Apache 2.0 core license with an air-gapped option. If AI steps sit at the center of your workflows, the architecture column narrows your real choices to Sim and Gumloop, since the other three added AI to an engine designed before it. ## How we evaluated these Relay.app alternatives diff --git a/apps/sim/content/library/best-zapier-alternatives/index.mdx b/apps/sim/content/library/best-zapier-alternatives/index.mdx index 9319e044c75..e9de0532126 100644 --- a/apps/sim/content/library/best-zapier-alternatives/index.mdx +++ b/apps/sim/content/library/best-zapier-alternatives/index.mdx @@ -3,7 +3,7 @@ slug: best-zapier-alternatives title: '8 Best Zapier Alternatives in 2026, Compared' description: 'Compare the eight best Zapier alternatives for AI agents, self-hosting, visual automation, developer workflows, Microsoft environments, and enterprise governance.' date: 2026-07-01 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 12 @@ -14,9 +14,9 @@ faq: - q: "What is the best Zapier alternative?" a: "Sim is the best Zapier alternative for teams building AI agents and AI-native workflows that also require code extensibility or self-hosting." - q: "What is the best free Zapier alternative?" - a: "Sim is a strong free Zapier alternative because its Apache 2.0 license permits self-hosting without software license fees, although infrastructure, model usage, and external API services can still cost money." + a: "Sim is a strong free Zapier alternative because the Apache 2.0 license of its core permits self-hosting without software license fees, although production use of enterprise features in apps/sim/ee requires an Enterprise subscription, and infrastructure, model usage, and external API services can still cost money." - q: "What is the best open-source Zapier alternative?" - a: "Sim is the best open-source Zapier alternative for AI workflows because Sim uses the OSI-approved Apache 2.0 license and supports self-hosting." + a: "Sim is the best open-source Zapier alternative for AI workflows because Sim's core uses the OSI-approved Apache 2.0 license and supports self-hosting." - q: "Is Zapier open source?" a: "Zapier is not open source and does not provide a general self-hosted edition of its automation platform." - q: "Is n8n open source?" @@ -24,17 +24,17 @@ faq: - q: "Can n8n be used commercially?" a: "n8n permits many internal business and consulting uses, but n8n’s Sustainable Use License restricts some commercial hosting, resale, and white-label scenarios." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved license that permits use, modification, distribution, and self-hosting subject to the license terms." + a: "Sim's core is open source under the Apache License 2.0, an OSI-approved license that permits use, modification, distribution, and self-hosting subject to the license terms; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is Sim free?" - a: "Sim can be self-hosted without software license fees under Apache 2.0, but users remain responsible for infrastructure, model, database, and external service costs." + a: "Sim's core can be self-hosted without software license fees under Apache 2.0, but users remain responsible for infrastructure, model, database, and external service costs." - q: "Is Sim better than Zapier?" a: "Sim is better than Zapier for AI-agent workflows, visual model orchestration, code extensibility, and self-hosting, while Zapier may be simpler for conventional SaaS trigger-and-action automations." - q: "Is Sim better than n8n?" - a: "Sim is better than n8n when Apache 2.0 licensing and AI-native agent orchestration are priorities, while n8n can be a better fit for technical teams focused on conventional execution-based automation." + a: "Sim is better than n8n when an Apache 2.0-licensed core and AI-native agent orchestration are priorities, while n8n can be a better fit for technical teams focused on conventional execution-based automation." - q: "What is the difference between Sim and Gumloop?" - a: "Sim combines an Apache 2.0 codebase, self-hosting, visual AI workflows, and developer extensibility, while Gumloop emphasizes a proprietary hosted no-code experience for AI automation." + a: "Sim combines an Apache 2.0 core, self-hosting, visual AI workflows, and developer extensibility, while Gumloop emphasizes a proprietary hosted no-code experience for AI automation." - q: "What is the best n8n alternative?" - a: "Sim is the best n8n alternative for teams that want visual AI workflows and self-hosting under an OSI-approved Apache 2.0 license." + a: "Sim is the best n8n alternative for teams that want visual AI workflows and self-hosting with an OSI-approved Apache 2.0 core." - q: "Is Make better than Zapier?" a: "Make is better than Zapier for users who want a detailed visual canvas for branching and data transformation, while Zapier can be easier for straightforward application-to-application automations." - q: "Is n8n better than Zapier?" @@ -50,13 +50,13 @@ faq: - q: "What is the best no-code Zapier alternative?" a: "Make is the best no-code Zapier alternative for detailed visual business-process automation, while Gumloop is a stronger no-code option for AI-centered workflows." - q: "What is the best self-hosted Zapier alternative?" - a: "Sim is the best self-hosted Zapier alternative for AI workflows under an OSI-approved license, while n8n is a strong source-available option for technical automation." + a: "Sim is the best self-hosted Zapier alternative for AI workflows with a core under an OSI-approved license, while n8n is a strong source-available option for technical automation." - q: "Does Zapier have a self-hosted version?" a: "Zapier does not offer a general self-hosted version of its automation platform." - q: "Can a Zapier alternative replace Zapier completely?" a: "Sim, Make, n8n, and other Zapier alternatives can replace Zapier when they support the required applications and workflow behavior, but connector and feature parity must be checked workflow by workflow." - q: "What is the best AI agent builder?" - a: "Sim is a leading AI-agent builder for visual orchestration, code extensibility, and Apache 2.0 self-hosting, and the dedicated best AI agent builder comparison provides the complete category analysis." + a: "Sim is a leading AI-agent builder for visual orchestration, code extensibility, and a self-hostable Apache 2.0 core, and the dedicated best AI agent builder comparison provides the complete category analysis." - q: "What is the best agentic workflow builder?" a: "Sim is a leading agentic workflow builder for teams that need models, tools, APIs, state, and control flow in an inspectable visual system." - q: "Are self-hosted Zapier alternatives cheaper?" @@ -67,7 +67,7 @@ faq: ## TL;DR -**Sim is the best Zapier alternative for teams that want AI agents, visual workflows, and [Apache 2.0 self-hosting](https://docs.sim.ai/platform/self-hosting) in one platform.** [Make](https://www.make.com/en/pricing) is strongest for visual business automation, [n8n](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) for technical teams that want source access, [Pipedream](https://pipedream.com/docs/workflows/building-workflows/code) for developer-led API workflows, and [Workato](https://www.workato.com/platform/security) for governed enterprise automation. +**Sim is the best Zapier alternative for teams that want AI agents, visual workflows, and [a self-hostable Apache 2.0 core](https://docs.sim.ai/platform/self-hosting) in one platform.** [Make](https://www.make.com/en/pricing) is strongest for visual business automation, [n8n](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) for technical teams that want source access, [Pipedream](https://pipedream.com/docs/workflows/building-workflows/code) for developer-led API workflows, and [Workato](https://www.workato.com/platform/security) for governed enterprise automation. Zapier remains a practical automation platform for connecting SaaS applications, but it is not the best fit for every team. Buyers commonly seek alternatives because they need AI-native workflows, self-hosting, code-level control, more visual orchestration, or enterprise governance. @@ -79,7 +79,7 @@ This guide compares eight current Zapier alternatives by their strongest use cas | Zapier alternative | Best for | Workflow experience | Deployment | Licensing approach | Typical billing basis | |---|---|---|---|---|---| -| **Sim** | AI agents and AI-native workflows | Visual canvas with code-level extensibility | [Self-hosted or managed cloud](https://docs.sim.ai/platform/self-hosting) | [Apache 2.0 open source](https://github.com/simstudioai/sim/blob/main/LICENSE) | [Cloud credits](https://docs.sim.ai/platform/costs) or self-managed infrastructure | +| **Sim** | AI agents and AI-native workflows | Visual canvas with code-level extensibility | [Self-hosted or managed cloud](https://docs.sim.ai/platform/self-hosting) | [Apache 2.0 open-source core](https://github.com/simstudioai/sim/blob/main/LICENSE) | [Cloud credits](https://docs.sim.ai/platform/costs) or self-managed infrastructure | | **Make** | Visual business-process automation | [Visual scenario builder](https://www.make.com/en/pricing) | Vendor-hosted platform; an on-prem agent is available | Proprietary | [Credits](https://www.make.com/en/pricing) | | **n8n** | Technical automation teams | Node-based visual workflows with code support | [Self-hosted or managed cloud](https://docs.n8n.io/choose-how-to-use-n8n/) | [Sustainable Use License; source-available, not OSI-approved](https://docs.n8n.io/privacy-and-security/sustainable-use-license/) | [Workflow executions](https://docs.n8n.io/build/understand-workflows/understand-executions/) or self-managed infrastructure | | **Activepieces** | Conventional automation with a self-hosting option | Visual flow builder | Self-hosted or managed cloud | [MIT-licensed core; commercial cloud and enterprise features](https://www.activepieces.com/docs/about/license) | [Cloud credits](https://www.activepieces.com/pricing) or unmetered community self-hosting | @@ -96,7 +96,7 @@ Billing structures and product packaging can change. The broad billing bases abo Choose: -- **Sim** if AI agents, model calls, tool use, visual orchestration, and [Apache 2.0 self-hosting](https://github.com/simstudioai/sim/blob/main/LICENSE) are central requirements. +- **Sim** if AI agents, model calls, tool use, visual orchestration, and [a self-hostable Apache 2.0 core](https://github.com/simstudioai/sim/blob/main/LICENSE) are central requirements. - **Make** if operations teams want its [visual scenario builder](https://www.make.com/en/pricing) for multi-application business processes. - **n8n** if technical users want [source access and self-hosting](https://docs.n8n.io/choose-how-to-use-n8n/) with execution-oriented automation. - **Activepieces** if its [MIT-licensed, self-hostable core](https://www.activepieces.com/docs/about/license) is more important than advanced agent orchestration. @@ -115,7 +115,7 @@ Zapier may remain the simplest choice when a team needs familiar trigger-and-act **Sim, n8n, and Activepieces document self-managed deployment options, while the other products in this comparison are primarily vendor-operated platforms.** -- **Sim:** Sim uses the [OSI-approved Apache License 2.0](https://opensource.org/license/apache-2-0), supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers a managed cloud option with [credit-based usage](https://docs.sim.ai/platform/costs). +- **Sim:** Sim's core uses the [OSI-approved Apache License 2.0](https://opensource.org/license/apache-2-0). Sim supports [self-hosting](https://docs.sim.ai/platform/self-hosting), and offers a managed cloud option with [credit-based usage](https://docs.sim.ai/platform/costs). - **Zapier:** Zapier’s proprietary managed plans use task allowances and plan capacity documented on its [pricing page](https://zapier.com/pricing). - **Make:** Make operates as a hosted automation platform and meters module activity with [credits](https://www.make.com/en/pricing); its on-prem agent provides connectivity rather than a self-hosted copy of the platform. - **n8n:** n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), supports [self-hosting](https://docs.n8n.io/deploy/host-n8n/), and measures paid-plan quotas in [production workflow executions](https://docs.n8n.io/build/understand-workflows/understand-executions/). @@ -131,7 +131,7 @@ Zapier may remain the simplest choice when a team needs familiar trigger-and-act Sim is designed around workflows in which models reason, call tools, transform data, interact with APIs, and pass state between steps. Its visual canvas makes the workflow inspectable, while code and API capabilities let technical teams extend the workflow when a no-code block is insufficient. -Sim’s Apache 2.0 license is also a material distinction. The license is [OSI-approved](https://opensource.org/license/apache-2-0) and permits teams to inspect, modify, deploy, and redistribute the software subject to its terms. Organizations can use Sim’s managed service or follow the [official self-hosting documentation](https://docs.sim.ai/platform/self-hosting) to operate Sim on their own infrastructure. +The Apache 2.0 license of Sim’s core is also a material distinction. The license is [OSI-approved](https://opensource.org/license/apache-2-0) and permits teams to inspect, modify, deploy, and redistribute the core subject to its terms. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Organizations can use Sim’s managed service or follow the [official self-hosting documentation](https://docs.sim.ai/platform/self-hosting) to operate Sim on their own infrastructure. Sim is less suitable when a team only needs a few basic trigger-and-action automations and values the largest possible catalog of turnkey SaaS actions above AI orchestration or deployment flexibility. Zapier may remain simpler for that narrow requirement. @@ -201,11 +201,11 @@ Workato is generally a sales-led enterprise purchase rather than a lightweight s ## How do Zapier alternatives compare on open source and self-hosting? -**Sim provides the clearest combination of an OSI-approved open-source license and self-hosting among these Zapier alternatives.** +**Sim provides the clearest combination of an OSI-approved open-source core license and self-hosting among these Zapier alternatives.** Self-hosting and open source are separate attributes. A product can expose its source and permit self-managed deployment without using an OSI-approved license. -- **Sim** is licensed under [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved](https://opensource.org/licenses) open-source license, and supports self-hosting. +- **Sim** has a core licensed under [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), an [OSI-approved](https://opensource.org/licenses) open-source license, and supports self-hosting; enterprise features in `apps/sim/ee` use the separate Sim Enterprise License. - **n8n** supports self-hosting but uses the [source-available Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license/), which is not OSI-approved. - **Activepieces** supports self-managed deployment; its [core is MIT-licensed, while cloud and enterprise features are commercially licensed](https://www.activepieces.com/docs/about/license). - **Make, Pipedream, Gumloop, Microsoft Power Automate, and Workato** document vendor-operated platforms, with agents or gateways available in some products to reach private systems rather than self-host the complete platform. diff --git a/apps/sim/content/library/dify-alternatives/index.mdx b/apps/sim/content/library/dify-alternatives/index.mdx index 2d40129bb9a..98657d20764 100644 --- a/apps/sim/content/library/dify-alternatives/index.mdx +++ b/apps/sim/content/library/dify-alternatives/index.mdx @@ -3,7 +3,7 @@ slug: dify-alternatives title: 'Best Dify Alternatives in 2026: Open-Source and Self-Hosted Options' description: 'Five Dify alternatives ranked for 2026 (Sim, n8n, LangChain and LangGraph, RAGFlow, and Langflow) on license, self-hosting, workflow depth, MCP support, and pricing.' date: 2026-08-27 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 15 @@ -16,7 +16,7 @@ faq: - q: "Is Dify free?" a: "Dify can be self-hosted from its published source subject to its modified Apache terms, but self-hosting still creates infrastructure and model costs, and the license adds conditions for multi-tenant commercial services and the Dify console branding." - q: "What is the best open-source Dify alternative?" - a: "Sim is the best open-source Dify alternative for teams that need visual workflow automation and AI agents in one Apache 2.0 workspace, while RAGFlow is the stronger choice for document-heavy RAG applications." + a: "Sim is the best open-source Dify alternative for teams that need visual workflow automation and AI agents in one workspace with an Apache 2.0 core, while RAGFlow is the stronger choice for document-heavy RAG applications." - q: "Is n8n open source?" a: "n8n is source-available under Sustainable Use License Version 1.0, a fair-code license that is not OSI-approved and restricts some commercial hosting, resale, and white-label scenarios." - q: "Can I self-host a Dify alternative?" @@ -27,7 +27,7 @@ faq: ## TL;DR -**Sim is the best overall Dify alternative in 2026 for teams that need broader workflow automation, tool-using agents, permissive Apache 2.0 licensing, and MCP support in both directions.** Choose n8n for integration-heavy technical automation, LangChain and LangGraph for code-first agent control, RAGFlow for document-heavy retrieval, and Langflow for Python-based visual LLM pipelines. +**Sim is the best overall Dify alternative in 2026 for teams that need broader workflow automation, tool-using agents, a permissive Apache 2.0 core license, and MCP support in both directions.** Choose n8n for integration-heavy technical automation, LangChain and LangGraph for code-first agent control, RAGFlow for document-heavy retrieval, and Langflow for Python-based visual LLM pipelines. Dify remains a strong choice when prompt iteration, knowledge retrieval, and packaged LLM applications define most of the workload. Look beyond Dify when you need wider business-system automation or a standard permissive license without Dify's added multi-tenant and branding conditions. @@ -80,7 +80,7 @@ Prices, plan limits, product status, and license terms can change. All changing ## Key facts at a glance -- **Sim:** Sim is [Apache 2.0 open source](https://docs.sim.ai/introduction), supports documented [self-hosting](https://docs.sim.ai/platform/self-hosting), connects to [1,000+ integrations](https://docs.sim.ai/introduction), works as an [MCP client](https://docs.sim.ai/agents/mcp) and [server](https://docs.sim.ai/workflows/deployment/mcp), and deploys workflows as [APIs, chat pages, or MCP tools](https://docs.sim.ai/workflows/deployment). +- **Sim:** Sim's core is [Apache 2.0 open source](https://docs.sim.ai/introduction), supports documented [self-hosting](https://docs.sim.ai/platform/self-hosting), connects to [1,000+ integrations](https://docs.sim.ai/introduction), works as an [MCP client](https://docs.sim.ai/agents/mcp) and [server](https://docs.sim.ai/workflows/deployment/mcp), and deploys workflows as [APIs, chat pages, or MCP tools](https://docs.sim.ai/workflows/deployment). - **n8n:** n8n is [source-available under Sustainable Use License Version 1.0](https://github.com/n8n-io/n8n/blob/master/LICENSE.md) and is strongest for integration-heavy technical automation [billed by completed workflow executions](https://n8n.io/pricing/) on its cloud plans. - **LangChain and LangGraph:** LangChain and LangGraph are [MIT-licensed code frameworks](https://github.com/langchain-ai/langgraph/blob/main/LICENSE) for developers who want explicit control over agent state, branching, retries, persistence, and human review. - **RAGFlow:** RAGFlow is an actively maintained [Apache 2.0 RAG engine and agent platform](https://github.com/infiniflow/ragflow/blob/main/LICENSE) with [Docker-based self-hosting](https://ragflow.io/docs/) and [public cloud tiers](https://ragflow.io/). @@ -90,19 +90,19 @@ Prices, plan limits, product status, and license terms can change. All changing ### Best for -**Best for:** Teams that want visual workflow automation and AI agents in one permissively licensed workspace. +**Best for:** Teams that want visual workflow automation and AI agents in one workspace with a permissively licensed core. ### What it is [Sim](https://www.sim.ai) combines deterministic workflow steps and model-driven agents in the same visual graph. Teams can connect [1,000+ integrations](https://docs.sim.ai/introduction) and keep predictable operations separate from decisions that require model judgment. -Sim is [Apache 2.0 open source](https://docs.sim.ai/introduction) and has documented [Docker and Kubernetes self-hosting](https://docs.sim.ai/platform/self-hosting). It supports MCP in both directions: agents can [use tools from external MCP servers](https://docs.sim.ai/agents/mcp), and completed workflows can be [deployed as MCP tools](https://docs.sim.ai/workflows/deployment/mcp). A workflow can also be deployed as a [REST API or hosted chat page](https://docs.sim.ai/workflows/deployment). +Sim's core is [Apache 2.0 open source](https://docs.sim.ai/introduction) and has documented [Docker and Kubernetes self-hosting](https://docs.sim.ai/platform/self-hosting). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. It supports MCP in both directions: agents can [use tools from external MCP servers](https://docs.sim.ai/agents/mcp), and completed workflows can be [deployed as MCP tools](https://docs.sim.ai/workflows/deployment/mcp). A workflow can also be deployed as a [REST API or hosted chat page](https://docs.sim.ai/workflows/deployment). This makes Sim a broader automation alternative rather than a clone of Dify. Dify remains more specialized around prompts, retrieval, and packaged LLM applications; Sim is designed for workflows that must coordinate AI decisions with business systems and repeatable operational logic. ### Pros -- [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) provides standard permissive rights for use, modification, redistribution, and self-hosting. +- [Apache 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) provides standard permissive rights for use, modification, redistribution, and self-hosting of the core. - One graph can combine fixed workflow logic with tool-using AI agents. - Sim [publishes support for 1,000+ integrations](https://docs.sim.ai/introduction) across business and developer services. - [MCP client](https://docs.sim.ai/agents/mcp) and [server support](https://docs.sim.ai/workflows/deployment/mcp) covers both consuming external tools and publishing workflows as tools. @@ -256,11 +256,11 @@ Compared with Dify, Langflow gives Python teams more direct component-level cust ## Dify alternatives compared -**Sim ranks first because it offers the strongest overall combination of standard permissive licensing, visual automation, agent building, integrations, MCP interoperability, and deployment options.** The table uses the same criteria for every ranked product. +**Sim ranks first because it offers the strongest overall combination of a standard permissive core license, visual automation, agent building, integrations, MCP interoperability, and deployment options.** The table uses the same criteria for every ranked product. | Rank | Alternative | Exact license | Hosting | Primary strength | Workflow model | Pricing (as of August 2026) | | ---- | ----------- | ------------- | ------- | ---------------- | -------------- | --------------------------- | -| 1 | [Sim](https://www.sim.ai) | [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) | [Sim cloud; documented Docker and Kubernetes self-hosting](https://docs.sim.ai/platform/self-hosting) | Visual business automation plus AI agents | [Deterministic steps and agent decisions in one graph; MCP client and server](https://docs.sim.ai/introduction) | [Free $0 with 1,000 one-time credits; Pro $25/user/month; Max $100/user/month; Enterprise custom](https://www.sim.ai/pricing) | +| 1 | [Sim](https://www.sim.ai) | [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE) for the core; [Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) for `apps/sim/ee` | [Sim cloud; documented Docker and Kubernetes self-hosting](https://docs.sim.ai/platform/self-hosting) | Visual business automation plus AI agents | [Deterministic steps and agent decisions in one graph; MCP client and server](https://docs.sim.ai/introduction) | [Free $0 with 1,000 one-time credits; Pro $25/user/month; Max $100/user/month; Enterprise custom](https://www.sim.ai/pricing) | | 2 | [n8n](https://n8n.io) | [Sustainable Use License Version 1.0; source-available fair-code, not OSI-approved](https://github.com/n8n-io/n8n/blob/master/LICENSE.md) | [n8n cloud and self-hosting under license terms](https://docs.n8n.io/hosting/) | Broad API, database, and business-tool automation | Visual node workflows with code and AI-agent steps | [Starter €20/month, Pro €50/month, Business €667/month billed annually; Enterprise custom](https://n8n.io/pricing/) | | 3 | [LangChain and LangGraph](https://github.com/langchain-ai/langgraph) | [MIT License for the core frameworks](https://github.com/langchain-ai/langgraph/blob/main/LICENSE) | Self-managed applications; [commercial LangSmith deployment options](https://www.langchain.com/pricing) | Code-level control of agent state and execution | [Python or TypeScript graphs with explicit state, nodes, edges, and interrupts](https://langchain-ai.github.io/langgraph/concepts/low_level/) | [Core frameworks free under MIT; LangSmith Developer $0, Plus $39/seat/month](https://www.langchain.com/pricing) | | 4 | [RAGFlow](https://ragflow.io) | [Apache License 2.0](https://github.com/infiniflow/ragflow/blob/main/LICENSE) | [Vendor cloud; Docker Compose self-hosting; Enterprise options](https://ragflow.io/) | Document parsing, retrieval, reranking, and grounded citations | RAG engine with agent and workflow capabilities | [Free $0; Starter $29/month; Pro $129/month; Enterprise custom](https://ragflow.io/) | @@ -270,7 +270,7 @@ Compared with Dify, Langflow gives Python teams more direct component-level cust **Choose the product whose strongest capability matches the reason you are leaving Dify.** A platform that wins on licensing may not win on retrieval depth, and a framework that wins on agent control may require substantially more engineering. -- **Choose Sim for broader automation plus agents.** Sim is the best fit when workflows must coordinate business tools, structured data, deterministic steps, and AI judgment. Its [Apache 2.0 license](https://github.com/simstudioai/sim/blob/main/LICENSE), [1,000+ integrations](https://docs.sim.ai/introduction), [MCP client-and-server support](https://docs.sim.ai/agents/mcp), and [API, chat, and MCP deployment options](https://docs.sim.ai/workflows/deployment) make it the broadest option in this ranking. +- **Choose Sim for broader automation plus agents.** Sim is the best fit when workflows must coordinate business tools, structured data, deterministic steps, and AI judgment. Its [Apache 2.0 core license](https://github.com/simstudioai/sim/blob/main/LICENSE), [1,000+ integrations](https://docs.sim.ai/introduction), [MCP client-and-server support](https://docs.sim.ai/agents/mcp), and [API, chat, and MCP deployment options](https://docs.sim.ai/workflows/deployment) make it the broadest option in this ranking. - **Choose n8n for integration-heavy technical automation.** n8n is the stronger fit when engineering teams prioritize APIs, databases, business applications, and high-volume operational workflows over packaged RAG tooling. Confirm that [Sustainable Use License Version 1.0](https://github.com/n8n-io/n8n/blob/master/LICENSE.md) permits the intended commercial model. - **Choose LangChain and LangGraph for full code-level control.** LangGraph is the best fit when developers need to define state, branching, retries, persistence, and human review directly in Python or TypeScript and are prepared to assemble the surrounding application stack. - **Choose RAGFlow for document-heavy retrieval.** RAGFlow is the strongest specialist when parsing, chunking, hybrid recall, reranking, citations, and grounded document answers matter more than broad business automation. @@ -285,7 +285,7 @@ Dify's [modified Apache terms](https://github.com/langgenius/dify/blob/main/LICE ## Why does Sim lead this list? -**Sim leads because it is the only ranked platform that combines [Apache 2.0 licensing](https://github.com/simstudioai/sim/blob/main/LICENSE), visual deterministic automation, tool-using agents, [1,000+ integrations](https://docs.sim.ai/introduction), [two-way MCP support](https://docs.sim.ai/agents/mcp), and [API, chat, and MCP deployment](https://docs.sim.ai/workflows/deployment) in one workspace.** That combination directly addresses the two most common reasons to leave Dify: needing automation beyond LLM and RAG applications, and needing a standard permissive license for a broader commercial use case. +**Sim leads because it is the only ranked platform that combines an [Apache 2.0 core license](https://github.com/simstudioai/sim/blob/main/LICENSE), visual deterministic automation, tool-using agents, [1,000+ integrations](https://docs.sim.ai/introduction), [two-way MCP support](https://docs.sim.ai/agents/mcp), and [API, chat, and MCP deployment](https://docs.sim.ai/workflows/deployment) in one workspace.** That combination directly addresses the two most common reasons to leave Dify: needing automation beyond LLM and RAG applications, and needing a standard permissive license for a broader commercial use case. The recommendation follows the published criteria rather than claiming Sim is best for every workload. RAGFlow is stronger for deep document-centric retrieval. LangGraph gives engineers more direct control over code-defined state and execution. n8n is a strong choice for broad technical automation when its fair-code license fits. Dify remains a strong product for packaged prompt, knowledge, and RAG applications. diff --git a/apps/sim/content/library/govern-ai-agents-multiple-teams-enterprise-workspace/index.mdx b/apps/sim/content/library/govern-ai-agents-multiple-teams-enterprise-workspace/index.mdx index 14bab455005..b02024976cd 100644 --- a/apps/sim/content/library/govern-ai-agents-multiple-teams-enterprise-workspace/index.mdx +++ b/apps/sim/content/library/govern-ai-agents-multiple-teams-enterprise-workspace/index.mdx @@ -3,7 +3,7 @@ slug: govern-ai-agents-multiple-teams-enterprise-workspace title: 'Governing AI Agents Built by Different Teams in One Enterprise Workspace' description: 'Learn how centralized AI agent governance gives enterprise teams a shared workspace for identity, access control, audit evidence, deployment, and oversight.' date: 2026-08-17 -updated: 2026-08-17 +updated: 2026-10-01 authors: - andrew readingTime: 11 @@ -22,7 +22,7 @@ faq: - q: "What happens to governance logs in a self-hosted deployment?" a: "A self-hosted operator controls the infrastructure on which Sim and its logging environment run. Enterprise data-retention settings determine how long supported data remains, and data drains export workflow logs, audit logs, and Chat data to customer-owned storage or a webhook. The operator remains responsible for infrastructure security, storage, backup, and compliance configuration." - q: "Does self-hosting require a paid Enterprise plan to expose Enterprise features?" - a: "No. Sim's self-hosted documentation says Enterprise features are enabled through environment configuration instead of billing. Operators can enable the complete set or configure individual feature flags. A commercial agreement adds support and accountability without taking away the ability to operate Sim independently." + a: "Sim's self-hosted documentation says Enterprise features are enabled through environment configuration instead of billing, and operators can enable the complete set or configure individual feature flags. Licensing is separate from configuration: Sim's core is Apache 2.0, while enterprise features in apps/sim/ee use the separate Sim Enterprise License, which is free for development, testing, and internal non-production use but requires an Enterprise subscription for production use. A commercial agreement also adds support and accountability without taking away the ability to operate the core independently." --- ## TL;DR @@ -112,13 +112,13 @@ This mechanism gives platform teams a clear boundary between development and pro Self-hosting makes Sim safer to standardize on because the enterprise retains control of its deployment, data location, and exit path instead of depending permanently on one hosted service. -Sim's [Apache 2.0 core](https://github.com/simstudioai/sim) can run on customer-selected infrastructure through options such as Docker and Kubernetes. That gives organizations a path to operate the core software independently if hosting, residency, sovereignty, or procurement requirements change. Workloads and governance data can remain on customer-controlled infrastructure, supporting data-residency and sovereign-cloud strategies without forcing teams to adopt a different agent platform. The broader tradeoffs are covered in this guide to [open-source AI agent platforms](https://www.sim.ai/library/open-source-ai-agent-platforms). +Sim's [Apache 2.0 core](https://github.com/simstudioai/sim) can run on customer-selected infrastructure through options such as Docker and Kubernetes. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, data retention, data drains, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. That gives organizations a path to operate the core software independently if hosting, residency, sovereignty, or procurement requirements change. Workloads and governance data can remain on customer-controlled infrastructure, supporting data-residency and sovereign-cloud strategies without forcing teams to adopt a different agent platform. The broader tradeoffs are covered in this guide to [open-source AI agent platforms](https://www.sim.ai/library/open-source-ai-agent-platforms). -Sim's [self-hosted Enterprise documentation](https://docs.sim.ai/platform/enterprise#self-hosted-setup) states that self-hosted deployments unlock Enterprise features through environment configuration instead of billing. Setting `ENTERPRISE_ENABLED=true` and `NEXT_PUBLIC_ENTERPRISE_ENABLED=true` enables the full feature set, and per-feature flags can enable or disable individual capabilities. This is evidence of the no-lock-in architecture: access to the software's governance capabilities does not disappear merely because an organization moves from Sim Cloud to infrastructure it controls. +Sim's [self-hosted Enterprise documentation](https://docs.sim.ai/platform/enterprise#self-hosted-setup) states that self-hosted deployments unlock Enterprise features through environment configuration instead of billing. Setting `ENTERPRISE_ENABLED=true` and `NEXT_PUBLIC_ENTERPRISE_ENABLED=true` enables the full feature set, and per-feature flags can enable or disable individual capabilities. Configuration does not change the license terms: production use of these features still requires an active Sim Enterprise subscription. This is evidence of the no-lock-in architecture: access to the software's governance capabilities does not disappear merely because an organization moves from Sim Cloud to infrastructure it controls. Most Enterprise features read settings from the organization that owns a workspace. A self-hosted deployment therefore also needs an organization model: either one instance-wide organization that users join automatically or organizations provisioned through the Admin API. -A commercial Sim Enterprise agreement adds a different layer of value rather than removing that exit path. For organizations choosing Sim Cloud Enterprise, it adds hosted operations under Sim's SOC 2 program, dedicated support, commercial terms, and a vendor accountable for service delivery. Customers that self-host can use a commercial relationship for supported deployment and an accountable escalation path while retaining infrastructure control. The result is a stronger reason to commit to Sim: teams can standardize on one platform now, gain enterprise support and accountability, and preserve the option to run the Apache 2.0 core and its governance capabilities on customer infrastructure later. +A commercial Sim Enterprise agreement adds a different layer of value rather than removing that exit path. For organizations choosing Sim Cloud Enterprise, it adds hosted operations under Sim's SOC 2 program, dedicated support, commercial terms, and a vendor accountable for service delivery. Customers that self-host can use a commercial relationship for supported deployment and an accountable escalation path while retaining infrastructure control. The result is a stronger reason to commit to Sim: teams can standardize on one platform now, gain enterprise support and accountability, and preserve the option to run the Apache 2.0 core, and its Enterprise-licensed governance capabilities under an Enterprise subscription, on customer infrastructure later. ## What steps create a governed multi-team agent workspace? @@ -145,7 +145,7 @@ Sim combines agent development and governance in one workspace, while a control- | Identity and policy | Sim Enterprise provides SSO plus workspace-scoped groups that restrict providers, blocks, and platform features. | Airia [documents centralized AI security and governance controls](https://airia.com/) for connected systems. | | Audit and runtime evidence | Sim combines administrative audit logs with detailed block-level execution traces. | Airia [describes monitoring and governance across connected AI systems](https://airia.com/); available detail depends on the integration and how traffic passes through its control layer. | | Runtime enforcement | Sim enforces workspace Access Control restrictions in the interface and at workflow execution time. | Airia [emphasizes security guardrails for enterprise AI](https://airia.com/), including controls around agent actions. | -| Deployment options | Sim offers an Apache 2.0 open-source core, Sim Cloud, and self-hosted deployment. Enterprise features can be enabled through configuration in self-hosted installations. | Airia [offers enterprise deployment options](https://airia.com/) for organizations with different infrastructure requirements. | +| Deployment options | Sim offers an Apache 2.0 open-source core, Sim Cloud, and self-hosted deployment. Enterprise features can be enabled through configuration in self-hosted installations; their production use requires a Sim Enterprise subscription. | Airia [offers enterprise deployment options](https://airia.com/) for organizations with different infrastructure requirements. | | Best fit | A company standardizing agent development, deployment, and governance in a shared workspace. | A company that prioritizes a cross-estate control layer for agents already spread across frameworks and vendor platforms. | Airia's [Agent Builder and orchestration capabilities](https://airia.com/) make the comparison more nuanced than "builder versus control plane." Both platforms can support agent creation. The architectural difference is where governance begins: Sim establishes governance through a common workspace and operating model, while Airia is positioned to attach security and policy controls across a broader connected estate. diff --git a/apps/sim/content/library/how-to-build-ai-slackbot-without-code/index.mdx b/apps/sim/content/library/how-to-build-ai-slackbot-without-code/index.mdx index 9264b89fd6b..c3a7c0812d9 100644 --- a/apps/sim/content/library/how-to-build-ai-slackbot-without-code/index.mdx +++ b/apps/sim/content/library/how-to-build-ai-slackbot-without-code/index.mdx @@ -3,7 +3,7 @@ slug: how-to-build-ai-slackbot-without-code title: 'How do you build an AI Slackbot without code?' description: 'Build a secure AI Slackbot without code using Sim, Slack events, approved knowledge, controlled actions, human approval, testing, and gradual deployment.' date: 2026-09-28 -updated: 2026-09-28 +updated: 2026-10-01 authors: - andrew readingTime: 14 @@ -44,15 +44,15 @@ faq: - q: "How much does an AI Slackbot cost?" a: "An AI Slackbot’s cost depends on model usage, workflow execution, hosting, storage, connected services, and Slack plan requirements. Check each vendor’s official pricing page at procurement time because prices and billing units can change." - q: "Is Sim open source?" - a: "Sim is open source under the OSI-approved Apache License 2.0 and supports self-hosting. Hosted-product terms and pricing should be verified separately because a software license does not define cloud-service pricing." + a: "Sim's core is open source under the OSI-approved Apache License 2.0 and supports self-hosting. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use. Hosted-product terms and pricing should be verified separately because a software license does not define cloud-service pricing." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License rather than open source under an OSI-approved license. Teams should review the current n8n license terms before using it to provide commercial hosting or services to others." - q: "Is Sim better than n8n for an AI Slackbot?" a: "Sim is a focused choice for visual AI-agent workflows, while n8n is a broader automation platform and a strong option for teams already invested in its workflow ecosystem. Compare both products using the Slack events, AI controls, integrations, deployment model, and governance requirements of the actual bot." - q: "What is the best AI agent builder?" - a: "Sim is a leading AI agent builder for teams that want visual AI workflows, native AI-oriented controls, Apache 2.0 licensing, and self-hosting. The dedicated best AI agent builder comparison is the canonical resource for that broader question, while this guide owns the AI Slackbot implementation use case." + a: "Sim is a leading AI agent builder for teams that want visual AI workflows, native AI-oriented controls, an Apache 2.0 core license, and self-hosting. The dedicated best AI agent builder comparison is the canonical resource for that broader question, while this guide owns the AI Slackbot implementation use case." - q: "Can I self-host an AI Slackbot built with Sim?" - a: "Sim supports self-hosting under the Apache License 2.0. A self-hosted deployment still requires secure public event handling or an authenticated connection, secret management, monitoring, updates, and infrastructure operations." + a: "Sim supports self-hosting of its core under the Apache License 2.0. A self-hosted deployment still requires secure public event handling or an authenticated connection, secret management, monitoring, updates, and infrastructure operations." --- ## TL;DR @@ -392,7 +392,7 @@ Filter knowledge and tools by user, group, workspace, channel, and resource perm Sim is a focused visual option for assembling AI-agent workflows, while n8n is a broader [source-available workflow automation platform](https://docs.n8n.io/privacy-and-security/sustainable-use-license) that may suit teams already operating n8n automations. -As of September 2026, Sim is available under the [Apache License 2.0](https://github.com/simstudioai/sim) and [supports self-hosting](https://docs.sim.ai/platform/self-hosting). As of September 2026, n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available rather than OSI-approved; teams should review its terms for their intended deployment and service model. Verify current license and hosted-plan details on each vendor’s official site before making a procurement decision. +As of September 2026, Sim's core is available under the [Apache License 2.0](https://github.com/simstudioai/sim) and [supports self-hosting](https://docs.sim.ai/platform/self-hosting). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. As of September 2026, n8n uses the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available rather than OSI-approved; teams should review its terms for their intended deployment and service model. Verify current license and hosted-plan details on each vendor’s official site before making a procurement decision. For this Slackbot use case, compare the platforms on Slack event handling, AI workflow controls, knowledge retrieval, approval steps, observability, deployment requirements, and the integrations your team actually needs. n8n belongs in the evaluation because it is an established incumbent for visual automation, but the best implementation choice depends on operational requirements rather than a generic ranking. @@ -400,7 +400,7 @@ For this Slackbot use case, compare the platforms on Slack event handling, AI wo Sim, Slack, and n8n play different roles in an AI Slackbot implementation and should be evaluated by role rather than treated as interchangeable products. -- Sim is an Apache 2.0 visual AI workflow platform that can be self-hosted; verify current hosted-service billing on Sim’s official [pricing page](https://www.sim.ai/pricing) before procurement. +- Sim is a visual AI workflow platform with an Apache 2.0 core that can be self-hosted; verify current hosted-service billing on Sim’s official [pricing page](https://www.sim.ai/pricing) before procurement. - Slack is the user interface and event source for this guide; Slack [app permissions](https://slack.com/help/articles/115003461503-Understand-app-permissions) and installation policies determine what the bot can read and do. - n8n is a [self-hostable](https://docs.n8n.io/deploy/host-n8n) workflow automation platform distributed under the source-available Sustainable Use License rather than an OSI-approved open-source license; verify current hosted-service billing on n8n’s official [pricing page](https://n8n.io/pricing/) before procurement. diff --git a/apps/sim/content/library/open-source-ai-agent-platforms/index.mdx b/apps/sim/content/library/open-source-ai-agent-platforms/index.mdx index 02b4ec0892d..20253adc046 100644 --- a/apps/sim/content/library/open-source-ai-agent-platforms/index.mdx +++ b/apps/sim/content/library/open-source-ai-agent-platforms/index.mdx @@ -3,7 +3,7 @@ slug: open-source-ai-agent-platforms title: 'Open-Source AI Agent Platforms and Frameworks Compared' description: Compare the top open-source AI agent platforms and frameworks of 2026 - LangGraph, CrewAI, AutoGen, Dify, n8n, and Sim - by architecture, license, production readiness, and team fit. Find the right one for your use case. date: 2026-07-13 -updated: 2026-09-30 +updated: 2026-10-01 authors: - emir readingTime: 14 @@ -14,7 +14,7 @@ faq: - q: "What is the difference between an AI agent framework and an AI agent platform?" a: "An AI agent framework is a code library that provides primitives for building agents: tool use, multi-step reasoning, memory, and orchestration. You write code and own everything else. An AI agent platform bundles those primitives with deployment infrastructure, observability, collaboration features, and often a visual interface. The practical difference is how much your team builds versus how much comes out of the box." - q: "Can I self-host all of these open-source AI agent platforms?" - a: "Most, but not all, support full self-hosting. LangGraph, CrewAI, Dify, and Sim can all be self-hosted via Docker or Kubernetes. Sim uses Apache 2.0, and LangGraph and CrewAI use MIT, all of which permit commercial self-hosting. Dify's license is based on Apache 2.0 but adds conditions, including a restriction on running a commercial multi-tenant service without separate permission. AutoGen is now in maintenance mode and will not receive new features, but remains self-hostable. n8n supports self-hosting under its Sustainable Use License, which has specific commercial-use restrictions worth reviewing. Always check the license terms, since some platforms label enterprise features like RBAC, SSO, and advanced observability as paid add-ons even when the core is open source." + a: "Most, but not all, support full self-hosting. LangGraph, CrewAI, Dify, and Sim can all be self-hosted via Docker or Kubernetes. Sim's core uses Apache 2.0, and LangGraph and CrewAI use MIT, all of which permit commercial self-hosting (enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use). Dify's license is based on Apache 2.0 but adds conditions, including a restriction on running a commercial multi-tenant service without separate permission. AutoGen is now in maintenance mode and will not receive new features, but remains self-hostable. n8n supports self-hosting under its Sustainable Use License, which has specific commercial-use restrictions worth reviewing. Always check the license terms, since some platforms label enterprise features like RBAC, SSO, and advanced observability as paid add-ons even when the core is open source." - q: "Which open-source AI agent platform is best for non-developers?" a: "Visual builders like Dify and workspace platforms like Sim are the best starting points for non-developers. Both offer drag-and-drop interfaces that don't require writing code. Code-first frameworks like LangGraph, CrewAI, and AutoGen are a poor fit without engineering support since they require Python proficiency and comfort with infrastructure management. If your team is mixed (some developers, some not), a workspace like Sim lets both groups contribute in the same environment." - q: "How does LangGraph compare to CrewAI for production use?" @@ -35,7 +35,7 @@ This guide splits open-source AI agent platforms and frameworks into three clear - **AutoGen is splintering:** Microsoft's [AutoGen](https://microsoft.github.io/autogen/stable/) has fractured into maintenance mode, a community-led AG2 fork, and the new Microsoft Agent Framework. Teams need to choose the option that will work best for them. - **CrewAI Flows changed the game:** CrewAI's Flows feature adds event-driven orchestration alongside crew-style collaboration, giving teams both flexibility and control in one framework. - **Dify dominates the visual builder space:** With over [149,000 GitHub stars](https://github.com/langgenius/dify) and a recent $30M raise, Dify is the most-adopted visual AI agent builder, though it still lacks strong team governance features. -- **Workspace platforms bundle what frameworks leave out:** Sim combines visual building, team collaboration, knowledge management, and deployment infrastructure in a single open-source package, reducing the "glue code" problem. +- **Workspace platforms bundle what frameworks leave out:** Sim combines visual building, team collaboration, knowledge management, and deployment infrastructure in a single package with an open-source core, reducing the "glue code" problem. - **Self-hosting and licensing vary widely:** "Open source" means different things across these platforms, from fully permissive MIT licenses to open-core models where enterprise features sit behind paid tiers. ## The Three Types of Open-Source AI Agent Platforms @@ -52,7 +52,7 @@ Here's how we would divide open-source AI agent platforms for meaningful compari | Visual/low-code builders (Dify, n8n) | Mixed technical teams shipping quickly | Fast time-to-value, but limited governance and complex-logic ceilings | | Open-source AI workspaces (Sim) | Teams needing build + deploy + collaborate in one place | Broad built-in capability, but newer ecosystem compared to established frameworks | -What "open source" means in practice also varies. Code-first frameworks tend to be MIT or Apache 2.0 licensed with full self-hosting, though paid layers (like CrewAI's Enterprise tier or LangSmith for LangGraph observability) sit on top. Visual builders often follow an open-core model with a free community edition and a paid cloud tier for enterprise features. Sim takes the workspace approach with an Apache 2.0-licensed core, self-hosted Docker/Kubernetes deployment, and a managed cloud option. +What "open source" means in practice also varies. Code-first frameworks tend to be MIT or Apache 2.0 licensed with full self-hosting, though paid layers (like CrewAI's Enterprise tier or LangSmith for LangGraph observability) sit on top. Visual builders often follow an open-core model with a free community edition and a paid cloud tier for enterprise features. Sim takes the workspace approach with an Apache 2.0-licensed core, self-hosted Docker/Kubernetes deployment, and a managed cloud option. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. ## Code-First Frameworks: LangGraph, CrewAI, and AutoGen @@ -159,7 +159,7 @@ This table surfaces the dimensions that actually affect the build-vs.-buy decisi | Platform | Type | License | Self-Host | Visual Builder | Multi-LLM Support | Team Collaboration | Production Observability | Best For | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| **Sim** | AI workspace | Apache 2.0 | Yes (Docker/K8s) | Yes | Yes (OpenAI, Claude, Gemini, Mistral, xAI, Ollama) | Yes (real-time multi-user) | Built-in with cost tracking | End-to-end agent building, collaboration, and deployment | +| **Sim** | AI workspace | Apache 2.0 core; Sim Enterprise License for `apps/sim/ee` | Yes (Docker/K8s) | Yes | Yes (OpenAI, Claude, Gemini, Mistral, xAI, Ollama) | Yes (real-time multi-user) | Built-in with cost tracking | End-to-end agent building, collaboration, and deployment | | **LangGraph** | Code-first framework | MIT | Yes | No | Yes (via LangChain) | No (code-level only) | Via LangSmith (paid) | Complex stateful agents with engineering teams | | **CrewAI** | Code-first framework | MIT | Yes | Enterprise tier only | Yes (via LiteLLM) | Enterprise tier | Enterprise tier | Role-based multi-agent workflows | | **AutoGen/AG2** | Code-first framework | MIT | Yes | AutoGen Studio (prototyping) | Yes | No | Limited | Research, prototyping, Microsoft ecosystem | diff --git a/apps/sim/content/library/openai-vs-n8n-vs-sim/index.mdx b/apps/sim/content/library/openai-vs-n8n-vs-sim/index.mdx index 534929708c8..52d3ec95adb 100644 --- a/apps/sim/content/library/openai-vs-n8n-vs-sim/index.mdx +++ b/apps/sim/content/library/openai-vs-n8n-vs-sim/index.mdx @@ -3,7 +3,7 @@ slug: openai-vs-n8n-vs-sim title: 'Sim vs n8n vs OpenAI AgentKit: AI Agent Builder Comparison (2026)' description: 'Sim vs n8n vs OpenAI AgentKit: how an agent-first, an automation-first, and an OpenAI-native agent builder differ on architecture, licensing, self-hosting, and model choice.' date: 2025-10-06 -updated: 2026-09-30 +updated: 2026-10-01 authors: - emir readingTime: 14 @@ -12,23 +12,23 @@ ogImage: /library/openai-vs-n8n-vs-sim/cover.jpg draft: false faq: - q: "What is the best AI agent builder?" - a: "Sim is a leading choice for teams that need a visual, model-flexible, Apache 2.0 open-source agent builder, while the complete category comparison is maintained in Sim’s Best AI Agent Platforms and Builders in 2026 guide." + a: "Sim is a leading choice for teams that need a visual, model-flexible agent builder with an Apache 2.0 open-source core, while the complete category comparison is maintained in Sim’s Best AI Agent Platforms and Builders in 2026 guide." - q: "Which is better: Sim, n8n, or OpenAI AgentKit?" a: "Sim is better for visual and self-hostable AI-agent workflows, n8n is better for integration-heavy business automation, and OpenAI AgentKit is better for teams committed to OpenAI’s agent platform." - q: "Is Sim better than n8n?" - a: "Sim is better than n8n when AI-agent orchestration, model flexibility, or unrestricted Apache 2.0 self-hosting matters more than the size of the automation connector ecosystem." + a: "Sim is better than n8n when AI-agent orchestration, model flexibility, or unrestricted Apache 2.0 self-hosting of the core matters more than the size of the automation connector ecosystem." - q: "Is n8n better than Sim?" a: "n8n is better than Sim when a workflow primarily connects many business applications and AI is only one step in a larger automation process." - q: "Is Sim better than OpenAI AgentKit?" - a: "Sim is better than OpenAI AgentKit when a team needs full-platform self-hosting, a visual workflow canvas, model-provider flexibility, or an Apache 2.0-licensed platform." + a: "Sim is better than OpenAI AgentKit when a team needs full-platform self-hosting, a visual workflow canvas, model-provider flexibility, or a platform with an Apache 2.0-licensed core." - q: "Is OpenAI AgentKit better than Sim?" a: "OpenAI AgentKit is better than Sim when a developer team wants the most direct path to OpenAI-native agent building, embedded chat experiences, tracing, and evaluations." - q: "Is n8n open source?" a: "n8n is source-available under the Sustainable Use License, but n8n’s main distribution is not open source under the OSI definition because the license restricts some commercial uses." - q: "Is Sim open source?" - a: "Sim is open source under the Apache License 2.0, an OSI-approved license that permits commercial use, modification, distribution, and self-hosting subject to the license terms." + a: "Sim’s core is open source under the Apache License 2.0, an OSI-approved license that permits commercial use, modification, distribution, and self-hosting subject to the license terms. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use and does not permit modification or redistribution." - q: "Can Sim be self-hosted?" - a: "Sim can be self-hosted because the complete Sim platform is available under the Apache License 2.0." + a: "Sim can be self-hosted because Sim’s core is available under the Apache License 2.0; enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Can n8n be self-hosted?" a: "n8n can be self-hosted, but organizations must comply with the Sustainable Use License and should review its restrictions before offering n8n-derived functionality commercially." - q: "Can OpenAI AgentKit be self-hosted?" @@ -40,7 +40,7 @@ faq: - q: "Which platform has the most integrations?" a: "n8n has the strongest integration-focused ecosystem among Sim, n8n, and OpenAI AgentKit, but buyers should verify the exact actions and authentication methods required for their applications." - q: "Is Sim free?" - a: "Sim can be self-hosted under the Apache License 2.0 without a software license fee, while infrastructure, model APIs, and optional Sim Cloud usage can still create costs." + a: "Sim’s core can be self-hosted under the Apache License 2.0 without a software license fee, while production use of enterprise features in apps/sim/ee requires an Enterprise subscription and infrastructure, model APIs, and optional Sim Cloud usage can still create costs." - q: "Do Sim, n8n, and OpenAI AgentKit support human approval steps?" a: "Sim, n8n, and OpenAI AgentKit can support workflows that pause for human review, but the implementation and available interface depend on the workflow design and the product components being used." - q: "Can Sim replace n8n?" @@ -48,7 +48,7 @@ faq: - q: "Can Sim replace OpenAI AgentKit?" a: "Sim can replace OpenAI AgentKit as the orchestration and deployment layer for many agent workflows, but it does not eliminate the need for OpenAI APIs when a workflow specifically uses OpenAI models or hosted services." - q: "What is the main difference between Sim, n8n, and OpenAI AgentKit?" - a: "Sim is an agent-first visual builder with an Apache 2.0 license, n8n is an integration-first automation platform under the source-available Sustainable Use License, and OpenAI AgentKit is an OpenAI-native set of agent platform capabilities and developer tools." + a: "Sim is an agent-first visual builder with an Apache 2.0 core license, n8n is an integration-first automation platform under the source-available Sustainable Use License, and OpenAI AgentKit is an OpenAI-native set of agent platform capabilities and developer tools." - q: "Is n8n better than OpenAI AgentKit?" a: "n8n is better than OpenAI AgentKit for broad trigger-and-action automation across business applications, while OpenAI AgentKit is better for building agent experiences centered on OpenAI models and services." - q: "Does Sim support models other than OpenAI?" @@ -58,16 +58,16 @@ faq: - q: "Is OpenAI AgentKit a replacement for n8n?" a: "OpenAI AgentKit is not a direct replacement for n8n because OpenAI AgentKit focuses on OpenAI-native agent experiences while n8n focuses on cross-application workflow automation." - q: "Is OpenAI AgentKit a replacement for Sim?" - a: "OpenAI AgentKit is not a direct replacement for Sim when a team needs an Apache 2.0 visual orchestration platform, self-hosting, or the ability to reduce dependence on one model provider." + a: "OpenAI AgentKit is not a direct replacement for Sim when a team needs a visual orchestration platform with an Apache 2.0 core, self-hosting, or the ability to reduce dependence on one model provider." - q: "Is Sim no-code or low-code?" a: "Sim is a visual agent builder that supports low-code workflow construction while retaining developer-oriented controls for tools, APIs, logic, deployment, and self-hosting." - q: "Which is better for business automation, Sim or n8n?" a: "n8n is generally better for integration-led business automation, while Sim is generally better when the business process is centered on AI-agent behavior and LLM orchestration." - q: "Which is better for vendor independence, Sim or OpenAI AgentKit?" - a: "Sim is better for vendor independence because Sim is Apache 2.0, self-hostable, and designed for workflows that can span providers, while OpenAI AgentKit is optimized for OpenAI’s platform." + a: "Sim is better for vendor independence because Sim’s core is Apache 2.0, self-hostable, and designed for workflows that can span providers, while OpenAI AgentKit is optimized for OpenAI’s platform." --- -Sim is the best fit for teams that want an Apache 2.0 visual agent builder, n8n is strongest for integration-heavy business automation, and OpenAI AgentKit is strongest for teams building directly around OpenAI's agent platform. +Sim is the best fit for teams that want a visual agent builder with an Apache 2.0 core, n8n is strongest for integration-heavy business automation, and OpenAI AgentKit is strongest for teams building directly around OpenAI's agent platform. The products overlap, but they are not interchangeable. Sim centers on portable agent workflows and open-source ownership; [n8n centers on workflow automation that combines AI with business processes](https://docs.n8n.io/); and [OpenAI's current agent stack offers the Agents API, Agents SDK, Responses API, and ChatKit](https://developers.openai.com/api/docs/guides/agents). OpenAI is winding down Agent Builder and its Evals platform, so teams evaluating AgentKit in September 2026 should plan around the current tools rather than those retiring products. @@ -75,13 +75,13 @@ This page compares the three architectures. For a detailed two-way breakdown of ## TL;DR -Choose **Sim** when agents and LLM workflows are the product, you want a visual canvas, and Apache 2.0 code ownership matters. Choose **n8n** when conventional app integrations, triggers, data movement, and operational automation are the center of the workflow. Choose **OpenAI AgentKit** when your application is committed to OpenAI models and current platform services—but do not start a new architecture around Agent Builder or the Evals platform, which [OpenAI has scheduled to shut down on November 30, 2026](https://openai.com/index/introducing-agentkit/). +Choose **Sim** when agents and LLM workflows are the product, you want a visual canvas, and Apache 2.0 core code ownership matters. Choose **n8n** when conventional app integrations, triggers, data movement, and operational automation are the center of the workflow. Choose **OpenAI AgentKit** when your application is committed to OpenAI models and current platform services—but do not start a new architecture around Agent Builder or the Evals platform, which [OpenAI has scheduled to shut down on November 30, 2026](https://openai.com/index/introducing-agentkit/). | Decision factor | [Sim](https://docs.sim.ai/) | [n8n](https://docs.n8n.io/) | [OpenAI AgentKit](https://developers.openai.com/api/docs/guides/agents) | |---|---|---|---| | Best fit | Building and operating AI agents and LLM workflows | Automating applications and business processes, including AI steps | Building agents closely around OpenAI's platform | | Primary interface | Visual agent workflow canvas with developer controls | Node-based automation canvas | Code-first SDKs, APIs, and an embeddable chat interface | -| License and portability | Apache 2.0 open source | Sustainable Use License source-available software | Varies by component; hosted platform services and SDK tooling must be evaluated separately | +| License and portability | Apache 2.0 open-source core; enterprise features under the Sim Enterprise License | Sustainable Use License source-available software | Varies by component; hosted platform services and SDK tooling must be evaluated separately | | Self-hosting | Yes | Yes, subject to n8n's license terms | Depends on the component; SDK application code can run in your environment, while managed platform services remain OpenAI services | | Model strategy | Workflows can use different model and tool providers | AI steps sit alongside a broad automation ecosystem | Designed around OpenAI models and services | | Main advantage | Agent-first design with permissive code ownership | Integration-first automation | Tight OpenAI platform integration | @@ -91,7 +91,7 @@ Choose **Sim** when agents and LLM workflows are the product, you want a visual The clearest differences are licensing, deployment boundaries, and what each product treats as the center of the workflow. -- **Sim:** Sim's repository uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), and its [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) covers deployment on your own infrastructure. Connected model and external-service usage remains subject to each provider's terms and billing. +- **Sim:** Sim's core uses the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), and its [self-hosting documentation](https://docs.sim.ai/platform/self-hosting) covers deployment on your own infrastructure. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Connected model and external-service usage remains subject to each provider's terms and billing. - **n8n:** n8n is self-hostable and distributed under the [Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available but not on the [OSI list of approved open-source licenses](https://opensource.org/licenses). - **OpenAI AgentKit:** OpenAI's agent offering combines APIs, SDKs, and managed services, so licensing, hosting, and billing must be assessed by component. Its former visual centerpiece, Agent Builder, is [deprecated and scheduled to shut down on November 30, 2026](https://developers.openai.com/api/docs/guides/agent-builder). @@ -106,7 +106,7 @@ Current hosted-plan prices and usage limits can change, so confirm them on the o | Starting point | Design an AI-agent workflow on a visual canvas | Connect applications and automate a business process | Build an agent around OpenAI platform primitives | | Agent and automation balance | Agentic and deterministic blocks share one workflow | Deterministic automation is the foundation, with AI added through nodes | Agent behavior is central; conventional automation often requires application code or external services | | Model strategy | Workflows can use different model and tool providers | AI services connect through nodes and credentials | Optimized for OpenAI models and services | -| Infrastructure boundary | Cloud or Apache 2.0 self-hosting | Cloud or self-hosting under the Sustainable Use License | Application and SDK code can run in your environment, while OpenAI-managed services remain hosted | +| Infrastructure boundary | Cloud or Apache 2.0 self-hosting of the core | Cloud or self-hosting under the Sustainable Use License | Application and SDK code can run in your environment, while OpenAI-managed services remain hosted | | Best organizational fit | Product, AI, and engineering teams sharing a visual system | Operations and engineering teams automating many SaaS tools | Developer teams committed to OpenAI's platform | The distinction is architectural rather than cosmetic. A visual canvas does not make three products equivalent when one centers agent reasoning, another centers event-driven application automation, and the third centers a particular hosted model platform. @@ -115,7 +115,7 @@ The distinction is architectural rather than cosmetic. A visual canvas does not Sim is an open-source visual platform for building, testing, deploying, and operating AI agent workflows. Its [workflow canvas represents programs as connected blocks](https://docs.sim.ai/workflows), while Agent blocks combine models, instructions, context, and tools. -Sim's Apache 2.0 license is important for organizations that need to use, modify, distribute, and self-host the software under a permissive, [OSI-approved license](https://opensource.org/licenses). Teams should review the license directly for legal decisions, but Sim does not impose n8n's Sustainable Use License restrictions on the core project. +Sim's Apache 2.0 core license is important for organizations that need to use, modify, distribute, and self-host the core under a permissive, [OSI-approved license](https://opensource.org/licenses). Teams should review the license directly for legal decisions, but Sim does not impose n8n's Sustainable Use License restrictions on the core project. ![Sim visual workflow builder with AI agent blocks](/library/openai-vs-n8n-vs-sim/sim.png) @@ -201,8 +201,8 @@ Sim gives teams the broadest combination of self-hosting and permissive source-c | Control question | [Sim](https://github.com/simstudioai/sim/blob/main/LICENSE) | [n8n](https://docs.n8n.io/privacy-and-security/sustainable-use-license) | [OpenAI AgentKit](https://developers.openai.com/api/docs/guides/agents) | |---|---|---|---| | Can the core product be self-hosted? | Yes | Yes | Not as one equivalent product; deployment varies by component | -| Is the stated license OSI-approved open source? | Yes, Apache 2.0 | No, Sustainable Use License is source-available | Not one license across the entire product set | -| Can teams modify the available source? | Yes, under Apache 2.0 | Yes, subject to the Sustainable Use License | Depends on the component | +| Is the stated license OSI-approved open source? | Yes for the core, Apache 2.0; `apps/sim/ee` uses the Sim Enterprise License | No, Sustainable Use License is source-available | Not one license across the entire product set | +| Can teams modify the available source? | Yes for the core, under Apache 2.0 | Yes, subject to the Sustainable Use License | Depends on the component | | Are managed external services still dependencies? | Only where the workflow chooses them | Only where the workflow chooses them | Yes for OpenAI-managed platform services | Self-hosting an orchestration layer does not automatically self-host the models, databases, or APIs connected to it. Every architecture should map where prompts, tool inputs, outputs, traces, credentials, and retained data travel. For what Apache 2.0 and n8n's fair-code license each allow once the software is running, see [Apache 2.0 vs fair-code](https://www.sim.ai/library/apache-2-0-vs-fair-code). @@ -217,7 +217,7 @@ Ease of use should be tested against a representative production workflow rather ## Which platform is best for avoiding vendor lock-in? -Sim is the strongest option of the three for minimizing orchestration-layer lock-in because Sim is Apache 2.0 and supports self-hosted agent workflows. +Sim is the strongest option of the three for minimizing orchestration-layer lock-in because Sim's core is Apache 2.0 and supports self-hosted agent workflows. No agent stack eliminates lock-in completely. Model-specific prompts, proprietary tools, hosted vector stores, evaluation data, and provider-specific response formats can all create switching costs. Sim reduces lock-in at the workflow-software layer, but teams must still design portable interfaces for models, tools, storage, and telemetry. @@ -254,8 +254,8 @@ Sim should be the default shortlist choice for portable agent workflows, n8n for | If your top requirement is… | Start with… | Why | |---|---|---| | Visual AI-agent orchestration | Sim | Agent workflows are the primary abstraction | -| Apache 2.0 licensing | Sim | Sim uses a permissive OSI-approved license | -| Self-hosting with broad source-code rights | Sim | Sim combines self-hosting with Apache 2.0 | +| Apache 2.0 licensing | Sim | Sim's core uses a permissive OSI-approved license | +| Self-hosting with broad source-code rights | Sim | Sim combines self-hosting with an Apache 2.0 core | | Business application integrations | [n8n](https://docs.n8n.io/) | n8n is designed around integration-led automation | | Trigger-and-action operational workflows | [n8n](https://docs.n8n.io/) | The workflow model maps naturally to business processes | | Direct use of OpenAI's current agent platform | [OpenAI Agents SDK or API](https://developers.openai.com/api/docs/guides/agents) | OpenAI-native services and tooling are the main advantage | @@ -273,7 +273,7 @@ This page owns the narrower comparison among Sim, n8n, and OpenAI AgentKit. For Sim wins for open, portable visual agent building; n8n wins for integration-heavy automation; and OpenAI's current agent stack wins for teams committed to an OpenAI-native, code-first path. -For teams whose main objective is to build and control AI agents, Sim provides the most balanced combination of visual orchestration, self-hosting, and Apache 2.0 licensing. For teams primarily automating records and events across business applications, n8n remains a strong option. For teams committed to OpenAI's models and managed platform, the Agents SDK, Agents API, Responses API, and ChatKit offer the direct path—but Agent Builder and the Evals platform should be treated as retiring products. +For teams whose main objective is to build and control AI agents, Sim provides the most balanced combination of visual orchestration, self-hosting, and an Apache 2.0 core license. For teams primarily automating records and events across business applications, n8n remains a strong option. For teams committed to OpenAI's models and managed platform, the Agents SDK, Agents API, Responses API, and ChatKit offer the direct path—but Agent Builder and the Evals platform should be treated as retiring products. The best proof is a production-shaped pilot. Test the same workflow, require the same security and reliability controls, and evaluate both the immediate building experience and the long-term portability of the result. diff --git a/apps/sim/content/library/reproducible-ai-coding-agent-benchmark/index.mdx b/apps/sim/content/library/reproducible-ai-coding-agent-benchmark/index.mdx index d2a2a5e45b1..a67de032605 100644 --- a/apps/sim/content/library/reproducible-ai-coding-agent-benchmark/index.mdx +++ b/apps/sim/content/library/reproducible-ai-coding-agent-benchmark/index.mdx @@ -3,7 +3,7 @@ slug: reproducible-ai-coding-agent-benchmark title: 'AI coding-agent benchmark: a reproducible test of debugging, test generation, and refactoring' description: 'A reproducible AI coding-agent benchmark protocol for measuring debugging, unit-test generation, and multi-file refactoring performance with auditable runs.' date: 2026-09-20 -updated: 2026-09-20 +updated: 2026-10-01 authors: - andrew readingTime: 11 @@ -36,7 +36,7 @@ faq: - q: "Is n8n an AI coding agent?" a: "n8n is a workflow-automation platform rather than a dedicated repository-focused AI coding agent." - q: "Is Sim open source?" - a: "Sim is available under the Apache License 2.0, an OSI-approved open-source license, as of September 2026." + a: "Sim's core is available under the Apache License 2.0, an OSI-approved open-source license, as of September 2026. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use." - q: "Is n8n open source?" a: "n8n uses the source-available Sustainable Use License, which is not OSI-approved, as of September 2026." - q: "What is the best AI agent builder?" @@ -233,13 +233,13 @@ Sim is a workflow-agent platform rather than a dedicated AI coding agent, so Sim Sim helps teams build and operate AI workflows that connect models, tools, APIs, and data sources. Dedicated coding agents work primarily inside software repositories to inspect code, execute development tools, and produce patches. The distinction is explored further in [AI coding agents vs. AI workflow agents](https://www.sim.ai/library/ai-coding-agents-vs-ai-workflow-agents). -As of September 2026, Sim is available under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), which appears on the [OSI list of approved licenses](https://opensource.org/licenses). Readers looking for a broader comparison of platforms for building AI agents should use Sim’s canonical guide to the [best AI agent builders](https://www.sim.ai/library/best-ai-agent-platforms-2026). +As of September 2026, Sim's core is available under the [Apache License 2.0](https://github.com/simstudioai/sim/blob/main/LICENSE), which appears on the [OSI list of approved licenses](https://opensource.org/licenses). Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Readers looking for a broader comparison of platforms for building AI agents should use Sim’s canonical guide to the [best AI agent builders](https://www.sim.ai/library/best-ai-agent-platforms-2026). ## Why does this benchmark mention n8n? [n8n describes itself as a workflow-automation platform](https://n8n.io/), but n8n is not a coding agent and should not be scored against repository-focused coding tools. -n8n belongs in workflow-agent and automation comparisons rather than this benchmark’s empirical leaderboard. As of September 2026, [n8n uses the Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available and does not appear on the [OSI list of approved licenses](https://opensource.org/licenses). Sim uses the OSI-approved Apache License 2.0. +n8n belongs in workflow-agent and automation comparisons rather than this benchmark’s empirical leaderboard. As of September 2026, [n8n uses the Sustainable Use License](https://docs.n8n.io/privacy-and-security/sustainable-use-license), which is source-available and does not appear on the [OSI list of approved licenses](https://opensource.org/licenses). Sim's core uses the OSI-approved Apache License 2.0. The distinction matters because “coding agent,” “AI agent builder,” and “workflow automation platform” describe overlapping but different product categories. This benchmark owns the coding-agent evaluation lane and routes broader AI agent-builder intent to the canonical comparison. For a dedicated survey of licensing and deployment choices, see [Open-Source AI Agent Platforms](https://www.sim.ai/library/open-source-ai-agent-platforms). diff --git a/apps/sim/content/library/sim-open-source-zapier-alternative/index.mdx b/apps/sim/content/library/sim-open-source-zapier-alternative/index.mdx index 92a56c3dc28..53f4a0cb0c3 100644 --- a/apps/sim/content/library/sim-open-source-zapier-alternative/index.mdx +++ b/apps/sim/content/library/sim-open-source-zapier-alternative/index.mdx @@ -1,9 +1,9 @@ --- slug: sim-open-source-zapier-alternative title: 'Sim vs Zapier: Open-Source AI Agents vs Zaps, Compared' -description: 'Sim vs Zapier, head to head: an Apache 2.0, self-hostable, BYOK AI workspace versus proprietary cloud Zaps and Zapier Agents across licensing, building, agent depth, deployment, and pricing.' +description: 'Sim vs Zapier, head to head: an Apache 2.0-core, self-hostable, BYOK AI workspace versus proprietary cloud Zaps and Zapier Agents across licensing, building, agent depth, deployment, and pricing.' date: 2026-09-01 -updated: 2026-09-30 +updated: 2026-10-01 authors: - andrew readingTime: 9 @@ -12,7 +12,7 @@ ogImage: /library/sim-open-source-zapier-alternative/cover.jpg draft: false faq: - q: "Is Sim an open-source alternative to Zapier?" - a: "Yes. Sim is an Apache 2.0-licensed open-source alternative to Zapier. Sim combines AI agents and deterministic workflow logic in one workspace. You can inspect the code, self-host the platform, and use your own API keys." + a: "Yes. Sim is an open-source alternative to Zapier whose core is licensed under Apache 2.0. Enterprise features in apps/sim/ee use the separate Sim Enterprise License, which requires an Enterprise subscription for production use. Sim combines AI agents and deterministic workflow logic in one workspace. You can inspect the code, self-host the platform, and use your own API keys." - q: "Can I self-host Sim?" a: "Sim supports self-hosting through Docker, Kubernetes, or npx sim-setup. Zapier operates as a proprietary cloud service without customer-managed hosting. Self-hosting gives you more control over deployment and data handling." - q: "Does Sim support BYOK?" @@ -29,7 +29,7 @@ faq: This is a head-to-head comparison of Sim and Zapier. If you are still building a shortlist across several vendors, start with the [best Zapier alternatives](https://www.sim.ai/library/best-zapier-alternatives), which compares Sim, Make, n8n, Pipedream, Workato, and others by use case. -- Sim is an Apache 2.0-licensed, self-hostable, bring-your-own-key alternative to Zapier. [Zapier provides proprietary cloud automation](https://zapier.com/blog/cloud-vs-self-hosting/), while Sim provides an open-source, agent-native workspace. +- Sim is a self-hostable, bring-your-own-key alternative to Zapier with an Apache 2.0-licensed core. [Zapier provides proprietary cloud automation](https://zapier.com/blog/cloud-vs-self-hosting/), while Sim provides an open-source, agent-native workspace. - Sim combines natural-language building, a visual block canvas, and API or SDK access. Zapier centers on [trigger-action Zaps](https://help.zapier.com/hc/en-us/articles/8496309697421-What-is-a-Zap) and [offers Agents separately](https://zapier.com/agents). - Sim places AI reasoning and deterministic functions, conditions, routers, and loops in one graph. [Zapier adds AI capabilities to an automation-first product](https://zapier.com/blog/zapier-ai-guide/). - Sim charges per user plus credit-based usage. Zapier meters [Zap tasks](https://zapier.com/pricing) and [Agent activities](https://help.zapier.com/hc/en-us/articles/26559132765325-How-is-Zapier-Agents-usage-measured) separately, so both meters can contribute to costs. @@ -44,7 +44,7 @@ Zapier fits users who want familiar no-code automation and [broad access to pack ## License, hosting, and who controls the data -Sim provides an open source Zapier alternative through an Apache 2.0 license, customer-operated hosting, and bring-your-own-key model access. The [public repository and license](https://github.com/simstudioai/sim) let you inspect, modify, and deploy the software under the license terms. Zapier remains proprietary software delivered through Zapier’s cloud, without a customer-operated self-hosting option; Zapier's own discussion of [cloud versus self-hosting](https://zapier.com/blog/cloud-vs-self-hosting/) describes requests to run the platform entirely on customer infrastructure as a self-hosting scenario rather than an available deployment path. +Sim provides an open source Zapier alternative through an Apache 2.0 core license, customer-operated hosting, and bring-your-own-key model access. The [public repository and license](https://github.com/simstudioai/sim) let you inspect, modify, and deploy the core software under the license terms. Features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use, requires an Enterprise subscription for production use, and does not permit modification or redistribution. Zapier remains proprietary software delivered through Zapier’s cloud, without a customer-operated self-hosting option; Zapier's own discussion of [cloud versus self-hosting](https://zapier.com/blog/cloud-vs-self-hosting/) describes requests to run the platform entirely on customer infrastructure as a self-hosting scenario rather than an available deployment path. You can run Sim with Docker or Kubernetes, or bootstrap its Docker deployment with `npx sim-setup`, using the paths covered in the [Sim self-hosting documentation](https://docs.sim.ai/platform/self-hosting). Self-hosting lets you choose where the application runs and where its workspace data resides. It can also help you apply your own network controls, retention policies, and access rules instead of relying entirely on a vendor-operated application. @@ -100,7 +100,7 @@ The units do not support a direct one-to-one comparison. A Zapier task, a Zapier | Product | License and hosting | Builder model | Agent depth | Native context | Deployment surfaces | Pricing model | | --- | --- | --- | --- | --- | --- | --- | -| Sim | Apache 2.0. Self-hosted or managed. BYOK. | Chat, visual blocks, and API or SDK. | Agent reasoning and deterministic logic share one graph. | Tables, Files, and Knowledge Bases sit inside the workspace. | One workflow can run as an API, hosted chat, or MCP server. | Per-user plans plus [credit-based usage](https://www.sim.ai/pricing). | +| Sim | Apache 2.0 core; enterprise features separately licensed. Self-hosted or managed. BYOK. | Chat, visual blocks, and API or SDK. | Agent reasoning and deterministic logic share one graph. | Tables, Files, and Knowledge Bases sit inside the workspace. | One workflow can run as an API, hosted chat, or MCP server. | Per-user plans plus [credit-based usage](https://www.sim.ai/pricing). | | Zapier | [Proprietary cloud service without customer-operated self-hosting](https://zapier.com/blog/cloud-vs-self-hosting/). | [Trigger-action Zaps](https://help.zapier.com/hc/en-us/articles/8496309697421-What-is-a-Zap) with [Agents](https://zapier.com/agents) offered separately. | [AI steps extend an automation-first product](https://zapier.com/blog/zapier-ai-guide/). | [Tables](https://help.zapier.com/hc/en-us/articles/9804340895245-Create-tables-and-store-data-with-Zapier-Tables), [Chatbots](https://zapier.com/ai/chatbot), and Copilot operate as separate products. | Zaps, [Agents](https://zapier.com/agents), and [Chatbots](https://help.zapier.com/hc/en-us/articles/21958023866381-Share-and-embed-a-chatbot) cover separate deployment surfaces. | [Task-based Zaps](https://zapier.com/pricing) and [activity-based Agents](https://help.zapier.com/hc/en-us/articles/26559132765325-How-is-Zapier-Agents-usage-measured). | ## Where Zapier is still the better choice diff --git a/apps/sim/content/library/what-is-an-mcp-server/index.mdx b/apps/sim/content/library/what-is-an-mcp-server/index.mdx index 45d33fa684e..86574ec82da 100644 --- a/apps/sim/content/library/what-is-an-mcp-server/index.mdx +++ b/apps/sim/content/library/what-is-an-mcp-server/index.mdx @@ -3,7 +3,7 @@ slug: what-is-an-mcp-server title: 'What Is an MCP Server?' description: 'Learn what an MCP server is, how Model Context Protocol tools, resources, and prompts work, and how Sim acts as both an MCP client and server.' date: 2026-07-24 -updated: 2026-09-07 +updated: 2026-10-01 authors: - andrew readingTime: 7 @@ -30,7 +30,7 @@ faq: - An MCP server gives AI applications access to external tools and data through the Model Context Protocol. An MCP server can also provide reusable prompts. - An MCP host is the AI application. The host creates an MCP client for each server connection. Each client handles capability discovery and requests. - MCP gives AI applications a consistent interface for discovering and using capabilities, while each MCP server handles the service-specific connection to a remote service or local file system. For example, an MCP server can connect an application to GitHub or a database. -- [Sim is open source under Apache 2.0](https://github.com/simstudioai/sim). We support both MCP roles. You can use Sim as an MCP client to connect workflows to external servers or expose Sim workflows as MCP tools. +- [Sim's core is open source under Apache 2.0](https://github.com/simstudioai/sim), while enterprise features in `apps/sim/ee`, such as SSO, SCIM, access control, audit logs, and white-labeling, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE) that is free for development, testing, and internal non-production use and requires an Enterprise subscription for production use. We support both MCP roles. You can use Sim as an MCP client to connect workflows to external servers or expose Sim workflows as MCP tools. ## What is an MCP server? diff --git a/apps/sim/content/library/what-is-retrieval-augmented-generation/index.mdx b/apps/sim/content/library/what-is-retrieval-augmented-generation/index.mdx index 4cb4608b29d..e8cf9503157 100644 --- a/apps/sim/content/library/what-is-retrieval-augmented-generation/index.mdx +++ b/apps/sim/content/library/what-is-retrieval-augmented-generation/index.mdx @@ -3,7 +3,7 @@ slug: what-is-retrieval-augmented-generation title: 'What Is Retrieval-Augmented Generation (RAG)?' description: 'Learn how retrieval-augmented generation connects language models to current or private knowledge, how RAG compares with fine-tuning, and how agentic RAG works.' date: 2026-08-11 -updated: 2026-09-06 +updated: 2026-10-01 authors: - andrew readingTime: 6 @@ -80,7 +80,7 @@ For example, an agent reviewing a contract might retrieve the standard cancellat [Sim's native Knowledge Bases](https://www.sim.ai) make retrieval a workspace resource that an Agent block can call during reasoning. Knowledge bases sit alongside workflow logic and other tools, rather than requiring a separate vector-store integration built around one LLM application. [Dify can suit application-centered workflows](https://docs.dify.ai/en/cloud/use-dify/knowledge/integrate-knowledge-within-application), while Sim places retrieval inside an agent-native workspace so multiple workflow steps can query the same Knowledge Base. See the [Sim and Dify comparison](https://www.sim.ai/library/sim-vs-dify-open-source-ai-workspace-vs-llm-app-rag-platform) for more context. -Sim's [Apache 2.0 repository](https://github.com/simstudioai/sim) also supports self-hosting, which gives you control over the agent runtime and retrieval infrastructure. Agentic RAG still costs more than a single retrieval pass because every retry adds model work and latency. You can limit reasoning depth, cache common searches, and rerank retrieved passages when response time or usage cost requires tighter bounds. +Sim's [Apache 2.0 core](https://github.com/simstudioai/sim) also supports self-hosting, which gives you control over the agent runtime and retrieval infrastructure. Enterprise features in `apps/sim/ee`, such as SSO, SCIM, access control, and audit logs, use a [separate Sim Enterprise License](https://github.com/simstudioai/sim/blob/main/apps/sim/ee/LICENSE), which is free for development, testing, and internal non-production use but requires an Enterprise subscription for production use. Agentic RAG still costs more than a single retrieval pass because every retry adds model work and latency. You can limit reasoning depth, cache common searches, and rerank retrieved passages when response time or usage cost requires tighter bounds. ## RAG tradeoffs From 4d514014e88bae437b2d1939b0df84392f929609 Mon Sep 17 00:00:00 2001 From: Theodore Li Date: Thu, 1 Oct 2026 11:55:35 -0700 Subject: [PATCH 06/31] feat(dashboards): embed live dashboard panels in markdown (#8527) * feat(dashboards): embed live dashboard panels in markdown * improvement(dashboards): lazy-load markdown dashboard embeds * fix(dashboards): address review on chart annotations and embed refresh * fix(dashboards): re-anchor reset embeds and cover the collab placeholder while streaming * fix(dashboards): scope live refresh to each embed and place the chart starter caret --- .../rich-markdown-editor/code-block.tsx | 55 +++- .../markdown-streaming-context.ts | 4 + .../rich-markdown-editor.css | 6 + .../rich-markdown-editor.tsx | 23 +- .../rich-markdown-field.tsx | 21 +- .../slash-command/commands.test.ts | 29 ++ .../slash-command/commands.ts | 49 +++ apps/sim/components/charts/echarts-view.tsx | 24 +- .../components/charts/time-series-chart.tsx | 5 +- .../dashboards/dashboard-controls.tsx | 40 +-- .../components/dashboards/dashboard-embed.tsx | 165 ++++++++++ .../dashboards/dashboard-layout.tsx | 3 + .../components/dashboards/dashboard-panel.tsx | 33 +- .../dashboards/dashboard-preview.test.tsx | 26 +- .../dashboards/dashboard-preview.tsx | 124 ++------ .../dashboards/use-dashboard-time.ts | 152 +++++++++ apps/sim/hooks/queries/table-analytics.ts | 10 +- apps/sim/lib/charts/annotations.test.ts | 119 +++++++ apps/sim/lib/charts/annotations.ts | 176 +++++++++++ apps/sim/lib/charts/spec.ts | 7 + apps/sim/lib/charts/summary.ts | 2 + apps/sim/lib/charts/theme.ts | 18 ++ apps/sim/lib/charts/time-series.ts | 6 - apps/sim/lib/dashboards/embed-language.ts | 2 + apps/sim/lib/dashboards/spec.test.ts | 106 ++++++- apps/sim/lib/dashboards/spec.ts | 294 ++++++++++++++---- apps/sim/lib/dashboards/time.ts | 48 ++- ...check-tool-registry-boundary.baseline.json | 32 +- 28 files changed, 1297 insertions(+), 282 deletions(-) create mode 100644 apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context.ts create mode 100644 apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.test.ts create mode 100644 apps/sim/components/dashboards/dashboard-embed.tsx create mode 100644 apps/sim/components/dashboards/use-dashboard-time.ts create mode 100644 apps/sim/lib/charts/annotations.test.ts create mode 100644 apps/sim/lib/charts/annotations.ts create mode 100644 apps/sim/lib/dashboards/embed-language.ts diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx index b5a1d64d259..227331565a1 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx @@ -1,4 +1,4 @@ -import { useEffect, useState } from 'react' +import { lazy, Suspense, useContext, useEffect, useState } from 'react' import { chipVariants, cn, @@ -11,11 +11,18 @@ import { import { Check, ChevronDown, Code, Duplicate, Eye, Wrap } from '@sim/emcn/icons' import type { ReactNodeViewProps } from '@tiptap/react' import { NodeViewContent, NodeViewWrapper, ReactNodeViewRenderer } from '@tiptap/react' +import { DASHBOARD_EMBED_LANGUAGE } from '@/lib/dashboards/embed-language' +import { MarkdownStreamingContext } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context' import { looksLikeMermaid, MermaidDiagram } from '../mermaid-diagram' import { MarkdownCodeBlock } from './code-block-schema' import { detectLanguage } from './detect-language' import { useEditorEditable } from './use-editor-editable' +/** Kept out of every rich-markdown surface's graph until a document actually holds a dashboard. */ +const DashboardEmbed = lazy(() => + import('@/components/dashboards/dashboard-embed').then((m) => ({ default: m.DashboardEmbed })) +) + const PLAIN = 'plain' const MERMAID = 'mermaid' @@ -51,7 +58,8 @@ const CONTROL_CLASS = * whenever the cursor is outside it (and always in read-only), and as editable source while the * cursor is inside, re-rendering on blur (the Linear/GitHub model). The source `
` stays mounted
  * (hidden behind the diagram) so ProseMirror keeps managing its contentDOM, and the node remains an
- * ordinary code block, so markdown round-trips unchanged.
+ * ordinary code block, so markdown round-trips unchanged. A ```dashboard fence renders live
+ * dashboard panels the same way.
  */
 function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeViewProps) {
   const [wrap, setWrap] = useState(false)
@@ -60,16 +68,19 @@ function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeView
   const [peekSource, setPeekSource] = useState(false)
   const { copied, copy } = useCopyToClipboard({ resetMs: 1500 })
   const editable = useEditorEditable(editor)
+  const isStreaming = useContext(MarkdownStreamingContext)
 
   const explicitLanguage = node.attrs.language as string | null
   const text = node.textContent
   const isMermaid = explicitLanguage === MERMAID || (!explicitLanguage && looksLikeMermaid(text))
+  const isDashboard = explicitLanguage === DASHBOARD_EMBED_LANGUAGE
+  const isRendered = isMermaid || isDashboard
 
   // Editable Mermaid shows source while the caret is focused inside the block and re-renders the
   // diagram on blur (the Linear/GitHub model). The Show source / Show diagram control drives this by
   // focusing into / blurring the block; read-only uses {@link peekSource} since there is no caret.
   useEffect(() => {
-    if (!isMermaid || !editable) {
+    if (!isRendered || !editable) {
       setEditingInline(false)
       return
     }
@@ -92,13 +103,13 @@ function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeView
       editor.off('focus', sync)
       editor.off('blur', sync)
     }
-  }, [editor, getPos, isMermaid, editable])
+  }, [editor, getPos, isRendered, editable])
 
   const showSource = editable ? editingInline : peekSource
-  const showDiagram = isMermaid && text.trim().length > 0 && !showSource
+  const showRendered = isRendered && text.trim().length > 0 && !showSource
 
-  // Skip language detection on the mermaid path — the picker/label never render there.
-  const language = explicitLanguage ?? (isMermaid ? null : detectLanguage(text)) ?? PLAIN
+  // Skip language detection on rendered blocks — the picker/label never render there.
+  const language = explicitLanguage ?? (isRendered ? null : detectLanguage(text)) ?? PLAIN
   const label =
     LANGUAGE_OPTIONS.find((option) => option.value === language)?.label ??
     explicitLanguage ??
@@ -133,10 +144,12 @@ function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeView
         )}
         contentEditable={false}
       >
-        {isMermaid && (
+        {isRendered && (
           
         )}
-        {!isMermaid &&
+        {!isRendered &&
           (editable ? (
             // Editable: a language picker. Read-only: a static label — selecting a language calls
             // updateAttributes, which would mutate a doc that must not change.
@@ -180,7 +193,7 @@ function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeView
               {label}
             
           ))}
-        {!isMermaid && editable && (
+        {!isRendered && editable && (
           
       
-
+      
          as='code' />
       
- {showDiagram && ( - // Clicking the diagram selects the whole node (same selection ring as an image/code block) - // instead of dropping a caret inside — preventDefault stops ProseMirror placing the caret, - // which would otherwise flip to source. Editing is an explicit Show source / blur action. + {showRendered && ( + // Select the whole node instead of placing a caret, which would flip the block to source.
{ + const target = event.target + if (!(target instanceof Element) || !event.currentTarget.contains(target)) return + if (target.closest('button, input')) return event.preventDefault() const pos = typeof getPos === 'function' ? getPos() : null if (typeof pos === 'number') editor.commands.setNodeSelection(pos) }} > - + {isDashboard ? ( + + + + ) : ( + + )}
)} diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context.ts b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context.ts new file mode 100644 index 00000000000..6600bb4fc74 --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context.ts @@ -0,0 +1,4 @@ +import { createContext } from 'react' + +/** True while agent output is streaming into the editor, for node views that render its content. */ +export const MarkdownStreamingContext = createContext(false) diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.css b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.css index b6a13b7e214..7fd223915c6 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.css +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.css @@ -391,6 +391,12 @@ margin: 1rem 0; } +/* A dashboard results table sizes its own columns and sits flush in its panel. */ +.rich-markdown-nodes .dashboard-embed table { + table-layout: auto; + margin: 0; +} + .rich-markdown-nodes th > p, .rich-markdown-nodes td > p { margin: 0; diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.tsx b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.tsx index 7fde7d1a97d..4c3dc3564d0 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.tsx +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-editor.tsx @@ -68,6 +68,7 @@ import { } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-fidelity' import { parseMarkdownToDoc } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-parse' import { isPlainTextPaste } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-paste' +import { MarkdownStreamingContext } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context' import { useEditorMentions } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/mention' import { EditorBubbleMenu } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/menus/bubble-menu' import { LinkHoverCard } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/menus/link-hover-card' @@ -1439,17 +1440,19 @@ export function LoadedRichMarkdownEditor({ void insertImagesRef.current(images, range) }} /> - {showPlaceholder && placeholder && ( - + {showPlaceholder && placeholder && ( + + )} + - )} - +
) diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx index 2a568884595..40d4d1ea85c 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx @@ -27,6 +27,7 @@ import { } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-fidelity' import { parseMarkdownToDoc } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-parse' import { isPlainTextPaste } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-paste' +import { MarkdownStreamingContext } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context' import { useEditorMentions } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/mention' import { EditorBubbleMenu } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/menus/bubble-menu' import { LinkHoverCard } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/menus/link-hover-card' @@ -444,15 +445,17 @@ function LoadedRichMarkdownField({ }} /> )} - + + + ) } diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.test.ts b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.test.ts new file mode 100644 index 00000000000..08b4f65a381 --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.test.ts @@ -0,0 +1,29 @@ +/** + * @vitest-environment jsdom + */ +import { Editor } from '@tiptap/core' +import { describe, expect, it } from 'vitest' +import { createMarkdownEditorExtensions } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/editor-extensions' +import { SLASH_COMMANDS } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands' + +const chart = SLASH_COMMANDS.find((command) => command.title === 'Chart') + +describe('Chart slash command', () => { + it.each([ + ['an empty paragraph', '

/chart

'], + ['the end of existing text', '

Intro /chart

'], + ])('puts the caret at the start of the starter fence from %s', (_, content) => { + const editor = new Editor({ + extensions: createMarkdownEditorExtensions({ placeholder: '' }), + content, + }) + const end = editor.state.doc.content.size - 1 + const from = editor.state.doc.textBetween(0, end).indexOf('/') + 1 + chart?.run({ editor, range: { from, to: end } }) + const { $from } = editor.state.selection + expect($from.parent.type.name).toBe('codeBlock') + expect($from.parent.attrs.language).toBe('dashboard') + expect($from.parentOffset).toBe(0) + editor.destroy() + }) +}) diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.ts b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.ts index 000a434eba4..8fb32c79595 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.ts +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/slash-command/commands.ts @@ -1,5 +1,6 @@ import type { ComponentType, SVGProps } from 'react' import { + ChartColumn, Code, Heading1, Heading2, @@ -14,6 +15,27 @@ import { TextQuote, } from '@sim/emcn/icons' import type { Editor, Range } from '@tiptap/core' +import { TextSelection } from '@tiptap/pm/state' +import { DASHBOARD_EMBED_LANGUAGE } from '@/lib/dashboards/embed-language' + +/** A time-series starter; the table id is left for the author to fill in. */ +const DASHBOARD_EMBED_STARTER = `title: Rows over time +time: 7d +source: + tableId: # table id +blocks: + - chart: Rows per day + source: + groupBy: [createdAt] + bucket: day + aggregate: + rows: { op: count } + option: + xAxis: { type: time } + yAxis: { type: value } + series: + - { type: line, encode: { x: createdAt, y: rows } } +` export interface SlashCommandContext { editor: Editor @@ -121,6 +143,33 @@ export const SLASH_COMMANDS: readonly SlashCommandItem[] = [ shortcut: '⌘⌥C', run: ({ editor, range }) => editor.chain().focus().deleteRange(range).toggleCodeBlock().run(), }, + { + title: 'Chart', + group: 'Blocks', + icon: ChartColumn, + aliases: ['dashboard', 'graph', 'metric', 'live data'], + run: ({ editor, range }) => + editor + .chain() + .focus() + .deleteRange(range) + .insertContent({ + type: 'codeBlock', + attrs: { language: DASHBOARD_EMBED_LANGUAGE }, + content: [{ type: 'text', text: DASHBOARD_EMBED_STARTER }], + }) + .command(({ tr }) => { + let fence = -1 + tr.doc.nodesBetween(range.from - 1, tr.doc.content.size, (node, pos) => { + if (fence < 0 && node.type.name === 'codeBlock') fence = pos + return fence < 0 + }) + if (fence < 0) return false + tr.setSelection(TextSelection.create(tr.doc, fence + 1)) + return true + }) + .run(), + }, { title: 'Table', group: 'Blocks', diff --git a/apps/sim/components/charts/echarts-view.tsx b/apps/sim/components/charts/echarts-view.tsx index 041ef6c5945..aed2276c426 100644 --- a/apps/sim/components/charts/echarts-view.tsx +++ b/apps/sim/components/charts/echarts-view.tsx @@ -5,9 +5,14 @@ import { cn } from '@sim/emcn' import { getErrorMessage } from '@sim/utils/errors' import type { EChartsType } from 'echarts' import { useTheme } from 'next-themes' +import { applyChartAnnotations, type ChartAnnotations } from '@/lib/charts/annotations' import { installBarRowHighlight } from '@/lib/charts/bar-row-highlight' import { chartSummaryExtension } from '@/lib/charts/summary' -import { applyChartTooltipDefaults, readEmcnChartTheme } from '@/lib/charts/theme' +import { + applyChartTooltipDefaults, + readChartTonePalette, + readEmcnChartTheme, +} from '@/lib/charts/theme' interface EChartsViewProps { option: Record @@ -15,6 +20,8 @@ interface EChartsViewProps { className?: string createController?: (chart: EChartsType) => EChartsController revision?: string + /** Highlights and thresholds, coloured from the theme at render time. */ + annotations?: ChartAnnotations } export interface EChartsController { @@ -33,6 +40,7 @@ export function EChartsView({ className, createController, revision, + annotations, }: EChartsViewProps) { const containerRef = useRef(null) const chartRef = useRef(null) @@ -40,15 +48,23 @@ export function EChartsView({ const rowHighlightRef = useRef<(() => void) | null>(null) const { resolvedTheme } = useTheme() const [status, setStatus] = useState<{ theme: string | undefined; error?: string } | null>(null) - const optionKey = JSON.stringify(option) + const optionKey = JSON.stringify({ option, annotations }) const applyOption = useEffectEvent((chart: EChartsType, nextOption: string) => { try { controllerRef.current?.dispose() controllerRef.current = null const controller = createController?.(chart) controllerRef.current = controller ?? null - const parsed = applyChartTooltipDefaults(JSON.parse(nextOption)) - chart.setOption(controller ? controller.prepareOption(parsed) : parsed, { notMerge: true }) + const next: { option: Record; annotations?: ChartAnnotations } = + JSON.parse(nextOption) + const parsed = applyChartTooltipDefaults(next.option) + // Annotate after the bar helpers read the option: the label series would read as a non-bar chart. + const rendered = next.annotations + ? applyChartAnnotations(parsed, next.annotations, readChartTonePalette(chart.getDom())) + : parsed + chart.setOption(controller ? controller.prepareOption(rendered) : rendered, { + notMerge: true, + }) controller?.afterUpdate() rowHighlightRef.current?.() rowHighlightRef.current = installBarRowHighlight(chart, parsed) diff --git a/apps/sim/components/charts/time-series-chart.tsx b/apps/sim/components/charts/time-series-chart.tsx index 8554a8f4df0..9b1755e78cf 100644 --- a/apps/sim/components/charts/time-series-chart.tsx +++ b/apps/sim/components/charts/time-series-chart.tsx @@ -3,6 +3,7 @@ import { useRef, useState } from 'react' import { cn, scrollFadeAttributes, scrollFadeXClass, useScrollEdges } from '@sim/emcn' import { EChartsView } from '@/components/charts/echarts-view' +import type { ChartAnnotations } from '@/lib/charts/annotations' import { bindTimeSeriesInteractions, type ChartReadout, @@ -13,9 +14,10 @@ import { dashboardTimeLabel } from '@/lib/dashboards/time' interface TimeSeriesChartProps extends Omit { label: string option: Record + annotations?: ChartAnnotations } -export function TimeSeriesChart({ label, option, ...config }: TimeSeriesChartProps) { +export function TimeSeriesChart({ label, option, annotations, ...config }: TimeSeriesChartProps) { const valuesRef = useRef(null) const edges = useScrollEdges(valuesRef, { axis: 'x' }) const [readout, setReadout] = useState(null) @@ -58,6 +60,7 @@ export function TimeSeriesChart({ label, option, ...config }: TimeSeriesChartPro void onRefresh: () => void } -const RANGE_OPTIONS = [ - { value: '1h', label: 'Last hour' }, - { value: '24h', label: 'Last 24 hours' }, - { value: '7d', label: 'Last 7 days' }, - { value: '30d', label: 'Last 30 days' }, - { value: '90d', label: 'Last 90 days' }, -] as const +const RANGE_OPTIONS = DASHBOARD_RANGES.map((value) => ({ + value, + label: DASHBOARD_RANGE_LABELS[value], +})) export function DashboardControls({ period, @@ -54,26 +56,14 @@ export function DashboardControls({ new Intl.DateTimeFormat('en-US', { timeZone: localTimeZone, timeZoneName: 'short' }) .formatToParts(new Date(range.to)) .find((part) => part.type === 'timeZoneName')?.value ?? localTimeZone - const from = new Date(range.from) - const to = new Date(Date.parse(range.to) - 1) - const fromLocal = zonedWallClock(from, timeZone) - const toLocal = zonedWallClock(to, timeZone) - const sameDay = fromLocal.slice(0, 10) === toLocal.slice(0, 10) - const dates = new Intl.DateTimeFormat('en-US', { - timeZone, - month: 'short', - day: 'numeric', - year: fromLocal.slice(0, 4) === toLocal.slice(0, 4) ? undefined : 'numeric', - hour: sameDay ? '2-digit' : undefined, - minute: sameDay ? '2-digit' : undefined, - hourCycle: 'h23', - }) + const fromLocal = zonedWallClock(new Date(range.from), timeZone) + const toLocal = zonedWallClock(new Date(Date.parse(range.to) - 1), timeZone) const label = period === 'custom' ? rangeError ? 'Custom: choose range' - : `Custom: ${dates.formatRange(from, to)}` - : RANGE_OPTIONS.find((option) => option.value === period)!.label + : `Custom: ${dashboardRangeText(range, timeZone)}` + : DASHBOARD_RANGE_LABELS[period] return (
+ {children} +
+ ) +} + +/** + * A ```dashboard fence rendered as live panels. Data is read with the viewer's own session, so + * the fence only renders inside its workspace; a public share shows a notice instead of querying. + */ +export function DashboardEmbed({ source, isStreaming }: DashboardEmbedProps) { + const params = useParams() + const workspaceId = typeof params.workspaceId === 'string' ? params.workspaceId : null + if (isStreaming) return The chart loads when Sim finishes writing. + if (!workspaceId) + return Open this document in its workspace to see live data. + return ( + + + + ) +} + +function LiveDashboardEmbed({ source, workspaceId }: LiveDashboardEmbedProps) { + const parsed = useMemo(() => parseDashboardEmbed(source), [source]) + if (!parsed.spec) return {parsed.error} + return +} + +function EmbedView({ spec, workspaceId }: EmbedViewProps) { + const rootRef = useRef(null) + const embedId = useId() + const [inView, setInView] = useState(false) + const [state, setState] = useState({ + range: null, + from: null, + to: null, + zone: 'local', + }) + const time = useDashboardTime({ + state, + setState: (update) => setState((current) => ({ ...current, ...update })), + time: spec.time, + workspaceId, + tableIds: dashboardTableIds(spec.blocks, spec.source), + live: inView, + }) + useEffect(() => { + const root = rootRef.current + if (!root) return + const observer = new IntersectionObserver(([entry]) => setInView(entry.isIntersecting)) + observer.observe(root) + return () => observer.disconnect() + }, []) + const zoomed = state.range !== null + const { period } = time.controls + const caption = + period === 'custom' + ? dashboardRangeText(time.range, time.interactions.timeZone) + : DASHBOARD_RANGE_LABELS[period] + return ( +
+ {spec.title && ( +

{spec.title}

+ )} +

+ + + {caption} + + + {dashboardTimeLabel(time.range.from, time.interactions.timeZone)} –{' '} + {dashboardTimeLabel(time.range.to, time.interactions.timeZone)} + + + {zoomed && ( + <> + {' · '} + + + )} +

+ {time.rangeError ? ( +

+ {time.rangeError} +

+ ) : ( + + + + )} +
+ ) +} diff --git a/apps/sim/components/dashboards/dashboard-layout.tsx b/apps/sim/components/dashboards/dashboard-layout.tsx index f3c876c97bb..e60470aae99 100644 --- a/apps/sim/components/dashboards/dashboard-layout.tsx +++ b/apps/sim/components/dashboards/dashboard-layout.tsx @@ -4,6 +4,7 @@ import { cn, TabStrip } from '@sim/emcn' import { useQueryState } from 'nuqs' import { DashboardPanel } from '@/components/dashboards/dashboard-panel' import { dashboardTabParser, dashboardUrlOptions } from '@/components/dashboards/search-params' +import type { ChartHighlight } from '@/lib/charts/annotations' import type { DashboardBlock, DashboardSource, DashboardTabs } from '@/lib/dashboards/spec' import type { DashboardTimeRange } from '@/lib/dashboards/time' @@ -16,6 +17,8 @@ interface DashboardLayoutProps { now: number path?: string startIndex?: number + embedded?: boolean + highlights?: ChartHighlight[] } interface DashboardTabsProps extends Omit { block: DashboardTabs diff --git a/apps/sim/components/dashboards/dashboard-panel.tsx b/apps/sim/components/dashboards/dashboard-panel.tsx index 2a35bf9b98c..f75d4f924aa 100644 --- a/apps/sim/components/dashboards/dashboard-panel.tsx +++ b/apps/sim/components/dashboards/dashboard-panel.tsx @@ -15,8 +15,9 @@ import { EChartsView } from '@/components/charts/echarts-view' import { TimeSeriesChart } from '@/components/charts/time-series-chart' import { useDashboardInteractions } from '@/components/dashboards/dashboard-interactions' import type { QueryTableAnalyticsResponse } from '@/lib/api/contracts/table-analytics' +import type { ChartAnnotations, ChartHighlight } from '@/lib/charts/annotations' import { buildChartRenderOption, horizontalBarChartHeight } from '@/lib/charts/option' -import { isTimeSeriesOption } from '@/lib/charts/time-series' +import { isTimeSeriesOption } from '@/lib/charts/spec' import { type DashboardDataBlock, type DashboardSource, @@ -36,6 +37,10 @@ interface DashboardPanelProps { workspaceId: string range: DashboardTimeRange now: number + /** Inside a markdown document, where headings belong to the document outline. */ + embedded?: boolean + /** Drawn on time-series charts only. */ + highlights?: ChartHighlight[] } function displayValue(value: string | number | boolean | null | undefined): string { @@ -83,7 +88,15 @@ function ResultsTable({ data, timeField, timeZone }: ResultsTableProps) { ) } -export function DashboardPanel({ block, defaults, workspaceId, range, now }: DashboardPanelProps) { +export function DashboardPanel({ + block, + defaults, + workspaceId, + range, + now, + embedded = false, + highlights, +}: DashboardPanelProps) { const interactions = useDashboardInteractions() const source = resolveDashboardSource(defaults, block.source) const panelRange = source.range ? relativeDashboardRange(source.range, now) : range @@ -97,6 +110,7 @@ export function DashboardPanel({ block, defaults, workspaceId, range, now }: Das ...panelRange, }, }, + refreshedAt: now, }) const title = 'stat' in block ? block.stat : 'chart' in block ? block.chart : block.table const data = query.data @@ -109,6 +123,10 @@ export function DashboardPanel({ block, defaults, workspaceId, range, now }: Das rows: data.rows, }) : null + const annotations: ChartAnnotations | undefined = + 'chart' in block && ((timeSeries && highlights) || block.thresholds) + ? { highlights: timeSeries ? highlights : undefined, thresholds: block.thresholds } + : undefined const barChartHeight = 'chart' in block ? horizontalBarChartHeight(block.option, data?.rows.length ?? 10) : null const times = @@ -144,6 +162,15 @@ export function DashboardPanel({ block, defaults, workspaceId, range, now }: Das unit={!query.isError && metric != null ? block.unit : undefined} loading={query.isPending} /> + ) : embedded ? ( +

+ {title} +

) : (

)} {data.rows.length === 0 && ( diff --git a/apps/sim/components/dashboards/dashboard-preview.test.tsx b/apps/sim/components/dashboards/dashboard-preview.test.tsx index ad877bbac24..d53a3d1496d 100644 --- a/apps/sim/components/dashboards/dashboard-preview.test.tsx +++ b/apps/sim/components/dashboards/dashboard-preview.test.tsx @@ -7,8 +7,10 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { DashboardPreview } from '@/components/dashboards/dashboard-preview' import { tableAnalyticsKeys } from '@/hooks/queries/table-analytics' +const { layout } = vi.hoisted(() => ({ layout: vi.fn((_props: { now: number }) => null) })) + /** Panels would issue analytics requests; this suite exercises the controls and query cache. */ -vi.mock('@/components/dashboards/dashboard-layout', () => ({ DashboardLayout: () => null })) +vi.mock('@/components/dashboards/dashboard-layout', () => ({ DashboardLayout: layout })) const content = 'title: Example\ntime: 7d\nsource: {tableId: table-1}\nblocks: [{stat: Total, source: {aggregate: {total: {op: count}}}}]' @@ -20,6 +22,7 @@ describe('dashboard refresh', () => { let client: QueryClient beforeEach(() => { vi.stubGlobal('IS_REACT_ACT_ENVIRONMENT', true) + layout.mockClear() client = new QueryClient() container = document.createElement('div') document.body.append(container) @@ -31,12 +34,13 @@ describe('dashboard refresh', () => { client.clear() }) - it('refreshes only this workspace and this dashboard tables', async () => { + it('refreshes by re-keying its panels to a new now, leaving other caches alone', async () => { const key = (workspaceId: string, tableId: string) => - tableAnalyticsKeys.query(tableId, { - workspaceId, - query: { ...range, aggregate: { total: { op: 'count' } } }, - }) + tableAnalyticsKeys.query( + tableId, + { workspaceId, query: { ...range, aggregate: { total: { op: 'count' } } } }, + 0 + ) const matching = key('workspace-1', 'table-1') const otherTable = key('workspace-1', 'other-table') const otherWorkspace = key('workspace-2', 'table-1') @@ -60,9 +64,13 @@ describe('dashboard refresh', () => { if (!button) throw new Error('Refresh button not rendered') return button }) + const before = layout.mock.lastCall?.[0].now + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(Date.now() + 60_000) await act(async () => refresh.click()) - expect(client.getQueryState(matching)?.isInvalidated).toBe(true) - expect(client.getQueryState(otherTable)?.isInvalidated).toBe(false) - expect(client.getQueryState(otherWorkspace)?.isInvalidated).toBe(false) + vi.useRealTimers() + expect(layout.mock.lastCall?.[0].now).toBeGreaterThan(before ?? Number.POSITIVE_INFINITY) + for (const queryKey of [matching, otherTable, otherWorkspace]) + expect(client.getQueryState(queryKey)?.isInvalidated).toBe(false) }) }) diff --git a/apps/sim/components/dashboards/dashboard-preview.tsx b/apps/sim/components/dashboards/dashboard-preview.tsx index d2337d1b9bc..1449853d6b9 100644 --- a/apps/sim/components/dashboards/dashboard-preview.tsx +++ b/apps/sim/components/dashboards/dashboard-preview.tsx @@ -1,9 +1,6 @@ 'use client' -import { Suspense, useMemo, useRef, useState } from 'react' -import { getErrorMessage } from '@sim/utils/errors' -import { toRecord } from '@sim/utils/object' -import { useIsFetching, useQueryClient } from '@tanstack/react-query' +import { Suspense, useMemo } from 'react' import { useQueryStates } from 'nuqs' import { DashboardControls } from '@/components/dashboards/dashboard-controls' import { DashboardInteractionContext } from '@/components/dashboards/dashboard-interactions' @@ -13,21 +10,8 @@ import { dashboardUrlKeys, dashboardUrlOptions, } from '@/components/dashboards/search-params' -import { getBrowserTimezone } from '@/lib/core/utils/timezone' -import { - type DashboardBlock, - type DashboardSpec, - parseDashboardSpec, - resolveDashboardSource, -} from '@/lib/dashboards/spec' -import { - type DashboardTimeRange, - dashboardRangeFromCalendar, - parseDashboardCustomRange, - relativeDashboardRange, -} from '@/lib/dashboards/time' -import { tableAnalyticsKeys } from '@/hooks/queries/table-analytics' -import { createDashboardCursorStore, type DashboardCursorStore } from '@/stores/dashboards/cursor' +import { useDashboardTime } from '@/components/dashboards/use-dashboard-time' +import { type DashboardSpec, dashboardTableIds, parseDashboardSpec } from '@/lib/dashboards/spec' interface DashboardPreviewProps { content: string @@ -42,60 +26,21 @@ interface DashboardViewProps { dashboardId: string } -function dashboardTableIds(spec: DashboardSpec): Set { - const ids = new Set() - const visit = (blocks: DashboardBlock[]) => { - for (const block of blocks) { - if ('row' in block) visit(block.row) - else if ('tabs' in block) Object.values(block.tabs).forEach(visit) - else if (!('text' in block)) - ids.add(resolveDashboardSource(spec.source, block.source).tableId) - } - } - visit(spec.blocks) - return ids -} - function DashboardView({ spec, workspaceId, dashboardId }: DashboardViewProps) { - const cursorStoreRef = useRef(null) - cursorStoreRef.current ??= createDashboardCursorStore() const [state, setState] = useQueryStates(dashboardParsers, { ...dashboardUrlOptions, urlKeys: dashboardUrlKeys(dashboardId), }) - const [now, setNow] = useState(() => Date.now()) - const [inputError, setInputError] = useState(null) - const queryClient = useQueryClient() - const tableIds = dashboardTableIds(spec) - const queryFilter = { - queryKey: tableAnalyticsKeys.queries(), - predicate: (query: { queryKey: readonly unknown[] }) => { - return ( - toRecord(query.queryKey[3]).workspaceId === workspaceId && - tableIds.has(String(query.queryKey[2])) - ) - }, - } - const isFetching = useIsFetching(queryFilter) > 0 - const localTimeZone = getBrowserTimezone() - const timeZone = state.zone === 'local' ? localTimeZone : 'UTC' - const period = state.range ?? spec.time ?? '7d' + const time = useDashboardTime({ + state, + setState: (update) => void setState(update), + time: spec.time, + workspaceId, + tableIds: dashboardTableIds(spec.blocks, spec.source), + }) const firstBlock = spec.blocks[0] const description = firstBlock && 'text' in firstBlock ? firstBlock.text : null const startIndex = description === null ? 0 : 1 - let range = relativeDashboardRange(period === 'custom' ? '7d' : period, now) - let rangeError: string | null = null - if (period === 'custom') { - try { - range = parseDashboardCustomRange(state.from ?? '', state.to ?? '') - } catch (error) { - rangeError = getErrorMessage(error, 'Choose a custom range') - } - } - const onZoom = (selected: DashboardTimeRange) => { - setInputError(null) - void setState({ range: 'custom', ...selected }) - } return (
@@ -109,59 +54,28 @@ function DashboardView({ spec, workspaceId, dashboardId }: DashboardViewProps) {

)}
- { - setInputError(null) - cursorStoreRef.current?.getState().clearCursor() - setNow(Date.now()) - void setState({ range: value, from: null, to: null }) - }} - onCalendarChange={(from, to) => { - try { - const selected = dashboardRangeFromCalendar(from, to, timeZone) - setInputError(null) - void setState({ range: 'custom', ...selected }) - return true - } catch (error) { - setInputError(getErrorMessage(error, 'Invalid range')) - return false - } - }} - onRefresh={() => { - setNow(Date.now()) - cursorStoreRef.current?.getState().clearCursor() - if (period === 'custom') void queryClient.invalidateQueries(queryFilter) - }} - onZoneChange={(zone) => void setState({ zone })} - /> + - {inputError && ( + {time.inputError && (

- {inputError} + {time.inputError}

)} - {rangeError ? ( + {time.rangeError ? (

- {rangeError} + {time.rangeError}

) : ( - + )} diff --git a/apps/sim/components/dashboards/use-dashboard-time.ts b/apps/sim/components/dashboards/use-dashboard-time.ts new file mode 100644 index 00000000000..f96b8ad0b3b --- /dev/null +++ b/apps/sim/components/dashboards/use-dashboard-time.ts @@ -0,0 +1,152 @@ +'use client' + +import { useEffect, useEffectEvent, useRef, useState } from 'react' +import { getErrorMessage } from '@sim/utils/errors' +import { toRecord } from '@sim/utils/object' +import { useIsFetching } from '@tanstack/react-query' +import { getBrowserTimezone } from '@/lib/core/utils/timezone' +import type { DashboardRange, DashboardTime } from '@/lib/dashboards/spec' +import { + type DashboardTimeRange, + dashboardRangeFromCalendar, + parseDashboardCustomRange, + relativeDashboardRange, +} from '@/lib/dashboards/time' +import { tableAnalyticsKeys } from '@/hooks/queries/table-analytics' +import { createDashboardCursorStore, type DashboardCursorStore } from '@/stores/dashboards/cursor' + +/** How often a relative range advances its end to the present while the page is visible. */ +const LIVE_TICK_MS = 60_000 + +export interface DashboardTimeState { + range: DashboardRange | 'custom' | null + from: string | null + to: string | null + zone: 'utc' | 'local' +} + +interface UseDashboardTimeProps { + state: DashboardTimeState + setState: (update: Partial) => void + /** The authored default; the viewer's selection in `state` wins over it. */ + time: DashboardTime | undefined + workspaceId: string + tableIds: ReadonlySet + /** + * Refreshes every minute while true and the page is visible: relative ranges advance to the + * present, fixed and zoomed ranges refetch in place. + */ + live?: boolean +} + +/** + * Resolves the active time range and wires the range controls, zoom, refresh, and cursor sync. + * The caller owns where the selection lives: the URL on the dashboard page, local state in an + * embedded fence. + */ +export function useDashboardTime({ + state, + setState, + time, + workspaceId, + tableIds, + live = false, +}: UseDashboardTimeProps) { + const cursorStoreRef = useRef(null) + cursorStoreRef.current ??= createDashboardCursorStore() + const cursorStore = cursorStoreRef.current + const [now, setNow] = useState(() => Date.now()) + const [inputError, setInputError] = useState(null) + const queryFilter = { + queryKey: tableAnalyticsKeys.queries(), + predicate: (query: { queryKey: readonly unknown[] }) => + toRecord(query.queryKey[3]).workspaceId === workspaceId && + tableIds.has(String(query.queryKey[2])), + } + const isFetching = useIsFetching(queryFilter) > 0 + const timeZone = state.zone === 'local' ? getBrowserTimezone() : 'UTC' + const fixed = typeof time === 'object' ? time : null + const preset = typeof time === 'string' ? time : '7d' + const period = state.range ?? (fixed ? 'custom' : preset) + + let range: DashboardTimeRange = relativeDashboardRange(period === 'custom' ? '7d' : period, now) + let rangeError: string | null = null + if (period === 'custom') { + const custom = state.range === 'custom' ? { from: state.from ?? '', to: state.to ?? '' } : fixed + try { + if (!custom) throw new Error('Choose a custom range') + range = parseDashboardCustomRange(custom.from, custom.to) + } catch (error) { + rangeError = getErrorMessage(error, 'Choose a custom range') + } + } + + /** Advancing `now` rolls relative ranges and re-keys fixed ones, refetching only this view. */ + const tick = useEffectEvent(() => { + if (document.visibilityState === 'visible') setNow(Date.now()) + }) + /** A view that resumes after a pause catches up at once instead of on the next tick. */ + const catchUp = useEffectEvent(() => { + if (Date.now() - now >= LIVE_TICK_MS) tick() + }) + useEffect(() => { + if (!live) return + catchUp() + const interval = setInterval(tick, LIVE_TICK_MS) + document.addEventListener('visibilitychange', tick) + return () => { + clearInterval(interval) + document.removeEventListener('visibilitychange', tick) + } + }, [live]) + + const onZoom = (selected: DashboardTimeRange) => { + setInputError(null) + setState({ range: 'custom', ...selected }) + } + /** Returns to the authored range, re-anchored to the present. */ + const reset = () => { + cursorStore.getState().clearCursor() + setNow(Date.now()) + setState({ range: null, from: null, to: null }) + } + + return { + range, + now, + rangeError, + inputError, + reset, + interactions: { cursorStore, timeZone, onZoom }, + controls: { + period, + range, + timeZone, + zone: state.zone, + isFetching, + rangeError: rangeError !== null, + onPeriodChange: (value: DashboardRange | 'custom') => { + setInputError(null) + cursorStore.getState().clearCursor() + setNow(Date.now()) + setState({ range: value, from: null, to: null }) + }, + onCalendarChange: (from: string, to: string) => { + try { + const selected = dashboardRangeFromCalendar(from, to, timeZone) + setInputError(null) + setState({ range: 'custom', ...selected }) + return true + } catch (error) { + setInputError(getErrorMessage(error, 'Invalid range')) + return false + } + }, + onRefresh: () => { + setNow(Date.now()) + cursorStore.getState().clearCursor() + }, + onZoneChange: (zone: 'utc' | 'local') => setState({ zone }), + }, + } +} diff --git a/apps/sim/hooks/queries/table-analytics.ts b/apps/sim/hooks/queries/table-analytics.ts index 9ed2648a3cc..1bb6aefed77 100644 --- a/apps/sim/hooks/queries/table-analytics.ts +++ b/apps/sim/hooks/queries/table-analytics.ts @@ -12,17 +12,19 @@ export const TABLE_ANALYTICS_STALE_TIME = 60_000 export const tableAnalyticsKeys = { all: ['table-analytics'] as const, queries: () => [...tableAnalyticsKeys.all, 'query'] as const, - query: (tableId: string, body: QueryTableAnalyticsBody) => - [...tableAnalyticsKeys.queries(), tableId, body] as const, + /** `refreshedAt` re-keys a fixed range when its view refreshes, so only that view refetches. */ + query: (tableId: string, body: QueryTableAnalyticsBody, refreshedAt: number) => + [...tableAnalyticsKeys.queries(), tableId, body, refreshedAt] as const, } interface UseTableAnalyticsProps { tableId: string body: QueryTableAnalyticsBody + refreshedAt: number } -export function useTableAnalytics({ tableId, body }: UseTableAnalyticsProps) { +export function useTableAnalytics({ tableId, body, refreshedAt }: UseTableAnalyticsProps) { return useQuery({ - queryKey: tableAnalyticsKeys.query(tableId, body), + queryKey: tableAnalyticsKeys.query(tableId, body, refreshedAt), queryFn: async ({ signal }) => { const result = await requestJson(queryTableAnalyticsContract, { params: { tableId }, diff --git a/apps/sim/lib/charts/annotations.test.ts b/apps/sim/lib/charts/annotations.test.ts new file mode 100644 index 00000000000..60aafc426bd --- /dev/null +++ b/apps/sim/lib/charts/annotations.test.ts @@ -0,0 +1,119 @@ +import { describe, expect, it } from 'vitest' +import { applyChartAnnotations, CHART_ANNOTATION_SERIES_ID } from '@/lib/charts/annotations' + +const palette = { tones: { neutral: 'grey', error: 'red', info: 'blue' } } +const timeSeries = { + xAxis: { type: 'time' }, + yAxis: { type: 'value' }, + series: [{ type: 'line' }, { type: 'line' }], +} +type Series = Record + +describe('chart annotations', () => { + it('draws marks under the first series and their labels on a series above', () => { + const option = applyChartAnnotations( + timeSeries, + { + highlights: [ + { + from: '2026-08-21T00:00:00.000Z', + to: '2026-09-07T00:00:00.000Z', + label: 'Outage', + tone: 'error', + }, + { at: '2026-09-10T14:00:00.000Z', tone: 'info' }, + ], + thresholds: [{ value: 50, label: 'SLO' }], + }, + palette + ) + const [first, second, labels] = option.series as Series[] + expect(first.markArea).toMatchObject({ + data: [ + [ + { + xAxis: '2026-08-21T00:00:00.000Z', + itemStyle: { color: 'red' }, + label: { show: false }, + }, + { xAxis: '2026-09-07T00:00:00.000Z' }, + ], + ], + }) + expect(first.markLine).toMatchObject({ + z: 1, + data: [ + { xAxis: '2026-09-10T14:00:00.000Z', lineStyle: { color: 'blue', opacity: 1 } }, + { yAxis: 50, lineStyle: { color: 'grey', opacity: 1 }, label: { show: false } }, + ], + }) + expect(second).toEqual({ type: 'line' }) + expect(labels).toMatchObject({ + id: CHART_ANNOTATION_SERIES_ID, + data: [], + z: 3, + markArea: { + data: [ + [{ name: 'Outage', itemStyle: { opacity: 0 }, label: { show: true, color: 'red' } }, {}], + ], + }, + markLine: { + data: [{ yAxis: 50, name: 'SLO', lineStyle: { opacity: 0 }, label: { color: 'grey' } }], + }, + }) + }) + + it('labels vertical lines at the visual top of an inverted bar axis', () => { + const option = applyChartAnnotations( + { + xAxis: { type: 'value' }, + yAxis: { type: 'category', inverse: true }, + series: [{ type: 'bar' }], + }, + { thresholds: [{ value: 100, label: 'Needs tuning' }] }, + palette + ) + const [, labels] = option.series as Series[] + expect(labels.markLine).toMatchObject({ + data: [{ xAxis: 100, label: { position: 'start' } }], + }) + }) + + it('puts the label series on the same axes as the first series', () => { + const option = applyChartAnnotations( + { + xAxis: { type: 'time' }, + yAxis: [{ type: 'value' }, { type: 'value' }], + series: [{ type: 'line', yAxisIndex: 1 }], + }, + { thresholds: [{ value: 5, label: 'Limit' }] }, + palette + ) + expect((option.series as Series[])[1]).toMatchObject({ yAxisIndex: 1 }) + }) + + it('rejects thresholds on a chart without axes', () => { + expect(() => + applyChartAnnotations({ series: [{ type: 'pie' }] }, { thresholds: [{ value: 1 }] }, palette) + ).toThrow('Thresholds require a value axis') + }) + + it('adds no label series when nothing is labelled', () => { + const option = applyChartAnnotations(timeSeries, { thresholds: [{ value: 1 }] }, palette) + expect(option.series).toHaveLength(2) + }) + + it('leaves an option without annotations untouched', () => { + expect(applyChartAnnotations(timeSeries, {}, palette)).toBe(timeSeries) + }) + + it('refuses to merge with hand-written mark components', () => { + expect(() => + applyChartAnnotations( + { ...timeSeries, series: [{ type: 'line', markLine: { data: [] } }] }, + { thresholds: [{ value: 1 }] }, + palette + ) + ).toThrow('Use highlights and thresholds instead of markArea or markLine on the series') + }) +}) diff --git a/apps/sim/lib/charts/annotations.ts b/apps/sim/lib/charts/annotations.ts new file mode 100644 index 00000000000..79c41a1e2f6 --- /dev/null +++ b/apps/sim/lib/charts/annotations.ts @@ -0,0 +1,176 @@ +import { filterUndefined, toRecord } from '@sim/utils/object' + +export const CHART_TONES = ['neutral', 'error', 'info'] as const +export type ChartTone = (typeof CHART_TONES)[number] +/** Tone colours for bands, lines, and their labels; neutral matches the axis labels. */ +export interface ChartTonePalette { + tones: Record +} + +/** A shaded stretch of time, or a single instant drawn as a vertical line. */ +export type ChartHighlight = + | { from: string; to: string; label?: string; tone?: ChartTone } + | { at: string; label?: string; tone?: ChartTone } + +/** A horizontal (or, on horizontal bars, vertical) reference line on the value axis. */ +export interface ChartThreshold { + value: number + label?: string + tone?: ChartTone +} + +export interface ChartAnnotations { + /** Applied only to charts with a time x-axis. */ + highlights?: readonly ChartHighlight[] + thresholds?: readonly ChartThreshold[] +} + +const BAND_OPACITY = 0.08 + +function firstAxis(axis: unknown): Record { + return toRecord(Array.isArray(axis) ? axis[0] : axis) +} + +/** The axis a threshold is measured on; ECharts defaults an unspecified yAxis to a value axis. */ +export function valueAxisKey(option: Record): 'xAxis' | 'yAxis' { + if (option.xAxis === undefined && option.yAxis === undefined) + throw new Error('Thresholds require a value axis') + const y = firstAxis(option.yAxis) + if (y.type === undefined || y.type === 'value' || y.type === 'log') return 'yAxis' + const x = firstAxis(option.xAxis) + if (x.type === 'value' || x.type === 'log') return 'xAxis' + throw new Error('Thresholds require a value axis') +} + +/** + * A vertical line runs from the y-axis start to its end, so its visual top is the start when the + * y axis is inverted (as horizontal bar charts usually are). + */ +function verticalTop(option: Record): 'start' | 'end' { + return firstAxis(option.yAxis).inverse === true ? 'start' : 'end' +} + +/** A chart annotations can draw on: a first series with no hand-written marks to collide with. */ +export function assertAnnotatable(option: Record): void { + const series = Array.isArray(option.series) ? option.series : [option.series] + if (series[0] === undefined) throw new Error('Highlights and thresholds require a series') + const first = toRecord(series[0]) + if (first.markArea !== undefined || first.markLine !== undefined) + throw new Error('Use highlights and thresholds instead of markArea or markLine on the series') +} + +/** + * Id of the empty series that carries annotation labels. ECharts draws a mark's label at the mark's + * depth, so the marks sit under the data and their labels ride on this series above it. + */ +export const CHART_ANNOTATION_SERIES_ID = '\u0000annotations' + +interface AnnotationMark { + label?: string + color: string + position: string +} + +/** Vertical lines label upright past their top end; ECharts rotates `inside*` labels. */ +function markLabel(mark: AnnotationMark) { + return { show: true, formatter: '{b}', color: mark.color, fontSize: 12, position: mark.position } +} + +/** + * Draws highlights and thresholds as `markArea` and `markLine` under the first series, with their + * labels on a silent series drawn above every series, coloured from the theme palette. + */ +export function applyChartAnnotations( + option: Record, + annotations: ChartAnnotations, + palette: ChartTonePalette +): Record { + const highlights = annotations.highlights ?? [] + const thresholds = annotations.thresholds ?? [] + if (highlights.length === 0 && thresholds.length === 0) return option + assertAnnotatable(option) + const series = Array.isArray(option.series) ? option.series : [option.series] + const first = toRecord(series[0]) + + const bands: Array = [] + const lines: Array }> = [] + for (const highlight of highlights) { + const color = palette.tones[highlight.tone ?? 'neutral'] + if ('from' in highlight) bands.push({ ...highlight, color, position: 'insideTop' }) + else + lines.push({ + label: highlight.label, + color, + position: verticalTop(option), + coord: { xAxis: highlight.at }, + }) + } + for (const threshold of thresholds) { + const axis = valueAxisKey(option) + lines.push({ + label: threshold.label, + color: palette.tones[threshold.tone ?? 'neutral'], + position: axis === 'xAxis' ? verticalTop(option) : 'insideEndTop', + coord: { [axis]: threshold.value }, + }) + } + + const band = (mark: (typeof bands)[number], visible: boolean) => [ + { + name: mark.label ?? '', + xAxis: mark.from, + itemStyle: { color: mark.color, opacity: visible ? BAND_OPACITY : 0 }, + label: visible ? { show: false } : markLabel(mark), + }, + { xAxis: mark.to }, + ] + const line = (mark: (typeof lines)[number], visible: boolean) => ({ + name: mark.label ?? '', + ...mark.coord, + lineStyle: { color: mark.color, type: 'dashed', width: 1, opacity: visible ? 1 : 0 }, + label: visible ? { show: false } : markLabel(mark), + }) + const markArea = (marks: typeof bands, visible: boolean) => + marks.length > 0 && { markArea: { silent: true, data: marks.map((m) => band(m, visible)) } } + const markLine = (marks: typeof lines, visible: boolean) => + marks.length > 0 && { + markLine: { + silent: true, + symbol: ['none', 'none'], + z: 1, + data: marks.map((m) => line(m, visible)), + }, + } + + const labelled = (marks: T[]) => marks.filter((mark) => mark.label) + const labelSeries = [ + ...(labelled(bands).length + labelled(lines).length > 0 + ? [ + { + id: CHART_ANNOTATION_SERIES_ID, + type: 'line', + ...filterUndefined({ + xAxisIndex: first.xAxisIndex, + yAxisIndex: first.yAxisIndex, + xAxisId: first.xAxisId, + yAxisId: first.yAxisId, + }), + data: [], + silent: true, + z: 3, + tooltip: { show: false }, + ...markArea(labelled(bands), false), + ...markLine(labelled(lines), false), + }, + ] + : []), + ] + return { + ...option, + series: [ + { ...first, ...markArea(bands, true), ...markLine(lines, true) }, + ...series.slice(1), + ...labelSeries, + ], + } +} diff --git a/apps/sim/lib/charts/spec.ts b/apps/sim/lib/charts/spec.ts index 37bc55f0aad..c27f8a089b8 100644 --- a/apps/sim/lib/charts/spec.ts +++ b/apps/sim/lib/charts/spec.ts @@ -5,6 +5,7 @@ */ import { getErrorMessage } from '@sim/utils/errors' +import { toRecord } from '@sim/utils/object' import { getColumnId } from '@/lib/table/column-keys' import type { ColumnDefinition } from '@/lib/table/types' @@ -270,3 +271,9 @@ export function shapeTableRows( } return out } + +/** Cartesian charts with one horizontal time axis share dashboard interactions. */ +export function isTimeSeriesOption(option: Record): boolean { + const axes = Array.isArray(option.xAxis) ? option.xAxis : [option.xAxis] + return axes.length === 1 && toRecord(axes[0]).type === 'time' +} diff --git a/apps/sim/lib/charts/summary.ts b/apps/sim/lib/charts/summary.ts index 0744eb5bf91..1157bbb358f 100644 --- a/apps/sim/lib/charts/summary.ts +++ b/apps/sim/lib/charts/summary.ts @@ -1,5 +1,6 @@ import { toRecord } from '@sim/utils/object' import type { EChartsType, registerUpdateLifecycle } from 'echarts' +import { CHART_ANNOTATION_SERIES_ID } from '@/lib/charts/annotations' import type { ChartReadout, ChartReadoutValue } from '@/lib/charts/time-series' type ChartModel = Parameters>[1]>[0] @@ -43,6 +44,7 @@ export function summarizeChart(model: ChartModel, labels: Record const values: ChartReadoutValue[] = [] model.eachSeries((series) => { if (series.get('coordinateSystem') !== 'cartesian2d') return + if (series.id === CHART_ANNOTATION_SERIES_ID) return const data = series.getRawData() const axis = model.getComponent('yAxis', Number(toRecord(series.option).yAxisIndex ?? 0)) const format = toRecord(toRecord(axis?.option).axisLabel).formatter diff --git a/apps/sim/lib/charts/theme.ts b/apps/sim/lib/charts/theme.ts index 86063bae94f..9846fb932d4 100644 --- a/apps/sim/lib/charts/theme.ts +++ b/apps/sim/lib/charts/theme.ts @@ -1,4 +1,5 @@ import { isRecordLike, toRecord } from '@sim/utils/object' +import type { ChartTonePalette } from '@/lib/charts/annotations' import { CHART_BAR_MAX_WIDTH, mapTooltipEntries } from '@/lib/charts/option' import { formatChartValue } from '@/lib/charts/summary' @@ -97,6 +98,23 @@ export function applyChartTooltipDefaults(option: Record) { return option } +/** Colours for highlights and thresholds; neutral matches the axis labels. */ +export function readChartTonePalette(element: HTMLElement): ChartTonePalette { + const styles = getComputedStyle(element) + const token = (name: string) => { + const value = styles.getPropertyValue(name).trim() + if (!value) throw new Error(`Missing chart theme token ${name}`) + return value + } + return { + tones: { + neutral: token('--text-tertiary'), + error: token('--text-error'), + info: token('--brand-blue'), + }, + } +} + /** Canvas cannot resolve CSS variables; read the same tokens as EMCN at its own container. */ export function readEmcnChartTheme(element: HTMLElement): Record { const styles = getComputedStyle(element) diff --git a/apps/sim/lib/charts/time-series.ts b/apps/sim/lib/charts/time-series.ts index b5ef62073a5..06b9b5e7cf2 100644 --- a/apps/sim/lib/charts/time-series.ts +++ b/apps/sim/lib/charts/time-series.ts @@ -30,12 +30,6 @@ export interface TimeSeriesInteractionOptions { onZoom?: (range: DashboardTimeRange) => void } -/** Cartesian charts with one horizontal time axis share dashboard interactions. */ -export function isTimeSeriesOption(option: Record): boolean { - const axes = Array.isArray(option.xAxis) ? option.xAxis : [option.xAxis] - return axes.length === 1 && toRecord(axes[0]).type === 'time' -} - /** Use ECharts' resolved encodings and colors, including transformed datasets. */ export function readTimeSeriesTooltip( params: unknown, diff --git a/apps/sim/lib/dashboards/embed-language.ts b/apps/sim/lib/dashboards/embed-language.ts new file mode 100644 index 00000000000..4780070c70b --- /dev/null +++ b/apps/sim/lib/dashboards/embed-language.ts @@ -0,0 +1,2 @@ +/** The fence language that renders a dashboard embed in markdown; kept apart from the parser. */ +export const DASHBOARD_EMBED_LANGUAGE = 'dashboard' diff --git a/apps/sim/lib/dashboards/spec.test.ts b/apps/sim/lib/dashboards/spec.test.ts index 2df96eb0eba..7907c851ec7 100644 --- a/apps/sim/lib/dashboards/spec.test.ts +++ b/apps/sim/lib/dashboards/spec.test.ts @@ -1,5 +1,9 @@ import { describe, expect, it } from 'vitest' -import { parseDashboardSpec, resolveDashboardSource } from '@/lib/dashboards/spec' +import { + parseDashboardEmbed, + parseDashboardSpec, + resolveDashboardSource, +} from '@/lib/dashboards/spec' import { dashboardRangeFromCalendar, parseDashboardCustomRange, @@ -161,3 +165,103 @@ describe('dashboard parse errors', () => { ) }) }) + +describe('fixed time ranges', () => { + it('accepts fixed instants and rejects reversed bounds', () => { + expect( + parseDashboardSpec( + 'title: T\ntime: {from: 2026-09-20T00:00:00Z, to: 2026-09-27T00:00:00Z}\nsource: {tableId: tbl_1}\nblocks:\n - stat: Total\n source: {aggregate: {n: {op: count}}}\n' + ).spec?.time + ).toEqual({ from: '2026-09-20T00:00:00Z', to: '2026-09-27T00:00:00Z' }) + expect( + parseDashboardSpec( + 'title: T\ntime: {from: 2026-09-27T00:00:00Z, to: 2026-09-20T00:00:00Z}\nsource: {tableId: tbl_1}\nblocks:\n - stat: Total\n source: {aggregate: {n: {op: count}}}\n' + ).error + ).toContain('The start must be earlier than the end') + }) +}) + +describe('dashboard embeds', () => { + it('parses data blocks and rows without a title', () => { + const parsed = parseDashboardEmbed( + 'time: 24h\nsource: {tableId: tbl_1}\nblocks:\n - row:\n - stat: Total\n source: {aggregate: {n: {op: count}}}\n - table: Rows\n source: {columns: [status]}\n' + ) + expect(parsed.error).toBeUndefined() + expect(parsed.spec?.title).toBeUndefined() + }) + + it('rejects text and tabs, which the surrounding document provides', () => { + expect(parseDashboardEmbed('source: {tableId: tbl_1}\nblocks:\n - text: hi\n').error).toBe( + 'blocks.0: expected a block with one of stat, chart, table, row; unknown key "text"' + ) + expect( + parseDashboardEmbed( + 'source: {tableId: tbl_1}\nblocks:\n - tabs: {A: [{table: Rows, source: {columns: [id]}}]}\n' + ).error + ).toContain('unknown key "tabs"') + }) + + it('caps the number of embedded blocks', () => { + const row = ` - row:\n${' - stat: Total\n source: {aggregate: {n: {op: count}}}\n'.repeat(6)}` + expect(parseDashboardEmbed(`source: {tableId: tbl_1}\nblocks:\n${row}${row}`).error).toBe( + 'Dashboard exceeds 12 blocks' + ) + }) +}) + +describe('highlights and thresholds', () => { + const chart = + 'blocks:\n - chart: Weekly\n source: {groupBy: [createdAt], bucket: week, aggregate: {n: {op: count}}}\n option: {xAxis: {type: time}, yAxis: {type: value}, series: [{type: line}]}\n' + + it('normalizes highlight instants to UTC ISO', () => { + const parsed = parseDashboardEmbed( + `source: {tableId: tbl_1}\nhighlights:\n - {from: 2026-08-21T00:00, to: 2026-09-07T00:00:00Z, label: Outage}\n - {at: 2026-09-10T14:00:00Z, tone: info}\n${chart}` + ) + expect(parsed.error).toBeUndefined() + expect(parsed.spec?.highlights).toEqual([ + { from: '2026-08-21T00:00:00.000Z', to: '2026-09-07T00:00:00.000Z', label: 'Outage' }, + { at: '2026-09-10T14:00:00.000Z', tone: 'info' }, + ]) + }) + + it('rejects reversed highlights and unknown tones', () => { + expect( + parseDashboardEmbed( + `source: {tableId: tbl_1}\nhighlights: [{from: 2026-09-07T00:00:00Z, to: 2026-08-21T00:00:00Z}]\n${chart}` + ).error + ).toContain('A highlight must start before it ends') + expect( + parseDashboardEmbed( + `source: {tableId: tbl_1}\nhighlights: [{at: 2026-09-07T00:00:00Z, tone: red}]\n${chart}` + ).error + ).toBe('highlights.0.tone: Invalid option: expected one of "neutral"|"error"|"info"') + }) + + it('rejects highlights or thresholds on a chart with hand-written marks', () => { + const marked = + 'blocks:\n - chart: Weekly\n source: {groupBy: [createdAt], bucket: week, aggregate: {n: {op: count}}}\n option: {xAxis: {type: time}, yAxis: {type: value}, series: [{type: line, markLine: {data: []}}]}\n' + const message = 'Use highlights and thresholds instead of markArea or markLine on the series' + expect( + parseDashboardEmbed( + `source: {tableId: tbl_1}\nhighlights: [{at: 2026-09-10T00:00:00Z}]\n${marked}` + ).error + ).toBe(message) + expect( + parseDashboardEmbed(`source: {tableId: tbl_1}\n${marked} thresholds: [{value: 5}]\n`).error + ).toBe(message) + expect(parseDashboardEmbed(`source: {tableId: tbl_1}\n${marked}`).error).toBeUndefined() + }) + + it('accepts thresholds on a chart with a value axis and rejects them without one', () => { + expect( + parseDashboardEmbed( + `source: {tableId: tbl_1}\n${chart} thresholds: [{value: 50, label: SLO}]\n` + ).error + ).toBeUndefined() + expect( + parseDashboardEmbed( + 'source: {tableId: tbl_1}\nblocks:\n - chart: No value axis\n source: {groupBy: [status], aggregate: {n: {op: count}}}\n option: {xAxis: {type: category}, yAxis: {type: category}, series: [{type: bar}]}\n thresholds: [{value: 5}]\n' + ).error + ).toBe('Thresholds require a value axis') + }) +}) diff --git a/apps/sim/lib/dashboards/spec.ts b/apps/sim/lib/dashboards/spec.ts index 3b21faa071a..e9c54557013 100644 --- a/apps/sim/lib/dashboards/spec.ts +++ b/apps/sim/lib/dashboards/spec.ts @@ -2,7 +2,19 @@ import { getErrorMessage } from '@sim/utils/errors' import { omit } from '@sim/utils/object' import { JSON_SCHEMA, load } from 'js-yaml' import { z } from 'zod' -import { parseChartSpec } from '@/lib/charts/spec' +import { + assertAnnotatable, + CHART_TONES, + type ChartHighlight, + type ChartThreshold, + valueAxisKey, +} from '@/lib/charts/annotations' +import { isTimeSeriesOption, parseChartSpec } from '@/lib/charts/spec' +import { + type DashboardTimeRange, + parseDashboardCustomRange, + parseDashboardInstant, +} from '@/lib/dashboards/time' import { measureYamlExpansion } from '@/lib/file-parsers/yaml-limits' import { type AnalyticsSelection, @@ -12,11 +24,28 @@ import { /** Dashboard YAML is bounded before parsing; writes report the same limit without decoding. */ export const MAX_DASHBOARD_SOURCE_BYTES = 128 * 1024 +const MAX_DASHBOARD_EMBED_SOURCE_BYTES = 32 * 1024 +const MAX_DASHBOARD_EMBED_BLOCKS = 12 export const DASHBOARD_SOURCE_TOO_LARGE = 'Dashboard source exceeds 128 KB' export const DASHBOARD_RANGES = ['1h', '24h', '7d', '30d', '90d'] as const export type DashboardRange = (typeof DASHBOARD_RANGES)[number] const rangeSchema = z.enum(DASHBOARD_RANGES) +/** A relative preset, or fixed instants for a chart that must not slide with the clock. */ +export type DashboardTime = DashboardRange | DashboardTimeRange +const timeSchema = z.union([ + rangeSchema, + z + .object({ from: z.string(), to: z.string() }) + .strict() + .superRefine((value, ctx) => { + try { + parseDashboardCustomRange(value.from, value.to) + } catch (error) { + ctx.addIssue({ code: 'custom', message: getErrorMessage(error, 'Invalid time range') }) + } + }), +]) const titleSchema = z.string().min(1).max(160) export const dashboardSourceSchema = analyticsSelectionSchema .extend({ @@ -42,6 +71,7 @@ export interface DashboardChart extends BlockBase { chart: string source?: DashboardSource option: Record + thresholds?: ChartThreshold[] } export interface DashboardTable extends BlockBase { table: string @@ -57,19 +87,81 @@ export type DashboardDataBlock = DashboardStat | DashboardChart | DashboardTable export type DashboardBlock = DashboardText | DashboardDataBlock | DashboardRow | DashboardTabs export interface DashboardSpec { title: string - time?: DashboardRange + time?: DashboardTime source?: DashboardSource + /** Time bands and instants drawn on every time-series chart. */ + highlights?: ChartHighlight[] blocks: DashboardBlock[] } +export type DashboardEmbedBlock = DashboardDataBlock | DashboardEmbedRow +export interface DashboardEmbedRow extends BlockBase { + row: DashboardEmbedBlock[] +} +/** A dashboard fragment inside markdown: the surrounding document supplies text and structure. */ +export interface DashboardEmbedSpec { + title?: string + time?: DashboardTime + source?: DashboardSource + highlights?: ChartHighlight[] + blocks: DashboardEmbedBlock[] +} + +const toneSchema = z.enum(CHART_TONES) +const annotationLabelSchema = z.string().min(1).max(80) +/** Instants are normalized to ISO with `Z` so the canvas never reads them as local time. */ +const instantSchema = z.string().transform((value, ctx) => { + try { + return parseDashboardInstant(value) + } catch (error) { + ctx.addIssue({ code: 'custom', message: getErrorMessage(error, 'Invalid time') }) + return z.NEVER + } +}) +const highlightSchema = z.union([ + z + .object({ + from: instantSchema, + to: instantSchema, + label: annotationLabelSchema.optional(), + tone: toneSchema.optional(), + }) + .strict() + .refine((value) => value.from < value.to, 'A highlight must start before it ends'), + z + .object({ + at: instantSchema, + label: annotationLabelSchema.optional(), + tone: toneSchema.optional(), + }) + .strict(), +]) +const highlightsSchema = z.array(highlightSchema).min(1).max(20) +const thresholdSchema = z + .object({ + value: z.number().finite(), + label: annotationLabelSchema.optional(), + tone: toneSchema.optional(), + }) + .strict() const base = { flex: z.number().int().min(1).max(12).optional() } const source = { ...base, source: dashboardSourceSchema.optional() } +const dataBlockSchemas = [ + z.object({ ...source, stat: titleSchema, unit: z.string().max(24).optional() }).strict(), + z + .object({ + ...source, + chart: titleSchema, + option: z.record(z.string(), z.unknown()), + thresholds: z.array(thresholdSchema).min(1).max(5).optional(), + }) + .strict(), + z.object({ ...source, table: titleSchema }).strict(), +] as const const blockSchema: z.ZodType = z.lazy(() => z.union([ z.object({ ...base, text: z.string().min(1).max(10000) }).strict(), - z.object({ ...source, stat: titleSchema, unit: z.string().max(24).optional() }).strict(), - z.object({ ...source, chart: titleSchema, option: z.record(z.string(), z.unknown()) }).strict(), - z.object({ ...source, table: titleSchema }).strict(), + ...dataBlockSchemas, z.object({ ...base, row: z.array(blockSchema).min(1).max(12) }).strict(), z .object({ @@ -87,11 +179,27 @@ const blockSchema: z.ZodType = z.lazy(() => const dashboardSchema: z.ZodType = z .object({ title: titleSchema, - time: rangeSchema.optional(), + time: timeSchema.optional(), source: dashboardSourceSchema.optional(), + highlights: highlightsSchema.optional(), blocks: z.array(blockSchema).min(1).max(48), }) .strict() +const embedBlockSchema: z.ZodType = z.lazy(() => + z.union([ + ...dataBlockSchemas, + z.object({ ...base, row: z.array(embedBlockSchema).min(1).max(12) }).strict(), + ]) +) +const dashboardEmbedSchema: z.ZodType = z + .object({ + title: titleSchema.optional(), + time: timeSchema.optional(), + source: dashboardSourceSchema.optional(), + highlights: highlightsSchema.optional(), + blocks: z.array(embedBlockSchema).min(1).max(MAX_DASHBOARD_EMBED_BLOCKS), + }) + .strict() function queryMode(source: DashboardSource | undefined): 'columns' | 'aggregate' | undefined { return source?.columns ? 'columns' : source?.aggregate ? 'aggregate' : undefined @@ -127,6 +235,7 @@ export function resolveDashboardSource( } const BLOCK_KINDS = ['text', 'stat', 'chart', 'table', 'row', 'tabs'] as const +const EMBED_BLOCK_KINDS = ['stat', 'chart', 'table', 'row'] as const /** * One line per issue, `path: message`. A block reports the errors of the kind it declares; a @@ -135,6 +244,7 @@ const BLOCK_KINDS = ['text', 'stat', 'chart', 'table', 'row', 'tabs'] as const */ function describeSchemaIssues( issues: readonly z.core.$ZodIssue[], + kinds: readonly string[], prefix: PropertyKey[] = [] ): string[] { return issues.flatMap((issue) => { @@ -148,17 +258,17 @@ function describeSchemaIssues( (entry) => entry.code === 'invalid_type' && entry.path.length === 1 && - BLOCK_KINDS.some((kind) => kind === entry.path[0]) + kinds.some((kind) => kind === entry.path[0]) ) ) const isBlock = issue.errors.some((branch) => - branch.some((entry) => entry.path.length === 1 && entry.path[0] === 'text') + branch.some((entry) => entry.path.length === 1 && entry.path[0] === kinds[0]) ) if (!isBlock || declared) { const branch = - declared ?? + (isBlock ? declared : undefined) ?? issue.errors.reduce((fewest, next) => (next.length < fewest.length ? next : fewest)) - return describeSchemaIssues(branch, path) + return describeSchemaIssues(branch, kinds, path) } const rejected = issue.errors.map( (branch) => @@ -171,71 +281,129 @@ function describeSchemaIssues( const unknown = [...rejected[0]].filter((key) => rejected.every((keys) => keys.has(key))) const keys = unknown.map((key) => `"${key}"`).join(', ') return [ - `${at}expected a block with one of ${BLOCK_KINDS.join(', ')}${keys ? `; unknown key ${keys}` : ''}`, + `${at}expected a block with one of ${kinds.join(', ')}${keys ? `; unknown key ${keys}` : ''}`, ] }) } -/** Strict YAML validation, including expansion limits before recursive parsing. */ -export function parseDashboardSpec( - content: string -): { spec: DashboardSpec; error?: never } | { error: string; spec?: never } { - try { - if (new TextEncoder().encode(content).byteLength > MAX_DASHBOARD_SOURCE_BYTES) - throw new Error(DASHBOARD_SOURCE_TOO_LARGE) - const raw: unknown = load(content, { schema: JSON_SCHEMA }) - const measured = measureYamlExpansion(raw, { - maxNodes: 10000, - maxDepth: 24, - maxSerializedBytes: 256 * 1024, - }) - if (!measured.within) throw new Error(measured.reason) - const parsed = dashboardSchema.safeParse(raw) - if (!parsed.success) throw new Error(describeSchemaIssues(parsed.error.issues).join('\n')) - const spec = parsed.data - let count = 0 - const visit = (blocks: DashboardBlock[], depth: number): void => { - if (depth > 4) throw new Error('Dashboard layout exceeds 4 levels') - for (const block of blocks) { - if (++count > 48) throw new Error('Dashboard exceeds 48 blocks') - if ('row' in block) visit(block.row, depth + 1) - else if ('tabs' in block) - Object.values(block.tabs).forEach((children) => visit(children, depth + 1)) - else if (!('text' in block)) { - const resolved = resolveDashboardSource(spec.source, block.source) - const { tableId: _tableId, range: _range, ...selection } = resolved - const parsed = analyticsQuerySchema.safeParse({ - ...selection, - from: '2026-01-01T00:00:00Z', - to: '2026-01-02T00:00:00Z', - }) - if (!parsed.success) - throw new Error(parsed.error.issues.map((issue) => issue.message).join('; ')) - if ( - 'stat' in block && - (!selection.aggregate || - Object.keys(selection.aggregate).length !== 1 || - selection.groupBy) - ) { - throw new Error(`Stat "${block.stat}" requires exactly one aggregate and no groupBy`) - } - if ('chart' in block) { - const chart = parseChartSpec( - JSON.stringify({ schema_version: 1, option: block.option }) - ) - if (!chart.spec) throw new Error(chart.error) - block.option = chart.spec.option - } +type ParseResult = { spec: T; error?: never } | { error: string; spec?: never } + +/** Bounded YAML decoding: byte size and expansion limits before recursive schema parsing. */ +function loadBoundedYaml(content: string, maxBytes: number, tooLarge: string): unknown { + if (new TextEncoder().encode(content).byteLength > maxBytes) throw new Error(tooLarge) + const raw: unknown = load(content, { schema: JSON_SCHEMA }) + const measured = measureYamlExpansion(raw, { + maxNodes: 10000, + maxDepth: 24, + maxSerializedBytes: 256 * 1024, + }) + if (!measured.within) throw new Error(measured.reason) + return raw +} + +/** + * Validates each data block's resolved query and sanitizes chart options in place. Limits nesting + * and the total block count, including nested rows and tabs. + */ +function validateDashboardBlocks( + blocks: DashboardBlock[], + defaults: DashboardSource | undefined, + maxBlocks: number, + highlights: ChartHighlight[] | undefined +): void { + let count = 0 + const visit = (children: DashboardBlock[], depth: number): void => { + if (depth > 4) throw new Error('Dashboard layout exceeds 4 levels') + for (const block of children) { + if (++count > maxBlocks) throw new Error(`Dashboard exceeds ${maxBlocks} blocks`) + if ('row' in block) visit(block.row, depth + 1) + else if ('tabs' in block) Object.values(block.tabs).forEach((tab) => visit(tab, depth + 1)) + else if (!('text' in block)) { + const resolved = resolveDashboardSource(defaults, block.source) + const { tableId: _tableId, range: _range, ...selection } = resolved + const parsed = analyticsQuerySchema.safeParse({ + ...selection, + from: '2026-01-01T00:00:00Z', + to: '2026-01-02T00:00:00Z', + }) + if (!parsed.success) + throw new Error(parsed.error.issues.map((issue) => issue.message).join('; ')) + if ( + 'stat' in block && + (!selection.aggregate || + Object.keys(selection.aggregate).length !== 1 || + selection.groupBy) + ) { + throw new Error(`Stat "${block.stat}" requires exactly one aggregate and no groupBy`) + } + if ('chart' in block) { + const chart = parseChartSpec(JSON.stringify({ schema_version: 1, option: block.option })) + if (!chart.spec) throw new Error(chart.error) + block.option = chart.spec.option + if (block.thresholds) valueAxisKey(block.option) + if (block.thresholds || (highlights && isTimeSeriesOption(block.option))) + assertAnnotatable(block.option) } } } - visit(spec.blocks, 1) - return { spec } + } + visit(blocks, 1) +} + +/** Strict YAML validation, including expansion limits before recursive parsing. */ +export function parseDashboardSpec(content: string): ParseResult { + try { + const raw = loadBoundedYaml(content, MAX_DASHBOARD_SOURCE_BYTES, DASHBOARD_SOURCE_TOO_LARGE) + const parsed = dashboardSchema.safeParse(raw) + if (!parsed.success) + throw new Error(describeSchemaIssues(parsed.error.issues, BLOCK_KINDS).join('\n')) + validateDashboardBlocks(parsed.data.blocks, parsed.data.source, 48, parsed.data.highlights) + return { spec: parsed.data } } catch (error) { return { error: getErrorMessage(error, 'Invalid dashboard') } } } +/** Parses a markdown ```dashboard fence body with the dashboard grammar, minus text and tabs. */ +export function parseDashboardEmbed(content: string): ParseResult { + try { + const raw = loadBoundedYaml( + content, + MAX_DASHBOARD_EMBED_SOURCE_BYTES, + 'Dashboard embed exceeds 32 KB' + ) + const parsed = dashboardEmbedSchema.safeParse(raw) + if (!parsed.success) + throw new Error(describeSchemaIssues(parsed.error.issues, EMBED_BLOCK_KINDS).join('\n')) + validateDashboardBlocks( + parsed.data.blocks, + parsed.data.source, + MAX_DASHBOARD_EMBED_BLOCKS, + parsed.data.highlights + ) + return { spec: parsed.data } + } catch (error) { + return { error: getErrorMessage(error, 'Invalid dashboard embed') } + } +} + +/** Every table a dashboard's data blocks read, for scoping refresh and fetch state. */ +export function dashboardTableIds( + blocks: readonly DashboardBlock[], + defaults: DashboardSource | undefined +): Set { + const ids = new Set() + const visit = (children: readonly DashboardBlock[]) => { + for (const block of children) { + if ('row' in block) visit(block.row) + else if ('tabs' in block) Object.values(block.tabs).forEach(visit) + else if (!('text' in block)) ids.add(resolveDashboardSource(defaults, block.source).tableId) + } + } + visit(blocks) + return ids +} + export function dashboardSelection(source: ResolvedDashboardSource): AnalyticsSelection { const { tableId: _tableId, range: _range, ...selection } = source return selection diff --git a/apps/sim/lib/dashboards/time.ts b/apps/sim/lib/dashboards/time.ts index 0446823ab3b..d3a8dfd9a83 100644 --- a/apps/sim/lib/dashboards/time.ts +++ b/apps/sim/lib/dashboards/time.ts @@ -17,18 +17,20 @@ export function relativeDashboardRange(range: DashboardRange, now: number): Dash return { from: new Date(now - RANGE_MS[range]).toISOString(), to: new Date(now).toISOString() } } +/** A UTC instant written with or without a trailing `Z`, normalized to ISO. */ +export function parseDashboardInstant(value: string): string { + const normalized = value.endsWith('Z') ? value.slice(0, -1) : value + if (!/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d{3})?)?$/.test(normalized)) + throw new Error('Choose a complete start and end date') + const date = new Date(`${normalized}Z`) + if (!Number.isFinite(date.getTime()) || !date.toISOString().startsWith(normalized)) + throw new Error('Invalid UTC date and time') + return date.toISOString() +} + /** URL bounds are instants; legacy offset-free URLs continue to mean UTC. */ export function parseDashboardCustomRange(from: string, to: string): DashboardTimeRange { - const instant = (value: string) => { - const normalized = value.endsWith('Z') ? value.slice(0, -1) : value - if (!/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d{3})?)?$/.test(normalized)) - throw new Error('Choose a complete start and end date') - const date = new Date(`${normalized}Z`) - if (!Number.isFinite(date.getTime()) || !date.toISOString().startsWith(normalized)) - throw new Error('Invalid UTC date and time') - return date.toISOString() - } - const range = { from: instant(from), to: instant(to) } + const range = { from: parseDashboardInstant(from), to: parseDashboardInstant(to) } if (range.from >= range.to) throw new Error('The start must be earlier than the end') return range } @@ -73,6 +75,32 @@ export function dashboardZoomRange( return { from: new Date(from).toISOString(), to: new Date(to).toISOString() } } +export const DASHBOARD_RANGE_LABELS: Record = { + '1h': 'Last hour', + '24h': 'Last 24 hours', + '7d': 'Last 7 days', + '30d': 'Last 30 days', + '90d': 'Last 90 days', +} + +/** Compact dates for a range; times appear only when it starts and ends on the same day. */ +export function dashboardRangeText(range: DashboardTimeRange, timeZone: string): string { + const from = new Date(range.from) + const to = new Date(Date.parse(range.to) - 1) + const fromLocal = zonedWallClock(from, timeZone) + const toLocal = zonedWallClock(to, timeZone) + const sameDay = fromLocal.slice(0, 10) === toLocal.slice(0, 10) + return new Intl.DateTimeFormat('en-US', { + timeZone, + month: 'short', + day: 'numeric', + year: fromLocal.slice(0, 4) === toLocal.slice(0, 4) ? undefined : 'numeric', + hour: sameDay ? '2-digit' : undefined, + minute: sameDay ? '2-digit' : undefined, + hourCycle: 'h23', + }).formatRange(from, to) +} + export function dashboardTimeLabel(instant: number | string, timeZone: string): string { return new Intl.DateTimeFormat('en-US', { timeZone, diff --git a/scripts/check-tool-registry-boundary.baseline.json b/scripts/check-tool-registry-boundary.baseline.json index d3dfa6f67d5..b3b2fc3e2c9 100644 --- a/scripts/check-tool-registry-boundary.baseline.json +++ b/scripts/check-tool-registry-boundary.baseline.json @@ -442,16 +442,16 @@ } }, "app/workspace/[workspaceId]/skills/[skillId]/page.tsx": { - "modules": 1421, + "modules": 1457, "gateways": { - "apps/sim/app/workspace/[workspaceId]/skills/[skillId]/skill-detail.tsx": 1420, + "apps/sim/app/workspace/[workspaceId]/skills/[skillId]/skill-detail.tsx": 1456, "apps/sim/triggers/registry.ts": 528, - "apps/sim/blocks/registry.ts": 376, - "apps/sim/blocks/registry-maps.ts": 374, - "apps/sim/app/workspace/[workspaceId]/skills/components/skill-fields/index.ts": 232, - "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx": 229, - "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/editor-extensions.ts": 93, - "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/mention/index.ts": 82 + "apps/sim/blocks/registry.ts": 374, + "apps/sim/blocks/registry-maps.ts": 372, + "apps/sim/app/workspace/[workspaceId]/skills/components/skill-fields/index.ts": 268, + "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx": 265, + "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/editor-extensions.ts": 117, + "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/mention/index.ts": 92 } }, "app/workspace/[workspaceId]/skills/error.tsx": { @@ -459,16 +459,16 @@ "gateways": {} }, "app/workspace/[workspaceId]/skills/new/page.tsx": { - "modules": 1419, + "modules": 1455, "gateways": { - "apps/sim/app/workspace/[workspaceId]/skills/new/skill-create.tsx": 1418, + "apps/sim/app/workspace/[workspaceId]/skills/new/skill-create.tsx": 1454, "apps/sim/triggers/registry.ts": 528, - "apps/sim/blocks/registry.ts": 376, - "apps/sim/blocks/registry-maps.ts": 374, - "apps/sim/app/workspace/[workspaceId]/skills/components/skill-fields/index.ts": 232, - "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx": 229, - "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/editor-extensions.ts": 93, - "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/mention/index.ts": 82 + "apps/sim/blocks/registry.ts": 374, + "apps/sim/blocks/registry-maps.ts": 372, + "apps/sim/app/workspace/[workspaceId]/skills/components/skill-fields/index.ts": 268, + "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/rich-markdown-field.tsx": 265, + "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/editor-extensions.ts": 117, + "apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/mention/index.ts": 92 } }, "app/workspace/[workspaceId]/skills/page.tsx": { From b6a598e71307b62259a9e1b7ae8bb851a219e1ca Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 11:57:57 -0700 Subject: [PATCH 07/31] fix(accounts): scope organization OAuth outbound requests (#8533) * fix(accounts): scope organization OAuth outbound requests * fix(accounts): preserve outbound ownership during reconnect --- ...ganization-account-outbound.integration.ts | 275 ++++++++++++++++++ .../application/organization-accounts.ts | 71 ++--- .../personal-organization-accounts.ts | 31 +- 3 files changed, 329 insertions(+), 48 deletions(-) create mode 100644 apps/sim/lib/credential-groups/__integration__/organization-account-outbound.integration.ts diff --git a/apps/sim/lib/credential-groups/__integration__/organization-account-outbound.integration.ts b/apps/sim/lib/credential-groups/__integration__/organization-account-outbound.integration.ts new file mode 100644 index 00000000000..503a432e2ff --- /dev/null +++ b/apps/sim/lib/credential-groups/__integration__/organization-account-outbound.integration.ts @@ -0,0 +1,275 @@ +import { createHash } from 'node:crypto' +import { createServer } from 'node:http' +import { db } from '@sim/db' +import { credential, credentialGroupEnrollment, member, organization, user } from '@sim/db/schema' +import { readTestRedisUrl } from '@sim/db/testing/test-infrastructure' +import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' +import { generateId } from '@sim/utils/id' +import { toRecord } from '@sim/utils/object' +import { eq, inArray } from 'drizzle-orm' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' +import { env } from '@/lib/core/config/env' +import { + resolveCurrentOutboundRoute, + runWithOutboundOrganization, +} from '@/lib/core/network/context.server' +import { startOrganizationAccountConnection } from '@/lib/credential-groups/application/organization-accounts' +import { reconnectPersonalOrganizationAccount } from '@/lib/credential-groups/application/personal-organization-accounts' +import { createManagedMcpConnector } from '@/lib/credential-groups/managed-mcp-service' +import { consumeCredentialGroupMcpOAuthAttempt } from '@/lib/credential-groups/mcp-oauth-state' +import { createViewerCredentialGroupEnrollment } from '@/lib/credential-groups/self-enrollment' +import { ensureWorkspaceAccountsGroup } from '@/lib/credential-groups/service' +import { encryptManagedMcpTokens } from '@/lib/credentials/managed-mcp' +import * as oauth from '@/lib/mcp/oauth/auth' +import { createSsrfGuardedMcpFetch } from '@/lib/mcp/pinned-fetch' + +const RESOURCE = 'https://mcp.lucid.app/mcp/readonly' +const ISSUER = 'https://oauth.fixture.test' +const owner = generateId() +const blockedOwner = generateId() +const outsider = generateId() +const directOrg = generateId() +const blockedOrg = generateId() +const servers = new Map() +const grants = new Map() +const requests: string[] = [] + +/** Real OAuth discovery and registration over a socket; only the remote destination is replaced. */ +const providerServer = createServer(async (request, response) => { + const path = request.url ?? '/' + requests.push(path) + response.setHeader('content-type', 'application/json') + if (path.includes('oauth-protected-resource')) { + response.end(JSON.stringify({ resource: RESOURCE, authorization_servers: [ISSUER] })) + } else if (path.includes('oauth-authorization-server') || path.includes('openid-configuration')) { + response.end( + JSON.stringify({ + issuer: ISSUER, + authorization_endpoint: `${ISSUER}/authorize`, + token_endpoint: `${ISSUER}/token`, + registration_endpoint: `${ISSUER}/register`, + response_types_supported: ['code'], + code_challenge_methods_supported: ['S256'], + token_endpoint_auth_methods_supported: ['none'], + }) + ) + } else if (path === '/register' && request.method === 'POST') { + const chunks: Buffer[] = [] + for await (const chunk of request) chunks.push(Buffer.from(chunk)) + const metadata: unknown = JSON.parse(Buffer.concat(chunks).toString()) + response.writeHead(201).end( + JSON.stringify({ + ...toRecord(metadata), + client_id: 'fixture-dynamic-client', + }) + ) + } else { + response.writeHead(404).end(JSON.stringify({ error: 'unknown fixture endpoint' })) + } +}) + +beforeAll(async () => { + Object.assign(env, { + REDIS_URL: readTestRedisUrl(), + OUTBOUND_ROUTING_SOURCE: 'env', + OUTBOUND_ROUTING_CONFIG: JSON.stringify({ + schemaVersion: 1, + revision: 'account-oauth-fixture', + organizations: { [directOrg]: { kind: 'direct' }, [blockedOrg]: { kind: 'blocked' } }, + }), + OUTBOUND_GATEWAYS: JSON.stringify({ + direct: { + organizationId: directOrg, + url: 'https://direct.fixture.test/', + credentialId: 'direct', + }, + blocked: { + organizationId: blockedOrg, + url: 'https://blocked.fixture.test/', + credentialId: 'blocked', + }, + }), + OUTBOUND_GATEWAY_CREDENTIALS: JSON.stringify({ + direct: { token: 'd'.repeat(32) }, + blocked: { token: 'b'.repeat(32) }, + }), + }) + await db.insert(user).values( + [owner, blockedOwner, outsider].map((id) => ({ + id, + name: 'OAuth fixture', + email: `${id}@fixture.test`, + emailVerified: true, + createdAt: new Date(), + updatedAt: new Date(), + })) + ) + for (const organizationId of [directOrg, blockedOrg]) { + const userId = organizationId === directOrg ? owner : blockedOwner + await db + .insert(organization) + .values({ id: organizationId, name: 'OAuth fixture', slug: organizationId }) + await db.insert(member).values({ id: generateId(), organizationId, userId, role: 'owner' }) + const group = await ensureWorkspaceAccountsGroup( + { kind: 'organization', organizationId }, + userId + ) + const { mcpServer } = await db.transaction((tx) => + createManagedMcpConnector( + { + organizationId, + credentialGroupId: group.id, + userId, + validated: { input: { connectorId: 'lucid' }, url: RESOURCE }, + }, + tx + ) + ) + servers.set(organizationId, mcpServer.id) + const { enrollment } = await createViewerCredentialGroupEnrollment({ + organizationId, + credentialGroupId: group.id, + userId, + }) + const credentialId = `mcp-cg-${generateId()}` + await db.insert(credential).values({ + id: credentialId, + organizationId, + type: 'managed_mcp', + displayName: 'OAuth fixture', + grantedAt: new Date(), + credentialGroupEnrollmentId: enrollment.id, + mcpServerId: mcpServer.id, + managedOauthStatus: 'active', + mcpTools: [], + encryptedOauthTokenSet: await encryptManagedMcpTokens({ + access_token: 'fixture-access', + token_type: 'Bearer', + }), + }) + grants.set(organizationId, { credentialId, enrollmentId: enrollment.id }) + } + await new Promise((resolve) => providerServer.listen(0, '127.0.0.1', resolve)) + const address = providerServer.address() + if (!address || typeof address === 'string') throw new Error('OAuth fixture did not bind') + const origin = `http://127.0.0.1:${address.port}` + const guardedFetch = createSsrfGuardedMcpFetch({ serverUrl: origin }) + const authenticate = oauth.mcpAuthGuarded + vi.spyOn(oauth, 'mcpAuthGuarded').mockImplementation((provider, options) => + authenticate(provider, { + ...options, + fetchFn: (input, init) => { + const remote = new URL(input instanceof Request ? input.url : input) + if (![new URL(RESOURCE).origin, ISSUER].includes(remote.origin)) { + throw new Error('Unexpected OAuth fixture origin') + } + return guardedFetch(new URL(`${remote.pathname}${remote.search}`, origin), init) + }, + }) + ) +}) + +afterAll(async () => { + if (providerServer.listening) + await new Promise((resolve, reject) => { + providerServer.close((error) => (error ? reject(error) : resolve())) + providerServer.closeAllConnections() + }) + await db.delete(organization).where(inArray(organization.id, [directOrg, blockedOrg])) + await db.delete(user).where(inArray(user.id, [owner, blockedOwner, outsider])) +}) + +function connect( + organizationId: string, + userId = owner, + mcpServerId = servers.get(organizationId)! +) { + return startOrganizationAccountConnection.execute({ + principal: createSessionPrincipal({ userId, sessionId: generateId() }), + input: { + organizationId, + mcpServerId, + oauthCompletionId: generateId(), + returnTo: 'integrations', + }, + }) +} + +function reconnect(organizationId: string, userId = owner) { + return reconnectPersonalOrganizationAccount.execute({ + principal: createSessionPrincipal({ userId, sessionId: generateId() }), + input: { + credentialId: grants.get(organizationId)!.credentialId, + oauthCompletionId: generateId(), + }, + }) +} + +async function verifyAuthorization(start: typeof connect | typeof reconnect) { + const result = await start(directOrg) + if (!result.authorizationUrl) throw new Error('OAuth authorization URL is missing') + const authorization = new URL(result.authorizationUrl) + expect(`${authorization.origin}${authorization.pathname}`).toBe(`${ISSUER}/authorize`) + expect(authorization.searchParams.get('client_id')).toBe('fixture-dynamic-client') + expect(authorization.searchParams.get('code_challenge_method')).toBe('S256') + const state = authorization.searchParams.get('state')! + const attempt = await consumeCredentialGroupMcpOAuthAttempt(state) + expect(attempt).toMatchObject({ + organizationId: directOrg, + userId: owner, + mcpServerId: servers.get(directOrg), + returnTo: 'integrations', + }) + expect(authorization.searchParams.get('code_challenge')).toBe( + createHash('sha256').update(attempt!.codeVerifier).digest('base64url') + ) + expect(await consumeCredentialGroupMcpOAuthAttempt(state)).toBeNull() +} + +describe.each([ + { name: 'Connect', start: connect }, + { name: 'Reconnect', start: reconnect }, +])('$name account OAuth outbound ownership', ({ start }) => { + it('starts dynamic OAuth with a bound single-use attempt when no ambient scope exists', async () => { + await verifyAuthorization(start) + await expect(resolveCurrentOutboundRoute()).rejects.toMatchObject({ code: 'MISSING_SCOPE' }) + }) + it('uses authorized ownership instead of an ambient blocked organization and restores the outer scope', async () => { + await runWithOutboundOrganization(blockedOrg, async () => { + await verifyAuthorization(start) + await expect(resolveCurrentOutboundRoute()).rejects.toMatchObject({ code: 'ROUTE_BLOCKED' }) + }) + }) + it('does not bypass an organization block through ambient platform scope', async () => { + const before = requests.length + await runWithOutboundOrganization(null, async () => { + await expect(start(blockedOrg, blockedOwner)).rejects.toMatchObject({ + code: 'ROUTE_BLOCKED', + }) + expect(await resolveCurrentOutboundRoute()).toEqual({ kind: 'direct' }) + }) + expect(requests.length).toBe(before) + }) +}) + +describe('Account authorization before outbound OAuth', () => { + it('denies nonmembers and cross-organization providers before contacting OAuth', async () => { + const before = requests.length + await expect(connect(directOrg, outsider)).rejects.toMatchObject({ code: 'not_found' }) + await expect(connect(directOrg, owner, servers.get(blockedOrg))).rejects.toMatchObject({ + code: 'not_found', + }) + expect(requests.length).toBe(before) + }) + it('denies another contributor and a revoked enrollment during reconnect', async () => { + const before = requests.length + await expect(reconnect(directOrg, outsider)).rejects.toMatchObject({ code: 'not_found' }) + const enrollmentId = grants.get(directOrg)!.enrollmentId + await db + .update(credentialGroupEnrollment) + .set({ status: 'revoked' }) + .where(eq(credentialGroupEnrollment.id, enrollmentId)) + await expect(reconnect(directOrg)).rejects.toMatchObject({ code: 'forbidden' }) + expect(requests.length).toBe(before) + }) +}) diff --git a/apps/sim/lib/credential-groups/application/organization-accounts.ts b/apps/sim/lib/credential-groups/application/organization-accounts.ts index a5dcfe6dec2..3cfe7c5e6e0 100644 --- a/apps/sim/lib/credential-groups/application/organization-accounts.ts +++ b/apps/sim/lib/credential-groups/application/organization-accounts.ts @@ -12,6 +12,7 @@ import { defineOrganizationOperation, type OrganizationOperation, } from '@/lib/core/application/organization-operation' +import { runWithOutboundOrganization } from '@/lib/core/network/context.server' import { OrchestrationError } from '@/lib/core/orchestration/types' import { validateUpdateCredentialGroupInput } from '@/lib/credential-groups/application/validation' import { loadScopedAccountsCredentialListContext } from '@/lib/credential-groups/credentials' @@ -119,41 +120,43 @@ export function defineOrganizationAccountsUseCase< if (group) await requireOrganizationAccountsSetup(context.organizationId, group.credentialGroupId) } - const result = await definition.execute({ input, context }).catch((error: unknown) => { - if (error instanceof ManagedMcpConnectorError) - throw new OrchestrationError( - error.code === 'bad_gateway' ? 'internal' : error.code, - error.message - ) - if (error instanceof CredentialGroupEnrollmentError) - throw new OrchestrationError( - error.status === 404 - ? 'not_found' - : error.status === 409 - ? 'conflict' - : error.status === 400 - ? 'validation' - : 'internal', - error.message - ) - throw error - }) - const audit = definition.projectAudit?.(result) - if (audit) - recordAudit({ - ...audit, - actorId: context.userId, - action: AuditAction.CREDENTIAL_GROUP_UPDATED, - resourceType: AuditResourceType.CREDENTIAL_GROUP, - metadata: { - organizationId: context.organizationId, - operation: definition.operation.id, - actor: resolvePrincipalAuditAttribution(principal).actor, - }, - request, + return runWithOutboundOrganization(context.organizationId, async () => { + const result = await definition.execute({ input, context }).catch((error: unknown) => { + if (error instanceof ManagedMcpConnectorError) + throw new OrchestrationError( + error.code === 'bad_gateway' ? 'internal' : error.code, + error.message + ) + if (error instanceof CredentialGroupEnrollmentError) + throw new OrchestrationError( + error.status === 404 + ? 'not_found' + : error.status === 409 + ? 'conflict' + : error.status === 400 + ? 'validation' + : 'internal', + error.message + ) + throw error }) - await definition.afterSuccess?.({ result, context }) - return result + const audit = definition.projectAudit?.(result) + if (audit) + recordAudit({ + ...audit, + actorId: context.userId, + action: AuditAction.CREDENTIAL_GROUP_UPDATED, + resourceType: AuditResourceType.CREDENTIAL_GROUP, + metadata: { + organizationId: context.organizationId, + operation: definition.operation.id, + actor: resolvePrincipalAuditAttribution(principal).actor, + }, + request, + }) + await definition.afterSuccess?.({ result, context }) + return result + }) }, } } diff --git a/apps/sim/lib/credential-groups/application/personal-organization-accounts.ts b/apps/sim/lib/credential-groups/application/personal-organization-accounts.ts index e600968903c..ade2baabfa7 100644 --- a/apps/sim/lib/credential-groups/application/personal-organization-accounts.ts +++ b/apps/sim/lib/credential-groups/application/personal-organization-accounts.ts @@ -12,6 +12,7 @@ import { sha256Hex } from '@sim/security/hash' import { generateId } from '@sim/utils/id' import { and, asc, eq, gt, inArray, isNotNull } from 'drizzle-orm' import { defineOperation } from '@/lib/core/application' +import { runWithOutboundOrganization } from '@/lib/core/network/context.server' import { OrchestrationError } from '@/lib/core/orchestration/types' import { sameResourceScopeCondition } from '@/lib/core/resource-scope.server' import { createCredentialGroupOAuthStartUrl } from '@/lib/credential-groups/enrollment-links' @@ -156,20 +157,22 @@ export const reconnectPersonalOrganizationAccount = defineAuthorizedCredentialUs completionId: input.oauthCompletionId, returnTo: 'integrations' as const, } - if (account.type === 'managed_oauth' && account.optionId) { - return startViewerCredentialGroupOAuth({ - ...connection, - optionId: account.optionId, - connectionIntent: { kind: 'reconnect', credentialId: account.credentialId }, - }) - } - if (account.type === 'managed_mcp' && account.mcpServerId) { - return startViewerCredentialGroupMcpOAuth({ - ...connection, - mcpServerId: account.mcpServerId, - }) - } - throw new OrchestrationError('not_found', 'This account provider is no longer available') + return runWithOutboundOrganization(account.organizationId, () => { + if (account.type === 'managed_oauth' && account.optionId) { + return startViewerCredentialGroupOAuth({ + ...connection, + optionId: account.optionId, + connectionIntent: { kind: 'reconnect', credentialId: account.credentialId }, + }) + } + if (account.type === 'managed_mcp' && account.mcpServerId) { + return startViewerCredentialGroupMcpOAuth({ + ...connection, + mcpServerId: account.mcpServerId, + }) + } + throw new OrchestrationError('not_found', 'This account provider is no longer available') + }) } const { invitationLink } = await createViewerCredentialGroupEnrollment({ organizationId: account.organizationId, From 7dad107e1b9f5d30828627f0c844db1456baf2d7 Mon Sep 17 00:00:00 2001 From: Theodore Li Date: Thu, 1 Oct 2026 12:22:08 -0700 Subject: [PATCH 08/31] fix(wiza): surface Wiza's nested error messages (#8532) * fix(wiza): surface Wiza's nested error messages * fix(wiza): keep plain-text errors and parse reveal polling failures * fix(wiza): only accept trimmed string messages in error extractor --- apps/sim/tools/error-extractors.test.ts | 42 +++++++++++++++++++++++ apps/sim/tools/error-extractors.ts | 15 ++++++++ apps/sim/tools/wiza/company_enrichment.ts | 2 ++ apps/sim/tools/wiza/get_credits.ts | 2 ++ apps/sim/tools/wiza/individual_reveal.ts | 17 ++++++++- apps/sim/tools/wiza/prospect_search.ts | 2 ++ 6 files changed, 79 insertions(+), 1 deletion(-) diff --git a/apps/sim/tools/error-extractors.test.ts b/apps/sim/tools/error-extractors.test.ts index 29f2315bdf2..cc05f3f78ce 100644 --- a/apps/sim/tools/error-extractors.test.ts +++ b/apps/sim/tools/error-extractors.test.ts @@ -362,4 +362,46 @@ describe('Error Extractors', () => { ) }) }) + + describe('wiza-errors', () => { + it('extracts the message nested under status', () => { + const errorInfo: ErrorInfo = { + status: 400, + statusText: 'Bad Request', + data: { status: { code: 400, message: 'The size parameter is not allowed.' } }, + } + + expect(extractErrorMessage(errorInfo, ErrorExtractorId.WIZA_ERRORS)).toBe( + 'The size parameter is not allowed.' + ) + }) + + it('falls back to a top-level message', () => { + const errorInfo: ErrorInfo = { status: 401, data: { message: 'Unauthorized' } } + + expect(extractErrorMessage(errorInfo, ErrorExtractorId.WIZA_ERRORS)).toBe('Unauthorized') + }) + + it('keeps plain-text bodies', () => { + const errorInfo: ErrorInfo = { status: 502, data: 'Bad gateway' } + + expect(extractErrorMessage(errorInfo, ErrorExtractorId.WIZA_ERRORS)).toBe('Bad gateway') + }) + + it('ignores a non-string top-level message', () => { + const errorInfo: ErrorInfo = { status: 422, data: { message: ['bad filter'] } } + + expect(extractErrorMessage(errorInfo, ErrorExtractorId.WIZA_ERRORS)).toBe( + 'Request failed with status 422' + ) + }) + + it('falls back to the status when Wiza sends an empty message', () => { + const errorInfo: ErrorInfo = { status: 400, data: { status: { code: 400, message: '' } } } + + expect(extractErrorMessage(errorInfo, ErrorExtractorId.WIZA_ERRORS)).toBe( + 'Request failed with status 400' + ) + }) + }) }) diff --git a/apps/sim/tools/error-extractors.ts b/apps/sim/tools/error-extractors.ts index 1177c38c642..057a49f85b4 100644 --- a/apps/sim/tools/error-extractors.ts +++ b/apps/sim/tools/error-extractors.ts @@ -514,6 +514,20 @@ const ERROR_EXTRACTORS: ErrorExtractorConfig[] = [ return parts.length > 0 ? parts.join(': ') : undefined }, }, + { + id: 'wiza-errors', + description: + 'Wiza API error envelope: {status: {code, message}}, plus plain-text bodies. The message is nested under status, so the generic extractors miss it', + examples: ['Wiza'], + extract: (errorInfo) => { + const data = errorInfo?.data + const candidates = [data, data?.status?.message, data?.message] + for (const candidate of candidates) { + if (typeof candidate === 'string' && candidate.trim()) return candidate.trim() + } + return undefined + }, + }, { id: 'crunchbase-errors', description: @@ -680,6 +694,7 @@ export const ErrorExtractorId = { POSTHOG_ERRORS: 'posthog-errors', QUICKBOOKS_FAULT: 'quickbooks-fault', PROSPEO_ERRORS: 'prospeo-errors', + WIZA_ERRORS: 'wiza-errors', CRUNCHBASE_ERRORS: 'crunchbase-errors', PITCHBOOK_ERRORS: 'pitchbook-errors', SPLUNK_ERRORS: 'splunk-errors', diff --git a/apps/sim/tools/wiza/company_enrichment.ts b/apps/sim/tools/wiza/company_enrichment.ts index 39089404f1c..e7da372122d 100644 --- a/apps/sim/tools/wiza/company_enrichment.ts +++ b/apps/sim/tools/wiza/company_enrichment.ts @@ -1,3 +1,4 @@ +import { ErrorExtractorId } from '@/tools/error-extractors' import type { ToolConfig } from '@/tools/types' import { wizaHosting } from '@/tools/wiza/hosting' import type { WizaCompanyEnrichmentParams, WizaCompanyEnrichmentResponse } from '@/tools/wiza/types' @@ -11,6 +12,7 @@ export const wizaCompanyEnrichmentTool: ToolConfig< description: 'Enrich a company by name, domain, LinkedIn ID, or LinkedIn slug with detailed firmographic data', version: '1.0.0', + errorExtractor: ErrorExtractorId.WIZA_ERRORS, hosting: wizaHosting((_params, output) => { // 2 API credits per successful company match; no charge on a no-match. diff --git a/apps/sim/tools/wiza/get_credits.ts b/apps/sim/tools/wiza/get_credits.ts index 3ff8f02c2e7..676306ae5c8 100644 --- a/apps/sim/tools/wiza/get_credits.ts +++ b/apps/sim/tools/wiza/get_credits.ts @@ -1,3 +1,4 @@ +import { ErrorExtractorId } from '@/tools/error-extractors' import type { ToolConfig } from '@/tools/types' import type { WizaGetCreditsParams, WizaGetCreditsResponse } from '@/tools/wiza/types' @@ -6,6 +7,7 @@ export const wizaGetCreditsTool: ToolConfig((_params, output) => { let credits = 0 @@ -238,9 +240,22 @@ export const wizaIndividualRevealTool: ToolConfig< consecutiveErrors += 1 if (consecutiveErrors >= MAX_CONSECUTIVE_POLL_ERRORS) { const errorText = await statusResponse.text().catch(() => '') + let errorData: unknown + try { + errorData = JSON.parse(errorText) + } catch { + errorData = errorText + } return { success: false, - error: `Wiza API error: ${statusResponse.status} - ${errorText}`, + error: extractErrorMessage( + { + status: statusResponse.status, + statusText: statusResponse.statusText, + data: errorData, + }, + ErrorExtractorId.WIZA_ERRORS + ), output: result.output, } } diff --git a/apps/sim/tools/wiza/prospect_search.ts b/apps/sim/tools/wiza/prospect_search.ts index 5031a76df62..34e7ae6bb69 100644 --- a/apps/sim/tools/wiza/prospect_search.ts +++ b/apps/sim/tools/wiza/prospect_search.ts @@ -1,4 +1,5 @@ import { isRecordLike } from '@sim/utils/object' +import { ErrorExtractorId } from '@/tools/error-extractors' import type { ToolConfig } from '@/tools/types' import { wizaHosting } from '@/tools/wiza/hosting' import type { WizaProspectSearchParams, WizaProspectSearchResponse } from '@/tools/wiza/types' @@ -11,6 +12,7 @@ export const wizaProspectSearchTool: ToolConfig< name: 'Wiza Prospect Search', description: "Search Wiza's database of prospects using person, company, and financial filters", version: '1.0.0', + errorExtractor: ErrorExtractorId.WIZA_ERRORS, hosting: wizaHosting(() => { // Prospect search returns profiles without contact data and consumes no credits; From 9a2b64dcb74f6c6d115d48a6738aacc759ee7f91 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 12:25:46 -0700 Subject: [PATCH 09/31] fix(chat): preserve per-chat resource panel widths (#8531) * fix(chat): preserve per-chat resource panel widths * fix(chat): finalize panel resizing against current layout * fix(chat): keep active resizes aligned across chat adoption --- .github/workflows/desktop-e2e.yml | 8 + apps/desktop/e2e/chat-panel.spec.ts | 429 ++++++++++++++++++ .../components/get-started/get-started.tsx | 10 +- .../home/organization-home.tsx | 2 +- .../organization-secret-input.tsx | 5 +- .../components/special-tags/special-tags.tsx | 2 + .../app/workspace/[workspaceId]/home/home.tsx | 2 +- .../[workspaceId]/home/hooks/use-chat.ts | 5 + .../home/hooks/use-mothership-resize.ts | 379 +++++++++------- .../home/hooks/use-resource-panel.ts | 21 +- .../sidebar-footer/sidebar-footer.tsx | 6 +- .../settings/settings-guarded-link.tsx | 8 +- .../components/settings/settings-sidebar.tsx | 3 +- apps/sim/hooks/use-oauth-return.ts | 2 +- .../sim/hooks/use-settings-navigation.test.ts | 16 +- apps/sim/hooks/use-settings-navigation.ts | 45 +- apps/sim/lib/browser-agent/transport.ts | 7 +- .../lib/navigation/settings-return.test.ts | 35 ++ apps/sim/lib/navigation/settings-return.ts | 27 ++ apps/sim/scripts/fixtures/chat-panel.tsx | 120 +++++ apps/sim/stores/chat-panel/store.test.ts | 40 ++ apps/sim/stores/chat-panel/store.ts | 116 +++++ 22 files changed, 1033 insertions(+), 255 deletions(-) create mode 100644 apps/desktop/e2e/chat-panel.spec.ts create mode 100644 apps/sim/lib/navigation/settings-return.test.ts create mode 100644 apps/sim/lib/navigation/settings-return.ts create mode 100644 apps/sim/scripts/fixtures/chat-panel.tsx create mode 100644 apps/sim/stores/chat-panel/store.test.ts create mode 100644 apps/sim/stores/chat-panel/store.ts diff --git a/.github/workflows/desktop-e2e.yml b/.github/workflows/desktop-e2e.yml index c9d4f92a539..e3ed8a6a745 100644 --- a/.github/workflows/desktop-e2e.yml +++ b/.github/workflows/desktop-e2e.yml @@ -11,6 +11,14 @@ on: - '.github/workflows/desktop-release.yml' - 'apps/desktop/**' - 'apps/sim/app/_shell/desktop-update-*.tsx' + - 'apps/sim/app/workspace/**/home/hooks/use-mothership-resize.ts' + - 'apps/sim/app/workspace/**/home/hooks/use-resource-panel.ts' + - 'apps/sim/app/workspace/**/home/components/chat-panel-layout.tsx' + - 'apps/sim/stores/chat-panel/**' + - 'apps/sim/stores/constants.ts' + - 'apps/sim/lib/browser-agent/transport.ts' + - 'apps/sim/lib/core/utils/separator-keys.ts' + - 'apps/sim/scripts/fixtures/chat-panel.tsx' - 'apps/sim/app/layout.tsx' - 'apps/sim/hooks/use-desktop-update-state.ts' - 'apps/sim/lib/desktop/**' diff --git a/apps/desktop/e2e/chat-panel.spec.ts b/apps/desktop/e2e/chat-panel.spec.ts new file mode 100644 index 00000000000..920be2ce387 --- /dev/null +++ b/apps/desktop/e2e/chat-panel.spec.ts @@ -0,0 +1,429 @@ +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createServer } from 'node:http' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { _electron as electron, expect, test } from '@playwright/test' +import type { SimDesktopApi } from '@sim/desktop-bridge' +import { getErrorMessage } from '@sim/utils/errors' +import { build } from 'esbuild' +import postcss from 'postcss' +import loadPostcssConfig from 'postcss-load-config' + +const DESKTOP_DIR = fileURLToPath(new URL('..', import.meta.url)) +const SIM_DIR = fileURLToPath(new URL('../../sim/', import.meta.url)) +const FIXTURE = fileURLToPath(new URL('../../sim/scripts/fixtures/chat-panel.tsx', import.meta.url)) + +test('chat panel sizes survive navigation, chat switches, collapse, and layout constraints', async () => { + const reportPath = process.env.CHAT_PANEL_REPORT_PATH ?? test.info().outputPath('chat-panel.json') + const checks: { + name: string + status: 'passed' | 'failed' + durationMs: number + error?: string + }[] = [] + const check = async (name: string, run: () => Promise) => { + const started = Date.now() + try { + await test.step(name, run) + checks.push({ name, status: 'passed', durationMs: Date.now() - started }) + } catch (error) { + checks.push({ + name, + status: 'failed', + durationMs: Date.now() - started, + error: getErrorMessage(error), + }) + throw error + } + } + const userData = mkdtempSync(join(tmpdir(), 'sim-chat-panel-e2e-')) + let app: Awaited> | undefined + const errors: string[] = [] + let desktopExit: { code: number | null; signal: string | null } | null = null + let rendererCrashed = false + let passed = false + let javascript = '' + let stylesheet = '' + const server = createServer((request, response) => { + const path = new URL(request.url ?? '/', 'http://localhost').pathname + if (path === '/page') { + response.setHeader('Content-Type', 'text/html') + response.end('Native browser resize fixture') + } else if (path === '/fixture.js' || path === '/fixture.css') { + response.setHeader('Content-Type', path.endsWith('.js') ? 'text/javascript' : 'text/css') + response.end(path.endsWith('.js') ? javascript : stylesheet) + } else if (path.startsWith('/api/')) { + response.setHeader('Content-Type', 'application/json') + response.end( + path === '/api/auth/get-session' + ? JSON.stringify({ user: { id: 'fixture-user' }, session: { id: 'fixture-session' } }) + : '{}' + ) + } else { + response.setHeader('Content-Type', 'text/html') + response.setHeader( + 'Set-Cookie', + 'better-auth.session_token=fixture; HttpOnly; SameSite=Lax; Path=/' + ) + response.end( + '
' + ) + } + }) + + try { + await check('load the production resource panel resize hook in Electron', async () => { + const config = await loadPostcssConfig({}, SIM_DIR) + const cssPath = join(SIM_DIR, 'app/_styles/globals.css') + const css = await postcss(config.plugins).process( + `${readFileSync(cssPath, 'utf8')}\n@source ${JSON.stringify(FIXTURE)};`, + { from: cssPath } + ) + const bundle = await build({ + entryPoints: [FIXTURE], + bundle: true, + write: false, + outfile: test.info().outputPath('fixture.js'), + external: ['node:async_hooks'], + banner: { js: 'var process={env:{NODE_ENV:"development"},browser:true};' }, + format: 'iife', + platform: 'browser', + tsconfig: join(SIM_DIR, 'tsconfig.json'), + define: { 'process.env.NODE_ENV': '"development"' }, + }) + javascript = bundle.outputFiles.find((file) => file.path.endsWith('.js'))?.text ?? '' + stylesheet = `${css.css}\n${bundle.outputFiles.find((file) => file.path.endsWith('.css'))?.text ?? ''}` + await new Promise((resolve) => server.listen(0, resolve)) + const address = server.address() + if (!address || typeof address === 'string') throw new Error('Missing fixture address') + app = await electron.launch({ + args: [process.env.SIM_DESKTOP_E2E_MAIN ?? '.'], + cwd: DESKTOP_DIR, + env: { + ...process.env, + SIM_DESKTOP_ORIGIN: `http://127.0.0.1:${address.port}`, + SIM_DESKTOP_USER_DATA: userData, + }, + }) + }) + if (!app) throw new Error('Electron did not launch') + app.process().once('exit', (code, signal) => { + desktopExit = { code, signal } + }) + const shell = app + const page = await shell.firstWindow() + page.on('pageerror', (error) => errors.push(error.message)) + page.on('crash', () => { + rendererCrashed = true + }) + await shell.evaluate(({ app, BrowserWindow }) => { + const window = BrowserWindow.getAllWindows()[0] + // Keep the physical window inside the small displays used by macOS CI. + window.setMinimumSize(0, 0) + window.setContentSize(720, 400) + window.webContents.setBackgroundThrottling(false) + app.focus({ steal: true }) + window.focus() + }) + await page.reload() + await shell.evaluate(({ app, BrowserWindow }) => { + const window = BrowserWindow.getAllWindows()[0] + window.webContents.setZoomFactor(0.5) + app.focus({ steal: true }) + window.focus() + }) + expect(errors).toEqual([]) + await expect + .poll(() => + shell.evaluate(({ BrowserWindow }) => BrowserWindow.getAllWindows()[0].isFocused()) + ) + .toBe(true) + await expect.poll(() => page.evaluate(() => window.innerWidth)).toBe(1440) + const panel = page.locator('[data-mothership-panel]') + const divider = page.getByRole('separator', { name: 'Resize resource view' }) + const width = () => panel.evaluate((element) => element.getBoundingClientRect().width) + const expectWidth = async (expected: number) => { + await expect.poll(width).toBeCloseTo(expected, 0) + } + const beginDrag = async () => { + await shell.evaluate(({ app, BrowserWindow }) => { + app.focus({ steal: true }) + BrowserWindow.getAllWindows()[0].focus() + }) + await expect + .poll(() => + shell.evaluate(({ BrowserWindow }) => BrowserWindow.getAllWindows()[0].isFocused()) + ) + .toBe(true) + await divider.hover({ position: { x: 4, y: 100 } }) + const rect = await panel.boundingBox() + if (!rect) throw new Error('Missing panel bounds') + await page.mouse.down() + await expect + .poll(() => divider.evaluate((element) => element.hasPointerCapture(1))) + .toBe(true) + return rect + } + const dragTo = async (target: number) => { + const rect = await beginDrag() + await page.mouse.move(rect.x + rect.width - target, rect.y + 100, { steps: 12 }) + await page.mouse.up() + await expectWidth(target) + } + + await check('a chosen width survives a settings round trip and reload', async () => { + await expectWidth(720) + await dragTo(620) + await page.getByRole('button', { name: 'Settings', exact: true }).click() + await expect(panel).toHaveCount(0) + await page.getByRole('button', { name: 'Back', exact: true }).click() + await expectWidth(620) + await page.reload() + await expectWidth(620) + }) + + await check( + 'workspace and organization chats restore independent widths without remounting', + async () => { + await page.getByRole('button', { name: 'workspace-chat-b', exact: true }).click() + await expectWidth(720) + await dragTo(830) + await page.getByRole('button', { name: 'organization-chat-a', exact: true }).click() + await expectWidth(720) + await dragTo(560) + await page.getByRole('button', { name: 'Settings', exact: true }).click() + await page.getByRole('button', { name: 'Back', exact: true }).click() + await expectWidth(560) + await page.getByRole('button', { name: 'workspace-chat-a', exact: true }).click() + await expectWidth(620) + await page.getByRole('button', { name: 'workspace-chat-b', exact: true }).click() + await expectWidth(830) + } + ) + + await check( + 'collapse preserves the expanded preference, including keyboard changes', + async () => { + await page.getByRole('button', { name: 'Collapse resource view' }).click() + await expectWidth(0) + await page.getByRole('button', { name: 'Expand resource view' }).click() + await expectWidth(830) + await divider.focus() + await page.keyboard.press('ArrowRight') + await expectWidth(798) + await page.getByRole('button', { name: 'Settings', exact: true }).click() + await page.getByRole('button', { name: 'Back', exact: true }).click() + await expectWidth(798) + } + ) + + await check('container and window clamps do not overwrite the preferred width', async () => { + await page.getByRole('button', { name: 'Resize container' }).click() + await expectWidth(520) + await page.getByRole('button', { name: 'Resize container' }).click() + await expectWidth(798) + await shell.evaluate(({ BrowserWindow }) => + BrowserWindow.getAllWindows()[0].setContentSize(525, 400) + ) + await expectWidth(570) + await page.getByRole('button', { name: 'Settings', exact: true }).click() + await page.getByRole('button', { name: 'Back', exact: true }).click() + await expectWidth(570) + await shell.evaluate(({ BrowserWindow }) => + BrowserWindow.getAllWindows()[0].setContentSize(720, 400) + ) + await expectWidth(798) + }) + + await check('resizing writes storage only when the gesture ends', async () => { + const before = await page.evaluate(() => JSON.stringify(localStorage)) + const rect = await beginDrag() + await page.mouse.move(rect.x + 100, rect.y + 100, { steps: 12 }) + await expectWidth(698) + expect(await page.evaluate(() => JSON.stringify(localStorage))).toBe(before) + await page.mouse.up() + expect(await page.evaluate(() => JSON.stringify(localStorage))).not.toBe(before) + }) + + for (const interruption of [ + 'pointercancel', + 'capture loss', + 'blur', + 'detach', + 'chat switch', + ] as const) { + await check(`${interruption} keeps the previous saved width`, async () => { + const before = await page.evaluate(() => JSON.stringify(localStorage)) + const rect = await beginDrag() + await page.mouse.move(rect.x + 100, rect.y + 100, { steps: 12 }) + await expectWidth(598) + if (interruption === 'pointercancel') { + await divider.dispatchEvent('pointercancel', { pointerId: 1 }) + } else if (interruption === 'capture loss') { + await divider.evaluate((element) => element.releasePointerCapture(1)) + await page.mouse.move(rect.x + 101, rect.y + 100) + } else if (interruption === 'blur') { + await page.evaluate(() => window.dispatchEvent(new Event('blur'))) + } else if (interruption === 'chat switch') { + await page + .getByRole('button', { name: 'organization-chat-a', exact: true }) + .evaluate((element: HTMLButtonElement) => element.click()) + await expectWidth(560) + } else { + await page + .getByRole('button', { name: 'Settings', exact: true }) + .evaluate((element: HTMLButtonElement) => element.click()) + await expect(panel).toHaveCount(0) + } + await page.mouse.up() + if (interruption === 'chat switch') { + await page.getByRole('button', { name: 'workspace-chat-b', exact: true }).click() + } + if (interruption === 'detach') { + await page.getByRole('button', { name: 'Back', exact: true }).click() + } + await expectWidth(698) + expect(await page.evaluate(() => JSON.stringify(localStorage))).toBe(before) + }) + } + + await check('a viewport change during a drag clamps the committed display width', async () => { + await page.getByRole('button', { name: 'Resize container' }).click() + await expectWidth(520) + const rect = await beginDrag() + await page.mouse.move(rect.x + 20, rect.y + 100, { steps: 12 }) + await expectWidth(500) + await shell.evaluate(({ BrowserWindow }) => + BrowserWindow.getAllWindows()[0].setContentSize(300, 400) + ) + await expect.poll(() => page.evaluate(() => window.innerWidth)).toBe(600) + await page.mouse.up() + await expectWidth(480) + await shell.evaluate(({ BrowserWindow }) => + BrowserWindow.getAllWindows()[0].setContentSize(720, 400) + ) + await expectWidth(500) + await page.getByRole('button', { name: 'Resize container' }).click() + await dragTo(698) + }) + + await check('another account cannot inherit the current chat width', async () => { + await page.getByRole('button', { name: 'Switch account' }).click() + await expectWidth(720) + await dragTo(580) + await page.getByRole('button', { name: 'Switch account' }).click() + await expectWidth(698) + }) + await check('assigning a permanent chat ID lets an active drag finish', async () => { + await page.getByRole('button', { name: 'pending:chat', exact: true }).click() + await dragTo(620) + const rect = await beginDrag() + await page.mouse.move(rect.x + 100, rect.y + 100, { steps: 12 }) + await expectWidth(520) + await page + .getByRole('button', { name: 'Assign chat ID', exact: true }) + .evaluate((element: HTMLButtonElement) => element.click()) + await expect(page.getByRole('button', { name: 'Assign chat ID', exact: true })).toBeDisabled() + expect(await divider.evaluate((element) => element.hasPointerCapture(1))).toBe(true) + await page.mouse.up() + await expectWidth(520) + await page.getByRole('button', { name: 'workspace-chat-b', exact: true }).click() + await expectWidth(698) + await page.getByRole('button', { name: 'assigned-chat', exact: true }).click() + await expectWidth(520) + await page.getByRole('button', { name: 'workspace-chat-b', exact: true }).click() + await expectWidth(698) + }) + + await check( + 'cancelling before the first animation frame restores native browser bounds', + async () => { + await page.getByRole('button', { name: 'pending:native', exact: true }).click() + await dragTo(698) + await shell.evaluate(({ app, BrowserWindow }) => { + app.focus({ steal: true }) + BrowserWindow.getAllWindows()[0].focus() + }) + await expect + .poll(() => + shell.evaluate(({ BrowserWindow }) => BrowserWindow.getAllWindows()[0].isFocused()) + ) + .toBe(true) + await page.getByRole('button', { name: 'Start browser', exact: true }).click() + const nativeBounds = () => + shell.evaluate(({ BrowserWindow, WebContentsView }) => { + const view = BrowserWindow.getAllWindows()[0].contentView.children.find( + (child) => + child instanceof WebContentsView && child.webContents.getURL().endsWith('/page') + ) + return view?.getVisible() ? view.getBounds() : null + }) + await expect.poll(nativeBounds).not.toBeNull() + const before = await nativeBounds() + await beginDrag() + await divider.evaluate((element) => { + const rect = element.getBoundingClientRect() + element.dispatchEvent( + new PointerEvent('pointermove', { pointerId: 1, clientX: rect.x + 104, bubbles: true }) + ) + element.dispatchEvent(new PointerEvent('pointercancel', { pointerId: 1, bubbles: true })) + }) + // The bridge round trip observes main-process geometry after the queued bounds messages. + await page.evaluate(() => + ( + globalThis as typeof globalThis & { simDesktop: SimDesktopApi } + ).simDesktop.browserAgent.capturePanelSnapshot('pending:native') + ) + await page.mouse.up() + await expectWidth(698) + expect(await nativeBounds()).toEqual(before) + + await check('native predictions follow a chat ID assigned during a drag', async () => { + if (!before) throw new Error('Missing native browser bounds') + await beginDrag() + await page + .getByRole('button', { name: 'Assign chat ID', exact: true }) + .evaluate((element: HTMLButtonElement) => element.click()) + await expect( + page.getByRole('button', { name: 'Assign chat ID', exact: true }) + ).toBeDisabled() + expect(await divider.evaluate((element) => element.hasPointerCapture(1))).toBe(true) + await divider.evaluate((element) => { + const rect = element.getBoundingClientRect() + element.dispatchEvent( + new PointerEvent('pointermove', { + pointerId: 1, + clientX: rect.x + 104, + bubbles: true, + }) + ) + }) + await page.evaluate(() => + ( + globalThis as typeof globalThis & { simDesktop: SimDesktopApi } + ).simDesktop.browserAgent.capturePanelSnapshot('assigned-chat') + ) + expect(await nativeBounds()).toEqual({ + ...before, + x: before.x + 50, + width: before.width - 50, + }) + await page.mouse.up() + await expectWidth(598) + }) + } + ) + expect(errors).toEqual([]) + passed = true + } finally { + mkdirSync(dirname(reportPath), { recursive: true }) + writeFileSync( + reportPath, + JSON.stringify({ passed, checks, errors, desktopExit, rendererCrashed }, null, 2) + ) + await app?.close() + await new Promise((resolve) => server.close(() => resolve())) + rmSync(userData, { recursive: true, force: true }) + } +}) diff --git a/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx b/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx index 90d272b5883..3c18b5fb2e3 100644 --- a/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx +++ b/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx @@ -3,8 +3,8 @@ import { useEffect, useState } from 'react' import { cn } from '@sim/emcn' import { ArrowRight } from '@sim/emcn/icons' -import Link from 'next/link' import { HomeSection } from '@/components/home/home-section' +import { SettingsGuardedLink } from '@/components/settings/settings-guarded-link' import { OAUTH_SEARCH_READ_SCOPE, oauthScopeSatisfies } from '@/lib/auth/oauth-provider' import type { ResourceScope } from '@/lib/core/resource-scope' import { organizationRoutes } from '@/lib/navigation/paths' @@ -142,7 +142,11 @@ export function GetStarted() { {steps.map((step, i) => { const complete = completed[step.id] return ( - 0 && 'border-t')}> + 0 && 'border-t')} + > - + ) })} diff --git a/apps/sim/app/o/[organizationId]/home/organization-home.tsx b/apps/sim/app/o/[organizationId]/home/organization-home.tsx index 7baabe357df..e6d7d080c28 100644 --- a/apps/sim/app/o/[organizationId]/home/organization-home.tsx +++ b/apps/sim/app/o/[organizationId]/home/organization-home.tsx @@ -141,7 +141,7 @@ function OrganizationHomeContent({ !hasChat && mothershipAvailable && canBuild && (searchAccess.memberScoped || planEnabled) const liveSearch = getDeploymentShape().features.liveEnterpriseSearch === true const assistantSearchLevel = 'fast' - const panel = useChatResourcePanel(chat, controller) + const panel = useChatResourcePanel(chat, controller, userId) const addResource = panel.addResourceFromUser /** Restore only an explicitly selected results tab on an empty Home; closing it clears the URL. */ useEffect(() => { diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/organization-secret-input.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/organization-secret-input.tsx index 4ff7e8bf0f6..3e18dde09d6 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/organization-secret-input.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/organization-secret-input.tsx @@ -1,6 +1,7 @@ 'use client' import { createContext, type ReactNode, useContext } from 'react' +import { SettingsGuardedLink } from '@/components/settings/settings-guarded-link' import { ApiClientError } from '@/lib/api/client/errors' import { useSession } from '@/lib/auth/auth-client' import { organizationRoutes } from '@/lib/navigation/paths' @@ -75,12 +76,12 @@ export function OrganizationSecretInputHost({ {organizationContext.viewer.isAdmin ? ( <> Enable Generic Secrets in{' '} - organization Integrations settings - + , then return here to enter the keys. ) : ( diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/special-tags.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/special-tags.tsx index 3665d705d12..fd03be2d91a 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/special-tags.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/special-tags.tsx @@ -53,6 +53,7 @@ import { parseSearchConnectionBody, searchConnectionTargetSchema, } from '@/lib/knowledge/search/connection-target' +import { rememberSettingsReturnUrl } from '@/lib/navigation/settings-return' import { OAUTH_PROVIDERS } from '@/lib/oauth/oauth' import { getServiceConfigByProviderId } from '@/lib/oauth/utils' import { organizationSecretNameSchema } from '@/lib/organization-secrets/validation' @@ -3488,6 +3489,7 @@ function UsageUpgradeDisplay({ data }: { data: UsageUpgradeTagData }) { {canManageBilling ? ( rememberSettingsReturnUrl(href)} variant='border' rightIcon={hosted ? ArrowRight : SquareArrowUpRight} target={hosted ? undefined : '_blank'} diff --git a/apps/sim/app/workspace/[workspaceId]/home/home.tsx b/apps/sim/app/workspace/[workspaceId]/home/home.tsx index a41c43e3d8c..c7b50e9ec2e 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/home.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/home.tsx @@ -178,7 +178,7 @@ function HomeContent({ chatId, userName, userId }: HomeProps) { dispatchingHeadId, getCurrentRequestId, } = chat - const panel = useChatResourcePanel(chat, controller) + const panel = useChatResourcePanel(chat, controller, userId) const { isResourceCollapsed, skipResourceTransition, diff --git a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts index 805b8fb6604..bb40fe65b80 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts +++ b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts @@ -126,6 +126,7 @@ import { getWorkflowById, getWorkflows } from '@/hooks/queries/utils/workflow-ca import { getWorkflowListQueryOptions } from '@/hooks/queries/utils/workflow-list-query' import { workflowKeys } from '@/hooks/queries/workflows' import { snapAllSmoothText } from '@/hooks/use-smooth-text' +import { useChatPanelStore } from '@/stores/chat-panel/store' import { useMothershipEffortStore } from '@/stores/mothership-effort/store' import { useMothershipQueueStore } from '@/stores/mothership-queue/store' import type { @@ -1183,6 +1184,9 @@ export function useChat( : pendingChatKeyRef.current chatIdRef.current = chatId const resolvedDesktopScopeId = desktopChatScopeId(scopeKey, chatId) + if (wasPending) { + useChatPanelStore.getState().migrate(pendingDesktopScopeId, resolvedDesktopScopeId) + } const activeActivityTracker = resourceActivityTrackerRef.current if (activeActivityTracker?.generation === streamGenRef.current) { if (wasPending) { @@ -1753,6 +1757,7 @@ export function useChat( return } + useChatPanelStore.getState().migrate(previousDesktopScopeId, resolvedChatId) await migrateDesktopChatScopes(previousDesktopScopeId, resolvedChatId) if (pendingChatKey) { useMothershipQueueStore.getState().migrate(pendingChatKey, resolvedChatId) diff --git a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-mothership-resize.ts b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-mothership-resize.ts index f9e543a4f8e..53d4587176c 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-mothership-resize.ts +++ b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-mothership-resize.ts @@ -1,6 +1,7 @@ -import { useCallback, useEffect, useRef } from 'react' +import { useCallback, useLayoutEffect, useRef } from 'react' import { beginBrowserPanelDividerDrag } from '@/lib/browser-agent/transport' import { readSeparatorKey, type SeparatorKey } from '@/lib/core/utils/separator-keys' +import { useChatPanelStore } from '@/stores/chat-panel/store' import { MOTHERSHIP_WIDTH } from '@/stores/constants' /** @@ -97,6 +98,16 @@ function writeWidthInstantly(el: HTMLElement, width: number) { el.style.transition = prevTransition } +/** Restores the preference within the current layout without changing the saved width. */ +function restorePanelWidth(el: HTMLElement, preferred: number | undefined) { + if (preferred === undefined) { + el.style.removeProperty('width') + return + } + const width = Math.min(preferred, measureMaxWidth(el)) + if (el.style.width !== `${width}px`) writeWidthInstantly(el, width) +} + /** Mirrors the panel's current width and bounds onto the divider for assistive tech. */ function syncDividerValue(handle: HTMLElement, el: HTMLElement, maxWidth = measureMaxWidth(el)) { handle.setAttribute('aria-valuemin', String(MOTHERSHIP_WIDTH.MIN)) @@ -104,6 +115,17 @@ function syncDividerValue(handle: HTMLElement, el: HTMLElement, maxWidth = measu handle.setAttribute('aria-valuenow', String(Math.round(el.getBoundingClientRect().width))) } +/** Synchronous storage hydration also covers panels that arrive after a lazy fallback. */ +function readPreferredWidth(userId: string | undefined, scopeId: string): number | undefined { + if (!useChatPanelStore.persist.hasHydrated()) void useChatPanelStore.persist.rehydrate() + return userId ? useChatPanelStore.getState().widths[`${userId}:${scopeId}`] : undefined +} + +interface MothershipResizeOptions { + userId?: string + collapsed: boolean +} + /** * Hook for managing resize of the MothershipView resource panel. * @@ -113,184 +135,209 @@ function syncDividerValue(handle: HTMLElement, el: HTMLElement, maxWidth = measu * `handleResizePointerDown` to the drag handle's onPointerDown. * Bind `handleResizeKeyDown` and `handleResizeFocus` to the same handle so it is * keyboard-adjustable and reports its value to assistive tech. - * Call `clearWidth` when the panel collapses so the CSS class retakes control. */ -export function useMothershipResize(desktopScopeId: string) { +export function useMothershipResize( + desktopScopeId: string, + { userId, collapsed }: MothershipResizeOptions +) { + const scopeRef = useRef(desktopScopeId) const mothershipRef = useRef(null) const cleanupRef = useRef<(() => void) | null>(null) const focusedDividerRef = useRef(null) - const desktopScopeIdRef = useRef(desktopScopeId) - desktopScopeIdRef.current = desktopScopeId - - const handleResizePointerDown = useCallback((e: React.PointerEvent) => { - e.preventDefault() + const preferredWidthRef = useRef(undefined) + + const rememberWidth = useCallback( + (width: number) => { + preferredWidthRef.current = width + if (userId) useChatPanelStore.getState().setWidth(userId, desktopScopeId, width) + }, + [userId, desktopScopeId] + ) + const restoreWidth = useCallback(() => { const el = mothershipRef.current - if (!el) return - // Single-flight: a second press while a drag is live must not stack listeners - if (cleanupRef.current) return - - const handle = e.currentTarget as HTMLElement - const pointerId = e.pointerId - handle.setPointerCapture(pointerId) - - // Pin to current rendered width so drag starts from the visual position - const startRect = el.getBoundingClientRect() - el.style.width = `${startRect.width}px` - - // Nothing moves the panel's right edge mid-drag, and the pointer keeps the - // offset it grabbed at, so one measurement serves the whole gesture. - const geometry: DragGeometry = { - panelRight: startRect.right, - grabOffset: e.clientX - startRect.left, - maxWidth: measureMaxWidth(el), - } - - // The panel's left edge IS the divider. Handing it to the browser - // transport lets the native browser view (when one is showing) be - // repositioned arithmetically per pointer move instead of waiting for the - // renderer's layout → measure → report round-trip; no-op (null) when no - // browser resource is live - const predictBrowserBounds = beginBrowserPanelDividerDrag( - startRect.left, - desktopScopeIdRef.current - ) + if (!el || cleanupRef.current) return + restorePanelWidth(el, collapsed ? undefined : preferredWidthRef.current) + const divider = focusedDividerRef.current + if (divider && document.activeElement === divider) syncDividerValue(divider, el) + }, [collapsed]) + + useLayoutEffect(() => { + const store = useChatPanelStore.getState() + if (store.resolveChatId(scopeRef.current) !== desktopScopeId) cleanupRef.current?.() + scopeRef.current = desktopScopeId + preferredWidthRef.current = readPreferredWidth(userId, desktopScopeId) + restoreWidth() + }, [desktopScopeId, userId, restoreWidth]) + + /** DOM attachment owns gesture cleanup; pending chat adoption leaves the same panel attached. */ + const attachPanel = useCallback( + (el: HTMLDivElement | null) => { + if (!el) return + mothershipRef.current = el + preferredWidthRef.current = readPreferredWidth(userId, scopeRef.current) + let rafId: number | null = null + const scheduleRestore = () => { + rafId ??= requestAnimationFrame(() => { + rafId = null + restoreWidth() + }) + } + restoreWidth() + const observer = new ResizeObserver(scheduleRestore) + if (el.parentElement) observer.observe(el.parentElement) + window.addEventListener('resize', scheduleRestore) + return () => { + cleanupRef.current?.() + observer.disconnect() + window.removeEventListener('resize', scheduleRestore) + if (rafId !== null) cancelAnimationFrame(rafId) + mothershipRef.current = null + } + }, + [userId, restoreWidth] + ) - // Disable CSS transition to prevent animation lag during drag - const prevTransition = el.style.transition - el.style.transition = 'none' - document.body.style.cursor = 'ew-resize' - document.body.style.userSelect = 'none' + const handleResizePointerDown = useCallback( + (e: React.PointerEvent) => { + e.preventDefault() - let rafId: number | null = null - let lastClientX: number | null = null + const el = mothershipRef.current + if (!el) return + // Single-flight: a second press while a drag is live must not stack listeners + if (cleanupRef.current) return + + const handle = e.currentTarget as HTMLElement + const pointerId = e.pointerId + handle.setPointerCapture(pointerId) + + // Pin to current rendered width so drag starts from the visual position + const startRect = el.getBoundingClientRect() + el.style.width = `${startRect.width}px` + + // Snapshot geometry avoids layout reads on every move; release applies fresh bounds. + const geometry: DragGeometry = { + panelRight: startRect.right, + grabOffset: e.clientX - startRect.left, + maxWidth: measureMaxWidth(el), + } - const applyWidth = (clientX: number) => { - el.style.width = `${panelWidthAt(clientX, geometry)}px` - } + // The panel's left edge IS the divider. Handing it to the browser + // transport lets the native browser view (when one is showing) be + // repositioned arithmetically per pointer move instead of waiting for the + // renderer's layout → measure → report round-trip; no-op (null) when no + // browser resource is live + const predictBrowserBounds = beginBrowserPanelDividerDrag(startRect.left, desktopScopeId) - // AbortController removes all listeners at once on cleanup/cancel/unmount - const ac = new AbortController() - const { signal } = ac + // Disable CSS transition to prevent animation lag during drag + const prevTransition = el.style.transition + el.style.transition = 'none' + document.body.style.cursor = 'ew-resize' + document.body.style.userSelect = 'none' - const cleanup = () => { - ac.abort() - if (rafId !== null) { - cancelAnimationFrame(rafId) - rafId = null - } - // Land on the exact final pointer position before transitions come back, - // so a fast flick whose last move never got a frame is not lost. The - // flush is what stops that catch-up delta from animating: without it the - // width write and the transition restore land in one style change, and - // the panel eases into its final width over 200ms while the native view - // chases it. - if (lastClientX !== null) applyWidth(lastClientX) - void el.offsetWidth - el.style.transition = prevTransition - document.body.style.cursor = '' - document.body.style.userSelect = '' - cleanupRef.current = null - syncDividerValue(handle, el) - } - cleanupRef.current = cleanup - - handle.addEventListener( - 'pointermove', - (moveEvent: PointerEvent) => { - if (moveEvent.pointerId !== pointerId) return - lastClientX = moveEvent.clientX - // Fast path first: hand the native browser view its next rect at - // pointer-event time (clamped exactly like the width write below), a - // full layout pass ahead of the measured geometry report - predictBrowserBounds?.(dividerXAt(moveEvent.clientX, geometry)) - // Coalesce to one width write per frame: pointermove can outpace the - // display refresh, and every unbatched write forces an extra layout - // pass that the embedded browser view then has to chase - rafId ??= requestAnimationFrame(() => { - rafId = null - if (lastClientX !== null) applyWidth(lastClientX) - }) - }, - { signal } - ) + let rafId: number | null = null + let lastClientX: number | null = null - handle.addEventListener( - 'pointerup', - (upEvent: PointerEvent) => { - if (upEvent.pointerId !== pointerId) return - handle.releasePointerCapture(upEvent.pointerId) - cleanup() - }, - { signal } - ) + const applyWidth = (clientX: number) => { + el.style.width = `${panelWidthAt(clientX, geometry)}px` + } - // Browser fires pointercancel when it reclaims the gesture (scroll, palm rejection, etc.) - // Without this, body cursor/userSelect and transition would be permanently stuck - handle.addEventListener('pointercancel', cleanup, { signal }) - // A blur mid-drag (cmd-tab, window switch) would otherwise strand the - // body cursor/userSelect overrides with no pointerup coming - window.addEventListener('blur', cleanup, { signal }) - }, []) + // AbortController removes all listeners at once on cleanup/cancel/unmount + const ac = new AbortController() + const { signal } = ac - // Tear down any active drag if the component unmounts mid-drag - useEffect(() => { - return () => { - cleanupRef.current?.() - } - }, []) + const finish = (commit: boolean) => { + ac.abort() + if (rafId !== null) { + cancelAnimationFrame(rafId) + rafId = null + } + if (commit && lastClientX !== null) { + rememberWidth(panelWidthAt(lastClientX, geometry)) + } + // Flush the restored width before transitions return, so the native view + // does not chase a 200ms catch-up animation. + restorePanelWidth(el, preferredWidthRef.current) + void el.offsetWidth + // A cancelled frame may never change DOM size, so ResizeObserver cannot undo its prediction. + const restoredRect = el.getBoundingClientRect() + if ( + restoredRect.left === startRect.left && + restoredRect.top === startRect.top && + restoredRect.width === startRect.width && + restoredRect.height === startRect.height + ) { + predictBrowserBounds?.(restoredRect.left, scopeRef.current) + } + el.style.transition = prevTransition + document.body.style.cursor = '' + document.body.style.userSelect = '' + cleanupRef.current = null + if (handle.hasPointerCapture(pointerId)) handle.releasePointerCapture(pointerId) + syncDividerValue(handle, el) + } + const cancel = () => finish(false) + const cancelPointer = (event: PointerEvent) => { + if (event.pointerId === pointerId) cancel() + } + cleanupRef.current = cancel + + handle.addEventListener( + 'pointermove', + (moveEvent: PointerEvent) => { + if (moveEvent.pointerId !== pointerId) return + lastClientX = moveEvent.clientX + // Fast path first: hand the native browser view its next rect at + // pointer-event time (clamped exactly like the width write below), a + // full layout pass ahead of the measured geometry report + predictBrowserBounds?.(dividerXAt(moveEvent.clientX, geometry), scopeRef.current) + // Coalesce to one width write per frame: pointermove can outpace the + // display refresh, and every unbatched write forces an extra layout + // pass that the embedded browser view then has to chase + rafId ??= requestAnimationFrame(() => { + rafId = null + if (lastClientX !== null) applyWidth(lastClientX) + }) + }, + { signal } + ) + + handle.addEventListener( + 'pointerup', + (upEvent: PointerEvent) => { + if (upEvent.pointerId !== pointerId) return + finish(true) + }, + { signal } + ) + + // Browser fires pointercancel when it reclaims the gesture (scroll, palm rejection, etc.) + // Without this, body cursor/userSelect and transition would be permanently stuck + handle.addEventListener('pointercancel', cancelPointer, { signal }) + handle.addEventListener('lostpointercapture', cancelPointer, { signal }) + // A blur mid-drag (cmd-tab, window switch) would otherwise strand the + // body cursor/userSelect overrides with no pointerup coming + window.addEventListener('blur', cancel, { signal }) + }, + [desktopScopeId, rememberWidth] + ) - // Re-clamp panel width when the viewport is resized (inline px width can exceed max after narrowing). - // Shares `measureMaxWidth` with the drag so a window resize can never leave - // the panel at a width the drag would have refused, or vice versa. - // Coalesced to one frame, and the pinned width is read off the inline style - // rather than the box: resize events can outpace the display during a live - // window-edge drag, so measuring in the handler would flush layout per event. - // The container measurement `computeMaxWidth` does need costs one flush, but - // it happens inside the coalesced frame, not per event. - // The clamp also has to land without animating — the transition on the panel - // would otherwise make the embedded browser view chase a moving rect for - // 200ms after the drag stops. - useEffect(() => { - let rafId: number | null = null - - const clampWidth = () => { - rafId = null + /** Steps the panel width from the focused divider, never during a live drag. */ + const handleResizeKeyDown = useCallback( + (e: React.KeyboardEvent) => { + const key = readSeparatorKey(e) const el = mothershipRef.current - if (!el) return - const pinned = el.style.width - const divider = focusedDividerRef.current - const reportsToDivider = divider !== null && document.activeElement === divider - if (!pinned && !reportsToDivider) return + if (!key || !el || cleanupRef.current) return const maxWidth = measureMaxWidth(el) - if (pinned && Number.parseFloat(pinned) > maxWidth) writeWidthInstantly(el, maxWidth) - if (reportsToDivider) syncDividerValue(divider, el, maxWidth) - } - - const handleWindowResize = () => { - if (rafId !== null) return - rafId = requestAnimationFrame(clampWidth) - } - - window.addEventListener('resize', handleWindowResize) - return () => { - window.removeEventListener('resize', handleWindowResize) - if (rafId !== null) cancelAnimationFrame(rafId) - } - }, []) - - /** Steps the panel width from the focused divider, never during a live drag. */ - const handleResizeKeyDown = useCallback((e: React.KeyboardEvent) => { - const key = readSeparatorKey(e) - const el = mothershipRef.current - if (!key || !el || cleanupRef.current) return - const maxWidth = measureMaxWidth(el) - const width = keyboardPanelWidth(key, el.getBoundingClientRect().width, maxWidth) - e.preventDefault() - e.stopPropagation() - writeWidthInstantly(el, width) - syncDividerValue(e.currentTarget, el, maxWidth) - }, []) + const width = keyboardPanelWidth(key, el.getBoundingClientRect().width, maxWidth) + e.preventDefault() + e.stopPropagation() + writeWidthInstantly(el, width) + rememberWidth(width) + syncDividerValue(e.currentTarget, el, maxWidth) + }, + [rememberWidth] + ) /** Reports the current width when the divider takes focus, and while it keeps focus. */ const handleResizeFocus = useCallback((e: React.FocusEvent) => { @@ -299,16 +346,10 @@ export function useMothershipResize(desktopScopeId: string) { if (el) syncDividerValue(e.currentTarget, el) }, []) - /** Remove inline width so the collapse CSS class retakes control */ - const clearWidth = useCallback(() => { - mothershipRef.current?.style.removeProperty('width') - }, []) - return { - mothershipRef, + mothershipRef: attachPanel, handleResizePointerDown, handleResizeKeyDown, handleResizeFocus, - clearWidth, } } diff --git a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-resource-panel.ts b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-resource-panel.ts index 2778516a241..046da128f3f 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-resource-panel.ts +++ b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-resource-panel.ts @@ -154,7 +154,8 @@ export function useChatResourcePanel( | 'removeResource' | 'setActiveResourceId' >, - controller: ReturnType + controller: ReturnType, + userId?: string ) { const { desktopScopeId, @@ -182,22 +183,16 @@ export function useChatResourcePanel( effectiveActiveResourceIdRef, onResourceEvent: handleResourceEvent, } = controller - const { - mothershipRef, - handleResizePointerDown, - handleResizeKeyDown, - handleResizeFocus, - clearWidth, - } = useMothershipResize(desktopScopeId) + const { mothershipRef, handleResizePointerDown, handleResizeKeyDown, handleResizeFocus } = + useMothershipResize(desktopScopeId, { userId, collapsed: isResourceCollapsed }) effectiveActiveResourceIdRef.current = activeResourceId const resourceAttentionChatIdRef = useRef(resolvedChatId) const collapseResource = useCallback(() => { resourceCollapseOwnedByUserRef.current = true resourceSelectionOwnedByUserRef.current = true - clearWidth() setResourceCollapsed(true) - }, [clearWidth, setResourceCollapsed]) + }, [setResourceCollapsed]) const clearResourceActivity = useCallback((resourceId: string) => { setResourceActivityIds((current) => { @@ -279,7 +274,6 @@ export function useChatResourcePanel( const previousChatId = resourceAttentionChatIdRef.current resourceAttentionChatIdRef.current = resolvedChatId if (!resolvedChatId) { - clearWidth() setResourceCollapsed(true) } if (!resolvedChatId || (previousChatId && previousChatId !== resolvedChatId)) { @@ -287,7 +281,7 @@ export function useChatResourcePanel( resourceSelectionOwnedByUserRef.current = false setResourceActivityIds(new Set()) } - }, [resolvedChatId, clearWidth, setResourceCollapsed]) + }, [resolvedChatId, setResourceCollapsed]) useEffect(() => { if ( @@ -304,10 +298,9 @@ export function useChatResourcePanel( useEffect(() => { if (resources.length === 0 && !isResourceCollapsedRef.current) { - clearWidth() setResourceCollapsed(true) } - }, [resources, clearWidth, setResourceCollapsed]) + }, [resources, setResourceCollapsed]) useEffect(() => { const resourceIds = new Set(resources.map(getChatResourceSelectionId)) diff --git a/apps/sim/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-footer/sidebar-footer.tsx b/apps/sim/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-footer/sidebar-footer.tsx index 1da14a9bf22..6fd710207d1 100644 --- a/apps/sim/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-footer/sidebar-footer.tsx +++ b/apps/sim/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-footer/sidebar-footer.tsx @@ -24,6 +24,7 @@ import { SettingsIntentLink } from '@/components/settings/settings-intent-link' import { ANONYMOUS_USER_ID } from '@/lib/auth/constants' import { signOutAndRedirect } from '@/lib/auth/sign-out' import { getDesktopUpdates } from '@/lib/desktop' +import { rememberSettingsReturnUrl } from '@/lib/navigation/settings-return' import { getUserColor } from '@/lib/workspaces/colors' import { SidebarTooltip } from '@/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-tooltip' import { @@ -227,7 +228,10 @@ export function SidebarFooter({ href={href} onNavigate={(event) => { event.preventDefault() - useSettingsDirtyStore.getState().requestLeave(onNavigate) + useSettingsDirtyStore.getState().requestLeave(() => { + rememberSettingsReturnUrl(href) + onNavigate() + }) }} > diff --git a/apps/sim/components/settings/settings-guarded-link.tsx b/apps/sim/components/settings/settings-guarded-link.tsx index eb0565c27ee..337e3f57f51 100644 --- a/apps/sim/components/settings/settings-guarded-link.tsx +++ b/apps/sim/components/settings/settings-guarded-link.tsx @@ -3,6 +3,7 @@ import type { ComponentProps } from 'react' import Link from 'next/link' import { useRouter } from 'next/navigation' +import { rememberSettingsReturnUrl } from '@/lib/navigation/settings-return' import { useSettingsDirtyStore } from '@/stores/settings/dirty/store' interface SettingsGuardedLinkProps @@ -23,7 +24,12 @@ export function SettingsGuardedLink({ href, onNavigate, ...props }: SettingsGuar const { isDirty, navigationBlocked, requestLeave } = useSettingsDirtyStore.getState() if (isDirty || navigationBlocked) { event.preventDefault() - requestLeave(() => router.push(href)) + requestLeave(() => { + rememberSettingsReturnUrl(href) + router.push(href) + }) + } else { + rememberSettingsReturnUrl(href) } onNavigate?.() }} diff --git a/apps/sim/components/settings/settings-sidebar.tsx b/apps/sim/components/settings/settings-sidebar.tsx index e26d05a36a7..d8c389c83e8 100644 --- a/apps/sim/components/settings/settings-sidebar.tsx +++ b/apps/sim/components/settings/settings-sidebar.tsx @@ -25,6 +25,7 @@ import { import { SettingsIntentLink } from '@/components/settings/settings-intent-link' import { usePendingSettingsSelection } from '@/components/settings/use-pending-settings-selection' import { APP_ENTRY_PATH } from '@/lib/navigation/paths' +import { popSettingsReturnUrl } from '@/lib/navigation/settings-return' import { SidebarSection } from '@/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-section' import { SidebarTooltip } from '@/app/workspace/[workspaceId]/w/components/sidebar/components/sidebar-tooltip' import { @@ -134,7 +135,7 @@ export function SettingsSidebar
({ fullWidth leftIcon={ChevronLeft} className={SIDEBAR_RAIL_CHIP_CLASS} - onClick={() => requestLeave(() => router.push(backHref))} + onClick={() => requestLeave(() => router.push(popSettingsReturnUrl(backHref)))} > Back diff --git a/apps/sim/hooks/use-oauth-return.ts b/apps/sim/hooks/use-oauth-return.ts index 8509d0d33be..ef40c531c2d 100644 --- a/apps/sim/hooks/use-oauth-return.ts +++ b/apps/sim/hooks/use-oauth-return.ts @@ -27,6 +27,7 @@ import { } from '@/lib/credentials/oauth-chat-attempt' import { getDesktopBridge } from '@/lib/desktop' import { organizationRoutes } from '@/lib/navigation/paths' +import { SETTINGS_RETURN_URL_KEY } from '@/lib/navigation/settings-return' import { stripMicrosoftDataverseEnvironmentFromOAuthCallback } from '@/lib/oauth/microsoft-dataverse' import { searchSetupAccessParam } from '@/lib/sim-search/search-params' import { organizationSearchSetupPath } from '@/lib/sim-search/setup-navigation' @@ -36,7 +37,6 @@ import { workspaceCredentialKeys, } from '@/hooks/queries/utils/credential-keys' import { requireWorkspaceCredentialListResponse } from '@/hooks/queries/utils/fetch-workspace-credentials' -import { SETTINGS_RETURN_URL_KEY } from '@/hooks/use-settings-navigation' const OAUTH_CREDENTIAL_UPDATED_EVENT = 'oauth-credentials-updated' const CONTEXT_MAX_AGE_MS = 15 * 60 * 1000 diff --git a/apps/sim/hooks/use-settings-navigation.test.ts b/apps/sim/hooks/use-settings-navigation.test.ts index 1e8aa042086..e3ee76c8ece 100644 --- a/apps/sim/hooks/use-settings-navigation.test.ts +++ b/apps/sim/hooks/use-settings-navigation.test.ts @@ -11,7 +11,7 @@ import type { WorkspaceHostContext } from '@/lib/api/contracts/workspaces' */ vi.mock('@/lib/auth/auth-client', () => authClientMock) -import { resolveSettingsHref, resolveSettingsReturnUrl } from '@/hooks/use-settings-navigation' +import { resolveSettingsHref } from '@/hooks/use-settings-navigation' const HOST_CONTEXT: WorkspaceHostContext = { workspace: { @@ -53,17 +53,3 @@ describe('resolveSettingsHref unified settings navigation', () => { ).toBe('/workspace/workspace-b/upgrade') }) }) - -describe('resolveSettingsReturnUrl', () => { - const fallback = '/workspace/workspace-b' - - it('discards a stored url captured in a workspace the user has since left', () => { - expect( - resolveSettingsReturnUrl({ - storedUrl: '/workspace/workspace-a/w/workflow-a', - workspaceId: 'workspace-b', - fallback, - }) - ).toBe(fallback) - }) -}) diff --git a/apps/sim/hooks/use-settings-navigation.ts b/apps/sim/hooks/use-settings-navigation.ts index bed928bc466..534309ab677 100644 --- a/apps/sim/hooks/use-settings-navigation.ts +++ b/apps/sim/hooks/use-settings-navigation.ts @@ -6,11 +6,10 @@ import type { WorkspaceHostContext } from '@/lib/api/contracts/workspaces' import { useSession } from '@/lib/auth/auth-client' import { canManageWorkspaceBilling } from '@/lib/billing/workspace-permissions' import { APP_ENTRY_PATH } from '@/lib/navigation/paths' +import { popSettingsReturnUrl, rememberSettingsReturnUrl } from '@/lib/navigation/settings-return' import { useOptionalWorkspaceHostContext } from '@/app/workspace/[workspaceId]/providers/workspace-host-provider' import type { SettingsSection } from '@/app/workspace/[workspaceId]/settings/navigation' -export const SETTINGS_RETURN_URL_KEY = 'settings-return-url' - interface SettingsNavigationOptions { section?: SettingsSection mcpServerId?: string @@ -58,31 +57,6 @@ export function resolveSettingsHref({ return query ? `${pathname}?${query}` : pathname } -interface ResolveSettingsReturnUrlParams { - storedUrl: string | null - workspaceId?: string - fallback: string -} - -/** - * Resolves the stored settings return url, discarding it when it points at a - * different workspace than the one currently open. Switching workspaces from - * settings keeps the user on the new workspace, so a return url captured in the - * old one would silently navigate them back out of it. - */ -export function resolveSettingsReturnUrl({ - storedUrl, - workspaceId, - fallback, -}: ResolveSettingsReturnUrlParams): string { - if (!storedUrl) return fallback - const [, root, storedWorkspaceId] = storedUrl.split('/') - if (root === 'workspace' && storedWorkspaceId && storedWorkspaceId !== workspaceId) { - return fallback - } - return storedUrl -} - export function useSettingsNavigation(): UseSettingsNavigationReturn { const router = useRouter() const params = useParams<{ workspaceId?: string }>() @@ -103,28 +77,13 @@ export function useSettingsNavigation(): UseSettingsNavigationReturn { [hostContext, session?.user?.id, workspaceId] ) - const popSettingsReturnUrl = useCallback( - (fallback: string): string => { - try { - const storedUrl = sessionStorage.getItem(SETTINGS_RETURN_URL_KEY) - sessionStorage.removeItem(SETTINGS_RETURN_URL_KEY) - return resolveSettingsReturnUrl({ storedUrl, workspaceId, fallback }) - } catch { - return fallback - } - }, - [workspaceId] - ) - const navigateToSettings = useCallback( (options?: SettingsNavigationOptions) => { const currentPath = window.location.pathname if (currentPath.startsWith(settingsPrefix)) { router.replace(getSettingsHref(options), { scroll: false }) } else { - try { - sessionStorage.setItem(SETTINGS_RETURN_URL_KEY, currentPath) - } catch {} + rememberSettingsReturnUrl(getSettingsHref(options)) router.push(getSettingsHref(options)) } }, diff --git a/apps/sim/lib/browser-agent/transport.ts b/apps/sim/lib/browser-agent/transport.ts index 693a5440f70..bf368ce5942 100644 --- a/apps/sim/lib/browser-agent/transport.ts +++ b/apps/sim/lib/browser-agent/transport.ts @@ -659,6 +659,7 @@ export function setBrowserPanelOccluded( * with its left edge shifted by the divider's travel. Call at drag start with * the divider position (the panel's left edge in viewport CSS pixels); the * returned predictor reports a rect per pointer move, before layout runs. + * Pass the current scope with each prediction if a pending chat adopts its durable ID mid-drag. * Measured reports remain authoritative and correct any drift. * * Both `startDividerX` and every `dividerX` must be the panel's REAL viewport @@ -673,10 +674,10 @@ export function setBrowserPanelOccluded( export function beginBrowserPanelDividerDrag( startDividerX: number, scopeId = currentBrowserScopeId() -): ((dividerX: number) => void) | null { +): ((dividerX: number, reportScopeId?: string) => void) | null { const base = latestPanelBoundsByScope.get(scopeId) if (!bridge() || !base) return null - return (dividerX: number) => { + return (dividerX: number, reportScopeId = scopeId) => { const dx = Math.round(dividerX - startDividerX) const width = base.width - dx if (width <= 0) return @@ -686,7 +687,7 @@ export function beginBrowserPanelDividerDrag( reportBrowserPanelBounds( { x: base.x + dx, y: base.y, width, height: base.height }, { viewportWidth: window.innerWidth, viewportHeight: window.innerHeight, widthRatio: 0 }, - scopeId + reportScopeId ) } } diff --git a/apps/sim/lib/navigation/settings-return.test.ts b/apps/sim/lib/navigation/settings-return.test.ts new file mode 100644 index 00000000000..420da2a2103 --- /dev/null +++ b/apps/sim/lib/navigation/settings-return.test.ts @@ -0,0 +1,35 @@ +/** @vitest-environment jsdom */ +import { beforeEach, describe, expect, it } from 'vitest' +import { popSettingsReturnUrl, rememberSettingsReturnUrl } from '@/lib/navigation/settings-return' + +describe('settings round trips', () => { + beforeEach(() => sessionStorage.clear()) + + it.each(['/workspace/workspace-a', '/o/organization-a'])( + 'restores the complete chat URL in %s without replacing it on section changes', + (scope) => { + const original = `${scope}/chat/chat-a?resource=file-a&view=view-a#selection` + window.history.replaceState(null, '', original) + rememberSettingsReturnUrl(`${scope}/settings/general`) + window.history.replaceState(null, '', `${scope}/settings/general`) + rememberSettingsReturnUrl(`${scope}/settings/profile`) + window.history.replaceState(null, '', `${scope}/settings/profile`) + expect(popSettingsReturnUrl(`${scope}/home`)).toBe(original) + expect(popSettingsReturnUrl(`${scope}/home`)).toBe(`${scope}/home`) + } + ) + + it('does not carry a return destination into a different organization', () => { + window.history.replaceState(null, '', '/o/organization-a/chat/chat-a?resource=file-a') + rememberSettingsReturnUrl('/o/organization-a/settings/general') + window.history.replaceState(null, '', '/o/organization-b/settings/general') + expect(popSettingsReturnUrl('/o/organization-b/home')).toBe('/o/organization-b/home') + }) + + it('does not store an origin from outside the destination settings scope', () => { + window.history.replaceState(null, '', '/workspace/workspace-a/chat/chat-a') + rememberSettingsReturnUrl('/o/organization-a/settings/general') + window.history.replaceState(null, '', '/o/organization-a/settings/general') + expect(popSettingsReturnUrl('/o/organization-a/home')).toBe('/o/organization-a/home') + }) +}) diff --git a/apps/sim/lib/navigation/settings-return.ts b/apps/sim/lib/navigation/settings-return.ts new file mode 100644 index 00000000000..b2466ddc352 --- /dev/null +++ b/apps/sim/lib/navigation/settings-return.ts @@ -0,0 +1,27 @@ +export const SETTINGS_RETURN_URL_KEY = 'settings-return-url' + +function settingsScope(pathname: string): string | undefined { + return pathname.match(/^(\/(?:workspace|o)\/[^/?#]+)\/settings(?:\/|$)/)?.[1] +} + +/** Captures a complete return URL only when entering settings from the same owning surface. */ +export function rememberSettingsReturnUrl(settingsHref: string): void { + const scope = settingsScope(settingsHref) + const { pathname, search, hash } = window.location + if (!scope || !pathname.startsWith(`${scope}/`) || settingsScope(pathname)) return + try { + sessionStorage.setItem(SETTINGS_RETURN_URL_KEY, `${pathname}${search}${hash}`) + } catch {} +} + +/** Consumes the saved destination only when it belongs to the current settings owner. */ +export function popSettingsReturnUrl(fallback: string): string { + try { + const stored = sessionStorage.getItem(SETTINGS_RETURN_URL_KEY) + sessionStorage.removeItem(SETTINGS_RETURN_URL_KEY) + const scope = settingsScope(window.location.pathname) + return scope && stored?.startsWith(`${scope}/`) && !settingsScope(stored) ? stored : fallback + } catch { + return fallback + } +} diff --git a/apps/sim/scripts/fixtures/chat-panel.tsx b/apps/sim/scripts/fixtures/chat-panel.tsx new file mode 100644 index 00000000000..a6c68711c01 --- /dev/null +++ b/apps/sim/scripts/fixtures/chat-panel.tsx @@ -0,0 +1,120 @@ +import { lazy, StrictMode, Suspense, useRef, useState } from 'react' +import { createRoot } from 'react-dom/client' +import { + activateBrowserScope, + migrateBrowserScope, + openUrlInNewBrowserTab, + reportBrowserPanelBounds, +} from '@/lib/browser-agent/transport' +import { + ChatPanelContent, + ChatPanelLayout, +} from '@/app/workspace/[workspaceId]/home/components/chat-panel-layout' +import { useMothershipResize } from '@/app/workspace/[workspaceId]/home/hooks/use-mothership-resize' +import { useChatPanelStore } from '@/stores/chat-panel/store' + +const LazyContent = lazy(async () => ({ default: ChatPanelContent })) + +interface ChatFixtureProps { + chatId: string + userId: string +} + +function ChatFixture({ chatId, userId }: ChatFixtureProps) { + const browserHost = useRef(null) + const [collapsed, setCollapsed] = useState(false) + const panel = useMothershipResize(chatId, { userId, collapsed }) + const startBrowser = async () => { + await activateBrowserScope(chatId) + await openUrlInNewBrowserTab( + `${location.origin.replace('127.0.0.1', 'localhost')}/page`, + chatId + ) + const rect = browserHost.current?.getBoundingClientRect() + if (rect) + reportBrowserPanelBounds( + { x: rect.x, y: rect.y, width: rect.width, height: rect.height }, + null, + chatId + ) + } + return ( + setCollapsed(!collapsed)} + onResize={panel.handleResizePointerDown} + onResizeKeyDown={panel.handleResizeKeyDown} + onResizeFocus={panel.handleResizeFocus} + panel={ + + +
+ Resource +
+
+
+ } + > +
+ Chat + +
+
+ ) +} + +interface ChatPanelFixtureProps { + [key: string]: never +} + +function ChatPanelFixture(_props: ChatPanelFixtureProps) { + const [chatId, setChatId] = useState('workspace-chat-a') + const [userId, setUserId] = useState('user-a') + const [settings, setSettings] = useState(false) + const [narrow, setNarrow] = useState(false) + return ( + <> + +
+ {!settings && } +
+ + ) +} + +const root = document.getElementById('root') +if (!root) throw new Error('Missing fixture root') +createRoot(root).render( + + + +) diff --git a/apps/sim/stores/chat-panel/store.test.ts b/apps/sim/stores/chat-panel/store.test.ts new file mode 100644 index 00000000000..9f744d77943 --- /dev/null +++ b/apps/sim/stores/chat-panel/store.test.ts @@ -0,0 +1,40 @@ +/** @vitest-environment jsdom */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { useChatPanelStore } from '@/stores/chat-panel/store' + +describe('chat identity adoption during a resize', () => { + beforeEach(() => { + useChatPanelStore.getState().reset() + }) + + it.each([undefined, 600])( + 'lands a late gesture on the assigned chat (previous width: %s)', + (initialWidth) => { + const store = useChatPanelStore.getState() + if (initialWidth !== undefined) store.setWidth('user-a', 'pending:chat', initialWidth) + store.migrate('pending:chat', 'chat-a') + store.setWidth('user-a', 'pending:chat', 700) + expect(useChatPanelStore.getState().widths).toEqual({ 'user-a:chat-a': 700 }) + } + ) +}) + +describe('cold chat panel preferences', () => { + it.each(['migrate', 'resize'] as const)( + 'preserves saved chats when %s precedes panel attachment', + async (action) => { + vi.resetModules() + localStorage.setItem( + 'chat-panel-widths', + JSON.stringify({ state: { widths: { 'user-a:existing-chat': 720 } }, version: 0 }) + ) + const { useChatPanelStore: coldStore } = await import('@/stores/chat-panel/store') + if (action === 'migrate') coldStore.getState().migrate('pending:chat', 'chat-a') + else coldStore.getState().setWidth('user-a', 'chat-a', 700) + expect(coldStore.getState().widths['user-a:existing-chat']).toBe(720) + expect( + JSON.parse(localStorage.getItem('chat-panel-widths')!).state.widths['user-a:existing-chat'] + ).toBe(720) + } + ) +}) diff --git a/apps/sim/stores/chat-panel/store.ts b/apps/sim/stores/chat-panel/store.ts new file mode 100644 index 00000000000..d5c10a4f245 --- /dev/null +++ b/apps/sim/stores/chat-panel/store.ts @@ -0,0 +1,116 @@ +import { createLogger } from '@sim/logger' +import { toRecord } from '@sim/utils/object' +import { LRUCache } from 'lru-cache' +import { create } from 'zustand' +import { createJSONStorage, devtools, persist } from 'zustand/middleware' +import { MOTHERSHIP_WIDTH } from '@/stores/constants' +import { registerUserDataReset } from '@/stores/user-data-reset-registry' + +const STORAGE_KEY = 'chat-panel-widths' +const MAX_SAVED_CHATS = 200 +const logger = createLogger('ChatPanelStore') +/** A drag can finish after the server assigns its pending chat a durable ID. */ +const adoptedChatKeys = new LRUCache({ max: MAX_SAVED_CHATS }) + +interface ChatPanelState { + widths: Record + resolveChatId: (chatId: string) => string + setWidth: (userId: string, chatId: string, width: number) => void + migrate: (fromChatId: string, toChatId: string) => void + reset: () => void +} + +function validWidth(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) && value >= MOTHERSHIP_WIDTH.MIN +} + +/** Bounds device-local history by most recently adjusted chat, including restored storage. */ +function boundedWidths(value: unknown): Record { + const widths: Record = {} + for (const [key, width] of Object.entries(toRecord(value)).slice(-MAX_SAVED_CHATS)) { + if (key && validWidth(width)) widths[key] = width + } + return widths +} + +/** Storage failure must not prevent resizing or retaining preferences for the current session. */ +const storage = createJSONStorage>(() => ({ + getItem: (key) => { + try { + return window.localStorage.getItem(key) + } catch { + return null + } + }, + setItem: (key, value) => { + try { + window.localStorage.setItem(key, value) + } catch { + logger.warn('Unable to save chat panel preferences') + } + }, + removeItem: (key) => { + try { + window.localStorage.removeItem(key) + } catch { + logger.warn('Unable to clear chat panel preferences') + } + }, +})) + +export const useChatPanelStore = create()( + devtools( + persist( + (set, get) => ({ + widths: {}, + resolveChatId: (chatId) => adoptedChatKeys.get(chatId) ?? chatId, + setWidth: (userId, chatId, width) => { + if (!validWidth(width)) return + if (!useChatPanelStore.persist.hasHydrated()) void useChatPanelStore.persist.rehydrate() + const key = `${userId}:${get().resolveChatId(chatId)}` + if (get().widths[key] === width) return + set((state) => { + const { [key]: _previous, ...rest } = state.widths + return { widths: boundedWidths({ ...rest, [key]: width }) } + }) + }, + migrate: (fromChatId, toChatId) => { + if (fromChatId === toChatId) return + if (!useChatPanelStore.persist.hasHydrated()) void useChatPanelStore.persist.rehydrate() + adoptedChatKeys.set(fromChatId, toChatId) + const previous = get().widths + const widths = { ...previous } + let changed = false + for (const [key, width] of Object.entries(previous)) { + if (!key.endsWith(`:${fromChatId}`)) continue + const destination = `${key.slice(0, -fromChatId.length)}${toChatId}` + widths[destination] ??= width + delete widths[key] + changed = true + } + if (changed) set({ widths }) + }, + reset: () => { + adoptedChatKeys.clear() + set({ widths: {} }) + }, + }), + { + name: STORAGE_KEY, + storage, + skipHydration: true, + partialize: ({ widths }) => ({ widths }), + merge: (persisted, current) => ({ + ...current, + widths: boundedWidths(toRecord(persisted).widths), + }), + } + ), + { name: 'chat-panel' } + ) +) + +registerUserDataReset(STORAGE_KEY, () => { + useChatPanelStore.getState().reset() + void useChatPanelStore.persist.clearStorage() +}) From a1e573aed0d25c5ae51e698dadb69fe66bfd41ce Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 13:03:01 -0700 Subject: [PATCH 10/31] chore(search): remove legacy indexed enterprise search (#8528) * chore(search): remove legacy indexed enterprise search * fix(search): preserve live onboarding and Slack scope policies * fix(search): close retired document writes and refresh Search consent * fix(search): preserve ordinary knowledge base classification --- .github/workflows/ci.yml | 1 - .github/workflows/desktop-e2e.yml | 1 - .../self-hosting/integrations-oauth.mdx | 2 +- apps/docs/content/docs/search/index.mdx | 2 +- apps/docs/content/docs/search/mcp.mdx | 2 +- .../connectors/directory-sync/route.test.ts | 19 - apps/sim/app/api/knowledge/search/route.ts | 64 +- .../api/knowledge/sim-search/connect/route.ts | 20 - .../sim-search/integrations/overview/route.ts | 23 - .../sim-search/personal-source-setup/route.ts | 70 -- .../sim-search/sources/overview/route.ts | 23 - .../sim-search/sources/progress/route.ts | 23 - .../sim-search/sources/route.test.ts | 44 +- .../api/knowledge/sim-search/stats/route.ts | 24 - .../[id]/connected-accounts/indexing/route.ts | 22 - apps/sim/app/api/v1/knowledge/search/route.ts | 3 - .../components/composer/composer.test.tsx | 9 +- .../components/get-started/get-started.tsx | 70 +- .../home/organization-home.test.tsx | 16 +- .../home/organization-home.tsx | 10 +- .../indexed/github-member-integration.tsx | 91 -- .../integrations/indexed/index.ts | 1 - .../indexed/member-integration-row.tsx | 223 ---- .../indexed/member-integrations-list.tsx | 268 ---- .../integrations/indexed/source-status.ts | 44 - .../indexed/use-member-enrollment.test.tsx | 358 ------ .../indexed/use-member-enrollment.ts | 416 ------ .../integrations/integrations.test.tsx | 1041 --------------- .../integrations/integrations.tsx | 12 +- .../integrations/page.test.tsx | 25 - .../[knowledgeBaseId]/[documentId]/page.tsx | 91 -- .../[documentId]/search-params.ts | 25 - .../components/integrations/indexed/index.ts | 1 - ...xed-organization-integrations-settings.tsx | 139 --- .../organization-integrations-setup.tsx | 188 --- .../organization-search-stats-period.tsx | 89 -- .../indexed/organization-source-people.tsx | 79 -- .../indexed/organization-source-stats.tsx | 226 ---- ...rganization-integrations-settings.test.tsx | 456 ------- .../organization-integrations-settings.tsx | 9 +- .../organization-search-status.ts | 27 - .../integrations/search-source-setup.test.tsx | 150 +-- .../integrations/search-source-setup.tsx | 19 +- .../providers/[connectorType]/page.test.tsx | 23 +- .../providers/[connectorType]/page.tsx | 3 +- .../[connectorType]/provider-detail.tsx | 125 +- .../sources/[connectorId]/search-params.ts | 14 - .../[connectorId]/source-detail.test.tsx | 68 +- .../sources/[connectorId]/source-detail.tsx | 168 +-- .../knowledge-search-results/indexed/index.ts | 1 - .../indexed/indexed-search-results.tsx | 214 ---- .../knowledge-search-results.tsx | 5 +- .../search-transitions.test.tsx | 49 +- .../search-integration-connection.tsx | 30 +- .../atlassian-source-setup-modal.tsx | 254 ---- .../search-sources/source-setup-modal.tsx | 131 -- .../home/hooks/use-chat.dom.test.tsx | 4 +- .../[workspaceId]/home/hooks/use-chat.ts | 7 +- .../add-connector-modal.tsx | 3 +- .../connector-documents.tsx | 12 - .../connector-selector-field.test.tsx | 51 - .../connector-selector-field.tsx | 22 +- .../connector-settings-fields.tsx | 3 +- .../use-connector-settings-form.test.tsx | 13 +- .../use-connector-settings-form.ts | 3 +- apps/sim/bootstrap.ts | 2 - apps/sim/connectors/coda/README.md | 2 +- .../search-params.ts | 22 - .../lib/copy/copy-resources.test.ts | 15 +- .../lib/copy/copy-resources.ts | 59 +- apps/sim/executor/utils/credential-token.ts | 5 +- apps/sim/hooks/queries/kb/connectors.test.ts | 2 +- apps/sim/hooks/queries/kb/connectors.ts | 222 +--- apps/sim/hooks/queries/kb/knowledge.test.ts | 47 +- apps/sim/hooks/queries/kb/knowledge.ts | 32 +- .../hooks/queries/organization-accounts.ts | 4 - .../queries/organization-search-stats.ts | 28 - .../hooks/queries/personal-source-setup.ts | 67 - apps/sim/hooks/queries/search-integrations.ts | 3 +- apps/sim/hooks/queries/selectors.test.tsx | 78 -- apps/sim/hooks/queries/selectors.ts | 127 +- .../reset-organization-search-access.test.ts | 55 +- .../utils/reset-organization-search-access.ts | 8 +- .../hooks/queries/utils/search-source-keys.ts | 12 - .../use-personal-source-account.test.tsx | 130 -- apps/sim/hooks/use-personal-source-account.ts | 135 -- ...use-search-integration-connection.test.tsx | 46 +- .../use-search-integration-connection.ts | 144 +-- .../api/contracts/desktop-source-connect.ts | 6 - .../lib/api/contracts/knowledge/connectors.ts | 176 +-- apps/sim/lib/api/contracts/knowledge/mcp.ts | 48 - .../knowledge/personal-integrations.ts | 13 - .../knowledge/personal-source-setup.test.ts | 36 - .../knowledge/personal-source-setup.ts | 86 -- .../api/contracts/knowledge/search-stats.ts | 76 -- .../mothership-management-tools.test.ts | 2 +- .../contracts/mothership-search-sources.ts | 1 - .../api/contracts/organization-accounts.ts | 26 - apps/sim/lib/api/contracts/workspaces.ts | 1 - apps/sim/lib/core/config/deployment-shape.ts | 2 - apps/sim/lib/core/config/env-flags.ts | 4 - apps/sim/lib/core/config/env.ts | 3 - apps/sim/lib/credential-groups/README.md | 4 +- .../organization-account-indexing.test.ts | 134 -- .../organization-account-indexing.ts | 62 - .../organization-settings-delegation.test.ts | 2 - .../self-enrollment-oauth.ts | 2 +- apps/sim/lib/credential-groups/service.ts | 78 +- .../slack-managed-user-scopes.ts | 5 +- .../slack-managed-users.test.ts | 38 +- .../credential-groups/slack-managed-users.ts | 15 +- .../credential-groups/slack-provider.test.ts | 59 +- .../lib/credential-groups/slack-provider.ts | 12 +- ...esolve-organization-personal-token.test.ts | 54 +- .../resolve-organization-personal-token.ts | 59 +- .../sim/lib/credentials/managed-oauth.test.ts | 5 +- apps/sim/lib/desktop/source-browser.ts | 16 +- .../__integration__/coda-live.integration.ts | 110 +- ...dormant-processing-recovery.integration.ts | 5 - .../dormant-search-processing.integration.ts | 7 +- .../embedding-insert-batches.integration.ts | 16 +- .../excluded-member-documents.integration.ts | 11 - .../github-member.integration.ts | 313 +---- .../gitlab-live.integration.ts | 614 +++++---- .../gmail-member.integration.ts | 11 - .../google-calendar-member.integration.ts | 11 - .../jira-member.integration.ts | 11 - .../kb-block-search.integration.ts | 41 +- .../knowledge-projection.integration.ts | 1111 ++++------------- .../organization-mcp-search.integration.ts | 988 --------------- ...rganization-search-overview.integration.ts | 432 ------- .../processing-lock-scope.integration.ts | 16 +- ...rovider-processing-recovery.integration.ts | 93 +- .../read-indexed-document.integration.ts | 343 ----- .../search-latency.integration.ts | 765 +----------- .../search-mcp-setup.integration.ts | 177 ++- .../search-source-pagination.integration.ts | 79 +- .../search-source-progress.integration.ts | 179 +-- .../search-source-setup.integration.ts | 134 +- .../stored-document-recovery.integration.ts | 68 +- .../unfilled-projection-source.integration.ts | 402 ------ ...orkspace-kb-document-access.integration.ts | 15 +- .../knowledge/access/predicate.integration.ts | 208 +-- .../lib/knowledge/access/predicate.test.ts | 116 -- apps/sim/lib/knowledge/api/route-policies.ts | 21 +- ...onnect-personal-search-integration.test.ts | 36 - .../connect-personal-search-integration.ts | 30 +- .../knowledge/application/connectors.test.ts | 91 -- .../lib/knowledge/application/connectors.ts | 109 +- .../lib/knowledge/application/documents.ts | 10 + .../knowledge/application/operations.test.ts | 15 - .../lib/knowledge/application/operations.ts | 79 -- .../organization-search-overview.test.ts | 181 --- .../organization-search-overview.ts | 307 ----- .../organization-search-stats.test.ts | 94 -- .../application/organization-search-stats.ts | 24 - .../personal-search-account.test.ts | 125 -- .../application/personal-search-account.ts | 68 - .../personal-search-integration-pages.ts | 49 - .../personal-search-integrations.test.ts | 117 +- .../personal-search-integrations.ts | 13 - .../application/personal-source-setup.test.ts | 271 ---- .../application/personal-source-setup.ts | 305 ----- .../application/search-diagnostics.ts | 27 - .../application/search-integrations.test.ts | 52 +- .../application/search-integrations.ts | 63 +- .../application/search-source-overview.ts | 220 ---- .../application/search-source-progress.ts | 110 -- .../application/search-sources.test.ts | 145 +-- .../knowledge/application/search-sources.ts | 144 +-- .../lib/knowledge/application/search.test.ts | 51 +- apps/sim/lib/knowledge/application/search.ts | 16 +- .../knowledge/application/sim-search.test.ts | 256 +--- .../lib/knowledge/application/sim-search.ts | 299 +---- .../connectors/indexing-policy.test.ts | 27 - .../knowledge/connectors/indexing-policy.ts | 5 +- .../organization-account-indexing.test.ts | 110 -- .../organization-account-indexing.ts | 136 -- .../connectors/viewer-member-sync-error.ts | 33 - .../connectors/viewer-source-accounts.test.ts | 76 -- .../connectors/viewer-source-accounts.ts | 76 -- apps/sim/lib/knowledge/constants.ts | 2 - .../lib/knowledge/documents/ocr-recovery.md | 2 +- .../lib/knowledge/mcp/route-handler.test.ts | 5 +- apps/sim/lib/knowledge/mcp/route-handler.ts | 3 +- .../lib/knowledge/mcp/server.protocol.test.ts | 165 +-- apps/sim/lib/knowledge/mcp/server.test.ts | 89 +- apps/sim/lib/knowledge/mcp/server.ts | 172 ++- .../lib/knowledge/orchestration/connectors.ts | 11 +- apps/sim/lib/knowledge/projection/enqueue.ts | 12 +- apps/sim/lib/knowledge/projection/run.ts | 23 +- .../lib/knowledge/search/activity-stats.ts | 100 -- .../sim/lib/knowledge/search/activity.test.ts | 67 - apps/sim/lib/knowledge/search/activity.ts | 38 - apps/sim/lib/knowledge/search/author.ts | 34 - .../knowledge/search/connection-attempt.ts | 1 - .../search/connection-target.test.ts | 5 +- .../lib/knowledge/search/connection-target.ts | 29 +- apps/sim/lib/knowledge/search/diagnostics.ts | 44 +- .../lib/knowledge/search/keyword-ranking.ts | 4 +- apps/sim/lib/knowledge/search/prewarm.test.ts | 9 +- apps/sim/lib/knowledge/search/prewarm.ts | 8 +- apps/sim/lib/knowledge/search/queries.ts | 26 +- .../search/source-vector-indexes.test.ts | 21 - .../knowledge/search/source-vector-indexes.ts | 24 - apps/sim/lib/knowledge/search/stats.test.ts | 63 - apps/sim/lib/knowledge/search/stats.ts | 74 -- .../load-search-integrations.test.ts | 75 +- .../application/load-search-integrations.ts | 10 - .../assistant/connected-account-tool.test.ts | 9 +- .../assistant/connected-account-tool.ts | 8 +- .../lib/mothership/assistant/tool-policy.ts | 5 +- apps/sim/lib/mothership/chat/payload.ts | 8 +- .../server/knowledge/workspace-search.test.ts | 229 +--- .../server/knowledge/workspace-search.ts | 237 +--- .../tools/server/search-sources.test.ts | 10 +- .../server/settings-connected-accounts.ts | 13 - apps/sim/lib/organizations/surface.test.ts | 21 + apps/sim/lib/organizations/surface.ts | 4 + .../application/execute-selector.test.ts | 38 - .../selectors/application/execute-selector.ts | 30 +- .../lib/selectors/server/credentials.test.ts | 88 -- apps/sim/lib/selectors/server/credentials.ts | 54 - .../providers/credential-bundle.test.ts | 49 - .../server/providers/credential-bundle.ts | 15 +- apps/sim/lib/selectors/server/types.ts | 5 - apps/sim/lib/selectors/types.ts | 7 - apps/sim/lib/sim-search/connectors.ts | 38 +- apps/sim/lib/sim-search/indexed/README.md | 39 - .../documents/read-indexed-document.ts | 230 ---- .../documents/read-search-document.test.ts | 267 ---- .../indexed/documents/read-search-document.ts | 173 --- apps/sim/lib/sim-search/indexed/gate.ts | 42 - apps/sim/lib/sim-search/indexed/index.ts | 18 - .../personal-account-ownership.ts | 29 - .../integrations/personal-inventory.ts | 43 - .../personal-search-integrations.ts | 164 --- .../sim-search/indexed/mcp/register-tools.ts | 148 --- .../indexed/retrieval/access-plan.ts | 188 --- .../lib/sim-search/indexed/retrieval/index.ts | 10 - .../sim-search/indexed/retrieval/keyword.ts | 325 ----- .../sim-search/indexed/retrieval/legs.test.ts | 995 --------------- .../lib/sim-search/indexed/retrieval/legs.ts | 128 -- .../sim-search/indexed/retrieval/permitted.ts | 424 ------- .../indexed/retrieval/projection-access.ts | 234 ---- .../indexed/retrieval/projection-fill.test.ts | 78 -- .../indexed/retrieval/projection-fill.ts | 102 -- .../retrieval/source-vector-indexes.ts | 33 - .../retrieval/tin-keyword-readiness.test.ts | 22 - .../indexed/retrieval/tin-keyword.test.ts | 41 - .../indexed/retrieval/tin-keyword.ts | 65 - .../indexed/retrieval/tin-query.test.ts | 23 - .../sim-search/indexed/retrieval/tin-query.ts | 195 --- .../sim-search/indexed/retrieval/vector.ts | 491 -------- .../search/scoped-search.activity.test.ts | 135 -- .../indexed/search/scoped-search.test.ts | 100 -- .../indexed/search/scoped-search.ts | 184 --- apps/sim/lib/sim-search/live/README.md | 2 +- apps/sim/lib/sim-search/live/application.ts | 8 - apps/sim/lib/sim-search/live/scopes.ts | 2 +- .../lib/sim-search/personal-source-setup.ts | 2 - apps/sim/lib/slack-search/connections.test.ts | 3 +- .../__integration__/fork-sync.integration.ts | 203 +++ .../fixtures/desktop-source-connect.tsx | 25 +- apps/sim/tools/index.test.ts | 3 - apps/sim/tools/index.ts | 4 +- docker/app.Dockerfile | 2 - packages/db/knowledge-projection.test.ts | 245 ---- packages/db/knowledge-projection.ts | 272 +--- packages/db/schema.ts | 8 +- .../search-embedding-retirement.md | 4 +- .../src/mocks/deployment-shape.mock.ts | 2 - packages/testing/src/mocks/env-flags.mock.ts | 2 - .../src/mocks/indexed-org-search.mock.ts | 22 - .../src/mocks/kb-connectors-queries.mock.ts | 15 +- scripts/test-patterns-baseline.json | 1 - 276 files changed, 2147 insertions(+), 23554 deletions(-) delete mode 100644 apps/sim/app/api/knowledge/sim-search/connect/route.ts delete mode 100644 apps/sim/app/api/knowledge/sim-search/integrations/overview/route.ts delete mode 100644 apps/sim/app/api/knowledge/sim-search/personal-source-setup/route.ts delete mode 100644 apps/sim/app/api/knowledge/sim-search/sources/overview/route.ts delete mode 100644 apps/sim/app/api/knowledge/sim-search/sources/progress/route.ts delete mode 100644 apps/sim/app/api/knowledge/sim-search/stats/route.ts delete mode 100644 apps/sim/app/api/organizations/[id]/connected-accounts/indexing/route.ts delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/github-member-integration.tsx delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/index.ts delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/source-status.ts delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx delete mode 100644 apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts delete mode 100644 apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx delete mode 100644 apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx delete mode 100644 apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/search-params.ts delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup.tsx delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period.tsx delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people.tsx delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx delete mode 100644 apps/sim/app/o/[organizationId]/settings/components/integrations/organization-search-status.ts delete mode 100644 apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/search-params.ts delete mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/index.ts delete mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/indexed-search-results.tsx delete mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/search-sources/atlassian-source-setup-modal.tsx delete mode 100644 apps/sim/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal.tsx delete mode 100644 apps/sim/ee/organization-search-stats/search-params.ts delete mode 100644 apps/sim/hooks/queries/organization-search-stats.ts delete mode 100644 apps/sim/hooks/queries/personal-source-setup.ts delete mode 100644 apps/sim/hooks/use-personal-source-account.test.tsx delete mode 100644 apps/sim/hooks/use-personal-source-account.ts delete mode 100644 apps/sim/lib/api/contracts/knowledge/personal-source-setup.test.ts delete mode 100644 apps/sim/lib/api/contracts/knowledge/personal-source-setup.ts delete mode 100644 apps/sim/lib/api/contracts/knowledge/search-stats.ts delete mode 100644 apps/sim/lib/credential-groups/application/organization-account-indexing.test.ts delete mode 100644 apps/sim/lib/credential-groups/application/organization-account-indexing.ts delete mode 100644 apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts delete mode 100644 apps/sim/lib/knowledge/__integration__/organization-search-overview.integration.ts delete mode 100644 apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts delete mode 100644 apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts delete mode 100644 apps/sim/lib/knowledge/application/organization-search-overview.test.ts delete mode 100644 apps/sim/lib/knowledge/application/organization-search-overview.ts delete mode 100644 apps/sim/lib/knowledge/application/organization-search-stats.test.ts delete mode 100644 apps/sim/lib/knowledge/application/organization-search-stats.ts delete mode 100644 apps/sim/lib/knowledge/application/personal-search-account.test.ts delete mode 100644 apps/sim/lib/knowledge/application/personal-search-account.ts delete mode 100644 apps/sim/lib/knowledge/application/personal-search-integration-pages.ts delete mode 100644 apps/sim/lib/knowledge/application/personal-source-setup.test.ts delete mode 100644 apps/sim/lib/knowledge/application/personal-source-setup.ts delete mode 100644 apps/sim/lib/knowledge/application/search-source-overview.ts delete mode 100644 apps/sim/lib/knowledge/application/search-source-progress.ts delete mode 100644 apps/sim/lib/knowledge/connectors/indexing-policy.test.ts delete mode 100644 apps/sim/lib/knowledge/connectors/organization-account-indexing.test.ts delete mode 100644 apps/sim/lib/knowledge/connectors/organization-account-indexing.ts delete mode 100644 apps/sim/lib/knowledge/connectors/viewer-member-sync-error.ts delete mode 100644 apps/sim/lib/knowledge/connectors/viewer-source-accounts.test.ts delete mode 100644 apps/sim/lib/knowledge/connectors/viewer-source-accounts.ts delete mode 100644 apps/sim/lib/knowledge/search/activity-stats.ts delete mode 100644 apps/sim/lib/knowledge/search/activity.test.ts delete mode 100644 apps/sim/lib/knowledge/search/activity.ts delete mode 100644 apps/sim/lib/knowledge/search/author.ts delete mode 100644 apps/sim/lib/knowledge/search/source-vector-indexes.test.ts delete mode 100644 apps/sim/lib/knowledge/search/source-vector-indexes.ts delete mode 100644 apps/sim/lib/knowledge/search/stats.test.ts delete mode 100644 apps/sim/lib/knowledge/search/stats.ts delete mode 100644 apps/sim/lib/sim-search/indexed/README.md delete mode 100644 apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts delete mode 100644 apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/documents/read-search-document.ts delete mode 100644 apps/sim/lib/sim-search/indexed/gate.ts delete mode 100644 apps/sim/lib/sim-search/indexed/index.ts delete mode 100644 apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts delete mode 100644 apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts delete mode 100644 apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts delete mode 100644 apps/sim/lib/sim-search/indexed/mcp/register-tools.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/index.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/keyword.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/legs.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/permitted.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/tin-query.ts delete mode 100644 apps/sim/lib/sim-search/indexed/retrieval/vector.ts delete mode 100644 apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts delete mode 100644 apps/sim/lib/sim-search/indexed/search/scoped-search.ts delete mode 100644 apps/sim/lib/sim-search/personal-source-setup.ts delete mode 100644 packages/db/knowledge-projection.test.ts delete mode 100644 packages/testing/src/mocks/indexed-org-search.mock.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0a30f3f9eec..547db6f3459 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -222,7 +222,6 @@ jobs: platforms: linux/amd64 tags: ${{ steps.login-ecr.outputs.registry }}/${{ steps.ecr-repo.outputs.name }}:${{ github.sha }}-dev build-args: | - SIM_SEARCH_LIVE_DEFAULT=true MSHIP_PLAN_MODE_DEFAULT=true max-cache-size-mb: ${{ matrix.cache_mb }} diff --git a/.github/workflows/desktop-e2e.yml b/.github/workflows/desktop-e2e.yml index e3ed8a6a745..a86ea559a71 100644 --- a/.github/workflows/desktop-e2e.yml +++ b/.github/workflows/desktop-e2e.yml @@ -28,7 +28,6 @@ on: - 'apps/sim/hooks/queries/personal-search-integrations.ts' - 'apps/sim/hooks/use-search-integration-connection.ts' - 'apps/sim/hooks/use-github-installation-setup.ts' - - 'apps/sim/app/o/**/integrations/indexed/use-member-enrollment.ts' - 'apps/sim/lib/api/contracts/desktop-source-connect.ts' - 'apps/sim/scripts/fixtures/desktop-source-connect.tsx' - 'apps/sim/app/workspace/**/browser-session/**' diff --git a/apps/docs/content/docs/platform/self-hosting/integrations-oauth.mdx b/apps/docs/content/docs/platform/self-hosting/integrations-oauth.mdx index 392b0785f8d..99c522a3f49 100644 --- a/apps/docs/content/docs/platform/self-hosting/integrations-oauth.mdx +++ b/apps/docs/content/docs/platform/self-hosting/integrations-oauth.mdx @@ -207,7 +207,7 @@ The private key must include its PEM header, footer, and contents. Sim accepts a Keep the private key and client secret in the deployment's server configuration; organization admins select installations in Sim without entering these secrets. -The live Search application needs these five variables. If a worker also indexes ordinary GitHub knowledge bases, or Search has explicitly reverted to `SIM_SEARCH_LIVE=false`, configure the variables in that worker’s matching environment too. Updating the app’s secret store does not update a separately configured worker. Live Search does not schedule a GitHub indexing worker. +The live Search application needs these five variables. If a worker also indexes ordinary GitHub knowledge bases, configure the variables in that worker’s matching environment too. Updating the app’s secret store does not update a separately configured worker. Live Search does not schedule a GitHub indexing worker. The **Client ID** is different from the numeric **App ID**. Use credentials from **Developer settings → GitHub Apps**. `GITHUB_CLIENT_ID` and `GITHUB_CLIENT_SECRET` belong to the separate GitHub sign-in integration and remain unchanged. Search does not read `GITHUB_REPO_CLIENT_ID` or `GITHUB_REPO_CLIENT_SECRET`. diff --git a/apps/docs/content/docs/search/index.mdx b/apps/docs/content/docs/search/index.mdx index cb498ac484c..59c23f27f13 100644 --- a/apps/docs/content/docs/search/index.mdx +++ b/apps/docs/content/docs/search/index.mdx @@ -80,6 +80,6 @@ Conversations remain private to their author. Connecting an external account doe ## Legacy indexing and workspace knowledge bases -Live Search is the default. An operator can explicitly set `SIM_SEARCH_LIVE=false` to restore the legacy indexed Search backend. Its content sync, document processing, and stored permission maintenance are specific to that mode. These guides describe live Search. +Enterprise Search uses live provider queries. Search sources do not run content indexing or stored permission maintenance; ordinary knowledge-base connectors retain their indexing behavior. Ordinary workspace [knowledge bases](/knowledgebase) still ingest, chunk, and index documents for their own features. Their connector setup, processing status, and access settings remain separate. diff --git a/apps/docs/content/docs/search/mcp.mdx b/apps/docs/content/docs/search/mcp.mdx index 25fab917a3d..0e838e09fca 100644 --- a/apps/docs/content/docs/search/mcp.mdx +++ b/apps/docs/content/docs/search/mcp.mdx @@ -44,7 +44,7 @@ The connection applies to the organization whose URL you copied. Sim checks curr `chat` accepts questions up to 8,192 characters. Tool responses are limited to 1 MiB. API rate limits apply; follow any retry delay. Provider failures or incomplete retrieval are reported explicitly, and the tools do not fall back to an old index when a live provider is unavailable. -These are the live-backend schemas. If an operator explicitly selects legacy Search with `SIM_SEARCH_LIVE=false`, the server advertises its indexed search/read schemas instead. Refresh your MCP client's tool discovery after changing backends. +These are the Search MCP schemas. Refresh tool discovery in clients that previously connected to the retired indexed backend. ## Reconnect or revoke access diff --git a/apps/sim/app/api/knowledge/connectors/directory-sync/route.test.ts b/apps/sim/app/api/knowledge/connectors/directory-sync/route.test.ts index f9c6c7c53ae..d3fded53e48 100644 --- a/apps/sim/app/api/knowledge/connectors/directory-sync/route.test.ts +++ b/apps/sim/app/api/knowledge/connectors/directory-sync/route.test.ts @@ -4,7 +4,6 @@ import { hasMockCondition, resetEnvFlagsMock, schemaMock, - setEnvFlags, } from '@sim/testing' import { authInternalMock, authInternalMockFns } from '@sim/testing/mocks/auth-internal.mock' import { dbChainMockFns } from '@sim/testing/mocks/database.mock' @@ -108,24 +107,6 @@ describe('connector directory sync scheduler', () => { await expect(run()).resolves.toMatchObject({ dispatched: 1, failed: 1 }) }) - it.each([true, false])( - 'excludes Search directories from scheduled pages only when live Search is %s', - async (liveSearch) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: liveSearch }) - mockConnectorRows.mockResolvedValue([]) - await run() - expect( - hasMockCondition( - mockWhere.mock.calls[0][0], - (node) => - node.type === 'eq' && - node.left === schemaMock.knowledgeBase.isSearchIndex && - node.right === false - ) - ).toBe(liveSearch) - } - ) - it('does not enqueue a connector another scheduler claimed or paused', async () => { mockConnectorRows.mockResolvedValue([connector()]) mockClaim.mockResolvedValueOnce([]) diff --git a/apps/sim/app/api/knowledge/search/route.ts b/apps/sim/app/api/knowledge/search/route.ts index 1186b6da0f0..610507ea53e 100644 --- a/apps/sim/app/api/knowledge/search/route.ts +++ b/apps/sim/app/api/knowledge/search/route.ts @@ -6,68 +6,9 @@ import { } from '@/lib/api/server/routes' import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { DEFAULT_RERANKER_MODEL } from '@/lib/knowledge/reranker-models' -import { sourceAuthor } from '@/lib/knowledge/search/author' -import { searchScopedKnowledge } from '@/lib/sim-search/indexed' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { searchLiveKnowledge } from '@/lib/sim-search/live/application' -const DIRECT_SEARCH_VECTOR_BUDGET_MS = 3000 - -const indexedSearchRoute = defineInternalJsonRoute({ - contract: searchWorkspaceKnowledgeContract, - auth: internalSessionAuth, - operation: knowledgeOperations.search, - rateLimit: internalRateLimits.none({ - reason: - 'A person typing queries; the embedding call is metered against the canonical search owner', - }), - errorPolicy: internalKnowledgeErrorPolicies.search, - mapInput: ({ body }, { request }) => ({ - workspaceId: body.workspaceId, - organizationId: body.organizationId, - filters: body.filters, - query: body.query, - topK: body.topK, - allowPartialResults: true, - vectorBudgetMs: DIRECT_SEARCH_VECTOR_BUDGET_MS, - /** - * A person's search is reranked by a cross-encoder whenever the workspace or the platform - * holds a key for one; the use case checks that before spending a call, and reranking stays - * best-effort, so a provider outage leaves the fused order in place. - */ - rerankerEnabled: true, - rerankerModel: DEFAULT_RERANKER_MODEL, - surface: 'dashboard' as const, - signal: request.signal, - }), - useCase: searchScopedKnowledge, - present: ({ results, knowledgeBases, retrieval }, { input }) => { - const knowledgeBaseNames = new Map(knowledgeBases.map((kb) => [kb.id, kb.name])) - return { - success: true as const, - data: { - query: input.query ?? '', - retrieval, - results: results.map((result) => ({ - documentId: result.documentId, - knowledgeBaseId: result.knowledgeBaseId, - knowledgeBaseName: knowledgeBaseNames.get(result.knowledgeBaseId) ?? '', - documentName: result.documentName, - sourceUrl: result.sourceUrl, - connectorType: result.connectorType, - sourceModifiedAt: result.sourceModifiedAt?.toISOString() ?? null, - author: sourceAuthor(result.metadata), - content: result.content, - chunkIndex: result.chunkIndex, - similarity: result.similarity, - })), - }, - } - }, -}) - -const liveSearchRoute = defineInternalJsonRoute({ +export const POST = defineInternalJsonRoute({ contract: searchWorkspaceKnowledgeContract, auth: internalSessionAuth, operation: knowledgeOperations.search, @@ -80,6 +21,3 @@ const liveSearchRoute = defineInternalJsonRoute({ useCase: searchLiveKnowledge, present: (data) => ({ success: true as const, data }), }) - -/** Indexed organization search is dormant unless its gate is on; Live Search serves otherwise. */ -export const POST = isIndexedOrgSearchEnabled() ? indexedSearchRoute : liveSearchRoute diff --git a/apps/sim/app/api/knowledge/sim-search/connect/route.ts b/apps/sim/app/api/knowledge/sim-search/connect/route.ts deleted file mode 100644 index 521b6b570ef..00000000000 --- a/apps/sim/app/api/knowledge/sim-search/connect/route.ts +++ /dev/null @@ -1,20 +0,0 @@ -import { connectSimSearchConnectorContract } from '@/lib/api/contracts/knowledge' -import { - defineInternalJsonRoute, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { connectSimSearchConnector } from '@/lib/knowledge/application/sim-search' - -export const POST = defineInternalJsonRoute({ - contract: connectSimSearchConnectorContract, - auth: internalSessionAuth, - operation: knowledgeOperations.simSearchConnect, - rateLimit: internalRateLimits.none({ reason: 'One click per source; mints a single-use link' }), - errorPolicy: internalKnowledgeErrorPolicies.connectAccount, - mapInput: ({ body }) => body, - useCase: connectSimSearchConnector, - present: (result) => ({ success: true as const, data: result }), -}) diff --git a/apps/sim/app/api/knowledge/sim-search/integrations/overview/route.ts b/apps/sim/app/api/knowledge/sim-search/integrations/overview/route.ts deleted file mode 100644 index 325f40ea7c8..00000000000 --- a/apps/sim/app/api/knowledge/sim-search/integrations/overview/route.ts +++ /dev/null @@ -1,23 +0,0 @@ -import { readOrganizationSearchOverviewContract } from '@/lib/api/contracts/knowledge/connectors' -import { - defineInternalJsonRoute, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { readOrganizationSearchOverview } from '@/lib/knowledge/application/organization-search-overview' - -export const GET = defineInternalJsonRoute({ - contract: readOrganizationSearchOverviewContract, - auth: internalSessionAuth, - operation: knowledgeOperations.readOrganizationSearchOverview, - rateLimit: internalRateLimits.none({ - reason: 'Bounded provider operational aggregates for organization integration settings', - }), - errorPolicy: internalKnowledgeErrorPolicies.connectors, - mapInput: ({ query }) => query, - useCase: readOrganizationSearchOverview, - present: (overview) => ({ success: true as const, data: overview }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) diff --git a/apps/sim/app/api/knowledge/sim-search/personal-source-setup/route.ts b/apps/sim/app/api/knowledge/sim-search/personal-source-setup/route.ts deleted file mode 100644 index 8fff7a8ff6e..00000000000 --- a/apps/sim/app/api/knowledge/sim-search/personal-source-setup/route.ts +++ /dev/null @@ -1,70 +0,0 @@ -import { - listPersonalSourceSetupAccountsContract, - personalSourceSetupContract, -} from '@/lib/api/contracts/knowledge/personal-source-setup' -import { - defineInternalJsonRoute, - extendInternalErrorPolicy, - internalErrorResponse, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { - listPersonalSourceSetupAccounts, - personalSourceSetup, -} from '@/lib/knowledge/application/personal-source-setup' -import { - SelectorConnectionUnavailableError, - SelectorContextUnavailableError, - SelectorOptionsUnavailableError, -} from '@/lib/selectors/server/errors' -import { IntegrationNotAllowedError } from '@/ee/access-control/utils/permission-check' - -const errorPolicy = extendInternalErrorPolicy( - internalKnowledgeErrorPolicies.connectAccount, - (error) => { - if (error instanceof SelectorConnectionUnavailableError) - return internalErrorResponse(error.status, { - error: 'Reconnect your account to choose projects or spaces', - }) - if (error instanceof SelectorContextUnavailableError) - return internalErrorResponse(400, { - error: 'Enter your Atlassian site to choose projects or spaces', - }) - if (error instanceof SelectorOptionsUnavailableError) - return internalErrorResponse(error.status, { - error: - 'Could not load projects or spaces. Check the site and account access, then try again.', - }) - if (error instanceof IntegrationNotAllowedError) - return internalErrorResponse(403, { error: error.message }) - return null - } -) - -export const GET = defineInternalJsonRoute({ - contract: listPersonalSourceSetupAccountsContract, - auth: internalSessionAuth, - operation: knowledgeOperations.listPersonalSourceSetupAccounts, - rateLimit: internalRateLimits.user({ bucketName: 'knowledge.search.personal-setup.accounts' }), - errorPolicy, - mapInput: ({ query }) => query, - useCase: listPersonalSourceSetupAccounts, - present: (data) => ({ success: true as const, data }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) - -export const POST = defineInternalJsonRoute({ - contract: personalSourceSetupContract, - auth: internalSessionAuth, - operation: knowledgeOperations.personalSourceSetup, - rateLimit: internalRateLimits.user({ bucketName: 'knowledge.search.personal-setup' }), - errorPolicy, - parseOptions: { maxBodyBytes: 384 * 1024 }, - mapInput: ({ body }) => body, - useCase: personalSourceSetup, - present: (data) => ({ success: true as const, data }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) diff --git a/apps/sim/app/api/knowledge/sim-search/sources/overview/route.ts b/apps/sim/app/api/knowledge/sim-search/sources/overview/route.ts deleted file mode 100644 index f6bdf9abb5e..00000000000 --- a/apps/sim/app/api/knowledge/sim-search/sources/overview/route.ts +++ /dev/null @@ -1,23 +0,0 @@ -import { readSearchSourceOverviewContract } from '@/lib/api/contracts/knowledge/connectors' -import { - defineInternalJsonRoute, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { readSearchSourceOverview } from '@/lib/knowledge/application/search-source-overview' - -export const GET = defineInternalJsonRoute({ - contract: readSearchSourceOverviewContract, - auth: internalSessionAuth, - operation: knowledgeOperations.readSearchSourceOverview, - rateLimit: internalRateLimits.none({ - reason: 'Bounded provider existence probes for source setup and indexing progress', - }), - errorPolicy: internalKnowledgeErrorPolicies.connectors, - mapInput: ({ query }) => query, - useCase: readSearchSourceOverview, - present: (overview) => ({ success: true as const, data: overview }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) diff --git a/apps/sim/app/api/knowledge/sim-search/sources/progress/route.ts b/apps/sim/app/api/knowledge/sim-search/sources/progress/route.ts deleted file mode 100644 index a6a793a28af..00000000000 --- a/apps/sim/app/api/knowledge/sim-search/sources/progress/route.ts +++ /dev/null @@ -1,23 +0,0 @@ -import { readSearchSourceProgressContract } from '@/lib/api/contracts/knowledge/connectors' -import { - defineInternalJsonRoute, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { readSearchSourceProgress } from '@/lib/knowledge/application/search-source-progress' - -export const POST = defineInternalJsonRoute({ - contract: readSearchSourceProgressContract, - auth: internalSessionAuth, - operation: knowledgeOperations.readSearchSourceProgress, - rateLimit: internalRateLimits.none({ - reason: 'Bounded viewer-authorized indexing progress polling', - }), - errorPolicy: internalKnowledgeErrorPolicies.connectors, - mapInput: ({ body }) => body, - useCase: readSearchSourceProgress, - present: ({ sources }) => ({ success: true as const, data: sources }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) diff --git a/apps/sim/app/api/knowledge/sim-search/sources/route.test.ts b/apps/sim/app/api/knowledge/sim-search/sources/route.test.ts index 4b2f0c54243..d255e78bcf6 100644 --- a/apps/sim/app/api/knowledge/sim-search/sources/route.test.ts +++ b/apps/sim/app/api/knowledge/sim-search/sources/route.test.ts @@ -5,32 +5,16 @@ import type { SearchSourceSummary } from '@/lib/api/contracts/knowledge/connecto const mocks = vi.hoisted(() => ({ execute: vi.fn(), - overview: vi.fn(), - adminOverview: vi.fn(), -})) -vi.mock('@/lib/knowledge/application/organization-search-overview', () => ({ - readOrganizationSearchOverview: { - operation: { id: 'knowledge.search.integrations.overview' }, - execute: mocks.adminOverview, - }, })) vi.mock('@/lib/knowledge/application/search-sources', () => ({ listSearchSources: { operation: { id: 'knowledge.search.sources.list' }, execute: mocks.execute }, })) -vi.mock('@/lib/knowledge/application/search-source-overview', () => ({ - readSearchSourceOverview: { - operation: { id: 'knowledge.search.sources.overview' }, - execute: mocks.overview, - }, -})) vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) vi.mock('@/lib/knowledge/application/upload-sessions', () => ({ KnowledgeDocumentUnsupportedMediaTypeError: class extends Error {}, })) import { NoWorkspaceAccessError } from '@/lib/core/application/workspace-authorization' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { GET as getAdminOverview } from '@/app/api/knowledge/sim-search/integrations/overview/route' import { GET } from '@/app/api/knowledge/sim-search/sources/route' const WORKSPACE_ID = '7d28e5e2-fb03-4118-9c52-4ab77ccff369' @@ -42,15 +26,7 @@ const source = { accessMode: 'admin', availability: 'available', enabled: true, - isSyncing: false, - lastSyncAt: null, - hasSyncError: false, - hasViewerDocuments: false, - viewerFailedDocumentCount: 0, - viewerEmailVerified: true, - viewerAccounts: [], - connectionRequired: false, - viewerMembership: null, + isGitHubInstallation: false, } satisfies SearchSourceSummary beforeEach(() => { @@ -112,21 +88,3 @@ describe('Search pagination boundary', () => { expect(mocks.execute).not.toHaveBeenCalled() }) }) - -describe('organization administration overview boundary', () => { - it('preserves a role refusal without exposing health data', async () => { - mocks.adminOverview.mockRejectedValue( - new OrchestrationError('forbidden', 'Organization administrator access is required') - ) - const response = await getAdminOverview( - createMockRequest( - 'GET', - undefined, - {}, - `http://localhost/api/knowledge/sim-search/integrations/overview?organizationId=${WORKSPACE_ID}` - ) - ) - expect(response.status).toBe(403) - expect(await response.json()).not.toHaveProperty('data') - }) -}) diff --git a/apps/sim/app/api/knowledge/sim-search/stats/route.ts b/apps/sim/app/api/knowledge/sim-search/stats/route.ts deleted file mode 100644 index e7ae7eb9777..00000000000 --- a/apps/sim/app/api/knowledge/sim-search/stats/route.ts +++ /dev/null @@ -1,24 +0,0 @@ -import { readOrganizationSearchStatsContract } from '@/lib/api/contracts/knowledge/search-stats' -import { - defineInternalJsonRoute, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { internalKnowledgeErrorPolicies } from '@/lib/knowledge/api/route-policies' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { readOrganizationSearchStats } from '@/lib/knowledge/application/organization-search-stats' - -export const GET = defineInternalJsonRoute({ - contract: readOrganizationSearchStatsContract, - auth: internalSessionAuth, - operation: knowledgeOperations.readOrganizationSearchStats, - rateLimit: internalRateLimits.user({ - bucketName: 'organization-search-stats', - config: { maxTokens: 30, refillRate: 30, refillIntervalMs: 60_000 }, - }), - errorPolicy: internalKnowledgeErrorPolicies.connectors, - mapInput: ({ query }) => query, - useCase: readOrganizationSearchStats, - present: (data) => ({ success: true as const, data }), - staticResponseHeaders: { 'Cache-Control': 'private, no-store' }, -}) diff --git a/apps/sim/app/api/organizations/[id]/connected-accounts/indexing/route.ts b/apps/sim/app/api/organizations/[id]/connected-accounts/indexing/route.ts deleted file mode 100644 index d79a286c00f..00000000000 --- a/apps/sim/app/api/organizations/[id]/connected-accounts/indexing/route.ts +++ /dev/null @@ -1,22 +0,0 @@ -import { updateOrganizationAccountIndexingContract } from '@/lib/api/contracts/organization-accounts' -import { - defineInternalJsonRoute, - internalOrchestrationErrorPolicy, - internalRateLimits, - internalSessionAuth, -} from '@/lib/api/server/routes' -import { - updateOrganizationAccountIndexing, - updateOrganizationAccountIndexingOperation, -} from '@/lib/credential-groups/application/organization-account-indexing' - -export const PUT = defineInternalJsonRoute({ - contract: updateOrganizationAccountIndexingContract, - auth: internalSessionAuth, - operation: updateOrganizationAccountIndexingOperation, - rateLimit: internalRateLimits.user({ bucketName: 'organization-account-indexing' }), - errorPolicy: internalOrchestrationErrorPolicy, - mapInput: ({ params, body }) => ({ organizationId: params.id, ...body }), - useCase: updateOrganizationAccountIndexing, - present: ({ enabled, knowledgeBaseIds }) => ({ enabled, knowledgeBaseIds }), -}) diff --git a/apps/sim/app/api/v1/knowledge/search/route.ts b/apps/sim/app/api/v1/knowledge/search/route.ts index 2a1b97b5b3e..91c1a56a91c 100644 --- a/apps/sim/app/api/v1/knowledge/search/route.ts +++ b/apps/sim/app/api/v1/knowledge/search/route.ts @@ -24,7 +24,6 @@ import { import { getDocumentTagDefinitions } from '@/lib/knowledge/tags/service' import { buildUndefinedTagsError, validateTagValue } from '@/lib/knowledge/tags/utils' import type { StructuredFilter } from '@/lib/knowledge/types' -import { usesIndexedRetrieval } from '@/lib/sim-search/indexed/gate' import { checkKnowledgeBaseAccess, type KnowledgeBaseAccessResult } from '@/app/api/knowledge/utils' import { handleError, resolveV1KnowledgeReadAccess } from '@/app/api/v1/knowledge/utils' import { @@ -251,7 +250,6 @@ export const POST = withRouteHandler(async (request: NextRequest) => { accessProvider, searchMode, boostRecency, - indexedRetrieval: usesIndexedRetrieval(accessibleKbs), structuredFilters, }) } else if (hasQuery) { @@ -268,7 +266,6 @@ export const POST = withRouteHandler(async (request: NextRequest) => { accessProvider, searchMode, boostRecency, - indexedRetrieval: usesIndexedRetrieval(accessibleKbs), query, queryVector: { vector: JSON.stringify(queryEmbeddingResult.embedding), diff --git a/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx b/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx index 72e85c1c61c..8e365f58e08 100644 --- a/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx +++ b/apps/sim/app/o/[organizationId]/home/components/composer/composer.test.tsx @@ -18,7 +18,6 @@ import type { useSpeechToText } from '@/hooks/use-speech-to-text' import { useMothershipEffortStore } from '@/stores/mothership-effort/store' const mocks = vi.hoisted(() => ({ - live: false, plan: false, advanced: false, speech: vi.fn(), @@ -90,10 +89,9 @@ import type { ChatRequestMode } from '@/app/workspace/[workspaceId]/home/types' import { FeatureFlagsProvider } from '@/app/workspace/[workspaceId]/providers/feature-flags-provider' import { useFileAttachments } from '@/app/workspace/[workspaceId]/w/[workflowId]/components/panel/components/copilot/components/user-input/hooks/use-file-attachments' -const liveShape = () => - createMockDeploymentShape({ features: { liveEnterpriseSearch: mocks.live } }) -deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(liveShape) -deploymentShapeMockFns.mockGetDeploymentShape.mockImplementation(liveShape) +const deploymentShape = () => createMockDeploymentShape() +deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(deploymentShape) +deploymentShapeMockFns.mockGetDeploymentShape.mockImplementation(deploymentShape) organizationProviderMockFns.mockUseOrganizationContext.mockReturnValue({ organization: { id: 'organization-a' }, }) @@ -109,7 +107,6 @@ beforeEach(() => { modelSelection: { model: 'gpt-6-astra', fastMode: false }, }) mocks.plan = false - mocks.live = false vi.clearAllMocks() mocks.workspaces = [ { diff --git a/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx b/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx index 3c18b5fb2e3..f9e42b544c8 100644 --- a/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx +++ b/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx @@ -6,11 +6,15 @@ import { ArrowRight } from '@sim/emcn/icons' import { HomeSection } from '@/components/home/home-section' import { SettingsGuardedLink } from '@/components/settings/settings-guarded-link' import { OAUTH_SEARCH_READ_SCOPE, oauthScopeSatisfies } from '@/lib/auth/oauth-provider' -import type { ResourceScope } from '@/lib/core/resource-scope' import { organizationRoutes } from '@/lib/navigation/paths' +import { + liveSearchProviderForCredential, + supportsLiveSearchMode, +} from '@/lib/sim-search/live/provider-catalog' import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' -import { useSearchSourceOverview } from '@/hooks/queries/kb/connectors' import { useAuthorizedApps } from '@/hooks/queries/oauth-provider' +import { useOrganizationAccounts } from '@/hooks/queries/organization-accounts' +import { useSearchIntegrations } from '@/hooks/queries/search-integrations' type StepId = 'connect-integration' | 'connect-sim-search' @@ -67,14 +71,18 @@ function StepMark({ complete }: { complete: boolean }) { * The organization home's onboarding list under the composer. Same chrome as * the workspace home's suggested actions: a hover-revealed disclosure header * over hairline-separated rows. Each step leads to the page that completes it, - * and reads as done from the organization's real state: a source the viewer can - * search and an OAuth app authorized to use Search. + * and reads as done from the organization's real state: a connected account and an OAuth app authorized to use Search. */ export function GetStarted() { - const { organization, viewer } = useOrganizationContext() + const { organization, viewer, connectedAccountsAvailable } = useOrganizationContext() const routes = organizationRoutes(organization.id) - const scope: ResourceScope = { kind: 'organization', organizationId: organization.id } - const { data: overview } = useSearchSourceOverview(scope) + const canConnectIntegrations = viewer.canConnectSearchIntegrations && connectedAccountsAvailable + const { data: accounts } = useOrganizationAccounts( + canConnectIntegrations ? organization.id : undefined + ) + const { data: integrations } = useSearchIntegrations(organization.id, { + enabled: canConnectIntegrations, + }) const { data: authorizedApps, fetchNextPage, @@ -86,6 +94,46 @@ export function GetStarted() { authorizedApps?.pages.some((page) => page.apps.some((app) => oauthScopeSatisfies(app.scopes, OAUTH_SEARCH_READ_SCOPE)) ) ?? false + const approvedProviders = new Set( + integrations + ?.filter((integration) => integration.approved && integration.available !== false) + .map((integration) => integration.connectorType) + ) + const readyOptions = new Set( + accounts?.credentialGroup?.options + .filter((option) => { + const provider = liveSearchProviderForCredential(option.provider) + return ( + option.status === 'active' && + option.configurationStatus === 'ready' && + provider && + supportsLiveSearchMode(provider, 'member') && + approvedProviders.has(provider) + ) + }) + .map((option) => option.id) + ) + const readyMcpServers = new Set( + accounts?.credentialGroup?.mcpServers + .filter((server) => { + const provider = liveSearchProviderForCredential(`mcp:${server.managedConnectorId}`) + return ( + server.enabled && + provider && + approvedProviders.has(provider) && + accounts.availableMcpConnectors.some((id) => id === server.managedConnectorId) + ) + }) + .map((server) => server.id) + ) + const hasSearchConnection = + accounts?.credentialGroup?.status === 'active' && + (accounts.viewerAccounts?.some( + (account) => account.status === 'active' && readyOptions.has(account.optionId) + ) || + accounts.viewerMcpAccounts?.some( + (account) => account.status === 'active' && readyMcpServers.has(account.mcpServerId) + )) const hrefs: Record = { 'connect-integration': viewer.isAdmin @@ -94,10 +142,12 @@ export function GetStarted() { 'connect-sim-search': routes.settingsSection('search-mcp'), } const completed: Record = { - 'connect-integration': overview?.hasSearchableDocuments === true, + 'connect-integration': Boolean(hasSearchConnection), 'connect-sim-search': hasSearchAuthorization, } - const steps = STEPS.filter((step) => step.id !== 'connect-sim-search' || viewer.canUseSearchMcp) + const steps = STEPS.filter((step) => + step.id === 'connect-sim-search' ? viewer.canUseSearchMcp : canConnectIntegrations + ) const [expanded, setExpanded] = useState(true) /** @@ -132,6 +182,8 @@ export function GetStarted() { setExpanded((prev) => !prev) } + if (steps.length === 0) return null + return ( ({ - live: false, plan: false, resourcePanel: vi.fn(), chat: vi.fn(), @@ -99,18 +95,15 @@ import { OrganizationHome } from '@/app/o/[organizationId]/home/organization-hom const mockSession = authClientMockFns.mockUseSession const mockContext = organizationProviderMockFns.mockUseOrganizationContext -const mockSources = kbConnectorsQueriesMockFns.mockUseSearchSourceOverview -const liveShape = () => - createMockDeploymentShape({ features: { liveEnterpriseSearch: mocks.live } }) -deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(liveShape) -deploymentShapeMockFns.mockGetDeploymentShape.mockImplementation(liveShape) +const deploymentShape = () => createMockDeploymentShape() +deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(deploymentShape) +deploymentShapeMockFns.mockGetDeploymentShape.mockImplementation(deploymentShape) let root: Root let container: HTMLDivElement beforeEach(() => { useMothershipDraftsStore.setState({ drafts: {} }) mocks.plan = false - mocks.live = false mocks.activeResource = null mockSession.mockReturnValue({ data: { user: { id: 'reader' } } }) useOrganizationChatModeStore.setState({ modes: {}, assistantSearchLevels: {} }) @@ -131,7 +124,6 @@ beforeEach(() => { canBuild: true, viewer: { isAdmin: false, canUseSearchMcp: true }, }) - mockSources.mockReturnValue({ data: { providers: [], hasSearchableDocuments: false } }) mocks.apiKeys.mockReturnValue({ data: { personalKeys: [] } }) mockAuthorizedApps([{ apps: [], nextCursor: null }]) mocks.chat.mockReturnValue({ diff --git a/apps/sim/app/o/[organizationId]/home/organization-home.tsx b/apps/sim/app/o/[organizationId]/home/organization-home.tsx index e6d7d080c28..f6b150b2049 100644 --- a/apps/sim/app/o/[organizationId]/home/organization-home.tsx +++ b/apps/sim/app/o/[organizationId]/home/organization-home.tsx @@ -9,7 +9,6 @@ import { requestJson } from '@/lib/api/client/request' import type { WorkspaceSearchFilters } from '@/lib/api/contracts/knowledge' import { getWorkspaceHostContextContract } from '@/lib/api/contracts/workspaces' import { useSession } from '@/lib/auth/auth-client' -import { getDeploymentShape } from '@/lib/core/config/deployment-shape' import { MothershipHandoffStorage } from '@/lib/core/utils/browser-storage' import { getMothershipAttachmentPreviewUrl, @@ -139,7 +138,6 @@ function OrganizationHomeContent({ const hasChat = Boolean(chatId || chat.messages.length) const canSelectMode = !hasChat && mothershipAvailable && canBuild && (searchAccess.memberScoped || planEnabled) - const liveSearch = getDeploymentShape().features.liveEnterpriseSearch === true const assistantSearchLevel = 'fast' const panel = useChatResourcePanel(chat, controller, userId) const addResource = panel.addResourceFromUser @@ -257,7 +255,7 @@ function OrganizationHomeContent({ ...(handoff.assistantSearch ? { assistantSearch: handoff.assistantSearch } : {}), }) } - }, [chatId, organization.id, requestMode, sendMessage, assistantSearchLevel, liveSearch]) + }, [chatId, organization.id, requestMode, sendMessage, assistantSearchLevel]) const send = ( message: string, @@ -390,11 +388,7 @@ function OrganizationHomeContent({ chatId={chat.resolvedChatId} composer={composer} onWorkspaceResourceSelect={requestMode !== 'assistant' ? selectResource : undefined} - initialScrollBlocked={ - (requestMode !== 'assistant' || liveSearch) && - chat.resources.length > 0 && - panel.isResourceCollapsed - } + initialScrollBlocked={chat.resources.length > 0 && panel.isResourceCollapsed} /> ) : ( - canConnect: boolean -} - -/** A member authorizes GitHub once, independently of the repositories added by admins. */ -export function GitHubMemberIntegration({ - organizationId, - inventory, - canConnect, -}: GitHubMemberIntegrationProps) { - const connect = useConnectOrganizationAccount() - const reconnect = useReconnectPersonalOrganizationAccount() - const accounts = - inventory.data?.viewerAccounts?.filter( - (account) => account.providerId === 'github-repositories' - ) ?? [] - const account = accounts.find((entry) => entry.status === 'needs_reauth') ?? accounts[0] - const option = - inventory.data?.credentialGroup?.status === 'active' - ? inventory.data.credentialGroup.options.find( - (entry) => entry.provider === 'github-repositories' && entry.status === 'active' - ) - : undefined - const loading = inventory.isPending && !inventory.data - const failed = inventory.isError - const meta = CONNECTOR_META_REGISTRY.github - const onError = (error: Error) => toast.error(error.message) - const description = account - ? `${accounts.map((entry) => entry.displayName).join(', ')} · ${account.status === 'needs_reauth' ? 'Reconnect required' : 'Connected'}` - : loading - ? 'Loading connection' - : failed - ? 'Could not load connection' - : option - ? 'Connect once to search the repositories your admin adds' - : 'An admin needs to reconnect GitHub' - - return ( - : undefined} - title='GitHub' - description={description} - trailing={ -
- - {failed ? ( - void inventory.refetch()}> - {inventory.isFetching ? 'Retrying' : 'Retry'} - - ) : account?.status === 'needs_reauth' && option?.id === account.optionId ? ( - reconnect.mutate(account.credentialId, { onError })} - > - Reconnect - - ) : !account && !loading && canConnect && option ? ( - connect.mutate({ organizationId, optionId: option.id }, { onError })} - > - Connect - - ) : null} -
- } - /> - ) -} diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/index.ts b/apps/sim/app/o/[organizationId]/integrations/indexed/index.ts deleted file mode 100644 index 9b1281fd906..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/indexed/index.ts +++ /dev/null @@ -1 +0,0 @@ -export { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/indexed/member-integrations-list' diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx deleted file mode 100644 index 6619a285d71..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/indexed/member-integration-row.tsx +++ /dev/null @@ -1,223 +0,0 @@ -'use client' - -import { Chip, ChipLink } from '@sim/emcn' -import { organizationRoutes } from '@/lib/navigation/paths' -import { connectorDisplayName } from '@/lib/sim-search/connectors' -import { DisconnectAccountMenu } from '@/app/o/[organizationId]/integrations/disconnect-account-menu' -import { getSearchSourceStatus } from '@/app/o/[organizationId]/integrations/indexed/source-status' -import { - CONNECTABLE_MEMBERSHIPS, - enrollmentActionLabel, - type useMemberEnrollment, -} from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' -import { IntegrationTile } from '@/app/workspace/[workspaceId]/integrations/components/integrations-showcase' -import type { RowAction } from '@/app/workspace/[workspaceId]/settings/components/row-actions-menu' -import { SettingsResourceRow } from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' -import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' -import type { useSearchSources } from '@/hooks/queries/kb/connectors' - -interface MemberIntegrationRowProps { - organizationId: string - connectorType: string - configured: boolean - sources: ReturnType - enrollment: ReturnType - memberAccessAvailable: boolean - mirroredAccessAvailable: boolean - onCreate?: () => void - addLabel?: string -} - -/** One flat account row per integration, independent of how many content scopes it indexes. */ -export function MemberIntegrationRow({ - organizationId, - connectorType, - configured, - sources, - enrollment, - memberAccessAvailable, - mirroredAccessAvailable, - onCreate, - addLabel, -}: MemberIntegrationRowProps) { - const name = connectorDisplayName(connectorType) - const meta = CONNECTOR_META_REGISTRY[connectorType] - const rows = sources.data ?? [] - const accounts = [ - ...new Map( - rows - .flatMap((source) => source.viewerAccounts) - .map((account) => [account.credentialId, account]) - ).values(), - ] - const accountLabels = new Map() - for (const account of accounts) { - if (accounts.filter((other) => other.displayName === account.displayName).length < 2) continue - const descriptions = [ - ...new Set( - rows - .filter((source) => - source.viewerAccounts.some((other) => other.credentialId === account.credentialId) - ) - .map((source) => source.sourceDescription) - .filter(Boolean) - ), - ] - const context = [ - account.status === 'needs_reauth' ? 'Reconnect required' : undefined, - descriptions.length > 1 ? `${descriptions[0]} +${descriptions.length - 1}` : descriptions[0], - ].filter(Boolean) - accountLabels.set(account.credentialId, [...context, account.displayName].join(' · ')) - } - for (const label of new Set(accountLabels.values())) { - const matching = accounts.filter((account) => accountLabels.get(account.credentialId) === label) - if (matching.length < 2) continue - matching.forEach((account, index) => - accountLabels.set(account.credentialId, `Connection ${index + 1} · ${label}`) - ) - } - const isUsable = (source: (typeof rows)[number]) => - source.availability === 'available' && - (source.accessMode === 'members' - ? memberAccessAvailable - : mirroredAccessAvailable && (!source.connectionRequired || memberAccessAvailable)) - const eligible = rows.filter( - (source) => - source.connectionRequired && - source.viewerEmailVerified && - source.enabled && - source.approved !== false && - isUsable(source) && - source.viewerMembership !== null && - CONNECTABLE_MEMBERSHIPS.has(source.viewerMembership) - ) - const target = - eligible.find((source) => source.viewerMembership === 'needs_reauth') ?? eligible[0] - const hasLoadError = configured && sources.isError && !sources.isFetchNextPageError - const ready = !configured || (!sources.isPending && !hasLoadError) - const allCentral = - rows.length > 0 && !sources.hasNextPage && rows.every((source) => !source.connectionRequired) - const needsEmailVerification = rows.some( - (source) => - source.enabled && source.approved !== false && isUsable(source) && !source.viewerEmailVerified - ) - const waiting = target - ? enrollment.isAwaiting(target.connectorId) - : enrollment.isAwaitingSource(connectorType) - const status = (source: (typeof rows)[number]) => - getSearchSourceStatus({ - source, - scopeKind: 'organization', - supported: meta?.search === true, - usable: isUsable(source), - connectable: eligible.includes(source), - waiting: enrollment.isAwaiting(source.connectorId), - }) - function description() { - if (!configured) return waiting ? 'Finish connecting in the other tab' : 'Not connected' - if (hasLoadError) return 'Could not load connection' - if (sources.isPending) return 'Loading connection' - if (target) { - if (waiting) return 'Finish connecting in the other tab' - if (target.viewerMembership === 'needs_reauth') return 'Reconnect your account' - const hasConnectedContent = rows.some( - (source) => - source.enabled && - source.approved !== false && - isUsable(source) && - (!source.connectionRequired || source.viewerMembership === 'connected') - ) - return hasConnectedContent ? 'Additional connection required' : 'Not connected' - } - if (rows.length === 1 && !sources.hasNextPage) return status(rows[0]) - if (needsEmailVerification) return 'Verify your email' - if ( - rows.some( - (source) => - !source.enabled || - source.approved === false || - !isUsable(source) || - source.viewerMembership === 'revoked' || - (source.connectionRequired && source.viewerMembership === null) - ) - ) - return 'Some connections need attention' - if (rows.some((source) => source.hasSyncError || source.viewerFailedDocumentCount > 0)) - return 'Sync needs attention' - if (rows.some((source) => source.isSyncing)) return 'Indexing' - if (sources.hasNextPage) - return sources.isFetchNextPageError - ? 'Could not check remaining connections' - : 'More connections to check' - if (allCentral) return 'Connected by your organization' - return accounts.length ? 'Connected' : 'No connected content' - } - const actions: RowAction[] = [] - if (configured && ready && !sources.hasNextPage && !allCentral && addLabel && onCreate) - actions.push({ label: addLabel, onSelect: onCreate, disabled: enrollment.isPending }) - const canCheckConnections = - configured && ready && sources.hasNextPage && !target && !needsEmailVerification - const canConnect = ready && (target || (!configured && onCreate)) - - return ( - : undefined} - title={name} - description={description()} - trailing={ -
- - {ready && needsEmailVerification && ( - - Verify email - - )} - {hasLoadError && ( - void sources.refetch()}> - {sources.isFetching ? 'Retrying' : 'Retry'} - - )} - {canCheckConnections && ( - void sources.fetchNextPage({ cancelRefetch: false })} - > - {sources.isFetchingNextPage - ? 'Checking' - : sources.isFetchNextPageError - ? 'Retry' - : 'Check connections'} - - )} - {canConnect && ( - - target - ? enrollment.connect(target.knowledgeBaseId, target.connectorId) - : onCreate?.() - } - > - {target?.viewerMembership - ? enrollmentActionLabel(target.viewerMembership, waiting) - : waiting - ? 'Open again' - : 'Connect'} - - )} -
- } - /> - ) -} diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx deleted file mode 100644 index bcb1692d8eb..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/indexed/member-integrations-list.tsx +++ /dev/null @@ -1,268 +0,0 @@ -'use client' - -import { useMemo } from 'react' -import { toast } from '@sim/emcn' -import type { ResourceScope } from '@/lib/core/resource-scope' -import { getSearchConnectionLabels } from '@/lib/sim-search/connection-labels' -import { - getConnectorAccessAvailability, - SEARCH_CONNECTORS, - SEARCH_SOURCE_TYPES, - type SearchConnector, -} from '@/lib/sim-search/connectors' -import { GitHubMemberIntegration } from '@/app/o/[organizationId]/integrations/indexed/github-member-integration' -import { MemberIntegrationRow } from '@/app/o/[organizationId]/integrations/indexed/member-integration-row' -import { useMemberEnrollment } from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' -import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' -import { SourceSetupModal } from '@/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal' -import { - SettingsEmptyState, - SettingsQueryErrorState, -} from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' -import { RESOURCE_LIST_STACK } from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' -import { useSearchSourceOverview, useSearchSources } from '@/hooks/queries/kb/connectors' -import { - organizationAccountsKeys, - useOrganizationAccounts, -} from '@/hooks/queries/organization-accounts' -import { usePersonalSearchIntegrations } from '@/hooks/queries/personal-search-integrations' -import { useSearchIntegrations } from '@/hooks/queries/search-integrations' -import { searchSourceKeys } from '@/hooks/queries/utils/search-source-keys' -import { usePermissionConfig } from '@/hooks/use-permission-config' - -interface MemberIntegrationsListProps { - search?: string - showEmpty?: boolean -} - -/** Provider existence comes from the complete overview; account status uses bounded source pages. */ -export function MemberIntegrationsList({ - search = '', - showEmpty = true, -}: MemberIntegrationsListProps = {}) { - const { organization, searchAccess } = useOrganizationContext() - const scope: ResourceScope = { kind: 'organization', organizationId: organization.id } - const overview = useSearchSourceOverview(scope) - const integrations = useSearchIntegrations(organization.id) - const organizationAccounts = useOrganizationAccounts(organization.id) - const githubAccounts = - organizationAccounts.data?.viewerAccounts?.filter( - (account) => account.providerId === 'github-repositories' - ) ?? [] - const usesGitHubInventory = - organizationAccounts.isPending || - organizationAccounts.isError || - organizationAccounts.data?.viewerAccounts !== undefined - const slackInventory = usePersonalSearchIntegrations({ - organizationId: organization.id, - connectorType: 'slack', - }) - const canConnectSharedSlack = - !slackInventory.isError && - slackInventory.data?.available.some( - (entry) => entry.target.connectorType === 'slack' && !entry.target.connectorId - ) === true - const availability = usePermissionConfig() - const configured = new Map( - overview.data?.providers.map((provider) => [provider.connectorType, provider]) - ) - const approved = new Set( - integrations.data - ?.filter((integration) => integration.approved) - .map((integration) => integration.connectorType) - ) - const providers = SEARCH_SOURCE_TYPES.flatMap(([type, meta]) => { - const connector = SEARCH_CONNECTORS.find((entry) => entry.type === type) - const canCreate = Boolean( - connector && - (type !== 'slack' || canConnectSharedSlack) && - approved.has(type) && - getConnectorAccessAvailability(meta, availability.integrationAvailability, { - memberAccessAvailable: searchAccess.memberScoped, - mirroredAccessAvailable: searchAccess.sourceMirrored, - oauthServiceAvailability: availability.oauthServiceAvailability, - isIntegrationAvailabilityReady: availability.isIntegrationAvailabilityReady, - }).members - ) - const hasGitHubAccount = - type === 'github' && - (githubAccounts.length > 0 || organizationAccounts.isPending || organizationAccounts.isError) - return configured.has(type) || canCreate || hasGitHubAccount - ? [{ type, meta, connector, canCreate, configured: configured.has(type) }] - : [] - }) - const failedQuery = overview.isError ? overview : integrations.isError ? integrations : null - const query = search.trim().toLowerCase() - const showSlackSetupError = - slackInventory.isError && approved.has('slack') && 'slack'.includes(query) - const visible = providers.filter( - (provider) => - provider.meta.name.toLowerCase().includes(query) || - (provider.type === 'github' && - githubAccounts.some((account) => account.displayName.toLowerCase().includes(query))) - ) - const githubProvider = visible.find((provider) => provider.type === 'github') - const githubRow = - githubProvider && usesGitHubInventory ? ( - - ) : null - - return ( - <> -
- {failedQuery ? ( - <> - void failedQuery.refetch()} - variant='inline' - /> - {githubRow} - - ) : overview.isPending || integrations.isPending ? ( - <> - Loading integrations - {githubRow} - - ) : ( - <> - {showSlackSetupError && ( - void slackInventory.refetch()} - variant='inline' - /> - )} - {availability.integrationAvailabilityError && ( - void availability.refetchIntegrationAvailability()} - variant='inline' - /> - )} - {providers.map((provider) => ( - - ))} - {showEmpty && - visible.length === 0 && - !availability.integrationAvailabilityError && - !showSlackSetupError && ( - - {!availability.isIntegrationAvailabilityReady || - (approved.has('slack') && 'slack'.includes(query) && slackInventory.isPending) - ? 'Loading integrations' - : search - ? 'No matching integrations.' - : 'No integrations are available to connect.'} - - )} - - )} -
- - ) -} - -interface MemberIntegrationProps { - scope: ResourceScope & { kind: 'organization' } - connectorType: string - connector?: SearchConnector - configured: boolean - canCreate: boolean - memberAccessAvailable: boolean - mirroredAccessAvailable: boolean -} - -/** Each provider loads one bounded page; additional content is loaded explicitly. */ -function MemberIntegration({ - scope, - connectorType, - connector, - configured, - canCreate, - memberAccessAvailable, - mirroredAccessAvailable, -}: MemberIntegrationProps) { - const sources = useSearchSources(scope, { connectorType, enabled: configured }) - const membershipQueryKeys = useMemo( - () => [searchSourceKeys.list(scope), organizationAccountsKeys.detail(scope.organizationId)], - [scope.organizationId] - ) - const connectedConnectorIds = useMemo( - () => - new Set( - sources.data - ?.filter((source) => source.viewerMembership === 'connected') - .map((source) => source.connectorId) - ), - [sources.data] - ) - const enrollment = useMemberEnrollment({ - membershipQueryKeys, - connectedConnectorIds, - directOAuth: true, - onConnectionError: toast.error, - }) - return ( - <> - enrollment.connectSearchSource(scope, connector) - : undefined - } - addLabel={ - connector?.setupFields.length - ? getSearchConnectionLabels(connectorType, 'members').add - : undefined - } - /> - {enrollment.setupConnector && ( - - enrollment.connectSource(scope, enrollment.setupConnector!.type, config) - } - /> - )} - - ) -} diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/source-status.ts b/apps/sim/app/o/[organizationId]/integrations/indexed/source-status.ts deleted file mode 100644 index 77769aea4a6..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/indexed/source-status.ts +++ /dev/null @@ -1,44 +0,0 @@ -import type { SearchSourceSummary } from '@/lib/api/contracts/knowledge/connectors' -import type { ResourceScope } from '@/lib/core/resource-scope' - -interface SearchSourceStatusInput { - source: SearchSourceSummary - scopeKind: ResourceScope['kind'] - supported: boolean - usable: boolean - connectable: boolean - waiting: boolean -} - -/** Source and integration rows share one member-facing status priority. */ -export function getSearchSourceStatus({ - source, - scopeKind, - supported, - usable, - connectable, - waiting, -}: SearchSourceStatusInput): string { - const membership = source.viewerMembership - let status: string - if (!supported) status = 'Available in its knowledge base' - else if (source.approved === false) status = 'Deactivated by an organization admin' - else if (!usable) status = `Not available in this ${scopeKind}` - else if (!source.enabled) status = 'Syncing is paused' - else if (!source.viewerEmailVerified || membership === 'unverified_email') - status = 'Verify your email to search this source' - else if (membership === 'revoked') status = 'Your access was removed by an admin' - else if (source.connectionRequired && membership === null) status = 'Needs admin attention' - else if (connectable) - status = waiting - ? 'Finish connecting in the other tab' - : membership === 'needs_reauth' - ? 'Your account needs to be reconnected' - : 'Connect your account to search this source' - else if (source.hasSyncError || source.viewerFailedDocumentCount > 0) - status = 'Sync needs attention' - else if (source.isSyncing) status = 'Indexing' - else if (source.hasViewerDocuments) status = 'Ready to search' - else status = source.lastSyncAt ? 'No searchable documents yet' : 'Waiting for the first sync' - return status -} diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx b/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx deleted file mode 100644 index 7bba56306ae..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.test.tsx +++ /dev/null @@ -1,358 +0,0 @@ -/** - * @vitest-environment jsdom - */ -import { act } from 'react' -import { - kbConnectorsQueriesMock, - kbConnectorsQueriesMockFns, -} from '@sim/testing/mocks/kb-connectors-queries.mock' -import { reactQueryMock } from '@sim/testing/mocks/react-query.mock' -import { createRoot, type Root } from 'react-dom/client' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const mocks = vi.hoisted(() => ({ - enrollmentMutate: vi.fn(), - sourceConnectionMutate: vi.fn(), - connectionError: vi.fn(), - channels: [] as Array<{ - name: string - onmessage: ((event: MessageEvent) => void) | null - close: ReturnType - }>, -})) - -vi.mock('@tanstack/react-query', () => reactQueryMock) -vi.mock('@/hooks/queries/kb/connectors', () => kbConnectorsQueriesMock) - -import { useMemberEnrollment } from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' - -type Enrollment = ReturnType - -let latest: Enrollment | null = null -let root: Root | null = null -let container: HTMLDivElement | null = null -let enrollmentTab: { location: { href: string }; closed: boolean; close: () => void } - -function Harness({ - connected, - directOAuth, - onConnectionError, -}: { - connected: ReadonlySet - directOAuth?: boolean - onConnectionError?: (message: string) => void -}) { - latest = useMemberEnrollment({ - membershipQueryKeys: [['test-memberships']], - connectedConnectorIds: connected, - directOAuth, - onConnectionError, - }) - return null -} - -function mount( - connected: ReadonlySet = new Set(), - directOAuth = false, - onConnectionError?: (message: string) => void -) { - ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true - container = document.createElement('div') - document.body.appendChild(container) - root = createRoot(container) - act(() => - root?.render( - - ) - ) -} - -function enrollment(): Enrollment { - if (!latest) throw new Error('Hook did not render') - return latest -} - -beforeEach(() => { - kbConnectorsQueriesMockFns.mockUseStartConnectorMemberEnrollment.mockReturnValue({ - mutate: mocks.enrollmentMutate, - submittedAt: 0, - isPending: false, - error: null, - }) - kbConnectorsQueriesMockFns.mockUseConnectSimSearchConnector.mockReturnValue({ - mutate: mocks.sourceConnectionMutate, - submittedAt: 0, - isPending: false, - error: null, - }) - vi.useFakeTimers() - mocks.channels.length = 0 - vi.stubGlobal( - 'BroadcastChannel', - class { - onmessage: ((event: MessageEvent) => void) | null = null - close = vi.fn() - constructor(public name: string) { - mocks.channels.push(this) - } - } - ) - enrollmentTab = { - location: { href: '' }, - closed: false, - close: vi.fn(), - } - vi.spyOn(window, 'open').mockReturnValue(enrollmentTab as unknown as Window) -}) - -afterEach(() => { - if (root) act(() => root?.unmount()) - container?.remove() - root = null - container = null - latest = null - vi.useRealTimers() -}) - -describe('useMemberEnrollment', () => { - it('reports an OAuth failure once per attempt and allows the same error on a later retry', () => { - mount(new Set(), true, mocks.connectionError) - for (let index = 0; index < 2; index += 1) { - act(() => enrollment().connect('kb-1', 'connector-1')) - act(() => - mocks.enrollmentMutate.mock.calls[index][1].onSuccess({ - url: 'https://provider.test/authorize', - }) - ) - act(() => - mocks.channels[index].onmessage?.( - new MessageEvent('message', { data: 'permissions_required' }) - ) - ) - act(() => - mocks.channels[index].onmessage?.( - new MessageEvent('message', { data: 'permissions_required' }) - ) - ) - expect(mocks.connectionError).toHaveBeenCalledTimes(index + 1) - expect(enrollment().isAwaiting('connector-1')).toBe(false) - } - expect(mocks.connectionError).toHaveBeenLastCalledWith( - 'All requested permissions are required to connect this account.' - ) - act(() => vi.advanceTimersByTime(10 * 60_000)) - expect(mocks.connectionError).toHaveBeenCalledTimes(2) - }) - - it.each(['existing', 'new'] as const)( - 'does not expire a superseded %s source authorization after its retry connects', - (source) => { - mount(new Set(), true, mocks.connectionError) - const mutation = source === 'existing' ? mocks.enrollmentMutate : mocks.sourceConnectionMutate - for (let index = 0; index < 2; index += 1) { - act(() => { - if (source === 'existing') enrollment().connect('kb-1', 'connector-1') - else enrollment().connectSource('workspace-1', 'jira') - }) - act(() => - mutation.mock.calls[index][1].onSuccess({ - url: `https://provider.test/attempt-${index}`, - connectorId: 'connector-1', - }) - ) - } - act(() => mocks.channels[1].onmessage?.(new MessageEvent('message', { data: 'connected' }))) - act(() => vi.advanceTimersByTime(10 * 60_000)) - act(() => - mocks.channels[0].onmessage?.(new MessageEvent('message', { data: 'permissions_required' })) - ) - expect(mocks.connectionError).not.toHaveBeenCalled() - expect(enrollment().error).toBeNull() - expect(enrollment().isAwaiting('connector-1')).toBe(false) - expect(mocks.channels[0].close).toHaveBeenCalledOnce() - expect(mocks.channels[1].close).toHaveBeenCalledOnce() - } - ) - - it.each([ - ['existing', 'permissions_required'], - ['existing', 'denied'], - ['existing', 'expired'], - ['new', 'permissions_required'], - ['new', 'denied'], - ['new', 'expired'], - ] as const)( - 'ignores the previous %s source’s %s while its retry request is pending', - (source, failure) => { - mount(new Set(), true, mocks.connectionError) - const mutation = source === 'existing' ? mocks.enrollmentMutate : mocks.sourceConnectionMutate - const connect = () => { - if (source === 'existing') enrollment().connect('kb-1', 'connector-1') - else enrollment().connectSource('workspace-1', 'jira', { projectKey: 'ENG' }) - } - act(connect) - act(() => - mutation.mock.calls[0][1].onSuccess({ - url: 'https://provider.test/previous', - connectorId: 'connector-1', - }) - ) - act(() => vi.advanceTimersByTime(9 * 60_000)) - act(connect) - if (failure !== 'expired') { - act(() => mocks.channels[0].onmessage?.(new MessageEvent('message', { data: failure }))) - } - act(() => vi.advanceTimersByTime(60_000)) - expect(mocks.connectionError).not.toHaveBeenCalled() - expect(enrollment().error).toBeNull() - act(() => - mutation.mock.calls[1][1].onSuccess({ - url: 'https://provider.test/retry', - connectorId: 'connector-1', - }) - ) - expect(enrollment().isAwaiting('connector-1')).toBe(true) - expect(mocks.channels[1].close).not.toHaveBeenCalled() - } - ) - - it.each([ - ['existing', 'success'], - ['existing', 'failure'], - ['new', 'success'], - ['new', 'failure'], - ] as const)('ignores a superseded %s source request’s late %s', (source, outcome) => { - mount(new Set(), true, mocks.connectionError) - const mutation = source === 'existing' ? mocks.enrollmentMutate : mocks.sourceConnectionMutate - const retryTab = { location: { href: '' }, closed: false, close: vi.fn() } - vi.mocked(window.open) - .mockReturnValueOnce(enrollmentTab as unknown as Window) - .mockReturnValueOnce(retryTab as unknown as Window) - for (let index = 0; index < 2; index += 1) { - act(() => { - if (source === 'existing') enrollment().connect('kb-1', 'connector-1') - else enrollment().connectSource('workspace-1', 'jira', { projectKey: 'ENG' }) - }) - } - act(() => - mutation.mock.calls[1][1].onSuccess({ - url: 'https://provider.test/retry', - connectorId: 'connector-1', - }) - ) - act(() => { - if (outcome === 'failure') { - mutation.mock.calls[0][1].onError(new Error('Previous request failed')) - } else { - mutation.mock.calls[0][1].onSuccess({ - url: 'https://provider.test/previous', - connectorId: 'connector-1', - }) - } - }) - expect(enrollmentTab.location.href).toBe('') - expect(retryTab.location.href).toBe('https://provider.test/retry') - expect(retryTab.close).not.toHaveBeenCalled() - expect(enrollment().isAwaiting('connector-1')).toBe(true) - expect(mocks.channels[1].close).not.toHaveBeenCalled() - expect(mocks.connectionError).not.toHaveBeenCalled() - expect(enrollment().error).toBeNull() - }) - - it('does not let a delayed first-source response replace its newer connector authorization', () => { - mount(new Set(), true, mocks.connectionError) - act(() => enrollment().connectSource('workspace-1', 'jira', { projectKey: 'ENG' })) - act(() => enrollment().connect('kb-1', 'connector-1')) - act(() => - mocks.enrollmentMutate.mock.calls[0][1].onSuccess({ url: 'https://provider.test/retry' }) - ) - act(() => - mocks.sourceConnectionMutate.mock.calls[0][1].onSuccess({ - url: 'https://provider.test/previous', - connectorId: 'connector-1', - }) - ) - expect(enrollmentTab.location.href).toBe('https://provider.test/retry') - expect(mocks.channels[1].close).not.toHaveBeenCalled() - expect(enrollment().isAwaiting('connector-1')).toBe(true) - }) - - it('ignores first-source success while a newer request for its connector is still pending', () => { - mount(new Set(), true, mocks.connectionError) - const retryTab = { location: { href: '' }, closed: false, close: vi.fn() } - vi.mocked(window.open) - .mockReturnValueOnce(enrollmentTab as unknown as Window) - .mockReturnValueOnce(retryTab as unknown as Window) - act(() => enrollment().connectSource('workspace-1', 'jira', { projectKey: 'ENG' })) - act(() => enrollment().connect('kb-1', 'connector-1')) - act(() => - mocks.sourceConnectionMutate.mock.calls[0][1].onSuccess({ - url: 'https://provider.test/previous', - connectorId: 'connector-1', - }) - ) - expect(enrollmentTab.location.href).toBe('') - expect(enrollment().isAwaiting('connector-1')).toBe(false) - act(() => mocks.channels[0].onmessage?.(new MessageEvent('message', { data: 'denied' }))) - expect(mocks.connectionError).not.toHaveBeenCalled() - act(() => - mocks.enrollmentMutate.mock.calls[0][1].onSuccess({ url: 'https://provider.test/retry' }) - ) - expect(retryTab.location.href).toBe('https://provider.test/retry') - expect(enrollment().isAwaiting('connector-1')).toBe(true) - expect(mocks.channels[1].close).not.toHaveBeenCalled() - }) - - it('keeps overlapping provider authorizations separate and reports a rejected one on the original page', () => { - mount(new Set(), true) - act(() => enrollment().connect('kb-1', 'connector-1')) - act(() => - mocks.enrollmentMutate.mock.calls[0][1].onSuccess({ url: 'https://provider.test/one' }) - ) - act(() => enrollment().connect('kb-1', 'connector-2')) - act(() => - mocks.enrollmentMutate.mock.calls[1][1].onSuccess({ url: 'https://provider.test/two' }) - ) - expect(mocks.channels[0].name).not.toBe(mocks.channels[1].name) - act(() => mocks.channels[0].onmessage?.(new MessageEvent('message', { data: 'denied' }))) - expect(enrollment().isAwaiting('connector-1')).toBe(false) - expect(enrollment().isAwaiting('connector-2')).toBe(true) - expect(enrollment().error).toContain('Authorization was canceled') - act(() => mocks.channels[1].onmessage?.(new MessageEvent('message', { data: 'unrecognized' }))) - expect(enrollment().isAwaiting('connector-2')).toBe(true) - }) - - it('passes direct authorization correlation through first-source setup and cleans it up on failure', () => { - mount(new Set(), true) - act(() => - enrollment().connectSource({ kind: 'organization', organizationId: 'org-1' }, 'gmail') - ) - const [input, handlers] = mocks.sourceConnectionMutate.mock.calls[0] - expect(input).toMatchObject({ - organizationId: 'org-1', - connectorType: 'gmail', - oauthCompletionId: expect.any(String), - }) - act(() => handlers.onError(new Error('Unavailable'))) - expect(mocks.channels[0].close).toHaveBeenCalledOnce() - expect(enrollmentTab.close).toHaveBeenCalledOnce() - }) - - it('does not navigate or await a tab closed before the enrollment request completes', () => { - mount() - act(() => enrollment().connectSource('workspace-1', 'google_drive')) - enrollmentTab.closed = true - const [, handlers] = mocks.sourceConnectionMutate.mock.calls[0] - act(() => - handlers.onSuccess({ url: 'https://example.test/enroll', connectorId: 'connector-1' }) - ) - - expect(enrollmentTab.location.href).toBe('') - expect(enrollment().isAwaiting('connector-1')).toBe(false) - expect(enrollment().isAwaitingSource('google_drive')).toBe(false) - }) -}) diff --git a/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts b/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts deleted file mode 100644 index 76faa926cc8..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/indexed/use-member-enrollment.ts +++ /dev/null @@ -1,416 +0,0 @@ -'use client' - -import { useCallback, useEffect, useRef, useState } from 'react' -import { createLogger } from '@sim/logger' -import { generateId } from '@sim/utils/id' -import { type QueryKey, useMutation, useQueryClient } from '@tanstack/react-query' -import type { DesktopSourceRequest } from '@/lib/api/contracts/desktop-source-connect' -import { - type ResourceScope, - resourceScopeFields, - resourceScopeKey, -} from '@/lib/core/resource-scope' -import { - CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES, - credentialGroupOAuthCompletionChannel, - isCredentialGroupOAuthFailure, -} from '@/lib/credential-groups/oauth-completion' -import { isDesktopApp } from '@/lib/desktop' -import { connectDesktopSource } from '@/lib/desktop/source-connect' -import type { SearchConnector } from '@/lib/sim-search/connectors' -import { - useConnectSimSearchConnector, - useStartConnectorMemberEnrollment, - type ViewerConnectorMembership, -} from '@/hooks/queries/kb/connectors' - -const logger = createLogger('MemberEnrollment') - -/** How often the membership queries are refreshed while a member connects in another tab. */ -const AWAITING_CONNECTION_POLL_MS = 4_000 -/** How long a connection is awaited before the surface stops refreshing on its own. */ -const AWAITING_CONNECTION_TIMEOUT_MS = 10 * 60_000 -const POPUP_BLOCKED_MESSAGE = 'Allow pop-ups for this site to connect your account.' - -/** Memberships the viewer can act on themselves. */ -export const CONNECTABLE_MEMBERSHIPS: ReadonlySet = new Set([ - 'needs_reauth', - 'invited', - 'not_enrolled', -]) - -/** The label of the one action a connectable membership offers. */ -export function enrollmentActionLabel( - membership: ViewerConnectorMembership, - waiting: boolean -): string { - if (waiting) return 'Open again' - return membership === 'needs_reauth' ? 'Reconnect' : 'Connect' -} - -/** An enrollment tab this surface opened that has not connected yet. */ -interface AwaitingEnrollment { - since: number - tab: Window - /** - * The Sim Search source whose connect created the connector, so the source - * can be told it is awaited before its membership row exists to look it up by. - */ - connectorType: string | null - oauthCompletionId?: string -} - -interface UseMemberEnrollmentProps { - /** Queries this surface reads memberships from, refreshed while a connection is awaited. */ - membershipQueryKeys: readonly QueryKey[] - /** Connector ids the viewer is now connected to; awaiting stops for them. */ - connectedConnectorIds: ReadonlySet - /** Main Integrations skips the invitation page; invitation-based surfaces keep their flow. */ - directOAuth?: boolean - onConnectionError?: (message: string) => void -} - -/** - * Lets the viewer connect their own account to a per-member connector, by - * connector or by Sim Search source. Enrollment opens in a new tab, and the - * membership queries are polled meanwhile so the surface that started it - * updates on its own once the account is connected. - * - * The tab is opened in the click itself, before the enrollment link is - * minted, because a tab opened after a network round trip is outside the - * click's activation window and popup blockers swallow it. - */ -export function useMemberEnrollment({ - membershipQueryKeys, - connectedConnectorIds, - directOAuth = false, - onConnectionError, -}: UseMemberEnrollmentProps) { - const connectedRef = useRef(connectedConnectorIds) - const oauthPopups = useRef( - new Map< - string, - { - channel: BroadcastChannel - timer: ReturnType - attemptKey: string - connectorId?: string - } - >() - ) - const queryClient = useQueryClient() - const nativeAbort = useRef(null) - useEffect(() => () => nativeAbort.current?.abort(), []) - const nativeConnection = useMutation({ - mutationFn: async (request: DesktopSourceRequest) => { - nativeAbort.current?.abort() - const controller = new AbortController() - nativeAbort.current = controller - return connectDesktopSource(request, controller.signal) - }, - onSettled: () => - Promise.all( - membershipQueryKeys.map((queryKey) => queryClient.invalidateQueries({ queryKey })) - ), - onError: (error) => onConnectionError?.(error.message), - onSuccess: () => setSetupConnector(null), - }) - const enrollment = useStartConnectorMemberEnrollment() - const sourceConnection = useConnectSimSearchConnector() - const [awaitingSince, setAwaitingSince] = useState>( - () => new Map() - ) - const [popupBlocked, setPopupBlocked] = useState(false) - const [oauthError, setOAuthError] = useState(null) - - const refreshMemberships = useCallback(() => { - for (const queryKey of membershipQueryKeys) { - void queryClient.invalidateQueries({ queryKey }) - } - }, [membershipQueryKeys, queryClient]) - - const clearOAuth = (completionId: string) => { - const popup = oauthPopups.current.get(completionId) - if (!popup) return false - clearTimeout(popup.timer) - popup.channel.close() - oauthPopups.current.delete(completionId) - setAwaitingSince( - (current) => - new Map([...current].filter(([, entry]) => entry.oauthCompletionId !== completionId)) - ) - return true - } - - const finishOAuth = (completionId: string, error: string | null) => { - if (!clearOAuth(completionId)) return - setOAuthError(error) - if (error) onConnectionError?.(error) - refreshMemberships() - } - - useEffect(() => { - const popups = oauthPopups.current - return () => { - for (const popup of popups.values()) { - clearTimeout(popup.timer) - popup.channel.close() - } - popups.clear() - } - }, []) - - useEffect(() => { - connectedRef.current = connectedConnectorIds - }, [connectedConnectorIds]) - - /** - * Polls while any connection is awaited, and once more after the last one - * connects: that tick drops the connected ids, so a token that later needs - * reauthorization is not mistaken for a connection still being awaited. - * Direct OAuth waits for its completion message: provider window isolation - * can report a closed handle while authorization is still in progress. - */ - const awaiting = awaitingSince.size > 0 - useEffect(() => { - if (!awaiting) return - const timer = setInterval(() => { - const now = Date.now() - setAwaitingSince((current) => { - const next = new Map( - [...current].filter( - ([id, { since, tab, oauthCompletionId }]) => - (Boolean(oauthCompletionId) || !tab.closed) && - (Boolean(oauthCompletionId) || !connectedRef.current.has(id)) && - now - since < AWAITING_CONNECTION_TIMEOUT_MS - ) - ) - return next.size === current.size ? current : next - }) - refreshMemberships() - }, AWAITING_CONNECTION_POLL_MS) - return () => clearInterval(timer) - }, [awaiting, refreshMemberships]) - - /** Opens the tab inside the click, then sends it wherever `start` mints. */ - const openEnrollment = ( - attemptKey: string, - start: (handlers: { - oauthCompletionId?: string - onSuccess: (url: string, connectorId: string, connectorType?: string) => boolean - onError: () => boolean - }) => void - ) => { - const tab = window.open('about:blank', '_blank') - if (!tab) { - setPopupBlocked(true) - onConnectionError?.(POPUP_BLOCKED_MESSAGE) - return - } - tab.opener = null - setPopupBlocked(false) - setOAuthError(null) - const oauthCompletionId = directOAuth ? generateId() : undefined - if (oauthCompletionId) { - for (const [previousId, previous] of oauthPopups.current) { - if ( - previous.attemptKey === attemptKey || - (previous.connectorId && `connector:${previous.connectorId}` === attemptKey) - ) { - clearOAuth(previousId) - } - } - const channel = new BroadcastChannel(credentialGroupOAuthCompletionChannel(oauthCompletionId)) - channel.onmessage = ({ data }: MessageEvent) => { - if (data === 'connected') finishOAuth(oauthCompletionId, null) - else if (isCredentialGroupOAuthFailure(data)) - finishOAuth(oauthCompletionId, CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES[data]) - } - const timer = setTimeout(() => { - finishOAuth(oauthCompletionId, CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES.expired) - }, AWAITING_CONNECTION_TIMEOUT_MS) - oauthPopups.current.set(oauthCompletionId, { channel, timer, attemptKey }) - } - start({ - ...(oauthCompletionId ? { oauthCompletionId } : {}), - onSuccess: (url, connectorId, connectorType) => { - if (tab.closed || (oauthCompletionId && !oauthPopups.current.has(oauthCompletionId))) { - if (oauthCompletionId) - finishOAuth(oauthCompletionId, CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES.denied) - return false - } - if (oauthCompletionId) { - const popup = oauthPopups.current.get(oauthCompletionId)! - const connectorAttemptKey = `connector:${connectorId}` - const latestAttempt = [...oauthPopups.current] - .reverse() - .find( - ([id, entry]) => - id === oauthCompletionId || - entry.connectorId === connectorId || - entry.attemptKey === connectorAttemptKey - ) - if (latestAttempt?.[0] !== oauthCompletionId) { - clearOAuth(oauthCompletionId) - return false - } - for (const [previousId, previous] of oauthPopups.current) { - if ( - previousId === oauthCompletionId || - (previous.connectorId !== connectorId && previous.attemptKey !== connectorAttemptKey) - ) - continue - clearOAuth(previousId) - } - popup.connectorId = connectorId - } - tab.location.href = url - setAwaitingSince((current) => - new Map(current).set(connectorId, { - since: Date.now(), - tab, - connectorType: connectorType ?? null, - ...(oauthCompletionId ? { oauthCompletionId } : {}), - }) - ) - return true - }, - onError: () => { - const active = !oauthCompletionId || oauthPopups.current.has(oauthCompletionId) - if (oauthCompletionId) finishOAuth(oauthCompletionId, null) - tab.close() - return active - }, - }) - } - - const connect = (knowledgeBaseId: string, connectorId: string) => { - if (isDesktopApp()) { - nativeConnection.mutate({ - kind: 'member-enrollment', - params: { id: knowledgeBaseId, connectorId }, - ...(directOAuth ? { completionId: generateId() } : {}), - }) - return - } - openEnrollment(`connector:${connectorId}`, ({ onSuccess, onError, oauthCompletionId }) => { - enrollment.mutate( - { knowledgeBaseId, connectorId, ...(oauthCompletionId ? { oauthCompletionId } : {}) }, - { - onSuccess: ({ url }) => onSuccess(url, connectorId), - onError: (err) => { - if (!onError()) return - onConnectionError?.(err.message) - logger.error('Failed to start member enrollment', { error: err.message }) - }, - } - ) - }) - } - - /** - * Connects a Sim Search source: its per-member connector exists afterwards, - * and the viewer enrolls. The setup fields are read only when this connect - * creates the connector. - */ - const connectSource = ( - owner: string | ResourceScope, - connectorType: string, - sourceConfig?: Record - ) => { - const scope = - typeof owner === 'string' ? { kind: 'workspace' as const, workspaceId: owner } : owner - if (isDesktopApp()) { - nativeConnection.mutate({ - kind: 'search-source', - body: { ...resourceScopeFields(scope), connectorType, sourceConfig }, - ...(directOAuth ? { completionId: generateId() } : {}), - }) - return - } - const configKey = JSON.stringify( - Object.entries(sourceConfig ?? {}).sort(([left], [right]) => left.localeCompare(right)) - ) - openEnrollment( - `source:${resourceScopeKey(scope)}:${connectorType}:${configKey}`, - ({ onSuccess, onError, oauthCompletionId }) => { - sourceConnection.mutate( - { - ...resourceScopeFields(scope), - connectorType, - sourceConfig, - ...(oauthCompletionId ? { oauthCompletionId } : {}), - }, - { - onSuccess: ({ url, connectorId }) => { - if (onSuccess(url, connectorId, connectorType)) setSetupConnector(null) - }, - onError: (err) => { - if (!onError()) return - onConnectionError?.(err.message) - logger.error('Failed to connect a Sim Search source', { error: err.message }) - }, - } - ) - } - ) - } - - const [setupConnector, setSetupConnector] = useState(null) - - /** - * One click on a new Sim Search source: ask for its setup fields when it - * needs them, and otherwise create it and enroll in one step. - */ - const connectSearchSource = (owner: string | ResourceScope, connector: SearchConnector) => { - if (connector.setupFields.length > 0) { - setSetupConnector(connector) - return - } - connectSource(owner, connector.type) - } - - const isAwaiting = (connectorId: string) => - (nativeConnection.isPending && - nativeConnection.variables?.kind === 'member-enrollment' && - nativeConnection.variables.params.connectorId === connectorId) || - (awaitingSince.has(connectorId) && - (Boolean(awaitingSince.get(connectorId)?.oauthCompletionId) || - !connectedConnectorIds.has(connectorId))) - - /** - * Whether a Sim Search source is awaited by the connect that created its - * connector: the membership list has no row for it until it refetches, so - * the source cannot be looked up by connector id yet. - */ - const isAwaitingSource = (connectorType: string) => - (nativeConnection.isPending && - nativeConnection.variables?.kind === 'search-source' && - nativeConnection.variables.body.connectorType === connectorType) || - [...awaitingSince].some( - ([id, awaiting]) => - awaiting.connectorType === connectorType && - (Boolean(awaiting.oauthCompletionId) || !connectedConnectorIds.has(id)) - ) - - /** The surface reports the latest attempt, whichever path made it. */ - const latest = - enrollment.submittedAt >= sourceConnection.submittedAt ? enrollment : sourceConnection - return { - connect, - connectSource, - connectSearchSource, - setupConnector, - closeSetup: () => { - nativeAbort.current?.abort() - nativeAbort.current = null - setSetupConnector(null) - }, - isAwaiting, - isAwaitingSource, - isPending: nativeConnection.isPending || enrollment.isPending || sourceConnection.isPending, - error: popupBlocked - ? POPUP_BLOCKED_MESSAGE - : (nativeConnection.error?.message ?? oauthError ?? latest.error?.message ?? null), - } -} diff --git a/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx b/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx deleted file mode 100644 index 243d053ed17..00000000000 --- a/apps/sim/app/o/[organizationId]/integrations/integrations.test.tsx +++ /dev/null @@ -1,1041 +0,0 @@ -/** @vitest-environment jsdom */ -import { act, type ReactNode } from 'react' -import { toast } from '@sim/emcn' -import { - createMockDeploymentShape, - deploymentShapeMock, - deploymentShapeMockFns, -} from '@sim/testing/mocks/deployment-shape.mock' -import { - kbConnectorsQueriesMock, - kbConnectorsQueriesMockFns, -} from '@sim/testing/mocks/kb-connectors-queries.mock' -import { - organizationAccountsQueriesMock, - organizationAccountsQueriesMockFns, -} from '@sim/testing/mocks/organization-accounts-queries.mock' -import { - organizationProviderMock, - organizationProviderMockFns, -} from '@sim/testing/mocks/organization-provider.mock' -import { NuqsTestingAdapter } from 'nuqs/adapters/testing' -import { createRoot, type Root } from 'react-dom/client' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import type { SearchSourceSummary } from '@/lib/api/contracts/knowledge/connectors' -import type { SearchConnector } from '@/lib/sim-search/connectors' - -const mocks = vi.hoisted(() => ({ - live: false, - integrations: vi.fn(), - slackInventory: vi.fn(), - filters: vi.fn(), - connect: vi.fn(), - connectSearchSource: vi.fn(), - availability: vi.fn(), - refetch: vi.fn(), - nextPage: vi.fn(), - enrollment: vi.fn(), - accountMenu: vi.fn(), - request: vi.fn(), - updateUrl: vi.fn(), - setupConnector: null as SearchConnector | null, - connectOrganizationAccount: vi.fn(), - reconnectOrganizationAccount: vi.fn(), - refetchAccounts: vi.fn(), -})) -vi.mock('@/lib/core/config/deployment-shape', () => deploymentShapeMock) -vi.mock('@/hooks/queries/organization-secrets', () => ({ - useOrganizationSecretSource: () => ({ data: { source: null } }), -})) -vi.mock('@/hooks/queries/organization-accounts', () => organizationAccountsQueriesMock) -vi.mock('@/app/o/[organizationId]/integrations/slack-search-actions', () => ({ - SlackSearchActions: () => Return to Slack, -})) -vi.mock( - '@/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/search-integration-connection', - () => ({ - SearchIntegrationConnection: (props: unknown) => { - mocks.request(props) - return Requested connection - }, - }) -) -vi.mock('@/hooks/queries/personal-search-integrations', () => ({ - usePersonalSearchIntegrations: mocks.slackInventory, -})) -vi.mock('@/hooks/queries/search-integrations', () => ({ - useSearchIntegrations: mocks.integrations, -})) -vi.mock('@/hooks/use-permission-config', () => ({ usePermissionConfig: mocks.availability })) -vi.mock('@/app/o/[organizationId]/components/organization-page', () => ({ - OrganizationPage: ({ action, children }: { action?: ReactNode; children?: ReactNode }) => ( - <> - {action} - {children} - - ), -})) -vi.mock( - '@/app/o/[organizationId]/components/organization-page/use-organization-page-filters', - () => ({ useOrganizationPageFilters: mocks.filters }) -) -vi.mock('@/app/o/[organizationId]/providers/organization-provider', () => organizationProviderMock) -vi.mock('@/app/workspace/[workspaceId]/integrations/components/integrations-showcase', () => ({ - IntegrationTile: () => null, -})) -vi.mock('@/app/o/[organizationId]/integrations/disconnect-account-menu', () => ({ - DisconnectAccountMenu: (props: { - integrationName: string - accounts: { credentialId: string }[] - actions?: RowAction[] - }) => { - mocks.accountMenu(props) - const actions = [ - ...(props.actions ?? []), - ...props.accounts.map((account) => ({ - label: `Disconnect ${account.credentialId}`, - onSelect: vi.fn(), - })), - ] - return actions.length ? ( - - ) : null - }, -})) -vi.mock('@/hooks/queries/kb/connectors', () => kbConnectorsQueriesMock) -vi.mock('@/app/o/[organizationId]/integrations/indexed/use-member-enrollment', () => ({ - enrollmentActionLabel: (membership: string, waiting: boolean) => - waiting ? 'Open again' : membership === 'needs_reauth' ? 'Reconnect' : 'Connect', - CONNECTABLE_MEMBERSHIPS: new Set(['invited', 'not_enrolled', 'needs_reauth']), - useMemberEnrollment: (options: unknown) => { - mocks.enrollment(options) - return { - connect: mocks.connect, - connectSearchSource: mocks.connectSearchSource, - isAwaiting: () => false, - isAwaitingSource: () => false, - isPending: false, - setupConnector: mocks.setupConnector, - closeSetup: vi.fn(), - } - }, -})) -vi.mock('@/hooks/use-oauth-return', () => ({ - useDesktopOAuthConnectListener: () => undefined, - useOAuthReturnRouter: () => undefined, -})) - -import { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/indexed' -import { OrganizationIntegrations } from '@/app/o/[organizationId]/integrations/integrations' -import { - type RowAction, - RowActionsMenu, -} from '@/app/workspace/[workspaceId]/settings/components/row-actions-menu' -import { organizationAccountsKeys } from '@/hooks/queries/organization-accounts' - -deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(() => - createMockDeploymentShape({ features: { liveEnterpriseSearch: mocks.live } }) -) - -const mockUseOrganizationContext = organizationProviderMockFns.mockUseOrganizationContext -const mockUseOrganizationAccounts = organizationAccountsQueriesMockFns.mockUseOrganizationAccounts -const mockUseSearchSources = kbConnectorsQueriesMockFns.mockUseSearchSources -const mockUseSearchSourceOverview = kbConnectorsQueriesMockFns.mockUseSearchSourceOverview - -const scope = { kind: 'organization', organizationId: 'organization-a' } as const -const account = { - credentialId: 'account', - displayName: 'My work account', - status: 'active' as const, -} -const memberSource: SearchSourceSummary = { - knowledgeBaseId: 'search-index', - connectorId: 'source-a', - connectorType: 'gmail', - sourceDescription: 'Inbox', - accessMode: 'members', - availability: 'available', - enabled: true, - isSyncing: false, - lastSyncAt: null, - hasSyncError: false, - hasViewerDocuments: false, - viewerFailedDocumentCount: 0, - viewerEmailVerified: true, - viewerAccounts: [], - connectionRequired: true, - viewerMembership: 'not_enrolled', - approved: true, -} -const centralSource: SearchSourceSummary = { - ...memberSource, - connectorId: 'central', - connectorType: 'google_drive', - sourceDescription: 'Shared Drive', - accessMode: 'admin', - connectionRequired: false, - viewerMembership: null, -} - -let root: Root -let container: HTMLDivElement -let rows: SearchSourceSummary[] -let queryOverrides: Record -beforeEach(() => { - mocks.live = false - vi.spyOn(toast, 'error').mockReturnValue('toast') - vi.stubGlobal('IS_REACT_ACT_ENVIRONMENT', true) - mocks.setupConnector = null - organizationAccountsQueriesMockFns.mockUseConnectOrganizationAccount.mockReturnValue({ - mutate: mocks.connectOrganizationAccount, - isPending: false, - }) - organizationAccountsQueriesMockFns.mockUseReconnectPersonalOrganizationAccount.mockReturnValue({ - mutate: mocks.reconnectOrganizationAccount, - isPending: false, - }) - mockUseOrganizationAccounts.mockReturnValue({ - data: { credentialGroup: null, viewerAccounts: [] }, - isPending: false, - isError: false, - refetch: mocks.refetchAccounts, - }) - rows = [memberSource] - queryOverrides = {} - mockUseOrganizationContext.mockReturnValue({ - organization: { id: scope.organizationId }, - viewer: { isAdmin: false }, - searchAccess: { memberScoped: true, sourceMirrored: true }, - }) - mocks.filters.mockReturnValue({ search: '' }) - mocks.slackInventory.mockReturnValue({ - data: { available: [] }, - isPending: false, - isError: false, - }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'gmail', approved: true }], - isPending: false, - }) - mockUseSearchSourceOverview.mockReturnValue({ - data: { - providers: [{ connectorType: 'gmail', isSyncing: false }], - hasSearchableDocuments: false, - }, - isPending: false, - }) - mocks.availability.mockReturnValue({ - integrationAvailability: new Map(), - oauthServiceAvailability: new Map([ - ['google-email', true], - ['confluence', true], - ['jira', true], - ]), - isIntegrationAvailabilityReady: true, - integrationAvailabilityError: null, - }) - mockUseSearchSources.mockImplementation( - (_scope: unknown, options: { enabled: boolean; connectorType?: string }) => ({ - data: options.enabled - ? rows.filter((row) => row.connectorType === options.connectorType) - : undefined, - isPending: false, - isError: false, - isFetching: false, - isFetchNextPageError: false, - hasNextPage: false, - fetchNextPage: mocks.nextPage, - refetch: mocks.refetch, - ...queryOverrides, - }) - ) - container = document.createElement('div') - document.body.appendChild(container) - root = createRoot(container) -}) -afterEach(async () => { - await act(async () => root.unmount()) - container.remove() -}) -async function render(searchParams = '', element: ReactNode = ) { - await act(async () => - root.render( - - {element} - - ) - ) -} -function buttons(label: string) { - return Array.from(document.querySelectorAll('button')).filter( - (button) => button.textContent?.trim() === label - ) -} - -async function openMenu(name: string) { - const trigger = document.querySelector( - `[aria-label="${name} integration actions"]` - )! - await act(async () => - trigger.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', bubbles: true })) - ) -} -function menuItem(label: string) { - return [...document.querySelectorAll('[role="menuitem"]')].find( - (item) => item.textContent === label - )! -} - -describe('GitHub member account inventory', () => { - const githubAccount = { - credentialId: 'github-account', - providerId: 'github-repositories', - groupId: 'accounts-group', - optionId: 'github-option', - displayName: 'My GitHub', - status: 'active' as const, - } - const githubGroup = { - id: 'accounts-group', - status: 'active', - options: [{ id: 'github-option', provider: 'github-repositories', status: 'active' }], - } - - beforeEach(() => { - rows = ['repo-one', 'repo-two'].map((connectorId) => ({ - ...memberSource, - connectorId, - connectorType: 'github', - sourceDescription: connectorId, - })) - mockUseSearchSourceOverview.mockReturnValue({ - data: { providers: [{ connectorType: 'github' }] }, - isPending: false, - }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'github', approved: true }], - isPending: false, - }) - mocks.availability.mockReturnValue({ - integrationAvailability: new Map(), - oauthServiceAvailability: new Map([['github-repositories', true]]), - isIntegrationAvailabilityReady: true, - }) - mockUseOrganizationAccounts.mockReturnValue({ - data: { credentialGroup: githubGroup, viewerAccounts: [] }, - isPending: false, - isError: false, - refetch: mocks.refetchAccounts, - }) - }) - - it('connects once through the account operation', async () => { - await render() - expect(buttons('Connect')).toHaveLength(1) - expect(container.textContent).toContain('Connect once') - await act(async () => buttons('Connect')[0].click()) - expect(mocks.connectOrganizationAccount).toHaveBeenCalledExactlyOnceWith( - { organizationId: scope.organizationId, optionId: 'github-option' }, - expect.any(Object) - ) - expect(mockUseSearchSources).not.toHaveBeenCalled() - expect(mocks.connect).not.toHaveBeenCalled() - expect(mocks.connectSearchSource).not.toHaveBeenCalled() - expect(document.querySelector('[role="dialog"]')).toBeNull() - expect(container.textContent).not.toContain('repo-one') - }) - - it('keeps one account row when an admin adds another repository', async () => { - mockUseOrganizationAccounts.mockReturnValue({ - data: { credentialGroup: githubGroup, viewerAccounts: [githubAccount] }, - isPending: false, - }) - await render() - rows.push({ - ...rows[0], - connectorId: 'future-repository', - sourceDescription: 'future-repository', - }) - await render() - expect(document.querySelectorAll('[aria-label="GitHub integration actions"]')).toHaveLength(1) - expect(mocks.accountMenu).toHaveBeenLastCalledWith( - expect.objectContaining({ accounts: [githubAccount] }) - ) - expect(buttons('Connect')).toHaveLength(0) - expect(buttons('Reconnect')).toHaveLength(0) - expect(mockUseSearchSources).not.toHaveBeenCalled() - expect(container.textContent).not.toContain('future-repository') - }) - - it('keeps an owned account visible before any repository source exists', async () => { - mockUseSearchSourceOverview.mockReturnValue({ data: { providers: [] }, isPending: false }) - mocks.integrations.mockReturnValue({ data: [], isPending: false }) - mockUseOrganizationAccounts.mockReturnValue({ - data: { credentialGroup: githubGroup, viewerAccounts: [githubAccount] }, - isPending: false, - }) - await render('', ) - expect(container.textContent).toContain('My GitHub · Connected') - expect(document.querySelectorAll('[aria-label="GitHub integration actions"]')).toHaveLength(1) - expect(mockUseSearchSources).not.toHaveBeenCalled() - expect(buttons('Connect')).toHaveLength(0) - expect(container.textContent).not.toContain('No integrations are available') - }) - - it.each(['group', 'option'] as const)( - 'keeps Disconnect but hides Reconnect when the canonical %s is disabled', - async (disabled) => { - const expired = { ...githubAccount, status: 'needs_reauth' } - mockUseOrganizationAccounts.mockReturnValue({ - data: { - credentialGroup: { - ...githubGroup, - status: disabled === 'group' ? 'disabled' : 'active', - options: [ - { ...githubGroup.options[0], status: disabled === 'option' ? 'disabled' : 'active' }, - ], - }, - viewerAccounts: [expired], - }, - isPending: false, - }) - await render() - expect(buttons('Reconnect')).toHaveLength(0) - expect(mocks.accountMenu).toHaveBeenLastCalledWith( - expect.objectContaining({ accounts: [expired] }) - ) - await openMenu('GitHub') - expect(menuItem('Disconnect github-account')).toBeDefined() - } - ) - - it('allows personal reauthorization while Search is disabled', async () => { - mockUseSearchSourceOverview.mockReturnValue({ data: { providers: [] }, isPending: false }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'github', approved: false }], - isPending: false, - }) - mockUseOrganizationContext.mockReturnValue({ - organization: { id: scope.organizationId }, - searchAccess: { memberScoped: false, sourceMirrored: false }, - }) - mockUseOrganizationAccounts.mockReturnValue({ - data: { - credentialGroup: githubGroup, - viewerAccounts: [{ ...githubAccount, status: 'needs_reauth' }], - }, - isPending: false, - }) - await render() - expect(buttons('Reconnect')).toHaveLength(1) - await act(async () => buttons('Reconnect')[0].click()) - expect(mocks.reconnectOrganizationAccount).toHaveBeenCalledExactlyOnceWith( - 'github-account', - expect.any(Object) - ) - expect(mocks.connect).not.toHaveBeenCalled() - }) - - it('does not reconnect an account through a different active option', async () => { - mockUseOrganizationAccounts.mockReturnValue({ - data: { - credentialGroup: githubGroup, - viewerAccounts: [{ ...githubAccount, optionId: 'other-option', status: 'needs_reauth' }], - }, - isPending: false, - }) - await render() - expect(buttons('Reconnect')).toHaveLength(0) - expect(document.querySelector('[aria-label="GitHub integration actions"]')).not.toBeNull() - }) - - it.each(['pending', 'error'] as const)( - 'does not fall through to repository setup when the inventory is %s', - async (state) => { - mockUseOrganizationAccounts.mockReturnValue({ - data: undefined, - isPending: state === 'pending', - isError: state === 'error', - error: state === 'error' ? new Error('Could not load accounts') : null, - isFetching: false, - refetch: mocks.refetchAccounts, - }) - await render() - expect(container.textContent).toContain('GitHub') - expect(mockUseSearchSources).not.toHaveBeenCalled() - expect(buttons('Connect')).toHaveLength(0) - expect(mocks.connectSearchSource).not.toHaveBeenCalled() - if (state === 'error') { - await act(async () => buttons('Retry')[0].click()) - expect(mocks.refetchAccounts).toHaveBeenCalledOnce() - } - } - ) - - it('retains legacy account management only after a successful response omits the inventory', async () => { - mockUseOrganizationAccounts.mockReturnValue({ - data: { credentialGroup: githubGroup }, - isPending: false, - isError: false, - }) - rows = rows.map((source) => ({ - ...source, - viewerMembership: 'connected', - viewerAccounts: [githubAccount], - })) - await render() - expect(mockUseSearchSources).toHaveBeenCalledWith(scope, { - connectorType: 'github', - enabled: true, - }) - expect(document.querySelectorAll('[aria-label="GitHub integration actions"]')).toHaveLength(1) - expect(mocks.accountMenu).toHaveBeenLastCalledWith( - expect.objectContaining({ accounts: [githubAccount] }) - ) - expect(buttons('Connect')).toHaveLength(0) - }) -}) - -describe('grouped member integrations', () => { - it('renders one provider row and loads bounded pages per configured provider', async () => { - mockUseSearchSourceOverview.mockReturnValue({ - data: { providers: [{ connectorType: 'gmail' }, { connectorType: 'google_drive' }] }, - isPending: false, - }) - rows = Array.from({ length: 25 }, (_, index) => ({ - ...memberSource, - connectorId: `gmail-${index}`, - })) - await render() - expect(buttons('Connect')).toHaveLength(1) - expect(container.textContent).toContain('Google Drive') - expect(container.textContent).not.toContain('Inbox') - expect(mockUseSearchSources).toHaveBeenCalledWith(scope, { - connectorType: 'gmail', - enabled: true, - }) - expect(mockUseSearchSources).toHaveBeenCalledWith(scope, { - connectorType: 'google_drive', - enabled: true, - }) - expect(document.querySelector('[role="dialog"]')).toBeNull() - }) - it('keeps the list flat even when an old details URL is opened', async () => { - rows = [ - memberSource, - { ...memberSource, connectorId: 'source-b', sourceDescription: 'Archive' }, - ] - await render('?integration=gmail') - expect(document.querySelector('[role="region"]')).toBeNull() - expect(document.querySelector('[role="dialog"]')).toBeNull() - expect(container.textContent).not.toContain('Inbox') - expect(container.textContent).not.toContain('Archive') - expect(document.querySelector('[aria-label="Gmail integration actions"]')).toBeNull() - expect(buttons('Connect')).toHaveLength(1) - }) - it('connects one configured target even with many same-provider content scopes', async () => { - rows = [ - memberSource, - { ...memberSource, connectorId: 'source-b', sourceDescription: 'Archive' }, - ] - await render('?integration=gmail') - expect(buttons('Connect')).toHaveLength(1) - await act(async () => buttons('Connect')[0].click()) - expect(mocks.connect).toHaveBeenCalledExactlyOnceWith('search-index', 'source-a') - }) - it('deduplicates the same account across scopes', async () => { - rows = [memberSource, { ...memberSource, connectorId: 'source-b' }].map((source) => ({ - ...source, - viewerMembership: 'connected', - viewerAccounts: [account], - })) - await render('?integration=gmail') - expect(mocks.accountMenu).toHaveBeenCalledWith(expect.objectContaining({ accounts: [account] })) - expect(buttons('Connect')).toHaveLength(0) - expect(buttons('Reconnect')).toHaveLength(0) - expect(container.textContent).toContain('Connected') - }) - - it('distinguishes same-name accounts by content and renewal state only when needed', async () => { - rows = [ - { ...memberSource, sourceDescription: 'Engineering', viewerAccounts: [account] }, - { - ...memberSource, - connectorId: 'source-b', - sourceDescription: 'Handbook', - viewerAccounts: [{ ...account, credentialId: 'expired', status: 'needs_reauth' }], - }, - ] - await render() - expect(mocks.accountMenu.mock.calls.at(-1)?.[0].accountLabels).toEqual( - new Map([ - ['account', 'Engineering · My work account'], - ['expired', 'Reconnect required · Handbook · My work account'], - ]) - ) - }) - - it('gives otherwise identical accounts distinct connection labels', async () => { - rows = [ - { - ...memberSource, - viewerAccounts: [account, { ...account, credentialId: 'second' }], - }, - ] - await render() - expect(mocks.accountMenu.mock.calls.at(-1)?.[0].accountLabels).toEqual( - new Map([ - ['account', 'Connection 1 · Inbox · My work account'], - ['second', 'Connection 2 · Inbox · My work account'], - ]) - ) - }) - it.each(['personal', 'central'] as const)( - 'keeps existing %s content connected when another source needs authorization', - async (kind) => { - const connected: SearchSourceSummary = - kind === 'personal' - ? { ...memberSource, viewerMembership: 'connected', viewerAccounts: [account] } - : { ...centralSource, connectorType: 'gmail' } - rows = [connected, { ...memberSource, connectorId: 'additional-source' }] - await render() - expect(container.textContent).toContain('Additional connection required') - expect(container.textContent).not.toContain('Not connected') - expect(buttons('Connect')).toHaveLength(1) - expect(buttons('Reconnect')).toHaveLength(0) - await act(async () => buttons('Connect')[0].click()) - expect(mocks.connect).toHaveBeenCalledExactlyOnceWith('search-index', 'additional-source') - } - ) - it('does not ask for authorization again when connected content fails to sync', async () => { - rows = [memberSource, { ...memberSource, connectorId: 'source-b' }].map((source) => ({ - ...source, - viewerMembership: 'connected', - viewerAccounts: [account], - hasSyncError: true, - })) - await render() - expect(container.textContent).toContain('Sync needs attention') - expect(buttons('Connect')).toHaveLength(0) - expect(buttons('Reconnect')).toHaveLength(0) - }) - it('reconnects the expired target before any connected account', async () => { - rows = [ - { ...memberSource, viewerMembership: 'connected', viewerAccounts: [account] }, - { - ...memberSource, - connectorId: 'expired-source', - viewerMembership: 'needs_reauth', - viewerAccounts: [{ ...account, credentialId: 'expired-account', status: 'needs_reauth' }], - }, - { ...memberSource, connectorId: 'unconnected-source' }, - ] - await render('?integration=gmail') - await act(async () => buttons('Reconnect')[0].click()) - expect(mocks.connect).toHaveBeenCalledExactlyOnceWith('search-index', 'expired-source') - }) - it('explains central connections without asking for a personal account', async () => { - mockUseSearchSourceOverview.mockReturnValue({ - data: { providers: [{ connectorType: 'google_drive', isSyncing: false }] }, - isPending: false, - }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'google_drive', approved: true }], - isPending: false, - }) - rows = [centralSource] - await render('?integration=google_drive') - expect(container.textContent).toContain('Google Drive') - expect(buttons('Connect')).toHaveLength(0) - }) - it.each([{ enabled: false }, { approved: false }, { availability: 'unavailable' as const }])( - 'retains own-account removal when a source cannot connect: %o', - async (override) => { - rows = [ - { - ...memberSource, - viewerMembership: 'needs_reauth', - viewerAccounts: [account], - ...override, - }, - ] - await render('?integration=gmail') - expect(mocks.accountMenu).toHaveBeenCalledWith( - expect.objectContaining({ accounts: [account] }) - ) - expect(buttons('Reconnect')).toHaveLength(0) - } - ) - it.each(['members', 'admin'] as const)( - 'does not claim connected when multi-scope %s access is disabled', - async (accessMode) => { - mockUseOrganizationContext.mockReturnValue({ - organization: { id: scope.organizationId }, - searchAccess: { memberScoped: false, sourceMirrored: false }, - }) - rows = [memberSource, { ...memberSource, connectorId: 'source-b' }].map((source) => ({ - ...source, - accessMode, - isSyncing: true, - viewerMembership: 'connected', - viewerAccounts: [account], - })) - await render() - expect(container.textContent).toContain('Some connections need attention') - expect(container.textContent).not.toContain('Indexing') - expect(buttons('Connect')).toHaveLength(0) - expect(mocks.accountMenu).toHaveBeenCalledWith( - expect.objectContaining({ accounts: [account] }) - ) - } - ) - it('keeps email verification as account recovery and returns to the integration list', async () => { - rows = [{ ...memberSource, viewerEmailVerified: false, viewerMembership: 'unverified_email' }] - await render() - expect(container.textContent).toContain('Verify your email') - expect(buttons('Connect')).toHaveLength(0) - const recovery = container.querySelector('a[href^="/verify"]')! - expect(new URL(recovery.href).searchParams.get('redirectAfter')).toBe( - '/o/organization-a/integrations' - ) - }) - it.each([{ enabled: false }, { approved: false }, { availability: 'unavailable' as const }])( - 'does not offer verification for blocked content: %o', - async (override) => { - rows = [memberSource, { ...memberSource, connectorId: 'source-b' }].map((source) => ({ - ...source, - ...override, - viewerEmailVerified: false, - viewerMembership: 'unverified_email', - })) - await render() - expect(container.querySelector('a[href^="/verify"]')).toBeNull() - expect(container.textContent).toContain('Some connections need attention') - expect(buttons('Connect')).toHaveLength(0) - } - ) - it('preserves explicit pagination instead of pretending loaded scope counts are complete', async () => { - queryOverrides = { hasNextPage: true } - rows = [] - await render('?integration=gmail') - expect(container.textContent).toContain('More connections to check') - await act(async () => buttons('Check connections')[0].click()) - expect(mocks.nextPage).toHaveBeenCalledOnce() - expect(buttons('Connect')).toHaveLength(0) - }) - it.each(['needs_reauth', 'not_enrolled'] as const)( - 'exposes an older %s source without marking a partial inventory connected', - async (membership) => { - rows = Array.from({ length: 25 }, (_, index) => ({ - ...memberSource, - connectorId: `source-${index}`, - viewerMembership: 'connected', - viewerAccounts: [account], - })) - queryOverrides = { hasNextPage: true } - await render() - expect(container.textContent).toContain('More connections to check') - expect(container.textContent).not.toContain('Connected') - expect(buttons('Connect')).toHaveLength(0) - await act(async () => buttons('Check connections')[0].click()) - expect(mocks.nextPage).toHaveBeenCalledOnce() - queryOverrides = { hasNextPage: true, isFetchingNextPage: true, isFetching: true } - await render() - expect(buttons('Checking')[0]).toBeDisabled() - rows = [ - ...rows, - { ...memberSource, connectorId: 'older-source', viewerMembership: membership }, - ] - queryOverrides = { hasNextPage: false } - await render() - expect(buttons('Check connections')).toHaveLength(0) - const action = membership === 'needs_reauth' ? 'Reconnect' : 'Connect' - await act(async () => buttons(action)[0].click()) - expect(mocks.connect).toHaveBeenCalledExactlyOnceWith('search-index', 'older-source') - } - ) - it('retries a failed connection read directly from its flat row', async () => { - queryOverrides = { isError: true, error: new Error('Could not load') } - await render() - expect(buttons('Connect')).toHaveLength(0) - expect(document.querySelector('[role="region"]')).toBeNull() - await act(async () => buttons('Retry')[0].click()) - expect(mocks.refetch).toHaveBeenCalledOnce() - }) - it('keeps later connection pages retryable without an expanded section', async () => { - queryOverrides = { hasNextPage: true, isError: true, isFetchNextPageError: true } - rows = [] - await render() - expect(container.textContent).toContain('Could not check remaining connections') - await act(async () => buttons('Retry')[0].click()) - expect(mocks.nextPage).toHaveBeenCalledOnce() - expect(document.querySelector('[role="region"]')).toBeNull() - }) - it('offers direct Connect only for a new eligible provider, with no duplicate scope row', async () => { - mockUseSearchSourceOverview.mockReturnValue({ data: { providers: [] }, isPending: false }) - await render() - expect(buttons('Connect')).toHaveLength(1) - await act(async () => buttons('Connect')[0].click()) - expect(mocks.connectSearchSource).toHaveBeenCalledWith( - scope, - expect.objectContaining({ type: 'gmail' }) - ) - expect(document.querySelector('[role="dialog"]')).toBeNull() - }) - it.each([ - { data: undefined, isPending: true, isError: false }, - { data: undefined, isPending: false, isError: true, error: new Error('Slack unavailable') }, - ])( - 'keeps other integrations usable when Slack inventory is unavailable: %o', - async (inventory) => { - mocks.slackInventory.mockReturnValue(inventory) - await render() - expect(buttons('Connect')).toHaveLength(1) - await act(async () => buttons('Connect')[0].click()) - expect(mocks.connect).toHaveBeenCalledExactlyOnceWith('search-index', 'source-a') - expect(container.textContent).toContain('Gmail') - } - ) - it.each([false, true])( - 'preserves shared Slack onboarding without a duplicate row (configured: %s)', - async (configured) => { - mockUseSearchSourceOverview.mockReturnValue({ - data: { providers: configured ? [{ connectorType: 'slack' }] : [] }, - isPending: false, - }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'slack', approved: true }], - isPending: false, - }) - mocks.availability.mockReturnValue({ - integrationAvailability: new Map([['slack_v2', { state: 'ready', oauthAvailable: true }]]), - oauthServiceAvailability: new Map([['slack', true]]), - isIntegrationAvailabilityReady: true, - }) - mocks.slackInventory.mockReturnValue({ - data: { available: [{ target: { connectorType: 'slack' } }] }, - isPending: false, - isError: false, - }) - rows = configured ? [{ ...memberSource, connectorType: 'slack' }] : [] - await render() - expect(buttons('Connect')).toHaveLength(1) - expect(mocks.slackInventory).toHaveBeenCalledWith({ - organizationId: scope.organizationId, - connectorType: 'slack', - }) - await act(async () => buttons('Connect')[0].click()) - if (configured) { - expect(mocks.connect).toHaveBeenCalledExactlyOnceWith('search-index', 'source-a') - expect(mocks.connectSearchSource).not.toHaveBeenCalled() - } else { - expect(mocks.connectSearchSource).toHaveBeenCalledExactlyOnceWith( - scope, - expect.objectContaining({ type: 'slack' }) - ) - expect(mocks.connect).not.toHaveBeenCalled() - } - expect(document.querySelector('[role="dialog"]')).toBeNull() - } - ) - it('keeps Slack setup errors relevant to the selected integration filter', async () => { - mocks.integrations.mockReturnValue({ - data: [ - { connectorType: 'gmail', approved: true }, - { connectorType: 'slack', approved: true }, - ], - isPending: false, - }) - mocks.slackInventory.mockReturnValue({ - isPending: false, - isError: true, - error: new Error('Could not load Slack setup'), - }) - await render('', ) - expect(container.textContent).not.toContain('Could not load Slack setup') - expect(buttons('Connect')).toHaveLength(1) - await render('', ) - expect(container.textContent).toContain('Could not load Slack setup') - expect(container.textContent).not.toContain('No integrations are available to connect') - expect(container.textContent).not.toContain('No matching integrations') - }) - it.each([ - { data: { available: [] }, isPending: false, isError: false }, - { - data: { available: [{ target: { connectorType: 'slack', connectorId: 'existing' } }] }, - isPending: false, - isError: false, - }, - { data: undefined, isPending: true, isError: false }, - { data: undefined, isPending: false, isError: true, error: new Error('Could not load Slack') }, - ])('withholds new Slack setup without a ready shared-app target: %o', async (inventory) => { - mockUseSearchSourceOverview.mockReturnValue({ data: { providers: [] }, isPending: false }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'slack', approved: true }], - isPending: false, - }) - mocks.availability.mockReturnValue({ - integrationAvailability: new Map([['slack_v2', { state: 'ready', oauthAvailable: true }]]), - oauthServiceAvailability: new Map([['slack', true]]), - isIntegrationAvailabilityReady: true, - }) - mocks.slackInventory.mockReturnValue(inventory) - await render() - expect(buttons('Connect')).toHaveLength(0) - expect(mocks.connectSearchSource).not.toHaveBeenCalled() - }) - it('withholds new setup on failed/incomplete provider data', async () => { - mockUseSearchSourceOverview.mockReturnValue({ - data: { providers: [{ connectorType: 'confluence', isSyncing: false }] }, - isPending: false, - }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'confluence', approved: true }], - isPending: false, - }) - rows = [{ ...memberSource, connectorType: 'confluence' }] - queryOverrides = { isError: true, error: new Error('Could not load'), hasNextPage: false } - await render('?integration=confluence') - expect( - mocks.accountMenu.mock.calls.at(-1)?.[0].actions.map((action: RowAction) => action.label) - ).toEqual([]) - queryOverrides = { hasNextPage: true } - await render('?integration=confluence') - expect( - mocks.accountMenu.mock.calls.at(-1)?.[0].actions.map((action: RowAction) => action.label) - ).toEqual([]) - }) - it('keeps the integration menu open during background indexing refreshes', async () => { - mockUseSearchSourceOverview.mockReturnValue({ - data: { providers: [{ connectorType: 'confluence', isSyncing: true }] }, - isPending: false, - }) - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'confluence', approved: true }], - isPending: false, - }) - rows = [{ ...memberSource, connectorType: 'confluence' }] - await render('?integration=confluence') - const trigger = document.querySelector( - '[aria-label="Confluence integration actions"]' - )! - await act(async () => - trigger.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', bubbles: true })) - ) - expect(document.querySelector('[role="menu"]')).not.toBeNull() - queryOverrides = { isFetching: true } - await render('?integration=confluence') - expect(document.querySelector('[aria-label="Confluence integration actions"]')).toBe(trigger) - expect(document.querySelector('[role="menu"]')).not.toBeNull() - }) - it('retains typed connection requests and Slack onboarding on the main page', async () => { - const connectionRequest = { - userId: 'person', - target: { - type: 'link' as const, - provider: 'google-email', - connectorType: 'gmail', - connectorId: 'source-a', - }, - } - await render( - '', - - ) - expect(mocks.request).toHaveBeenCalledWith( - expect.objectContaining({ ...connectionRequest, organizationId: scope.organizationId }) - ) - expect(container.textContent).toContain('Return to Slack') - }) - it('keeps scoped enrollment invalidations and toast error handling', async () => { - await render('?integration=gmail') - const options = mocks.enrollment.mock.calls.at(-1)?.[0] as { - membershipQueryKeys: unknown[] - onConnectionError: (message: string) => void - } - expect(options.membershipQueryKeys).toContainEqual( - organizationAccountsKeys.detail(scope.organizationId) - ) - options.onConnectionError('Choose the matching account') - expect(toast.error).toHaveBeenCalledExactlyOnceWith('Choose the matching account') - expect(document.body.textContent).not.toContain('Choose the matching account') - }) -}) - -describe('live integrations backend selection', () => { - it('connects through existing OAuth enrollment without loading indexed sources', async () => { - mocks.live = true - mocks.integrations.mockReturnValue({ - data: [{ connectorType: 'google_drive', approved: true }], - }) - mockUseOrganizationAccounts.mockReturnValue({ - data: { - availableMcpConnectors: [], - credentialGroup: { - status: 'active', - mcpServers: [], - options: [ - { - id: 'drive', - provider: 'google-drive', - label: 'Google Drive', - status: 'active', - configurationStatus: 'ready', - }, - ], - }, - viewerAccounts: [], - }, - isError: false, - }) - await render('', ) - await act(async () => buttons('Connect')[0].click()) - expect(mocks.connectOrganizationAccount).toHaveBeenCalledWith( - { organizationId: scope.organizationId, optionId: 'drive' }, - expect.any(Object) - ) - expect(mockUseSearchSources).not.toHaveBeenCalled() - expect(mockUseSearchSourceOverview).not.toHaveBeenCalled() - expect(mocks.integrations).toHaveBeenCalledWith(scope.organizationId) - expect(mocks.connectSearchSource).not.toHaveBeenCalled() - }) - it('offers reconnect for an existing personal grant', async () => { - mocks.live = true - mocks.integrations.mockReturnValue({ data: [{ connectorType: 'slack', approved: true }] }) - mockUseOrganizationAccounts.mockReturnValue({ - data: { - availableMcpConnectors: [], - credentialGroup: { - status: 'active', - mcpServers: [], - options: [ - { - id: 'slack', - provider: 'slack', - label: 'Slack', - status: 'active', - configurationStatus: 'ready', - }, - ], - }, - viewerAccounts: [ - { - credentialId: 'my-slack', - providerId: 'slack', - optionId: 'slack', - displayName: 'My Slack', - status: 'needs_reauth', - }, - ], - }, - isError: false, - }) - await render('', ) - await act(async () => buttons('Reconnect')[0].click()) - expect(mocks.reconnectOrganizationAccount).toHaveBeenCalledWith('my-slack', expect.any(Object)) - expect(container.textContent).toContain('Reconnect needed') - }) -}) diff --git a/apps/sim/app/o/[organizationId]/integrations/integrations.tsx b/apps/sim/app/o/[organizationId]/integrations/integrations.tsx index 5a92f870368..0869432d66d 100644 --- a/apps/sim/app/o/[organizationId]/integrations/integrations.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/integrations.tsx @@ -1,11 +1,9 @@ 'use client' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' import { SEARCH_DEBOUNCE_MS } from '@/lib/url-state' import { OrganizationPage } from '@/app/o/[organizationId]/components/organization-page' import { useOrganizationPageFilters } from '@/app/o/[organizationId]/components/organization-page/use-organization-page-filters' -import { MemberIntegrationsList } from '@/app/o/[organizationId]/integrations/indexed' import { LiveMemberIntegrations } from '@/app/o/[organizationId]/integrations/live-member-integrations' import { SlackSearchActions } from '@/app/o/[organizationId]/integrations/slack-search-actions' import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' @@ -25,16 +23,12 @@ export function OrganizationIntegrations({ useOAuthReturnRouter() useDesktopOAuthConnectListener() const { organization } = useOrganizationContext() - const { features } = useDeploymentShape() const { search } = useOrganizationPageFilters() const sourceSearch = useDebounce(search.trim(), SEARCH_DEBOUNCE_MS) return ( )} - {features.liveEnterpriseSearch ? ( - - ) : ( - - )} + ) } diff --git a/apps/sim/app/o/[organizationId]/integrations/page.test.tsx b/apps/sim/app/o/[organizationId]/integrations/page.test.tsx index af60c390fb9..5728acdeee4 100644 --- a/apps/sim/app/o/[organizationId]/integrations/page.test.tsx +++ b/apps/sim/app/o/[organizationId]/integrations/page.test.tsx @@ -25,31 +25,6 @@ beforeEach(() => { }) describe('integrations page Slack context', () => { - it('preserves a requested connection across login and validates it in the existing organization page', async () => { - const selected = { - ...props, - searchParams: Promise.resolve({ - connectorType: 'gmail', - connectorId: 'source', - credentialId: 'account', - }), - } - const page = await OrganizationIntegrationsPage(selected) - expect(page.props.connectionRequest).toMatchObject({ - userId: 'viewer', - target: { - type: 'link', - connectorType: 'gmail', - connectorId: 'source', - credentialId: 'account', - }, - }) - authMockFns.mockGetSession.mockResolvedValue(null) - await expect(OrganizationIntegrationsPage(selected)).rejects.toThrow('NEXT_REDIRECT') - expect(mockRedirect).toHaveBeenCalledWith( - `/login?callbackUrl=${encodeURIComponent('/o/organization-a/integrations?connectorType=gmail&connectorId=source&credentialId=account')}` - ) - }) it('rejects unknown providers and reconnects without a source', async () => { for (const query of [ { connectorType: 'invented' }, diff --git a/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx b/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx deleted file mode 100644 index 35ad5bdf2f7..00000000000 --- a/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/page.tsx +++ /dev/null @@ -1,91 +0,0 @@ -import { ChipLink } from '@sim/emcn' -import { notFound, redirect } from 'next/navigation' -import type { SearchParams } from 'nuqs/server' -import { readSearchDocumentResultSchema } from '@/lib/api/contracts/knowledge/documents' -import { getSession } from '@/lib/auth' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { readSearchDocument } from '@/lib/sim-search/indexed' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import { buildAuthCrossLink } from '@/app/(auth)/auth-redirect' -import { - loadDocumentReadParams, - serializeDocumentReadParams, -} from '@/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/search-params' -import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -interface OrganizationDocumentPageProps { - params: Promise<{ organizationId: string; knowledgeBaseId: string; documentId: string }> - searchParams: Promise -} - -export default async function OrganizationDocumentPage({ - params, - searchParams, -}: OrganizationDocumentPageProps) { - if (!isIndexedOrgSearchEnabled()) notFound() - const { organizationId, knowledgeBaseId, documentId } = await params - const position = await loadDocumentReadParams(searchParams, { strict: true }).catch(() => - notFound() - ) - const href = `/o/${encodeURIComponent(organizationId)}/knowledge/${encodeURIComponent(knowledgeBaseId)}/${encodeURIComponent(documentId)}` - const session = await getSession() - if (!session?.user) { - redirect( - buildAuthCrossLink('/login', { - callbackUrl: serializeDocumentReadParams(href, position), - isInviteFlow: false, - }) - ) - } - const registry = new ResolvedSecretTraceRegistry() - let result: Awaited> - try { - result = await readSearchDocument.execute({ - principal: { kind: 'session', userId: session.user.id, sessionId: session.session.id }, - input: { - documentId, - assertedOrganizationId: organizationId, - ...position, - limit: 3, - resultSecretRegistry: registry, - }, - }) - } catch (error) { - if ( - error instanceof OrchestrationError && - (error.code === 'not_found' || error.code === 'forbidden' || error.code === 'validation') - ) - notFound() - throw error - } - if (result.knowledgeBaseId !== knowledgeBaseId) notFound() - const projected = projectResolvedSecretModelContent(result, registry, 1024 * 1024) - if (!projected.safe) return

This document cannot be displayed safely.

- const document = readSearchDocumentResultSchema.parse(projected.value) - return ( -
-
-

- {document.documentName ?? 'Document'} -

- {document.chunks.map((chunk) => ( -

- {chunk.content} -

- ))} - -
-
- ) -} diff --git a/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/search-params.ts b/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/search-params.ts deleted file mode 100644 index 42f9dd224e8..00000000000 --- a/apps/sim/app/o/[organizationId]/knowledge/[knowledgeBaseId]/[documentId]/search-params.ts +++ /dev/null @@ -1,25 +0,0 @@ -import { createLoader, createParser, createSerializer } from 'nuqs/server' - -const parseAsDocumentPosition = createParser({ - parse: (value) => { - if (!/^\d+$/.test(value)) return null - const position = Number(value) - return Number.isSafeInteger(position) && position <= 2147483647 ? position : null - }, - serialize: String, -}).withDefault(0) - -export const documentReadParams = { - startChunkIndex: parseAsDocumentPosition, - startOffset: parseAsDocumentPosition, -} - -const documentReadUrlKeys = { - urlKeys: { - startChunkIndex: 'start-chunk-index', - startOffset: 'start-offset', - }, -} as const - -export const loadDocumentReadParams = createLoader(documentReadParams, documentReadUrlKeys) -export const serializeDocumentReadParams = createSerializer(documentReadParams, documentReadUrlKeys) diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts deleted file mode 100644 index 658a877fbc1..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/index.ts +++ /dev/null @@ -1 +0,0 @@ -export { IndexedOrganizationIntegrationsSettings } from '@/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings' diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx deleted file mode 100644 index 714b6b164ec..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/indexed-organization-integrations-settings.tsx +++ /dev/null @@ -1,139 +0,0 @@ -'use client' - -import { useState } from 'react' -import { Chip, ChipConfirmModal, ChipModalError, ChipSwitch, toast } from '@sim/emcn' -import { useQueryStates } from 'nuqs' -import { getOrganizationAccountUpdateOptions } from '@/lib/credential-groups/organization-account-options' -import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' -import { OrganizationIntegrationsSetup } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup' -import { OrganizationSourcePeople } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people' -import { OrganizationSourceStats } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats' -import { - organizationIntegrationsTabParam, - organizationPeopleIntegrationParam, -} from '@/app/o/[organizationId]/settings/components/integrations/search-params' -import { RowActionsMenu } from '@/app/workspace/[workspaceId]/settings/components/row-actions-menu' -import { - SettingsEmptyState, - SettingsQueryErrorState, -} from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' -import { - useOrganizationAccounts, - useUpdateOrganizationAccounts, -} from '@/hooks/queries/organization-accounts' - -/** - * Organization Integrations settings for indexed organization search: the Sources, People, and - * Stats tabs over the connectors that crawl the search index. Rendered only while the deployment - * serves indexed search (`features.liveEnterpriseSearch === false`). - */ -export function IndexedOrganizationIntegrationsSettings() { - const { organization, viewer } = useOrganizationContext() - const [{ tab }, setNavigation] = useQueryStates({ - [organizationIntegrationsTabParam.key]: organizationIntegrationsTabParam.parser, - [organizationPeopleIntegrationParam.key]: organizationPeopleIntegrationParam.parser, - }) - const accounts = useOrganizationAccounts(viewer.isAdmin ? organization.id : undefined) - const update = useUpdateOrganizationAccounts() - const [refreshOpen, setRefreshOpen] = useState(false) - const group = accounts.data?.credentialGroup - const refreshConnections = () => { - if (!group || update.isPending) return - update.mutate( - { - organizationId: organization.id, - groupId: group.id, - update: { options: getOrganizationAccountUpdateOptions(group) }, - }, - { - onSuccess: () => { - setRefreshOpen(false) - toast.success('Sign-in settings updated') - }, - } - ) - } - if (!viewer.isAdmin) return null - - const tabs = ( - void setNavigation({ tab: value, integration: null })} - options={[ - { value: 'providers', label: 'Sources' }, - { value: 'people', label: 'People' }, - { value: 'stats', label: 'Stats' }, - ]} - /> - ) - - return ( -
- {tab === 'providers' && ( -
- {tabs} - {!accounts.error && group && group.options.length > 0 && ( - { - update.reset() - setRefreshOpen(true) - }, - }, - ]} - /> - )} -
- )} - { - if (!update.isPending) setRefreshOpen(open) - }} - title='Update sign-in settings?' - text='Apply Sim’s current OAuth app and permission settings to member connections. People whose settings changed must reconnect. This does not sync content.' - confirm={{ label: 'Update', pending: update.isPending, onClick: refreshConnections }} - > - {update.error?.message} - - {tab === 'providers' && } - {tab === 'stats' && } - {tab === 'people' && ( - void accounts.refetch()} - variant='inline' - /> - ) : !accounts.data ? ( - Loading connected accounts - ) : !accounts.data.credentialGroup ? ( -
- - Add a source that uses member accounts before requesting connections. - - void setNavigation({ tab: 'providers', integration: null })}> - View sources - -
- ) : undefined - } - /> - )} -
- ) -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup.tsx deleted file mode 100644 index 83fd54b8354..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup.tsx +++ /dev/null @@ -1,188 +0,0 @@ -'use client' - -import { toast } from '@sim/emcn' -import { Plus } from '@sim/emcn/icons' -import { useQueryStates } from 'nuqs' -import { SettingsPanel } from '@/components/settings/settings-panel' -import { organizationRoutes } from '@/lib/navigation/paths' -import { getConnectorAccessAvailability, SEARCH_SOURCE_TYPES } from '@/lib/sim-search/connectors' -import { searchSetupAccessParam, searchSetupParam } from '@/lib/sim-search/search-params' -import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' -import { AddOrganizationSourceModal } from '@/app/o/[organizationId]/settings/components/integrations/add-organization-source-modal' -import { organizationSearchStatusLabel } from '@/app/o/[organizationId]/settings/components/integrations/organization-search-status' -import { SearchSourceSetup } from '@/app/o/[organizationId]/settings/components/integrations/search-source-setup' -import { OrganizationSlackAccountSetup } from '@/app/o/[organizationId]/settings/components/integrations/slack-account-setup' -import { IntegrationTile } from '@/app/workspace/[workspaceId]/integrations/components/integrations-showcase' -import { - SettingsEmptyState, - SettingsQueryErrorState, -} from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' -import { - RESOURCE_LIST_STACK, - SettingsResourceRow, -} from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' -import { useSettingsSearch } from '@/app/workspace/[workspaceId]/settings/components/use-settings-search' -import { useOrganizationSearchOverview } from '@/hooks/queries/kb/connectors' -import { useUpdateSearchIntegration } from '@/hooks/queries/search-integrations' -import { usePermissionConfig } from '@/hooks/use-permission-config' - -export function OrganizationIntegrationsSetup() { - const { organization, viewer, searchAccess } = useOrganizationContext() - const [search, setSearch] = useSettingsSearch() - const [setup, setSetup] = useQueryStates( - { - [searchSetupParam.key]: searchSetupParam.parser, - [searchSetupAccessParam.key]: searchSetupAccessParam.parser, - }, - { history: 'replace' } - ) - const overview = useOrganizationSearchOverview(organization.id, { enabled: viewer.isAdmin }) - const availability = usePermissionConfig() - const approval = useUpdateSearchIntegration() - const providers = new Map( - overview.data?.providers.map((provider) => [provider.connectorType, provider]) - ) - const sources = SEARCH_SOURCE_TYPES.map(([type, meta]) => ({ - type, - meta, - access: getConnectorAccessAvailability(meta, availability.integrationAvailability, { - memberAccessAvailable: searchAccess.memberScoped, - mirroredAccessAvailable: searchAccess.sourceMirrored, - oauthServiceAvailability: availability.oauthServiceAvailability, - isIntegrationAvailabilityReady: availability.isIntegrationAvailabilityReady, - }), - })) - const query = search.trim().toLowerCase() - const visible = sources.flatMap((source) => { - const provider = providers.get(source.type) - return provider && - (provider.approved || provider.sourceCount > 0) && - source.meta.name.toLowerCase().includes(query) - ? [{ ...source, provider }] - : [] - }) - const ready = Boolean( - !overview.isPending && - !overview.isError && - availability.isIntegrationAvailabilityReady && - !availability.integrationAvailabilityError - ) - const closePicker = () => { - if (!approval.isPending) void setSetup({ addConnector: null, 'source-access': null }) - } - const selectSource = (type: string, accessMode: 'admin' | 'members') => { - const selectedType = searchSetupParam.parser.parse(type) - if (!selectedType || !ready || approval.isPending) return - const startSetup = () => - void setSetup({ - addConnector: selectedType, - 'source-access': accessMode === 'members' ? 'members' : null, - }) - if (providers.get(type)?.approved) { - startSetup() - return - } - approval.mutate( - { organizationId: organization.id, connectorType: type, approved: true }, - { onSuccess: startSetup, onError: (error) => toast.error(error.message) } - ) - } - if (!viewer.isAdmin) return null - if (!searchAccess.memberScoped && !searchAccess.sourceMirrored) - return ( - - Search sources are not enabled for this organization. - - ) - - const feedback = overview.isError ? ( - void overview.refetch()} - variant='inline' - /> - ) : availability.integrationAvailabilityError ? ( - void availability.refetchIntegrationAvailability()} - variant='inline' - /> - ) : null - - return ( - <> - void setSetup({ addConnector: '', 'source-access': null }), - }, - ]} - search={{ value: search, onChange: setSearch, placeholder: 'Search sources' }} - > - {setup.addConnector !== '' && feedback} -
- {overview.isError ? null : overview.isPending ? ( - Loading sources - ) : visible.length === 0 ? ( - - {query ? 'No matching sources' : 'No sources yet. Add a source to get started.'} - - ) : ( - visible.map(({ type, meta, access, provider }) => { - const available = access.admin || access.members - const status = - provider.approved && ready && !available - ? 'Unavailable in this deployment' - : organizationSearchStatusLabel(provider) - return ( - } - title={meta.name} - description={[ - status, - provider.sourceCount > 0 - ? `${provider.sourceCount} ${provider.sourceCount === 1 ? 'connection' : 'connections'}` - : undefined, - ] - .filter(Boolean) - .join(' · ')} - href={organizationRoutes(organization.id).searchProvider(type)} - clickLabel={`Manage ${meta.name}`} - navigable - /> - ) - }) - )} -
-
- {setup.addConnector === '' ? ( - - ) : ( - - )} - - - ) -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period.tsx deleted file mode 100644 index cff874b2785..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period.tsx +++ /dev/null @@ -1,89 +0,0 @@ -'use client' - -import { useRef, useState } from 'react' -import { Calendar, ChipCombobox, Popover, PopoverAnchor, PopoverContent, toast } from '@sim/emcn' -import { formatDateShort } from '@/lib/core/utils/date-display' -import { getSearchStatsRangeError, type SEARCH_STATS_PERIODS } from '@/lib/knowledge/search/stats' - -const PERIOD_OPTIONS = [ - { value: 'today', label: 'Today' }, - { value: '3d', label: 'Past 3 days' }, - { value: '7d', label: 'Past 7 days' }, - { value: '14d', label: 'Past 14 days' }, - { value: '30d', label: 'Past 30 days' }, - { value: '90d', label: 'Past 90 days' }, - { value: 'custom', label: 'Custom range' }, -] satisfies { value: (typeof SEARCH_STATS_PERIODS)[number]; label: string }[] - -interface SearchStatsPeriodSelection { - period: (typeof SEARCH_STATS_PERIODS)[number] - startDate: string | null - endDate: string | null -} - -interface OrganizationSearchStatsPeriodProps extends SearchStatsPeriodSelection { - onChange: (selection: SearchStatsPeriodSelection) => void -} - -export function OrganizationSearchStatsPeriod({ - period, - startDate, - endDate, - onChange, -}: OrganizationSearchStatsPeriodProps) { - const triggerContainerRef = useRef(null) - const calendarRef = useRef(null) - const [calendarOpen, setCalendarOpen] = useState(false) - const label = - period === 'custom' && startDate && endDate && !getSearchStatsRangeError({ startDate, endDate }) - ? `${formatDateShort(startDate)} – ${formatDateShort(endDate)}` - : PERIOD_OPTIONS.find((option) => option.value === period)?.label - - return ( -
- { - const selected = PERIOD_OPTIONS.find((option) => option.value === value) - if (!selected) return - if (selected.value === 'custom') setCalendarOpen(true) - else onChange({ period: selected.value, startDate: null, endDate: null }) - }} - /> - - - calendarRef.current?.focus()} - onCloseAutoFocus={() => - triggerContainerRef.current?.querySelector('[role="combobox"]')?.focus() - } - > - setCalendarOpen(false)} - onRangeChange={(start, end) => { - const error = getSearchStatsRangeError({ startDate: start, endDate: end }) - if (error) { - toast.error(error) - return - } - onChange({ period: 'custom', startDate: start, endDate: end }) - setCalendarOpen(false) - }} - /> - - -
- ) -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people.tsx deleted file mode 100644 index 338ba91c5d9..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-people.tsx +++ /dev/null @@ -1,79 +0,0 @@ -'use client' - -import type { ComponentProps, ReactNode } from 'react' -import { ChipSelect } from '@sim/emcn' -import { useQueryState } from 'nuqs' -import type { CredentialGroupOption } from '@/lib/api/contracts/credential-groups' -import { getCredentialGroupIndexingConnector } from '@/lib/credential-groups/indexing' -import { organizationPeopleIntegrationParam } from '@/app/o/[organizationId]/settings/components/integrations/search-params' -import { OrganizationAccountPeople } from '@/ee/credential-groups/components/organization-account-people' - -interface OrganizationSourcePeopleProps - extends Omit, 'searchConnection' | 'filters'> { - options: CredentialGroupOption[] - tabs: ReactNode -} - -export function OrganizationSourcePeople({ - options, - tabs, - ...props -}: OrganizationSourcePeopleProps) { - const [integration, setIntegration] = useQueryState( - organizationPeopleIntegrationParam.key, - organizationPeopleIntegrationParam.parser - ) - const integrations = options - .flatMap((option) => { - const connector = getCredentialGroupIndexingConnector(option.provider) - return option.status === 'active' && connector - ? [ - { - optionId: option.id, - type: connector.type, - name: connector.meta.name, - icon: connector.meta.icon, - needsSetup: option.provider === 'slack' && option.configurationStatus !== 'ready', - }, - ] - : [] - }) - .sort((a, b) => a.name.localeCompare(b.name)) - const selected = integrations.find((item) => item.type === integration) - - return ( - -
- {tabs} - void setIntegration(value === 'all' ? null : value)} - disabled={props.enabled === false || Boolean(props.setupFallback)} - options={[ - { value: 'all', label: 'All integrations' }, - ...integrations.map((item) => ({ - value: item.type, - label: item.name, - icon: item.icon, - })), - ]} - /> -
- {selected?.needsSetup && ( -

- Update the Slack app from Sources before requesting connections. -

- )} - - } - /> - ) -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx deleted file mode 100644 index 0048318bfaa..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats.tsx +++ /dev/null @@ -1,226 +0,0 @@ -'use client' - -import { type ReactNode, useMemo } from 'react' -import { BarChart, Chip, ChipSelect, Tooltip } from '@sim/emcn' -import { CircleInfo } from '@sim/emcn/icons' -import { useQueryStates } from 'nuqs' -import { - SEARCH_STATS_PEOPLE_LIMIT, - SEARCH_STATS_SURFACE_LABELS, - SEARCH_STATS_SURFACES, -} from '@/lib/knowledge/search/stats' -import { OrganizationSearchStatsPeriod } from '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-search-stats-period' -import { SettingsEmptyState } from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' -import { SettingsPanel } from '@/app/workspace/[workspaceId]/settings/components/settings-panel' -import { - RESOURCE_LIST_STACK, - SettingsResourceRow, -} from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' -import { SettingsSection } from '@/app/workspace/[workspaceId]/settings/components/settings-section/settings-section' -import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' -import { - organizationSearchStatsParsers, - organizationSearchStatsUrlOptions, -} from '@/ee/organization-search-stats/search-params' -import { useOrganizationSearchStats } from '@/hooks/queries/organization-search-stats' - -const SURFACE_OPTIONS = [ - { value: 'all', label: 'All surfaces' }, - ...SEARCH_STATS_SURFACES.map((surface) => ({ - value: surface, - label: SEARCH_STATS_SURFACE_LABELS[surface], - })), -] - -interface OrganizationSourceStatsProps { - organizationId: string - tabs?: ReactNode -} - -function sourceLabel(sourceType: string) { - return ( - CONNECTOR_META_REGISTRY[sourceType]?.name ?? (sourceType === 'uploads' ? 'Uploads' : sourceType) - ) -} - -export function OrganizationSourceStats({ organizationId, tabs }: OrganizationSourceStatsProps) { - const [{ period, surface, startDate, endDate }, setFilters] = useQueryStates( - organizationSearchStatsParsers, - organizationSearchStatsUrlOptions - ) - const stats = useOrganizationSearchStats({ - organizationId, - period, - surface: surface ?? undefined, - ...(period === 'custom' - ? { startDate: startDate ?? undefined, endDate: endDate ?? undefined } - : {}), - }) - const series = useMemo( - () => - stats.data?.series.map((point) => ({ - timestamp: point.timestamp, - value: point.invocations, - })) ?? [], - [stats.data?.series] - ) - const data = stats.data - const totals = data?.totals - const metrics = totals - ? [ - { label: 'Search invocations', value: totals.invocations.toLocaleString() }, - { label: 'Active people', value: totals.activePeople.toLocaleString() }, - { label: 'Results returned', value: totals.results.toLocaleString() }, - ] - : [] - - return ( - -
- {tabs} -
- - void setFilters({ surface: organizationSearchStatsParsers.surface.parse(value) }) - } - /> - void setFilters(selection)} - /> -
-
- {stats.isError ? ( - - Couldn’t load Search stats. void stats.refetch()}>Try again - - ) : !data || !totals ? ( - Loading Search stats - ) : ( - <> -
- {metrics.map((metric) => ( -
-
{metric.label}
-
{metric.value}
-
- ))} -
- - - - - - Successful Search requests since tracking was enabled. Assistant and MCP counts - are Search tool calls. Results count each document once per request. One request - can return multiple sources. - - - } - action={ - - {period === 'custom' ? 'UTC' : 'UTC · Includes today'} - - } - > - - - {!surface && totals.invocations > 0 && ( - -
- {data.surfaces.map((row) => ( - - {row.invocations.toLocaleString()} - - } - /> - ))} -
-
- )} - - Invocations returning this source - - } - > - {data.sources.length ? ( -
- {data.sources.map((row) => { - const Icon = CONNECTOR_META_REGISTRY[row.sourceType]?.icon - return ( - : undefined} - title={sourceLabel(row.sourceType)} - badge={ - - {row.invocations.toLocaleString()} - - } - /> - ) - })} -
- ) : ( - - No sources returned in this period. - - )} -
- - Top {SEARCH_STATS_PEOPLE_LIMIT} · Invocations - - } - > - {data.people.length ? ( -
- {data.people.map((person) => ( - - {person.invocations.toLocaleString()} - - } - /> - ))} -
- ) : ( - - No active people in this period. - - )} -
- - )} -
- ) -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx deleted file mode 100644 index d5c61f0370a..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.test.tsx +++ /dev/null @@ -1,456 +0,0 @@ -/** @vitest-environment jsdom */ -import { act } from 'react' -import { toast } from '@sim/emcn' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { - organizationAccountsQueriesMock, - organizationAccountsQueriesMockFns, -} from '@sim/testing/mocks/organization-accounts-queries.mock' -import { - organizationProviderMock, - organizationProviderMockFns, -} from '@sim/testing/mocks/organization-provider.mock' -import { NuqsTestingAdapter } from 'nuqs/adapters/testing' -import { createRoot, type Root } from 'react-dom/client' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const mocks = vi.hoisted(() => ({ - invite: vi.fn(), - refetch: vi.fn(), - update: vi.fn(), - updatePending: false, - updateError: null as Error | null, - resetUpdate: vi.fn(), -})) -vi.mock('@/app/o/[organizationId]/providers/organization-provider', () => organizationProviderMock) -vi.mock( - '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-integrations-setup', - () => ({ OrganizationIntegrationsSetup: () =>
Provider setup
}) -) -vi.mock( - '@/app/o/[organizationId]/settings/components/integrations/indexed/organization-source-stats', - () => ({ - OrganizationSourceStats: ({ organizationId }: { organizationId: string }) => ( -
Stats for {organizationId}
- ), - }) -) -vi.mock('@/hooks/queries/organization-accounts', () => organizationAccountsQueriesMock) - -import { SettingsHeaderProvider, SettingsHeaderShell } from '@/components/settings/settings-header' -import { resetDeploymentShape } from '@/lib/core/config/deployment-shape' -import { OrganizationIntegrationsSettings } from '@/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings' - -const mockContext = organizationProviderMockFns.mockUseOrganizationContext -const { mockUseOrganizationAccounts: mockAccounts, mockUseOrganizationAccountPeople: mockPeople } = - organizationAccountsQueriesMockFns -organizationAccountsQueriesMockFns.mockUseUpdateOrganizationAccounts.mockImplementation(() => ({ - mutate: mocks.update, - isPending: mocks.updatePending, - error: mocks.updateError, - reset: mocks.resetUpdate, -})) -organizationAccountsQueriesMockFns.mockUseInviteOrganizationAccountPeople.mockImplementation( - () => ({ - mutateAsync: mocks.invite, - reset: vi.fn(), - }) -) -organizationAccountsQueriesMockFns.mockUseResendOrganizationAccountInvitation.mockImplementation( - () => ({}) -) -organizationAccountsQueriesMockFns.mockUseRevokeOrganizationAccountEnrollment.mockImplementation( - () => ({}) -) - -describe('organization integration invitations', () => { - let root: Root - let container: HTMLDivElement - - beforeEach(() => { - /** These cover the indexed organization Integrations settings. */ - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - resetDeploymentShape() - vi.spyOn(toast, 'success').mockReturnValue('toast-id') - vi.spyOn(toast, 'error').mockReturnValue('toast-id') - mocks.updatePending = false - mocks.updateError = null - vi.stubGlobal('IS_REACT_ACT_ENVIRONMENT', true) - mockContext.mockReturnValue({ organization: { id: 'org-a' }, viewer: { isAdmin: true } }) - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { credentialGroup: { id: 'group-a', options: [] } }, - error: null, - refetch: mocks.refetch, - }) - mockPeople.mockReturnValue({ data: { pages: [{ enrollments: [] }] } }) - mocks.invite.mockResolvedValue({ - sentCount: 2, - results: [ - { email: 'one@example.com', success: true }, - { email: 'two@example.com', success: true }, - ], - }) - container = document.createElement('div') - document.body.appendChild(container) - root = createRoot(container) - }) - - afterEach(async () => { - await act(async () => root.unmount()) - container.remove() - resetEnvFlagsMock() - resetDeploymentShape() - }) - - async function render(searchParams = '') { - await act(async () => - root.render( - - - - - - - - ) - ) - } - - function findButton(label: string) { - const button = Array.from(document.querySelectorAll('button')).find( - (element) => element.textContent === label || element.getAttribute('aria-label') === label - ) - if (!button) throw new Error(`Missing ${label} button`) - return button - } - - async function click(label: string) { - await act(async () => findButton(label).click()) - } - - async function openRefresh() { - expect(container.textContent).not.toContain('Update configurations') - await act(async () => - findButton('More source actions').dispatchEvent( - new MouseEvent('pointerdown', { bubbles: true, button: 0 }) - ) - ) - const item = document.querySelector('[role="menuitem"]') - expect(item?.textContent).toBe('Update sign-in settings') - await act(async () => item?.click()) - expect(document.body.textContent).toContain('People whose settings changed must reconnect.') - } - - it('keeps provider setup as the default and sends manual invitations from People to this org', async () => { - await render() - expect(container.textContent).toContain('Provider setup') - expect(mockAccounts).toHaveBeenLastCalledWith('org-a') - expect(mockPeople).not.toHaveBeenCalled() - - await click('People') - expect(container.textContent).not.toContain('Provider setup') - expect(mockAccounts).toHaveBeenLastCalledWith('org-a') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: true }) - expect(container.querySelector('input[placeholder="Search people..."]')).not.toBeNull() - expect(mocks.invite).not.toHaveBeenCalled() - - await click('Request connections') - const input = document.querySelector('input[placeholder="Enter emails"]') - if (!input) throw new Error('Missing invitation email input') - const paste = new Event('paste', { bubbles: true, cancelable: true }) - Object.defineProperty(paste, 'clipboardData', { - value: { getData: () => 'one@example.com two@example.com' }, - }) - await act(async () => input.dispatchEvent(paste)) - await click('Send requests') - expect(mocks.invite).toHaveBeenCalledExactlyOnceWith({ - organizationId: 'org-a', - emails: ['one@example.com', 'two@example.com'], - }) - expect(document.querySelector('[role="dialog"]')).toBeNull() - }) - - it('refreshes saved provider identities only after choosing the maintenance action and confirming', async () => { - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { - credentialGroup: { - id: 'group-a', - options: [ - { - id: 'github-option', - provider: 'github-repositories', - label: 'Engineering', - required: true, - }, - { - id: 'slack-option', - provider: 'slack', - label: 'Slack', - required: false, - slackBotCredentialId: 'slack-bot', - requiredScopes: ['search:read'], - }, - ], - }, - }, - error: null, - }) - mocks.update.mockImplementationOnce((_input, { onSuccess }) => onSuccess()) - await render() - await openRefresh() - await click('Update') - expect(mocks.update).toHaveBeenCalledWith( - { - organizationId: 'org-a', - groupId: 'group-a', - update: { - options: [ - { - id: 'github-option', - provider: 'github-repositories', - label: 'Engineering', - required: true, - }, - { - id: 'slack-option', - provider: 'slack', - label: 'Slack', - required: false, - slackBotCredentialId: 'slack-bot', - }, - ], - }, - }, - expect.any(Object) - ) - expect(toast.success).toHaveBeenCalledWith('Sign-in settings updated') - - expect(document.querySelector('[role="dialog"]')).toBeNull() - }) - - it('keeps failed refreshes open for retry and blocks duplicate submissions', async () => { - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { credentialGroup: { id: 'group-a', options: [{ provider: 'gmail' }] } }, - error: null, - }) - await render() - await openRefresh() - await click('Update') - mocks.updateError = new Error('Update denied') - await render() - expect(document.body.textContent).toContain('Update denied') - expect(document.querySelector('[role="dialog"]')).not.toBeNull() - mocks.updatePending = true - await render() - expect(findButton('Update')).toBeDisabled() - expect(mocks.update).toHaveBeenCalledOnce() - }) - - it('does not offer maintenance without saved providers', async () => { - await render() - expect(container.querySelector('[aria-label="More source actions"]')).toBeNull() - }) - - it('opens People directly from the saved URL', async () => { - await render('?tab=people') - expect(container.textContent).toContain('Request connections') - expect(container.textContent).not.toContain('Provider setup') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: true }) - }) - - it('waits for integration options before loading filtered people or allowing invitations', async () => { - mockAccounts.mockReturnValue({ - isSuccess: false, - data: undefined, - error: null, - isPending: true, - }) - await render('?tab=people&integration=jira') - expect(mockAccounts).toHaveBeenLastCalledWith('org-a') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: false }) - expect(container.textContent).toContain('Loading connected accounts') - expect(container.textContent).not.toContain('No people invited yet') - expect(findButton('Request connections')).toBeDisabled() - await click('Request connections') - expect(document.querySelector('[role="dialog"]')).toBeNull() - - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { credentialGroup: { id: 'group-a', options: [] } }, - error: null, - }) - await render('?tab=people&integration=jira') - expect(container.textContent).not.toContain('Loading connected accounts') - expect(findButton('Request connections')).not.toBeDisabled() - await click('Request connections') - expect(document.querySelector('[role="dialog"]')).not.toBeNull() - expect(mocks.invite).not.toHaveBeenCalled() - }) - - it('stops the people query when setup resolves without a pool and preserves the setup action', async () => { - mockAccounts.mockReturnValue({ - isSuccess: false, - data: undefined, - error: null, - isPending: true, - }) - mockPeople.mockReturnValue({ error: new Error('Organization accounts not configured') }) - await render('?tab=people&integration=jira') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: false }) - expect(container.textContent).not.toContain('Organization accounts not configured') - - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { credentialGroup: null }, - error: null, - }) - await render('?tab=people&integration=jira') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: false }) - expect(container.textContent).toContain('before requesting connections') - expect(container.textContent).not.toContain('Organization accounts not configured') - expect(findButton('Request connections')).toBeDisabled() - }) - - it('sends an org without a credential group back to provider setup before invitations', async () => { - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { credentialGroup: null }, - error: null, - }) - await render('?tab=people') - expect(container.textContent).toContain('before requesting connections') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: false }) - expect(findButton('Request connections')).toBeDisabled() - await click('View sources') - expect(container.textContent).toContain('Provider setup') - expect(mocks.invite).not.toHaveBeenCalled() - }) - - it('surfaces account lookup errors instead of treating them as missing setup', async () => { - mockAccounts.mockReturnValue({ - isSuccess: true, - error: new Error('Account access denied'), - refetch: mocks.refetch, - }) - await render('?tab=people') - expect(container.textContent).toContain('Account access denied') - expect(container.textContent).not.toContain('View sources') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: false }) - expect(findButton('Request connections')).toBeDisabled() - await click('Try again') - expect(mocks.refetch).toHaveBeenCalledOnce() - }) - - it('does not load admin account data or expose invitations to an ordinary member', async () => { - mockContext.mockReturnValue({ organization: { id: 'org-a' }, viewer: { isAdmin: false } }) - await render('?tab=people') - expect(container.textContent).toBe('') - expect(mockAccounts).toHaveBeenLastCalledWith(undefined) - expect(mockPeople).not.toHaveBeenCalled() - expect(mocks.invite).not.toHaveBeenCalled() - }) - it('opens organization stats without loading people', async () => { - await render() - await click('Stats') - expect(container.textContent).toContain('Stats for org-a') - expect(mockPeople).not.toHaveBeenCalled() - }) - - it('filters connection summaries and requests to the selected integration, then returns to All', async () => { - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { - credentialGroup: { - id: 'group-a', - options: [ - { id: 'jira-option', provider: 'jira', status: 'active' }, - { id: 'gmail-option', provider: 'gmail', status: 'active' }, - { id: 'old-option', provider: 'confluence', status: 'revoked' }, - ], - }, - }, - }) - await render('?tab=people&integration=jira&credential-group-people=alex') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', 'alex', { - enabled: true, - optionId: 'jira-option', - }) - expect(findButton('Filter people by integration').textContent).toContain('Jira') - await click('Request connections') - expect(document.querySelector('[role="dialog"]')?.textContent).toContain( - 'Request Jira connections' - ) - await click('Cancel') - await act(async () => - findButton('Filter people by integration').dispatchEvent( - new MouseEvent('pointerdown', { bubbles: true, button: 0 }) - ) - ) - const all = Array.from(document.querySelectorAll('[role="menuitem"]')).find( - (item) => item.textContent === 'All integrations' - ) - expect(all).toBeDefined() - expect(document.querySelector('[role="menu"]')?.textContent).not.toContain('Confluence') - await act(async () => all?.click()) - await vi.waitFor(() => - expect(mockPeople).toHaveBeenLastCalledWith('org-a', 'alex', { enabled: true }) - ) - expect(container.querySelector('input[placeholder="Search people..."]')).toHaveValue('alex') - }) - - it.each(['', '&integration=gmail'])( - 'defaults to All on navigation with one integration and initial filter %s', - async (filter) => { - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { - credentialGroup: { - id: 'group-a', - options: [{ id: 'gmail-option', provider: 'gmail', status: 'active' }], - }, - }, - }) - await render(`?tab=people${filter}`) - expect(findButton('Filter people by integration').textContent).toContain( - filter ? 'Gmail' : 'All integrations' - ) - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { - enabled: true, - ...(filter ? { optionId: 'gmail-option' } : {}), - }) - await click('Sources') - await click('People') - expect(findButton('Filter people by integration').textContent).toContain('All integrations') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { enabled: true }) - } - ) - - it('preserves Slack setup recovery in People without hiding existing connections', async () => { - mockAccounts.mockReturnValue({ - isSuccess: true, - data: { - credentialGroup: { - id: 'group-a', - options: [ - { - id: 'slack-option', - provider: 'slack', - status: 'active', - configurationStatus: 'needs_update', - }, - ], - }, - }, - }) - await render('?tab=people&integration=slack') - expect(mockPeople).toHaveBeenLastCalledWith('org-a', '', { - enabled: true, - optionId: 'slack-option', - }) - expect(findButton('Request connections')).toBeDisabled() - expect(container.textContent).toContain('Update the Slack app from Sources') - }) -}) diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx index 58942c418ad..035c95cb41c 100644 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-integrations-settings.tsx @@ -1,14 +1,7 @@ 'use client' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' -import { IndexedOrganizationIntegrationsSettings } from '@/app/o/[organizationId]/settings/components/integrations/indexed' import { LiveSearchSettings } from '@/app/o/[organizationId]/settings/components/integrations/live-search-settings' export function OrganizationIntegrationsSettings() { - const { features } = useDeploymentShape() - return features.liveEnterpriseSearch ? ( - - ) : ( - - ) + return } diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-search-status.ts b/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-search-status.ts deleted file mode 100644 index d4eed45ee37..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/organization-search-status.ts +++ /dev/null @@ -1,27 +0,0 @@ -import type { OrganizationSearchProviderSummary } from '@/lib/api/contracts/knowledge/connectors' - -const STATUS_LABELS: Record = { - needs_setup: 'Setup required', - waiting_for_connections: 'Waiting for connections', - indexing: 'Indexing', - needs_attention: 'Sync failed', - paused: 'Paused', - active: 'Ready to search', -} - -export function organizationSearchStatusLabel(provider: OrganizationSearchProviderSummary): string { - if (!provider.approved) return 'Deactivated' - if (provider.status === 'needs_setup' && provider.sourceCount > 0) return 'Waiting for first sync' - if (provider.status === 'needs_attention') { - const error = - provider.issue === 'account_sync_incomplete' - ? 'Some accounts are not up to date' - : provider.issue === 'permission_sync_incomplete' - ? 'Some permissions could not be verified' - : provider.issue === 'document_indexing_failed' - ? 'Some documents failed to index' - : 'Sync failed' - return provider.isSyncing ? `Indexing · ${error}` : error - } - return STATUS_LABELS[provider.status] -} diff --git a/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx b/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx index e228074fefb..f06463ec193 100644 --- a/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx +++ b/apps/sim/app/o/[organizationId]/settings/components/integrations/search-source-setup.test.tsx @@ -4,7 +4,6 @@ import { act, cloneElement, type ReactNode } from 'react' import { authClientMock, authClientMockFns } from '@sim/testing/mocks/auth-client.mock' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { kbConnectorsQueriesMock, kbConnectorsQueriesMockFns, @@ -406,16 +405,9 @@ afterEach(async () => { container?.remove() root = null container = null - resetEnvFlagsMock() resetDeploymentShape() }) -/** Selects the indexed backend, whose arms of these dialogs index and mirror sources. */ -function selectIndexedSearch() { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - resetDeploymentShape() -} - describe('Search source setup with real connector dialogs', () => { it.each([ ['source-one', '/o/org-1/settings/integrations/sources/source-one'], @@ -439,25 +431,6 @@ describe('Search source setup with real connector dialogs', () => { expect(document.querySelector('[role="dialog"]')).toBeNull() } ) - - it('prepares organization connected-account indexing in members mode even when central access is available', async () => { - selectIndexedSearch() - mocks.bases = [] - await render( - , - '?addConnector=slack' - ) - expect(mocks.prepare).toHaveBeenCalledWith({ - organizationId: 'org-1', - connectorType: 'slack', - accessMode: 'members', - }) - }) }) describe('member content credentials in real add and edit dialogs', () => { @@ -742,43 +715,6 @@ describe('administrator source prerequisites in real connector dialogs', () => { mocks.credentials = [driveCredential] }) - it.each(['admin', 'members'] as const)( - 'shows and saves Gmail’s Search default date window in %s mode', - async (accessMode) => { - selectIndexedSearch() - mocks.credentials = [ - { - id: 'gmail-service', - name: 'Gmail indexing', - provider: 'google-email', - type: 'service_account', - }, - ] - await render( - - ) - expect(document.body.textContent).toContain('Last 6 months') - expect(document.body.textContent).not.toContain('All time (default)') - if (accessMode === 'admin') await fill(adminEmailPlaceholder, 'admin@example.com') - await click(button(accessMode === 'admin' ? 'Connect & Sync' : 'Create & Invite')) - expect(mocks.create).toHaveBeenCalledWith( - expect.objectContaining({ - accessMode, - connectorType: 'gmail', - sourceConfig: expect.objectContaining({ dateRange: '6m' }), - }), - expect.any(Object) - ) - } - ) - it('preserves a deliberate Gmail date-range draft and keeps general KB defaults separate', async () => { const key = 'gmail-all-time' useConnectorSetupStore.getState().saveDraft(key, { @@ -929,14 +865,12 @@ describe('administrator source prerequisites in real connector dialogs', () => { ])( 'requires the Directory administrator email in $type administrator mode and refuses empty or blank subjects', async ({ type, provider }) => { - selectIndexedSearch() mocks.credentials = [{ ...driveCredential, provider }] await render( @@ -970,7 +904,6 @@ describe('administrator source prerequisites in real connector dialogs', () => { ])( 'excludes personal OAuth accounts and stale OAuth drafts from $type administrator setup', async ({ type, provider, name }) => { - selectIndexedSearch() const oauthCredential = { id: 'drive-personal', name: 'Personal Drive account', @@ -992,8 +925,7 @@ describe('administrator source prerequisites in real connector dialogs', () => { @@ -1097,42 +1029,6 @@ describe('administrator source prerequisites in real connector dialogs', () => { } ) - it('does not let an administrator erase the crawl subject from an existing mirrored Drive source', async () => { - selectIndexedSearch() - await render( - - ) - expect(document.body.textContent).toContain('Directory administrator email*') - await fill(adminEmailPlaceholder, '') - expect(button('Save')).toBeDisabled() - await click(button('Save')) - expect(mocks.update).not.toHaveBeenCalled() - await fill(adminEmailPlaceholder, 'replacement@example.com') - expect(button('Save')).toBeEnabled() - await click(button('Save')) - expect(mocks.update.mock.calls[0][0]).toMatchObject({ - connectorId: 'connector-1', - updates: { - sourceConfig: { - adminEmail: 'replacement@example.com', - fileType: 'documents', - }, - }, - }) - expect(mocks.applyAccess).not.toHaveBeenCalled() - }) - it('guides a general knowledge-base member source back to saving its crawl subject without losing drafts or combining mutations', async () => { const existing = connector({ connectorType: 'google_drive', @@ -1338,48 +1234,6 @@ describe('canonical Search connector safety', () => { ) }) - it('defaults an OAuth source to member accounts and never offers workspace-wide access', async () => { - selectIndexedSearch() - await render( - {}} - knowledgeBaseId='kb-search' - isSearchIndex - initialConnectorType='google_drive' - /> - ) - expect(button('Member accounts')).toHaveAttribute('aria-checked', 'true') - expect( - Array.from(document.querySelectorAll('button')).some( - (node) => node.textContent === 'Workspace' - ) - ).toBe(false) - expect(document.body.textContent).not.toContain('Everyone in this workspace') - await click(button('Choose another source')) - const gitlab = Array.from(document.querySelectorAll('button')).find( - (node) => node.getAttribute('aria-label') === 'GitLab' - ) - expect(gitlab).toBeDefined() - await click(gitlab!) - expect(button('Administrator token')).toHaveAttribute('aria-checked', 'true') - expect(button('Non-admin token')).toHaveAttribute('aria-checked', 'false') - expect(document.body.textContent).not.toContain('Connection method') - expect( - Array.from(document.querySelectorAll('button')).some( - (node) => node.textContent === 'Workspace' - ) - ).toBe(false) - await fill('Enter your GitLab PAT', 'fixture-pat') - await fill('gitlab.example.com', 'gitlab.example.test') - await fill('group/project or numeric ID', 'engineering/search') - await click(button('Connect & Sync')) - expect(mocks.create).toHaveBeenCalledWith( - expect.objectContaining({ accessMode: 'admin', connectorType: 'gitlab' }), - expect.any(Object) - ) - }) - it('keeps an existing Search member source out of workspace-wide mode', async () => { await render( - type === 'github' || (!liveSearch && (setup['source-access'] === 'members' || type === 'slack')) + type === 'github' || type === 'slack' || setup['source-access'] === 'members' ? ('members' as const) : ('admin' as const) @@ -199,10 +197,7 @@ export function SearchSourceSetup({ ) { if (selectedType && session?.user?.id) { const accessMode = initialMode(selectedType) - const setupMode = - liveSearch || !(selectedMeta?.mirrorsSourceAcls && selectedMeta.auth.mode === 'oauth') - ? accessMode - : 'choose' + const setupMode = accessMode return ( void setSelectedType(type !== null ? searchSetupParam.parser.parse(type) : null) @@ -224,7 +219,7 @@ export function SearchSourceSetup({ onCreated={async (type, connector) => { await setSelectedType(null) const destination = organizationRoutes(scope.organizationId).searchSource(connector.id) - if (liveSearch && accessMode === 'admin' && type !== 'gitlab') { + if (accessMode === 'admin' && type !== 'gitlab') { updateSearchIntegration.mutate( { organizationId: scope.organizationId, @@ -366,11 +361,7 @@ export function SearchSourceSetup({ isIntegrationAvailabilityReady, } ) - const available = - type === 'github' || - (!liveSearch && (setup['source-access'] === 'members' || type === 'slack')) - ? members - : central + const available = initialMode(type) === 'members' ? members : central return ( ({ authorize: vi.fn() })) vi.mock('@/lib/settings/application/organization-section-access', () => ({ @@ -27,16 +26,13 @@ import OrganizationProviderPage from '@/app/o/[organizationId]/settings/integrat const mockRedirect = nextNavigationMockFns.mockRedirect const mockGetSession = authMockFns.mockGetSession -/** Member providers such as Jira keep a provider page only under indexed organization search. */ beforeEach(() => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) mockGetSession.mockResolvedValue({ user: { id: 'admin-1' } }) mocks.authorize.mockResolvedValue(true) }) -afterEach(resetEnvFlagsMock) it.each(['jira', 'confluence'])( - 'moves legacy %s Accounts links to filtered People and preserves the search', + 'moves legacy %s Accounts links to current Sources settings', async (connectorType) => { await expect( OrganizationProviderPage({ @@ -49,10 +45,6 @@ it.each(['jira', 'confluence'])( ).rejects.toThrow('NEXT_REDIRECT') const url = new URL(mockRedirect.mock.lastCall![0], 'https://example.com') expect(url.pathname).toBe('/o/org-1/settings/integrations') - expect(url.searchParams.get('tab')).toBe('people') - expect(url.searchParams.get('integration')).toBe(connectorType) - expect(url.searchParams.get('credential-group-people')).toBe('alex+qa@example.com') - expect(url.searchParams.has('view')).toBe(false) } ) @@ -66,14 +58,3 @@ it('authorizes organization settings before redirecting a legacy link', async () ).rejects.toThrow('NEXT_NOT_FOUND') expect(mockRedirect).not.toHaveBeenCalled() }) - -it.each(['jira', ''])( - 'preserves an active setup in a legacy Accounts link (%s)', - async (addConnector) => { - await OrganizationProviderPage({ - params: Promise.resolve({ organizationId: 'org-1', connectorType: 'jira' }), - searchParams: Promise.resolve({ view: 'accounts', addConnector, 'source-access': 'members' }), - }) - expect(mockRedirect).not.toHaveBeenCalled() - } -) diff --git a/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/page.tsx b/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/page.tsx index f0782f40b20..6ecec2b7578 100644 --- a/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/page.tsx +++ b/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/page.tsx @@ -2,7 +2,6 @@ import { Suspense } from 'react' import type { Metadata } from 'next' import { notFound, redirect } from 'next/navigation' import { getSession } from '@/lib/auth' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { organizationRoutes } from '@/lib/navigation/paths' import { authorizeOrganizationSettingsSection } from '@/lib/settings/application/organization-section-access' import { SEARCH_SOURCE_TYPES } from '@/lib/sim-search/connectors' @@ -49,7 +48,7 @@ export default async function OrganizationProviderPage({ })) ) notFound() - if (isLiveEnterpriseSearchEnabled && !LIVE_SEARCH_SERVICE_PROVIDERS.includes(connectorType)) + if (!LIVE_SEARCH_SERVICE_PROVIDERS.includes(connectorType)) redirect(organizationRoutes(organizationId).settingsSection('integrations')) const query = await searchParams const activeSetup = diff --git a/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/provider-detail.tsx b/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/provider-detail.tsx index 84142d6f837..0c300e32171 100644 --- a/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/provider-detail.tsx +++ b/apps/sim/app/o/[organizationId]/settings/integrations/providers/[connectorType]/provider-detail.tsx @@ -1,21 +1,17 @@ 'use client' import { useState } from 'react' -import { ChipConfirmModal, ChipModalError } from '@sim/emcn' import { ArrowLeft, Plus } from '@sim/emcn/icons' -import { format } from 'date-fns' import { useRouter } from 'next/navigation' import { useQueryState, useQueryStates } from 'nuqs' import type { SettingsAction } from '@/components/settings/settings-header' import { SettingsPanel } from '@/components/settings/settings-panel' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import { organizationRoutes } from '@/lib/navigation/paths' import { getSearchConnectionLabels } from '@/lib/sim-search/connection-labels' import { getConnectorAccessAvailability } from '@/lib/sim-search/connectors' import { searchSetupAccessParam, searchSetupParam } from '@/lib/sim-search/search-params' import { SEARCH_DEBOUNCE_MS } from '@/lib/url-state' import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' -import { organizationSearchStatusLabel } from '@/app/o/[organizationId]/settings/components/integrations/organization-search-status' import { connectedAccountsParam } from '@/app/o/[organizationId]/settings/components/integrations/search-params' import { SearchSourcePagination } from '@/app/o/[organizationId]/settings/components/integrations/search-source-pagination' import { SearchSourceSetup } from '@/app/o/[organizationId]/settings/components/integrations/search-source-setup' @@ -31,9 +27,9 @@ import { } from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' import { useSettingsSearch } from '@/app/workspace/[workspaceId]/settings/components/use-settings-search' import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' -import { useOrganizationSearchOverview, useSearchSources } from '@/hooks/queries/kb/connectors' +import { useSearchSources } from '@/hooks/queries/kb/connectors' import { useOrganizationAccounts } from '@/hooks/queries/organization-accounts' -import { useUpdateSearchIntegration } from '@/hooks/queries/search-integrations' +import { useSearchIntegrations } from '@/hooks/queries/search-integrations' import { useDebounce } from '@/hooks/use-debounce' import { usePermissionConfig } from '@/hooks/use-permission-config' @@ -44,21 +40,18 @@ interface OrganizationProviderDetailProps { export function OrganizationProviderDetail({ connectorType }: OrganizationProviderDetailProps) { const { organization, viewer, searchAccess } = useOrganizationContext() const router = useRouter() - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch const meta = CONNECTOR_META_REGISTRY[connectorType] const [search, setSearch] = useSettingsSearch() const sourceSearch = useDebounce(search.trim(), SEARCH_DEBOUNCE_MS) - const [deactivating, setDeactivating] = useState(false) const [removingSlackAccounts, setRemovingSlackAccounts] = useState(false) const scope = { kind: 'organization', organizationId: organization.id } as const - const overview = useOrganizationSearchOverview(organization.id, { enabled: viewer.isAdmin }) + const overview = useSearchIntegrations(organization.id) const sources = useSearchSources(scope, { connectorType, search: sourceSearch, enabled: viewer.isAdmin, }) const availability = usePermissionConfig() - const approval = useUpdateSearchIntegration() const accounts = useOrganizationAccounts( viewer.isAdmin && connectorType === 'slack' ? organization.id : undefined ) @@ -73,7 +66,7 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid connectedAccountsParam.key, connectedAccountsParam.parser ) - const provider = overview.data?.providers.find((item) => item.connectorType === connectorType) + const provider = overview.data?.find((item) => item.connectorType === connectorType) const approved = provider?.approved === true const back = { text: 'Sources', @@ -103,13 +96,11 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid approved && unavailable ? 'Unavailable in this deployment' : provider - ? liveSearch - ? connectorType === 'gitlab' - ? 'Projects and permissions' - : connectorType === 'github' - ? 'GitHub App repositories' - : 'Service account connections' - : organizationSearchStatusLabel(provider) + ? connectorType === 'gitlab' + ? 'Projects and permissions' + : connectorType === 'github' + ? 'GitHub App repositories' + : 'Service account connections' : undefined, docsLink: meta.searchDocsUrl, search: searchField, @@ -135,30 +126,22 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid connectorType === 'slack' && (option?.provider !== 'slack' || option.configurationStatus !== 'ready') const pending = - overview.isPending || - overview.isError || - approval.isPending || - !availability.isIntegrationAvailabilityReady + overview.isPending || overview.isError || !availability.isIntegrationAvailabilityReady const startSource = () => void setSetup({ addConnector: searchSetupParam.parser.parse(connectorType), 'source-access': access.admin ? null : 'members', }) - const activate = () => - approval.mutate({ organizationId: organization.id, connectorType, approved: true }) const actions: SettingsAction[] = approved ? [ - ...(needsSlackSetup || - access.admin || - (connectorType === 'github' && access.members) || - (!liveSearch && access.members) + ...(needsSlackSetup || access.admin || (connectorType === 'github' && access.members) ? [ { text: needsSlackSetup ? 'Set up Slack app' - : liveSearch && connectorType === 'github' + : connectorType === 'github' ? 'Add repository' - : liveSearch && connectorType !== 'gitlab' + : connectorType !== 'gitlab' ? 'Add service account' : getSearchConnectionLabels(connectorType, access.admin ? 'admin' : 'members') .add, @@ -171,28 +154,15 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid }, ] : []), - ...(!liveSearch - ? [ - { - text: 'Deactivate', - disabled: approval.isPending, - onSelect: () => { - approval.reset() - setDeactivating(true) - }, - }, - ] - : []), ] : [ { - text: liveSearch ? 'View sources' : provider ? 'Activate' : 'Add integration', + text: 'View sources', variant: 'primary', disabled: pending || (!access.admin && !access.members), tooltip: unavailable ? 'This integration is unavailable in this deployment.' : undefined, - onSelect: liveSearch - ? () => router.push(organizationRoutes(organization.id).settingsSection('integrations')) - : activate, + onSelect: () => + router.push(organizationRoutes(organization.id).settingsSection('integrations')), }, ] actions.push(...removalActions) @@ -217,11 +187,6 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid const renderSources = () => ( - {approval.error && ( - - {approval.error.message} - - )} {availability.integrationAvailabilityError && ( - !liveSearch || source.accessMode === 'admin' || (connectorType === 'github' && source.isGitHubInstallation) ) @@ -263,41 +227,7 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid 0 - ? `${source.viewerFailedDocumentCount} ${source.viewerFailedDocumentCount === 1 ? 'document' : 'documents'} failed to index` - : source.isSyncing - ? 'Indexing' - : source.lastSyncAt - ? `Last synced ${format(new Date(source.lastSyncAt), 'MMM d, h:mm a')}` - : 'Waiting for the first sync', - ] - .filter(Boolean) - .join(' · ') - } + description={!approved ? 'Unavailable' : !source.enabled ? 'Paused' : undefined} href={organizationRoutes(organization.id).searchSource(source.connectorId)} clickLabel={`Open ${source.sourceDescription || meta.name}`} navigable @@ -305,7 +235,6 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid ))} {!sources.data?.some( (source) => - !liveSearch || source.accessMode === 'admin' || (connectorType === 'github' && source.isGitHubInstallation) ) && @@ -343,26 +272,6 @@ export function OrganizationProviderDetail({ connectorType }: OrganizationProvid onRemoved={() => setRemovingSlackAccounts(false)} /> )} - { - if (!approval.isPending) setDeactivating(open) - }} - title={`Deactivate ${meta.name}?`} - text='Its content will be unavailable in Search, Assistant, and MCP. Connections and accounts are preserved.' - confirm={{ - label: 'Deactivate', - variant: 'destructive', - pending: approval.isPending, - onClick: () => - approval.mutate( - { organizationId: organization.id, connectorType, approved: false }, - { onSuccess: () => setDeactivating(false) } - ), - }} - > - {approval.error?.message} - ) } diff --git a/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/search-params.ts b/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/search-params.ts deleted file mode 100644 index b4182204393..00000000000 --- a/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/search-params.ts +++ /dev/null @@ -1,14 +0,0 @@ -import { parseAsStringLiteral } from 'nuqs/server' -import { connectorDocumentFilterSchema } from '@/lib/api/contracts/knowledge/connectors' - -export const sourceViewParam = { - key: 'view', - parser: parseAsStringLiteral(['documents', 'settings', 'history']).withDefault('documents'), -} as const - -export const sourceDocumentFilterParam = { - key: 'document-filter', - parser: parseAsStringLiteral(connectorDocumentFilterSchema.options).withDefault('active'), -} as const - -export type SourceView = NonNullable> diff --git a/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.test.tsx b/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.test.tsx index 7abbbdeb7e3..e7d844179da 100644 --- a/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.test.tsx +++ b/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.test.tsx @@ -1,11 +1,6 @@ /** @vitest-environment jsdom */ import { act } from 'react' -import { - createMockDeploymentShape, - deploymentShapeMock, - deploymentShapeMockFns, -} from '@sim/testing/mocks/deployment-shape.mock' import { kbConnectorsQueriesMock, kbConnectorsQueriesMockFns, @@ -15,7 +10,6 @@ import { organizationProviderMock, organizationProviderMockFns, } from '@sim/testing/mocks/organization-provider.mock' -import { NuqsTestingAdapter } from 'nuqs/adapters/testing' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { ApiClientError } from '@/lib/api/client/errors' @@ -23,19 +17,14 @@ import type { ConnectorData } from '@/lib/api/contracts/knowledge/connectors' import type { ConnectorActionsOptions } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section/use-connector-actions' const mocks = vi.hoisted(() => ({ - live: false, admin: true, integrations: vi.fn(), - documents: vi.fn(), actions: vi.fn(), - recovery: vi.fn(), - history: vi.fn(), form: vi.fn(), dirty: false, saving: false, save: vi.fn(), })) -vi.mock('@/lib/core/config/deployment-shape', () => deploymentShapeMock) vi.mock('next/navigation', () => nextNavigationMock) vi.mock('@/app/o/[organizationId]/providers/organization-provider', () => organizationProviderMock) vi.mock('@/hooks/use-oauth-return', () => ({ useOAuthReturnForKBConnectors: vi.fn() })) @@ -52,25 +41,6 @@ vi.mock('@/connectors/registry', () => ({ }, }, })) -vi.mock( - '@/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-documents', - () => ({ - ConnectorDocuments: (props: unknown) => { - mocks.documents(props) - return

Source documents

- }, - }) -) -vi.mock('@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section', () => ({ - ConnectorRecovery: (props: { onEdit?: () => void }) => { - mocks.recovery(props) - return props.onEdit ? : null - }, - ConnectorSyncHistory: () => { - mocks.history() - return

Source sync history

- }, -})) vi.mock( '@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section/use-connector-actions', () => ({ @@ -101,9 +71,6 @@ nextNavigationMockFns.mockUsePathname.mockReturnValue( ) const mockIndex = kbConnectorsQueriesMockFns.mockUseSearchIndex const mockDetail = kbConnectorsQueriesMockFns.mockUseConnectorDetail -deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(() => - createMockDeploymentShape({ features: { liveEnterpriseSearch: mocks.live } }) -) organizationProviderMockFns.mockUseOrganizationContext.mockImplementation(() => ({ organization: { id: 'org-one' }, viewer: { isAdmin: mocks.admin }, @@ -143,7 +110,6 @@ describe('organization source detail navigation', () => { beforeEach(() => { vi.stubGlobal('IS_REACT_ACT_ENVIRONMENT', true) mocks.admin = true - mocks.live = false mocks.dirty = false mocks.saving = false mockIndex.mockReturnValue({ data: { knowledgeBaseId: 'index-one' }, isPending: false }) @@ -180,16 +146,14 @@ describe('organization source detail navigation', () => { await act(async () => root.unmount()) container.remove() }) - async function render(searchParams = '') { + async function render() { await act(async () => root.render( - - - - - - - + + + + + ) ) } @@ -214,7 +178,7 @@ describe('organization source detail navigation', () => { }) it('hides cached source data after access is revoked, even if the index also failed', async () => { - await render('?view=settings') + await render() mockIndex.mockReturnValue({ data: { knowledgeBaseId: 'index-one' }, isError: true, @@ -227,36 +191,36 @@ describe('organization source detail navigation', () => { error: new ApiClientError({ status: 403, message: 'Access denied', body: null }), refetch: vi.fn(), }) - await render('?view=settings') + await render() expect(container.textContent).toContain('Access denied') expect(container.textContent).not.toContain('Source configuration') }) it('hides cached source settings when integration status reports revoked access', async () => { - await render('?view=settings') + await render() mocks.integrations.mockReturnValue({ data: [{ connectorType: 'google_drive', approved: true }], isError: true, error: new ApiClientError({ status: 403, message: 'Access denied', body: null }), refetch: vi.fn(), }) - await render('?view=settings') + await render() expect(container.textContent).toContain('Access denied') expect(container.textContent).not.toContain('Source configuration') }) it('preserves the editable baseline across background connector updates', async () => { - await render('?view=settings') + await render() mockDetail.mockReturnValue({ data: { ...connector, status: 'syncing', sourceConfig: { folderId: 'changed-remotely' } }, }) - await render('?view=settings') + await render() expect(mocks.form).toHaveBeenLastCalledWith(expect.objectContaining({ connector })) }) it('uses the canonical saved row as the new settings baseline without leaving the source', async () => { mocks.dirty = true - await render('?view=settings') + await render() await click('Save') expect(mocks.save).toHaveBeenCalledOnce() @@ -267,16 +231,16 @@ describe('organization source detail navigation', () => { expect(container.textContent).toContain('Source configuration') mockDetail.mockReturnValue({ data: { ...connector, status: 'syncing' } }) - await render('?view=settings') + await render() expect(mocks.form).toHaveBeenLastCalledWith(expect.objectContaining({ connector: saved })) }) it('discards to the latest server settings only when explicitly requested', async () => { mocks.dirty = true - await render('?view=settings') + await render() const refreshed = { ...connector, sourceConfig: { folderId: 'latest-server-folder' } } mockDetail.mockReturnValue({ data: refreshed }) - await render('?view=settings') + await render() expect(mocks.form).toHaveBeenLastCalledWith(expect.objectContaining({ connector })) await click('Discard') diff --git a/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.tsx b/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.tsx index ba0b0812e79..f84cf589856 100644 --- a/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.tsx +++ b/apps/sim/app/o/[organizationId]/settings/integrations/sources/[connectorId]/source-detail.tsx @@ -1,34 +1,20 @@ 'use client' import { type ReactNode, useState } from 'react' -import { ChipLink, ChipModalTabs } from '@sim/emcn' +import { ChipLink } from '@sim/emcn' import { ArrowLeft } from '@sim/emcn/icons' import { useRouter } from 'next/navigation' -import { useQueryState } from 'nuqs' import { saveDiscardActions } from '@/components/settings/save-discard-actions' import type { SettingsAction, SettingsBackAction } from '@/components/settings/settings-header' import { SettingsPanel } from '@/components/settings/settings-panel' import { useSettingsUnsavedGuard } from '@/components/settings/use-settings-unsaved-guard' import { isApiClientError } from '@/lib/api/client/errors' import type { ConnectorData, ConnectorDetailData } from '@/lib/api/contracts/knowledge/connectors' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import type { ResourceScope } from '@/lib/core/resource-scope' -import { SOURCE_PERMISSION_ERROR } from '@/lib/knowledge/connectors/sync-limits' import { organizationRoutes } from '@/lib/navigation/paths' import { describeSearchSource } from '@/lib/sim-search/source-identity' -import { SEARCH_DEBOUNCE_MS } from '@/lib/url-state' import { useOrganizationContext } from '@/app/o/[organizationId]/providers/organization-provider' -import { - type SourceView, - sourceDocumentFilterParam, - sourceViewParam, -} from '@/app/o/[organizationId]/settings/integrations/sources/[connectorId]/search-params' import { UnsavedChangesModal } from '@/app/workspace/[workspaceId]/components/credential-detail/components/unsaved-changes-modal' -import { ConnectorDocuments } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-documents' -import { - ConnectorRecovery, - ConnectorSyncHistory, -} from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section' import { ConnectorActionFeedback } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section/connector-actions' import { getConnectorSyncState } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section/connector-sync-state' import { useConnectorActions } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connectors-section/use-connector-actions' @@ -39,7 +25,6 @@ import { SettingsQueryErrorState, } from '@/app/workspace/[workspaceId]/settings/components/settings-empty-state' import { SettingsResourceRow } from '@/app/workspace/[workspaceId]/settings/components/settings-resource-row' -import { useSettingsSearch } from '@/app/workspace/[workspaceId]/settings/components/use-settings-search' import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' import { isConnectorSyncingOrPending, @@ -47,15 +32,8 @@ import { useSearchIndex, } from '@/hooks/queries/kb/connectors' import { useSearchIntegrations } from '@/hooks/queries/search-integrations' -import { useDebounce } from '@/hooks/use-debounce' import { useOAuthReturnForKBConnectors } from '@/hooks/use-oauth-return' -const SOURCE_VIEWS = [ - { value: 'documents', label: 'Documents' }, - { value: 'settings', label: 'Settings' }, - { value: 'history', label: 'Sync history' }, -] as const - interface OrganizationSourceDetailProps { connectorId: string } @@ -63,7 +41,6 @@ interface OrganizationSourceDetailProps { export function OrganizationSourceDetail({ connectorId }: OrganizationSourceDetailProps) { const { organization, viewer } = useOrganizationContext() const router = useRouter() - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch const scope: ResourceScope = { kind: 'organization', organizationId: organization.id } const backHref = organizationRoutes(organization.id).settingsSection('integrations') const index = useSearchIndex(scope, { enabled: viewer.isAdmin }) @@ -128,7 +105,6 @@ export function OrganizationSourceDetail({ connectorId }: OrganizationSourceDeta
) if ( - liveSearch && detail.data.accessMode === 'members' && !(detail.data.connectorType === 'github' && detail.data.sourceConfig.githubRepositoryId) ) @@ -188,47 +164,22 @@ function SourceDetailContent({ const { organization } = useOrganizationContext() const integrations = useSearchIntegrations(organization.id) const router = useRouter() - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch - const [view, setView] = useQueryState( - sourceViewParam.key, - sourceViewParam.parser.withOptions({ history: 'replace' }) - ) - const [filter, setFilter] = useQueryState( - sourceDocumentFilterParam.key, - sourceDocumentFilterParam.parser - ) - const [search, setSearch] = useSettingsSearch() - const documentSearch = useDebounce(search.trim(), SEARCH_DEBOUNCE_MS) const meta = CONNECTOR_META_REGISTRY[connector.connectorType] const title = meta ? describeSearchSource(meta, connector.sourceConfig) || meta.name : 'Connection' - const { effectiveStatus, lastSyncError } = getConnectorSyncState(connector) - const permissionsIncomplete = - connector.lastSyncError?.split('\n').includes(SOURCE_PERMISSION_ERROR) ?? false + const { effectiveStatus } = getConnectorSyncState(connector) const status = effectiveStatus === 'paused' - ? liveSearch - ? 'Search paused' - : 'Sync paused' + ? 'Search paused' : effectiveStatus === 'disabled' - ? liveSearch - ? 'Search disabled' - : 'Sync disabled' - : effectiveStatus === 'error' - ? liveSearch - ? undefined - : 'Sync failed' - : undefined + ? 'Search disabled' + : undefined const description = [title === meta?.name ? undefined : meta?.name, status].filter(Boolean).join(' · ') || undefined const onBack = () => router.push(backHref) const onRemoved = () => router.replace(organizationRoutes(organization.id).settingsSection('integrations')) - const onViewChange = (value: string) => { - const next = sourceViewParam.parser.parse(value) - if (next) void setView(next) - } if ( integrations.isError && isApiClientError(integrations.error) && @@ -271,95 +222,18 @@ function SourceDetailContent({ )} ) - if (liveSearch || view === 'settings') - return ( - - ) return ( - - {integrationFeedback} - - {lastSyncError && - (effectiveStatus === 'active' || - (permissionsIncomplete && - (effectiveStatus === 'pending' || effectiveStatus === 'syncing'))) && ( - - )} - onViewChange('settings')} - /> - {view === 'documents' ? ( - void setFilter(next)} - progressScope={scope} - isSearchIndex - syncing={isConnectorSyncingOrPending(connector)} - /> - ) : ( - - )} - - ) -} - -interface SourceNavigationProps { - view: SourceView - onViewChange: (view: string) => void -} - -function SourceNavigation({ view, onViewChange }: SourceNavigationProps) { - return ( -
- -
+ /> ) } @@ -383,7 +257,6 @@ function SourcePanel({ children, ...panel }: SourcePanelProps) { - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch const lifecycle = useConnectorActions({ connector, knowledgeBaseId: connector.knowledgeBaseId, @@ -397,9 +270,9 @@ function SourcePanel({ {...panel} actions={[ ...lifecycle.actions - .filter((action) => !liveSearch || action.id !== 'sync') + .filter((action) => action.id !== 'sync') .map((action) => - liveSearch && action.id === 'pause' + action.id === 'pause' ? { ...action, text: connector.status === 'paused' ? 'Resume search' : 'Pause search', @@ -424,7 +297,6 @@ interface SourceSettingsEditorProps { backText: string onBack: () => void onRemoved: () => void - onViewChange: (view: string) => void } function SourceSettingsEditor(props: SourceSettingsEditorProps) { @@ -458,11 +330,9 @@ function SourceSettingsForm({ backText, onBack, onRemoved, - onViewChange, onSaved, onDiscard, }: SourceSettingsFormProps) { - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch const form = useConnectorSettingsForm({ connector: baseline, syncing: isConnectorSyncingOrPending(connector), @@ -491,15 +361,7 @@ function SourceSettingsForm({ })} > {queryError} - {!liveSearch && ( - { - if (next !== 'settings') guard.guardBack(() => onViewChange(next)) - }} - /> - )} - {liveSearch && connector.connectorType === 'gitlab' && ( + {connector.connectorType === 'gitlab' && ( (null) - const { - data: index, - isPending: basesPending, - isError: basesFailed, - isFetching: basesFetching, - refetch: refetchIndex, - } = useSearchIndex(scope) - const [filters] = useQueryStates(searchFilterParsers, resourceUrlKeys) - const custom = filters.updated === 'custom' - const pageFilters = useMemo( - () => searchFiltersFromParams(filters, searchedAt), - [filters.source, filters.updated, filters.from, filters.to, searchedAt] - ) - const searchFilters = suppliedFilters ?? pageFilters - const scopeId = scope.kind === 'organization' ? scope.organizationId : scope.workspaceId - useEffect(() => { - onSearchChange?.({ scope, query, filters: searchFilters, ...(topK ? { topK } : {}) }) - }, [scope.kind, scopeId, query, searchFilters, topK, onSearchChange]) - const filtersKey = JSON.stringify(searchFilters) - const expanded = expandedFor === filtersKey - /** A custom window is two-ended: until both days are chosen, nothing is searched. */ - const awaitingRange = !suppliedFilters && custom && !(filters.from && filters.to) - const { - data: search, - isPending, - isFetching, - isPlaceholderData, - isError: searchFailed, - refetch: refetchSearch, - } = useWorkspaceKnowledgeSearch( - scope, - awaitingRange ? '' : query, - searchFilters, - topK ?? - (expanded - ? WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.expanded - : WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.initial), - { retainAcrossLimits: topK === undefined } - ) - /** A full first page may collapse to few cards, yet more documents may still match. */ - const mayHaveMore = - topK === undefined && - !expanded && - (search?.results.length ?? 0) >= WORKSPACE_KNOWLEDGE_SEARCH_LIMITS.initial - const { data: overview } = useSearchSourceOverview(scope) - const indexing = (overview?.providers ?? []) - .filter((provider) => provider.isSyncing) - .map((provider) => connectorDisplayName(provider.connectorType)) - const documents = groupResultsByDocument(search?.results ?? []) - const sourceTypes = [ - ...new Set([ - ...(filters.source ? [filters.source] : []), - ...(overview?.providers.map((provider) => provider.connectorType) ?? []), - UPLOAD_SOURCE, - ]), - ].sort((left, right) => connectorDisplayName(left).localeCompare(connectorDisplayName(right))) - const failed = basesFailed || searchFailed - const pending = basesPending || isPending - const fetching = basesFetching || isFetching - const noSources = !basesPending && !basesFailed && !index?.knowledgeBaseId - const partial = search?.retrieval.status === 'partial' - const documentCount = documents.length === 1 ? '1 document' : `${documents.length} documents` - - const indexingNote = - indexing.length > 0 - ? `Still indexing ${indexing.join(', ')}; results grow as documents land.` - : null - - const showResults = !noSources && !failed && !basesPending && documents.length > 0 - /** A custom window waiting for its days must show the filters, or the picker is unreachable. */ - const showFilters = - hasShownFilters || - showResults || - awaitingRange || - (!noSources && !pending && !failed && !!search && !partial) - if (showFilters && !hasShownFilters) setHasShownFilters(true) - - return noSources ? ( -
-

No sources are set up yet.

- - View sources - -
- ) : ( -
-
-
- {awaitingRange ? ( -

- Choose the days to search. -

- ) : fetching ? ( - - ) : pending && !failed ? null : ( -

- {failed - ? 'Search couldn’t run.' - : partial - ? documents.length === 0 - ? 'Search timed out.' - : `${documentCount} · some results may be missing.` - : documents.length === 0 - ? 'Search found no results.' - : `${documentCount} · searched as you`} -

- )} - {indexingNote && !failed && !partial && ( -

{indexingNote}

- )} -
- {(failed || partial) && ( - void (basesFailed ? refetchIndex() : refetchSearch())} - > - {fetching ? 'Retrying' : 'Try again'} - - )} -
- {suppliedFilters === undefined && showFilters && } - {showResults && ( -
- {documents.map((result) => { - const source = toSource(result, query, scope) - return ( - - onSummarize(`Summarize "${cited.title ?? cited.url}"`, { - ...searchFilters, - documentIds: [result.documentId], - }) - } - /> - ) - })} - {mayHaveMore && ( -
- setExpandedFor(filtersKey)} - > - Show more - -
- )} -
- )} -
- ) -} diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx index 0505039e054..060e04932ff 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/knowledge-search-results/knowledge-search-results.tsx @@ -6,9 +6,7 @@ import { useQueryStates } from 'nuqs' import { ActivityStatus } from '@/components/ui/activity-status' import type { WorkspaceSearchFilters } from '@/lib/api/contracts/knowledge' import { useSession } from '@/lib/auth/auth-client' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import { type ResourceScope, resourceScopeKey } from '@/lib/core/resource-scope' -import { IndexedSearchResults } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed' import { SearchFilters } from '@/app/workspace/[workspaceId]/home/components/knowledge-search-results/search-filters' import { groupResultsByDocument, @@ -48,8 +46,7 @@ export function KnowledgeSearchResults({ const scope: ResourceScope = suppliedScope ?? { kind: 'workspace', workspaceId: workspaceId! } const { data: session } = useSession() const trimmed = query.trim() - const { features } = useDeploymentShape() - const Results = features.liveEnterpriseSearch ? LiveSearchResults : IndexedSearchResults + const Results = LiveSearchResults return ( ({ data: { knowledgeBaseId: 'index' }, isPending: false, })) -kbConnectorsQueriesMockFns.mockUseSearchSourceOverview.mockImplementation(() => ({ - data: { - providers: [ - { connectorType: 'slack', isSyncing: false }, - { connectorType: 'gmail', isSyncing: false }, - ], - }, -})) const mockRequestJson = apiClientRequestMockFns.mockRequestJson @@ -235,48 +226,12 @@ async function complete( }) } -describe('search refinement with the real query cache and URL state', () => { - /** The source refinement chips belong to the indexed search results. */ - beforeEach(() => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - resetDeploymentShape() - }) - afterEach(() => { - resetEnvFlagsMock() - resetDeploymentShape() - }) - - it('does not restore cleared access data as a placeholder', async () => { - await render() - await complete(0) - await act(async () => { - void client.resetQueries({ queryKey: knowledgeKeys.searches() }) - await vi.advanceTimersByTimeAsync(1) - }) - expect(container.textContent).not.toContain('Release plan') - await click('Gmail') - expect(container.textContent).not.toContain('Release plan') - }) - - it('clears displayed placeholder data when access is reset during a refinement', async () => { - await render() - await complete(0) - await click('Gmail') - expect(container.textContent).toContain('Release plan') - await act(async () => { - void client.resetQueries({ queryKey: knowledgeKeys.searches() }) - await vi.advanceTimersByTimeAsync(1) - }) - expect(container.textContent).not.toContain('Release plan') - }) -}) - describe('live search submission feedback', () => { it.each(['launch', 'edited draft', ''])( 'cancels with draft %j, ignores late results, and allows a fresh submission', async (draft) => { const shape = resolveDeploymentShape() - seedDeploymentShape({ ...shape, features: { ...shape.features, liveEnterpriseSearch: true } }) + seedDeploymentShape({ ...shape, features: shape.features }) await render({ organizationPage: true, params: '?q=launch' }) await act(async () => { await vi.advanceTimersByTimeAsync(1) @@ -328,7 +283,7 @@ describe('live search submission feedback', () => { it('acknowledges the submitted query before exposing refinement controls', async () => { const shape = resolveDeploymentShape() - seedDeploymentShape({ ...shape, features: { ...shape.features, liveEnterpriseSearch: true } }) + seedDeploymentShape({ ...shape, features: shape.features }) await render({ organizationPage: true, params: '?q=launch' }) await act(async () => { await vi.advanceTimersByTimeAsync(1) diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/search-integration-connection.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/search-integration-connection.tsx index 8e7e408da0f..29658e390a9 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/search-integration-connection.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/components/message-content/components/special-tags/search-integration-connection.tsx @@ -1,6 +1,5 @@ 'use client' -import { useState } from 'react' import { Chip } from '@sim/emcn' import { Check } from '@sim/emcn/icons' import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' @@ -9,7 +8,6 @@ import { InteractionCard, InteractionCardActionRow, } from '@/app/workspace/[workspaceId]/home/components/message-content/components/interaction-card' -import { SourceSetupModal } from '@/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal' import { BrandIcon } from '@/blocks/brand-icon' import { useSearchIntegrationConnection } from '@/hooks/use-search-integration-connection' @@ -42,7 +40,6 @@ function SearchIntegrationConnectionControl({ divided, onConnected, }: SearchIntegrationConnectionProps) { - const [setupOpen, setSetupOpen] = useState(false) const connector = SEARCH_CONNECTORS.find((entry) => entry.type === target.connectorType) const connection = useSearchIntegrationConnection({ organizationId, @@ -63,17 +60,7 @@ function SearchIntegrationConnectionControl({ : !connection.available ? `${name} connection is no longer available` : `${action} ${name}` - const handleConnect = () => { - if ( - target.connectionMode !== 'live' && - connector && - !connection.connectorId && - connector.setupFields.length && - !connection.pending - ) - setSetupOpen(true) - else void connection.connect() - } + const handleConnect = () => void connection.connect() const content = ( <> )} - {setupOpen && connector && ( - setSetupOpen(false)} - isPending={connection.isStarting} - error={connection.error} - onConnect={(config) => { - void connection.connect(config).then((started) => { - if (started) setSetupOpen(false) - }) - }} - /> - )} ) return embedded ? content : {content} diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/search-sources/atlassian-source-setup-modal.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/search-sources/atlassian-source-setup-modal.tsx deleted file mode 100644 index 92515383b1b..00000000000 --- a/apps/sim/app/workspace/[workspaceId]/home/components/search-sources/atlassian-source-setup-modal.tsx +++ /dev/null @@ -1,254 +0,0 @@ -'use client' - -import { useState } from 'react' -import { - Button, - Chip, - ChipCombobox, - ChipInput, - ChipModal, - ChipModalBody, - ChipModalField, - ChipModalFooter, - ChipModalHeader, - Tooltip, - toast, -} from '@sim/emcn' -import { ArrowLeftRight, Plus } from '@sim/emcn/icons' -import { getErrorMessage } from '@sim/utils/errors' -import type { PersonalSourceSetupQuery } from '@/lib/api/contracts/knowledge/personal-source-setup' -import type { SearchConnector } from '@/lib/sim-search/connectors' -import { MAX_PERSONAL_SOURCE_SETUP_KEYS } from '@/lib/sim-search/personal-source-setup' -import { ConnectorSelectorField } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field' -import { useConnectorConfigFields } from '@/app/workspace/[workspaceId]/knowledge/[id]/hooks/use-connector-config-fields' -import { useConnectPersonalSourceSetup } from '@/hooks/queries/personal-source-setup' -import { usePersonalSourceAccount } from '@/hooks/use-personal-source-account' - -interface AtlassianSourceSetupModalProps { - organizationId: string - connector: SearchConnector - connectorType: PersonalSourceSetupQuery['connectorType'] - onClose: () => void - onConnected?: (connection: { connectorId: string; credentialId: string }) => void -} - -export function AtlassianSourceSetupModal({ - organizationId, - connector, - connectorType, - onClose, - onConnected, -}: AtlassianSourceSetupModalProps) { - const account = usePersonalSourceAccount({ - organizationId, - connectorType, - onConnected: (id) => { - setSelectedAccount(id) - config.setSourceConfig((previous) => ({ domain: previous.domain ?? '' })) - }, - }) - const { mutateAsync: connect, isPending } = useConnectPersonalSourceSetup() - const config = useConnectorConfigFields({ - connectorConfig: connector.meta, - accessMode: 'members', - }) - const [selectedAccount, setSelectedAccount] = useState() - const accounts = account.accounts.data?.accounts ?? [] - const requestedAccount = - selectedAccount ?? - account.accounts.data?.completedCredentialId ?? - (accounts.length === 1 ? accounts[0].id : undefined) - const credentialId = accounts.find((item) => item.id === requestedAccount)?.id ?? null - const canonicalId = connectorType === 'jira' ? 'projectKey' : 'spaceKey' - const picker = connector.meta.configFields.find( - (field) => field.canonicalParamId === canonicalId && field.type === 'selector' - )! - const manual = connector.meta.configFields.find((field) => field.id === canonicalId)! - const advanced = config.canonicalModes[canonicalId] === 'advanced' - const domain = typeof config.sourceConfig.domain === 'string' ? config.sourceConfig.domain : '' - const resolved = config.resolveSourceConfig()[canonicalId] - const manualValue = config.sourceConfig[canonicalId] - const keys = Array.isArray(resolved) - ? resolved - .filter((key): key is string => typeof key === 'string' && Boolean(key.trim())) - .map((key) => key.trim()) - : [] - const pending = isPending || account.pending - const close = () => { - if (!isPending) onClose() - } - const chooseAccount = (id: string) => { - if (id === credentialId) return - setSelectedAccount(id) - config.setSourceConfig({ domain }) - } - const addAccount = () => { - void account.connect() - } - const submit = async () => { - if (!credentialId || !domain.trim() || !keys.length || pending) return - if (keys.length > MAX_PERSONAL_SOURCE_SETUP_KEYS) { - toast.error('Choose no more than 1,000 projects or spaces per source.') - return - } - try { - const result = await connect({ - action: 'connect', - organizationId, - connectorType, - credentialId, - domain: domain.trim(), - keys, - }) - onConnected?.({ connectorId: result.connectorId, credentialId }) - onClose() - } catch (error) { - toast.error(getErrorMessage(error, 'Could not connect the source')) - } - } - - return ( - { - if (!open) close() - }} - srTitle={`Connect ${connector.meta.name}`} - > - Connect {connector.meta.name} - - - {(aria) => ( - <> - ({ value: item.id, label: item.name })), - { - value: '__connect_new__', - label: `Connect ${connector.meta.name} account`, - icon: Plus, - onSelect: addAccount, - }, - ]} - placeholder={ - account.pending - ? 'Waiting for authorization' - : account.accounts.isPending - ? 'Loading accounts' - : 'Select your account' - } - disabled={pending || account.accounts.isPending} - /> - {account.pending && Cancel authorization} - {account.accounts.isError && ( - void account.accounts.refetch()}>Retry loading accounts - )} - - )} - - {credentialId && !account.pending && ( - <> - config.handleFieldChange('domain', value)} - placeholder='yoursite.atlassian.net' - autoComplete='off' - required - disabled={isPending} - /> - - - - - - {advanced ? 'Switch to selector' : 'Switch to manual input'} - - - } - > - {(aria) => ( - <> - {advanced ? ( - - config.handleFieldChange(canonicalId, event.target.value) - } - placeholder={manual.placeholder} - disabled={isPending} - /> - ) : picker.selectorKey ? ( - - config.handleFieldChange(picker.id, value, labels) - } - credentialId={credentialId} - sourceConfig={config.sourceConfig} - configFields={connector.meta.configFields} - canonicalModes={config.canonicalModes} - selectedLabels={config.selectionLabels[canonicalId]} - disabled={isPending} - /> - ) : null} - - )} - - - )} - - - window.open(connector.meta.searchDocsUrl, '_blank', 'noopener,noreferrer'), - }, - ] - : undefined - } - primaryAction={{ - label: isPending ? 'Connecting' : 'Connect & Sync', - onClick: () => void submit(), - disabled: !credentialId || !domain.trim() || !keys.length || pending, - }} - /> - - ) -} diff --git a/apps/sim/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal.tsx b/apps/sim/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal.tsx deleted file mode 100644 index 5570dd8917f..00000000000 --- a/apps/sim/app/workspace/[workspaceId]/home/components/search-sources/source-setup-modal.tsx +++ /dev/null @@ -1,131 +0,0 @@ -'use client' - -import { useState } from 'react' -import { - ChipModal, - ChipModalBody, - ChipModalField, - ChipModalFooter, - ChipModalHeader, -} from '@sim/emcn' -import type { SearchConnector } from '@/lib/sim-search/connectors' -import { AtlassianSourceSetupModal } from '@/app/workspace/[workspaceId]/home/components/search-sources/atlassian-source-setup-modal' - -interface SourceSetupModalProps { - organizationId?: string - onConnected?: (connection: { connectorId: string; credentialId: string }) => void - connector: SearchConnector - onClose: () => void - isPending?: boolean - error?: string | null - /** Connects the source with the filled-in fields; the caller opens the OAuth tab in this click. */ - onConnect: (sourceConfig: Record) => void -} - -/** - * The few fields a source needs before its first connect, such as a site and - * a space. Everyone after the first person clicks straight through. - */ -export function SourceSetupModal(props: SourceSetupModalProps) { - if ( - props.organizationId && - (props.connector.type === 'jira' || props.connector.type === 'confluence') - ) { - return ( - - ) - } - return -} - -function ManualSourceSetupModal({ - connector, - onClose, - onConnect, - isPending = false, - error, -}: SourceSetupModalProps) { - const docsUrl = connector.meta.searchDocsUrl - const fields = connector.setupFields - const [values, setValues] = useState>({}) - const complete = fields.every((field) => values[field.id]?.trim()) - - const submit = () => { - if (!complete || isPending) return - onConnect(Object.fromEntries(fields.map((field) => [field.id, values[field.id]?.trim() ?? '']))) - } - - return ( - { - if (!open) onClose() - }} - srTitle={`Connect ${connector.meta.name}`} - > - Connect {connector.meta.name} - - {fields.map((field) => - field.type === 'dropdown' ? ( - setValues((current) => ({ ...current, [field.id]: value }))} - options={(field.options ?? []).map((option) => ({ - value: option.id, - label: option.label, - }))} - placeholder={field.placeholder} - hint={field.description} - required - /> - ) : ( - setValues((current) => ({ ...current, [field.id]: value }))} - placeholder={field.placeholder} - hint={field.description} - autoComplete='off' - required - /> - ) - )} - {error && ( -

- {error} -

- )} -
- window.open(docsUrl, '_blank', 'noopener,noreferrer'), - }, - ] - : undefined - } - primaryAction={{ - label: isPending ? 'Connecting' : 'Connect', - onClick: submit, - disabled: !complete || isPending, - }} - /> -
- ) -} diff --git a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.dom.test.tsx b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.dom.test.tsx index acd0efa6d15..1781894479c 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.dom.test.tsx +++ b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.dom.test.tsx @@ -511,7 +511,7 @@ describe('useChat remount send recovery', () => { const shape = resolveDeploymentShape() seedDeploymentShape({ ...shape, - features: { ...shape.features, liveEnterpriseSearch: true }, + features: shape.features, }) const history: MothershipChatHistory = { id: 'chat-cited-search', @@ -632,7 +632,7 @@ describe('useChat remount send recovery', () => { it('restores an explicitly selected Search tab while reconnecting an active turn', async () => { const shape = resolveDeploymentShape() - seedDeploymentShape({ ...shape, features: { ...shape.features, liveEnterpriseSearch: true } }) + seedDeploymentShape({ ...shape, features: shape.features }) const history: MothershipChatHistory = { id: 'chat-reconnecting-search', title: 'Search', diff --git a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts index bb40fe65b80..7c6db2b8c06 100644 --- a/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts +++ b/apps/sim/app/workspace/[workspaceId]/home/hooks/use-chat.ts @@ -30,7 +30,6 @@ import type { MothershipTableViewContext } from '@/lib/api/contracts/mothership- import { useSession } from '@/lib/auth/auth-client' import { buildResourceAttachments } from '@/lib/browser-agent/attachments' import { cancelActiveBrowserTools, initBrowserAgentTransport } from '@/lib/browser-agent/transport' -import { getDeploymentShape } from '@/lib/core/config/deployment-shape' import { MothershipHandoffStorage } from '@/lib/core/utils/browser-storage' import { withinDeadline } from '@/lib/core/utils/deadline' import { readSSELines } from '@/lib/core/utils/sse' @@ -1927,7 +1926,6 @@ export function useChat( ) /** Recovery discards interim search tabs without taking an already visible panel away. */ if ( - getDeploymentShape().features.liveEnterpriseSearch && requestModeRef.current === 'assistant' && !sendingRef.current && (!activeStreamId || isTerminalStreamStatus(chatHistory.streamSnapshot?.status)) @@ -1956,7 +1954,6 @@ export function useChat( ? (reorderStoredChatResources(updatedResources, pendingOrder) ?? updatedResources) : updatedResources const keepSearchPanelStable = - getDeploymentShape().features.liveEnterpriseSearch && requestModeRef.current === 'assistant' && (sendingRef.current || (activeStreamId && !isTerminalStreamStatus(chatHistory.streamSnapshot?.status))) @@ -2173,9 +2170,7 @@ export function useChat( } const clearStreamResourceActivity = () => clearResourceActivity(activityTracker, true) const ctx = createStreamLoopContext({ - citedSourcesEnabled: - getDeploymentShape().features.liveEnterpriseSearch && - requestModeRef.current === 'assistant', + citedSourcesEnabled: requestModeRef.current === 'assistant', refreshRoute: () => router.refresh(), viewerId, workspaceId, diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx index 003454c9e36..d119f9d2a3c 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx @@ -18,7 +18,6 @@ import { } from '@sim/emcn' import { ArrowLeft, ChevronDown, ChevronRight, Plus, Search } from '@sim/emcn/icons' import type { ConnectorData } from '@/lib/api/contracts/knowledge/connectors' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import { type ResourceScope, resourceScopeFields } from '@/lib/core/resource-scope' import { asServiceAccountProviderId } from '@/lib/credentials/service-account-provider-ids' import { getIntegrationsForCredentialProvider } from '@/lib/integrations/credential-display' @@ -180,7 +179,7 @@ export function AddConnectorModal({ ) const { mutate: createConnector, isPending: isCreating } = useCreateConnector() - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch && isSearchIndex + const liveSearch = isSearchIndex const canSetUpGitHubInstallation = canAdmin && isSearchIndex && selectedType === 'github' && scope.kind === 'organization' const connectorConfig = liveSearchSourceMeta( diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-documents.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-documents.tsx index 7323a54c0ac..0bebc964de0 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-documents.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-documents.tsx @@ -3,7 +3,6 @@ import { Chip, ChipInput, ChipLink, Skeleton } from '@sim/emcn' import { RefreshCw, Search, SquareArrowUpRight } from '@sim/emcn/icons' import type { ConnectorDocumentFilter } from '@/lib/api/contracts/knowledge/connectors' -import type { ResourceScope } from '@/lib/core/resource-scope' import { getDocumentIndexingStatus } from '@/lib/knowledge/documents/types' import { ConnectorDocumentStatusFilter } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/connector-documents/connector-document-status-filter' import { @@ -27,9 +26,6 @@ interface ConnectorDocumentsProps { search?: string searchControl?: { value: string; onChange: (value: string) => void } showToolbar?: boolean - progressScope?: ResourceScope - isSearchIndex?: boolean - syncing?: boolean filter: ConnectorDocumentFilter onFilterChange: (filter: ConnectorDocumentFilter) => void } @@ -41,16 +37,11 @@ export function ConnectorDocuments({ search, searchControl, showToolbar = true, - progressScope, - isSearchIndex = false, - syncing, onFilterChange, }: ConnectorDocumentsProps) { const query = useConnectorDocuments(knowledgeBaseId, connectorId, { filter, search, - progressScope, - syncing, }) const { data, hasNextPage, isFetchingNextPage, fetchNextPage } = query const isLoading = query.isLoading || query.isPlaceholderData @@ -79,9 +70,6 @@ export function ConnectorDocuments({ return ( <>
- {isSearchIndex && ( -

Documents you can access

- )} {showToolbar && (
{searchControl && ( diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.test.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.test.tsx index ed1f9c3e86a..5463e820c43 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.test.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.test.tsx @@ -93,54 +93,3 @@ it.each([ } } ) - -it('keeps the prior selection when all personal setup options exceed its source limit', async () => { - mocks.loadAll.mockResolvedValue({ - status: 'complete', - options: Array.from({ length: 1001 }, (_, index) => ({ - id: `P${index}`, - label: `Project ${index}`, - })), - }) - const field: ConnectorConfigField & { selectorKey: 'jira.projectKeys' } = { - id: 'projects', - title: 'Projects', - type: 'selector', - selectorKey: 'jira.projectKeys', - multi: true, - allowSelectAll: true, - } - const container = document.createElement('div') - const root = createRoot(container) - try { - await act(async () => - root.render( - - ) - ) - const all = mocks.combobox.mock.lastCall![0].options.find((item) => item.label === 'All') - expect(container.textContent).not.toContain('Select all') - await act(async () => all?.onSelect?.()) - expect(mocks.loadAll).toHaveBeenCalledTimes(1) - expect(mocks.change).not.toHaveBeenCalled() - expect(container.querySelector('[role="alert"]')?.textContent).toContain( - 'Select up to 1,000 items' - ) - } finally { - await act(async () => root.unmount()) - } -}) diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.tsx index eae49ed34e5..1fd83e71c37 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/connector-selector-field/connector-selector-field.tsx @@ -7,8 +7,7 @@ import { useParams } from 'next/navigation' import { type ResourceScope, resourceScopeFromOwner } from '@/lib/core/resource-scope' import { projectSelectorContext } from '@/lib/selectors/context' import { getSelectorManifestEntry, type SelectorKey } from '@/lib/selectors/manifest' -import type { SelectorContext, SelectorSurface } from '@/lib/selectors/types' -import { MAX_PERSONAL_SOURCE_SETUP_KEYS } from '@/lib/sim-search/personal-source-setup' +import type { SelectorContext } from '@/lib/selectors/types' import type { SourceSelectionLabel } from '@/lib/sim-search/source-identity' import { SEARCH_DEBOUNCE_MS } from '@/lib/url-state' import { getDependsOnFields } from '@/lib/workflows/subblocks/dependencies' @@ -27,7 +26,6 @@ import { useDebounce } from '@/hooks/use-debounce' interface ConnectorSelectorFieldProps { controlAria?: ChipModalFieldAria scope?: ResourceScope - selectorSurface?: SelectorSurface field: ConnectorConfigField & { selectorKey: SelectorKey } value: ConfigFieldValue onChange: (value: ConfigFieldValue, selectedOptions?: SourceSelectionLabel[]) => void @@ -43,7 +41,6 @@ interface ConnectorSelectorFieldProps { export function ConnectorSelectorField({ controlAria, scope: explicitScope, - selectorSurface, field, value, onChange, @@ -112,9 +109,7 @@ export function ConnectorSelectorField({ }, [field.dependsOn, sourceConfig, configFields, canonicalModes]) const isEnabled = !disabled && !!credentialId && depsResolved - const missingDependencyMessage = selectorSurface - ? 'Enter your Atlassian site first' - : `Select ${getDependencyLabel(field, configFields)} first` + const missingDependencyMessage = `Select ${getDependencyLabel(field, configFields)} first` const debouncedSearch = useDebounce(searchTerm.trim(), SEARCH_DEBOUNCE_MS) const { data: options = [], @@ -131,7 +126,6 @@ export function ConnectorSelectorField({ } = useSelectorOptions(field.selectorKey, { context, scope, - surface: selectorSurface, search: debouncedSearch, enabled: isEnabled, surfaceId: `connector:${field.id}`, @@ -155,7 +149,6 @@ export function ConnectorSelectorField({ { context, scope, - surface: selectorSurface, detailIds: isEnabled ? missingSelectedIds : [], surfaceId: `connector:${field.id}`, } @@ -171,7 +164,6 @@ export function ConnectorSelectorField({ const { data: searchedOption } = useSelectorOptionDetail(field.selectorKey, { context, scope, - surface: selectorSurface, detailId: resolvesUnknownIds && isEnabled && debouncedSearch.length > 0 ? debouncedSearch : undefined, surfaceId: `connector:${field.id}`, @@ -254,16 +246,6 @@ export function ConnectorSelectorField({ }) return } - if ( - selectorSurface?.kind === 'personal-search-setup' && - result.options.length > MAX_PERSONAL_SOURCE_SETUP_KEYS - ) { - setBulkError({ - context, - message: `Select up to ${MAX_PERSONAL_SOURCE_SETUP_KEYS.toLocaleString()} items. Choose a smaller set to continue.`, - }) - return - } onChange( result.options.map((option) => option.id), result.options.map((option) => ({ id: option.id, label: option.label })) diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/connector-settings-fields.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/connector-settings-fields.tsx index 3e07a27b34c..0142cfa7513 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/connector-settings-fields.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/connector-settings-fields.tsx @@ -11,7 +11,6 @@ import { } from '@sim/emcn' import { ChevronDown, ChevronRight, Plus } from '@sim/emcn/icons' import type { ConnectorAccessMode } from '@/lib/api/contracts/knowledge/connectors' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import { type ResourceScope, resourceScopeFields } from '@/lib/core/resource-scope' import { asServiceAccountProviderId } from '@/lib/credentials/service-account-provider-ids' import { @@ -159,7 +158,7 @@ export function ConnectorSettingsFields({ onContentCredentialChange, onWorkspaceCredentialChange, }: ConnectorSettingsFieldsProps) { - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch && isSearchIndex + const liveSearch = isSearchIndex const connectorConfig = liveSearchSourceMeta(originalConfig, Boolean(liveSearch), { githubInstallation: usesGitHubInstallation && scope.kind === 'organization', }) diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.test.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.test.tsx index 4be5a4f9a53..0113ffce7b4 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.test.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.test.tsx @@ -2,11 +2,7 @@ * @vitest-environment jsdom */ import { act } from 'react' -import { - createMockDeploymentShape, - deploymentShapeMock, - deploymentShapeMockFns, -} from '@sim/testing/mocks/deployment-shape.mock' +import { deploymentShapeMock } from '@sim/testing/mocks/deployment-shape.mock' import { kbConnectorsQueriesMock, kbConnectorsQueriesMockFns, @@ -16,7 +12,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { ConnectorData } from '@/lib/api/contracts/knowledge/connectors' const mocks = vi.hoisted(() => ({ - live: false, update: vi.fn(), applyAccess: vi.fn(), settingsPending: false, @@ -54,9 +49,6 @@ vi.mock('@/hooks/use-permission-config', () => ({ import { useConnectorSettingsForm } from '@/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form' -deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(() => - createMockDeploymentShape({ features: { liveEnterpriseSearch: mocks.live } }) -) kbConnectorsQueriesMockFns.mockUseUpdateConnector.mockImplementation(() => ({ mutate: mocks.update, isPending: mocks.settingsPending, @@ -120,7 +112,6 @@ describe('shared connector settings form', () => { } beforeEach(() => { - mocks.live = false mocks.settingsPending = false mocks.accessPending = false ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true @@ -209,7 +200,7 @@ describe('shared connector settings form', () => { }) it('preserves the GitHub repository and pending connection after an incompatible replacement is refused', () => { - const sourceConfig = { repository: 'acme/platform', branch: 'main' } + const sourceConfig = { repository: 'acme/platform' } render( connector({ connectorType: 'github', diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.ts b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.ts index 32696d582b4..300366d0e7a 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.ts +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/edit-connector-modal/use-connector-settings-form.ts @@ -4,7 +4,6 @@ import { useCallback, useMemo, useState } from 'react' import { createLogger } from '@sim/logger' import { isEqual } from 'es-toolkit' import type { UpdateConnectorBody } from '@/lib/api/contracts/knowledge/connectors' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import type { ResourceScope } from '@/lib/core/resource-scope' import { isContentEngineAccessMode } from '@/lib/knowledge/connectors/access-modes' import { getConnectorAccessAvailability } from '@/lib/sim-search/connectors' @@ -137,7 +136,7 @@ export function useConnectorSettingsForm({ onSaved, syncing = isConnectorSyncingOrPending(connector), }: UseConnectorSettingsFormOptions) { - const liveSearch = useDeploymentShape().features.liveEnterpriseSearch && isSearchIndex + const liveSearch = isSearchIndex const connectorConfig = liveSearchSourceMeta( CONNECTOR_META_REGISTRY[connector.connectorType] ?? null, Boolean(liveSearch), diff --git a/apps/sim/bootstrap.ts b/apps/sim/bootstrap.ts index 80574b20329..e5e08c65b52 100644 --- a/apps/sim/bootstrap.ts +++ b/apps/sim/bootstrap.ts @@ -9,8 +9,6 @@ await loadRuntimeSecrets() // Explicit runtime configuration wins over deployment defaults. process.env.MSHIP_PLAN_MODE ??= process.env.MSHIP_PLAN_MODE_DEFAULT ?? (process.env.NODE_ENV === 'development' ? 'true' : 'false') -process.env.SIM_SEARCH_LIVE ??= process.env.SIM_SEARCH_LIVE_DEFAULT ?? 'true' -process.env.NEXT_PUBLIC_SIM_SEARCH_LIVE = process.env.SIM_SEARCH_LIVE // `server.js` is the Next standalone build artifact, a sibling of this file in // the image; it does not exist at type-check time, so the specifier is held in a // variable to keep it out of static module resolution. diff --git a/apps/sim/connectors/coda/README.md b/apps/sim/connectors/coda/README.md index b0f2d04d92d..adf4015a5ee 100644 --- a/apps/sim/connectors/coda/README.md +++ b/apps/sim/connectors/coda/README.md @@ -1,6 +1,6 @@ # Coda indexed connector decisions and verification -This document covers ordinary knowledge-base ingestion and the explicitly selected legacy Search backend (`SIM_SEARCH_LIVE=false`). Its historical verification notes refer to that indexed path. Default live Search uses personal Coda MCP authorization plus optional service-source verification; see [live Search](../../lib/sim-search/live/README.md) and the [Coda Search guide](../../../docs/content/docs/search/coda.mdx). Background Coda content/ACL/directory builds are not part of live Search. +This document covers ordinary knowledge-base ingestion. Historical organization-indexing verification notes describe the retired indexed Search path. Default live Search uses personal Coda MCP authorization plus optional service-source verification; see [live Search](../../lib/sim-search/live/README.md) and the [Coda Search guide](../../../docs/content/docs/search/coda.mdx). Background Coda content/ACL/directory builds are not part of live Search. ## Precedent and authentication diff --git a/apps/sim/ee/organization-search-stats/search-params.ts b/apps/sim/ee/organization-search-stats/search-params.ts deleted file mode 100644 index 0514a128245..00000000000 --- a/apps/sim/ee/organization-search-stats/search-params.ts +++ /dev/null @@ -1,22 +0,0 @@ -import { parseAsString, parseAsStringLiteral } from 'nuqs/server' -import { SEARCH_STATS_PERIODS, SEARCH_STATS_SURFACES } from '@/lib/knowledge/search/stats' - -export const organizationSearchStatsParsers = { - period: parseAsStringLiteral(SEARCH_STATS_PERIODS).withDefault('30d'), - /** An absent surface includes every Search entry point. */ - surface: parseAsStringLiteral(SEARCH_STATS_SURFACES), - startDate: parseAsString, - endDate: parseAsString, -} - -export const organizationSearchStatsUrlOptions = { - history: 'replace', - shallow: true, - clearOnDefault: true, - urlKeys: { - period: 'stats-period', - surface: 'stats-surface', - startDate: 'stats-start', - endDate: 'stats-end', - }, -} as const diff --git a/apps/sim/ee/workspace-forking/lib/copy/copy-resources.test.ts b/apps/sim/ee/workspace-forking/lib/copy/copy-resources.test.ts index 37b3cf2df7b..248dccc9cd9 100644 --- a/apps/sim/ee/workspace-forking/lib/copy/copy-resources.test.ts +++ b/apps/sim/ee/workspace-forking/lib/copy/copy-resources.test.ts @@ -3,6 +3,7 @@ import { sha256Hex } from '@sim/security/hash' import { dbChainMockFns, flattenMockConditions, + queueTableRows, resetDbChainMock, schemaMock, storageServiceMock, @@ -105,6 +106,12 @@ function mappedDocumentPlan(): ForkContentPlan { describe('copyForkResourceContent', () => { beforeEach(() => { resetDbChainMock() + for (let attempt = 0; attempt < 4; attempt++) + queueTableRows(schemaMock.knowledgeBase, [ + { id: 'src-kb', isSearchIndex: false }, + { id: 'child-kb', isSearchIndex: false }, + { id: 'existing-target-kb', isSearchIndex: false }, + ]) dbChainMockFns.returning.mockResolvedValue([{ id: 'activated-document' }]) dbChainMockFns.for.mockResolvedValue([{ workspaceId: 'child-ws' }]) storageServiceMockFns.mockHeadObject.mockResolvedValue(null) @@ -1138,11 +1145,15 @@ describe('planForkMappedKbDocumentCopies', () => { let selectCalls = 0 const tx = { select: () => { - const rows = selectCalls++ === 0 ? docs : existingTargets return { - from: () => ({ + from: (table: unknown) => ({ where: (condition: unknown) => { + if (table === schemaMock.knowledgeBase) { + const rows = Promise.resolve([{ id: 'target-kb' }]) + return Object.assign(rows, { for: () => rows }) + } wheres.push(condition) + const rows = selectCalls++ === 0 ? docs : existingTargets return Promise.resolve(rows) }, }), diff --git a/apps/sim/ee/workspace-forking/lib/copy/copy-resources.ts b/apps/sim/ee/workspace-forking/lib/copy/copy-resources.ts index c5cd463b170..57e0772d7d4 100644 --- a/apps/sim/ee/workspace-forking/lib/copy/copy-resources.ts +++ b/apps/sim/ee/workspace-forking/lib/copy/copy-resources.ts @@ -48,6 +48,10 @@ import { hashDurableSecretProvenanceValue, } from '@/lib/execution/durable-secret-provenance' import { WORKSPACE_ACCESS_TOKEN } from '@/lib/knowledge/access/types' +import { + connectorIndexingCondition, + requiresConnectorIndexing, +} from '@/lib/knowledge/connectors/indexing-policy' import { createKnowledgeDocumentSourceValue, type KnowledgeDocumentSourceValue, @@ -758,6 +762,7 @@ export async function copyForkResourceContainers( and( inArray(knowledgeBase.id, selection.knowledgeBases), eq(knowledgeBase.workspaceId, sourceWorkspaceId), + connectorIndexingCondition(), isNull(knowledgeBase.deletedAt) ) ) @@ -951,17 +956,43 @@ export async function planForkMappedKbDocumentCopies(params: { .where( and( inArray(document.id, candidateIds), + exists( + tx + .select({ id: knowledgeBase.id }) + .from(knowledgeBase) + .where( + and( + eq(knowledgeBase.id, document.knowledgeBaseId), + connectorIndexingCondition(), + isNull(knowledgeBase.deletedAt) + ) + ) + ), isNull(document.connectorId), isNull(document.deletedAt), isNull(document.archivedAt) ) ) - const planned = docs.flatMap((doc) => { + const candidates = docs.flatMap((doc) => { const targetKbId = resolver('knowledge-base', doc.knowledgeBaseId) if (targetKbId == null) return [] return [{ doc, targetKbId, childDocId: deriveCopyIdentity('document', targetKbId, doc.id) }] }) + if (candidates.length === 0) return { documents, docIdMap, mappingEntries } + const targets = await tx + .select({ id: knowledgeBase.id }) + .from(knowledgeBase) + .where( + and( + inArray(knowledgeBase.id, [...new Set(candidates.map(({ targetKbId }) => targetKbId))]), + connectorIndexingCondition(), + isNull(knowledgeBase.deletedAt) + ) + ) + .for('share') + const targetIds = new Set(targets.map(({ id }) => id)) + const planned = candidates.filter(({ targetKbId }) => targetIds.has(targetKbId)) const existingTargets = planned.length === 0 ? [] @@ -1690,7 +1721,10 @@ async function finalizeKbDocument(params: { } = params return db.transaction(async (tx) => { const [lockedKnowledgeBase] = await tx - .select({ workspaceId: knowledgeBase.workspaceId }) + .select({ + workspaceId: knowledgeBase.workspaceId, + isSearchIndex: knowledgeBase.isSearchIndex, + }) .from(knowledgeBase) .where(eq(knowledgeBase.id, childKnowledgeBaseId)) .for('update') @@ -1698,6 +1732,9 @@ async function finalizeKbDocument(params: { if (!lockedKnowledgeBase) { throw new Error(`Copied document knowledge base ${childKnowledgeBaseId} is missing`) } + if (!requiresConnectorIndexing(lockedKnowledgeBase.isSearchIndex)) { + throw new Error('Retired Search documents cannot be activated by a workspace copy') + } if (lockedKnowledgeBase.workspaceId !== billingContext.workspaceId) { throw new Error( `Copied document knowledge base ${childKnowledgeBaseId} moved from workspace ${billingContext.workspaceId}; refusing stale storage charge` @@ -1809,6 +1846,24 @@ async function copyKbDocument(params: { billingContext, } = params assertForkCopyActive(params.control) + const bases = await db + .select({ id: knowledgeBase.id, isSearchIndex: knowledgeBase.isSearchIndex }) + .from(knowledgeBase) + .where( + and( + inArray(knowledgeBase.id, [source.knowledgeBaseId, childKnowledgeBaseId]), + isNull(knowledgeBase.deletedAt) + ) + ) + const basesById = new Map(bases.map((base) => [base.id, base])) + if ( + [source.knowledgeBaseId, childKnowledgeBaseId].some((id) => { + const base = basesById.get(id) + return !base || !requiresConnectorIndexing(base.isSearchIndex) + }) + ) { + throw new Error('Workspace copies require active ordinary knowledge bases') + } const sourceSecretContext = await loadKnowledgeDocumentDurableSecretProvenance(source.id) const sourceSnapshotHash = hashDurableSecretProvenanceValue( createKnowledgeDocumentSourceValue(source) diff --git a/apps/sim/executor/utils/credential-token.ts b/apps/sim/executor/utils/credential-token.ts index 074e5361ca1..78e7e66cc33 100644 --- a/apps/sim/executor/utils/credential-token.ts +++ b/apps/sim/executor/utils/credential-token.ts @@ -1,6 +1,5 @@ import { createLogger } from '@sim/logger' import { AuthType } from '@/lib/auth/hybrid' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { createCopilotManagedOAuthPrincipal } from '@/lib/credentials/application/copilot-managed-oauth-delegation' import { bindExecutorManagedOAuthDelegation } from '@/lib/credentials/application/managed-oauth-delegation' import { authorizePersonalCredential } from '@/lib/credentials/application/personal-credentials' @@ -64,9 +63,7 @@ export async function resolveExecutorCredentialToken( throw new Error('Assistant credential use requires the authenticated person for this turn.') } const original = toolId ? getToolMetadata(toolId) : undefined - const tool = original - ? projectAssistantConnectedAccountTool(original, isLiveEnterpriseSearchEnabled) - : undefined + const tool = original ? projectAssistantConnectedAccountTool(original) : undefined if ( !tool?.oauth?.required || (!copilotExecutionContext.workspaceId && !copilotExecutionContext.organizationId) || diff --git a/apps/sim/hooks/queries/kb/connectors.test.ts b/apps/sim/hooks/queries/kb/connectors.test.ts index e6bfc7d9c97..054e6388420 100644 --- a/apps/sim/hooks/queries/kb/connectors.test.ts +++ b/apps/sim/hooks/queries/kb/connectors.test.ts @@ -287,7 +287,7 @@ describe('useSearchSources', () => { expect(sources.queryKey).not.toEqual(searchSourceKeys.list('scope-1')) await sources.queryFn({ signal }) expect(mockRequestJson).toHaveBeenLastCalledWith(listSearchSourcesContract, { - query: { organizationId: 'scope-1', search: '', mine: false }, + query: { organizationId: 'scope-1', search: '' }, signal, }) }) diff --git a/apps/sim/hooks/queries/kb/connectors.ts b/apps/sim/hooks/queries/kb/connectors.ts index 6347604ac6c..6674d38ff01 100644 --- a/apps/sim/hooks/queries/kb/connectors.ts +++ b/apps/sim/hooks/queries/kb/connectors.ts @@ -1,4 +1,3 @@ -import { useEffect } from 'react' import { keepPreviousData, type QueryClient, @@ -13,8 +12,6 @@ import { type ConnectorDetailData, type ConnectorDocumentsData, type ConnectorMemberSummary, - type ConnectSimSearchConnectorBody, - connectSimSearchConnectorContract, createKnowledgeConnectorContract, deleteKnowledgeConnectorContract, getKnowledgeConnectorContract, @@ -41,13 +38,9 @@ import { type ConnectorDocumentsQuery, type PrepareSearchSourceBody, prepareSearchSourceContract, - readOrganizationSearchOverviewContract, readSearchIndexContract, - readSearchSourceOverviewContract, - readSearchSourceProgressContract, type SearchConnectionOAuthQuery, type SearchSourcePage, - type SearchSourceProgress, } from '@/lib/api/contracts/knowledge/connectors' import { type ResourceScope, @@ -55,10 +48,7 @@ import { resourceScopeFromOwner, resourceScopeKey, } from '@/lib/core/resource-scope' -import { - MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_PAGE_SIZE, - MAX_SEARCH_SOURCE_PROGRESS_ITEMS, -} from '@/lib/knowledge/constants' +import { MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_PAGE_SIZE } from '@/lib/knowledge/constants' import { organizationAccountsKeys } from '@/hooks/queries/organization-accounts' import { credentialGroupKeys } from '@/hooks/queries/utils/credential-group-queries' import { knowledgeKeys } from '@/hooks/queries/utils/knowledge-keys' @@ -91,13 +81,6 @@ export const connectorKeys = { details: (knowledgeBaseId?: string) => [...connectorKeys.all(knowledgeBaseId), 'detail'] as const, detail: (knowledgeBaseId?: string, connectorId?: string) => [...connectorKeys.details(knowledgeBaseId), connectorId ?? ''] as const, - progress: (knowledgeBaseId?: string, connectorId?: string, scope?: ResourceScope) => - [ - ...connectorKeys.progresses(knowledgeBaseId, connectorId), - scope ? resourceScopeKey(scope) : '', - ] as const, - progresses: (knowledgeBaseId?: string, connectorId?: string) => - [...connectorKeys.detail(knowledgeBaseId, connectorId), 'progress'] as const, } async function fetchConnectors( @@ -437,80 +420,30 @@ export function useSearchIndex(scope: ResourceScope, options?: { enabled?: boole }) } -/** Full viewer counts update less often than bounded progress probes. */ -const SEARCH_SOURCE_SUMMARY_POLL_MS = 30_000 - -export function useSearchSourceOverview(scope: ResourceScope, options?: { enabled?: boolean }) { - return useQuery({ - queryKey: searchSourceKeys.overview(scope), - queryFn: async ({ signal }) => - ( - await requestJson(readSearchSourceOverviewContract, { - query: resourceScopeFields(scope), - signal, - }) - ).data, - enabled: options?.enabled ?? true, - staleTime: CONNECTOR_LIST_STALE_TIME, - refetchInterval: (query) => - query.state.data?.providers.some((provider) => provider.isSyncing) - ? SEARCH_SOURCE_SUMMARY_POLL_MS - : false, - }) -} - -/** Administrative health is independent of the viewer's account and document permissions. */ -export function useOrganizationSearchOverview( - organizationId: string, - options?: { enabled?: boolean } -) { - return useQuery({ - queryKey: searchSourceKeys.organizationOverview(organizationId), - queryFn: async ({ signal }) => - ( - await requestJson(readOrganizationSearchOverviewContract, { - query: { organizationId }, - signal, - }) - ).data, - enabled: Boolean(organizationId) && (options?.enabled ?? true), - staleTime: CONNECTOR_LIST_STALE_TIME, - refetchInterval: (query) => - query.state.data?.providers.some((provider) => provider.isSyncing || provider.hasPendingSync) - ? SEARCH_SOURCE_SUMMARY_POLL_MS - : false, - }) -} - export function useSearchSources( owner?: string | ResourceScope, options?: { enabled?: boolean search?: string - mine?: boolean connectorType?: string excludeConnectorType?: string } ) { - const queryClient = useQueryClient() const scope = typeof owner === 'string' ? owner ? { kind: 'workspace' as const, workspaceId: owner } : undefined : owner - const workspaceId = scope?.kind === 'workspace' ? scope.workspaceId : undefined - const organizationId = scope?.kind === 'organization' ? scope.organizationId : undefined const enabled = Boolean(scope) && (options?.enabled ?? true) const filters = { search: options?.search?.trim().toLowerCase() ?? '', - mine: options?.mine ?? false, ...(options?.connectorType?.trim() ? { connectorType: options.connectorType.trim() } : {}), ...(options?.excludeConnectorType?.trim() ? { excludeConnectorType: options.excludeConnectorType.trim() } : {}), } - const summary = useInfiniteQuery({ + return useInfiniteQuery({ queryKey: searchSourceKeys.pages(scope, filters), initialPageParam: null as string | null, getNextPageParam: (page: SearchSourcePage) => page.nextCursor, @@ -528,73 +461,7 @@ export function useSearchSources( ).data, enabled, staleTime: CONNECTOR_LIST_STALE_TIME, - refetchInterval: (query) => - query.state.data?.pages.some((page) => page.sources.some((source) => source.isSyncing)) - ? SEARCH_SOURCE_SUMMARY_POLL_MS - : false, }) - const activeIds = (summary.data ?? []) - .filter((source) => source.isSyncing) - .map((source) => source.connectorId) - .sort() - const progress = useQuery({ - queryKey: searchSourceKeys.progress(scope, activeIds), - queryFn: async ({ signal }) => { - if (!scope) throw new Error('A Search source scope is required') - const sources: SearchSourceProgress[] = [] - for (let offset = 0; offset < activeIds.length; offset += MAX_SEARCH_SOURCE_PROGRESS_ITEMS) { - const response = await requestJson(readSearchSourceProgressContract, { - body: { - ...resourceScopeFields(scope), - connectorIds: activeIds.slice(offset, offset + MAX_SEARCH_SOURCE_PROGRESS_ITEMS), - }, - signal, - }) - sources.push(...response.data) - } - return sources - }, - enabled: enabled && activeIds.length > 0, - staleTime: CONNECTOR_SYNC_POLL_INTERVAL_MS, - refetchInterval: (query) => - query.state.dataUpdateCount < 20 ? CONNECTOR_SYNC_POLL_INTERVAL_MS : 15_000, - }) - - /** Reconcile exact viewer counts when the cheaper probe observes a state transition. */ - useEffect(() => { - if (!enabled || !progress.data || progress.dataUpdatedAt <= summary.dataUpdatedAt) return - const states = new Map(progress.data.map((source) => [source.connectorId, source])) - const changed = summary.data?.some((source) => { - if (!source.isSyncing) return false - const state = states.get(source.connectorId) - return ( - !state || - state.isSyncing !== source.isSyncing || - state.hasSyncError !== source.hasSyncError || - state.hasIndexingError !== source.viewerFailedDocumentCount > 0 - ) - }) - if (changed) { - const owner = organizationId - ? { kind: 'organization' as const, organizationId } - : { kind: 'workspace' as const, workspaceId: workspaceId! } - queryClient.invalidateQueries( - { queryKey: searchSourceKeys.list(owner) }, - { cancelRefetch: false } - ) - } - }, [ - enabled, - progress.data, - progress.dataUpdatedAt, - summary.data, - summary.dataUpdatedAt, - queryClient, - workspaceId, - organizationId, - ]) - - return summary } /** Mints the viewer's enrollment link for a per-member connector; the caller navigates to it. */ @@ -751,11 +618,7 @@ export function useTriggerSync() { queryClient.invalidateQueries({ queryKey: connectorKeys.all(knowledgeBaseId) }) } }, - onSuccess: (_data, { knowledgeBaseId, connectorId }) => { - queryClient.invalidateQueries({ - queryKey: connectorKeys.progresses(knowledgeBaseId, connectorId), - }) - }, + /** An early poll can read idle before dispatch marks pending; reconcile after the request settles. */ onSettled: (_data, error, { knowledgeBaseId, connectorId }) => Promise.all([ @@ -807,9 +670,8 @@ async function fetchConnectorDocuments( export function useConnectorDocuments( knowledgeBaseId?: string, connectorId?: string, - options?: ConnectorDocumentListOptions & { progressScope?: ResourceScope; syncing?: boolean } + options?: ConnectorDocumentListOptions ) { - const queryClient = useQueryClient() const query = { includeExcluded: options?.filter ? undefined : (options?.includeExcluded ?? false), failedOnly: options?.filter ? undefined : (options?.failedOnly ?? false), @@ -838,54 +700,6 @@ export function useConnectorDocuments( staleTime: CONNECTOR_DOCUMENT_LIST_STALE_TIME, placeholderData: keepPreviousData, }) - const scope = options?.progressScope - const hasProgressScope = Boolean(scope) - const syncing = options?.syncing ?? false - const progress = useQuery({ - queryKey: connectorKeys.progress(knowledgeBaseId, connectorId, scope), - queryFn: async ({ signal }) => { - if (!scope || !connectorId) - throw new Error('A Search source scope and connector are required') - return ( - await requestJson(readSearchSourceProgressContract, { - body: { ...resourceScopeFields(scope), connectorIds: [connectorId] }, - signal, - }) - ).data - }, - enabled: Boolean(scope && knowledgeBaseId && connectorId), - staleTime: CONNECTOR_SYNC_POLL_INTERVAL_MS, - refetchInterval: (query) => - syncing || query.state.data?.some((source) => source.isSyncing) - ? query.state.dataUpdateCount < 20 - ? CONNECTOR_SYNC_POLL_INTERVAL_MS - : 15_000 - : false, - }) - - /** Refresh document pages once indexing settles, rather than polling every loaded page. */ - useEffect(() => { - if ( - !hasProgressScope || - syncing || - !progress.data || - progress.data.some((source) => source.isSyncing) || - progress.dataUpdatedAt <= documents.dataUpdatedAt - ) - return - void queryClient.invalidateQueries({ - queryKey: connectorDocumentKeys.lists(knowledgeBaseId, connectorId), - }) - }, [ - hasProgressScope, - syncing, - progress.data, - progress.dataUpdatedAt, - documents.dataUpdatedAt, - queryClient, - knowledgeBaseId, - connectorId, - ]) return documents } @@ -976,34 +790,6 @@ export function useRestoreConnectorDocument() { }) } -async function connectSimSearchConnector(body: ConnectSimSearchConnectorBody) { - const result = await requestJson(connectSimSearchConnectorContract, { body }) - return result.data -} - -/** - * One click on a Sim Search source: the source's per-member connector exists - * afterwards and the viewer has their enrollment link. The member list and the - * base list both gain a row on a first connect. - */ -export function useConnectSimSearchConnector() { - const queryClient = useQueryClient() - return useMutation({ - mutationFn: connectSimSearchConnector, - onSuccess: (data) => { - /** A first connect added a connector to the base; its own list is open on the settings page. */ - queryClient.invalidateQueries({ queryKey: connectorKeys.all(data.knowledgeBaseId) }) - }, - onSettled: () => { - queryClient.invalidateQueries({ queryKey: searchIndexKeys.details() }) - queryClient.invalidateQueries({ queryKey: searchSourceKeys.lists() }) - queryClient.invalidateQueries({ queryKey: searchIntegrationKeys.lists() }) - queryClient.invalidateQueries({ queryKey: knowledgeKeys.lists() }) - void invalidateConnectorAccounts(queryClient) - }, - }) -} - export function usePrepareSearchSource() { const queryClient = useQueryClient() return useMutation({ diff --git a/apps/sim/hooks/queries/kb/knowledge.test.ts b/apps/sim/hooks/queries/kb/knowledge.test.ts index b14daea048f..a3d9329521e 100644 --- a/apps/sim/hooks/queries/kb/knowledge.test.ts +++ b/apps/sim/hooks/queries/kb/knowledge.test.ts @@ -1,10 +1,6 @@ import { apiClientRequestMock } from '@sim/testing/mocks/api-client-request.mock' import { authClientMock, authClientMockFns } from '@sim/testing/mocks/auth-client.mock' -import { - createMockDeploymentShape, - deploymentShapeMock, - deploymentShapeMockFns, -} from '@sim/testing/mocks/deployment-shape.mock' +import { deploymentShapeMock } from '@sim/testing/mocks/deployment-shape.mock' import { emcnMock } from '@sim/testing/mocks/emcn.mock' import { reactQueryMock, reactQueryMockFns } from '@sim/testing/mocks/react-query.mock' import { beforeEach, describe, expect, it, vi } from 'vitest' @@ -36,9 +32,6 @@ const mocks = { getQueryData: reactQueryMockFns.mockQueryClient.getQueryData, } authClientMockFns.mockUseSession.mockReturnValue({ data: { user: { id: 'reader' } } }) -deploymentShapeMockFns.mockUseDeploymentShape.mockImplementation(() => - createMockDeploymentShape({ features: { liveEnterpriseSearch: mocks.live } }) -) interface CapturedQuery { queryKey: readonly unknown[] @@ -101,49 +94,13 @@ describe('knowledge query placeholder scope', () => { ).toBeUndefined() }) - it('retains successful refinements only for the same reader, query, scope, and result limit', () => { - const query = captureQuery(() => - useWorkspaceKnowledgeSearch('workspace-1', 'release', { source: 'slack' }, 5) - ) - const previous = { results: [{ documentId: 'private-document' }] } - mocks.getQueryData.mockReturnValue(previous) - const placeholder = (scope: string, text: string, topK: number, userId: string) => - query.placeholderData?.(previous, { - queryKey: knowledgeKeys.search(scope, text, {}, topK, userId), - state: { status: 'success', isInvalidated: false }, - }) - expect(placeholder('workspace-1', 'release', 5, 'reader')).toBe(previous) - expect(placeholder('workspace-1', 'release', 20, 'reader')).toBeUndefined() - expect(placeholder('workspace-1', 'release', 5, 'other')).toBeUndefined() - expect(placeholder('workspace-2', 'release', 5, 'reader')).toBeUndefined() - expect(placeholder('workspace-1', 'different', 5, 'reader')).toBeUndefined() - }) - - it('retains a page-owned search across result limits only for the same reader', () => { - const query = captureQuery(() => - useWorkspaceKnowledgeSearch('workspace-1', 'release', { source: 'slack' }, 50, { - retainAcrossLimits: true, - }) - ) - const previous = { results: [{ documentId: 'private-document' }] } - mocks.getQueryData.mockReturnValue(previous) - const placeholder = (topK: number, userId: string) => - query.placeholderData?.(previous, { - queryKey: knowledgeKeys.search('workspace-1', 'release', {}, topK, userId), - state: { status: 'success', isInvalidated: false }, - }) - expect(placeholder(20, 'reader')).toBe(previous) - expect(placeholder(50, 'reader')).toBe(previous) - expect(placeholder(20, 'other')).toBeUndefined() - }) - it('partitions search cache entries by filter and reader', () => { const query = captureQuery(() => useWorkspaceKnowledgeSearch('workspace-1', 'new query', { source: 'slack' }) ) expect(query.queryKey).toEqual([ ...knowledgeKeys.search('workspace-1', 'new query', { source: 'slack' }, 20, 'reader'), - 'indexed', + 'live', ]) expect(knowledgeKeys.search('workspace-1', 'query', { source: 'slack' })).not.toEqual( knowledgeKeys.search('workspace-1', 'query', { source: 'gitlab' }) diff --git a/apps/sim/hooks/queries/kb/knowledge.ts b/apps/sim/hooks/queries/kb/knowledge.ts index ce2833fd81c..8a9a4e1d349 100644 --- a/apps/sim/hooks/queries/kb/knowledge.ts +++ b/apps/sim/hooks/queries/kb/knowledge.ts @@ -58,7 +58,6 @@ import type { WorkspaceSearchFilters } from '@/lib/api/contracts/knowledge/searc import type { NativeSearchQuery } from '@/lib/api/contracts/mothership-assistant-tools' import { useSession } from '@/lib/auth/auth-client' import type { ChunkingStrategy, StrategyOptions } from '@/lib/chunkers/types' -import { useDeploymentShape } from '@/lib/core/config/deployment-shape' import { type ResourceScope, resourceScopeFields, @@ -1214,15 +1213,9 @@ async function searchWorkspaceKnowledge( interface WorkspaceKnowledgeSearchOptions { nativeQueries?: NativeSearchQuery[] reuseFreshResult?: boolean - /** - * Keeps the previous result painted when only the result limit changes. Set it when the - * surface owns the limit (Show more widens the same search); leave it off when the limit is - * part of what was asked for, so a new limit is a new search that never shows the old one. - */ - retainAcrossLimits?: boolean } -/** Searches the canonical index under the signed-in person's ACLs. */ +/** Searches connected providers with the signed-in person's access. */ export function useWorkspaceKnowledgeSearch( owner: string | ResourceScope | undefined, query: string, @@ -1230,10 +1223,7 @@ export function useWorkspaceKnowledgeSearch( topK = 20, options?: WorkspaceKnowledgeSearchOptions ) { - const { features } = useDeploymentShape() - const live = features.liveEnterpriseSearch === true const { data: session } = useSession() - const queryClient = useQueryClient() const userId = session?.user?.id const trimmed = query.trim() const scope = @@ -1247,7 +1237,7 @@ export function useWorkspaceKnowledgeSearch( return useQuery({ queryKey: [ ...knowledgeKeys.search(scopeKey, trimmed, filters, topK, userId, options?.nativeQueries), - live ? 'live' : 'indexed', + 'live', ], queryFn: ({ signal }) => searchWorkspaceKnowledge( @@ -1256,7 +1246,7 @@ export function useWorkspaceKnowledgeSearch( query: trimmed, filters, topK, - ...(live && options?.nativeQueries ? { nativeQueries: options.nativeQueries } : {}), + ...(options?.nativeQueries ? { nativeQueries: options.nativeQueries } : {}), }, signal ), @@ -1269,21 +1259,7 @@ export function useWorkspaceKnowledgeSearch( filters?.modifiedAfter || filters?.modifiedBefore ), - staleTime: live - ? options?.reuseFreshResult - ? 60_000 - : 0 - : WORKSPACE_KNOWLEDGE_SEARCH_STALE_TIME, + staleTime: options?.reuseFreshResult ? WORKSPACE_KNOWLEDGE_SEARCH_STALE_TIME : 0, retry: false, - placeholderData: (previous, previousQuery) => { - if (live || !userId || previousQuery?.state.status !== 'success') return undefined - if (previousQuery.state.isInvalidated) return undefined - const prefix = knowledgeKeys.searchQuery(scopeKey, trimmed, userId) - if (!prefix.every((part, index) => previousQuery.queryKey[index] === part)) return undefined - /** `search()` appends filters, then the limit, after the reader/query prefix. */ - const previousTopK = previousQuery.queryKey[prefix.length + 1] - if (!options?.retainAcrossLimits && previousTopK !== topK) return undefined - return queryClient.getQueryData(previousQuery.queryKey) === previous ? previous : undefined - }, }) } diff --git a/apps/sim/hooks/queries/organization-accounts.ts b/apps/sim/hooks/queries/organization-accounts.ts index 9ea8c72b571..571895a5a25 100644 --- a/apps/sim/hooks/queries/organization-accounts.ts +++ b/apps/sim/hooks/queries/organization-accounts.ts @@ -48,7 +48,6 @@ import { connectDesktopSource } from '@/lib/desktop/source-connect' import { personalCredentialKeys } from '@/hooks/queries/personal-credentials' import { mcpKeys } from '@/hooks/queries/utils/mcp-keys' import { resetOrganizationSearchAccess } from '@/hooks/queries/utils/reset-organization-search-access' -import { searchSourceKeys } from '@/hooks/queries/utils/search-source-keys' import { invalidateSelectorQueries } from '@/hooks/queries/utils/selector-keys' import { slackSearchKeys } from '@/hooks/queries/utils/slack-search-keys' @@ -242,9 +241,6 @@ export function useUpdateOrganizationAccounts() { queryClient.invalidateQueries({ queryKey: slackSearchKeys.organizationManifests(organizationId), }), - queryClient.invalidateQueries({ - queryKey: searchSourceKeys.organizationOverview(organizationId), - }), ]), }) } diff --git a/apps/sim/hooks/queries/organization-search-stats.ts b/apps/sim/hooks/queries/organization-search-stats.ts deleted file mode 100644 index a11c471103b..00000000000 --- a/apps/sim/hooks/queries/organization-search-stats.ts +++ /dev/null @@ -1,28 +0,0 @@ -'use client' - -import { useQuery } from '@tanstack/react-query' -import { requestJson } from '@/lib/api/client/request' -import { - type OrganizationSearchStatsQuery, - readOrganizationSearchStatsContract, -} from '@/lib/api/contracts/knowledge/search-stats' - -export const ORGANIZATION_SEARCH_STATS_STALE_TIME = 60_000 - -export const organizationSearchStatsKeys = { - all: ['organization-search-stats'] as const, - summaries: () => [...organizationSearchStatsKeys.all, 'summary'] as const, - summary: (query: OrganizationSearchStatsQuery) => - [...organizationSearchStatsKeys.summaries(), query] as const, -} - -export function useOrganizationSearchStats(query: OrganizationSearchStatsQuery) { - return useQuery({ - queryKey: organizationSearchStatsKeys.summary(query), - queryFn: async ({ signal }) => { - const result = await requestJson(readOrganizationSearchStatsContract, { query, signal }) - return result.data - }, - staleTime: ORGANIZATION_SEARCH_STATS_STALE_TIME, - }) -} diff --git a/apps/sim/hooks/queries/personal-source-setup.ts b/apps/sim/hooks/queries/personal-source-setup.ts deleted file mode 100644 index 645b8bd8c97..00000000000 --- a/apps/sim/hooks/queries/personal-source-setup.ts +++ /dev/null @@ -1,67 +0,0 @@ -'use client' - -import { useMutation, useQuery, useQueryClient } from '@tanstack/react-query' -import { requestJson } from '@/lib/api/client/request' -import { - listPersonalSourceSetupAccountsContract, - type PersonalSourceSetupBody, - type PersonalSourceSetupQuery, - personalSourceSetupContract, -} from '@/lib/api/contracts/knowledge/personal-source-setup' -import { organizationAccountsKeys } from '@/hooks/queries/organization-accounts' -import { personalSearchIntegrationKeys } from '@/hooks/queries/personal-search-integrations' -import { searchSourceKeys } from '@/hooks/queries/utils/search-source-keys' - -export const PERSONAL_SOURCE_SETUP_STALE_TIME = 15_000 - -export const personalSourceSetupKeys = { - all: ['personal-source-setup'] as const, - lists: () => [...personalSourceSetupKeys.all, 'list'] as const, - list: (query: PersonalSourceSetupQuery) => [...personalSourceSetupKeys.lists(), query] as const, -} - -export function usePersonalSourceSetupAccounts(query: PersonalSourceSetupQuery) { - return useQuery({ - queryKey: personalSourceSetupKeys.list(query), - queryFn: async ({ signal }) => - (await requestJson(listPersonalSourceSetupAccountsContract, { query, signal })).data, - staleTime: PERSONAL_SOURCE_SETUP_STALE_TIME, - refetchInterval: (state) => - query.completionId && !state.state.data?.completedCredentialId ? 1_500 : false, - }) -} - -export function useAuthorizePersonalSourceSetup() { - return useMutation({ - mutationFn: async (body: Extract) => { - const { data } = await requestJson(personalSourceSetupContract, { body }) - if (data.kind !== 'authorization') throw new Error('Could not start account authorization') - return data - }, - }) -} - -export function useConnectPersonalSourceSetup() { - const client = useQueryClient() - return useMutation({ - mutationFn: async (body: Extract) => { - const { data } = await requestJson(personalSourceSetupContract, { body }) - if (data.kind !== 'connected') throw new Error('Could not connect the source') - return data - }, - onSuccess: (_data, body) => - Promise.all([ - client.invalidateQueries({ queryKey: personalSourceSetupKeys.lists() }), - client.invalidateQueries({ queryKey: personalSearchIntegrationKeys.lists() }), - client.invalidateQueries({ - queryKey: searchSourceKeys.list({ - kind: 'organization', - organizationId: body.organizationId, - }), - }), - client.invalidateQueries({ - queryKey: organizationAccountsKeys.detail(body.organizationId), - }), - ]), - }) -} diff --git a/apps/sim/hooks/queries/search-integrations.ts b/apps/sim/hooks/queries/search-integrations.ts index 85829c16b43..ba870ae1a85 100644 --- a/apps/sim/hooks/queries/search-integrations.ts +++ b/apps/sim/hooks/queries/search-integrations.ts @@ -12,9 +12,10 @@ import { searchIntegrationKeys } from '@/hooks/queries/utils/search-integration- export const SEARCH_INTEGRATIONS_STALE_TIME = 30_000 -export function useSearchIntegrations(organizationId: string) { +export function useSearchIntegrations(organizationId: string, options?: { enabled?: boolean }) { return useQuery({ queryKey: searchIntegrationKeys.list(organizationId), + enabled: Boolean(organizationId) && (options?.enabled ?? true), queryFn: async ({ signal }) => (await requestJson(listSearchIntegrationsContract, { query: { organizationId }, signal })) .data, diff --git a/apps/sim/hooks/queries/selectors.test.tsx b/apps/sim/hooks/queries/selectors.test.tsx index 5b6a5170813..f766668d124 100644 --- a/apps/sim/hooks/queries/selectors.test.tsx +++ b/apps/sim/hooks/queries/selectors.test.tsx @@ -3,10 +3,6 @@ */ import { act } from 'react' -import { - apiClientRequestMock, - apiClientRequestMockFns, -} from '@sim/testing/mocks/api-client-request.mock' import { QueryClient, QueryClientProvider } from '@tanstack/react-query' import { createRoot, type Root } from 'react-dom/client' import { afterEach, describe, expect, it, vi } from 'vitest' @@ -19,12 +15,8 @@ vi.mock('@/lib/selectors/client/execute-selector', () => ({ executeSelectorRequest: mockExecuteSelectorRequest, })) -vi.mock('@/lib/api/client/request', () => apiClientRequestMock) - import { useSelectorOptionDetail, useSelectorOptions } from '@/hooks/queries/selectors' -const mockRequestJson = apiClientRequestMockFns.mockRequestJson - interface HookHarness { getResult: () => T queryClient: QueryClient @@ -104,76 +96,6 @@ afterEach(() => { }) describe('generic selector queries', () => { - it('uses the dedicated personal setup contract and isolates it from ordinary browsing', async () => { - const personalItems = [{ id: 'PERSONAL', label: 'Personal project' }] - mockRequestJson.mockResolvedValue({ - success: true, - data: { kind: 'list', items: personalItems }, - }) - mockExecuteSelectorRequest.mockResolvedValue({ - kind: 'list', - items: [{ id: 'ADMIN', label: 'Admin project' }], - }) - const hook = renderHookWithClient(() => - useSelectorOptions('jira.projectKeys', { - context: { oauthCredential: 'credential-1', domain: 'example.atlassian.net' }, - scope: { kind: 'organization', organizationId: 'org-1' }, - surface: { kind: 'personal-search-setup', organizationId: 'org-1', connectorType: 'jira' }, - surfaceId: 'projects', - }) - ) - await waitFor(() => expect(hook.getResult().data).toEqual(personalItems)) - expect(mockExecuteSelectorRequest).not.toHaveBeenCalled() - expect(mockRequestJson).toHaveBeenCalledWith( - expect.objectContaining({ path: '/api/knowledge/sim-search/personal-source-setup' }), - expect.objectContaining({ - body: { - action: 'options', - organizationId: 'org-1', - connectorType: 'jira', - credentialId: 'credential-1', - domain: 'example.atlassian.net', - request: { kind: 'list' }, - }, - signal: expect.any(AbortSignal), - }) - ) - hook.rerender(() => - useSelectorOptions('jira.projectKeys', { - context: { oauthCredential: 'credential-1', domain: 'example.atlassian.net' }, - scope: { kind: 'organization', organizationId: 'org-1' }, - surfaceId: 'projects', - }) - ) - await waitFor(() => - expect(hook.getResult().data).toEqual([{ id: 'ADMIN', label: 'Admin project' }]) - ) - expect(mockExecuteSelectorRequest).toHaveBeenCalledTimes(1) - }) - - it.each(['selector', 'organization'] as const)( - 'rejects a mismatched personal setup %s before sending a request', - async (mismatch) => { - const hook = renderHookWithClient(() => - useSelectorOptions(mismatch === 'selector' ? 'confluence.spaces' : 'jira.projectKeys', { - context: { oauthCredential: 'credential-1', domain: 'example.atlassian.net' }, - scope: { - kind: 'organization', - organizationId: mismatch === 'organization' ? 'org-2' : 'org-1', - }, - surface: { - kind: 'personal-search-setup', - organizationId: 'org-1', - connectorType: 'jira', - }, - }) - ) - await waitFor(() => expect(hook.getResult().error).not.toBeNull()) - expect(mockRequestJson).not.toHaveBeenCalled() - expect(mockExecuteSelectorRequest).not.toHaveBeenCalled() - } - ) - it('transports supported search and keeps context and request plaintext out of query keys', async () => { const credentialReference = '{{SHARED_GOOGLE_CREDENTIAL}}' const search = 'private search phrase' diff --git a/apps/sim/hooks/queries/selectors.ts b/apps/sim/hooks/queries/selectors.ts index a92dcf537c1..1dd37f1995c 100644 --- a/apps/sim/hooks/queries/selectors.ts +++ b/apps/sim/hooks/queries/selectors.ts @@ -2,12 +2,7 @@ import { useCallback, useEffect, useId, useMemo, useRef, useState } from 'react' import { useInfiniteQuery, useQueries, useQuery } from '@tanstack/react-query' -import { requestJson } from '@/lib/api/client/request' -import { personalSourceSetupContract } from '@/lib/api/contracts/knowledge/personal-source-setup' -import { - type ExecuteSelectorClientInput, - executeSelectorRequest, -} from '@/lib/selectors/client/execute-selector' +import { executeSelectorRequest } from '@/lib/selectors/client/execute-selector' import { projectSelectorContext } from '@/lib/selectors/context' import { MAX_SELECTOR_OPTIONS, MAX_SELECTOR_PAGES } from '@/lib/selectors/limits' import { @@ -21,7 +16,6 @@ import type { SelectorOption, SelectorPage, SelectorScope, - SelectorSurface, } from '@/lib/selectors/types' import { selectorKeys } from '@/hooks/queries/utils/selector-keys' @@ -40,38 +34,6 @@ interface SelectorHookArgs { search?: string enabled?: boolean surfaceId?: string - surface?: SelectorSurface -} - -async function executeForSurface( - input: ExecuteSelectorClientInput, - surface?: SelectorSurface -): Promise { - if (!surface) return executeSelectorRequest(input) - const expectedKey = surface.connectorType === 'jira' ? 'jira.projectKeys' : 'confluence.spaces' - if ( - input.selectorKey !== expectedKey || - input.scope?.kind !== 'organization' || - input.scope.organizationId !== surface.organizationId || - !input.context.oauthCredential || - !input.context.domain - ) - throw new Error('This selector is not available during personal source setup') - const result = await requestJson(personalSourceSetupContract, { - body: { - action: 'options', - organizationId: surface.organizationId, - connectorType: surface.connectorType, - credentialId: input.context.oauthCredential, - domain: input.context.domain, - request: input.request, - }, - signal: input.signal, - }) - if (result.data.kind !== 'list' && result.data.kind !== 'detail') { - throw new Error('Personal source setup returned an unexpected selector result') - } - return result.data } export interface SelectorOptionsResult { @@ -181,13 +143,7 @@ function usePreparedSelector( const context = projectSelectorContext(key, args.context) const scope = selectorScopeFromContext(args.context, args.scope) const contextValues = manifest.context.allowed.map((field) => context[field]) - const revision = useOpaqueRevision([ - ...contextValues, - ...requestValues, - args.surface?.kind, - args.surface?.organizationId, - args.surface?.connectorType, - ]) + const revision = useOpaqueRevision([...contextValues, ...requestValues]) const ready = args.enabled !== false && isSelectorReady(key, context) && @@ -199,7 +155,6 @@ function usePreparedSelector( revision, ready, surfaceId: args.surfaceId ?? generatedSurfaceId, - surface: args.surface, } } @@ -222,19 +177,16 @@ export function useSelectorOptions( // rq-lint-allow: context and search are represented by an opaque privacy revision. queryKey: baseKey, queryFn: async ({ signal }) => { - const result = await executeForSurface( - { - selectorKey: key, - scope: prepared.scope, - context: prepared.context, - request: { - kind: 'list', - ...(effectiveSearch !== undefined ? { search: effectiveSearch } : {}), - }, - signal, + const result = await executeSelectorRequest({ + selectorKey: key, + scope: prepared.scope, + context: prepared.context, + request: { + kind: 'list', + ...(effectiveSearch !== undefined ? { search: effectiveSearch } : {}), }, - prepared.surface - ) + signal, + }) if (result.kind !== 'list') throw new Error('Selector returned an unexpected detail result') return result }, @@ -247,20 +199,17 @@ export function useSelectorOptions( // rq-lint-allow: context and search are represented by an opaque privacy revision. queryKey: [...baseKey, 'paged'], queryFn: async ({ pageParam, signal }) => { - const result = await executeForSurface( - { - selectorKey: key, - scope: prepared.scope, - context: prepared.context, - request: { - kind: 'list', - ...(effectiveSearch !== undefined ? { search: effectiveSearch } : {}), - ...(typeof pageParam === 'string' ? { cursor: pageParam } : {}), - }, - signal, + const result = await executeSelectorRequest({ + selectorKey: key, + scope: prepared.scope, + context: prepared.context, + request: { + kind: 'list', + ...(effectiveSearch !== undefined ? { search: effectiveSearch } : {}), + ...(typeof pageParam === 'string' ? { cursor: pageParam } : {}), }, - prepared.surface - ) + signal, + }) if (result.kind !== 'list') throw new Error('Selector returned an unexpected detail result') return result }, @@ -441,16 +390,13 @@ export function useSelectorOptionDetail( prepared.revision ), queryFn: async ({ signal }) => { - const result = await executeForSurface( - { - selectorKey: key, - scope: prepared.scope, - context: prepared.context, - request: { kind: 'detail', id: args.detailId! }, - signal, - }, - prepared.surface - ) + const result = await executeSelectorRequest({ + selectorKey: key, + scope: prepared.scope, + context: prepared.context, + request: { kind: 'detail', id: args.detailId! }, + signal, + }) if (result.kind !== 'detail') throw new Error('Selector returned an unexpected list result') return result.item }, @@ -478,16 +424,13 @@ export function useSelectorOptionDetails( ordinal ), queryFn: async ({ signal }: { signal: AbortSignal }) => { - const result = await executeForSurface( - { - selectorKey: key, - scope: prepared.scope, - context: prepared.context, - request: { kind: 'detail', id: detailId }, - signal, - }, - prepared.surface - ) + const result = await executeSelectorRequest({ + selectorKey: key, + scope: prepared.scope, + context: prepared.context, + request: { kind: 'detail', id: detailId }, + signal, + }) if (result.kind !== 'detail') throw new Error('Selector returned an unexpected list result') return result.item }, diff --git a/apps/sim/hooks/queries/utils/reset-organization-search-access.test.ts b/apps/sim/hooks/queries/utils/reset-organization-search-access.test.ts index 3c44a253310..acf3fa91296 100644 --- a/apps/sim/hooks/queries/utils/reset-organization-search-access.test.ts +++ b/apps/sim/hooks/queries/utils/reset-organization-search-access.test.ts @@ -1,62 +1,9 @@ -import { QueryClient, QueryObserver } from '@tanstack/react-query' +import { QueryClient } from '@tanstack/react-query' import { expect, it, vi } from 'vitest' import type { WorkspaceKnowledgeSearchResult } from '@/lib/api/contracts/knowledge/search' import { resourceScopeKey } from '@/lib/core/resource-scope' import { knowledgeKeys } from '@/hooks/queries/utils/knowledge-keys' import { resetOrganizationSearchAccess } from '@/hooks/queries/utils/reset-organization-search-access' -import { searchSourceKeys } from '@/hooks/queries/utils/search-source-keys' - -it.each([true, false])( - 'keeps administrative rows visible while revalidating access, refresh success=%s', - async (success) => { - const client = new QueryClient({ defaultOptions: { queries: { retry: false } } }) - const scope = { kind: 'organization', organizationId: 'org-1' } as const - const adminKey = searchSourceKeys.organizationOverview(scope.organizationId) - const otherKey = searchSourceKeys.organizationOverview('org-2') - const viewerKeys = [ - searchSourceKeys.list(scope), - searchSourceKeys.overview(scope), - searchSourceKeys.pages(scope, { search: '', mine: false }), - ] - const before = { providers: [{ connectorType: 'gmail', approved: true }] } - const after = { providers: [{ connectorType: 'gmail', approved: false }] } - const response = Promise.withResolvers() - const fetchOverview = vi.fn(() => response.promise) - client.setQueryData(adminKey, before) - client.setQueryData(otherKey, before) - for (const key of viewerKeys) client.setQueryData(key, { privateContent: 'previous access' }) - const observer = new QueryObserver(client, { - queryKey: adminKey, - queryFn: fetchOverview, - staleTime: Number.POSITIVE_INFINITY, - }) - const observed = vi.fn() - const unsubscribe = observer.subscribe(observed) - try { - const refreshing = resetOrganizationSearchAccess(client, scope.organizationId) - expect(fetchOverview).toHaveBeenCalledOnce() - expect(observer.getCurrentResult()).toMatchObject({ data: before, isPending: false }) - for (const key of viewerKeys) expect(client.getQueryData(key)).toBeUndefined() - expect(client.getQueryState(otherKey)?.isInvalidated).toBe(false) - - if (success) response.resolve(after) - else response.reject(new Error('Could not refresh sources')) - await refreshing - - expect(observer.getCurrentResult()).toMatchObject({ - data: success ? after : before, - isError: !success, - isFetching: false, - }) - expect(observed.mock.calls.every(([result]) => result.data && !result.isPending)).toBe(true) - expect(client.getQueryData(otherKey)).toEqual(before) - } finally { - response.resolve(after) - unsubscribe() - client.clear() - } - } -) it.each([ { name: 'document', key: knowledgeKeys.document('kb-direct', 'document-direct') }, diff --git a/apps/sim/hooks/queries/utils/reset-organization-search-access.ts b/apps/sim/hooks/queries/utils/reset-organization-search-access.ts index 53f7b3a5cfb..f97837d750c 100644 --- a/apps/sim/hooks/queries/utils/reset-organization-search-access.ts +++ b/apps/sim/hooks/queries/utils/reset-organization-search-access.ts @@ -1,4 +1,4 @@ -import { matchQuery, type QueryClient } from '@tanstack/react-query' +import type { QueryClient } from '@tanstack/react-query' import { resourceScopeKey } from '@/lib/core/resource-scope' import { knowledgeKeys } from '@/hooks/queries/utils/knowledge-keys' import { searchSourceKeys } from '@/hooks/queries/utils/search-source-keys' @@ -9,10 +9,6 @@ export async function resetOrganizationSearchAccess( organizationId: string ) { const scope = { kind: 'organization', organizationId } as const - const adminOverview = { - queryKey: searchSourceKeys.organizationOverview(organizationId), - exact: true, - } await Promise.all([ queryClient.resetQueries({ queryKey: [...knowledgeKeys.searches(), resourceScopeKey(scope)], @@ -21,8 +17,6 @@ export async function resetOrganizationSearchAccess( queryClient.resetQueries({ queryKey: knowledgeKeys.details() }), queryClient.resetQueries({ queryKey: searchSourceKeys.list(scope), - predicate: (query) => !matchQuery(adminOverview, query), }), - queryClient.invalidateQueries(adminOverview), ]) } diff --git a/apps/sim/hooks/queries/utils/search-source-keys.ts b/apps/sim/hooks/queries/utils/search-source-keys.ts index ee508d3fb1b..1ceb82e5bf2 100644 --- a/apps/sim/hooks/queries/utils/search-source-keys.ts +++ b/apps/sim/hooks/queries/utils/search-source-keys.ts @@ -3,26 +3,14 @@ import { type ResourceScope, resourceScopeKey } from '@/lib/core/resource-scope' export const searchSourceKeys = { all: ['search-sources'] as const, lists: () => [...searchSourceKeys.all, 'list'] as const, - progress: (scope: ResourceScope | undefined, connectorIds: string[]) => - [ - ...searchSourceKeys.all, - 'progress', - scope ? resourceScopeKey(scope) : '', - connectorIds, - ] as const, pages: ( scope: string | ResourceScope | undefined, filters: { search: string - mine: boolean connectorType?: string excludeConnectorType?: string } ) => [...searchSourceKeys.list(scope), 'pages', filters] as const, - overview: (scope?: string | ResourceScope) => - [...searchSourceKeys.list(scope), 'overview'] as const, - organizationOverview: (organizationId: string) => - [...searchSourceKeys.list({ kind: 'organization', organizationId }), 'admin-overview'] as const, list: (scope?: string | ResourceScope) => [ ...searchSourceKeys.lists(), diff --git a/apps/sim/hooks/use-personal-source-account.test.tsx b/apps/sim/hooks/use-personal-source-account.test.tsx deleted file mode 100644 index 20e9aba58a1..00000000000 --- a/apps/sim/hooks/use-personal-source-account.test.tsx +++ /dev/null @@ -1,130 +0,0 @@ -/** - * @vitest-environment jsdom - */ -import { act } from 'react' -import { emcnMock, emcnMockFns } from '@sim/testing/mocks/emcn.mock' -import { reactQueryMock, reactQueryMockFns } from '@sim/testing/mocks/react-query.mock' -import { createRoot, type Root } from 'react-dom/client' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const mocks = vi.hoisted(() => ({ - authorize: vi.fn(), - refetch: vi.fn(), - completed: null as string | null, - connected: vi.fn(), -})) -vi.mock('@tanstack/react-query', () => reactQueryMock) -vi.mock('@sim/emcn', () => emcnMock) -vi.mock('@/hooks/queries/personal-source-setup', () => ({ - personalSourceSetupKeys: { list: (query: unknown) => ['personal-source-setup', query] }, - useAuthorizePersonalSourceSetup: () => ({ mutateAsync: mocks.authorize, isPending: false }), - usePersonalSourceSetupAccounts: () => ({ - data: { accounts: [], completedCredentialId: mocks.completed }, - refetch: mocks.refetch, - }), -})) - -import { usePersonalSourceAccount } from '@/hooks/use-personal-source-account' - -const mockToastError = emcnMockFns.mockToast.error -const mockSetQueryData = reactQueryMockFns.mockQueryClient.setQueryData - -describe('personal source account authorization', () => { - let root: Root - let container: HTMLDivElement - let current: ReturnType - let channels: Array<{ - onmessage: ((event: MessageEvent) => void) | null - close: ReturnType - }> - let tab: { - opener: unknown - location: { href: string } - focus: ReturnType - close: ReturnType - } - function Probe() { - current = usePersonalSourceAccount({ - organizationId: 'org-1', - connectorType: 'jira', - onConnected: mocks.connected, - }) - return null - } - beforeEach(() => { - vi.useFakeTimers() - mocks.completed = null - mocks.authorize.mockResolvedValue({ - kind: 'authorization', - url: 'https://auth.atlassian.com/authorize', - }) - channels = [] - vi.stubGlobal( - 'BroadcastChannel', - class { - onmessage = null - close = vi.fn() - constructor() { - channels.push(this) - } - } - ) - tab = { opener: {}, location: { href: 'about:blank' }, focus: vi.fn(), close: vi.fn() } - vi.spyOn(window, 'open').mockReturnValue(tab as unknown as Window) - ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true - container = document.createElement('div') - document.body.appendChild(container) - root = createRoot(container) - act(() => root.render()) - }) - afterEach(() => { - act(() => root.unmount()) - container.remove() - vi.useRealTimers() - }) - it('authorizes before a source exists and requires server confirmation of completion', async () => { - await act(async () => current.connect()) - expect(tab.opener).toBeNull() - expect(tab.location.href).toBe('https://auth.atlassian.com/authorize') - expect(mocks.authorize).toHaveBeenCalledWith( - expect.objectContaining({ - action: 'authorize', - organizationId: 'org-1', - connectorType: 'jira', - oauthCompletionId: expect.any(String), - }) - ) - expect(current.pending).toBe(true) - act(() => channels[0].onmessage?.({ data: 'connected' } as MessageEvent)) - expect(mocks.refetch).toHaveBeenCalledOnce() - expect(current.pending).toBe(true) - expect(tab.close).not.toHaveBeenCalled() - mocks.completed = 'my-account' - act(() => root.render()) - expect(current.pending).toBe(false) - expect(tab.close).toHaveBeenCalledOnce() - expect(mocks.connected).toHaveBeenCalledWith('my-account') - mocks.completed = null - act(() => root.render()) - expect(current.pending).toBe(false) - }) - it('ignores a late authorization response after cancellation', async () => { - let resolve!: (value: { kind: string; url: string }) => void - mocks.authorize.mockReturnValue( - new Promise((done) => { - resolve = done - }) - ) - let connecting!: Promise - act(() => { - connecting = current.connect() - }) - act(() => current.cancel()) - await act(async () => { - resolve({ kind: 'authorization', url: 'https://auth.atlassian.com/authorize' }) - await connecting - }) - expect(tab.location.href).toBe('about:blank') - expect(current.pending).toBe(false) - }) -}) diff --git a/apps/sim/hooks/use-personal-source-account.ts b/apps/sim/hooks/use-personal-source-account.ts deleted file mode 100644 index 3ac8447127d..00000000000 --- a/apps/sim/hooks/use-personal-source-account.ts +++ /dev/null @@ -1,135 +0,0 @@ -'use client' - -import { useCallback, useEffect, useRef, useState } from 'react' -import { toast } from '@sim/emcn' -import { getErrorMessage } from '@sim/utils/errors' -import { generateId } from '@sim/utils/id' -import { useQueryClient } from '@tanstack/react-query' -import type { PersonalSourceSetupQuery } from '@/lib/api/contracts/knowledge/personal-source-setup' -import { - CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES, - credentialGroupOAuthCompletionChannel, - isCredentialGroupOAuthFailure, -} from '@/lib/credential-groups/oauth-completion' -import { - personalSourceSetupKeys, - useAuthorizePersonalSourceSetup, - usePersonalSourceSetupAccounts, -} from '@/hooks/queries/personal-source-setup' - -interface PersonalSourceAccountProps { - organizationId: string - connectorType: PersonalSourceSetupQuery['connectorType'] - onConnected?: (credentialId: string) => void -} - -/** A popup message only refreshes the account inventory; the server proves completion. */ -export function usePersonalSourceAccount({ - organizationId, - connectorType, - onConnected, -}: PersonalSourceAccountProps) { - const [completionId, setCompletionId] = useState() - const attempt = useRef<{ id: string; tab: Window } | null>(null) - const accounts = usePersonalSourceSetupAccounts({ organizationId, connectorType, completionId }) - const { mutateAsync: authorize } = useAuthorizePersonalSourceSetup() - const queryClient = useQueryClient() - const onConnectedRef = useRef(onConnected) - const completed = accounts.data?.completedCredentialId - const { refetch } = accounts - - useEffect(() => { - onConnectedRef.current = onConnected - }, [onConnected]) - - useEffect(() => { - return () => { - attempt.current?.tab.close() - attempt.current = null - } - }, []) - - useEffect(() => { - if (!completionId || completed) return - const fail = (message: string) => { - if (attempt.current?.id !== completionId) return - attempt.current.tab.close() - attempt.current = null - setCompletionId(undefined) - toast.error(message) - } - const channel = new BroadcastChannel(credentialGroupOAuthCompletionChannel(completionId)) - channel.onmessage = ({ data }: MessageEvent) => { - if (isCredentialGroupOAuthFailure(data)) fail(CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES[data]) - else if (data === 'connected') void refetch() - } - const timer = window.setTimeout( - () => fail(CREDENTIAL_GROUP_OAUTH_FAILURE_MESSAGES.expired), - 10 * 60_000 - ) - return () => { - channel.close() - window.clearTimeout(timer) - } - }, [completionId, completed, refetch]) - - useEffect(() => { - if (completed && attempt.current && attempt.current.id === completionId) { - attempt.current.tab.close() - attempt.current = null - setCompletionId(undefined) - queryClient.setQueryData(personalSourceSetupKeys.list({ organizationId, connectorType }), { - ...accounts.data, - completedCredentialId: null, - }) - onConnectedRef.current?.(completed) - } - }, [completed, completionId, accounts.data, queryClient, organizationId, connectorType]) - - const connect = useCallback(async () => { - if (attempt.current) { - attempt.current.tab.focus() - return - } - const tab = window.open('about:blank', '_blank', 'width=600,height=700') - if (!tab) { - toast.error('Allow pop-ups for this site to connect your account.') - return - } - tab.opener = null - const id = generateId() - attempt.current = { id, tab } - setCompletionId(id) - try { - const result = await authorize({ - action: 'authorize', - organizationId, - connectorType, - oauthCompletionId: id, - }) - if (attempt.current?.id !== id) return - const url = new URL(result.url) - if ( - url.protocol !== 'https:' && - !(url.protocol === 'http:' && url.origin === window.location.origin) - ) { - throw new Error('The provider authorization URL is invalid') - } - tab.location.href = url.href - } catch (error) { - if (attempt.current?.id !== id) return - tab.close() - attempt.current = null - setCompletionId(undefined) - toast.error(getErrorMessage(error, 'Could not connect your account')) - } - }, [authorize, organizationId, connectorType]) - - const cancel = useCallback(() => { - attempt.current?.tab.close() - attempt.current = null - setCompletionId(undefined) - }, []) - - return { accounts, connect, cancel, pending: Boolean(completionId) } -} diff --git a/apps/sim/hooks/use-search-integration-connection.test.tsx b/apps/sim/hooks/use-search-integration-connection.test.tsx index 0f7c87a0a8d..078f1555e89 100644 --- a/apps/sim/hooks/use-search-integration-connection.test.tsx +++ b/apps/sim/hooks/use-search-integration-connection.test.tsx @@ -22,9 +22,8 @@ const m = vi.hoisted(() => ({ type: 'link' provider: string connectorType: string - connectorId?: string - connectionMode?: 'live' - optionId?: string + connectionMode: 'live' + optionId: string } | undefined, })) @@ -32,7 +31,8 @@ const target = { type: 'link', provider: 'gmail', connectorType: 'gmail', - connectorId: 'source', + connectionMode: 'live', + optionId: 'gmail-option', } as const vi.mock('@tanstack/react-query', () => reactQueryMock) vi.mock('@/hooks/queries/personal-search-integrations', () => ({ @@ -41,9 +41,7 @@ vi.mock('@/hooks/queries/personal-search-integrations', () => ({ usePersonalSearchIntegrations: (query: { completionId?: string }) => ({ data: { connections: [{ accounts: m.accounts }], - available: [ - { target: m.requestedTarget?.connectionMode === 'live' ? m.requestedTarget : target }, - ], + available: [{ target: m.requestedTarget ?? target }], completedCredentialId: m.receipts.get(query.completionId ?? '') ?? null, }, isSuccess: !m.queryError, @@ -123,8 +121,6 @@ beforeEach(() => { m.refetch.mockResolvedValue({ isSuccess: true, data: { connections: [] } }) m.mutate.mockResolvedValue({ url: 'https://provider.test/authorize', - connectorId: 'source', - knowledgeBaseId: 'kb', }) container = document.createElement('div') document.body.append(container) @@ -138,20 +134,6 @@ afterEach(() => { }) describe('Search connection card lifecycle', () => { - it('confirms account-first setup against inventory and reflects later revocation', () => { - m.requestedTarget = { type: 'link', provider: 'jira', connectorType: 'jira' } - render() - act(() => connection().completeSetup({ connectorId: 'new-source', credentialId: 'mine' })) - expect(connection().connectorId).toBe('new-source') - expect(connection().connected).toBe(false) - expect(m.mutate).not.toHaveBeenCalled() - m.accounts = [{ credentialId: 'mine', status: 'connected' }] - render() - expect(connection().connected).toBe(true) - m.accounts = [{ credentialId: 'mine', status: 'reconnect_needed' }] - render() - expect(connection().connected).toBe(false) - }) it('starts OAuth only on click and completes only when its receipt and current account agree', async () => { render() expect(m.mutate).not.toHaveBeenCalled() @@ -246,24 +228,6 @@ describe('Search connection card lifecycle', () => { expect(connection().connected).toBe(true) expect(windows[1].close).toHaveBeenCalled() }) - it('retries the exact source created by the first attempt after cancellation or reload', async () => { - m.requestedTarget = { type: 'link', provider: 'gmail', connectorType: 'gmail' } - render() - await act(async () => { - await connection().connect() - }) - expect(m.mutate.mock.calls[0][0].target.connectorId).toBeUndefined() - act(() => connection().cancel()) - act(() => root.unmount()) - root = createRoot(container) - render() - expect(connection().available).toBe(true) - await act(async () => { - await connection().connect() - }) - expect(m.mutate.mock.calls[1][0].target.connectorId).toBe('source') - expect(m.mutate.mock.calls[1][0].target.credentialId).toBeUndefined() - }) it('keeps failed starts actionable and rejects stale data after authorization errors', async () => { render() m.mutate.mockRejectedValueOnce(new Error('Source no longer available')) diff --git a/apps/sim/hooks/use-search-integration-connection.ts b/apps/sim/hooks/use-search-integration-connection.ts index ed7ba243072..3596fd7c0dc 100644 --- a/apps/sim/hooks/use-search-integration-connection.ts +++ b/apps/sim/hooks/use-search-integration-connection.ts @@ -61,13 +61,10 @@ export function useSearchIntegrationConnection({ const client = useQueryClient() const { mutateAsync, isPending } = useConnectPersonalSearchIntegration() const pending = attempt?.status === 'pending' - const connectorId = target.connectorId ?? attempt?.connectorId - const effectiveTarget = connectorId ? { ...target, connectorId } : target const query = usePersonalSearchIntegrations( { organizationId, connectorType: target.connectorType, - connectorId, completionId: attempt?.completionId, }, { pending } @@ -93,9 +90,7 @@ export function useSearchIntegrationConnection({ ] const available = query.isSuccess && - availableTargets.some( - (candidate) => JSON.stringify(candidate) === JSON.stringify(effectiveTarget) - ) + availableTargets.some((candidate) => JSON.stringify(candidate) === JSON.stringify(target)) useEffect(() => { const refresh = () => setAttempt(readSearchConnectionAttempt(key)) @@ -152,72 +147,66 @@ export function useSearchIntegrationConnection({ }, [connected, attempt, key, query.data?.completedCredentialId]) const { refetch } = query - const connect = useCallback( - async (sourceConfig?: Record) => { - if (starting.current || isPending || connected) return - if (pending && popup.current && !popup.current.closed) { - popup.current.focus() - return - } - const desktop = isDesktopApp() - if (desktop && pending) return - const tab = desktop ? null : window.open('about:blank', '_blank', 'width=600,height=700') - if (!desktop && !tab) { - setLocalError('Allow pop-ups for this site to connect your account.') - return + const connect = useCallback(async () => { + if (starting.current || isPending || connected) return + if (pending && popup.current && !popup.current.closed) { + popup.current.focus() + return + } + const desktop = isDesktopApp() + if (desktop && pending) return + const tab = desktop ? null : window.open('about:blank', '_blank', 'width=600,height=700') + if (!desktop && !tab) { + setLocalError('Allow pop-ups for this site to connect your account.') + return + } + if (tab) tab.opener = null + popup.current = tab + starting.current = true + const controller = new AbortController() + nativeAbort.current = controller + setLocalError(null) + let next: SearchConnectionAttempt | undefined + try { + if (!desktop) { + const fresh = await refetch() + if (!fresh.isSuccess) throw fresh.error } - if (tab) tab.opener = null - popup.current = tab - starting.current = true - const controller = new AbortController() - nativeAbort.current = controller - setLocalError(null) - let next: SearchConnectionAttempt | undefined - try { - if (!desktop) { - const fresh = await refetch() - if (!fresh.isSuccess) throw fresh.error - } - controller.signal.throwIfAborted() - next = { - completionId: generateId(), - requestedAt: Date.now(), - connectorId, - status: 'pending', - error: null, - } - writeSearchConnectionAttempt(key, next) - const result = await mutateAsync({ - organizationId, - target: connectorId ? { ...target, connectorId } : target, - sourceConfig, - oauthCompletionId: next.completionId, - signal: controller.signal, - }) - if (!result || !tab) return true - const url = new URL(result.url) - if ( - url.protocol !== 'https:' && - !(url.protocol === 'http:' && url.origin === window.location.origin) - ) - throw new Error('The provider authorization URL is invalid') - writeSearchConnectionAttempt(key, { ...next, connectorId: result.connectorId }) - tab.location.href = url.href - return true - } catch (error) { - tab?.close() - const message = getErrorMessage(error, 'Could not start the connection') - const current = readSearchConnectionAttempt(key) - if (next && current?.completionId === next.completionId && current.status === 'pending') - writeSearchConnectionAttempt(key, { ...current, status: 'failed', error: message }) - if (!controller.signal.aborted) setLocalError(message) - return false - } finally { - starting.current = false + controller.signal.throwIfAborted() + next = { + completionId: generateId(), + requestedAt: Date.now(), + status: 'pending', + error: null, } - }, - [isPending, connected, pending, refetch, mutateAsync, organizationId, target, connectorId, key] - ) + writeSearchConnectionAttempt(key, next) + const result = await mutateAsync({ + organizationId, + target, + oauthCompletionId: next.completionId, + signal: controller.signal, + }) + if (!result || !tab) return true + const url = new URL(result.url) + if ( + url.protocol !== 'https:' && + !(url.protocol === 'http:' && url.origin === window.location.origin) + ) + throw new Error('The provider authorization URL is invalid') + tab.location.href = url.href + return true + } catch (error) { + tab?.close() + const message = getErrorMessage(error, 'Could not start the connection') + const current = readSearchConnectionAttempt(key) + if (next && current?.completionId === next.completionId && current.status === 'pending') + writeSearchConnectionAttempt(key, { ...current, status: 'failed', error: message }) + if (!controller.signal.aborted) setLocalError(message) + return false + } finally { + starting.current = false + } + }, [isPending, connected, pending, refetch, mutateAsync, organizationId, target, key]) const cancel = useCallback(() => { nativeAbort.current?.abort() popup.current?.close() @@ -228,26 +217,11 @@ export function useSearchIntegrationConnection({ error: 'Connection canceled. You can try again.', }) }, [attempt, key]) - const completeSetup = useCallback( - (result: { connectorId: string; credentialId: string }) => { - writeSearchConnectionAttempt(key, { - completionId: generateId(), - requestedAt: Date.now(), - connectorId: result.connectorId, - credentialId: result.credentialId, - status: 'connected', - error: null, - }) - }, - [key] - ) return { - completeSetup, connect, cancel, inventoryError: query.error?.message, connected, - connectorId, pending, available, isStarting: isPending, diff --git a/apps/sim/lib/api/contracts/desktop-source-connect.ts b/apps/sim/lib/api/contracts/desktop-source-connect.ts index b18e749bbf2..c383303e362 100644 --- a/apps/sim/lib/api/contracts/desktop-source-connect.ts +++ b/apps/sim/lib/api/contracts/desktop-source-connect.ts @@ -1,6 +1,5 @@ import { z } from 'zod' import { startSlackCredentialGroupConfigurationBodySchema } from '@/lib/api/contracts/credential-groups' -import { connectSimSearchConnectorBodySchema } from '@/lib/api/contracts/knowledge/connectors' import { startGitHubSearchSetupBodySchema } from '@/lib/api/contracts/knowledge/github-setup' import { connectPersonalSearchIntegrationBodySchema } from '@/lib/api/contracts/knowledge/personal-integrations' import { knowledgeConnectorParamsSchema } from '@/lib/api/contracts/knowledge/shared' @@ -31,11 +30,6 @@ export const desktopSourceRequestSchema = z.discriminatedUnion('kind', [ params: knowledgeConnectorParamsSchema, completionId: z.string().uuid().optional(), }), - z.object({ - kind: z.literal('search-source'), - body: connectSimSearchConnectorBodySchema, - completionId: z.string().uuid().optional(), - }), z.object({ kind: z.literal('slack-managed-users'), owner: resourceOwnerSchema, diff --git a/apps/sim/lib/api/contracts/knowledge/connectors.ts b/apps/sim/lib/api/contracts/knowledge/connectors.ts index 962d101a731..91ac509ac86 100644 --- a/apps/sim/lib/api/contracts/knowledge/connectors.ts +++ b/apps/sim/lib/api/contracts/knowledge/connectors.ts @@ -8,11 +8,7 @@ import { knowledgeConnectorParamsSchema, successResponseSchema, } from '@/lib/api/contracts/knowledge/shared' -import { - booleanQueryFlagSchema, - organizationIdSchema, - resourceOwnerSchema, -} from '@/lib/api/contracts/primitives' +import { booleanQueryFlagSchema, resourceOwnerSchema } from '@/lib/api/contracts/primitives' import { defineRouteContract } from '@/lib/api/contracts/types' import { CONNECTOR_ACCESS_MODES } from '@/lib/knowledge/connectors/access-modes' import { @@ -20,9 +16,6 @@ import { MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_MUTATION_ITEMS, MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_PAGE_SIZE, MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_SEARCH_LENGTH, - MAX_SEARCH_SOURCE_PROGRESS_ITEMS, - MAX_SEARCH_SOURCE_PROVIDER_TYPES, - SEARCH_SOURCE_CANDIDATE_PAGE_SIZE, SEARCH_SOURCE_PAGE_SIZE, } from '@/lib/knowledge/constants' import { MEMBER_SYNC_STATUSES } from '@/lib/knowledge/types' @@ -350,41 +343,15 @@ const searchSourceSummaryFields = { availability: z.enum(['available', 'unavailable']), enabled: z.boolean(), approved: z.boolean().optional(), - isSyncing: z.boolean(), - lastSyncAt: z.string().datetime().nullable(), - hasSyncError: z.boolean(), - /** - * Whether the viewer can search at least one indexed document from this source. An - * existence flag rather than a count: counting means access-checking every visible document. - */ - hasViewerDocuments: z.boolean(), - viewerFailedDocumentCount: z.number().int().nonnegative().default(0), - viewerEmailVerified: z.boolean(), - viewerAccounts: z - .array( - z.object({ - credentialId: z.string().min(1).max(128), - displayName: z.string(), - status: z.enum(['active', 'needs_reauth']).optional(), - }) - ) - .max(SEARCH_SOURCE_CANDIDATE_PAGE_SIZE), } -export const searchSourceSummarySchema = z.discriminatedUnion('connectionRequired', [ - z.object({ - ...searchSourceSummaryFields, - connectionRequired: z.literal(true), - viewerMembership: viewerConnectorMembershipSchema.nullable(), - }), - z.object({ - ...searchSourceSummaryFields, - connectionRequired: z.literal(false), - viewerMembership: z.null(), - }), -]) +export const searchSourceSummarySchema = z.object(searchSourceSummaryFields) export type SearchSourceSummary = z.output -export type ViewerSearchSourceAccount = SearchSourceSummary['viewerAccounts'][number] +export interface ViewerSearchSourceAccount { + credentialId: string + displayName: string + status?: 'active' | 'needs_reauth' +} export const searchSourceCursorSchema = z.object({ createdAt: z.string().datetime(), @@ -402,7 +369,6 @@ export const listSearchSourcesQuerySchema = resourceOwnerSchema.safeExtend({ .max(100) .optional(), search: z.string().trim().max(200).optional(), - mine: booleanQueryFlagSchema.optional(), }) export type ListSearchSourcesQuery = z.input @@ -419,112 +385,6 @@ export const listSearchSourcesContract = defineRouteContract({ response: { mode: 'json', schema: successResponseSchema(searchSourcePageSchema) }, }) -export const searchSourceOverviewSchema = z.object({ - providers: z - .array( - z.object({ - connectorType: z.string().min(1).max(100), - isSyncing: z.boolean(), - }) - ) - .max(MAX_SEARCH_SOURCE_PROVIDER_TYPES), - hasSearchableDocuments: z.boolean(), -}) -export type SearchSourceOverview = z.output - -export const readSearchSourceOverviewContract = defineRouteContract({ - method: 'GET', - path: '/api/knowledge/sim-search/sources/overview', - query: resourceOwnerSchema, - response: { mode: 'json', schema: successResponseSchema(searchSourceOverviewSchema) }, -}) - -export const organizationSearchProviderStatusSchema = z.enum([ - 'needs_setup', - 'waiting_for_connections', - 'indexing', - 'needs_attention', - 'paused', - 'active', -]) -export type OrganizationSearchProviderStatus = z.output< - typeof organizationSearchProviderStatusSchema -> - -export const organizationSearchProviderSummarySchema = z.object({ - connectorType: z.string().min(1).max(100), - approved: z.boolean(), - sourceCount: z.number().int().nonnegative(), - status: organizationSearchProviderStatusSchema, - issue: z - .enum([ - 'sync_failed', - 'account_sync_incomplete', - 'document_indexing_failed', - 'permission_sync_incomplete', - ]) - .nullable(), - isSyncing: z.boolean(), - /** Older servers omit the continuation signal during a rolling deployment. */ - hasPendingSync: z.boolean().optional(), -}) -export type OrganizationSearchProviderSummary = z.output< - typeof organizationSearchProviderSummarySchema -> - -export const organizationSearchOverviewSchema = z.object({ - providers: z.array(organizationSearchProviderSummarySchema).max(MAX_SEARCH_SOURCE_PROVIDER_TYPES), -}) -export type OrganizationSearchOverview = z.output - -export const readOrganizationSearchOverviewQuerySchema = z.object({ - organizationId: organizationIdSchema, -}) -export type ReadOrganizationSearchOverviewQuery = z.input< - typeof readOrganizationSearchOverviewQuerySchema -> - -export const readOrganizationSearchOverviewContract = defineRouteContract({ - method: 'GET', - path: '/api/knowledge/sim-search/integrations/overview', - query: readOrganizationSearchOverviewQuerySchema, - response: { mode: 'json', schema: successResponseSchema(organizationSearchOverviewSchema) }, -}) - -export const searchSourceProgressSchema = z.object({ - connectorId: knowledgeConnectorParamsSchema.shape.connectorId, - isSyncing: z.boolean(), - hasSyncError: z.boolean(), - hasIndexingError: z.boolean(), -}) -export type SearchSourceProgress = z.output - -export const readSearchSourceProgressContract = defineRouteContract({ - method: 'POST', - path: '/api/knowledge/sim-search/sources/progress', - body: resourceOwnerSchema.safeExtend({ - connectorIds: z - .array(knowledgeConnectorParamsSchema.shape.connectorId.max(255)) - .min(1) - .max(MAX_SEARCH_SOURCE_PROGRESS_ITEMS), - }), - response: { - mode: 'json', - schema: successResponseSchema( - z.array(searchSourceProgressSchema).max(MAX_SEARCH_SOURCE_PROGRESS_ITEMS) - ), - }, -}) - -export const connectSimSearchConnectorBodySchema = resourceOwnerSchema.safeExtend({ - connectorType: z.string().min(1, 'connectorType cannot be empty').max(100), - connectorId: knowledgeConnectorParamsSchema.shape.connectorId.max(255).optional(), - /** Settings identify a compatible source, or assert the configuration of a selected source. */ - sourceConfig: z.record(z.string(), z.string().max(500)).optional(), - oauthCompletionId: searchConnectionOAuthQuerySchema.shape.oauthCompletionId, -}) -export type ConnectSimSearchConnectorBody = z.input - export const prepareSearchSourceBodySchema = resourceOwnerSchema.safeExtend({ connectorType: z.string().min(1, 'connectorType cannot be empty').max(100), accessMode: z.enum(['admin', 'members']).optional().default('admin'), @@ -547,28 +407,6 @@ export const prepareSearchSourceContract = defineRouteContract({ }, }) -/** - * One click on a Sim Search source: the workspace's Sim Search knowledge base - * and per-member connector exist afterwards, and the caller gets the link that - * connects their own account. - */ -export const connectSimSearchConnectorContract = defineRouteContract({ - method: 'POST', - path: '/api/knowledge/sim-search/connect', - body: connectSimSearchConnectorBodySchema, - response: { - mode: 'json', - schema: z.object({ - success: z.literal(true), - data: z.object({ - knowledgeBaseId: z.string(), - connectorId: z.string(), - url: z.string().url(), - }), - }), - }, -}) - export const deleteKnowledgeConnectorContract = defineRouteContract({ method: 'DELETE', path: '/api/knowledge/[id]/connectors/[connectorId]', diff --git a/apps/sim/lib/api/contracts/knowledge/mcp.ts b/apps/sim/lib/api/contracts/knowledge/mcp.ts index 285595da2ac..9a3d279959a 100644 --- a/apps/sim/lib/api/contracts/knowledge/mcp.ts +++ b/apps/sim/lib/api/contracts/knowledge/mcp.ts @@ -16,59 +16,11 @@ export const organizationKnowledgeMcpContract = defineRouteContract({ response: { mode: 'json', schema: mcpJsonRpcMessageSchema }, }) -const documentIdSchema = z.string().min(1, 'Document ID is required').max(255) - export const liveSearchMcpSchema = searchWorkspaceInputSchema.strict() export const readLiveDocumentMcpSchema = readDocumentInputSchema.strict() -export const searchMcpSchema = workspaceSearchFiltersSchema - .extend({ - query: z.string().trim().min(1, 'Search query is required').max(8192), - topK: z.number().int().min(1).max(50).default(10), - }) - .strict() - -export const readDocumentMcpSchema = z - .object({ - documentId: documentIdSchema.optional(), - url: z - .string() - .trim() - .url('Provide the original document URL') - .max(8192) - .refine((value) => { - if (!URL.canParse(value)) return false - const url = new URL(value) - return ['http:', 'https:'].includes(url.protocol) && !url.username && !url.password - }, 'Document URL must use HTTP or HTTPS without credentials') - .optional(), - limit: z.number().int().min(1).max(50).default(20), - offset: z.number().int().min(0).max(1_000_000).optional(), - aroundChunkIndex: z.number().int().min(0).max(1_000_000).optional(), - }) - .strict() - .superRefine((input, ctx) => { - if (Boolean(input.documentId) === Boolean(input.url)) { - ctx.addIssue({ - code: 'custom', - path: ['documentId'], - message: 'Provide either a document ID or its original URL', - }) - } - if (input.offset !== undefined && input.aroundChunkIndex !== undefined) { - ctx.addIssue({ - code: 'custom', - path: ['aroundChunkIndex'], - message: 'Use either an offset or a matching chunk index', - }) - } - }) - export const chatSearchMcpSchema = workspaceSearchFiltersSchema .extend({ query: z.string().trim().min(1, 'A question is required').max(8192), }) .strict() - -export type SearchMcpInput = z.input -export type ReadDocumentMcpInput = z.input diff --git a/apps/sim/lib/api/contracts/knowledge/personal-integrations.ts b/apps/sim/lib/api/contracts/knowledge/personal-integrations.ts index 1b2fa761bb7..16de7537189 100644 --- a/apps/sim/lib/api/contracts/knowledge/personal-integrations.ts +++ b/apps/sim/lib/api/contracts/knowledge/personal-integrations.ts @@ -8,8 +8,6 @@ export const personalSearchIntegrationSchema = z.object({ name: z.string().max(200), providerId: z.string().min(1).max(100), connectorType: z.string().min(1).max(100), - connectorId: z.string().min(1).max(200).optional(), - knowledgeBaseId: z.string().min(1).max(200).optional(), description: z.string().max(240), accounts: z .array( @@ -22,9 +20,6 @@ export const personalSearchIntegrationSchema = z.object({ ) .max(100), connectionStatus: z.enum(['connected', 'reconnect_needed', 'not_connected', 'unavailable']), - indexingStatus: z - .enum(['indexing', 'indexed', 'not_indexed', 'sync_failed', 'paused']) - .optional(), action: searchConnectionTargetSchema.nullable(), }) @@ -49,8 +44,6 @@ export const personalSearchIntegrationsQuerySchema = z.object({ completionId: z.string().uuid().optional(), organizationId: organizationIdSchema, connectorType: z.string().trim().min(1).max(100).optional(), - connectorId: z.string().min(1).max(200).optional(), - cursor: z.string().min(1).max(1024).optional(), }) export type PersonalSearchIntegrationsQuery = z.input @@ -65,10 +58,6 @@ export const connectPersonalSearchIntegrationBodySchema = z.object({ organizationId: organizationIdSchema, target: searchConnectionTargetSchema, oauthCompletionId: z.string().uuid(), - sourceConfig: z - .record(z.string().min(1).max(100), z.string().max(2000)) - .refine((config) => Object.keys(config).length <= 30, 'Too many source configuration fields') - .optional(), }) export type ConnectPersonalSearchIntegrationBody = z.input< typeof connectPersonalSearchIntegrationBodySchema @@ -82,8 +71,6 @@ export const connectPersonalSearchIntegrationContract = defineRouteContract({ schema: successResponseSchema( z.object({ url: z.string().url(), - connectorId: z.string().min(1).max(200).optional(), - knowledgeBaseId: z.string().min(1).max(200).optional(), }) ), }, diff --git a/apps/sim/lib/api/contracts/knowledge/personal-source-setup.test.ts b/apps/sim/lib/api/contracts/knowledge/personal-source-setup.test.ts deleted file mode 100644 index b0fcd945edf..00000000000 --- a/apps/sim/lib/api/contracts/knowledge/personal-source-setup.test.ts +++ /dev/null @@ -1,36 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { personalSourceSetupBodySchema } from '@/lib/api/contracts/knowledge/personal-source-setup' -import { executeSelectorBodySchema } from '@/lib/api/contracts/selectors/execute' - -const source = { - action: 'connect', - organizationId: 'organization-1', - connectorType: 'jira', - credentialId: 'account-1', - domain: 'example.atlassian.net', - keys: ['PROJECT'], -} - -describe('personal Search setup contracts', () => { - it('rejects arbitrary credential kinds, scope overrides and provider keys', () => { - for (const extra of [ - { workspaceId: 'workspace-1' }, - { selectorKey: 'jira.issues' }, - { accessMode: 'admin' }, - { connectorType: 'slack' }, - ]) { - expect(personalSourceSetupBodySchema.safeParse({ ...source, ...extra }).success).toBe(false) - } - }) - it('does not expose the trusted personal browsing marker through the generic selector API', () => { - expect( - executeSelectorBodySchema.safeParse({ - selectorKey: 'jira.projectKeys', - scope: { kind: 'organization', organizationId: 'organization-1' }, - context: { oauthCredential: 'account-1', domain: 'example.atlassian.net' }, - request: { kind: 'list' }, - personalSearchSetup: 'jira', - }).success - ).toBe(false) - }) -}) diff --git a/apps/sim/lib/api/contracts/knowledge/personal-source-setup.ts b/apps/sim/lib/api/contracts/knowledge/personal-source-setup.ts deleted file mode 100644 index c3e64253078..00000000000 --- a/apps/sim/lib/api/contracts/knowledge/personal-source-setup.ts +++ /dev/null @@ -1,86 +0,0 @@ -import { z } from 'zod' -import { defineRouteContract } from '@/lib/api/contracts' -import { successResponseSchema } from '@/lib/api/contracts/knowledge/shared' -import { organizationIdSchema } from '@/lib/api/contracts/primitives' -import { - executeSelectorResponseSchema, - selectorRequestSchema, -} from '@/lib/api/contracts/selectors/execute' -import { MAX_PERSONAL_SOURCE_SETUP_KEYS } from '@/lib/sim-search/personal-source-setup' - -const setupOwnerSchema = z.object({ - organizationId: organizationIdSchema, - connectorType: z.enum(['jira', 'confluence']), -}) -const setupCredentialSchema = setupOwnerSchema.extend({ - credentialId: z.string().min(1).max(128), - domain: z.string().trim().min(1, 'Enter your Atlassian site').max(253), -}) - -export const personalSourceSetupQuerySchema = setupOwnerSchema.extend({ - completionId: z.string().uuid().optional(), -}) -export type PersonalSourceSetupQuery = z.input - -export const personalSourceSetupAccountsSchema = z.object({ - accounts: z - .array( - z.object({ - id: z.string().min(1).max(128), - name: z.string().max(512), - provider: z.enum(['jira', 'confluence']), - type: z.literal('managed_oauth'), - scopes: z.array(z.string().max(200)).max(200), - }) - ) - .max(1000), - completedCredentialId: z.string().min(1).max(128).nullable(), -}) -export type PersonalSourceSetupAccounts = z.output - -export const personalSourceSetupBodySchema = z.discriminatedUnion('action', [ - setupOwnerSchema - .extend({ action: z.literal('authorize'), oauthCompletionId: z.string().uuid() }) - .strict(), - setupCredentialSchema - .extend({ - action: z.literal('connect'), - keys: z - .array(z.string().trim().min(1).max(255)) - .min(1, 'Select at least one project or space') - .max( - MAX_PERSONAL_SOURCE_SETUP_KEYS, - 'Choose no more than 1,000 projects or spaces per source' - ), - }) - .strict(), - setupCredentialSchema - .extend({ action: z.literal('options'), request: selectorRequestSchema }) - .strict(), -]) -export type PersonalSourceSetupBody = z.input - -export const personalSourceSetupResultSchema = z.discriminatedUnion('kind', [ - z.object({ kind: z.literal('authorization'), url: z.string().url() }), - z.object({ - kind: z.literal('connected'), - knowledgeBaseId: z.string().min(1).max(200), - connectorId: z.string().min(1).max(200), - }), - ...executeSelectorResponseSchema.options, -]) -export type PersonalSourceSetupResult = z.output - -export const listPersonalSourceSetupAccountsContract = defineRouteContract({ - method: 'GET', - path: '/api/knowledge/sim-search/personal-source-setup', - query: personalSourceSetupQuerySchema, - response: { mode: 'json', schema: successResponseSchema(personalSourceSetupAccountsSchema) }, -}) - -export const personalSourceSetupContract = defineRouteContract({ - method: 'POST', - path: '/api/knowledge/sim-search/personal-source-setup', - body: personalSourceSetupBodySchema, - response: { mode: 'json', schema: successResponseSchema(personalSourceSetupResultSchema) }, -}) diff --git a/apps/sim/lib/api/contracts/knowledge/search-stats.ts b/apps/sim/lib/api/contracts/knowledge/search-stats.ts deleted file mode 100644 index 63f30e2ec8f..00000000000 --- a/apps/sim/lib/api/contracts/knowledge/search-stats.ts +++ /dev/null @@ -1,76 +0,0 @@ -import { z } from 'zod' -import { organizationIdSchema } from '@/lib/api/contracts/primitives' -import { defineRouteContract } from '@/lib/api/contracts/types' -import { - getSearchStatsRangeError, - SEARCH_STATS_MAX_DAYS, - SEARCH_STATS_PEOPLE_LIMIT, - SEARCH_STATS_PERIODS, - SEARCH_STATS_SOURCE_LIMIT, - SEARCH_STATS_SURFACES, -} from '@/lib/knowledge/search/stats' - -export const organizationSearchStatsQuerySchema = z - .object({ - organizationId: organizationIdSchema, - period: z.enum(SEARCH_STATS_PERIODS).default('30d'), - surface: z.enum(SEARCH_STATS_SURFACES).optional(), - startDate: z.string().max(10).optional(), - endDate: z.string().max(10).optional(), - }) - .superRefine((query, context) => { - if (query.period === 'custom') { - const error = getSearchStatsRangeError(query) - if (error) context.addIssue({ code: 'custom', path: ['startDate'], message: error }) - } else if (query.startDate !== undefined || query.endDate !== undefined) { - context.addIssue({ - code: 'custom', - path: ['period'], - message: 'Choose Custom range to use start and end dates.', - }) - } - }) -export type OrganizationSearchStatsQuery = z.input - -const countSchema = z.number().int().nonnegative() -const sourceTypeSchema = z.string().min(1).max(100) -export const organizationSearchStatsSchema = z.object({ - start: z.string().datetime(), - end: z.string().datetime(), - totals: z.object({ - invocations: countSchema, - activePeople: countSchema, - results: countSchema, - }), - series: z - .array(z.object({ timestamp: z.string().datetime(), invocations: countSchema })) - .max(SEARCH_STATS_MAX_DAYS), - surfaces: z - .array(z.object({ surface: z.enum(SEARCH_STATS_SURFACES), invocations: countSchema })) - .max(7), - sources: z - .array(z.object({ sourceType: sourceTypeSchema, invocations: countSchema })) - .max(SEARCH_STATS_SOURCE_LIMIT), - people: z - .array( - z.object({ - userId: z.string().nullable(), - name: z.string().nullable(), - email: z.string().nullable(), - invocations: countSchema, - sourceTypes: z.array(sourceTypeSchema).max(SEARCH_STATS_SOURCE_LIMIT), - }) - ) - .max(SEARCH_STATS_PEOPLE_LIMIT), -}) -export type OrganizationSearchStats = z.output - -export const readOrganizationSearchStatsContract = defineRouteContract({ - method: 'GET', - path: '/api/knowledge/sim-search/stats', - query: organizationSearchStatsQuerySchema, - response: { - mode: 'json', - schema: z.object({ success: z.literal(true), data: organizationSearchStatsSchema }), - }, -}) diff --git a/apps/sim/lib/api/contracts/mothership-management-tools.test.ts b/apps/sim/lib/api/contracts/mothership-management-tools.test.ts index 07ee0bc30bf..e64a8040214 100644 --- a/apps/sim/lib/api/contracts/mothership-management-tools.test.ts +++ b/apps/sim/lib/api/contracts/mothership-management-tools.test.ts @@ -26,7 +26,7 @@ const examples = { { action: 'update', scope: 'account', section: 'profile', changes: { timezone: 'UTC' } }, ], search_sources: [ - { action: 'list', connectorType: 'google_drive', mine: true }, + { action: 'list', connectorType: 'google_drive' }, { action: 'get', connectorId: 'source-1' }, { action: 'providers' }, { action: 'setup', connectorType: 'google_drive', accessMode: 'admin' }, diff --git a/apps/sim/lib/api/contracts/mothership-search-sources.ts b/apps/sim/lib/api/contracts/mothership-search-sources.ts index 573cb1d3179..40bbdd611a0 100644 --- a/apps/sim/lib/api/contracts/mothership-search-sources.ts +++ b/apps/sim/lib/api/contracts/mothership-search-sources.ts @@ -17,7 +17,6 @@ export const organizationSearchSourcesInputSchema = z.discriminatedUnion('action cursor: z.string().min(1).max(1024).optional(), connectorType: connectorTypeSchema.optional(), search: z.string().trim().max(200).optional(), - mine: z.boolean().optional(), }), z.strictObject({ action: z.literal('get'), connectorId: z.string().min(1).max(255) }), z.strictObject({ action: z.literal('providers') }), diff --git a/apps/sim/lib/api/contracts/organization-accounts.ts b/apps/sim/lib/api/contracts/organization-accounts.ts index 8fd59a09d5b..e97824502f5 100644 --- a/apps/sim/lib/api/contracts/organization-accounts.ts +++ b/apps/sim/lib/api/contracts/organization-accounts.ts @@ -18,7 +18,6 @@ import { organizationIdSchema, workspaceIdSchema } from '@/lib/api/contracts/pri import { defineRouteContract } from '@/lib/api/contracts/types' import { ORGANIZATION_CREDENTIAL_TYPES } from '@/lib/credential-groups/credential-types' import { - ORGANIZATION_ACCOUNT_INDEXING_SOURCE_LIMIT, ORGANIZATION_ACCOUNT_WORKSPACE_LIMIT, ORGANIZATION_VIEWER_ACCOUNT_LIMIT, } from '@/lib/credential-groups/limits' @@ -140,31 +139,6 @@ export type OrganizationAccountsSettings = z.output< typeof getOrganizationAccountsContract.response.schema > -export const updateOrganizationAccountIndexingBodySchema = z - .object({ - optionId: z.string().min(1, 'Provider option is required').max(128), - enabled: z.boolean(), - }) - .strict() - -export const updateOrganizationAccountIndexingContract = defineRouteContract({ - method: 'PUT', - path: '/api/organizations/[id]/connected-accounts/indexing', - params: organizationAccountsParamsSchema, - body: updateOrganizationAccountIndexingBodySchema, - response: { - mode: 'json', - schema: z.object({ - enabled: z.boolean(), - knowledgeBaseIds: z - .array(z.string().min(1).max(128)) - .max(ORGANIZATION_ACCOUNT_INDEXING_SOURCE_LIMIT), - }), - }, -}) -export type UpdateOrganizationAccountIndexingBody = z.input< - NonNullable -> export type EnsureOrganizationAccountsBody = z.input< NonNullable > diff --git a/apps/sim/lib/api/contracts/workspaces.ts b/apps/sim/lib/api/contracts/workspaces.ts index 1ddf9880495..b7f03ef2826 100644 --- a/apps/sim/lib/api/contracts/workspaces.ts +++ b/apps/sim/lib/api/contracts/workspaces.ts @@ -214,7 +214,6 @@ export type WorkspaceOwnerBilling = z.output * these only off-hosted, where no subscription plan exists to decide entitlement. */ export const deploymentFeaturesSchema = z.object({ - liveEnterpriseSearch: z.boolean().optional(), accessControl: z.boolean(), auditLogs: z.boolean(), customBlocks: z.boolean(), diff --git a/apps/sim/lib/core/config/deployment-shape.ts b/apps/sim/lib/core/config/deployment-shape.ts index 0d31ab553d1..5d415b26020 100644 --- a/apps/sim/lib/core/config/deployment-shape.ts +++ b/apps/sim/lib/core/config/deployment-shape.ts @@ -13,7 +13,6 @@ import { isDataRetentionEnabled, isHosted, isInboxEnabled, - isLiveEnterpriseSearchEnabled, isSandboxesEnabled, isScimEnabled, isSessionPoliciesEnabled, @@ -89,7 +88,6 @@ export function resolveDeploymentShape(): DeploymentShape { azureConfigured: isAzureConfigured, cohereConfigured: isCohereConfigured, features: { - liveEnterpriseSearch: isLiveEnterpriseSearchEnabled, accessControl: isAccessControlEnabled, auditLogs: isAuditLogsEnabled, customBlocks: isCustomBlocksEnabled, diff --git a/apps/sim/lib/core/config/env-flags.ts b/apps/sim/lib/core/config/env-flags.ts index c44c19adb93..9ad4f597f0d 100644 --- a/apps/sim/lib/core/config/env-flags.ts +++ b/apps/sim/lib/core/config/env-flags.ts @@ -722,7 +722,3 @@ export function getCostMultiplier(): number { } /** Backend selector. Kept independent of enterprise entitlement overrides. */ -const liveEnterpriseSearchSetting = - typeof window === 'undefined' ? env.SIM_SEARCH_LIVE : getEnv('NEXT_PUBLIC_SIM_SEARCH_LIVE') -export const isLiveEnterpriseSearchEnabled = - liveEnterpriseSearchSetting === undefined || isTruthy(liveEnterpriseSearchSetting) diff --git a/apps/sim/lib/core/config/env.ts b/apps/sim/lib/core/config/env.ts index eb3975a08f8..7f232d483bb 100644 --- a/apps/sim/lib/core/config/env.ts +++ b/apps/sim/lib/core/config/env.ts @@ -600,7 +600,6 @@ export const env = createEnv({ MSHIP_PLAN_MODE: z.boolean().optional(), DASHBOARDS: z.boolean().optional(), MSHIP_MODEL_SELECTOR: z.boolean().optional(), - SIM_SEARCH_LIVE: z.boolean().optional(), // Query connected providers directly; false preserves indexed search INBOX_ENABLED: z.boolean().optional(), // Enable inbox (Sim Mailer) on self-hosted (bypasses hosted requirements) SANDBOXES_ENABLED: z.boolean().optional(), // Enable custom sandboxes on self-hosted (bypasses hosted requirements) @@ -766,7 +765,6 @@ export const env = createEnv({ NEXT_PUBLIC_ORGANIZATIONS_ENABLED: z.boolean().optional(), // Enable organizations on self-hosted (bypasses plan requirements) NEXT_PUBLIC_DISABLE_INVITATIONS: z.boolean().optional(), // Disable workspace invitations globally (for self-hosted deployments) NEXT_PUBLIC_DISABLE_PUBLIC_API: z.boolean().optional(), // Disable public API access UI toggle globally - NEXT_PUBLIC_SIM_SEARCH_LIVE: z.boolean().optional(), NEXT_PUBLIC_INBOX_ENABLED: z.boolean().optional(), // Enable inbox (Sim Mailer) on self-hosted NEXT_PUBLIC_CHAT_DISABLED: z.boolean().optional(), // Hide the Chat module (Chat is shown when unset) NEXT_PUBLIC_STATUS_NOTICE_PREVIEW: z.boolean().optional(), // Force the sidebar service-status notice into its critical preview state @@ -814,7 +812,6 @@ export const env = createEnv({ NEXT_PUBLIC_ORGANIZATIONS_ENABLED: process.env.NEXT_PUBLIC_ORGANIZATIONS_ENABLED, NEXT_PUBLIC_DISABLE_INVITATIONS: process.env.NEXT_PUBLIC_DISABLE_INVITATIONS, NEXT_PUBLIC_DISABLE_PUBLIC_API: process.env.NEXT_PUBLIC_DISABLE_PUBLIC_API, - NEXT_PUBLIC_SIM_SEARCH_LIVE: process.env.NEXT_PUBLIC_SIM_SEARCH_LIVE, NEXT_PUBLIC_INBOX_ENABLED: process.env.NEXT_PUBLIC_INBOX_ENABLED, NEXT_PUBLIC_CHAT_DISABLED: process.env.NEXT_PUBLIC_CHAT_DISABLED, NEXT_PUBLIC_STATUS_NOTICE_PREVIEW: process.env.NEXT_PUBLIC_STATUS_NOTICE_PREVIEW, diff --git a/apps/sim/lib/credential-groups/README.md b/apps/sim/lib/credential-groups/README.md index 5c46c85faf0..4f9eba0719c 100644 --- a/apps/sim/lib/credential-groups/README.md +++ b/apps/sim/lib/credential-groups/README.md @@ -37,7 +37,7 @@ The Providers tab lists only added providers. **Add provider** opens a searchabl - Databricks: **Add** collects and validates the organization's tenant MCP URL and registered OAuth client before creating an enabled provider in one transaction. Cancelling leaves nothing added. Organization owners and admins create it through `POST /api/organizations/[id]/connected-accounts/mcp-providers` and read or edit settings through `GET` / `PUT /api/organizations/[id]/connected-accounts/databricks`; no workspace configuration is used. Unfinished entries from the earlier flow appear in the Add catalog until their configuration is saved. The form never reads stored client secrets, and leaving the secret blank when editing preserves it. Endpoint/client identity changes invalidate affected grants and pending attempts, requiring people to reconnect. - Slack personal OAuth: the org admin supplies App ID, Slack workspace ID, client ID, and client secret, then verifies authorization. The org has one configured app/workspace. Existing workspace bots and their triggers remain separate. -Search availability and mode are managed through **Organization settings → Sources**. Member mode has no resource filters; service mode uses the configured source’s resource boundary. Live search/read adapters are registered under `lib/sim-search/live/`; they resolve only the acting member’s current grants. OAuth completion may invoke the shared dispatch helper, but live Search sources are rejected by queue and worker guards before indexing. Ordinary workspace KB sources and the explicit `SIM_SEARCH_LIVE=false` backend retain ingestion behavior. +Search availability and mode are managed through **Organization settings → Sources**. Member mode has no resource filters; service mode uses the configured source’s resource boundary. Live search/read adapters are registered under `lib/sim-search/live/`; they resolve only the acting member’s current grants. OAuth completion may invoke the shared dispatch helper, but live Search sources are rejected by queue and worker guards before indexing. Ordinary workspace KB sources retain ingestion behavior. Search requires both `CREDENTIAL_GROUPS` and `KNOWLEDGE_MEMBER_ACCESS` locally; Credential Groups alone requires only its own flag. Hosted deployments additionally enforce the routed org's feature rules and Enterprise availability. Owners and admins manage Search sources; existing Knowledge permission-group rules still apply. Managed MCP account connections remain available for live tool calls only and have no indexing switch. Separate API-key KB connectors for Fireflies, Granola, and Databricks do not consume these managed MCP connections. @@ -48,7 +48,7 @@ Search requires both `CREDENTIAL_GROUPS` and `KNOWLEDGE_MEMBER_ACCESS` locally; | Credential Groups settings page | `credential-groups` enabled, independently of Search | | Provider setup APIs, personal contributions, workspace access to the pool | `credential-groups` | | Organization Home/Assistant and chat pages, Sources, member Integrations, Search MCP settings | `credential-groups` and `knowledge-member-access` | -| Legacy Search content/member sync and persisted directory maintenance | Search features above and explicit `SIM_SEARCH_LIVE=false` | + | Organization Search MCP endpoint and organization knowledge search through internal/public APIs or trusted tools | `credential-groups` and `knowledge-member-access`, checked after current authorization | The Search gate uses the persisted knowledge base owner or the authenticated route's target organization. User, platform-admin, and workspace targeting cannot opt a different organization into Search. Disabled organizations receive `403 Search is not enabled for this organization` before index lookup or model execution; hiding navigation is not the authorization boundary. Organization Home, Search, and chat URLs open full settings in the viewer's most recent accessible workspace when Search is disabled. Default app entry uses that same destination, and Home, Integrations, chat history, and Assistant loading UI are hidden. Connected accounts settings and Workspaces remain available. Settings and source-setup URLs also enforce their gates. Ordinary workspace knowledge search keeps its existing behavior. The legacy indexed surface retains its pause controls when that backend is selected; live Sources does not expose indexing controls. diff --git a/apps/sim/lib/credential-groups/application/organization-account-indexing.test.ts b/apps/sim/lib/credential-groups/application/organization-account-indexing.test.ts deleted file mode 100644 index ea2d7468736..00000000000 --- a/apps/sim/lib/credential-groups/application/organization-account-indexing.test.ts +++ /dev/null @@ -1,134 +0,0 @@ -import { auditMock, auditMockFns, queueTableRows, resetDbChainMock, schemaMock } from '@sim/testing' -import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' -import { - credentialGroupsAvailabilityMock, - credentialGroupsAvailabilityMockFns, -} from '@sim/testing/mocks/credential-groups-availability.mock' -import { - credentialGroupsCredentialsMock, - credentialGroupsCredentialsMockFns, -} from '@sim/testing/mocks/credential-groups-credentials.mock' -import { credentialGroupsOrganizationSetupMock } from '@sim/testing/mocks/credential-groups-organization-setup.mock' -import { credentialGroupsSelfEnrollmentMock } from '@sim/testing/mocks/credential-groups-self-enrollment.mock' -import { credentialGroupsServiceMock } from '@sim/testing/mocks/credential-groups-service.mock' -import { - knowledgeAvailabilityMock, - knowledgeAvailabilityMockFns, -} from '@sim/testing/mocks/knowledge-availability.mock' -import { - knowledgeMemberQueueMock, - knowledgeMemberQueueMockFns, -} from '@sim/testing/mocks/knowledge-member-queue.mock' -import { permissionGroupsResolveMock } from '@sim/testing/mocks/permission-groups-resolve.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -const hoisted = vi.hoisted(() => ({ - setIndexing: vi.fn(), -})) -vi.mock('@sim/audit', () => auditMock) -vi.mock('@/lib/credential-groups/scoped-availability', () => credentialGroupsAvailabilityMock) -vi.mock('@/lib/credential-groups/credentials', () => credentialGroupsCredentialsMock) -vi.mock('@/lib/credential-groups/organization-setup', () => credentialGroupsOrganizationSetupMock) -vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) -vi.mock('@/lib/credential-groups/service', () => credentialGroupsServiceMock) -vi.mock('@/lib/credential-groups/provider-availability', () => ({ - listConfiguredCredentialGroupProviders: vi.fn(), -})) -vi.mock('@/lib/credential-groups/self-enrollment', () => credentialGroupsSelfEnrollmentMock) -vi.mock('@/lib/credential-groups/managed-mcp-service', () => ({ - ManagedMcpConnectorError: class extends Error {}, -})) -vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) -vi.mock('@/lib/knowledge/connectors/organization-account-indexing', () => ({ - setOrganizationAccountIndexing: hoisted.setIndexing, -})) -vi.mock('@/lib/knowledge/connectors/member-queue', () => knowledgeMemberQueueMock) - -import { updateOrganizationAccountIndexing } from '@/lib/credential-groups/application/organization-account-indexing' - -const mocks = { - ...hoisted, - available: credentialGroupsAvailabilityMockFns.mockIsScopedCredentialGroupsAvailable, - group: credentialGroupsCredentialsMockFns.mockLoadScopedAccountsCredentialListContext, - dispatch: knowledgeMemberQueueMockFns.mockDispatchMemberSyncsForCredentialOption, -} - -const feature = knowledgeAvailabilityMockFns.mockRequireKnowledgeMemberAccessAvailable -const principal = createSessionPrincipal({ userId: 'admin-1' }) -const input = { organizationId: 'org-1', optionId: 'option-1', enabled: true } - -describe('organization account indexing authorization', () => { - beforeEach(() => { - resetDbChainMock() - mocks.available.mockResolvedValue(true) - mocks.group.mockResolvedValue({ credentialGroupId: 'group-1' }) - feature.mockResolvedValue(undefined) - mocks.setIndexing.mockResolvedValue({ - enabled: true, - changed: true, - providerName: 'Gmail', - knowledgeBaseIds: ['kb-1'], - }) - }) - it.each(['member', null])('denies a %s before reading account data', async (role) => { - queueTableRows(schemaMock.member, role ? [{ role }] : []) - await expect(updateOrganizationAccountIndexing.execute({ principal, input })).rejects.toThrow() - expect(mocks.group).not.toHaveBeenCalled() - expect(mocks.setIndexing).not.toHaveBeenCalled() - }) - it('allows an org admin and dispatches only that organization option', async () => { - queueTableRows(schemaMock.member, [{ role: 'admin' }]) - await updateOrganizationAccountIndexing.execute({ principal, input }) - expect(mocks.setIndexing).toHaveBeenCalledWith({ ...input, credentialGroupId: 'group-1' }) - expect(mocks.dispatch).toHaveBeenCalledWith({ - organizationId: 'org-1', - credentialGroupOptionId: 'option-1', - }) - expect(auditMockFns.mockRecordAudit).toHaveBeenCalledWith( - expect.objectContaining({ - actorId: 'admin-1', - metadata: { - organizationId: 'org-1', - operation: 'organization_accounts.indexing.update', - actor: { kind: 'session', userId: 'admin-1' }, - }, - }) - ) - }) - it('requires indexing availability to enable a source', async () => { - queueTableRows(schemaMock.member, [{ role: 'admin' }]) - feature.mockRejectedValue(new Error('Search is not enabled')) - await expect(updateOrganizationAccountIndexing.execute({ principal, input })).rejects.toThrow( - 'Search is not enabled' - ) - expect(mocks.setIndexing).not.toHaveBeenCalled() - expect(mocks.dispatch).not.toHaveBeenCalled() - }) - it('allows pausing when the Search feature is off and does not dispatch', async () => { - queueTableRows(schemaMock.member, [{ role: 'admin' }]) - mocks.setIndexing.mockResolvedValue({ - enabled: false, - changed: true, - providerName: 'Gmail', - knowledgeBaseIds: ['kb-1'], - }) - await updateOrganizationAccountIndexing.execute({ - principal, - input: { ...input, enabled: false }, - }) - expect(feature).not.toHaveBeenCalled() - expect(mocks.dispatch).not.toHaveBeenCalled() - }) - it('does not audit or redispatch an unchanged setting', async () => { - queueTableRows(schemaMock.member, [{ role: 'admin' }]) - mocks.setIndexing.mockResolvedValue({ - enabled: true, - changed: false, - providerName: 'Gmail', - knowledgeBaseIds: ['kb-1'], - }) - await updateOrganizationAccountIndexing.execute({ principal, input }) - expect(auditMockFns.mockRecordAudit).not.toHaveBeenCalled() - expect(mocks.dispatch).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/credential-groups/application/organization-account-indexing.ts b/apps/sim/lib/credential-groups/application/organization-account-indexing.ts deleted file mode 100644 index fb346f46e83..00000000000 --- a/apps/sim/lib/credential-groups/application/organization-account-indexing.ts +++ /dev/null @@ -1,62 +0,0 @@ -import type { OrganizationMembershipContext } from '@/lib/core/application/organization-authorization' -import { defineOrganizationOperation } from '@/lib/core/application/organization-operation' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { defineOrganizationAccountsUseCase } from '@/lib/credential-groups/application/organization-accounts' -import { loadScopedAccountsCredentialListContext } from '@/lib/credential-groups/credentials' -import { requireKnowledgeMemberAccessAvailable } from '@/lib/knowledge/access/availability' -import { dispatchMemberSyncsForCredentialOption } from '@/lib/knowledge/connectors/member-queue' -import { setOrganizationAccountIndexing } from '@/lib/knowledge/connectors/organization-account-indexing' - -export const updateOrganizationAccountIndexingOperation = defineOrganizationOperation({ - id: 'organization_accounts.indexing.update', - minimumRole: 'admin', - principalKinds: ['session', 'organization_delegated'], - delegationAudience: 'sim:settings', - delegatedServices: ['copilot'], - capability: 'knowledge.use', -}) - -export const updateOrganizationAccountIndexing = defineOrganizationAccountsUseCase({ - operation: updateOrganizationAccountIndexingOperation, - async execute({ - input, - context, - }: { - input: { organizationId: string; optionId: string; enabled: boolean } - context: OrganizationMembershipContext - }) { - if (input.enabled) - await requireKnowledgeMemberAccessAvailable({ organizationId: context.organizationId }) - const group = await loadScopedAccountsCredentialListContext({ - kind: 'organization', - organizationId: context.organizationId, - }) - if (!group) - throw new OrchestrationError( - 'not_found', - 'Organization connected accounts are not configured' - ) - const result = await setOrganizationAccountIndexing({ - organizationId: context.organizationId, - credentialGroupId: group.credentialGroupId, - optionId: input.optionId, - enabled: input.enabled, - }) - return { ...result, credentialGroupId: group.credentialGroupId, optionId: input.optionId } - }, - projectAudit: (result) => - result.changed - ? { - resourceId: result.credentialGroupId, - resourceName: 'Connected accounts', - description: `${result.enabled ? 'Enabled' : 'Paused'} ${result.providerName} indexing`, - } - : null, - async afterSuccess({ context, result }) { - if (result.enabled && result.changed) - await dispatchMemberSyncsForCredentialOption({ - organizationId: context.organizationId, - credentialGroupOptionId: result.optionId, - }) - }, -}) diff --git a/apps/sim/lib/credential-groups/application/organization-settings-delegation.test.ts b/apps/sim/lib/credential-groups/application/organization-settings-delegation.test.ts index 2ea84440626..76abae115ad 100644 --- a/apps/sim/lib/credential-groups/application/organization-settings-delegation.test.ts +++ b/apps/sim/lib/credential-groups/application/organization-settings-delegation.test.ts @@ -11,7 +11,6 @@ vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveM import { authorizeOrganizationOperation } from '@/lib/core/application/organization-authorization' import { organizationAccountAccessOperations } from '@/lib/credential-groups/application/organization-access' -import { updateOrganizationAccountIndexingOperation } from '@/lib/credential-groups/application/organization-account-indexing' import { organizationAccountManagementOperations } from '@/lib/credential-groups/application/organization-account-management' import { organizationAccountOperations } from '@/lib/credential-groups/application/organization-accounts' @@ -19,7 +18,6 @@ const operations = [ ...Object.values(organizationAccountOperations), ...Object.values(organizationAccountAccessOperations), ...Object.values(organizationAccountManagementOperations), - updateOrganizationAccountIndexingOperation, ] const principal: OrganizationDelegatedPrincipal = { kind: 'organization_delegated', diff --git a/apps/sim/lib/credential-groups/self-enrollment-oauth.ts b/apps/sim/lib/credential-groups/self-enrollment-oauth.ts index 79e474798c0..fe3265b7306 100644 --- a/apps/sim/lib/credential-groups/self-enrollment-oauth.ts +++ b/apps/sim/lib/credential-groups/self-enrollment-oauth.ts @@ -8,7 +8,7 @@ import { startCredentialGroupOAuth } from '@/lib/credential-groups/oauth' import type { CredentialGroupConnectionIntent } from '@/lib/credential-groups/oauth-intent' import { createViewerCredentialGroupEnrollment } from '@/lib/credential-groups/self-enrollment' -/** Starts a user-initiated connection with the same scoped receipt for live and indexed Search. */ +/** Starts a user-initiated connection with a scoped receipt for Search and knowledge-base enrollment. */ export async function startViewerCredentialGroupOAuth(input: { userId: string organizationId?: string diff --git a/apps/sim/lib/credential-groups/service.ts b/apps/sim/lib/credential-groups/service.ts index e92b6684f6e..72b09db19f1 100644 --- a/apps/sim/lib/credential-groups/service.ts +++ b/apps/sim/lib/credential-groups/service.ts @@ -25,7 +25,7 @@ import { credentialGroupScopePolicyVersion } from '@/lib/credential-groups/provi import { decryptCredentialGroupProviderConfiguration } from '@/lib/credential-groups/provider-configuration' import { getCredentialGroupProviderAdapter } from '@/lib/credential-groups/provider-registry' import { - type CredentialGroupStandardOAuthProvider, + type CredentialGroupProvider, isCredentialGroupProvider, } from '@/lib/credential-groups/providers' import { credentialGroupScope } from '@/lib/credential-groups/scope' @@ -87,7 +87,13 @@ function scopesEqual(left: string[], right: string[]): boolean { async function buildOption( scope: ResourceScope, - option: CredentialGroupOptionInput, + option: { + provider: CredentialGroupProvider + label: string + required: boolean + slackBotCredentialId?: string + requiredScopes?: string[] + }, credentialGroupId?: string, executor: DbOrTx = db ): Promise { @@ -395,11 +401,11 @@ export async function ensureWorkspaceAccountsGroup( } } -/** Adds a provider explicitly selected by an organization administrator, preserving all grants. */ +/** Adds a provider or extends its required consent during an explicit administrator action. */ export async function addOrganizationAccountProvider( organizationId: string, userId: string, - option: { provider: CredentialGroupStandardOAuthProvider; label: string }, + option: { provider: CredentialGroupProvider; label: string; requiredScopes?: string[] }, executor: DbTransaction ): Promise<{ groupId: string; changed: boolean }> { const scope = { kind: 'organization', organizationId } as const @@ -418,11 +424,37 @@ export async function addOrganizationAccountProvider( `Connected accounts contains duplicate ${option.label} settings` ) if (matching[0]) { - if (matching[0].status !== 'active') + const current = matching[0] + if (current.status !== 'active') throw new OrchestrationError( 'validation', `Enable ${option.label} in Connected accounts first` ) + if (option.requiredScopes) { + const previousScopes = + current.provider === 'slack' + ? resolveSlackManagedUserScopes(current.requiredScopes) + : current.requiredScopes + const requiredScopes = [...new Set([...previousScopes, ...option.requiredScopes])] + const scopeVersion = credentialGroupScopePolicyVersion(requiredScopes) + if (!scopesEqual(requiredScopes, previousScopes) || scopeVersion !== current.scopeVersion) { + const [updated] = await executor + .update(credentialGroup) + .set({ + options: existing.options.map((entry) => + entry.id === current.id ? { ...entry, requiredScopes, scopeVersion } : entry + ), + updatedAt: new Date(), + }) + .where( + and(eq(credentialGroup.id, group.id), resourceScopeCondition(credentialGroup, scope)) + ) + .returning({ id: credentialGroup.id }) + if (!updated) throw new Error('Connected accounts policy update returned no row') + await invalidateOptionGrants(executor, group.id, [current.id]) + return { groupId: group.id, changed: true } + } + } return { groupId: group.id, changed: group.created } } if ( @@ -449,6 +481,27 @@ export async function addOrganizationAccountProvider( return { groupId: group.id, changed: true } } +async function invalidateOptionGrants( + executor: DbTransaction, + groupId: string, + optionIds: string[] +) { + const enrollmentIds = executor + .select({ id: credentialGroupEnrollment.id }) + .from(credentialGroupEnrollment) + .where(eq(credentialGroupEnrollment.credentialGroupId, groupId)) + await executor + .update(credential) + .set({ managedOauthStatus: 'needs_reauth', updatedAt: new Date() }) + .where( + and( + eq(credential.type, 'managed_oauth'), + inArray(credential.credentialGroupEnrollmentId, enrollmentIds), + inArray(credential.credentialGroupOptionId, optionIds) + ) + ) +} + /** * Refuses to remove account options while a knowledge * connector syncs per member through one of them: the connector would be left @@ -570,20 +623,7 @@ export async function updateCredentialGroup( if (!updated) throw new Error('Credential group update returned no row') if (invalidatedOptionIds.length > 0) { - const enrollmentIds = tx - .select({ id: credentialGroupEnrollment.id }) - .from(credentialGroupEnrollment) - .where(eq(credentialGroupEnrollment.credentialGroupId, groupId)) - await tx - .update(credential) - .set({ managedOauthStatus: 'needs_reauth', updatedAt: new Date() }) - .where( - and( - eq(credential.type, 'managed_oauth'), - inArray(credential.credentialGroupEnrollmentId, enrollmentIds), - inArray(credential.credentialGroupOptionId, invalidatedOptionIds) - ) - ) + await invalidateOptionGrants(tx, groupId, invalidatedOptionIds) } return toCredentialGroup(updated, await listLinkedMcpServers(updated.id, tx)) }) diff --git a/apps/sim/lib/credential-groups/slack-managed-user-scopes.ts b/apps/sim/lib/credential-groups/slack-managed-user-scopes.ts index 4be542eefdf..1193daefb55 100644 --- a/apps/sim/lib/credential-groups/slack-managed-user-scopes.ts +++ b/apps/sim/lib/credential-groups/slack-managed-user-scopes.ts @@ -1,3 +1,5 @@ +import { SLACK_RTS_USER_SCOPES } from '@/lib/sim-search/live/scopes' + /** * User-token policy requested and verified by Credential Group Slack OAuth. * This is independent of the custom bot manifest and its configuration UI. @@ -48,12 +50,13 @@ export const SLACK_CHANNEL_READ_SCOPES = [ export const SLACK_DM_READ_SCOPES = ['im:history', 'im:read', 'mpim:history', 'mpim:read'] as const -/** The shared organization app grants member access for channel and DM indexing. */ +/** Explicit Search setup grants channel, DM, and live retrieval permissions together. */ export const SLACK_SEARCH_USER_SCOPES = [ ...SLACK_CHANNEL_READ_SCOPES, ...SLACK_DM_READ_SCOPES, 'users:read', 'users:read.email', + ...SLACK_RTS_USER_SCOPES, ] as const /** Existing workflow options retain their scope policy; every user grant must attest identity. */ diff --git a/apps/sim/lib/credential-groups/slack-managed-users.test.ts b/apps/sim/lib/credential-groups/slack-managed-users.test.ts index 4a3e89d887f..ce9e8301da7 100644 --- a/apps/sim/lib/credential-groups/slack-managed-users.test.ts +++ b/apps/sim/lib/credential-groups/slack-managed-users.test.ts @@ -199,6 +199,7 @@ describe('Slack managed-user authorization', () => { existingScopes: undefined, requestedScopes: SLACK_MANAGED_USER_SCOPES, scopes: SLACK_SEARCH_USER_SCOPES, + upgradesSearchPolicy: false, }, { name: 'legacy workflow pool without explicit scopes', @@ -206,6 +207,7 @@ describe('Slack managed-user authorization', () => { existingScopes: undefined, requestedScopes: SLACK_SEARCH_USER_SCOPES, scopes: SLACK_MANAGED_USER_SCOPES, + upgradesSearchPolicy: false, }, { name: 'existing Search pool', @@ -213,10 +215,30 @@ describe('Slack managed-user authorization', () => { existingScopes: SLACK_SEARCH_USER_SCOPES, requestedScopes: SLACK_MANAGED_USER_SCOPES, scopes: SLACK_SEARCH_USER_SCOPES, + upgradesSearchPolicy: false, + }, + { + name: 'legacy Search pool', + existing: true, + existingScopes: [ + 'channels:history', + 'channels:read', + 'groups:history', + 'groups:read', + 'im:history', + 'im:read', + 'mpim:history', + 'mpim:read', + 'users:read', + 'users:read.email', + ], + requestedScopes: SLACK_MANAGED_USER_SCOPES, + scopes: SLACK_SEARCH_USER_SCOPES, + upgradesSearchPolicy: true, }, ])( - 'verifies an organization $name without replacing its scope policy or disconnecting members', - async ({ existing, existingScopes, requestedScopes, scopes }) => { + 'verifies an organization $name with its explicit Search or workflow scope policy', + async ({ existing, existingScopes, requestedScopes, scopes, upgradesSearchPolicy }) => { const updatedAt = new Date('2026-08-12T00:00:00Z') const group = { id: 'group-1', @@ -232,7 +254,9 @@ describe('Slack managed-user authorization', () => { required: true, authorizationAppId: 'slack:A123:T123', requiredScopes: existingScopes, - scopeVersion: credentialGroupScopePolicyVersion([...scopes]), + scopeVersion: credentialGroupScopePolicyVersion([ + ...(existingScopes ?? SLACK_MANAGED_USER_SCOPES), + ]), }, ] : [], @@ -264,7 +288,6 @@ describe('Slack managed-user authorization', () => { const attempt = await consumeSlackManagedUsersAttempt(created.state) expect(attempt?.requiredScopes).toEqual([...scopes]) if (!attempt) throw new Error('Expected an organization authorization attempt') - if (!existing) expect(attempt.requiredScopes).toHaveLength(10) queueTableRows(schemaMock.slackApp, [app]) queueTableRows(schemaMock.credentialGroup, [group]) @@ -301,9 +324,10 @@ describe('Slack managed-user authorization', () => { options: [expect.objectContaining({ requiredScopes: [...scopes] })], }) ) - expect(dbChainMockFns.set).not.toHaveBeenCalledWith( - expect.objectContaining({ managedOauthStatus: 'needs_reauth' }) - ) + if (!upgradesSearchPolicy) + expect(dbChainMockFns.set).not.toHaveBeenCalledWith( + expect.objectContaining({ managedOauthStatus: 'needs_reauth' }) + ) } ) diff --git a/apps/sim/lib/credential-groups/slack-managed-users.ts b/apps/sim/lib/credential-groups/slack-managed-users.ts index ce7982f217f..e7a12675a1e 100644 --- a/apps/sim/lib/credential-groups/slack-managed-users.ts +++ b/apps/sim/lib/credential-groups/slack-managed-users.ts @@ -24,6 +24,8 @@ import { } from '@/lib/credential-groups/provider-configuration' import { resolveSlackManagedUserScopes, + SLACK_CHANNEL_READ_SCOPES, + SLACK_DM_READ_SCOPES, SLACK_MANAGED_USER_CONFIGURATION_CALLBACK_PATH, SLACK_SEARCH_USER_SCOPES, } from '@/lib/credential-groups/slack-managed-user-scopes' @@ -532,8 +534,19 @@ export async function createSlackManagedUsersAttempt(params: { clientId = app.clientId clientSecret = app.clientSecret appRevision = app.revision + const retiredSearchScopes = new Set([ + ...SLACK_CHANNEL_READ_SCOPES, + ...SLACK_DM_READ_SCOPES, + 'users:read', + 'users:read.email', + ]) + const upgradesSearchPolicy = + existingOption?.requiredScopes?.length === retiredSearchScopes.size && + existingOption.requiredScopes.every((scope) => retiredSearchScopes.has(scope)) requiredScopes = resolveSlackManagedUserScopes( - existingOption ? existingOption.requiredScopes : SLACK_SEARCH_USER_SCOPES + !existingOption || upgradesSearchPolicy + ? SLACK_SEARCH_USER_SCOPES + : existingOption.requiredScopes ) } else { if (!params.slackBotCredentialId) diff --git a/apps/sim/lib/credential-groups/slack-provider.test.ts b/apps/sim/lib/credential-groups/slack-provider.test.ts index 51da5603c64..5acbe0394cf 100644 --- a/apps/sim/lib/credential-groups/slack-provider.test.ts +++ b/apps/sim/lib/credential-groups/slack-provider.test.ts @@ -1,5 +1,4 @@ import type { CredentialGroupOptionConfig } from '@sim/db/schema' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing' import { resetUrlsMock, urlsMockFns } from '@sim/testing/mocks/urls.mock' import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' @@ -34,14 +33,13 @@ afterAll(resetUrlsMock) describe('Slack member scope policy', () => { beforeEach(() => { - resetEnvFlagsMock() mocks.configuration.mockResolvedValue({ slackBotCredentialId: 'bot-1', clientId: 'client', clientSecret: 'secret', appId: 'A1', teamId: 'T1', - scopes: [...SLACK_MANAGED_USER_SCOPES], + scopes: [...new Set([...SLACK_MANAGED_USER_SCOPES, ...SLACK_SEARCH_USER_SCOPES])], }) mocks.exchange.mockResolvedValue({ appId: 'A1', @@ -80,45 +78,22 @@ describe('Slack member scope policy', () => { } } - it('requests RTS consent only with live search enabled while retaining the stored policy', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) - const current = context(SLACK_SEARCH_USER_SCOPES) - const policy = await adapter.getPolicy(current.option, { - workspaceId: current.workspaceId, - credentialGroupId: current.credentialGroupId, - }) - const authorization = await adapter.prepareAuthorization(current, policy) - const url = new URL( - await authorization.buildAuthorizationUrl({ state: 'state', nonce: 'nonce' }) - ) - expect(url.searchParams.get('user_scope')?.split(',')).toEqual( - expect.arrayContaining([ - 'search:read.public', - 'search:read.private', - 'search:read.im', - 'search:read.mpim', - 'search:read.files', - 'files:read', - ]) - ) - expect(policy.requiredScopes).toEqual([...SLACK_SEARCH_USER_SCOPES]) - }) - - it('uses the option policy for enrollment instead of widening it', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - const scopes = SLACK_MANAGED_USER_SCOPES - const current = context(scopes) - const policy = await adapter.getPolicy(current.option, { - workspaceId: current.workspaceId, - credentialGroupId: current.credentialGroupId, - }) - expect(policy.requiredScopes).toEqual([...scopes]) - const authorization = await adapter.prepareAuthorization(current, policy) - const url = new URL( - await authorization.buildAuthorizationUrl({ state: 'state', nonce: 'nonce' }) - ) - expect(url.searchParams.get('user_scope')?.split(',')).toEqual([...scopes]) - }) + it.each([{ scopes: SLACK_SEARCH_USER_SCOPES }, { scopes: SLACK_MANAGED_USER_SCOPES }])( + 'requests exactly the configured permissions without broadening workflow consent', + async ({ scopes }) => { + const current = context(scopes) + const policy = await adapter.getPolicy(current.option, { + workspaceId: current.workspaceId, + credentialGroupId: current.credentialGroupId, + }) + const authorization = await adapter.prepareAuthorization(current, policy) + const url = new URL( + await authorization.buildAuthorizationUrl({ state: 'state', nonce: 'nonce' }) + ) + expect(url.searchParams.get('user_scope')?.split(',')).toEqual([...scopes]) + expect(policy.requiredScopes).toEqual([...scopes]) + } + ) it('accepts a different provider email', async () => { const scopes = SLACK_SEARCH_USER_SCOPES diff --git a/apps/sim/lib/credential-groups/slack-provider.ts b/apps/sim/lib/credential-groups/slack-provider.ts index 51b0405af9b..f94d443e428 100644 --- a/apps/sim/lib/credential-groups/slack-provider.ts +++ b/apps/sim/lib/credential-groups/slack-provider.ts @@ -2,7 +2,6 @@ import { db } from '@sim/db' import { credentialGroup } from '@sim/db/schema' import { normalizeEmail } from '@sim/utils/string' import { and, eq } from 'drizzle-orm' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { resourceScopeColumns, resourceScopeFromOwner } from '@/lib/core/resource-scope' import { resourceScopeCondition } from '@/lib/core/resource-scope.server' import { getBaseUrl } from '@/lib/core/utils/urls' @@ -29,7 +28,6 @@ import { verifySlackUserIdentity, } from '@/lib/credential-groups/slack-managed-users' import type { DbOrTx } from '@/lib/db/types' -import { SLACK_RTS_USER_SCOPES } from '@/lib/sim-search/live/scopes' const PROVIDER = 'slack' as const @@ -181,15 +179,7 @@ export const slackCredentialGroupProviderAdapter: CredentialGroupProviderAdapter buildAuthorizationUrl: ({ state }) => { const authorizationUrl = new URL('https://slack.com/oauth/v2/authorize') authorizationUrl.searchParams.set('client_id', currentPolicy.clientId) - authorizationUrl.searchParams.set( - 'user_scope', - [ - ...new Set([ - ...policy.requiredScopes, - ...(isLiveEnterpriseSearchEnabled ? SLACK_RTS_USER_SCOPES : []), - ]), - ].join(',') - ) + authorizationUrl.searchParams.set('user_scope', policy.requiredScopes.join(',')) authorizationUrl.searchParams.set('redirect_uri', redirectUri) authorizationUrl.searchParams.set('state', state) authorizationUrl.searchParams.set('team', currentPolicy.teamId) diff --git a/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts b/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts index 4cf02aca8b1..1dc2ae6dfba 100644 --- a/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts +++ b/apps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts @@ -11,7 +11,7 @@ import { credentialsManagedOauthMock, credentialsManagedOauthMockFns, } from '@sim/testing/mocks/credentials-managed-oauth.mock' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' +import { resetEnvFlagsMock } from '@sim/testing/mocks/env-flags.mock' import { knowledgeSearchIntegrationPolicyMock, knowledgeSearchIntegrationPolicyMockFns, @@ -55,12 +55,6 @@ vi.mock('@/lib/sim-search/connectors', () => ({ ], })) vi.mock('@/lib/oauth/utils', () => oauthUtilsMock) -/** The barrel's other use cases need the application layer mocked above; ownership is exercised as is. */ -vi.mock('@/lib/sim-search/indexed', async () => ({ - ownsIndexedPersonalSearchAccount: ( - await import('@/lib/sim-search/indexed/integrations/personal-account-ownership') - ).ownsIndexedPersonalSearchAccount, -})) import { prepareOrganizationPersonalConnection, @@ -125,7 +119,6 @@ describe('organization personal token authorization', () => { }) it('uses current personal OAuth inventory in live mode without consulting indexed sources', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) mocks.liveAccounts.mockResolvedValue([ { id: 'own', providerId: 'google-drive', type: 'managed_oauth' }, ]) @@ -136,7 +129,6 @@ describe('organization personal token authorization', () => { expect(mocks.liveAccounts).toHaveBeenCalledWith({ organizationId: 'org' }, 'person') }) it('does not let direct integration tools bypass current organization scope restrictions', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) mocks.liveAccounts.mockResolvedValue([ { id: 'own', providerId: 'google-drive', type: 'managed_oauth' }, ]) @@ -156,7 +148,6 @@ describe('organization personal token authorization', () => { it.each(['service_account'])( 'never substitutes a %s for a personal OAuth account', async (type) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) mocks.liveAccounts.mockResolvedValue([{ id: 'own', providerId: 'google-drive', type }]) await expect(resolveOrganizationPersonalToken.execute({ principal, input })).rejects.toThrow( 'own connected account' @@ -166,14 +157,15 @@ describe('organization personal token authorization', () => { } ) it('observes a revoked live account without falling back to old indexing membership', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) await expect(resolveOrganizationPersonalToken.execute({ principal, input })).rejects.toThrow( 'own connected account' ) expect(mocks.token).not.toHaveBeenCalled() }) it('uses the authenticated person inventory and organization token scope without a workspace', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + mocks.liveAccounts.mockResolvedValue([ + { id: 'own', providerId: 'google-drive', type: 'managed_oauth' }, + ]) await expect(resolveOrganizationPersonalToken.execute({ principal, input })).resolves.toEqual({ accessToken: 'secret', refreshed: false, @@ -219,21 +211,7 @@ describe('organization personal token authorization', () => { expect(mocks.token).not.toHaveBeenCalled() }) - it('does not confuse paused indexing with account authorization', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - mocks.inventory.mockResolvedValue({ - connections: [ - { indexingStatus: 'paused', accounts: [{ credentialId: 'own', status: 'connected' }] }, - ], - nextCursor: null, - }) - await expect( - resolveOrganizationPersonalToken.execute({ principal, input }) - ).resolves.toHaveProperty('credentialType', 'managed_oauth') - }) - it('returns the live account target for a connection request without an indexed source', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) const target = { type: 'link', provider: 'google-drive', @@ -254,28 +232,4 @@ describe('organization personal token authorization', () => { expect(result).not.toHaveProperty('settingsPath') expect(mocks.liveAccounts).not.toHaveBeenCalled() }) - - it('finds an exact reconnect control on a later inventory page', async () => { - const target = { - type: 'link', - provider: 'google-drive', - connectorType: 'drive', - connectorId: 'connector', - credentialId: 'own', - } - mocks.inventory.mockResolvedValueOnce({ connections: [], available: [], nextCursor: 'next' }) - mocks.inventory.mockResolvedValueOnce({ - connections: [{ accounts: [{ credentialId: 'own', action: target }] }], - available: [], - nextCursor: null, - }) - await expect( - prepareOrganizationPersonalConnection.execute({ - principal, - input: { organizationId: 'org', providerName: 'Google Drive', credentialId: 'own' }, - }) - ).resolves.toEqual({ provider: 'Google Drive', providerId: 'google-drive', target }) - expect(mocks.inventory.mock.calls[1][0].input.cursor).toBe('next') - expect(mocks.token).not.toHaveBeenCalled() - }) }) diff --git a/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts b/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts index a1c7f8f8b29..2fc9b72a254 100644 --- a/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts +++ b/apps/sim/lib/credentials/application/resolve-organization-personal-token.ts @@ -11,12 +11,11 @@ import { } from '@/lib/credential-groups/credentials' import { resolveManagedOAuthToken } from '@/lib/credentials/managed-oauth' import { projectIntegrationToolsForViewer } from '@/lib/integrations/tool-projection' -import { personalSearchIntegrationPages } from '@/lib/knowledge/application/personal-search-integration-pages' +import { listPersonalSearchIntegrations } from '@/lib/knowledge/application/personal-search-integrations' import { requireOrganizationSearchApproval } from '@/lib/knowledge/search/integration-policy' import { providerIdsForService } from '@/lib/oauth/utils' import { getUserPermissionConfigForOrganization } from '@/lib/permission-groups/resolve.server' import { SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { listLiveAccounts } from '@/lib/sim-search/live/accounts' import { requiresScopedRetrieval } from '@/lib/sim-search/live/policy-schema' import { livePolicyFor, loadLiveSearchPolicies } from '@/lib/sim-search/live/policy-store' @@ -84,36 +83,25 @@ export const resolveOrganizationPersonalToken = { ) { throw new OrchestrationError('forbidden', 'This integration operation is unavailable.') } - const indexed = isIndexedOrgSearchEnabled() - /** Loaded lazily: this resolver is on the executor's credential path, which never needs it otherwise. */ - const owned = indexed - ? await (await import('@/lib/sim-search/indexed')).ownsIndexedPersonalSearchAccount( - principal, - { - organizationId: context.organizationId, - connectorType: connector.type, - credentialId: input.credentialId, - } - ) - : (await listLiveAccounts({ organizationId: context.organizationId }, context.userId)).some( - (account) => - account.id === input.credentialId && - account.type === 'managed_oauth' && - account.providerId === binding.providerId - ) + const owned = ( + await listLiveAccounts({ organizationId: context.organizationId }, context.userId) + ).some( + (account) => + account.id === input.credentialId && + account.type === 'managed_oauth' && + account.providerId === binding.providerId + ) if (!owned) throw new OrchestrationError( 'forbidden', 'Assistant can only use your own connected account for this integration.' ) - if (!indexed) { - const policies = await loadLiveSearchPolicies({ organizationId: context.organizationId }) - if (requiresScopedRetrieval(connector.type, livePolicyFor(policies, connector.type))) - throw new OrchestrationError( - 'forbidden', - 'This organization restricts search scope for this app. Use search_workspace with nativeQueries and read_document so those restrictions are enforced.' - ) - } + const policies = await loadLiveSearchPolicies({ organizationId: context.organizationId }) + if (requiresScopedRetrieval(connector.type, livePolicyFor(policies, connector.type))) + throw new OrchestrationError( + 'forbidden', + 'This organization restricts search scope for this app. Use search_workspace with nativeQueries and read_document so those restrictions are enforced.' + ) const token = await resolveManagedOAuthToken({ ...input, expectedProviderId: binding.providerId, @@ -177,17 +165,16 @@ export const prepareOrganizationPersonalConnection = { ) if (!connector) throw new OrchestrationError('validation', 'This integration is unavailable in Search.') - for await (const inventory of personalSearchIntegrationPages({ + const inventory = await listPersonalSearchIntegrations.execute({ principal, input: { organizationId: context.organizationId, connectorType: connector.type }, - })) { - const target = input.credentialId - ? inventory.connections - .flatMap((connection) => connection.accounts) - .find((account) => account.credentialId === input.credentialId)?.action - : inventory.available[0]?.target - if (target) return { provider: connector.meta.name, providerId: connector.providerId, target } - } + }) + const target = input.credentialId + ? inventory.connections + .flatMap((connection) => connection.accounts) + .find((account) => account.credentialId === input.credentialId)?.action + : inventory.available[0]?.target + if (target) return { provider: connector.meta.name, providerId: connector.providerId, target } throw new OrchestrationError( 'validation', 'No connection action is currently available. Check your integration connection status.' diff --git a/apps/sim/lib/credentials/managed-oauth.test.ts b/apps/sim/lib/credentials/managed-oauth.test.ts index 67935bb0fb2..74b8427d31d 100644 --- a/apps/sim/lib/credentials/managed-oauth.test.ts +++ b/apps/sim/lib/credentials/managed-oauth.test.ts @@ -410,7 +410,10 @@ describe('managed OAuth token resolution', () => { }) it('allows Search reads when Slack retains a broader grant', async () => { - seedSlackSearchCredential('option-1', 'active', SLACK_MANAGED_USER_SCOPES) + seedSlackSearchCredential('option-1', 'active', [ + ...SLACK_MANAGED_USER_SCOPES, + ...SLACK_SEARCH_USER_SCOPES, + ]) await expect( resolveManagedOAuthToken({ credentialId: 'credential-1', diff --git a/apps/sim/lib/desktop/source-browser.ts b/apps/sim/lib/desktop/source-browser.ts index 3494632bc43..ffe71f0c111 100644 --- a/apps/sim/lib/desktop/source-browser.ts +++ b/apps/sim/lib/desktop/source-browser.ts @@ -5,10 +5,7 @@ import { consumeDesktopSourceRequestContract, type DesktopSourceRequest, } from '@/lib/api/contracts/desktop-source-connect' -import { - connectSimSearchConnectorContract, - startKnowledgeConnectorMemberEnrollmentContract, -} from '@/lib/api/contracts/knowledge/connectors' +import { startKnowledgeConnectorMemberEnrollmentContract } from '@/lib/api/contracts/knowledge/connectors' import { gitHubSearchSetupScopeSchema, readGitHubSearchSetupContract, @@ -120,17 +117,6 @@ async function startRequest( : enrollmentMatch(result.data.url), } } - case 'search-source': { - const result = await requestJson(connectSimSearchConnectorContract, { - body: { ...request.body, oauthCompletionId: request.completionId }, - }) - return { - url: result.data.url, - match: request.completionId - ? { kind: 'completion', id: request.completionId } - : enrollmentMatch(result.data.url), - } - } case 'slack-managed-users': { const { owner, body, credentialGroupId } = request const result = owner.organizationId diff --git a/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts b/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts index ec084696138..e40cdb28090 100644 --- a/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/coda-live.integration.ts @@ -11,9 +11,7 @@ import { db } from '@sim/db' import { credential, document, - knowledgeBase, knowledgeConnector, - member, organization, session, user, @@ -24,14 +22,6 @@ import { generateId } from '@sim/utils/id' import { serializeSignedCookie } from 'better-call' import { eq } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' - -/** Its search-index knowledge bases are read through indexed organization search, dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) - import { z } from 'zod' const metrics = vi.hoisted(() => ({ embeddingCalls: 0 })) @@ -51,12 +41,8 @@ vi.mock('@/lib/embeddings', async () => ({ }, })) -import { - resolveBillingAttribution, - resolveOrganizationBillingAttribution, -} from '@/lib/billing/core/billing-attribution' +import { resolveBillingAttribution } from '@/lib/billing/core/billing-attribution' import { encryptSecret } from '@/lib/core/security/encryption' -import { createOrganizationCredential } from '@/lib/credentials/application/organization-credentials' import { assertCodaLiveFixture } from '@/lib/knowledge/__integration__/coda-live-fixture' import { seedKnowledgeAclFixture } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { listKnowledgeChunks } from '@/lib/knowledge/application/chunks' @@ -73,7 +59,6 @@ const tokenPath = process.env.CODA_CONNECTOR_LIVE_TOKEN_FILE const fixturePath = process.env.CODA_CONNECTOR_LIVE_FIXTURE_FILE const secondEmail = process.env.CODA_CONNECTOR_LIVE_SECOND_EMAIL const allowSharing = process.env.CODA_CONNECTOR_LIVE_ALLOW_SHARING !== 'false' -const organizationScope = process.env.CODA_CONNECTOR_LIVE_SCOPE === 'organization' const uiFixturePath = process.env.CODA_CONNECTOR_LIVE_UI_FIXTURE_FILE const fixtureSchema = z.object({ docId: z.string(), @@ -97,7 +82,7 @@ describe.skipIf(!tokenPath || !fixturePath || !secondEmail)( let token: string let fixture: z.infer let fixtureValidated = false - let credentialId = generateId() + const credentialId = generateId() const principal = (userId: string): Principal => ({ kind: 'session', userId, @@ -139,9 +124,7 @@ describe.skipIf(!tokenPath || !fixturePath || !secondEmail)( const result = await searchKnowledge.execute({ principal: as, input: { - ...(organizationScope - ? { organizationId: ids.organizationId } - : { workspaceId: ids.workspaceId }), + workspaceId: ids.workspaceId, knowledgeBaseIds: [ids.knowledgeBaseId], query: fixture.marker, searchMode: 'hybrid', @@ -154,15 +137,10 @@ describe.skipIf(!tokenPath || !fixturePath || !secondEmail)( async function sync() { const result = await executeSync(connectorId, { fullSync: false, - billingAttribution: organizationScope - ? await resolveOrganizationBillingAttribution({ - actorUserId: ids.aliceId, - organizationId: ids.organizationId, - }) - : await resolveBillingAttribution({ - actorUserId: ids.aliceId, - workspaceId: ids.workspaceId, - }), + billingAttribution: await resolveBillingAttribution({ + actorUserId: ids.aliceId, + workspaceId: ids.workspaceId, + }), }) expect(result.error).toBeUndefined() expect(result.skipReason).toBeUndefined() @@ -185,63 +163,28 @@ describe.skipIf(!tokenPath || !fixturePath || !secondEmail)( ids = await seedKnowledgeAclFixture() await db.update(user).set({ email: source.owner }).where(eq(user.id, ids.aliceId)) await db.update(user).set({ email: secondEmail! }).where(eq(user.id, ids.bobId)) - if (organizationScope) { - await db.insert(member).values([ - { - id: generateId(), - organizationId: ids.organizationId, - userId: ids.aliceId, - role: 'owner', - createdAt: new Date(), - }, - { - id: generateId(), - organizationId: ids.organizationId, - userId: ids.bobId, - role: 'member', - createdAt: new Date(), - }, - ]) - await db - .update(knowledgeBase) - .set({ workspaceId: null, organizationId: ids.organizationId, isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - const createdCredential = await createOrganizationCredential.execute({ - principal: principal(ids.aliceId), - input: { - organizationId: ids.organizationId, - type: 'service_account', - providerId: 'coda-service-account', - displayName: 'Disposable Coda live fixture', - apiToken: token, - }, - }) - credentialId = createdCredential.credential.id - } else - await db.insert(credential).values({ - id: credentialId, - workspaceId: ids.workspaceId, - createdBy: ids.aliceId, - type: 'service_account', - providerId: 'coda-service-account', - displayName: 'Disposable Coda live fixture', - encryptedServiceAccountKey: ( - await encryptSecret( - JSON.stringify({ - type: 'token_service_account', - providerId: 'coda-service-account', - apiToken: token, - }) - ) - ).encrypted, - }) + await db.insert(credential).values({ + id: credentialId, + workspaceId: ids.workspaceId, + createdBy: ids.aliceId, + type: 'service_account', + providerId: 'coda-service-account', + displayName: 'Disposable Coda live fixture', + encryptedServiceAccountKey: ( + await encryptSecret( + JSON.stringify({ + type: 'token_service_account', + providerId: 'coda-service-account', + apiToken: token, + }) + ) + ).encrypted, + }) const created = await createKnowledgeConnector.execute({ principal: principal(ids.aliceId), input: { knowledgeBaseId: ids.knowledgeBaseId, - ...(organizationScope - ? { assertedOrganizationId: ids.organizationId } - : { assertedWorkspaceId: ids.workspaceId }), + assertedWorkspaceId: ids.workspaceId, connectorType: 'coda', credentialId, accessMode: 'admin', @@ -321,8 +264,7 @@ describe.skipIf(!tokenPath || !fixturePath || !secondEmail)( workspaceId: ids.workspaceId, keyId: 'fixture', }) - if (organizationScope) await expect(keySearch).rejects.toThrow() - else expect(await keySearch).not.toContain(documentId) + expect(await keySearch).not.toContain(documentId) const chunks = await listKnowledgeChunks.execute({ principal: principal(ids.aliceId), input: { knowledgeBaseId: ids.knowledgeBaseId, documentId }, diff --git a/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts b/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts index fea881f3e4b..2ccf293562c 100644 --- a/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/dormant-processing-recovery.integration.ts @@ -24,11 +24,6 @@ import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' const fixture = vi.hoisted(() => ({ root: '' })) vi.mock('@/lib/core/config/trigger-runtime', () => ({ isInsideTriggerRun: () => false })) -/** Pinned to Live Search, whatever `SIM_SEARCH_LIVE` the run was started with. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => ({ - ...(await importOriginal>()), - isLiveEnterpriseSearchEnabled: true, -})) vi.mock('@/lib/uploads/core/setup.server', () => ({ get UPLOAD_DIR_SERVER() { return fixture.root diff --git a/apps/sim/lib/knowledge/__integration__/dormant-search-processing.integration.ts b/apps/sim/lib/knowledge/__integration__/dormant-search-processing.integration.ts index 9da06ae10c6..0ed419014ed 100644 --- a/apps/sim/lib/knowledge/__integration__/dormant-search-processing.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/dormant-search-processing.integration.ts @@ -2,7 +2,7 @@ import { db } from '@sim/db' import { document, knowledgeBase, organization, user, workspace } from '@sim/db/schema' import { generateId } from '@sim/utils/id' import { eq, inArray } from 'drizzle-orm' -import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, @@ -14,11 +14,6 @@ import { processDocumentsWithQueue, } from '@/lib/knowledge/documents/service' -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => ({ - ...(await importOriginal>()), - isLiveEnterpriseSearchEnabled: true, -})) - /** Queued work cannot revive dormant Search before retirement reaches its documents. */ describe('dormant Search document processing', () => { const ids = createKnowledgeAclFixtureIds() diff --git a/apps/sim/lib/knowledge/__integration__/embedding-insert-batches.integration.ts b/apps/sim/lib/knowledge/__integration__/embedding-insert-batches.integration.ts index 25a060da5fb..7c47768857e 100644 --- a/apps/sim/lib/knowledge/__integration__/embedding-insert-batches.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/embedding-insert-batches.integration.ts @@ -17,13 +17,6 @@ import { generateId } from '@sim/utils/id' import { eq, inArray } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' -/** These transaction checks exercise indexed Search, which Live Search normally disables. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) - const fixtures = vi.hoisted(() => ({ root: '', process: vi.fn(), embeddings: vi.fn() })) vi.mock('@/lib/uploads/core/setup.server', () => ({ get UPLOAD_DIR_SERVER() { @@ -56,11 +49,6 @@ describe('bounded embedding insert transactions', () => { beforeAll(async () => { fixtures.root = mkdtempSync(path.join(tmpdir(), 'sim-embedding-batches-')) await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) - /** A search index, the only kind of base whose chunks the keyword projection holds. */ - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) vi.spyOn(embeddingClient, 'assertKnowledgeEmbeddingCapacity').mockResolvedValue(undefined) }) @@ -163,7 +151,7 @@ describe('bounded embedding insert transactions', () => { .from(embedding) .where(eq(embedding.documentId, file.documentId)) ).toEqual([{ id: previousId }]) - for (const table of ['embedding_search', 'embedding_keyword_search']) { + for (const table of ['embedding_search']) { expect( await db.$client.unsafe(`SELECT id FROM ${table} WHERE document_id = $1`, [file.documentId]) ).toEqual([{ id: previousId }]) @@ -187,7 +175,7 @@ describe('bounded embedding insert transactions', () => { expect(await db.select().from(document).where(eq(document.id, file.documentId))).toMatchObject([ { processingStatus: 'completed', chunkCount: 205, processingError: null }, ]) - for (const table of ['embedding', 'embedding_search', 'embedding_keyword_search']) { + for (const table of ['embedding', 'embedding_search']) { expect( await db.$client.unsafe( `SELECT count(*)::int AS count FROM ${table} WHERE document_id = $1`, diff --git a/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts b/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts index 451f96df73b..79e4b2a111e 100644 --- a/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/excluded-member-documents.integration.ts @@ -3,7 +3,6 @@ import { db } from '@sim/db' import { document, embedding, - knowledgeBase, knowledgeConnector, knowledgeConnectorMember, knowledgeDocumentObservation, @@ -17,12 +16,6 @@ import { eq, inArray } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' const provider = vi.hoisted(() => ({ list: vi.fn(), get: vi.fn(), changes: vi.fn() })) -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/connectors/registry.server', () => ({ CONNECTOR_REGISTRY: { google_drive: { @@ -139,10 +132,6 @@ describe('excluded member documents retain current source authorization', () => }, ids.aliceId ) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) await db .update(knowledgeConnector) .set({ status: 'active', memberSyncStatus: 'idle', memberSyncLockToken: null }) diff --git a/apps/sim/lib/knowledge/__integration__/github-member.integration.ts b/apps/sim/lib/knowledge/__integration__/github-member.integration.ts index b680357e2f6..7829bd571eb 100644 --- a/apps/sim/lib/knowledge/__integration__/github-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/github-member.integration.ts @@ -32,12 +32,6 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, isNull, sql } from 'drizzle-orm' import { afterAll, afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, @@ -81,7 +75,6 @@ import { seedKnowledgeAclFixture, seedKnowledgeMemberFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { GITHUB_READ_SOURCE_TIMEOUT_MS } from '@/lib/knowledge/access/github-installation' import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' import { subjectToken } from '@/lib/knowledge/access/tokens' import { KnowledgeDocumentNotReadyError } from '@/lib/knowledge/application/chunk-errors' @@ -92,8 +85,6 @@ import { } from '@/lib/knowledge/application/connectors' import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' import { searchKnowledge } from '@/lib/knowledge/application/search' -import { readSearchSourceOverview } from '@/lib/knowledge/application/search-source-overview' -import { listSearchSources } from '@/lib/knowledge/application/search-sources' import { grantKnowledgeConnectorCredentialAccess } from '@/lib/knowledge/connectors/member-access' import * as memberSyncEngine from '@/lib/knowledge/connectors/member-sync-engine' import { @@ -101,11 +92,8 @@ import { MEMBER_TOMBSTONE_PURGE_DAYS, } from '@/lib/knowledge/connectors/sync-limits' import { getDocuments } from '@/lib/knowledge/documents/service' -import { getTagUsageStats } from '@/lib/knowledge/tags/service' -import { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' import { deleteFile } from '@/lib/uploads/core/storage-service' import { downloadFileFromUrl } from '@/lib/uploads/utils/file-utils.server' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' const redisUrl = readTestRedisUrl() @@ -144,7 +132,6 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { const { privateKey, publicKey } = generateKeyPairSync('rsa', { modulusLength: 2048 }) let organizationSource = false let installationSuspended = false - let referenceObserved: ((repository: string) => void) | undefined const installation = () => ({ id: 42, app_id: 1, @@ -349,7 +336,6 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { : Response.json({ message: 'Not Found' }, { status: 404 }) } if (match[2].startsWith('/git/ref/heads/')) { - referenceObserved?.(match[1]) if (source.stallRef) return new Promise((_resolve, reject) => { request.signal.addEventListener('abort', () => reject(request.signal.reason), { @@ -444,7 +430,6 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { refreshedUsers.clear() organizationSource = false installationSuspended = false - referenceObserved = undefined oauthStateKey = undefined oauthVerification = undefined Object.assign(env, { @@ -453,10 +438,6 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { }) ids = createKnowledgeAclFixtureIds() await seedKnowledgeAclFixture(ids) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) enrolled = await seedKnowledgeMemberFixture(ids) const policy = await getCredentialGroupProviderAdapter('github-repositories').getPolicy( undefined, @@ -682,7 +663,7 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { ]) await db .update(knowledgeBase) - .set({ workspaceId: null, organizationId: ids.organizationId }) + .set({ workspaceId: null, organizationId: ids.organizationId, isSearchIndex: true }) .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) await db .update(credentialGroup) @@ -760,10 +741,8 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { return installationCredentialId } - it('reuses connected organization members for a later installation source without another enrollment', async () => { + it('creates a live installation source without indexing or replacing connected member accounts', async () => { const installationCredentialId = await useOrganizationInstallation() - expect((await sync()).error).toBeUndefined() - const [shared] = await rows() const enrollmentsBefore = await db .select({ id: credentialGroupEnrollment.id, @@ -795,17 +774,10 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { await expect( createKnowledgeConnector.execute({ principal: actor(ids.bobId), input }) ).rejects.toThrow() - const dispatchedSync = vi.spyOn(memberSyncEngine, 'executeMemberSync') const { connector } = await createKnowledgeConnector.execute({ principal: actor(ids.aliceId), input, }) - try { - expect(dispatchedSync).toHaveBeenCalledExactlyOnceWith(connector.id, expect.any(Object)) - expect((await dispatchedSync.mock.results[0].value).error).toBeUndefined() - } finally { - dispatchedSync.mockRestore() - } const connectorId = connector.id expect(connector).toMatchObject({ credentialGroupId: enrolled.groupId, @@ -816,16 +788,7 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { .select() .from(knowledgeConnector) .where(eq(knowledgeConnector.id, connectorId)) - expect(current).toMatchObject({ memberSyncStatus: 'idle' }) - expect(current.lastMemberSyncAt).not.toBeNull() - const memberships = await db - .select() - .from(knowledgeConnectorMember) - .where(eq(knowledgeConnectorMember.connectorId, connectorId)) - expect(memberships).toHaveLength(2) - expect(memberships.map((row) => row.credentialId).sort()).toEqual( - credentialsBefore.map((row) => row.id).sort() - ) + expect(current).toMatchObject({ lastMemberSyncAt: null, nextMemberSyncAt: null }) expect( await db .select({ @@ -848,54 +811,17 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { .where(eq(credential.credentialGroupOptionId, enrolled.optionId)) .orderBy(credential.id) ).toEqual(credentialsBefore) - const [indexed] = await rows(connectorId) - expect(await search(actor(ids.aliceId))).toEqual([shared.id, indexed.id].sort()) - expect(await search(actor(ids.bobId))).toEqual([shared.id]) - await assertAccess(actor(ids.aliceId), indexed, true) - await assertAccess(actor(ids.bobId), indexed, false) - const indexedRead = (userId: string) => - readIndexedKnowledgeDocument.execute({ - principal: actor(userId), - input: { - organizationId: ids.organizationId, - target: { kind: 'id', documentId: indexed.id }, - limit: 10, - resultSecretRegistry: new ResolvedSecretTraceRegistry(), - }, - }) + expect(await rows(connectorId)).toEqual([]) expect( - (await indexedRead(ids.aliceId)).chunks?.map((chunk) => chunk.content).join('\n') - ).toContain('Orion later') - await expect(indexedRead(ids.bobId)).rejects.toThrow('Document not found') - /** Organization cache bytes are internal; members read through the authorized Search operation. */ - await expect( - downloadFileFromUrl(indexed.fileUrl, { userId: ids.aliceId, knowledgeAccess: 'user' }) - ).rejects.toThrow('Access denied') - await expect( - downloadFileFromUrl(indexed.fileUrl, { userId: ids.bobId, knowledgeAccess: 'user' }) - ).rejects.toThrow('Access denied') - for (const userId of [ids.aliceId, ids.bobId]) { - const { sources } = await listSearchSources.execute({ - principal: actor(userId), - input: { organizationId: ids.organizationId, connectorId }, - }) - expect(sources).toMatchObject([ - { - connectorId, - viewerMembership: 'connected', - hasViewerDocuments: userId === ids.aliceId, - }, - ]) - } - expect( - requests - .filter((entry) => entry.path.startsWith('/repos/fixture/later/git/blobs/')) - .map((entry) => entry.userId) - ).toEqual(['installation']) + await db + .select({ id: embedding.id }) + .from(embedding) + .where(eq(embedding.knowledgeBaseId, ids.knowledgeBaseId)) + ).toEqual([]) + expect(requests.filter((entry) => entry.path.includes('/git/blobs/'))).toEqual([]) }) - it('keeps actual skips, legacy skips, and provider failures distinct in authorized lists and source counts', async () => { - await useOrganizationInstallation() + it('keeps actual skips, legacy skips, and provider failures distinct in authorized document lists', async () => { const source = repositories.get('shared')! source.readers.delete(ids.bobId) source.files.set('empty.txt', '') @@ -908,18 +834,6 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { expect(binary).toMatchObject({ processingStatus: 'failed', storageKey: null }) expect(empty.contentHash).not.toBeNull() await db.update(document).set({ processingStatus: 'failed' }).where(eq(document.id, empty.id)) - const summary = async (userId: string) => - ( - await listSearchSources.execute({ - principal: actor(userId), - input: { organizationId: ids.organizationId, connectorId: enrolled.connectorId }, - }) - ).sources[0] - expect(await summary(ids.aliceId)).toMatchObject({ - hasViewerDocuments: true, - viewerFailedDocumentCount: 0, - hasSyncError: false, - }) source.files.set('unavailable.txt', 'Orion content whose blob cannot be fetched.') source.failedBlobs.add(shaFor(source.files.get('unavailable.txt')!)) await sync() @@ -932,7 +846,7 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { }) for (const userId of [ids.aliceId, ids.bobId]) { const provider = createKnowledgeAccessProvider(actor(userId), { - organizationId: ids.organizationId, + workspaceId: ids.workspaceId, knowledgeBaseIds: [ids.knowledgeBaseId], }) const listed = await getDocuments(ids.knowledgeBaseId, {}, 'github-skip-outcomes', provider) @@ -973,192 +887,7 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { : [] ) } - expect(await summary(userId)).toMatchObject({ - hasViewerDocuments: userId === ids.aliceId, - viewerFailedDocumentCount: userId === ids.aliceId ? 1 : 0, - }) - } - }) - - it('indexes an organization installation once and denies live user, app, and org revocations before search or reads', async () => { - const installationCredentialId = await useOrganizationInstallation() - const unrelatedSources = Array.from({ length: 105 }, () => generateId()) - await db.insert(knowledgeConnector).values( - unrelatedSources.map((id) => ({ - id, - knowledgeBaseId: ids.knowledgeBaseId, - connectorType: 'github', - accessMode: 'members', - credentialId: installationCredentialId, - credentialGroupId: enrolled.groupId, - credentialGroupOptionId: enrolled.optionId, - sourceConfig: { repository: 'fixture/shared', githubRepositoryId: '9001' }, - })) - ) - await db.insert(knowledgeConnectorMember).values( - unrelatedSources.map((connectorId) => ({ - id: generateId(), - organizationId: ids.organizationId, - connectorId, - credentialId: enrolled.members[0].credentialId, - subjectToken: enrolled.members[0].subjectToken, - })) - ) - const result = await sync() - expect(result.error).toBeUndefined() - expect(result.docsHydratedOnce).toBe(1) - const [indexed] = await rows() - expect(indexed).toBeDefined() - const provider = (userId: string) => - createKnowledgeAccessProvider(actor(userId), { - organizationId: ids.organizationId, - knowledgeBaseIds: [ids.knowledgeBaseId], - }) - const page = (userId: string, offset = 0) => - getDocuments( - ids.knowledgeBaseId, - { limit: 1, offset, sortBy: 'filename', sortOrder: 'asc' }, - 'github-candidate-regression', - provider(userId) - ) - expect(await page(ids.aliceId)).toMatchObject({ - documents: [{ id: indexed.id }], - pagination: { total: 1 }, - }) - expect( - ( - await readSearchSourceOverview.execute({ - principal: actor(ids.aliceId), - input: { organizationId: ids.organizationId }, - }) - ).hasSearchableDocuments - ).toBe(true) - await db.update(document).set({ tag1: 'fixture' }).where(eq(document.id, indexed.id)) - await db.update(embedding).set({ tag1: 'fixture' }).where(eq(embedding.documentId, indexed.id)) - expect( - await getTagUsageStats(ids.knowledgeBaseId, provider(ids.aliceId), 'github-tag-regression') - ).toEqual( - expect.arrayContaining([ - expect.objectContaining({ tagSlot: 'tag1', documentCount: 1, chunkCount: 1 }), - ]) - ) - expect( - ( - await readIndexedKnowledgeDocument.execute({ - principal: actor(ids.aliceId), - input: { - organizationId: ids.organizationId, - target: { kind: 'url', url: indexed.sourceUrl! }, - limit: 1, - resultSecretRegistry: new ResolvedSecretTraceRegistry(), - }, - }) - ).documentId - ).toBe(indexed.id) - expect( - requests.filter((entry) => entry.path.includes('/git/blobs/')).map((entry) => entry.userId) - ).toEqual(['installation']) - expect(await search(actor(ids.aliceId))).toEqual([indexed.id]) - expect(await search(actor(ids.bobId))).toEqual([indexed.id]) - await assertAccess(actor(ids.bobId), indexed, true) - const source = repositories.get('shared')! - source.readers.delete(ids.bobId) - expect(await page(ids.bobId)).toMatchObject({ documents: [], pagination: { total: 0 } }) - expect(await search(actor(ids.bobId))).toEqual([]) - await assertAccess(actor(ids.bobId), indexed, false) - expect(await search(actor(ids.aliceId))).toEqual([indexed.id]) - expect( - await db - .select() - .from(knowledgeDocumentObservation) - .where(eq(knowledgeDocumentObservation.documentId, indexed.id)) - ).toHaveLength(2) - source.readers.add(ids.bobId) - source.public = true - source.installed = false - expect(await search(actor(ids.aliceId))).toEqual([]) - await assertAccess(actor(ids.aliceId), indexed, false) - source.installed = true - installationSuspended = true - expect(await search(actor(ids.aliceId))).toEqual([]) - installationSuspended = false - expect(await search(actor(ids.aliceId))).toEqual([indexed.id]) - const slowRepository = repository('slow') - const slowSourceId = generateId() - await db.insert(knowledgeConnector).values({ - id: slowSourceId, - knowledgeBaseId: ids.knowledgeBaseId, - connectorType: 'github', - accessMode: 'members', - credentialId: installationCredentialId, - credentialGroupId: enrolled.groupId, - credentialGroupOptionId: enrolled.optionId, - sourceConfig: { repository: 'fixture/slow', githubRepositoryId: String(slowRepository.id) }, - }) - expect((await sync(slowSourceId)).error).toBeUndefined() - const [slowDocument] = await rows(slowSourceId) - expect(slowDocument).toBeDefined() - const deniedRepository = repository('denied-paging', [ids.aliceId]) - const deniedSourceId = generateId() - await db.insert(knowledgeConnector).values({ - id: deniedSourceId, - knowledgeBaseId: ids.knowledgeBaseId, - connectorType: 'github', - accessMode: 'members', - credentialId: installationCredentialId, - credentialGroupId: enrolled.groupId, - credentialGroupOptionId: enrolled.optionId, - sourceConfig: { - repository: 'fixture/denied-paging', - githubRepositoryId: String(deniedRepository.id), - }, - }) - expect((await sync(deniedSourceId)).error).toBeUndefined() - const [deniedDocument] = await rows(deniedSourceId) - deniedRepository.readers.delete(ids.aliceId) - for (const [id, filename] of [ - [indexed.id, 'alpha'], - [deniedDocument.id, 'beta'], - [slowDocument.id, 'gamma'], - ]) - await db.update(document).set({ filename }).where(eq(document.id, id)) - expect(await page(ids.aliceId, 1)).toMatchObject({ - documents: [{ id: slowDocument.id }], - pagination: { total: 2, offset: 1 }, - }) - slowRepository.stallRef = true - const sourceTimers: AbortController[] = [] - const nativeTimeout = AbortSignal.timeout.bind(AbortSignal) - const timerSpy = vi.spyOn(AbortSignal, 'timeout').mockImplementation((duration) => { - if (duration !== GITHUB_READ_SOURCE_TIMEOUT_MS) return nativeTimeout(duration) - const controller = new AbortController() - sourceTimers.push(controller) - return controller.signal - }) - try { - const observed = new Set() - const candidatesStarted = new Promise((resolve) => { - referenceObserved = (name) => { - observed.add(name) - if (observed.has('shared') && observed.has('slow')) resolve() - } - }) - const pending = search(actor(ids.aliceId), 'vector') - await candidatesStarted - /** Complete the fast response's microtasks before expiring the stalled candidate. */ - for (let turn = 0; turn < 20; turn++) await Promise.resolve() - for (const timer of sourceTimers) timer.abort(new Error('fixture source timeout')) - expect(await pending).toEqual([indexed.id]) - } finally { - timerSpy.mockRestore() - referenceObserved = undefined - slowRepository.stallRef = false } - await db - .delete(member) - .where(and(eq(member.organizationId, ids.organizationId), eq(member.userId, ids.bobId))) - await expect(search(actor(ids.bobId))).rejects.toThrow() - await assertAccess(actor(ids.bobId), indexed, false) }) it.runIf(Boolean(redisUrl))( @@ -1316,24 +1045,6 @@ describe('fixture-backed GitHub member search in PostgreSQL', () => { expect(privateFile.acl).toEqual([enrolled.members[0].subjectToken]) expect(await search(actor(ids.aliceId))).toEqual([shared.id, privateFile.id].sort()) expect(await search(actor(ids.bobId))).toEqual([shared.id]) - for (const userId of [ids.aliceId, ids.bobId]) { - const summaries = await listSearchSources.execute({ - principal: actor(userId), - input: { workspaceId: ids.workspaceId }, - }) - expect( - summaries.sources.map((source) => ({ - connectorId: source.connectorId, - hasViewerDocuments: source.hasViewerDocuments, - })) - ).toEqual( - expect.arrayContaining([ - { connectorId: enrolled.connectorId, hasViewerDocuments: true }, - { connectorId: privateId, hasViewerDocuments: userId === ids.aliceId }, - { connectorId: blockedId, hasViewerDocuments: false }, - ]) - ) - } expect(await search(workspaceKey())).toEqual([]) await assertAccess(actor(ids.bobId), shared, true) await assertAccess(actor(ids.bobId), privateFile, false) diff --git a/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts b/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts index 774b7d00846..974768e2295 100644 --- a/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/gitlab-live.integration.ts @@ -20,9 +20,7 @@ import { knowledgeConnector, knowledgeConnectorPermissionGrant, knowledgeConnectorPermissionSnapshot, - member, organization, - organizationSearchIntegration, permissions, session, user, @@ -36,14 +34,6 @@ import { serializeSignedCookie } from 'better-call' import { and, eq, inArray, isNull, or } from 'drizzle-orm' import { NextRequest } from 'next/server' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' - -/** Its search-index knowledge bases are read through indexed organization search, dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) - import type { EmbedOptions } from '@/lib/embeddings/types' const fixture = vi.hoisted(() => ({ embeddingCalls: 0 })) @@ -85,10 +75,7 @@ import type { UpdateConnectorBody, } from '@/lib/api/contracts/knowledge/connectors' import { decryptApiKey, encryptApiKey } from '@/lib/api-key/crypto' -import { - resolveBillingAttribution, - resolveOrganizationBillingAttribution, -} from '@/lib/billing/core/billing-attribution' +import { resolveBillingAttribution } from '@/lib/billing/core/billing-attribution' import { seedKnowledgeAclFixture } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { listKnowledgeChunks } from '@/lib/knowledge/application/chunks' import { updateKnowledgeConnectorAccess } from '@/lib/knowledge/application/connector-access' @@ -96,13 +83,11 @@ import { updateKnowledgeConnector } from '@/lib/knowledge/application/connectors import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' import { searchKnowledge } from '@/lib/knowledge/application/search' import { executeSync } from '@/lib/knowledge/connectors/sync-engine' -import { readSearchDocument } from '@/lib/sim-search/indexed/documents/read-search-document' import * as storage from '@/lib/uploads/core/storage-service' import { downloadFileFromUrl } from '@/lib/uploads/utils/file-utils.server' import { PATCH as updateConnectorRoute } from '@/app/api/knowledge/[id]/connectors/[connectorId]/route' import { POST as createConnectorRoute } from '@/app/api/knowledge/[id]/connectors/route' import { gitlabConnector } from '@/connectors/gitlab/gitlab' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' interface GitLabPerson { id: number @@ -700,336 +685,297 @@ describe.skipIf(!fixtureFile)('live self-hosted GitLab ingestion and permission throw new Error('Disposable connector sync did not finish within one minute') } - it.each([false, true])( - 'enforces non-admin CSV permissions through authenticated API, indexing and every read surface (Search=%s)', - async (isSearchIndex) => { - await api(`/projects/${projectId}`, 'PUT', { visibility: 'private' }) - await api(`/projects/${projectId}/issues/${issueIid}`, 'PUT', { confidential: false }) - const owner = isSearchIndex - ? { organizationId: ids.organizationId } - : { workspaceId: ids.workspaceId } - if (isSearchIndex) { - await db.insert(member).values( - Object.values(people).map((person) => ({ - id: generateId(), - userId: person.simId, - organizationId: ids.organizationId, - role: person.simId === ids.aliceId ? 'owner' : 'member', - createdAt: new Date(), - })) - ) - await db - .insert(organizationSearchIntegration) - .values({ organizationId: ids.organizationId, connectorType: 'gitlab', approved: true }) - } - const attribution = isSearchIndex - ? await resolveOrganizationBillingAttribution({ - actorUserId: ids.aliceId, - organizationId: ids.organizationId, - }) - : await resolveBillingAttribution({ - actorUserId: ids.aliceId, - workspaceId: ids.workspaceId, - }) - const knowledgeBaseId = generateId() - await db.insert(knowledgeBase).values({ - id: knowledgeBaseId, - userId: ids.aliceId, - ...owner, - name: `CSV live fixture ${isSearchIndex ? 'Search' : 'KB'}`, - isSearchIndex, - chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, - }) - const mapping = { - filename: 'users.csv', - content: `user_id,email\n${people.reporter.id},${people.reporter.email}\n${people.guest.id},${people.guest.email}\n`, - } - const projectPermissions = (userId: number) => ({ - filename: 'permissions.csv', - content: `project_path,user_id\n${config.project},${userId}\nunrelated/project,${people.guest.id}\n${config.project},999999999\n`, - }) - if (!auditorToken) await waitForProjectAccess(people.reporter, 200) - const indexingToken = auditorToken ?? people.reporter.token - const createBody: CreateConnectorBody = { - connectorType: 'gitlab', - accessMode: 'admin', - apiKey: indexingToken, - sourceConfig: { ...config, contentTypes: 'issues' }, - syncIntervalMinutes: 60, - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - userMapping: mapping, - projectPermissions: projectPermissions(people.reporter.id), - }, - } - const denied = await connectorRequest(knowledgeBaseId, createBody, undefined, readerCookie) - expect(denied.status).toBe(403) - const unsupported = await connectorRequest(knowledgeBaseId, { - ...createBody, - accessMode: 'workspace', - }) - expect(unsupported.status).toBe(400) - const inaccessibleProject = await connectorRequest(knowledgeBaseId, { - ...createBody, - sourceConfig: { ...createBody.sourceConfig, project: 'missing-fixture-project' }, - }) - expect(inaccessibleProject.status).toBe(400) - const createdResponse = await connectorRequest(knowledgeBaseId, createBody) - const created = await createdResponse.json() - expect(createdResponse.status, JSON.stringify(created)).toBe(201) - let connector = created.data as ConnectorData - expect(connector.permissionConfig).toMatchObject({ + it('enforces non-admin CSV permissions through authenticated API, indexing and every knowledge-base read surface', async () => { + await api(`/projects/${projectId}`, 'PUT', { visibility: 'private' }) + await api(`/projects/${projectId}/issues/${issueIid}`, 'PUT', { confidential: false }) + const owner = { workspaceId: ids.workspaceId } + const attribution = await resolveBillingAttribution({ + actorUserId: ids.aliceId, + workspaceId: ids.workspaceId, + }) + const knowledgeBaseId = generateId() + await db.insert(knowledgeBase).values({ + id: knowledgeBaseId, + userId: ids.aliceId, + ...owner, + name: 'CSV live fixture KB', + chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, + }) + const mapping = { + filename: 'users.csv', + content: `user_id,email\n${people.reporter.id},${people.reporter.email}\n${people.guest.id},${people.guest.email}\n`, + } + const projectPermissions = (userId: number) => ({ + filename: 'permissions.csv', + content: `project_path,user_id\n${config.project},${userId}\nunrelated/project,${people.guest.id}\n${config.project},999999999\n`, + }) + if (!auditorToken) await waitForProjectAccess(people.reporter, 200) + const indexingToken = auditorToken ?? people.reporter.token + const createBody: CreateConnectorBody = { + connectorType: 'gitlab', + accessMode: 'admin', + apiKey: indexingToken, + sourceConfig: { ...config, contentTypes: 'issues' }, + syncIntervalMinutes: 60, + permissionConfig: { provider: 'gitlab', mode: 'csv', - revision: 1, - userMapping: { rowCount: 2 }, - }) - expect(JSON.stringify(created)).not.toContain(indexingToken) - expect(JSON.stringify(created)).not.toContain(people.reporter.email) - expect(connector.sourceConfig).not.toHaveProperty('permissionConfig') - const storedConnector = await waitForSync(connector.id) - expect(storedConnector.encryptedApiKey).not.toBe(indexingToken) - expect( - (await decryptApiKey(storedConnector.encryptedApiKey!)).decrypted === indexingToken - ).toBe(true) - const docs = await db.select().from(document).where(eq(document.connectorId, connector.id)) - const ordinary = docs.find((doc) => doc.externalId === `issue:${issueIid}`)! - expect(ordinary.processingStatus).toBe('completed') - const excluded = docs.find((doc) => doc.externalId === `issue:${confidentialIid}`) - if (excluded) { - expect(excluded.acl).toEqual([]) - expect(excluded.storageKey).toBeNull() - } - const searchFor = async (person: GitLabPerson) => - ( - await searchKnowledge.execute({ + userMapping: mapping, + projectPermissions: projectPermissions(people.reporter.id), + }, + } + const denied = await connectorRequest(knowledgeBaseId, createBody, undefined, readerCookie) + expect(denied.status).toBe(403) + const unsupported = await connectorRequest(knowledgeBaseId, { + ...createBody, + accessMode: 'workspace', + }) + expect(unsupported.status).toBe(400) + const inaccessibleProject = await connectorRequest(knowledgeBaseId, { + ...createBody, + sourceConfig: { ...createBody.sourceConfig, project: 'missing-fixture-project' }, + }) + expect(inaccessibleProject.status).toBe(400) + const createdResponse = await connectorRequest(knowledgeBaseId, createBody) + const created = await createdResponse.json() + expect(createdResponse.status, JSON.stringify(created)).toBe(201) + let connector = created.data as ConnectorData + expect(connector.permissionConfig).toMatchObject({ + provider: 'gitlab', + mode: 'csv', + revision: 1, + userMapping: { rowCount: 2 }, + }) + expect(JSON.stringify(created)).not.toContain(indexingToken) + expect(JSON.stringify(created)).not.toContain(people.reporter.email) + expect(connector.sourceConfig).not.toHaveProperty('permissionConfig') + const storedConnector = await waitForSync(connector.id) + expect(storedConnector.encryptedApiKey).not.toBe(indexingToken) + expect( + (await decryptApiKey(storedConnector.encryptedApiKey!)).decrypted === indexingToken + ).toBe(true) + const docs = await db.select().from(document).where(eq(document.connectorId, connector.id)) + const ordinary = docs.find((doc) => doc.externalId === `issue:${issueIid}`)! + expect(ordinary.processingStatus).toBe('completed') + const excluded = docs.find((doc) => doc.externalId === `issue:${confidentialIid}`) + if (excluded) { + expect(excluded.acl).toEqual([]) + expect(excluded.storageKey).toBeNull() + } + const searchFor = async (person: GitLabPerson) => + ( + await searchKnowledge.execute({ + principal: principal(person), + input: { + ...owner, + knowledgeBaseIds: [knowledgeBaseId], + query: 'Orion', + topK: 100, + }, + }) + ).results.map((row) => row.documentId) + const assertReads = async (person: GitLabPerson, allowed: boolean) => { + expect((await searchFor(person)).includes(ordinary.id)).toBe(allowed) + const operations: Array<() => Promise> = [ + () => + readKnowledgeDocument.execute({ principal: principal(person), - input: { - ...owner, - knowledgeBaseIds: [knowledgeBaseId], - query: 'Orion', - topK: 100, - }, - }) - ).results.map((row) => row.documentId) - const assertReads = async (person: GitLabPerson, allowed: boolean) => { - expect((await searchFor(person)).includes(ordinary.id)).toBe(allowed) - const operations: Array<() => Promise> = [ - () => - readKnowledgeDocument.execute({ - principal: principal(person), - input: { knowledgeBaseId, documentId: ordinary.id }, - }), - () => - listKnowledgeChunks.execute({ - principal: principal(person), - input: { knowledgeBaseId, documentId: ordinary.id, limit: 10, offset: 0 }, - }), - ] - const download = () => - downloadFileFromUrl(ordinary.fileUrl!, { userId: person.simId, knowledgeAccess: 'user' }) - if (isSearchIndex) { - await expect(download()).rejects.toThrow('Access denied') - operations.push(() => - readSearchDocument.execute({ - principal: principal(person), - input: { - documentId: ordinary.id, - assertedOrganizationId: ids.organizationId, - limit: 3, - resultSecretRegistry: new ResolvedSecretTraceRegistry(), - }, - }) - ) - } else operations.push(download) - for (const operation of operations) { - if (allowed) await expect(operation()).resolves.toBeDefined() - else await expect(operation()).rejects.toBeDefined() - } + input: { knowledgeBaseId, documentId: ordinary.id }, + }), + () => + listKnowledgeChunks.execute({ + principal: principal(person), + input: { knowledgeBaseId, documentId: ordinary.id, limit: 10, offset: 0 }, + }), + ] + const download = () => + downloadFileFromUrl(ordinary.fileUrl!, { userId: person.simId, knowledgeAccess: 'user' }) + operations.push(download) + for (const operation of operations) { + if (allowed) await expect(operation()).resolves.toBeDefined() + else await expect(operation()).rejects.toBeDefined() } - await assertReads(people.reporter, true) - await assertReads(people.guest, false) - await assertReads(people.outsider, false) - const vectors = await db - .select({ content: embedding.content }) - .from(embedding) - .where(eq(embedding.documentId, ordinary.id)) - expect(vectors.map((row) => row.content).join('\n')).not.toContain( - 'INTERNAL_FIXTURE_NOTE_MUST_NOT_BE_INDEXED' - ) - const before = fixture.embeddingCalls - const tooLarge = await connectorRequest( - knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 1, - userMapping: { filename: 'large.csv', content: 'x'.repeat(4 * 1024 * 1024 + 1) }, - }, - }, - connector.id - ) - expect(tooLarge.status).toBe(400) - const overRequestLimit = await connectorRequest( - knowledgeBaseId, - { - sourceConfig: { description: 'x'.repeat(10 * 1024 * 1024) }, + } + await assertReads(people.reporter, true) + await assertReads(people.guest, false) + await assertReads(people.outsider, false) + const vectors = await db + .select({ content: embedding.content }) + .from(embedding) + .where(eq(embedding.documentId, ordinary.id)) + expect(vectors.map((row) => row.content).join('\n')).not.toContain( + 'INTERNAL_FIXTURE_NOTE_MUST_NOT_BE_INDEXED' + ) + const before = fixture.embeddingCalls + const tooLarge = await connectorRequest( + knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 1, + userMapping: { filename: 'large.csv', content: 'x'.repeat(4 * 1024 * 1024 + 1) }, }, - connector.id - ) - expect(overRequestLimit.status).toBe(413) - const invalid = await connectorRequest( - knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 1, - userMapping: { filename: 'bad.csv', content: '1,not-an-email' }, - }, + }, + connector.id + ) + expect(tooLarge.status).toBe(400) + const overRequestLimit = await connectorRequest( + knowledgeBaseId, + { + sourceConfig: { description: 'x'.repeat(10 * 1024 * 1024) }, + }, + connector.id + ) + expect(overRequestLimit.status).toBe(413) + const invalid = await connectorRequest( + knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 1, + userMapping: { filename: 'bad.csv', content: '1,not-an-email' }, }, - connector.id - ) - expect(invalid.status).toBe(400) - await assertReads(people.reporter, true) - const replace = await connectorRequest( - knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 1, - projectPermissions: projectPermissions(people.guest.id), - }, + }, + connector.id + ) + expect(invalid.status).toBe(400) + await assertReads(people.reporter, true) + const replace = await connectorRequest( + knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 1, + projectPermissions: projectPermissions(people.guest.id), }, - connector.id - ) - expect(replace.status).toBe(200) - connector = (await replace.json()).data - expect(connector.permissionConfig?.revision).toBe(2) - await assertReads(people.reporter, false) - await assertReads(people.guest, true) - expect(fixture.embeddingCalls).toBe(before) - const stale = await connectorRequest( - knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 1, - projectPermissions: projectPermissions(people.reporter.id), - }, + }, + connector.id + ) + expect(replace.status).toBe(200) + connector = (await replace.json()).data + expect(connector.permissionConfig?.revision).toBe(2) + await assertReads(people.reporter, false) + await assertReads(people.guest, true) + expect(fixture.embeddingCalls).toBe(before) + const stale = await connectorRequest( + knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 1, + projectPermissions: projectPermissions(people.reporter.id), }, - connector.id - ) - expect(stale.status).toBe(409) - const concurrent = await Promise.all( - [people.reporter, people.guest].map((person) => - connectorRequest( - knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 2, - projectPermissions: projectPermissions(person.id), - }, - }, - connector.id - ) - ) - ) - expect(concurrent.map((result) => result.status).sort()).toEqual([200, 409]) - const grants = await db - .select() - .from(knowledgeConnectorPermissionGrant) - .where(eq(knowledgeConnectorPermissionGrant.connectorId, connector.id)) - expect(grants).toHaveLength(1) - let winner = - grants[0].subjectToken === `u:${people.reporter.email}` ? people.reporter : people.guest - await assertReads(winner, true) - expect(fixture.embeddingCalls).toBe(before) - await db.update(user).set({ emailVerified: false }).where(eq(user.id, winner.simId)) - await assertReads(winner, false) - await db.update(user).set({ emailVerified: true }).where(eq(user.id, winner.simId)) - const removedMember = winner - const newlyMappedMember = winner === people.reporter ? people.guest : people.reporter - const remapped = await connectorRequest( - knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 3, - userMapping: { - filename: 'replacement-users.csv', - content: `${winner.id},${newlyMappedMember.email}`, + }, + connector.id + ) + expect(stale.status).toBe(409) + const concurrent = await Promise.all( + [people.reporter, people.guest].map((person) => + connectorRequest( + knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 2, + projectPermissions: projectPermissions(person.id), }, }, - }, - connector.id + connector.id + ) ) - expect(remapped.status).toBe(200) - await assertReads(removedMember, false) - winner = newlyMappedMember - await assertReads(winner, true) - expect(fixture.embeddingCalls).toBe(before) - const crossOwner = await connectorRequest( - ids.knowledgeBaseId, - { - permissionConfig: { - provider: 'gitlab', - mode: 'csv', - expectedRevision: 3, - projectPermissions: projectPermissions(people.reporter.id), + ) + expect(concurrent.map((result) => result.status).sort()).toEqual([200, 409]) + const grants = await db + .select() + .from(knowledgeConnectorPermissionGrant) + .where(eq(knowledgeConnectorPermissionGrant.connectorId, connector.id)) + expect(grants).toHaveLength(1) + let winner = + grants[0].subjectToken === `u:${people.reporter.email}` ? people.reporter : people.guest + await assertReads(winner, true) + expect(fixture.embeddingCalls).toBe(before) + await db.update(user).set({ emailVerified: false }).where(eq(user.id, winner.simId)) + await assertReads(winner, false) + await db.update(user).set({ emailVerified: true }).where(eq(user.id, winner.simId)) + const removedMember = winner + const newlyMappedMember = winner === people.reporter ? people.guest : people.reporter + const remapped = await connectorRequest( + knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 3, + userMapping: { + filename: 'replacement-users.csv', + content: `${winner.id},${newlyMappedMember.email}`, }, }, - connector.id - ) - expect(crossOwner.status).toBe(404) - await api(`/projects/${projectId}/issues/${issueIid}`, 'PUT', { confidential: true }) - const syncResult = await executeSync(connector.id, { - billingAttribution: attribution, - fullSync: true, - }) - expect(syncResult.docsFailed).toBe(0) - await assertReads(winner, false) - const [removed] = await db.select().from(document).where(eq(document.id, ordinary.id)) - expect(removed.storageKey).toBeNull() - expect(removed.acl).toEqual([]) - expect( - await db.select().from(embedding).where(eq(embedding.documentId, ordinary.id)) - ).toEqual([]) - const [snapshot] = await db - .select() - .from(knowledgeConnectorPermissionSnapshot) - .where(eq(knowledgeConnectorPermissionSnapshot.connectorId, connector.id)) - expect(snapshot.revision).toBe(4) - await connectorRequest(knowledgeBaseId, { status: 'paused' }, connector.id) - const switched = await connectorRequest( - knowledgeBaseId, - { - apiKey: adminToken, - permissionConfig: { provider: 'gitlab', mode: 'administrator', expectedRevision: 4 }, + }, + connector.id + ) + expect(remapped.status).toBe(200) + await assertReads(removedMember, false) + winner = newlyMappedMember + await assertReads(winner, true) + expect(fixture.embeddingCalls).toBe(before) + const crossOwner = await connectorRequest( + ids.knowledgeBaseId, + { + permissionConfig: { + provider: 'gitlab', + mode: 'csv', + expectedRevision: 3, + projectPermissions: projectPermissions(people.reporter.id), }, - connector.id - ) - expect(switched.status).toBe(200) - expect((await switched.json()).data.accessRewritePending).toBe(true) - await assertReads(winner, false) - await connectorRequest(knowledgeBaseId, { status: 'active' }, connector.id) - await executeSync(connector.id, { - billingAttribution: attribution, - fullSync: true, - }) - const [restored] = await db - .select() - .from(knowledgeConnector) - .where(eq(knowledgeConnector.id, connector.id)) - expect(restored.accessRewritePending).toBe(false) - }, - 180000 - ) + }, + connector.id + ) + expect(crossOwner.status).toBe(404) + await api(`/projects/${projectId}/issues/${issueIid}`, 'PUT', { confidential: true }) + const syncResult = await executeSync(connector.id, { + billingAttribution: attribution, + fullSync: true, + }) + expect(syncResult.docsFailed).toBe(0) + await assertReads(winner, false) + const [removed] = await db.select().from(document).where(eq(document.id, ordinary.id)) + expect(removed.storageKey).toBeNull() + expect(removed.acl).toEqual([]) + expect(await db.select().from(embedding).where(eq(embedding.documentId, ordinary.id))).toEqual( + [] + ) + const [snapshot] = await db + .select() + .from(knowledgeConnectorPermissionSnapshot) + .where(eq(knowledgeConnectorPermissionSnapshot.connectorId, connector.id)) + expect(snapshot.revision).toBe(4) + await connectorRequest(knowledgeBaseId, { status: 'paused' }, connector.id) + const switched = await connectorRequest( + knowledgeBaseId, + { + apiKey: adminToken, + permissionConfig: { provider: 'gitlab', mode: 'administrator', expectedRevision: 4 }, + }, + connector.id + ) + expect(switched.status).toBe(200) + expect((await switched.json()).data.accessRewritePending).toBe(true) + await assertReads(winner, false) + await connectorRequest(knowledgeBaseId, { status: 'active' }, connector.id) + await executeSync(connector.id, { + billingAttribution: attribution, + fullSync: true, + }) + const [restored] = await db + .select() + .from(knowledgeConnector) + .where(eq(knowledgeConnector.id, connector.id)) + expect(restored.accessRewritePending).toBe(false) + }, 180000) }) diff --git a/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts b/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts index 716af7b1b66..af955d4df2b 100644 --- a/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/gmail-member.integration.ts @@ -11,7 +11,6 @@ import { credentialGroupEnrollment, document, embedding, - knowledgeBase, knowledgeConnector, knowledgeConnectorMember, knowledgeDocumentObservation, @@ -24,12 +23,6 @@ import { eq, inArray } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' const counters = vi.hoisted(() => ({ embeddedTexts: 0 })) -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, @@ -224,10 +217,6 @@ describe('Gmail member ingestion and ACLs in PostgreSQL (provider fixtures)', () credentialGroupId: fixture.groupId, credentialGroupOptionId: fixture.optionId, }) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) await db .update(credentialGroup) .set({ diff --git a/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts b/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts index 2a66fdbf151..4d406455565 100644 --- a/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/google-calendar-member.integration.ts @@ -12,7 +12,6 @@ import { credentialGroup, credentialGroupEnrollment, document, - knowledgeBase, knowledgeConnector, knowledgeConnectorMember, knowledgeDocumentObservation, @@ -24,12 +23,6 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, isNull } from 'drizzle-orm' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, @@ -208,10 +201,6 @@ describe('Google Calendar member indexing and authorization in PostgreSQL', () = }) }) await seedKnowledgeAclFixture(ids) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) const policy = await getCredentialGroupProviderAdapter('google-calendar').getPolicy(undefined, { workspaceId: ids.workspaceId, }) diff --git a/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts b/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts index e9d76507746..b3ae46cd07f 100644 --- a/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/jira-member.integration.ts @@ -13,7 +13,6 @@ import { credentialGroupEnrollment, document, embedding, - knowledgeBase, knowledgeConnector, knowledgeConnectorMember, knowledgeDocumentObservation, @@ -25,12 +24,6 @@ import { generateId } from '@sim/utils/id' import { and, eq, inArray, isNull } from 'drizzle-orm' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/lib/embeddings', async () => ({ ...(await import('@/lib/embeddings/client')), assertKnowledgeEmbeddingCapacity: async () => {}, @@ -188,10 +181,6 @@ describe('Jira member indexing and authorization in PostgreSQL', () => { }) }) await seedKnowledgeAclFixture(ids) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) const policy = await getCredentialGroupProviderAdapter('jira').getPolicy(undefined, { workspaceId: ids.workspaceId, }) diff --git a/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts b/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts index 9eff697046f..7ebef980cfb 100644 --- a/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/kb-block-search.integration.ts @@ -3,19 +3,15 @@ import type { Principal } from '@sim/auth/principal' import { db } from '@sim/db' import { document, embedding, knowledgeBase, organization, user, workspace } from '@sim/db/schema' import { generateId } from '@sim/utils/id' -import { eq, inArray, sql } from 'drizzle-orm' +import { eq, inArray } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it } from 'vitest' import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' -import { VECTOR_PROBE_DOCUMENT_LIMIT } from '@/lib/knowledge/search/candidates' import { retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' import { embeddingVectorValues } from '@/lib/knowledge/vector-columns' -import { resolveSearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' -import { resolvePermittedDocuments } from '@/lib/sim-search/indexed/retrieval/permitted' -import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' describe('API-key KB block fan-out', () => { const ids = createKnowledgeAclFixtureIds() @@ -92,8 +88,6 @@ describe('API-key KB block fan-out', () => { it.each([false, true])( 'completes 18 concurrent KB searches with access checks intact (tag filter: %s)', async (withTags) => { - /** A cold process must also skip the global readiness probe for ordinary KBs. */ - forgetProjectionFilled() const previousDebug = db.$client.options.debug const statements: string[] = [] db.$client.options.debug = (_connection, query) => { @@ -159,37 +153,4 @@ describe('API-key KB block fan-out', () => { } } ) - - it('bounds the permitted set by the requested bases, not by what the tokens reach elsewhere', async () => { - const crowded = generateId() - await db.insert(knowledgeBase).values({ - id: crowded, - userId: ids.aliceId, - workspaceId: ids.workspaceId, - name: 'Crowded neighbour', - }) - try { - /** Baseline tokens are shared by every tenant, so another base can hold more than the limit. */ - await db.execute(sql` - INSERT INTO ${document} (id, knowledge_base_id, filename, file_url, file_size, mime_type, - processing_status, acl) - SELECT 'crowded-' || n, ${crowded}, 'crowded', 'https://fixture.invalid/crowded', 1, - 'text/plain', 'completed', ARRAY['ws']::text[] - FROM generate_series(1, ${VECTOR_PROBE_DOCUMENT_LIMIT + 1}) AS n - `) - const access = { kind: 'user' as const, userId: ids.bobId, tokens: ['pub', 'ws'] } - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: [bases[0].id], - access, - accessPlan: await resolveSearchAccessPlan([bases[0].id], access), - filtered: false, - }) - expect(permitted).toEqual({ - kind: 'bounded', - documents: [{ id: bases[0].visible, connectorId: null }], - }) - } finally { - await db.delete(knowledgeBase).where(eq(knowledgeBase.id, crowded)) - } - }) }) diff --git a/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts b/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts index 56ff0027e65..3b9f4f8a174 100644 --- a/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/knowledge-projection.integration.ts @@ -1,232 +1,43 @@ -/** - * The knowledge projector and the readers that must stay correct while it lags. A GitHub member - * source's document in a search index is changed by a writer in either projection mode — - * synchronous, as every writer now is, or deferred, as writers of earlier releases could be, - * leaving only a mark — and search is checked before the - * projector runs: a revoked member is refused and a granted one is served, on the vector and - * keyword legs and under the source filter, and a disabled or deleted chunk is gone at once. The - * projector's own contract follows: it converges the rows and removes the mark, keeps a mark that - * a write bumped during its pass, survives a document deleted under it, writes in pages bounded by - * chunk rows, releases workspace marks with nothing to project without a pass, and a deferred - * commit writes no projection row at all. - */ -import { createHash } from 'node:crypto' +/** Real PostgreSQL proof that legacy deferred vector repair survives indexed Search retirement. */ import { db } from '@sim/db' import { hasKnowledgeProjectionWork, - type KnowledgeProjection, releaseSettledMarks, runKnowledgeProjection, } from '@sim/db/knowledge-projection' import { - credential, - credentialGroup, - credentialGroupEnrollment, document, embedding, embeddingKeywordSearch, embeddingKeywordTin, embeddingSearch, knowledgeBase, - knowledgeConnector, - knowledgeConnectorMember, - knowledgeDocumentObservation, knowledgeProjectionDirty, organization, user, workspace, } from '@sim/db/schema' -import { sleep } from '@sim/utils/helpers' import { generateId } from '@sim/utils/id' -import { and, eq, inArray, sql } from 'drizzle-orm' +import { eq, inArray, sql } from 'drizzle-orm' import postgres from 'postgres' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' - -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) -/** The TINQL `resolveTinKeywordQuery` renders for `fixture`: its `english` stem, quoted. */ -vi.mock('@/lib/sim-search/indexed/retrieval/tin-keyword', () => ({ - resolveTinKeywordQuery: async () => '"fixtur"', -})) - -vi.mock('@/lib/core/config/feature-flags', () => ({ isFeatureEnabled: async () => false })) - import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import type { - GitHubInstallationReadGrant, - KnowledgeAccessProvider, - UserAccessScope, -} from '@/lib/knowledge/access/types' -import { leaseTransaction } from '@/lib/knowledge/connectors/sync-lock' -import { liveSourceAccessForConnectors } from '@/lib/knowledge/search/candidates' -import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' -import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' -import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' -import type { IndexedRetrievalContext } from '@/lib/sim-search/indexed/retrieval/permitted' -import { projectionCandidateAccessCondition } from '@/lib/sim-search/indexed/retrieval/projection-access' -import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' -import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' const ids = createKnowledgeAclFixtureIds() -const connectorId = generateId() -const otherConnectorId = generateId() -const contentCredentialId = generateId() -const groupId = generateId() -const optionId = generateId() +const searchBaseId = generateId() const documentId = generateId() -const chunkId = generateId() -const repositoryId = '4242' -const members = { - alice: { id: generateId(), subject: 'alice-gh', credentialId: generateId() }, - bob: { id: generateId(), subject: 'bob-gh', credentialId: generateId() }, -} -const subjectToken = (subject: string) => `s:github-repositories:-:${subject}` -const aclOf = (...who: Array<'alice' | 'bob'>) => - who.map((name) => subjectToken(members[name].subject)).sort() -/** - * A direction no other integration file writes, so no row of theirs ties with this file's chunks. - * Inside the first 512 dimensions, the only ones the candidate projection keeps. - */ +const searchDocumentId = generateId() const vector = Array.from({ length: 1536 }, (_, index) => (index === 511 ? 1 : 0)) -const queryVector = { - vector: JSON.stringify(vector), - dimensions: 1536 as const, - model: 'text-embedding-3-small', -} - -/** Alice holds the installation grant, so what she reads is decided by the document's ACL alone. */ -const scope: UserAccessScope = { - kind: 'user', - userId: ids.aliceId, - tokens: [ - 'pub', - subjectToken(members.alice.subject), - `u:${ids.aliceId}@fixture.test`, - 'ws', - ].sort(), -} -const plan: SearchAccessPlan = { - connectors: { - workspace: [], - admin: [], - members: [connectorId], - liveProofRequired: [connectorId], - }, - observers: { confirmed: [{ id: members.alice.id, connectorId }], observed: [] }, - memberSources: [connectorId], - connectorTypes: new Map([[connectorId, 'github']]), - uploads: false, -} -const grant: GitHubInstallationReadGrant = { - connectorId, - contentCredentialId, - readerCredentialId: members.alice.credentialId, - readerSubjectToken: subjectToken(members.alice.subject), - repositoryId, -} - -const searchInputs = { - knowledgeBaseIds: [ids.knowledgeBaseId], - topK: 20, - access: scope, - queryVector, -} - -/** A narrow reader of the search index, who proves the installation grant live. */ -function searchContext(): IndexedRetrievalContext { - const granted = { ...scope, githubInstallationGrants: [grant] } - const accessProvider: KnowledgeAccessProvider = { - get: async () => scope, - getForConnectors: async () => granted, - getForDocuments: async () => granted, - liveSourceConnectorCondition: async () => null, - } - return { - access: scope, - accessPlan: plan, - filtered: false, - permitted: { kind: 'unbounded', broad: false }, - liveSourceAccess: liveSourceAccessForConnectors( - plan.connectors.liveProofRequired, - accessProvider - ), - } -} - -const keywordIds = async () => - (await executeIndexedKeywordSearch({ ...searchInputs, query: 'fixture' }, searchContext())) - .map((row) => row.id) - .sort() - -/** - * Ranked exactly on the row, under the same visibility predicate the graph walk applies. The walk - * is approximate: in a graph the other files sharing this database crowd with degenerate vectors, - * a chunk can be pruned from every neighbour list and never be reached, however far the walk goes. - */ -const vectorIds = async () => - (await selectIndexedVectorResults({ ...searchInputs, distanceThreshold: 2 }, searchContext())) - .map((row) => row.id) - .sort() - -/** The chunks of the fixture document each ranking projection admits for Alice. */ -async function admitted(): Promise> { - const [vectorRows, keywordRows] = await Promise.all([ - db - .select({ id: embeddingSearch.id }) - .from(embeddingSearch) - .where( - and( - eq(embeddingSearch.documentId, documentId), - projectionCandidateAccessCondition(embeddingSearch, scope, plan, { filled: true }) - ) - ), - db - .select({ id: embeddingKeywordTin.id }) - .from(embeddingKeywordTin) - .where( - and( - eq(embeddingKeywordTin.documentId, documentId), - projectionCandidateAccessCondition(embeddingKeywordTin, scope, plan, { filled: true }) - ) - ), - ]) - return { - vector: vectorRows.map((row) => row.id).sort(), - keyword: keywordRows.map((row) => row.id).sort(), - } -} - -type Mode = 'sync' | 'async' - -/** - * What a writer of a release that deferred its projection selected first: the setting that skips - * the synchronous projection triggers, which the database still honours. - */ -const DEFER_PROJECTION = `SELECT set_config('sim.projection_mode', 'async', true)` - -/** Runs a write in a transaction of the given projection mode, as a knowledge writer would. */ -function write( - mode: Mode, - work: (tx: Parameters[0]>[0]) => Promise -) { - return db.transaction(async (tx) => { - if (mode === 'async') await tx.execute(sql.raw(DEFER_PROJECTION)) - await work(tx) - }) -} +let projector: postgres.Sql -function chunkRow(id: string, chunkIndex: number) { +function chunkRow(id: string, chunkIndex: number, search = false) { return { id, - documentId, - knowledgeBaseId: ids.knowledgeBaseId, + documentId: search ? searchDocumentId : documentId, + knowledgeBaseId: search ? searchBaseId : ids.knowledgeBaseId, chunkIndex, chunkHash: `fixture-hash-${chunkIndex}`, content: 'fixture readme', @@ -239,737 +50,311 @@ function chunkRow(id: string, chunkIndex: number) { } } -const markOf = async () => { - const [row] = await db - .select() - .from(knowledgeProjectionDirty) - .where(eq(knowledgeProjectionDirty.documentId, documentId)) - return row +/** Reproduces pending writes from the older release that deferred its projections. */ +async function defer( + work: (tx: Parameters[0]>[0]) => Promise +) { + await db.transaction(async (tx) => { + await tx.execute(sql`SELECT set_config('sim.projection_mode', 'async', true)`) + await work(tx) + }) } -const rowAcl = async (table: typeof embeddingSearch | typeof embeddingKeywordTin, id = chunkId) => { +const markOf = async (id = documentId) => { const [row] = await db - .select({ acl: table.acl, connectorId: table.connectorId, enabled: table.enabled }) - .from(table) - .where(eq(table.id, id)) + .select() + .from(knowledgeProjectionDirty) + .where(eq(knowledgeProjectionDirty.documentId, id)) return row } -/** The projector's connection: a pass holds its per-document advisory locks on it. */ -let projector: postgres.Sql -const project = (options: Partial[1]> = {}) => - runKnowledgeProjection(projector, { searchIndexes: true, ...options }) - -/** Shims for the Tin extension, which the test database does not carry. */ -let createdTinShims = false +const vectorsOf = (id = documentId) => + db + .select({ + id: embeddingSearch.id, + enabled: embeddingSearch.enabled, + vector512: embeddingSearch.vector512, + acl: embeddingSearch.acl, + connectorId: embeddingSearch.connectorId, + }) + .from(embeddingSearch) + .where(eq(embeddingSearch.documentId, id)) beforeAll(async () => { - projector = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) await seedKnowledgeAclFixture(ids) - const now = new Date() - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - await db.insert(credential).values({ - id: contentCredentialId, - workspaceId: ids.workspaceId, - type: 'service_account', - displayName: 'Fixture GitHub installation', - createdBy: ids.aliceId, - providerId: GITHUB_INSTALLATION_PROVIDER_ID, - }) - await db.insert(credentialGroup).values({ - id: groupId, + projector = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) + await db.insert(knowledgeBase).values({ + id: searchBaseId, + userId: ids.aliceId, workspaceId: ids.workspaceId, - publicId: generateId(), - name: 'GitHub readers', - options: [ - { - id: optionId, - provider: 'github-repositories', - label: 'GitHub fixture', - authorizationAppId: 'fixture-app', - requiredScopes: ['repo'], - scopeVersion: 1, - required: false, - status: 'active', - }, - ], - } as typeof credentialGroup.$inferInsert) - for (const who of ['alice', 'bob'] as const) { - const userId = who === 'alice' ? ids.aliceId : ids.bobId - const [enrollment] = await db - .insert(credentialGroupEnrollment) - .values({ - id: generateId(), - credentialGroupId: groupId, - userId, - email: `${userId}@fixture.test`, - status: 'completed', - invitationTokenHash: createHash('sha256').update(generateId()).digest('hex'), - invitationExpiresAt: new Date(Date.now() + 60 * 60 * 1000), - invitedAt: now, - }) - .returning({ id: credentialGroupEnrollment.id }) - await db.insert(credential).values({ - id: members[who].credentialId, - workspaceId: ids.workspaceId, - type: 'managed_oauth', - displayName: 'Fixture GitHub reader', - providerId: 'github-repositories', - authorizationAppId: 'fixture-app', - credentialGroupEnrollmentId: enrollment!.id, - credentialGroupOptionId: optionId, - managedOauthScopeVersion: 1, - providerSubjectId: members[who].subject, - providerTenantId: '', - managedOauthStatus: 'active', - grantedScopes: ['repo'], - encryptedOauthTokenSet: 'fixture-not-an-oauth-token', - grantedAt: now, - createdBy: userId, - }) - } - await db.insert(knowledgeConnector).values( - [connectorId, otherConnectorId].map((id) => ({ - id, - knowledgeBaseId: ids.knowledgeBaseId, - connectorType: 'github', - sourceConfig: { githubRepositoryId: repositoryId }, - accessMode: 'members', - status: 'active', - credentialId: contentCredentialId, - credentialGroupId: groupId, - credentialGroupOptionId: optionId, - })) - ) - await db.insert(knowledgeConnectorMember).values( - (['alice', 'bob'] as const).map((who) => ({ - id: members[who].id, - workspaceId: ids.workspaceId, - connectorId, - credentialId: members[who].credentialId, - subjectToken: subjectToken(members[who].subject), - status: 'active', - memberSyncedThrough: now, - })) - ) - await db.insert(document).values({ - id: documentId, - connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: 'fixture-file', - filename: 'readme.md', - fileUrl: 'https://fixture.test/readme', - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed', - acl: aclOf('alice', 'bob'), + name: 'Retired projection fixture', + chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, }) - await db.insert(knowledgeDocumentObservation).values( - (['alice', 'bob'] as const).map((who) => ({ - documentId, - memberId: members[who].id, - lastSeenAt: now, - runId: generateId(), +}) + +beforeEach(async () => { + await db + .delete(document) + .where(inArray(document.knowledgeBaseId, [ids.knowledgeBaseId, searchBaseId])) + await db + .update(knowledgeBase) + .set({ isSearchIndex: false }) + .where(inArray(knowledgeBase.id, [ids.knowledgeBaseId, searchBaseId])) + await db.insert(document).values( + [ + { id: documentId, knowledgeBaseId: ids.knowledgeBaseId }, + { id: searchDocumentId, knowledgeBaseId: searchBaseId }, + ].map((row) => ({ + ...row, + filename: 'repair.md', + fileUrl: 'https://fixture.test/repair', + fileSize: 12, + mimeType: 'text/plain', + processingStatus: 'completed', })) ) - const [tin] = await db.execute<{ present: boolean }>( - sql`SELECT to_regnamespace('tin') IS NOT NULL AS present` - ) - if (!tin?.present) { - createdTinShims = true - await db.execute( - sql.raw(`CREATE SCHEMA tin; - CREATE FUNCTION tin.full_score(tid) RETURNS double precision LANGUAGE sql IMMUTABLE AS 'SELECT 1.0::float8'; - CREATE FUNCTION knowledge_tin_base_token(text) RETURNS text LANGUAGE sql IMMUTABLE AS $$SELECT 'kb'$$; - CREATE FUNCTION knowledge_tin_stream(vector tsvector) RETURNS text LANGUAGE sql IMMUTABLE AS $$ - SELECT coalesce(string_agg(entry.lexeme, ' ' ORDER BY position), '') - FROM unnest(vector) AS entry(lexeme, positions, weights), unnest(entry.positions) AS position - $$; - CREATE FUNCTION tin_fixture_match(text, text) RETURNS boolean LANGUAGE sql IMMUTABLE AS 'SELECT true'; - CREATE OPERATOR ==> (LEFTARG = text, RIGHTARG = text, FUNCTION = tin_fixture_match);`) - ) - } -}, 60_000) +}) afterAll(async () => { - if (createdTinShims) { - await db.execute( - sql.raw(`DROP OPERATOR IF EXISTS ==> (text, text); - DROP FUNCTION IF EXISTS tin_fixture_match(text, text); - DROP FUNCTION IF EXISTS knowledge_tin_base_token(text); - DROP FUNCTION IF EXISTS knowledge_tin_stream(tsvector); - DROP SCHEMA IF EXISTS tin CASCADE;`) - ) - } await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) - await db.delete(credentialGroup).where(eq(credentialGroup.id, groupId)) await db.delete(organization).where(eq(organization.id, ids.organizationId)) await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) await projector?.end() - forgetProjectionFilled() -}) - -/** - * Every test starts from one converged chunk readable by Alice and Bob: whatever the last test - * left is removed, the chunk is written again, and the projector converges it. It is written - * deferred, so the projector writes all three projections: this database has no Tin extension, - * so no synchronous trigger would write the Tin row. - */ -beforeEach(async () => { - await db.delete(embedding).where(eq(embedding.documentId, documentId)) - await db - .update(document) - .set({ acl: aclOf('alice', 'bob'), connectorId }) - .where(eq(document.id, documentId)) - await write('async', (tx) => tx.insert(embedding).values(chunkRow(chunkId, 0))) - await project() - forgetProjectionFilled() }) -describe.each(['sync', 'async'] as const)('a %s writer', (mode) => { - it('revokes a grant before the projector runs, on both rankings', async () => { - await write(mode, (tx) => - tx - .update(document) - .set({ acl: aclOf('bob') }) - .where(eq(document.id, documentId)) - ) - expect(await markOf()).toMatchObject({ content: false }) - if (mode === 'async') { - /** The rows still name Alice: only the mark keeps her out. */ - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('alice', 'bob')) - expect((await rowAcl(embeddingKeywordTin))?.acl).toEqual(aclOf('alice', 'bob')) - } - expect(await admitted()).toEqual({ vector: [], keyword: [] }) - expect(await vectorIds()).toEqual([]) - expect(await keywordIds()).toEqual([]) - - await project() +describe('ordinary KB vector repair after indexed Search retirement', () => { + it('repairs deferred vectors in bounded pages without creating copied ACL or keyword rows', async () => { + const chunks = Array.from({ length: 5 }, (_, index) => chunkRow(generateId(), index)) + await defer((tx) => tx.insert(embedding).values(chunks)) + expect(await vectorsOf()).toEqual([]) + expect(await markOf()).toMatchObject({ content: true }) + const writes: number[] = [] + await runKnowledgeProjection(projector, { + pageSize: 2, + onPage: (page) => { + if (page.documentId === documentId) writes.push(page.written) + }, + }) + expect(writes).toEqual([2, 2, 1]) + const rows = await vectorsOf() + expect(rows.map((row) => row.id).sort()).toEqual(chunks.map((row) => row.id).sort()) + expect( + rows.every( + (row) => row.vector512?.[511] === 1 && row.acl === null && row.connectorId === null + ) + ).toBe(true) + expect( + await db + .select() + .from(embeddingKeywordSearch) + .where(eq(embeddingKeywordSearch.documentId, documentId)) + ).toEqual([]) + expect( + await db + .select() + .from(embeddingKeywordTin) + .where(eq(embeddingKeywordTin.documentId, documentId)) + ).toEqual([]) expect(await markOf()).toBeUndefined() - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) - expect((await rowAcl(embeddingKeywordTin))?.acl).toEqual(aclOf('bob')) - expect(await admitted()).toEqual({ vector: [], keyword: [] }) }) - it('serves a new grant before the projector runs, on both rankings', async () => { + it('settles retired Search content marks without recreating any candidate projection', async () => { await db - .update(document) - .set({ acl: aclOf('bob') }) - .where(eq(document.id, documentId)) - await project() - expect(await vectorIds()).toEqual([]) - - await write(mode, (tx) => - tx - .update(document) - .set({ acl: aclOf('alice', 'bob') }) - .where(eq(document.id, documentId)) - ) - if (mode === 'async') expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) - expect(await admitted()).toEqual({ vector: [chunkId], keyword: [chunkId] }) - expect(await vectorIds()).toEqual([chunkId]) - expect(await keywordIds()).toEqual([chunkId]) - }) - - it('decides a document that moved to a source outside the plan on the document', async () => { - await write(mode, (tx) => - tx.update(document).set({ connectorId: otherConnectorId }).where(eq(document.id, documentId)) - ) - if (mode === 'async') expect((await rowAcl(embeddingSearch))?.connectorId).toBe(connectorId) - expect(await admitted()).toEqual({ vector: [], keyword: [] }) - expect(await vectorIds()).toEqual([]) - expect(await keywordIds()).toEqual([]) - await project() - expect((await rowAcl(embeddingSearch))?.connectorId).toBe(otherConnectorId) - }) - - it('hides a disabled chunk at once', async () => { - await write(mode, (tx) => - tx.update(embedding).set({ enabled: false }).where(eq(embedding.id, chunkId)) - ) - expect(await markOf()).toMatchObject({ content: mode === 'async' }) - expect(await vectorIds()).toEqual([]) - expect(await keywordIds()).toEqual([]) - await project() - expect((await rowAcl(embeddingSearch))?.enabled).toBe(false) - }) - - it('removes a deleted chunk from every projection in the deleting statement', async () => { - await write(mode, (tx) => tx.delete(embedding).where(eq(embedding.id, chunkId))) - expect(await rowAcl(embeddingSearch)).toBeUndefined() - expect(await rowAcl(embeddingKeywordTin)).toBeUndefined() - expect(await vectorIds()).toEqual([]) + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where(eq(knowledgeBase.id, searchBaseId)) + await defer((tx) => tx.insert(embedding).values(chunkRow(generateId(), 0, true))) + expect(await markOf(searchDocumentId)).toMatchObject({ content: true }) + await runKnowledgeProjection(projector, {}) + expect(await vectorsOf(searchDocumentId)).toEqual([]) + expect( + await db + .select() + .from(embeddingKeywordSearch) + .where(eq(embeddingKeywordSearch.documentId, searchDocumentId)) + ).toEqual([]) + expect( + await db + .select() + .from(embeddingKeywordTin) + .where(eq(embeddingKeywordTin.documentId, searchDocumentId)) + ).toEqual([]) + expect(await markOf(searchDocumentId)).toBeUndefined() }) - it('makes a new chunk searchable, at once or once the projector runs', async () => { - const added = generateId() - await write(mode, (tx) => tx.insert(embedding).values(chunkRow(added, 1))) - expect(await markOf()).toMatchObject({ content: mode === 'async' }) - const expected = [chunkId, added].sort() - if (mode === 'sync') { - expect(await vectorIds()).toEqual(expected) - } else { - expect(await rowAcl(embeddingSearch, added)).toBeUndefined() - expect(await vectorIds()).toEqual([chunkId]) - } - await project() + it('settles permission-only marks without rewriting vector rows or canonical document access', async () => { + const chunk = chunkRow(generateId(), 0) + await db.insert(embedding).values(chunk) + const [before] = + await projector`SELECT xmin::text AS version FROM embedding_search WHERE id = ${chunk.id}` + await defer((tx) => tx.update(document).set({ acl: [] }).where(eq(document.id, documentId))) + expect(await markOf()).toMatchObject({ content: false }) + await runKnowledgeProjection(projector, {}) + const [after] = + await projector`SELECT xmin::text AS version FROM embedding_search WHERE id = ${chunk.id}` + expect(after.version).toBe(before.version) + const [canonical] = await db + .select({ acl: document.acl }) + .from(document) + .where(eq(document.id, documentId)) + expect(canonical.acl).toEqual([]) expect(await markOf()).toBeUndefined() - expect(await vectorIds()).toEqual(expected) - /** A synchronous writer's Tin row is its trigger's, which only a database with Tin has. */ - if (mode === 'async') expect(await keywordIds()).toEqual(expected) }) -}) -describe('the projector', () => { - it('keeps a mark that a write bumped during its pass, and settles it on the next', async () => { - await db - .update(document) - .set({ acl: aclOf('bob') }) - .where(eq(document.id, documentId)) + it('keeps a concurrent content change for the next pass instead of settling its newer generation', async () => { + const chunks = [chunkRow(generateId(), 0), chunkRow(generateId(), 1)] + await defer((tx) => tx.insert(embedding).values(chunks)) const before = await markOf() - let bumped = false - await project({ - onPage: async () => { - if (bumped) return - bumped = true - await db - .update(document) - .set({ acl: aclOf('alice') }) - .where(eq(document.id, documentId)) + let changed = false + await runKnowledgeProjection(projector, { + pageSize: 1, + onPage: async (page) => { + if (page.documentId !== documentId || changed) return + changed = true + await defer((tx) => + tx.update(embedding).set({ enabled: false }).where(eq(embedding.id, chunks[0].id)) + ) }, }) - expect(bumped).toBe(true) - const after = await markOf() - expect(after?.generation).toBe((before?.generation ?? 0) + 1) - await project() + expect((await markOf())?.generation).toBe((before?.generation ?? 0) + 1) + expect((await vectorsOf()).find((row) => row.id === chunks[0].id)?.enabled).toBe(true) + await runKnowledgeProjection(projector, {}) + expect((await vectorsOf()).find((row) => row.id === chunks[0].id)?.enabled).toBe(false) expect(await markOf()).toBeUndefined() - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('alice')) }) - it('keeps the mark when a synchronous ACL change commits under a page that then writes stale values', async () => { - /** A pending revocation of Bob: the rows still name both until a pass. */ - await write('async', (tx) => + it('stops repairing a base marked as Search between committed pages', async () => { + await defer((tx) => tx - .update(document) - .set({ acl: aclOf('alice') }) - .where(eq(document.id, documentId)) + .insert(embedding) + .values(Array.from({ length: 3 }, (_, index) => chunkRow(generateId(), index))) ) - const before = await markOf() - const [{ pid }] = await projector>`SELECT pg_backend_pid() AS pid` - /** - * A synchronous writer revokes Alice too and holds its transaction open: its fan-out has - * locked the projection rows and its mark bump is not yet visible. - */ - const writer = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) - let release: () => void = () => {} - const held = new Promise((resolve) => { - release = resolve - }) - let locked: () => void = () => {} - const fannedOut = new Promise((resolve) => { - locked = resolve + let changed = false + await runKnowledgeProjection(projector, { + pageSize: 1, + onPage: async (page) => { + if (page.documentId !== documentId || changed) return + changed = true + await db + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) + }, }) - try { - const writing = writer.begin(async (tx) => { - await tx`UPDATE document SET acl = ${aclOf('bob')} WHERE id = ${documentId}` - locked() - await held - }) - await fannedOut - /** - * The pass reads the committed document, Alice alone, and its page blocks on the rows the - * writer holds. Once the writer commits, the page rewrites those rows from what it read. - */ - const pass = project() - await vi.waitFor(async () => { - const [activity] = await db.execute<{ waiting: boolean }>( - sql`SELECT wait_event_type = 'Lock' AS waiting FROM pg_stat_activity WHERE pid = ${pid}` - ) - expect(activity?.waiting).toBe(true) - }) - release() - await writing - await pass - } finally { - release() - await writer.end() - } - /** The page wrote the value it read over the writer's newer one. */ - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('alice')) - /** The writer's bump is what the pass could not settle over: the rows stay decided on the document. */ - expect((await markOf())?.generation).toBe((before?.generation ?? 0) + 1) - expect(await admitted()).toEqual({ vector: [], keyword: [] }) - expect(await vectorIds()).toEqual([]) - await project() + expect(await vectorsOf()).toHaveLength(1) expect(await markOf()).toBeUndefined() - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) - expect((await rowAcl(embeddingKeywordTin))?.acl).toEqual(aclOf('bob')) - expect(await admitted()).toEqual({ vector: [], keyword: [] }) - }) - - it('passes over a document deleted while it is being projected', async () => { - const doomed = generateId() - await db.insert(document).values({ - id: doomed, - connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: 'doomed-file', - filename: 'doomed.md', - fileUrl: 'https://fixture.test/doomed', - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed', - acl: aclOf('alice'), - }) - await write('async', (tx) => - tx.insert(embedding).values( - Array.from({ length: 3 }, (_, index) => ({ - ...chunkRow(generateId(), index), - documentId: doomed, - })) - ) - ) - await expect( - project({ - pageSize: 1, - onPage: async (page) => { - if (page.documentId === doomed) await db.delete(document).where(eq(document.id, doomed)) - }, - }) - ).resolves.toMatchObject({ remaining: false }) - const [left] = await db - .select({ count: sql`count(*)::int` }) - .from(embeddingSearch) - .where(eq(embeddingSearch.documentId, doomed)) - expect(left?.count).toBe(0) - expect( - await db - .select() - .from(knowledgeProjectionDirty) - .where(eq(knowledgeProjectionDirty.documentId, doomed)) - ).toEqual([]) }) - it("rewrites a connector ACL page's projection rows in the page's own statement", async () => { - await leaseTransaction(connectorId)((tx) => + it('does not resurrect a document deleted after the first page commits', async () => { + await defer((tx) => tx - .update(document) - .set({ acl: aclOf('bob') }) - .where(eq(document.id, documentId)) + .insert(embedding) + .values(Array.from({ length: 3 }, (_, index) => chunkRow(generateId(), index))) ) - expect(await markOf()).toMatchObject({ content: false }) - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) - expect(await admitted()).toEqual({ vector: [], keyword: [] }) - await project() + await runKnowledgeProjection(projector, { + pageSize: 1, + onPage: async (page) => { + if (page.documentId === documentId) + await db.delete(document).where(eq(document.id, documentId)) + }, + }) + expect(await vectorsOf()).toEqual([]) expect(await markOf()).toBeUndefined() }) - it('splits the marks between passes that start together', async () => { - const documents = await Promise.all( - Array.from({ length: 6 }, async (_, index) => { - const id = generateId() - await db.insert(document).values({ - id, - connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: `split-${index}`, - filename: `split-${index}.md`, - fileUrl: `https://fixture.test/split-${index}`, - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed', - acl: aclOf('alice'), - }) - return id - }) - ) - await write('async', (tx) => - tx - .insert(embedding) - .values( - documents.map((id, index) => ({ ...chunkRow(generateId(), index), documentId: id })) - ) - ) - const second = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) + it('preserves work held by another pass until its advisory lock is released', async () => { + await defer((tx) => tx.insert(embedding).values(chunkRow(generateId(), 0))) + const other = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) try { - /** Each page lingers, so neither pass can finish the batch before the other starts. */ - const lingering = { onPage: () => sleep(25), searchIndexes: true } - const passes = await Promise.all([ - project(lingering), - runKnowledgeProjection(second, lingering), - ]) - expect(passes.every((pass) => pass.settled > 0)).toBe(true) - expect(passes[0].settled + passes[1].settled).toBe(documents.length) + await other`SELECT pg_advisory_lock(hashtextextended('knowledge_projection:' || ${documentId}, 0))` + await runKnowledgeProjection(projector, {}) + expect(await vectorsOf()).toEqual([]) + expect(await markOf()).toMatchObject({ content: true }) + await other`SELECT pg_advisory_unlock(hashtextextended('knowledge_projection:' || ${documentId}, 0))` + await runKnowledgeProjection(projector, {}) + expect(await vectorsOf()).toHaveLength(1) + expect(await markOf()).toBeUndefined() } finally { - await second.end() + await other.end() } - expect( - await db - .select() - .from(knowledgeProjectionDirty) - .where(inArray(knowledgeProjectionDirty.documentId, documents)) - ).toEqual([]) - await db.delete(document).where(inArray(document.id, documents)) }) - it('writes in pages bounded by chunk rows', async () => { - await write('async', (tx) => - tx - .insert(embedding) - .values(Array.from({ length: 4 }, (_, index) => chunkRow(generateId(), index + 1))) - ) - const pages: Array<{ projection: KnowledgeProjection; written: number }> = [] - await project({ pageSize: 2, onPage: (page) => void pages.push(page) }) - const vectorPages = pages.filter((page) => page.projection === 'embedding_search') - expect(vectorPages.map((page) => page.written)).toEqual([1, 2, 1]) - expect(pages.every((page) => page.written <= 2)).toBe(true) - expect(await markOf()).toBeUndefined() - }) - - it('writes keyword rows for search-index knowledge bases only', async () => { - const workspaceBaseId = generateId() - const workspaceDocument = generateId() - const workspaceChunk = generateId() - await db.insert(knowledgeBase).values({ - id: workspaceBaseId, - userId: ids.aliceId, - workspaceId: ids.workspaceId, - name: 'Workspace keyword fixture', - chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, - }) - try { - await db.insert(document).values({ - id: workspaceDocument, - knowledgeBaseId: workspaceBaseId, - filename: 'keyword.md', - fileUrl: 'https://fixture.test/keyword', + it('yields after one thousand deferred documents without losing any pending marks', async () => { + const documentIds = Array.from({ length: 1_001 }, () => generateId()) + await db.insert(document).values( + documentIds.map((id) => ({ + id, + knowledgeBaseId: ids.knowledgeBaseId, + filename: 'deferred.md', + fileUrl: 'https://fixture.test/deferred', fileSize: 12, mimeType: 'text/plain', processingStatus: 'completed', + })) + ) + await db + .insert(knowledgeProjectionDirty) + .values(documentIds.map((documentId) => ({ documentId, content: true }))) + const other = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) + try { + await other` + SELECT pg_advisory_lock(hashtextextended('knowledge_projection:' || id, 0)) + FROM unnest(${documentIds}::text[]) AS locked(id)` + expect(await runKnowledgeProjection(projector, {})).toMatchObject({ + deferred: 1_000, + written: 0, + remaining: true, }) - await write('async', (tx) => - tx.insert(embedding).values({ - ...chunkRow(workspaceChunk, 0), - documentId: workspaceDocument, - knowledgeBaseId: workspaceBaseId, - }) - ) - await project() - const keywordRows = await db - .select({ id: embeddingKeywordSearch.id }) - .from(embeddingKeywordSearch) - .where(inArray(embeddingKeywordSearch.id, [chunkId, workspaceChunk])) - expect(keywordRows).toEqual([{ id: chunkId }]) - expect(await rowAcl(embeddingSearch, workspaceChunk)).toBeDefined() + const [pending] = await db + .select({ count: sql`count(*)::int` }) + .from(knowledgeProjectionDirty) + .where(inArray(knowledgeProjectionDirty.documentId, documentIds)) + expect(pending.count).toBe(1_001) } finally { - await db.delete(knowledgeBase).where(eq(knowledgeBase.id, workspaceBaseId)) + await other.end() } }) - it.each(['sync', 'async'] as const)( - 'writes %s projection rows from a chunk commit only when the writer did not defer them', - async (mode) => { - /** - * The projector's single connection makes the write, so its own statistics can be flushed - * and read back without waiting on the collector's interval. - */ - const inserted = async () => { - await projector`SELECT pg_stat_force_next_flush()` - await projector`SELECT pg_stat_clear_snapshot()` - const [row] = await projector>` - SELECT n_tup_ins AS inserted FROM pg_stat_user_tables WHERE relname = 'embedding_search'` - return Number(row?.inserted ?? 0) - } - const chunks = Array.from({ length: 50 }, (_, index) => chunkRow(generateId(), index + 1)) - const before = await inserted() - await projector.begin(async (tx) => { - if (mode === 'async') await tx`SELECT set_config('sim.projection_mode', 'async', true)` - for (const chunk of chunks) { - await tx`INSERT INTO embedding (id, document_id, knowledge_base_id, chunk_index, chunk_hash, - content, content_length, token_count, start_offset, end_offset, embedding_model, embedding) - VALUES (${chunk.id}, ${documentId}, ${ids.knowledgeBaseId}, ${chunk.chunkIndex}, - ${chunk.chunkHash}, ${chunk.content}, 14, 2, 0, 14, ${chunk.embeddingModel}, - ${JSON.stringify(vector)}::vector)` - } - }) - expect((await inserted()) - before).toBe(mode === 'async' ? 0 : chunks.length) - expect(await markOf()).toMatchObject({ content: mode === 'async' }) - } - ) - - it('owes a pass for content and, while indexed search is on, search-index marks, asking only for content behind an undrained release', async () => { - const workspaceBaseId = generateId() - const [workspaceDocument, contentDocument] = [generateId(), generateId()] - await db.insert(knowledgeBase).values({ - id: workspaceBaseId, - userId: ids.aliceId, - workspaceId: ids.workspaceId, - name: 'Workspace work fixture', - chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, - }) - /** Thrown to roll the transaction back, so the marks of other suites are never touched for good. */ - const rollback = new Error('rollback') + it('resumes a bounded pass without rewriting an already current page', async () => { + const chunks = Array.from({ length: 3 }, (_, index) => chunkRow(generateId(), index)) + await defer((tx) => tx.insert(embedding).values(chunks)) + const startedAt = Date.now() + const clock = vi.spyOn(Date, 'now').mockReturnValue(startedAt) try { - await db.insert(document).values( - [workspaceDocument, contentDocument].map((id, index) => ({ - id, - knowledgeBaseId: workspaceBaseId, - filename: `work-${index}.md`, - fileUrl: `https://fixture.test/work-${index}`, - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed' as const, - })) - ) - const answers: Record = {} - const indexed = { searchIndexes: true } - const answer = async (tx: postgres.TransactionSql, label: string) => { - answers[label] = { - drained: await hasKnowledgeProjectionWork(tx, { drained: true }, indexed), - undrained: await hasKnowledgeProjectionWork(tx, { drained: false }, indexed), - dormant: await hasKnowledgeProjectionWork( - tx, - { drained: true }, - { searchIndexes: false } - ), - } - } - await projector - .begin(async (tx) => { - await tx`DELETE FROM knowledge_projection_dirty` - await tx`SELECT mark_knowledge_projection(ARRAY[${workspaceDocument}]::text[], false)` - await answer(tx, 'workspace') - await tx`SELECT mark_knowledge_projection(ARRAY[${documentId}]::text[], false)` - await answer(tx, 'search index') - await tx`SELECT mark_knowledge_projection(ARRAY[${contentDocument}]::text[], true)` - await answer(tx, 'content') - throw rollback - }) - .catch((error) => { - if (error !== rollback) throw error - }) - expect(answers).toEqual({ - workspace: { drained: false, undrained: false, dormant: false }, - 'search index': { drained: true, undrained: false, dormant: false }, - content: { drained: true, undrained: true, dormant: true }, + await runKnowledgeProjection(projector, { + pageSize: 1, + budgetMs: 1_000, + onPage: (page) => { + if (page.documentId === documentId) clock.mockReturnValue(startedAt + 1_000) + }, }) } finally { - await db.delete(knowledgeBase).where(eq(knowledgeBase.id, workspaceBaseId)) + clock.mockRestore() } + expect(await vectorsOf()).toHaveLength(1) + expect(await markOf()).toMatchObject({ content: true }) + const [before] = + await projector`SELECT xmin::text AS version FROM embedding_search WHERE id = ${chunks[0].id}` + await runKnowledgeProjection(projector, { pageSize: 1 }) + const [after] = + await projector`SELECT xmin::text AS version FROM embedding_search WHERE id = ${chunks[0].id}` + expect(after.version).toBe(before.version) + expect(await vectorsOf()).toHaveLength(3) + expect(await markOf()).toBeUndefined() }) - it('releases marks with nothing to project, keeping search-index marks for a pass only while indexed search is on', async () => { - const workspaceBaseId = generateId() - const [synced, deferred, held] = [generateId(), generateId(), generateId()] - await db.insert(knowledgeBase).values({ - id: workspaceBaseId, - userId: ids.aliceId, - workspaceId: ids.workspaceId, - name: 'Workspace fixture', - chunkingConfig: { maxSize: 1024, minSize: 1, overlap: 20 }, - }) - try { - await db.insert(document).values( - [synced, deferred, held].map((id, index) => ({ - id, - knowledgeBaseId: workspaceBaseId, - filename: `workspace-${index}.md`, - fileUrl: `https://fixture.test/workspace-${index}`, - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed' as const, - })) - ) - const workspaceChunk = (id: string, documentId: string) => ({ - ...chunkRow(id, 0), - documentId, - knowledgeBaseId: workspaceBaseId, - }) - const deferredChunk = generateId() - await write('sync', (tx) => - tx - .insert(embedding) - .values([workspaceChunk(generateId(), synced), workspaceChunk(generateId(), held)]) - ) - await write('async', (tx) => - tx.insert(embedding).values(workspaceChunk(deferredChunk, deferred)) - ) - /** A search-index mark with nothing but a source and ACL change is still a pass's. */ - await db - .update(document) - .set({ acl: aclOf('bob') }) - .where(eq(document.id, documentId)) - const marked = async () => - ( - await db - .select({ documentId: knowledgeProjectionDirty.documentId }) - .from(knowledgeProjectionDirty) - .where( - inArray(knowledgeProjectionDirty.documentId, [synced, deferred, held, documentId]) - ) - ) - .map((row) => row.documentId) - .sort() - - /** A writer re-marking `held` holds its mark until it commits; the release passes it over. */ - const writer = postgres(process.env.DATABASE_URL!, { max: 1, onnotice: () => undefined }) - let release: () => void = () => {} - const holding = new Promise((resolve) => { - release = resolve - }) - let locked: () => void = () => {} - const marking = new Promise((resolve) => { - locked = resolve - }) - try { - const writing = writer.begin(async (tx) => { - await tx`SELECT mark_knowledge_projection(ARRAY[${held}]::text[], false)` - locked() - await holding - }) - await marking - expect( - (await releaseSettledMarks(projector, Number.POSITIVE_INFINITY, { searchIndexes: true })) - .released - ).toBeGreaterThanOrEqual(1) - expect(await marked()).toEqual([deferred, held, documentId].sort()) - release() - await writing - } finally { - release() - await writer.end() - } - expect( - (await releaseSettledMarks(projector, Number.POSITIVE_INFINITY, { searchIndexes: true })) - .released - ).toBeGreaterThanOrEqual(1) - expect(await marked()).toEqual([deferred, documentId].sort()) - - /** The deferred chunk has no row until a pass writes it, and the pass settles both marks. */ - const deferredRow = async () => - db - .select({ id: embeddingSearch.id }) - .from(embeddingSearch) - .where(eq(embeddingSearch.id, deferredChunk)) - expect(await deferredRow()).toEqual([]) - await project() - expect(await deferredRow()).toEqual([{ id: deferredChunk }]) - expect(await marked()).toEqual([]) - expect((await rowAcl(embeddingSearch))?.acl).toEqual(aclOf('bob')) - - /** While indexed search is dormant, nothing reads a search-index mark, so it is released too. */ - await db - .update(document) - .set({ acl: aclOf('alice') }) - .where(eq(document.id, documentId)) - expect(await marked()).toEqual([documentId]) - await releaseSettledMarks(projector, Number.POSITIVE_INFINITY, { searchIndexes: false }) - expect(await marked()).toEqual([]) - } finally { - await db.delete(knowledgeBase).where(eq(knowledgeBase.id, workspaceBaseId)) - } + it('releases retired Search marks while retaining ordinary content repair for admission', async () => { + await db + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where(eq(knowledgeBase.id, searchBaseId)) + await defer((tx) => + tx.insert(embedding).values([chunkRow(generateId(), 0), chunkRow(generateId(), 0, true)]) + ) + await releaseSettledMarks(projector, Date.now() + 10_000) + expect(await markOf(searchDocumentId)).toBeUndefined() + expect(await markOf()).toMatchObject({ content: true }) + expect(await hasKnowledgeProjectionWork(projector)).toBe(true) + expect(await vectorsOf(searchDocumentId)).toEqual([]) }) }) diff --git a/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts b/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts deleted file mode 100644 index 7b2d0f2a111..00000000000 --- a/apps/sim/lib/knowledge/__integration__/organization-mcp-search.integration.ts +++ /dev/null @@ -1,988 +0,0 @@ -/** - * Real organization-owned ingestion, Postgres, API-key authentication, application - * authorization, source ACLs, and MCP SDK transport. Only embedding vectors are - * deterministic and storage is temporary; no provider or live credential is used. - */ -import { mkdtempSync } from 'node:fs' -import { rm } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import path from 'node:path' -import { Client } from '@modelcontextprotocol/sdk/client/index.js' -import { StreamableHTTPClientTransport } from '@modelcontextprotocol/sdk/client/streamableHttp.js' -import { CallToolResultSchema } from '@modelcontextprotocol/sdk/types.js' -import type { Principal } from '@sim/auth/principal' -import { db } from '@sim/db' -import { - apiKey, - document, - embedding, - embeddingSecretProvenance, - knowledgeBase, - knowledgeExternalGroup, - knowledgeExternalGroupMember, - member, - oauthAccessToken, - oauthClient, - oauthConsent, - organization, - organizationSearchIntegration, - organizationSearchMcpInvocation, - rateLimitBucket, - user, - workspace, -} from '@sim/db/schema' -import { sha256Hex } from '@sim/security/hash' -import { generateId } from '@sim/utils/id' -import { isPlainRecord } from '@sim/utils/object' -import { and, eq, inArray } from 'drizzle-orm' -import { NextRequest } from 'next/server' -import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' - -const fixtures = vi.hoisted(() => ({ - storageRoot: '', - afterResponse: [] as Array<() => Promise>, -})) -/** This ingestion suite covers the explicit rollback backend; live Search has its own suites. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) -vi.mock('@/lib/core/utils/after-response', () => ({ - afterResponse: (task: () => Promise) => fixtures.afterResponse.push(task), -})) - -async function flushAfterResponse() { - for (const task of fixtures.afterResponse.splice(0)) await task() -} -vi.mock('@/lib/uploads/core/setup.server', () => ({ - get UPLOAD_DIR_SERVER() { - return fixtures.storageRoot - }, -})) -vi.mock('@/lib/embeddings', async () => ({ - ...(await import('@/lib/embeddings/client')), - assertKnowledgeEmbeddingCapacity: async () => {}, - embedKnowledge: async (texts: string[]) => ({ - embeddings: texts.map(() => [1, ...Array(1535).fill(0)]), - totalTokens: texts.length, - billableTokens: 0, - isBYOK: true, - modelName: 'text-embedding-3-small', - pricingId: 'text-embedding-3-small', - }), -})) - -import { hashApiKey } from '@/lib/api-key/crypto' -import { hashOAuthToken } from '@/lib/auth/oauth-access-token' -import { OAUTH_ACCESS_TOKEN_PREFIX } from '@/lib/auth/oauth-provider' -import { resolveOrganizationBillingAttribution } from '@/lib/billing/core/billing-attribution' -import { encryptSecret } from '@/lib/core/security/encryption' -import { - createKnowledgeAclFixtureIds, - seedKnowledgeAclFixture, -} from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { confluencePageAcl } from '@/lib/knowledge/access/confluence-permissions' -import { listKnowledgeChunks } from '@/lib/knowledge/application/chunks' -import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' -import { listKnowledgeBases } from '@/lib/knowledge/application/knowledge-bases' -import { prepareSearchSource } from '@/lib/knowledge/application/sim-search' -import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' -import { addDocument, persistDocumentAcls } from '@/lib/knowledge/connectors/sync-persistence' -import { processDocumentAsync } from '@/lib/knowledge/documents/service' -import { getSearchMcpUrl } from '@/lib/knowledge/mcp/urls' -import { replaceKnowledgeEmbeddingSecretProvenanceInTx } from '@/lib/knowledge/secret-provenance' -import { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' -import { searchScopedKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' -import { DELETE, GET, POST } from '@/app/api/mcp/search/organizations/[organizationId]/route' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -describe('organization Search MCP rollback backend with real ingestion and current access', () => { - const ids = createKnowledgeAclFixtureIds() - const { - aliceId, - bobId, - workspaceId, - organizationId, - knowledgeBaseId, - connectorId, - lockId, - groupIds, - } = ids - const otherOrganizationId = generateId() - const workspaceKnowledgeBaseId = generateId() - const outsiderId = generateId() - const otherAdminId = generateId() - const bobMembershipId = generateId() - const tokens = { - alice: generateId(), - bob: generateId(), - outsider: generateId(), - workspace: generateId(), - expired: generateId(), - } - const oauthClientId = generateId() - const oauthTokens = { alice: generateId(), bob: generateId() } - const clients: Client[] = [] - const alicePrincipal: Principal = { kind: 'session', userId: aliceId, sessionId: generateId() } - const bobPrincipal: Principal = { kind: 'session', userId: bobId, sessionId: generateId() } - const otherAdminPrincipal: Principal = { - kind: 'session', - userId: otherAdminId, - sessionId: generateId(), - } - const bobSourceMembership = { - groupId: groupIds[2], - subjectToken: `u:${bobId}@fixture.test`, - } - let otherKnowledgeBaseId: string - let documentId: string - let alice: Client - let bob: Client - let aliceOAuth: Client - let bobOAuth: Client - - async function request( - token?: string, - target = organizationId, - options: { - method?: 'GET' | 'POST' | 'DELETE' - body?: unknown - raw?: string - headers?: Record - } = {} - ) { - const method = options.method ?? 'POST' - return { GET, POST, DELETE }[method]( - new NextRequest(`http://localhost:3000/api/mcp/search/organizations/${target}`, { - method, - headers: { - 'x-forwarded-for': '127.0.0.1', - 'content-type': 'application/json', - accept: 'application/json, text/event-stream', - ...(token ? { 'x-api-key': token } : {}), - ...options.headers, - }, - ...(method === 'POST' - ? { - body: - options.raw ?? - JSON.stringify(options.body ?? { jsonrpc: '2.0', id: 1, method: 'tools/list' }), - } - : {}), - }), - { params: Promise.resolve({ organizationId: target }) } - ) - } - - async function connect(token: string, bearer = false) { - const client = new Client({ name: 'Organization ACL fixture', version: '1.0.0' }) - clients.push(client) - await client.connect( - new StreamableHTTPClientTransport( - new URL(`http://localhost:3000/api/mcp/search/organizations/${organizationId}`), - { - requestInit: { - headers: bearer ? { authorization: `Bearer ${token}` } : { 'x-api-key': token }, - }, - fetch: async (url, init) => { - const req = new NextRequest(url instanceof Request ? url : String(url), { - ...init, - signal: init?.signal ?? undefined, - }) - req.headers.set('x-forwarded-for', '127.0.0.1') - const context = { params: Promise.resolve({ organizationId }) } - return req.method === 'GET' - ? GET(req, context) - : req.method === 'DELETE' - ? DELETE(req, context) - : POST(req, context) - }, - } - ) - ) - return client - } - - async function call(client: Client, name: string, args: Record) { - return CallToolResultSchema.parse(await client.callTool({ name, arguments: args })) - } - - async function value(client: Client, name: string, args: Record) { - const result = await call(client, name, args) - expect(result.isError).not.toBe(true) - const first = result.content[0] - if (first?.type !== 'text') throw new Error('Expected JSON text tool result') - const parsed: unknown = JSON.parse(first.text) - if (!isPlainRecord(parsed)) throw new Error('Expected JSON object tool result') - return parsed - } - - async function search(client: Client) { - const result = await value(client, 'search', { query: 'Orion', topK: 50 }) - if (!Array.isArray(result.results) || !result.results.every(isPlainRecord)) { - throw new Error('Expected search result objects') - } - return result.results - } - - async function applicationSearch(principal: Principal) { - const result = await searchScopedKnowledge.execute({ - principal, - input: { organizationId, query: 'Orion', searchMode: 'hybrid', topK: 50 }, - }) - return result.results - } - - async function expectDocumentHidden(client: Client) { - expect(await search(client)).toEqual([]) - expect(await call(client, 'read_document', { documentId })).toMatchObject({ - isError: true, - content: [{ type: 'text', text: 'Document not found' }], - }) - } - - async function revokeBobSourceAccess() { - await db - .delete(knowledgeExternalGroupMember) - .where( - and( - eq(knowledgeExternalGroupMember.groupId, bobSourceMembership.groupId), - eq(knowledgeExternalGroupMember.subjectToken, bobSourceMembership.subjectToken) - ) - ) - } - - beforeAll(async () => { - vi.stubGlobal('fetch', async () => { - throw new Error('Unexpected outbound organization MCP fixture request') - }) - fixtures.storageRoot = mkdtempSync(path.join(tmpdir(), 'sim-organization-mcp-integration-')) - await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) - await db.insert(user).values( - [outsiderId, otherAdminId].map((id) => ({ - id, - name: 'Other organization fixture', - email: `${id}@fixture.test`, - emailVerified: true, - createdAt: new Date(), - updatedAt: new Date(), - })) - ) - await db.insert(organization).values({ - id: otherOrganizationId, - name: 'Other organization MCP fixture', - slug: otherOrganizationId, - createdAt: new Date(), - }) - await db.insert(member).values([ - { id: generateId(), userId: aliceId, organizationId, role: 'owner' }, - { id: bobMembershipId, userId: bobId, organizationId, role: 'member' }, - { id: generateId(), userId: outsiderId, organizationId: otherOrganizationId, role: 'owner' }, - { - id: generateId(), - userId: otherAdminId, - organizationId: otherOrganizationId, - role: 'admin', - }, - ]) - /** Reuse source identities, but establish exclusive organization ownership before ingestion. */ - await db - .update(knowledgeBase) - .set({ - workspaceId: null, - organizationId, - isSearchIndex: true, - name: 'Renamed org index', - chunkingConfig: { maxSize: 256, minSize: 1, overlap: 20 }, - }) - .where(eq(knowledgeBase.id, knowledgeBaseId)) - await db - .update(knowledgeExternalGroup) - .set({ workspaceId: null, organizationId }) - .where(inArray(knowledgeExternalGroup.id, groupIds)) - const prepared = await prepareSearchSource.execute({ - principal: otherAdminPrincipal, - input: { organizationId: otherOrganizationId, connectorType: 'gitlab' }, - }) - otherKnowledgeBaseId = prepared.knowledgeBaseId - await db.insert(knowledgeBase).values({ - id: workspaceKnowledgeBaseId, - userId: bobId, - workspaceId, - name: 'Workspace documents', - }) - await db.insert(organizationSearchIntegration).values({ - organizationId, - connectorType: 'google_drive', - approved: true, - }) - await db.insert(apiKey).values( - Object.entries(tokens).map(([name, token]) => ({ - id: generateId(), - userId: name === 'bob' ? bobId : name === 'outsider' ? outsiderId : aliceId, - name, - key: `fixture-${generateId()}`, - keyHash: hashApiKey(token), - type: name === 'workspace' ? 'workspace' : 'personal', - workspaceId: name === 'workspace' ? workspaceId : null, - expiresAt: name === 'expired' ? new Date(0) : null, - })) - ) - await db.insert(oauthClient).values({ - id: oauthClientId, - clientId: oauthClientId, - name: 'Search MCP OAuth fixture', - public: true, - requirePKCE: true, - redirectUris: ['http://127.0.0.1/callback'], - tokenEndpointAuthMethod: 'none', - scopes: ['search:read'], - grantTypes: ['authorization_code'], - responseTypes: ['code'], - }) - for (const [userId, token] of [ - [aliceId, oauthTokens.alice], - [bobId, oauthTokens.bob], - ]) { - await db.insert(oauthConsent).values({ - id: generateId(), - clientId: oauthClientId, - userId, - scopes: ['search:read'], - createdAt: new Date(), - updatedAt: new Date(), - }) - await db.insert(oauthAccessToken).values({ - id: generateId(), - clientId: oauthClientId, - userId, - token: hashOAuthToken(token), - scopes: ['search:read'], - resource: getSearchMcpUrl(organizationId), - createdAt: new Date(), - expiresAt: new Date(Date.now() + 3600_000), - }) - } - const doc = await addDocument( - knowledgeBaseId, - connectorId, - 'google_drive', - { - externalId: 'organization-mcp-page', - mimeType: 'text/plain', - title: 'Orion organization project', - content: Array.from( - { length: 30 }, - (_, index) => - `Orion section ${index}: engineers approved the release checklist and documented the customer migration dependencies.` - ).join('\n\n'), - contentHash: 'fixture-organization-orion', - sourceUrl: 'https://fixture.atlassian.net/wiki/pages/organization-mcp', - }, - { userId: aliceId, workspaceId: null, organizationId }, - undefined, - 'admin', - createContentSyncLease(connectorId, lockId) - ) - documentId = doc.documentId - await processDocumentAsync( - knowledgeBaseId, - documentId, - doc, - {}, - await resolveOrganizationBillingAttribution({ actorUserId: aliceId, organizationId }) - ) - const [persisted] = await db - .select({ status: document.processingStatus, error: document.processingError }) - .from(document) - .where(eq(document.id, documentId)) - expect(persisted).toEqual({ status: 'completed', error: null }) - await persistDocumentAcls( - connectorId, - new Map([ - [ - 'organization-mcp-page', - confluencePageAcl({ - providerId: 'google-drive', - tenantId: 'fixture-tenant', - spacePrincipals: [{ kind: 'group', id: 'space' }], - restrictionChain: [[{ kind: 'group', id: 'page' }], [{ kind: 'group', id: 'parent' }]], - }), - ], - ]) - ) - alice = await connect(tokens.alice) - bob = await connect(tokens.bob, true) - aliceOAuth = await connect(OAUTH_ACCESS_TOKEN_PREFIX + oauthTokens.alice, true) - bobOAuth = await connect(OAUTH_ACCESS_TOKEN_PREFIX + oauthTokens.bob, true) - }) - - afterEach(flushAfterResponse) - - afterAll(async () => { - await Promise.all(clients.map((client) => client.close())) - await db.delete(oauthClient).where(eq(oauthClient.clientId, oauthClientId)) - await db - .delete(organization) - .where(inArray(organization.id, [organizationId, otherOrganizationId])) - await db.delete(workspace).where(eq(workspace.id, workspaceId)) - await db.delete(user).where(inArray(user.id, [aliceId, bobId, outsiderId, otherAdminId])) - if (fixtures.storageRoot) await rm(fixtures.storageRoot, { recursive: true, force: true }) - await db.$client.end() - vi.unstubAllGlobals() - }) - - it('creates an organization-only index, keeps it out of the workspace catalog, and separates actor from payer', async () => { - const input = { organizationId: otherOrganizationId, connectorType: 'gitlab' } - const results = await Promise.all([ - prepareSearchSource.execute({ principal: otherAdminPrincipal, input }), - prepareSearchSource.execute({ principal: otherAdminPrincipal, input }), - ]) - expect(results.map((result) => result.knowledgeBaseId)).toEqual([ - otherKnowledgeBaseId, - otherKnowledgeBaseId, - ]) - const indexes = await db - .select() - .from(knowledgeBase) - .where(eq(knowledgeBase.organizationId, otherOrganizationId)) - expect(indexes).toEqual([ - expect.objectContaining({ - id: otherKnowledgeBaseId, - workspaceId: null, - organizationId: otherOrganizationId, - isSearchIndex: true, - userId: otherAdminId, - }), - ]) - const catalog = await listKnowledgeBases.execute({ - principal: alicePrincipal, - input: { workspaceId }, - }) - expect(catalog.knowledgeBases.map(({ knowledgeBase }) => knowledgeBase.id)).toEqual([ - workspaceKnowledgeBaseId, - ]) - await expect( - resolveOrganizationBillingAttribution({ - actorUserId: otherAdminId, - organizationId: otherOrganizationId, - }) - ).resolves.toMatchObject({ - actorUserId: otherAdminId, - workspaceId: null, - organizationId: otherOrganizationId, - billedAccountUserId: outsiderId, - billingEntity: { type: 'organization', id: otherOrganizationId }, - }) - }) - - it('finds the canonical org-owned index and applies each current member’s source ACL to all tools', async () => { - const [owner] = await db - .select({ - workspaceId: knowledgeBase.workspaceId, - organizationId: knowledgeBase.organizationId, - }) - .from(knowledgeBase) - .where(eq(knowledgeBase.id, knowledgeBaseId)) - expect(owner).toEqual({ workspaceId: null, organizationId }) - expect((await alice.listTools()).tools.map((tool) => tool.name)).toEqual([ - 'search', - 'read_document', - 'chat', - ]) - const rows = await search(alice) - expect(rows.length).toBeGreaterThan(1) - expect(rows.every((row) => row.documentId === documentId && !('knowledgeBaseId' in row))).toBe( - true - ) - expect((await applicationSearch(alicePrincipal)).map((row) => row.documentId)).toEqual( - rows.map((row) => row.documentId) - ) - expect(await value(alice, 'read_document', { documentId })).toMatchObject({ - title: 'Orion organization project', - documentId, - chunks: expect.arrayContaining([ - expect.objectContaining({ chunkIndex: 0, content: expect.stringContaining('Orion') }), - ]), - pagination: expect.objectContaining({ limit: 20, offset: 0 }), - }) - const page = await value(alice, 'read_document', { - documentId, - limit: 1, - }) - expect(page.chunks).toEqual([ - expect.objectContaining({ chunkIndex: 0, content: expect.stringContaining('Orion') }), - ]) - expect(page.pagination).toMatchObject({ limit: 1, offset: 0, hasMore: true }) - const nextPage = await value(alice, 'read_document', { - documentId, - limit: 1, - offset: 1, - }) - expect(nextPage.chunks).toEqual([expect.objectContaining({ chunkIndex: 1 })]) - expect(nextPage.chunks).not.toEqual(page.chunks) - expect(nextPage.pagination).toMatchObject({ limit: 1, offset: 1 }) - await expectDocumentHidden(bob) - expect(await applicationSearch(bobPrincipal)).toEqual([]) - }) - - it('persists content-free per-client tool outcomes separately from search counters', async () => { - await db - .delete(organizationSearchMcpInvocation) - .where(eq(organizationSearchMcpInvocation.organizationId, organizationId)) - const clientName = 'MCP fixture client'.repeat(20) - await db - .update(oauthClient) - .set({ name: clientName }) - .where(eq(oauthClient.clientId, oauthClientId)) - try { - await aliceOAuth.listTools() - expect(fixtures.afterResponse).toHaveLength(0) - await search(aliceOAuth) - await value(aliceOAuth, 'read_document', { documentId }) - expect((await call(bob, 'read_document', { documentId })).isError).toBe(true) - expect(fixtures.afterResponse).toHaveLength(3) - await db - .update(oauthClient) - .set({ name: 'Renamed client' }) - .where(eq(oauthClient.clientId, oauthClientId)) - await flushAfterResponse() - const rows = await db - .select() - .from(organizationSearchMcpInvocation) - .where(eq(organizationSearchMcpInvocation.organizationId, organizationId)) - .orderBy(organizationSearchMcpInvocation.createdAt) - .limit(10) - expect(rows).toHaveLength(3) - expect(rows).toMatchObject([ - { - organizationId, - userId: aliceId, - authKind: 'oauth_access_token', - oauthClientId, - clientName: clientName.slice(0, 256), - toolName: 'search', - outcome: 'success', - }, - { - organizationId, - userId: aliceId, - authKind: 'oauth_access_token', - oauthClientId, - clientName: clientName.slice(0, 256), - toolName: 'read_document', - outcome: 'success', - }, - { - organizationId, - userId: bobId, - authKind: 'personal_api_key', - oauthClientId: null, - clientName: null, - toolName: 'read_document', - outcome: 'error', - }, - ]) - expect(rows.every((row) => row.durationMs >= 0)).toBe(true) - expect(JSON.stringify(rows)).not.toContain(documentId) - expect(JSON.stringify(rows)).not.toContain(oauthTokens.alice) - } finally { - await db - .update(oauthClient) - .set({ name: 'Search MCP OAuth fixture' }) - .where(eq(oauthClient.clientId, oauthClientId)) - } - }) - - it('enforces current document and organization access on Search OAuth clients', async () => { - expect((await aliceOAuth.listTools()).tools).toHaveLength(3) - expect(await search(aliceOAuth)).toEqual(await search(alice)) - expect(await value(aliceOAuth, 'read_document', { documentId })).toMatchObject({ - documentId, - chunks: expect.arrayContaining([ - expect.objectContaining({ content: expect.stringContaining('Orion') }), - ]), - }) - expect(await value(aliceOAuth, 'read_document', { documentId })).toHaveProperty('chunks') - await expectDocumentHidden(bobOAuth) - await db.insert(knowledgeExternalGroupMember).values(bobSourceMembership) - try { - expect((await search(bobOAuth)).length).toBeGreaterThan(0) - await db.delete(member).where(eq(member.id, bobMembershipId)) - await expect(bobOAuth.listTools()).rejects.toThrow() - await expect(search(bobOAuth)).rejects.toThrow() - } finally { - await db - .insert(member) - .values({ id: bobMembershipId, userId: bobId, organizationId, role: 'member' }) - .onConflictDoNothing() - await revokeBobSourceAccess() - } - await expectDocumentHidden(bobOAuth) - await db - .delete(oauthAccessToken) - .where(eq(oauthAccessToken.token, hashOAuthToken(oauthTokens.bob))) - await expect(bobOAuth.listTools()).rejects.toThrow() - }) - - it('resolves provider URLs in the organization index and focuses OAuth document reads', async () => { - const url = 'https://fixture.atlassian.net/wiki/pages/organization-mcp' - expect( - await value(aliceOAuth, 'read_document', { url, aroundChunkIndex: 1, limit: 1 }) - ).toMatchObject({ - documentId, - citationUrl: url, - chunks: [expect.objectContaining({ chunkIndex: 1 })], - pagination: { offset: 1, limit: 1 }, - }) - expect((await call(bob, 'read_document', { url })).isError).toBe(true) - expect( - (await call(alice, 'read_document', { url, knowledgeBaseId: generateId() })).isError - ).toBe(true) - await expect( - readIndexedKnowledgeDocument.execute({ - principal: alicePrincipal, - input: { - organizationId: otherOrganizationId, - target: { kind: 'url', url }, - limit: 20, - resultSecretRegistry: new ResolvedSecretTraceRegistry(), - }, - }) - ).rejects.toMatchObject({ code: 'not_found' }) - expect((await call(alice, 'read_document', { url, documentId })).isError).toBe(true) - expect( - (await call(alice, 'read_document', { url, offset: 0, aroundChunkIndex: 1 })).isError - ).toBe(true) - }) - - it('applies organization source, modified-date, and document filters through search', async () => { - const [original] = await db - .select({ sourceModifiedAt: document.sourceModifiedAt }) - .from(document) - .where(eq(document.id, documentId)) - await db - .update(document) - .set({ sourceModifiedAt: new Date('2026-01-02T00:00:00Z') }) - .where(eq(document.id, documentId)) - try { - const included = await value(alice, 'search', { - query: 'Orion', - source: 'google_drive', - modifiedAfter: '2026-01-01T00:00:00Z', - documentIds: [documentId], - }) - expect(included.results).toEqual( - expect.arrayContaining([expect.objectContaining({ documentId })]) - ) - for (const filters of [ - { source: 'slack' }, - { modifiedAfter: '2026-01-03T00:00:00Z' }, - { documentIds: [generateId()] }, - ]) { - expect(await value(alice, 'search', { query: 'Orion', ...filters })).toMatchObject({ - results: [], - }) - } - } finally { - await db.update(document).set(original).where(eq(document.id, documentId)) - } - }) - - it('denies nonmembers, other organizations and same-organization workspace keys before discovery', async () => { - expect((await request(tokens.outsider)).status).toBe(404) - expect((await request(tokens.alice, otherOrganizationId)).status).toBe(404) - expect((await request(tokens.workspace)).status).toBe(403) - expect( - ( - await call(alice, 'search', { - query: 'Orion', - knowledgeBaseIds: [otherKnowledgeBaseId], - }) - ).isError - ).toBe(true) - expect( - (await call(alice, 'read_document', { knowledgeBaseId: otherKnowledgeBaseId, documentId })) - .isError - ).toBe(true) - }) - - it('grants and revokes source ACL membership on existing clients', async () => { - await db.insert(knowledgeExternalGroupMember).values(bobSourceMembership) - try { - expect((await search(bob)).length).toBeGreaterThan(0) - expect(await value(bob, 'read_document', { documentId })).toHaveProperty( - 'documentId', - documentId - ) - expect(await value(bob, 'read_document', { documentId })).toHaveProperty('chunks') - } finally { - await revokeBobSourceAccess() - } - await expectDocumentHidden(bob) - }) - - it('hides source content immediately when organization approval is disabled', async () => { - const approval = and( - eq(organizationSearchIntegration.organizationId, organizationId), - eq(organizationSearchIntegration.connectorType, 'google_drive') - ) - await db.update(organizationSearchIntegration).set({ approved: false }).where(approval) - try { - await expectDocumentHidden(alice) - expect(await applicationSearch(alicePrincipal)).toEqual([]) - } finally { - await db.update(organizationSearchIntegration).set({ approved: true }).where(approval) - } - expect((await search(alice)).length).toBeGreaterThan(0) - }) - - it('excludes disabled documents from Search and MCP while preserving admin management previews', async () => { - const chunks = await db - .select({ enabled: embedding.enabled }) - .from(embedding) - .where(eq(embedding.documentId, documentId)) - expect(chunks.length).toBeGreaterThan(1) - expect(chunks.every((chunk) => chunk.enabled)).toBe(true) - await db.update(document).set({ enabled: false }).where(eq(document.id, documentId)) - try { - await expectDocumentHidden(alice) - expect(await applicationSearch(alicePrincipal)).toEqual([]) - const input = { knowledgeBaseId, documentId, assertedOrganizationId: organizationId } - const metadata = await readKnowledgeDocument.execute({ principal: alicePrincipal, input }) - expect(metadata.document.enabled).toBe(false) - const preview = await listKnowledgeChunks.execute({ principal: alicePrincipal, input }) - expect(preview.chunks.length).toBeGreaterThan(0) - } finally { - await db.update(document).set({ enabled: true }).where(eq(document.id, documentId)) - } - expect((await search(alice)).length).toBeGreaterThan(0) - }) - - it('revokes an existing client after current organization membership is removed despite retained source grants', async () => { - await db.insert(knowledgeExternalGroupMember).values(bobSourceMembership) - try { - expect((await search(bob)).length).toBeGreaterThan(0) - await db.delete(member).where(eq(member.id, bobMembershipId)) - expect((await request(tokens.bob)).status).toBe(404) - await expect(bob.listTools()).rejects.toThrow() - await expect(search(bob)).rejects.toThrow() - await expect(call(bob, 'read_document', { documentId })).rejects.toThrow() - await expect(applicationSearch(bobPrincipal)).rejects.toMatchObject({ code: 'not_found' }) - } finally { - await db - .insert(member) - .values({ id: bobMembershipId, userId: bobId, organizationId, role: 'member' }) - .onConflictDoNothing() - await revokeBobSourceAccess() - } - await expectDocumentHidden(bob) - }) - it('authenticates before protocol parsing and rejects unsupported or untrusted requests', async () => { - for (const method of ['initialize', 'tools/list', 'unknown']) { - for (const token of [undefined, 'invalid-fixture-key', tokens.expired]) { - expect( - (await request(token, organizationId, { body: { jsonrpc: '2.0', id: 1, method } })).status - ).toBe(401) - } - } - expect((await request(undefined, organizationId, { raw: '{' })).status).toBe(401) - for (const method of ['GET', 'DELETE'] as const) { - expect((await request(undefined, organizationId, { method })).status).toBe(401) - expect((await request(tokens.alice, organizationId, { method })).status).toBe(405) - } - expect( - ( - await request(tokens.alice, organizationId, { - headers: { authorization: `Bearer ${tokens.bob}` }, - }) - ).status - ).toBe(401) - expect( - ( - await request(tokens.alice, organizationId, { - headers: { origin: 'https://untrusted.example' }, - }) - ).status - ).toBe(403) - expect( - ( - await request(tokens.alice, organizationId, { - body: { jsonrpc: '2.0', id: 1, method: 'tools/list', ignored: 'x'.repeat(64 * 1024) }, - }) - ).status - ).toBe(413) - }) - - it('exposes no mutation tools and rejects invalid page bounds', async () => { - expect((await call(alice, 'delete_document', { documentId })).isError).toBe(true) - for (const page of [{ limit: 51 }, { offset: -1 }, { aroundChunkIndex: 1_000_001 }]) { - expect((await call(alice, 'read_document', { documentId, ...page })).isError).toBe(true) - } - expect((await call(alice, 'search', { query: 'Orion', topK: 51 })).isError).toBe(true) - }) - - it('returns an actionable empty index without creating one or accepting an alternate knowledge base', async () => { - await db - .update(knowledgeBase) - .set({ deletedAt: new Date() }) - .where(eq(knowledgeBase.id, knowledgeBaseId)) - try { - const empty = await value(alice, 'search', { query: 'Orion' }) - expect(empty.results).toEqual([]) - expect(empty.message).toContain('No Search index') - expect((await call(alice, 'read_document', { documentId })).isError).toBe(true) - expect( - (await call(alice, 'search', { query: 'Orion', knowledgeBaseIds: [knowledgeBaseId] })) - .isError - ).toBe(true) - } finally { - await db - .update(knowledgeBase) - .set({ deletedAt: null }) - .where(eq(knowledgeBase.id, knowledgeBaseId)) - } - }) - - it.each(['pending', 'processing', 'failed'])( - 'returns metadata without text when indexing status is %s', - async (processingStatus) => { - await db.update(document).set({ processingStatus }).where(eq(document.id, documentId)) - try { - const result = await value(alice, 'read_document', { documentId }) - expect(result).toMatchObject({ - documentId, - processingStatus, - title: 'Orion organization project', - }) - expect(result).not.toHaveProperty('chunks') - expect(result).not.toHaveProperty('pagination') - expect((await call(bob, 'read_document', { documentId })).isError).toBe(true) - } finally { - await db - .update(document) - .set({ processingStatus: 'completed' }) - .where(eq(document.id, documentId)) - } - } - ) - - it('rate limits discovery and reads through the existing shared buckets', async () => { - const metadataBucket = `v2:knowledge.search.index.read:user:${aliceId}` - const searchBucket = `v2:knowledge.search:user:${aliceId}` - await db - .update(rateLimitBucket) - .set({ tokens: '0', lastRefillAt: new Date() }) - .where(eq(rateLimitBucket.key, metadataBucket)) - try { - const denied = await request(tokens.alice) - expect(denied.status).toBe(429) - expect(denied.headers.get('retry-after')).not.toBeNull() - } finally { - await db.delete(rateLimitBucket).where(eq(rateLimitBucket.key, metadataBucket)) - } - await db - .update(rateLimitBucket) - .set({ tokens: '0', lastRefillAt: new Date() }) - .where(eq(rateLimitBucket.key, searchBucket)) - try { - expect((await call(alice, 'search', { query: 'Orion' })).isError).toBe(true) - expect((await alice.listTools()).tools).toHaveLength(3) - } finally { - await db.delete(rateLimitBucket).where(eq(rateLimitBucket.key, searchBucket)) - } - expect((await search(alice)).length).toBeGreaterThan(0) - }) - - it('bounds returned text and redacts source secrets without exposing their owner’s secret names', async () => { - const [original] = await db - .select({ - id: embedding.id, - content: embedding.content, - chunkHash: embedding.chunkHash, - secretProvenanceVersion: embedding.secretProvenanceVersion, - }) - .from(embedding) - .where(eq(embedding.documentId, documentId)) - .orderBy(embedding.chunkIndex) - .limit(1) - const [sidecar] = await db - .select() - .from(embeddingSecretProvenance) - .where(eq(embeddingSecretProvenance.embeddingId, original.id)) - try { - await db - .update(embedding) - .set({ content: 'Orion '.repeat(180_000), secretProvenanceVersion: null }) - .where(eq(embedding.id, original.id)) - const oversized = await call(alice, 'read_document', { - documentId, - limit: 1, - }) - expect(oversized.isError).toBe(true) - expect(JSON.stringify(oversized).length).toBeLessThan(1024) - - const secret = `fixture-secret-${generateId()}` - const text = `Orion source secret: ${secret}` - const encrypted = await encryptSecret(secret) - await db.transaction(async (tx) => { - await tx - .update(embedding) - .set({ content: text, chunkHash: sha256Hex(text) }) - .where(eq(embedding.id, original.id)) - await replaceKnowledgeEmbeddingSecretProvenanceInTx(tx, original.id, text, { - status: 'exact', - entries: [ - { - encryptedValue: encrypted.encrypted, - name: 'FIXTURE_PRIVATE_SECRET_NAME', - sourceUserId: aliceId, - }, - ], - }) - }) - for (const result of [ - await value(alice, 'read_document', { documentId, limit: 1 }), - await value(alice, 'search', { query: 'Orion' }), - ]) { - expect(JSON.stringify(result)).not.toContain(secret) - expect(JSON.stringify(result)).not.toContain('FIXTURE_PRIVATE_SECRET_NAME') - } - await db - .update(embeddingSecretProvenance) - .set({ entries: [{ encryptedValue: 'invalid-fixture-cipher', sourceUserId: aliceId }] }) - .where(eq(embeddingSecretProvenance.embeddingId, original.id)) - expect((await call(alice, 'read_document', { documentId, limit: 1 })).isError).toBe(true) - } finally { - await db - .update(embedding) - .set({ - content: original.content, - chunkHash: original.chunkHash, - secretProvenanceVersion: original.secretProvenanceVersion, - }) - .where(eq(embedding.id, original.id)) - if (sidecar) - await db - .insert(embeddingSecretProvenance) - .values(sidecar) - .onConflictDoUpdate({ target: embeddingSecretProvenance.embeddingId, set: sidecar }) - else - await db - .delete(embeddingSecretProvenance) - .where(eq(embeddingSecretProvenance.embeddingId, original.id)) - } - }) - - it('handles client cancellation without retaining or poisoning a session', async () => { - const controller = new AbortController() - controller.abort() - await expect( - alice.callTool({ name: 'search', arguments: { query: 'Orion' } }, undefined, { - signal: controller.signal, - }) - ).rejects.toThrow() - expect((await search(alice)).length).toBeGreaterThan(0) - }) -}) diff --git a/apps/sim/lib/knowledge/__integration__/organization-search-overview.integration.ts b/apps/sim/lib/knowledge/__integration__/organization-search-overview.integration.ts deleted file mode 100644 index f9565f1097e..00000000000 --- a/apps/sim/lib/knowledge/__integration__/organization-search-overview.integration.ts +++ /dev/null @@ -1,432 +0,0 @@ -import { db } from '@sim/db' -import { - account, - credential, - document, - knowledgeBase, - knowledgeConnector, - knowledgeConnectorMember, - knowledgeConnectorMemberSyncLog, - knowledgeConnectorSyncLog, - member, - organization, - organizationSearchIntegration, - user, - workspace, -} from '@sim/db/schema' -import { generateId } from '@sim/utils/id' -import { eq, inArray, sql } from 'drizzle-orm' -import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest' -import { - createKnowledgeAclFixtureIds, - seedKnowledgeAclFixture, -} from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { readOrganizationSearchOverview } from '@/lib/knowledge/application/organization-search-overview' -import { listSearchSources } from '@/lib/knowledge/application/search-sources' -import { SOURCE_PERMISSION_ERROR } from '@/lib/knowledge/connectors/sync-limits' - -const ids = createKnowledgeAclFixtureIds() -const indexId = generateId() -const driveId = generateId() -const pausedDriveId = generateId() -const gmailId = generateId() -const credentialId = generateId() -const accountId = generateId() -const memberId = generateId() -const documentId = generateId() -const sourceIds = [driveId, pausedDriveId, gmailId] -const principal = { kind: 'session', userId: ids.aliceId, sessionId: 'overview-admin' } as const -const input = { organizationId: ids.organizationId } - -beforeAll(async () => { - await seedKnowledgeAclFixture(ids) - await db.insert(member).values([ - { id: generateId(), organizationId: ids.organizationId, userId: ids.aliceId, role: 'admin' }, - { id: generateId(), organizationId: ids.organizationId, userId: ids.bobId, role: 'member' }, - ]) - await db.insert(knowledgeBase).values({ - id: indexId, - userId: ids.aliceId, - organizationId: ids.organizationId, - name: 'Overview fixture', - isSearchIndex: true, - }) - await db.insert(knowledgeConnector).values([ - { - id: driveId, - knowledgeBaseId: indexId, - connectorType: 'google_drive', - accessMode: 'admin', - sourceConfig: {}, - }, - { - id: pausedDriveId, - knowledgeBaseId: indexId, - connectorType: 'google_drive', - accessMode: 'admin', - sourceConfig: {}, - }, - { - id: gmailId, - knowledgeBaseId: indexId, - connectorType: 'gmail', - accessMode: 'members', - sourceConfig: {}, - }, - ]) - await db.insert(account).values({ - id: accountId, - accountId: ids.bobId, - userId: ids.bobId, - providerId: 'google-email', - createdAt: new Date(), - updatedAt: new Date(), - }) - await db.insert(credential).values({ - id: credentialId, - organizationId: ids.organizationId, - type: 'oauth', - displayName: 'Fixture connection', - providerId: 'google-email', - accountId, - createdBy: ids.bobId, - }) - await db.insert(knowledgeConnectorMember).values({ - id: memberId, - organizationId: ids.organizationId, - connectorId: gmailId, - credentialId, - subjectToken: `s:google-email:fixture:${ids.bobId}`, - }) - await db.insert(document).values({ - id: documentId, - knowledgeBaseId: indexId, - connectorId: driveId, - externalId: 'private-fixture', - filename: 'private-title.txt', - fileUrl: 'https://fixture.test/private', - fileSize: 5, - mimeType: 'text/plain', - processingStatus: 'completed', - acl: ['u:someone-else@fixture.test'], - aclVerifiedAt: new Date(), - }) -}) - -beforeEach(async () => { - await db - .delete(organizationSearchIntegration) - .where(eq(organizationSearchIntegration.organizationId, ids.organizationId)) - await db - .delete(knowledgeConnectorMemberSyncLog) - .where(inArray(knowledgeConnectorMemberSyncLog.connectorId, sourceIds)) - await db - .delete(knowledgeConnectorSyncLog) - .where(inArray(knowledgeConnectorSyncLog.connectorId, sourceIds)) - await db - .update(knowledgeConnector) - .set({ - status: 'active', - memberSyncStatus: 'idle', - lastSyncAt: new Date(), - lastMemberSyncAt: new Date(), - lastSyncError: null, - lastMemberSyncError: null, - listingCheckpoint: null, - directoryCheckpoint: null, - nextMemberSyncAt: null, - }) - .where(inArray(knowledgeConnector.id, sourceIds)) - await db - .update(knowledgeConnector) - .set({ status: 'paused' }) - .where(eq(knowledgeConnector.id, pausedDriveId)) - await db - .update(knowledgeConnectorMember) - .set({ - status: 'active', - lastCompleteListingAt: new Date(), - memberSyncedThrough: new Date(), - lastError: null, - consecutiveFailures: 0, - listingCheckpoint: null, - }) - .where(eq(knowledgeConnectorMember.id, memberId)) - await db - .update(document) - .set({ - processingStatus: 'completed', - enabled: true, - userExcluded: false, - contentHash: null, - storageKey: null, - fileUrl: 'https://fixture.test/private', - }) - .where(eq(document.id, documentId)) -}) - -afterAll(async () => { - await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) - await db.delete(organization).where(eq(organization.id, ids.organizationId)) - await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) -}) - -async function provider(connectorType: string) { - return (await readOrganizationSearchOverview.execute({ principal, input })).providers.find( - (item) => item.connectorType === connectorType - ) -} - -describe('organization operational overview with real SQL', () => { - it.each([ - SOURCE_PERMISSION_ERROR, - `Directory refresh incomplete: fixture\n${SOURCE_PERMISSION_ERROR}`, - `${SOURCE_PERMISSION_ERROR}\nSource listing failed for a fixture account`, - `Directory refresh incomplete: fixture\n${SOURCE_PERMISSION_ERROR}\nSource listing failed for a fixture account`, - ])( - 'recognizes a complete permission notice within composed diagnostics: %s', - async (lastSyncError) => { - await db - .update(knowledgeConnector) - .set({ lastSyncError }) - .where(eq(knowledgeConnector.id, driveId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'needs_attention', - issue: 'permission_sync_incomplete', - }) - } - ) - - it('does not classify a provider message containing the permission text as its own notice', async () => { - await db - .update(knowledgeConnector) - .set({ lastSyncError: `Provider message: ${SOURCE_PERMISSION_ERROR}` }) - .where(eq(knowledgeConnector.id, driveId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'needs_attention', - issue: 'sync_failed', - }) - }) - - it('ignores permission notices on paused sources', async () => { - await db - .update(knowledgeConnector) - .set({ lastSyncError: `Directory refresh incomplete: fixture\n${SOURCE_PERMISSION_ERROR}` }) - .where(eq(knowledgeConnector.id, pausedDriveId)) - expect(await provider('google_drive')).toMatchObject({ status: 'active', issue: null }) - }) - - it('counts configured sources independently of viewer ACLs, and excludes workspace and untouched providers', async () => { - const result = await readOrganizationSearchOverview.execute({ principal, input }) - expect(result.providers).toEqual( - expect.arrayContaining([ - { - connectorType: 'google_drive', - sourceCount: 2, - approved: true, - status: 'active', - issue: null, - isSyncing: false, - hasPendingSync: false, - }, - { - connectorType: 'gmail', - sourceCount: 1, - approved: true, - status: 'active', - issue: null, - isSyncing: false, - hasPendingSync: false, - }, - ]) - ) - expect(result.providers).toHaveLength(2) - expect(JSON.stringify(result)).not.toMatch(/private-title|someone-else|fixture connection/i) - const visible = await listSearchSources.execute({ - principal, - input: { ...input, connectorType: 'google_drive' }, - }) - expect(visible.sources).toHaveLength(2) - expect(visible.sources.every((source) => !source.hasViewerDocuments)).toBe(true) - }) - it('keeps explicit approvals and deactivations visible before source creation', async () => { - await db.insert(organizationSearchIntegration).values([ - { organizationId: ids.organizationId, connectorType: 'github', approved: true }, - { organizationId: ids.organizationId, connectorType: 'confluence', approved: false }, - { organizationId: ids.organizationId, connectorType: 'notion', approved: true }, - ]) - expect(await provider('github')).toMatchObject({ - sourceCount: 0, - approved: true, - status: 'needs_setup', - }) - expect(await provider('confluence')).toMatchObject({ - sourceCount: 0, - approved: false, - status: 'paused', - }) - expect(await provider('notion')).toBeUndefined() - }) - it('reports waiting accounts despite an empty member run having a completion timestamp', async () => { - await db - .update(knowledgeConnectorMember) - .set({ status: 'disabled' }) - .where(eq(knowledgeConnectorMember.id, memberId)) - await db - .update(knowledgeConnector) - .set({ memberSyncStatus: 'pending' }) - .where(eq(knowledgeConnector.id, gmailId)) - expect(await provider('gmail')).toMatchObject({ status: 'waiting_for_connections' }) - }) - it('distinguishes queued member continuation from active indexing and partial failure', async () => { - await db - .update(knowledgeConnectorMember) - .set({ listingCheckpoint: { cursor: 'fixture' } }) - .where(eq(knowledgeConnectorMember.id, memberId)) - const logId = generateId() - await db.insert(knowledgeConnectorMemberSyncLog).values({ - id: logId, - connectorId: gmailId, - status: 'partial', - membersIncomplete: 1, - docsFailed: 0, - processingDispatchFailed: 0, - completedAt: new Date(), - }) - expect(await provider('gmail')).toMatchObject({ - status: 'active', - isSyncing: false, - hasPendingSync: true, - }) - await db - .update(knowledgeConnectorMemberSyncLog) - .set({ membersFailed: 1 }) - .where(eq(knowledgeConnectorMemberSyncLog.id, logId)) - expect(await provider('gmail')).toMatchObject({ status: 'needs_attention' }) - await db - .update(knowledgeConnectorMemberSyncLog) - .set({ membersFailed: 0 }) - .where(eq(knowledgeConnectorMemberSyncLog.id, logId)) - await db - .update(knowledgeConnectorMember) - .set({ listingCheckpoint: null }) - .where(eq(knowledgeConnectorMember.id, memberId)) - expect(await provider('gmail')).toMatchObject({ status: 'needs_attention' }) - await db - .update(knowledgeConnector) - .set({ nextMemberSyncAt: sql`statement_timestamp() - interval '1 second'` }) - .where(eq(knowledgeConnector.id, gmailId)) - expect(await provider('gmail')).toMatchObject({ - status: 'active', - isSyncing: false, - hasPendingSync: true, - }) - await db - .update(knowledgeConnector) - .set({ nextMemberSyncAt: null }) - .where(eq(knowledgeConnector.id, gmailId)) - await db.insert(knowledgeConnectorMemberSyncLog).values({ - id: generateId(), - connectorId: gmailId, - status: 'completed', - docsFailed: 0, - processingDispatchFailed: 0, - startedAt: new Date(Date.now() + 1000), - completedAt: new Date(), - }) - expect(await provider('gmail')).toMatchObject({ status: 'active' }) - }) - it('ignores intentional skips while reporting inaccessible source and indexing failures', async () => { - await db - .update(document) - .set({ processingStatus: 'failed', contentHash: 'immutable-sha', fileUrl: '' }) - .where(eq(document.id, documentId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'active', - issue: null, - isSyncing: false, - }) - await db - .update(knowledgeConnector) - .set({ lastSyncError: 'previous sync failed' }) - .where(eq(knowledgeConnector.id, driveId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'needs_attention', - issue: 'sync_failed', - isSyncing: false, - }) - await db - .update(knowledgeConnector) - .set({ lastSyncError: null }) - .where(eq(knowledgeConnector.id, driveId)) - await db.update(document).set({ contentHash: null }).where(eq(document.id, documentId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'needs_attention', - issue: 'document_indexing_failed', - isSyncing: false, - }) - await db - .update(document) - .set({ contentHash: 'immutable-sha', storageKey: 'fixture-retained-artifact' }) - .where(eq(document.id, documentId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'needs_attention', - issue: 'document_indexing_failed', - isSyncing: false, - }) - await db.update(document).set({ userExcluded: true }).where(eq(document.id, documentId)) - expect(await provider('google_drive')).toMatchObject({ status: 'active' }) - await db - .update(document) - .set({ userExcluded: false, processingStatus: 'processing' }) - .where(eq(document.id, documentId)) - expect(await provider('google_drive')).toMatchObject({ status: 'indexing' }) - await db - .update(knowledgeConnector) - .set({ lastSyncError: 'previous sync failed' }) - .where(eq(knowledgeConnector.id, driveId)) - expect(await provider('google_drive')).toMatchObject({ - status: 'needs_attention', - isSyncing: true, - }) - }) - it('surfaces retained source errors and stale member permissions, while pause and deactivation take precedence', async () => { - await db - .update(knowledgeConnector) - .set({ lastSyncError: 'private-provider-error' }) - .where(eq(knowledgeConnector.id, driveId)) - expect(await provider('google_drive')).toMatchObject({ status: 'needs_attention' }) - await db - .update(knowledgeConnectorMember) - .set({ - memberSyncedThrough: new Date(Date.now() - 3 * 86_400_000), - lastCompleteListingAt: new Date(Date.now() - 3 * 86_400_000), - }) - .where(eq(knowledgeConnectorMember.id, memberId)) - expect(await provider('gmail')).toMatchObject({ status: 'needs_attention' }) - await db - .update(knowledgeConnector) - .set({ status: 'paused' }) - .where(eq(knowledgeConnector.id, driveId)) - expect(await provider('google_drive')).toMatchObject({ status: 'paused' }) - await db - .insert(organizationSearchIntegration) - .values({ organizationId: ids.organizationId, connectorType: 'gmail', approved: false }) - expect(await provider('gmail')).toMatchObject({ - approved: false, - status: 'paused', - isSyncing: false, - }) - }) - it('requires a current organization admin even for a workspace administrator', async () => { - await expect( - readOrganizationSearchOverview.execute({ - principal: { ...principal, userId: ids.bobId }, - input, - }) - ).rejects.toMatchObject({ code: 'forbidden' }) - await expect( - readOrganizationSearchOverview.execute({ principal, input: { organizationId: generateId() } }) - ).rejects.toMatchObject({ code: 'not_found' }) - }) -}) diff --git a/apps/sim/lib/knowledge/__integration__/processing-lock-scope.integration.ts b/apps/sim/lib/knowledge/__integration__/processing-lock-scope.integration.ts index 427f1e758de..fb793b32330 100644 --- a/apps/sim/lib/knowledge/__integration__/processing-lock-scope.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/processing-lock-scope.integration.ts @@ -13,13 +13,6 @@ import { eq, inArray } from 'drizzle-orm' import postgres from 'postgres' import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' -/** These transaction checks exercise indexed Search, which Live Search normally disables. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) - const fixtures = vi.hoisted(() => ({ root: '', process: vi.fn(), embeddings: vi.fn() })) vi.mock('@/lib/uploads/core/setup.server', () => ({ get UPLOAD_DIR_SERVER() { @@ -61,11 +54,6 @@ describe('document processing commit lock scope', () => { beforeAll(async () => { fixtures.root = mkdtempSync(path.join(tmpdir(), 'sim-processing-lock-scope-')) await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) - /** A search index, the only kind of base whose chunks the keyword projection holds. */ - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) vi.spyOn(embeddingClient, 'assertKnowledgeEmbeddingCapacity').mockResolvedValue(undefined) fixtures.process.mockResolvedValue({ chunks, @@ -164,7 +152,7 @@ describe('document processing commit lock scope', () => { expect(await db.select().from(document).where(eq(document.id, file.documentId))).toMatchObject([ { processingStatus: 'completed', chunkCount: 3 }, ]) - for (const projection of ['embedding_search', 'embedding_keyword_search']) { + for (const projection of ['embedding_search']) { expect( await db.$client.unsafe( `SELECT count(*)::int AS count FROM ${projection} WHERE document_id = $1`, @@ -228,7 +216,7 @@ describe('document processing commit lock scope', () => { .from(embedding) .where(eq(embedding.documentId, file.documentId)) ).toEqual([]) - for (const projection of ['embedding_search', 'embedding_keyword_search']) { + for (const projection of ['embedding_search']) { expect( await db.$client.unsafe(`SELECT id FROM ${projection} WHERE document_id = $1`, [ file.documentId, diff --git a/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts b/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts index 2003f25e31e..a3b567598f1 100644 --- a/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/provider-processing-recovery.integration.ts @@ -11,7 +11,6 @@ import { document, embedding, knowledgeBase, - member, organization, outboxEvent, rateLimitBucket, @@ -25,22 +24,13 @@ import { and, eq, inArray, sql } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' const fixtureStorage = vi.hoisted(() => ({ root: '' })) -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/lib/uploads/core/setup.server', () => ({ get UPLOAD_DIR_SERVER() { return fixtureStorage.root }, })) -import { - resolveBillingAttribution, - resolveOrganizationBillingAttribution, -} from '@/lib/billing/core/billing-attribution' +import { resolveBillingAttribution } from '@/lib/billing/core/billing-attribution' import { env } from '@/lib/core/config/env' import { processOutboxEventById } from '@/lib/core/outbox/service' import * as egress from '@/lib/core/security/input-validation.server' @@ -60,13 +50,12 @@ import { recordMemberObservations, } from '@/lib/knowledge/connectors/member-observations' import { createContentSyncLease, createMemberSyncLease } from '@/lib/knowledge/connectors/sync-lock' -import { addDocument, persistDocumentAcls } from '@/lib/knowledge/connectors/sync-persistence' +import { addDocument } from '@/lib/knowledge/connectors/sync-persistence' import { KNOWLEDGE_DOCUMENT_CONTINUATION_OUTBOX_EVENT } from '@/lib/knowledge/documents/processing-continuation-dispatch' import { knowledgeDocumentProcessingOutboxHandlers } from '@/lib/knowledge/documents/processing-outbox-handler' import { assertDocumentProcessingPayload } from '@/lib/knowledge/documents/processing-payload' import * as providerContinuation from '@/lib/knowledge/documents/processing-provider-continuation' import { processDocumentsWithQueue } from '@/lib/knowledge/documents/service' -import { searchScopedKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' const PNG = Buffer.from( 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+jRZkAAAAASUVORK5CYII=', @@ -114,7 +103,7 @@ describe('provider throttling resumes the shared indexing pipeline', () => { await db.$client.end() }) - it.each(['regular KB', 'member source', 'organization Search'] as const)( + it.each(['regular KB', 'member source'] as const)( 'recovers a %s after Mistral 429 without burning dispatches or charging twice', async (scope) => { vi.useRealTimers() @@ -129,28 +118,6 @@ describe('provider throttling resumes the shared indexing pipeline', () => { connectorId = memberFixture.connectorId lease = createMemberSyncLease(connectorId, memberFixture.runId) } - const orgOwned = scope === 'organization Search' - if (orgOwned) { - await db.insert(member).values([ - { - id: generateId(), - organizationId: ids.organizationId, - userId: ids.aliceId, - role: 'owner', - }, - { - id: generateId(), - organizationId: ids.organizationId, - userId: ids.bobId, - role: 'member', - }, - ]) - await db - .update(knowledgeBase) - .set({ workspaceId: null, organizationId: ids.organizationId, isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - } - const file = await addDocument( ids.knowledgeBaseId, connectorId, @@ -163,11 +130,9 @@ describe('provider throttling resumes the shared indexing pipeline', () => { contentHash: 'synthetic-scan-v1', sourceFile: { bytes: PNG, fileName: 'Orion scan.png', mimeType: 'image/png' }, }, - orgOwned - ? { userId: ids.aliceId, workspaceId: null, organizationId: ids.organizationId } - : { userId: ids.aliceId, workspaceId: ids.workspaceId }, + { userId: ids.aliceId, workspaceId: ids.workspaceId }, undefined, - scope === 'member source' ? 'members' : orgOwned ? 'admin' : 'workspace', + scope === 'member source' ? 'members' : 'workspace', lease ) if (scope === 'regular KB') { @@ -180,11 +145,6 @@ describe('provider throttling resumes the shared indexing pipeline', () => { memberFixture.runId ) await materializeDocumentAcls(connectorId, [file.documentId]) - } else { - await persistDocumentAcls( - connectorId, - new Map([['orion-scan', [`u:${ids.aliceId}@fixture.test`]]]) - ) } let ocrRequests = 0 @@ -260,15 +220,10 @@ describe('provider throttling resumes the shared indexing pipeline', () => { arrayBuffer: () => response.arrayBuffer(), } }) - const billing = orgOwned - ? await resolveOrganizationBillingAttribution({ - actorUserId: ids.aliceId, - organizationId: ids.organizationId, - }) - : await resolveBillingAttribution({ - actorUserId: ids.aliceId, - workspaceId: ids.workspaceId, - }) + const billing = await resolveBillingAttribution({ + actorUserId: ids.aliceId, + workspaceId: ids.workspaceId, + }) const requestId = generateId() const startedAt = Date.now() const holdParentHandoff = scope === 'regular KB' @@ -431,26 +386,16 @@ describe('provider throttling resumes the shared indexing pipeline', () => { expect(ocrRequests).toBe(2) expect(await charges()).toHaveLength(1) const principal = { kind: 'session' as const, userId: ids.aliceId, sessionId: generateId() } - const result = orgOwned - ? await searchScopedKnowledge.execute({ - principal, - input: { - organizationId: ids.organizationId, - query: 'Orion', - topK: 3, - searchMode: 'hybrid', - }, - }) - : await searchKnowledge.execute({ - principal, - input: { - workspaceId: ids.workspaceId, - knowledgeBaseIds: [ids.knowledgeBaseId], - query: 'Orion', - topK: 3, - searchMode: 'hybrid', - }, - }) + const result = await searchKnowledge.execute({ + principal, + input: { + workspaceId: ids.workspaceId, + knowledgeBaseIds: [ids.knowledgeBaseId], + query: 'Orion', + topK: 3, + searchMode: 'hybrid', + }, + }) expect(result.results.map((row) => row.documentId)).toContain(file.documentId) if (scope === 'member source') { const hidden = await searchKnowledge.execute({ diff --git a/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts b/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts deleted file mode 100644 index 9eea5b6cc4a..00000000000 --- a/apps/sim/lib/knowledge/__integration__/read-indexed-document.integration.ts +++ /dev/null @@ -1,343 +0,0 @@ -/** Real scoped document resolution, ACLs, ingestion, pagination, and provenance on disposable Postgres. */ -import { mkdtempSync } from 'node:fs' -import { rm } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import path from 'node:path' -import type { Principal } from '@sim/auth/principal' -import { db } from '@sim/db' -import { - document, - embedding, - knowledgeBase, - knowledgeExternalGroup, - member, - organization, - organizationSearchIntegration, - user, - workspace, -} from '@sim/db/schema' -import { generateId } from '@sim/utils/id' -import { eq, inArray } from 'drizzle-orm' -import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' - -const fixtures = vi.hoisted(() => ({ storageRoot: '' })) -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) -vi.mock('@/lib/uploads/core/setup.server', () => ({ - get UPLOAD_DIR_SERVER() { - return fixtures.storageRoot - }, -})) -vi.mock('@/lib/embeddings', async () => ({ - ...(await import('@/lib/embeddings/client')), - assertKnowledgeEmbeddingCapacity: async () => {}, - embedKnowledge: async (texts: string[]) => ({ - embeddings: texts.map(() => [1, ...Array(1535).fill(0)]), - totalTokens: texts.length, - billableTokens: 0, - isBYOK: true, - modelName: 'text-embedding-3-small', - pricingId: 'text-embedding-3-small', - }), -})) - -import { resolveOrganizationBillingAttribution } from '@/lib/billing/core/billing-attribution' -import { - createKnowledgeAclFixtureIds, - seedKnowledgeAclFixture, -} from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { confluencePageAcl } from '@/lib/knowledge/access/confluence-permissions' -import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' -import { addDocument, persistDocumentAcls } from '@/lib/knowledge/connectors/sync-persistence' -import { processDocumentAsync } from '@/lib/knowledge/documents/service' -import { - type ReadIndexedKnowledgeDocumentInput, - readIndexedKnowledgeDocument, -} from '@/lib/sim-search/indexed/documents/read-indexed-document' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -describe('indexed document references', () => { - const ids = createKnowledgeAclFixtureIds() - const other = createKnowledgeAclFixtureIds() - const alice: Principal = { kind: 'session', userId: ids.aliceId, sessionId: 'fixture-alice' } - const bob: Principal = { kind: 'session', userId: ids.bobId, sessionId: 'fixture-bob' } - const sourceUrl = 'https://fixture.atlassian.net/wiki/pages/target?view=all#section' - const content = Array.from( - { length: 240 }, - (_, index) => - `Document section ${index}: the customer setup checklist includes secure source connections, indexing, search, and citations. ` - ).join('\n\n') - let documentId: string - let hiddenDocumentId: string - - async function ingest(externalId: string) { - const doc = await addDocument( - ids.knowledgeBaseId, - ids.connectorId, - 'google_drive', - { - externalId, - mimeType: 'text/plain', - title: externalId, - content, - contentHash: externalId, - sourceUrl, - }, - { userId: ids.aliceId, workspaceId: null, organizationId: ids.organizationId }, - undefined, - 'admin', - createContentSyncLease(ids.connectorId, ids.lockId) - ) - await processDocumentAsync( - ids.knowledgeBaseId, - doc.documentId, - doc, - {}, - await resolveOrganizationBillingAttribution({ - actorUserId: ids.aliceId, - organizationId: ids.organizationId, - }) - ) - return doc.documentId - } - - function read(principal: Principal, input: Partial = {}) { - return readIndexedKnowledgeDocument.execute({ - principal, - input: { - organizationId: ids.organizationId, - target: { kind: 'url', url: sourceUrl }, - limit: 20, - resultSecretRegistry: new ResolvedSecretTraceRegistry(), - ...input, - }, - }) - } - - beforeAll(async () => { - fixtures.storageRoot = mkdtempSync(path.join(tmpdir(), 'sim-indexed-document-')) - vi.stubGlobal('fetch', async () => { - throw new Error('Indexed document reads must never fetch provider URLs') - }) - await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) - await seedKnowledgeAclFixture(other, { connectorType: 'google_drive' }) - await db - .update(knowledgeBase) - .set({ workspaceId: null, organizationId: ids.organizationId, isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - await db.insert(member).values([ - { id: generateId(), userId: ids.aliceId, organizationId: ids.organizationId, role: 'owner' }, - { id: generateId(), userId: ids.bobId, organizationId: ids.organizationId, role: 'member' }, - ]) - await db - .update(knowledgeExternalGroup) - .set({ workspaceId: null, organizationId: ids.organizationId }) - .where(inArray(knowledgeExternalGroup.id, ids.groupIds)) - await db - .insert(organizationSearchIntegration) - .values({ organizationId: ids.organizationId, connectorType: 'google_drive', approved: true }) - documentId = await ingest('visible-target') - hiddenDocumentId = await ingest('hidden-target') - const sourceAcl = confluencePageAcl({ - providerId: 'google-drive', - tenantId: 'fixture-tenant', - spacePrincipals: [{ kind: 'group', id: 'space' }], - restrictionChain: [[{ kind: 'group', id: 'page' }]], - }) - await persistDocumentAcls(ids.connectorId, new Map([['visible-target', sourceAcl]])) - }) - - afterAll(async () => { - for (const fixture of [ids, other]) { - await db.delete(workspace).where(eq(workspace.id, fixture.workspaceId)) - await db.delete(organization).where(eq(organization.id, fixture.organizationId)) - await db.delete(user).where(eq(user.id, fixture.aliceId)) - await db.delete(user).where(eq(user.id, fixture.bobId)) - } - await rm(fixtures.storageRoot, { recursive: true, force: true }) - vi.unstubAllGlobals() - await db.$client.end() - }) - - it('reads an exact URL without treating inaccessible duplicates as ambiguous', async () => { - const result = await read(alice) - expect(result.documentId).toBe(documentId) - expect(result.sourceUrl).toBe(sourceUrl) - expect(result.chunks?.length).toBeGreaterThan(3) - expect(result.chunks?.[0].content).toContain('Document section 0') - }) - - it('conceals URLs from a member without the provider permissions', async () => { - await expect(read(bob)).rejects.toThrow('Document not found') - }) - - it('rejects ambiguous accessible URLs and still accepts explicit document identifiers', async () => { - const [original] = await db.select().from(document).where(eq(document.id, documentId)) - await db - .update(document) - .set({ - acl: original.acl, - aclRequirements: original.aclRequirements, - aclVerifiedAt: original.aclVerifiedAt, - }) - .where(eq(document.id, hiddenDocumentId)) - try { - await expect(read(alice)).rejects.toThrow('Multiple accessible documents use this URL') - await expect(read(alice, { target: { kind: 'id', documentId } })).resolves.toMatchObject({ - documentId, - }) - } finally { - await db - .update(document) - .set({ acl: [], aclRequirements: [] }) - .where(eq(document.id, hiddenDocumentId)) - } - }) - - it('requires the canonical organization index for both URL and ID reads', async () => { - await db - .update(knowledgeBase) - .set({ deletedAt: new Date() }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - try { - await expect(read(alice)).rejects.toThrow('Document not found') - await expect(read(alice, { target: { kind: 'id', documentId } })).rejects.toThrow( - 'Document not found' - ) - } finally { - await db - .update(knowledgeBase) - .set({ deletedAt: null }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - } - }) - - it('does not resolve document URLs across organizations', async () => { - await expect(read(alice, { organizationId: other.organizationId })).rejects.toMatchObject({ - code: 'not_found', - }) - }) - - it('rejects caller kinds and revoked owner access before resolving the URL', async () => { - await expect( - read({ kind: 'workspace_api_key', workspaceId: other.workspaceId, keyId: 'foreign' }) - ).rejects.toMatchObject({ code: 'forbidden' }) - await expect( - read({ kind: 'session', userId: other.aliceId, sessionId: 'outsider' }) - ).rejects.toMatchObject({ code: 'not_found' }) - }) - - it.each([ - 'javascript:alert(1)', - 'https://user:secret@source.test/doc', - 'https:source.test/doc', - '/path', - 'https://source.test/a\nb', - ])('rejects malformed or unsafe provider URL %s', async (url) => { - await expect(read(alice, { target: { kind: 'url', url } })).rejects.toThrow( - 'HTTP or HTTPS document URL' - ) - }) - - it('does not rewrite provider queries or fragment identity during exact lookup', async () => { - await expect( - read(alice, { target: { kind: 'url', url: sourceUrl.replace('#section', '') } }) - ).rejects.toThrow('Document not found') - }) - - it.each(['enabled', 'userExcluded', 'archivedAt', 'deletedAt'] as const)( - 'omits documents when %s removes them from reading', - async (field) => { - const value = field === 'enabled' ? false : field === 'userExcluded' ? true : new Date() - await db - .update(document) - .set({ [field]: value }) - .where(eq(document.id, documentId)) - try { - await expect(read(alice)).rejects.toThrow('Document not found') - const target = { kind: 'id', documentId } as const - for (const page of [{}, { offset: 1 }, { aroundChunkIndex: 1 }]) { - await expect(read(alice, { target, ...page })).rejects.toThrow('Document not found') - } - } finally { - await db - .update(document) - .set({ [field]: field === 'enabled' ? true : field === 'userExcluded' ? false : null }) - .where(eq(document.id, documentId)) - } - } - ) - - it('centers enabled context around the actual matching chunk and resumes from the returned offset', async () => { - const all = await read(alice) - const first = all.chunks![0] - const target = all.chunks![4] - await db.update(embedding).set({ enabled: false }).where(eq(embedding.id, first.id)) - try { - const result = await read(alice, { aroundChunkIndex: target.chunkIndex, limit: 3 }) - expect(result.pagination?.offset).toBe(1) - expect(result.chunks?.map((chunk) => chunk.id)).toEqual( - all.chunks!.slice(2, 5).map((chunk) => chunk.id) - ) - const next = await read(alice, { - offset: result.pagination!.offset + result.chunks!.length, - limit: 3, - }) - expect(next.chunks?.[0].id).toBe(all.chunks![5].id) - const single = await read(alice, { aroundChunkIndex: target.chunkIndex, limit: 1 }) - expect(single.chunks?.map((chunk) => chunk.id)).toEqual([target.id]) - await expect(read(alice, { aroundChunkIndex: first.chunkIndex })).rejects.toThrow( - 'Document chunk not found' - ) - await expect(read(alice, { aroundChunkIndex: 999999 })).rejects.toThrow( - 'Document chunk not found' - ) - } finally { - await db.update(embedding).set({ enabled: true }).where(eq(embedding.id, first.id)) - } - }) - - it('bounds pagination and forbids ambiguous page coordinates', async () => { - for (const input of [ - { limit: 0 }, - { limit: 51 }, - { offset: -1 }, - { offset: 1_000_001 }, - { aroundChunkIndex: -1 }, - { aroundChunkIndex: 1.5 }, - ]) { - await expect(read(alice, input)).rejects.toThrow('Invalid document page bounds') - } - await expect(read(alice, { offset: 0, aroundChunkIndex: 1 })).rejects.toThrow( - 'Use offset or aroundChunkIndex, not both' - ) - }) - - it('retains metadata-only reads until indexing completes', async () => { - await db - .update(document) - .set({ processingStatus: 'processing' }) - .where(eq(document.id, documentId)) - try { - const result = await read(alice) - expect(result.processingStatus).toBe('processing') - expect(result).not.toHaveProperty('chunks') - expect(result).not.toHaveProperty('pagination') - } finally { - await db - .update(document) - .set({ processingStatus: 'completed' }) - .where(eq(document.id, documentId)) - } - }) - - it('stops cancelled reads', async () => { - const controller = new AbortController() - controller.abort() - await expect(read(alice, { signal: controller.signal })).rejects.toThrow() - }) -}) diff --git a/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts b/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts index 5eb6e667a39..687a872fa56 100644 --- a/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-latency.integration.ts @@ -3,37 +3,22 @@ import { readFileSync, statSync, writeFileSync } from 'node:fs' import type { Principal } from '@sim/auth/principal' import { db } from '@sim/db' import { - copilotChats, credential, credentialGroup, document, - embedding, - embeddingSearch, knowledgeBase, knowledgeConnector, knowledgeConnectorMember, knowledgeDocumentObservation, - member, organization, user, workspace, } from '@sim/db/schema' import { createLogger, Logger } from '@sim/logger' import { generateId } from '@sim/utils/id' -import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' -import { NextRequest } from 'next/server' +import { eq, inArray, sql } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, type MockInstance, vi } from 'vitest' - -/** Turns indexed organization search on: its search-index knowledge bases are read through it. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) - import { z } from 'zod' -import { workspaceKnowledgeSearchDataSchema } from '@/lib/api/contracts/knowledge/search' -import { internalSessionAuth } from '@/lib/api/server/routes' import { seedSearchReaderFixture } from '@/lib/knowledge/__integration__/seed-search-reader-fixture' import { createKnowledgeAclFixtureIds, @@ -41,7 +26,6 @@ import { seedKnowledgeMemberFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { type KnowledgeSearchTagFilter, searchKnowledge } from '@/lib/knowledge/application/search' -import type { KbEmbeddingDimensions } from '@/lib/knowledge/embedding-models' import { SearchBudget, SearchDeadlineError, @@ -49,18 +33,6 @@ import { } from '@/lib/knowledge/search/budget' import type { SearchStage } from '@/lib/knowledge/search/diagnostics' import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' -import { embeddingCandidateDistance } from '@/lib/knowledge/vector-columns' -import type { - MothershipStreamV1CheckpointPausePayload, - MothershipStreamV1ToolCallDescriptor, -} from '@/lib/mothership/generated/mothership-stream-v1' -import { isContractStreamEventEnvelope } from '@/lib/mothership/request/session/contract' -import { - readDocumentServerTool, - searchWorkspaceServerTool, -} from '@/lib/mothership/tools/server/knowledge/workspace-search' -import { POST as searchRoute } from '@/app/api/knowledge/search/route' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' /** Initialize controlled provider configuration before the real application modules load. */ vi.hoisted(() => { @@ -74,7 +46,6 @@ vi.hoisted(() => { } }) -const externalFetch = globalThis.fetch const enabled = process.env.KNOWLEDGE_SEARCH_PERFORMANCE_TEST === 'true' const batchSize = 1000 const MIN_CHUNK_COUNT = 5000 @@ -86,7 +57,6 @@ const unrelatedChunkCount = Number( const evictSharedBuffers = process.env.KNOWLEDGE_SEARCH_PERFORMANCE_EVICT_BUFFERS === 'true' const dimensions = 1536 const candidateDimensions = 512 -const HYBRID_CANDIDATE_LIMIT = 1600 const chunksPerDocument = 4 const logger = createLogger('SearchLatencyIntegration') const fixtureSchema = z.object({ @@ -119,7 +89,6 @@ const ids = reused?.fixture ?? createKnowledgeAclFixtureIds() const unrelated = reused?.unrelatedFixture ?? createKnowledgeAclFixtureIds() const fullWidthFixture = reused?.fullWidthFixture ?? createKnowledgeAclFixtureIds() const FULL_WIDTH_CHUNK_COUNT = 5000 -const organizationChatId = generateId() function topicVector(topic = 0) { const vector = Array.from({ length: dimensions }, (_, index) => Math.sin( @@ -137,20 +106,6 @@ const queryVector = topicVector() * The exact nearest chunks on the projection's stored halfvec, which is what the page's order * is measured against: the walk ranks on that column, and nothing rescores it. */ -async function exactProjectionNeighbors(vector: number[], limit: number, readerClause?: SQL) { - const distance = embeddingCandidateDistance( - dimensions as KbEmbeddingDimensions, - JSON.stringify(vector), - 'text-embedding-3-small' - ) - /** Unaliased: the distance expression qualifies its column with the table's own name. */ - return db.execute<{ id: string }>(sql`SELECT ${embeddingSearch.id} AS id FROM ${embeddingSearch} - INNER JOIN document d ON d.id = ${embeddingSearch.documentId} - WHERE ${embeddingSearch.knowledgeBaseId} = ${ids.knowledgeBaseId} - AND ${embeddingSearch.enabled} ${readerClause ?? sql``} - ORDER BY (${distance}) + 0, ${embeddingSearch.id} - LIMIT ${limit}`) -} const captured: CapturedQuery[] = [] const report: Record = { fixture: ids, @@ -170,7 +125,7 @@ const report: Record = { vectors: 'Normalized 512-dimensional topic/noise geometry with permuted copies across 1536 dimensions; verifies prefix candidate ranking, not semantic embedding quality', cache: evictSharedBuffers - ? 'Selected workspace and organization samples evict PostgreSQL shared buffers; operating-system cache is not cleared' + ? 'Selected workspace samples evict PostgreSQL shared buffers; operating-system cache is not cleared' : 'First and repeated samples; no claim of a cold operating-system cache', layout: reused ? 'Reused fixture; physical layout is inherited from its original report' @@ -283,19 +238,6 @@ function assertIndexedCandidates( expect(traversed['Actual Rows']).toBeLessThanOrEqual(candidateLimit) } -/** Small scopes must seek chunk metadata by document without reading the full vector projection. */ -function assertIndexedChunkProbe(node: ExplainNode): number { - let lookups = 0 - if (node['Relation Name'] === 'embedding_search') { - expect(['Index Scan', 'Index Only Scan']).toContain(node['Node Type']) - expect(node['Index Name']).toBe('embedding_search_document_lookup_idx') - expect((node.Output ?? []).join(' ')).not.toMatch(/(?:embedding_search\.)?(?:vector|binary)/) - lookups = node['Actual Loops'] - } - for (const child of node.Plans ?? []) lookups += assertIndexedChunkProbe(child) - return lookups -} - /** Keyword sort memory must scale with identities and scores, not the matched document text. */ function assertScalarKeywordSorts(node: ExplainNode) { if (node['Node Type'] === 'Sort') { @@ -310,7 +252,7 @@ function saveReport() { } /** Only the disposable fixture may evict shared buffers; the operating-system cache stays intact. */ -async function prepareOrganizationSample(label: string) { +async function prepareSample(label: string) { if (!evictSharedBuffers) return const [eviction] = await db.execute<{ buffers: number; evicted: number }>(sql` WITH cached AS MATERIALIZED ( @@ -354,6 +296,9 @@ const diagnosticSchema = z const resultSchema = z.object({ success: z.literal(true), data: z.object({ + retrieval: z + .object({ status: z.enum(['complete', 'partial']), timedOutLegs: z.array(z.string()) }) + .optional(), results: z.array( z.object({ documentId: z.string(), @@ -368,63 +313,22 @@ const resultSchema = z.object({ async function search( userId = ids.aliceId, query = 'Orion deployment', - filters: WorkspaceSearchFilters = {}, - organizationScope = false, - topK = 15 + filters: WorkspaceSearchFilters = {} ) { - return resultSchema.parse( - await searchWorkspaceServerTool.execute( - { query, topK, ...filters }, - { - userId, - ...(organizationScope - ? { organizationId: ids.organizationId, chatId: organizationChatId } - : { workspaceId: ids.workspaceId }), - requestMode: 'assistant', - toolCallId: generateId(), - copilotToolExecution: true, - resolvedSecretTraceRegistry: new ResolvedSecretTraceRegistry([], { - userId, - ...(organizationScope ? {} : { workspaceId: ids.workspaceId }), - }), - } - ) - ) -} - -async function searchDashboard( - query = 'Orion deployment', - userId = ids.aliceId, - organizationScope = false, - topK = 15 -) { - const authenticate = vi.spyOn(internalSessionAuth, 'authenticate').mockResolvedValue({ - kind: 'session', - userId, - sessionId: 'fixture-dashboard', + const data = await searchKnowledge.execute({ + principal: { kind: 'session', userId, sessionId: 'fixture-knowledge-search' }, + input: { + workspaceId: ids.workspaceId, + knowledgeBaseIds: [ids.knowledgeBaseId], + query, + topK: 15, + filters, + searchMode: 'hybrid', + allowPartialResults: true, + surface: 'api', + }, }) - try { - const response = await searchRoute( - new NextRequest('http://localhost/api/knowledge/search', { - method: 'POST', - headers: { 'content-type': 'application/json' }, - body: JSON.stringify({ - ...(organizationScope - ? { organizationId: ids.organizationId } - : { workspaceId: ids.workspaceId }), - query, - topK, - }), - }) - ) - expect(response.status).toBe(200) - return { - success: true as const, - data: workspaceKnowledgeSearchDataSchema.parse((await response.json()).data), - } - } finally { - authenticate.mockRestore() - } + return resultSchema.parse({ success: true, data }) } async function searchWorkspaceKb( @@ -489,15 +393,6 @@ async function sample( const diagnostics = diagnosticSchema.parse(completed[0][1]) expect(diagnostics.stages.embedding.count).toBe(1) expect(diagnostics.stages.retrieval.count).toBe(1) - if (diagnostics.surface === 'copilot') { - const passageBytes = result.data.results.map((row) => Buffer.byteLength(row.content)) - expect(diagnostics.passageBytes).toBe(passageBytes.reduce((total, bytes) => total + bytes, 0)) - expect(diagnostics.maxPassageBytes).toBe(Math.max(0, ...passageBytes)) - expect(diagnostics.uniqueDocumentCount).toBe( - new Set(result.data.results.map((row) => row.documentId)).size - ) - expect(diagnostics.toolResultBytes).toBeGreaterThan(diagnostics.passageBytes!) - } expect(captured.length).toBeLessThan(300) const searches = captured.filter( (item) => @@ -685,6 +580,7 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu workspaceId: ids.workspaceId, organizationId: null, embeddingModel: 'text-embedding-3-small', + isSearchIndex: false, }) .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) await db @@ -693,17 +589,6 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu .where(eq(knowledgeConnector.id, ids.connectorId)) await db.delete(credential).where(eq(credential.workspaceId, ids.workspaceId)) await db.delete(credentialGroup).where(eq(credentialGroup.workspaceId, ids.workspaceId)) - await db - .delete(copilotChats) - .where( - and( - eq(copilotChats.organizationId, ids.organizationId), - eq(copilotChats.userId, ids.aliceId) - ) - ) - await db - .delete(member) - .where(and(eq(member.organizationId, ids.organizationId), eq(member.userId, ids.aliceId))) await db.execute( sql`UPDATE document SET acl = ARRAY[${`u:${ids.aliceId}@fixture.test`}], user_excluded = false, acl_verified_at = statement_timestamp() WHERE knowledge_base_id = ${ids.knowledgeBaseId}` ) @@ -715,10 +600,6 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu .update(knowledgeBase) .set({ embeddingModel: 'text-embedding-3-small' }) .where(inArray(knowledgeBase.id, [ids.knowledgeBaseId, unrelated.knowledgeBaseId])) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) const indexes = await db.execute<{ indexname: string indexdef: string @@ -904,7 +785,7 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu }) it.each(['vector', 'keyword', 'both'] as const)( - 'keeps the Assistant budget when %s SQL branches are delayed', + 'keeps knowledge-search deadlines when %s SQL branches are delayed', async (delayedLegs) => { diagnosticLog?.mockClear() let vectorDelayed = false @@ -927,19 +808,7 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu }) as Promise }) try { - const result = await searchWorkspaceServerTool.execute( - { query: 'Orion deployment', topK: 15 }, - { - userId: ids.aliceId, - workspaceId: ids.workspaceId, - toolCallId: generateId(), - copilotToolExecution: true, - resolvedSecretTraceRegistry: new ResolvedSecretTraceRegistry([], { - userId: ids.aliceId, - workspaceId: ids.workspaceId, - }), - } - ) + const result = await search() const completed = diagnosticLog?.mock.calls.find( ([message]) => message === 'Knowledge search completed' ) @@ -971,52 +840,7 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu expect( parsed.data.results.every((row) => row.knowledgeBaseId === ids.knowledgeBaseId) ).toBe(true) - report[`assistant.deadline.${delayedLegs}`] = { resultCount: parsed.data.results.length } - } finally { - delayed.mockRestore() - } - }, - 30_000 - ) - - it.each(['vector', 'both'] as const)( - 'returns incomplete dashboard coverage when %s SQL branches exceed their deadline', - async (delayedLegs) => { - diagnosticLog?.mockClear() - const query = SearchBudget.prototype.query - const delayed = vi.spyOn(SearchBudget.prototype, 'query').mockImplementation(function ( - this: SearchBudget, - stage: SearchStage, - run: (executor: SearchExecutor) => PromiseLike - ): Promise { - return query.call(this, stage, async (tx) => { - if (delayedLegs === 'both' || this.leg === 'vector') - await tx.execute(sql`SELECT pg_sleep(${delayedLegs === 'both' ? 9 : 4})`) - return run(tx) - }) as Promise - }) - try { - const { data } = await searchDashboard() - const completed = diagnosticLog?.mock.calls.find( - ([message]) => message === 'Knowledge search completed' - ) - const diagnostics = diagnosticSchema.parse(completed?.[1]) - expect(diagnostics.outcome).toBe('partial') - expect(diagnostics.vectorBudgetMs).toBe(3000) - expect(diagnostics.stages.vector.totalMs).toBeGreaterThan(2500) - expect(diagnostics.stages.vector.totalMs).toBeLessThan(4000) - expect(data.retrieval).toEqual({ - status: 'partial', - timedOutLegs: delayedLegs === 'both' ? ['vector', 'keyword'] : ['vector'], - }) - if (delayedLegs === 'both') expect(data.results).toEqual([]) - else { - expect(data.results.length).toBeGreaterThan(0) - expect( - data.results.every((result) => result.knowledgeBaseId === ids.knowledgeBaseId) - ).toBe(true) - } - report[`dashboard.deadline.${delayedLegs}`] = { resultCount: data.results.length } + report[`knowledge.deadline.${delayedLegs}`] = { resultCount: parsed.data.results.length } } finally { delayed.mockRestore() } @@ -1024,216 +848,11 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu 30_000 ) - it('records first and repeated application searches with the actual SQL plans', async () => { - const before = embeddingCalls - for (let iteration = 0; iteration < 2; iteration++) { - const { result, plans, diagnostics } = await sample(`broad.${iteration}`, () => search()) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results).toHaveLength(15) - expect(result.data.results.every((row) => row.knowledgeBaseId === ids.knowledgeBaseId)).toBe( - true - ) - expect(plans.length).toBeGreaterThanOrEqual(2) - const vectorPlans = plans.filter((plan) => plan.kind === 'vector') - expect(vectorPlans).toHaveLength(1) - expect(vectorPlans[0].plan[0].Plan['Actual Rows']).toBeGreaterThan(0) - assertCompactCandidates(vectorPlans[0].plan[0].Plan) - expect(plans.some((plan) => plan.kind === 'page')).toBe(true) - const page = plans.find((plan) => plan.kind === 'page')! - const actual = await db.$client.unsafe(page.query, page.parameters).values() - const expected = await exactProjectionNeighbors(queryVector, actual.length) - const expectedIds = new Set(expected.map(({ id }) => id)) - const recall = actual.filter(([id]) => expectedIds.has(id)).length / expected.length - expect(recall).toBeGreaterThanOrEqual(0.95) - report[`recall.${iteration}`] = { neighbors: expected.length, recall } - saveReport() - } - expect(embeddingCalls - before).toBe(2) - }, 180_000) - - it('preserves exact-neighbor recall across different query vectors', async () => { - for (const topic of [3, 11, 23]) { - const { plans, diagnostics } = await sample(`topic.${topic}`, () => - search(ids.aliceId, `Topic ${topic} deployment`) - ) - expectCompleteVectorSearch(diagnostics) - const candidates = plans.find((plan) => plan.kind === 'vector')! - assertCompactCandidates(candidates.plan[0].Plan) - const page = plans.find((plan) => plan.kind === 'page')! - const actual = await db.$client.unsafe(page.query, page.parameters).values() - const expected = await exactProjectionNeighbors(topicVector(topic), actual.length) - const expectedIds = new Set(expected.map(({ id }) => id)) - const recall = actual.filter(([id]) => expectedIds.has(id)).length / expected.length - expect(recall).toBeGreaterThanOrEqual(0.95) - report[`recall.topic.${topic}`] = { neighbors: expected.length, recall } - saveReport() - } - }, 180_000) - - it('compares the Search tab and Assistant with the same person, query and index', async () => { - const dashboard = await sample('dashboard', searchDashboard) - const assistant = await sample('assistant.comparison', () => search()) - expectCompleteVectorSearch(dashboard.diagnostics) - expectCompleteVectorSearch(assistant.diagnostics) - expect(dashboard.diagnostics.surface).toBe('dashboard') - expect(assistant.diagnostics.surface).toBe('copilot') - expect(dashboard.diagnostics.stages.result_provenance).toBeUndefined() - expect(assistant.diagnostics.stages.result_provenance.count).toBe(1) - expect(dashboard.result.data.results).toHaveLength(15) - expect(assistant.result.data.results).toHaveLength(15) - const dashboardVector = dashboard.plans.filter((plan) => plan.kind === 'vector') - const assistantVector = assistant.plans.filter((plan) => plan.kind === 'vector') - expect(dashboardVector).toHaveLength(1) - expect(assistantVector).toHaveLength(1) - expect(dashboardVector[0].query).toBe(assistantVector[0].query) - expect(dashboardVector[0].parameters).toEqual(assistantVector[0].parameters) - expect(dashboardVector[0].plan[0].Plan['Actual Rows']).toBeGreaterThan(0) - }, 180_000) - it('keeps inaccessible content out of an otherwise identical search', async () => { const { result } = await sample('denied', () => search(ids.bobId)) expect(result.data.results).toEqual([]) }, 180_000) - it('preserves recall when the nearest topic is mostly inaccessible within a broad permission scope', async () => { - const reader = `u:${ids.aliceId}@fixture.test` - /** Four consecutive topics share each document; hide 99% of the query's topic cluster. */ - await db.execute(sql`UPDATE document - SET acl = ARRAY[${`u:${ids.bobId}@fixture.test`}] - WHERE knowledge_base_id = ${ids.knowledgeBaseId} - AND external_id::int % 8 = 0 AND external_id::int % 800 <> 0`) - try { - for (const surface of ['copilot', 'dashboard'] as const) { - const { result, plans, diagnostics } = await sample( - `filtered-neighborhood.${surface}`, - surface === 'copilot' ? () => search() : searchDashboard - ) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results).toHaveLength(15) - const page = plans.find((plan) => plan.kind === 'page')! - expect(page).toBeDefined() - const actual = await db.$client.unsafe(page.query, page.parameters).values() - const expected = await exactProjectionNeighbors( - queryVector, - actual.length, - sql`AND d.acl @> ARRAY[${reader}]::text[]` - ) - expect(expected.length).toBeGreaterThan(0) - const expectedIds = new Set(expected.map(({ id }) => id)) - const recall = actual.filter(([id]) => expectedIds.has(id)).length / expected.length - expect(recall).toBeGreaterThanOrEqual(0.95) - report[`recall.filtered-neighborhood.${surface}`] = { neighbors: expected.length, recall } - saveReport() - } - } finally { - await db - .update(document) - .set({ acl: [reader] }) - .where(eq(document.knowledgeBaseId, ids.knowledgeBaseId)) - } - }, 180_000) - - it('ranks a small permission scope by its bounded IDs without a corpus-wide vector probe', async () => { - const lastTopicDocument = Math.floor((chunkCount / chunksPerDocument - 1) / 8) * 8 - const documentIds = [0, 8, 16].map( - (offset) => `${ids.workspaceId}-doc-${lastTopicDocument - offset}` - ) - await db - .update(document) - .set({ acl: [`u:${ids.aliceId}@fixture.test`, `u:${ids.bobId}@fixture.test`] }) - .where(inArray(document.id, documentIds)) - try { - for (const surface of ['copilot', 'dashboard'] as const) { - const { result, plans, diagnostics } = await sample(`small-scope.${surface}`, () => - surface === 'copilot' ? search(ids.bobId) : searchDashboard('Orion deployment', ids.bobId) - ) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results.length).toBeGreaterThan(0) - expect(result.data.results.every((row) => documentIds.includes(row.documentId))).toBe(true) - const probe = plans.filter((plan) => plan.kind === 'probe') - expect(probe).toHaveLength(1) - expect(probe[0].query).not.toContain('<=>') - expect(probe[0].plan[0].Plan['Actual Rows']).toBe(12) - expect(assertIndexedChunkProbe(probe[0].plan[0].Plan)).toBe(documentIds.length) - /** The page reads the bounded ranking's identities from the projection, never the original vectors. */ - const page = plans.filter((plan) => plan.kind === 'page') - expect(page).toHaveLength(1) - expect(page[0].query).not.toContain('"embedding"."embedding"') - } - } finally { - await db - .update(document) - .set({ acl: [`u:${ids.aliceId}@fixture.test`] }) - .where(inArray(document.id, documentIds)) - } - }, 180_000) - - it.each([200, 1000, 1596, 1600, 2000])( - 'keeps a selective scope of %s chunks within both retrieval budgets', - async (count) => { - const documentCount = count / chunksPerDocument - const documentIds = Array.from( - { length: documentCount }, - (_, index) => `${ids.workspaceId}-doc-${chunkCount / chunksPerDocument - 1 - index}` - ) - await db - .update(document) - .set({ acl: [`u:${ids.aliceId}@fixture.test`, `u:${ids.bobId}@fixture.test`] }) - .where(inArray(document.id, documentIds)) - try { - for (const surface of ['copilot', 'dashboard'] as const) { - const { result, plans, diagnostics } = await sample( - `selective-${count}.${surface}`, - () => - surface === 'copilot' - ? search(ids.bobId) - : searchDashboard('Orion deployment', ids.bobId) - ) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results.length).toBeGreaterThan(0) - expect(result.data.results.every((row) => documentIds.includes(row.documentId))).toBe( - true - ) - const probe = plans.find((plan) => plan.kind === 'probe')! - expect(probe.plan[0].Plan['Actual Rows']).toBe(Math.min(count, HYBRID_CANDIDATE_LIMIT)) - expect(assertIndexedChunkProbe(probe.plan[0].Plan)).toBe( - Math.min(documentCount, HYBRID_CANDIDATE_LIMIT / chunksPerDocument) - ) - expect(plans.filter((plan) => plan.kind === 'vector')).toHaveLength( - count < HYBRID_CANDIDATE_LIMIT ? 0 : 1 - ) - if (count > HYBRID_CANDIDATE_LIMIT) { - const page = plans.find((plan) => plan.kind === 'page')! - const actual = await db.$client.unsafe(page.query, page.parameters).values() - const expected = await exactProjectionNeighbors( - queryVector, - actual.length, - sql`AND ${embeddingSearch.documentId} IN (${sql.join( - documentIds.map((id) => sql`${id}`), - sql`, ` - )})` - ) - const expectedIds = new Set(expected.map(({ id }) => id)) - const recall = actual.filter(([id]) => expectedIds.has(id)).length / expected.length - expect(recall).toBeGreaterThanOrEqual(0.95) - report[`recall.selective-${count}.${surface}`] = { - neighbors: expected.length, - recall, - candidateScan: diagnostics.vectorCandidateScan, - } - saveReport() - } - } - } finally { - await db - .update(document) - .set({ acl: [`u:${ids.aliceId}@fixture.test`] }) - .where(inArray(document.id, documentIds)) - } - }, - 180_000 - ) - it('bounds member-observation searches and rejects suspended readers with current ACL checks', async () => { const fixture = await seedKnowledgeMemberFixture(ids) const [alice, bob] = fixture.members @@ -1263,28 +882,13 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu .where(inArray(document.id, documentIds)) await db.execute(sql`ANALYZE document`) await db.execute(sql`ANALYZE knowledge_document_observation`) - for (const surface of ['copilot', 'dashboard'] as const) { - const broad = await sample(`member-broad.${surface}`, () => - surface === 'copilot' ? search() : searchDashboard() - ) - expectCompleteVectorSearch(broad.diagnostics) - expect(broad.result.data.results).toHaveLength(15) - const broadProbe = broad.plans.find((plan) => plan.kind === 'probe')! - expect(broadProbe.plan[0].Plan['Actual Rows']).toBe(HYBRID_CANDIDATE_LIMIT) - expect(assertIndexedChunkProbe(broadProbe.plan[0].Plan)).toBe( - HYBRID_CANDIDATE_LIMIT / chunksPerDocument - ) - const { result, plans, diagnostics } = await sample(`member-scope.${surface}`, () => - surface === 'copilot' ? search(ids.bobId) : searchDashboard('Orion deployment', ids.bobId) - ) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results.length).toBeGreaterThan(0) - expect(result.data.results.every((row) => documentIds.includes(row.documentId))).toBe(true) - const probe = plans.find((plan) => plan.kind === 'probe')! - expect(probe).toBeDefined() - expect(probe.query).toContain('knowledge_document_observation') - expect(assertIndexedChunkProbe(probe.plan[0].Plan)).toBe(documentIds.length) - } + const broad = await sample('member-broad', () => search()) + expectCompleteVectorSearch(broad.diagnostics) + expect(broad.result.data.results).toHaveLength(15) + const { result, diagnostics } = await sample('member-scope', () => search(ids.bobId)) + expectCompleteVectorSearch(diagnostics) + expect(result.data.results.length).toBeGreaterThan(0) + expect(result.data.results.every((row) => documentIds.includes(row.documentId))).toBe(true) await db .update(knowledgeConnectorMember) .set({ status: 'suspended' }) @@ -1328,21 +932,15 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu } }, 180_000) - it.each(['copilot', 'dashboard'] as const)( - 'keeps %s searches for common keyword terms complete', - async (surface) => { - const { result, diagnostics } = await sample(`common-keyword.${surface}`, () => - surface === 'dashboard' - ? searchDashboard('Engineering operations') - : search(ids.aliceId, 'Engineering operations') - ) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results).toHaveLength(15) - }, - 180_000 - ) + it('keeps common-keyword searches complete', async () => { + const { result, diagnostics } = await sample('common-keyword', () => + search(ids.aliceId, 'Engineering operations') + ) + expectCompleteVectorSearch(diagnostics) + expect(result.data.results).toHaveLength(15) + }, 180_000) - it('runs two independent Assistant searches concurrently', async () => { + it('runs two independent knowledge-base searches concurrently', async () => { diagnosticLog?.mockClear() const start = performance.now() const results = await Promise.all([search(), search(ids.aliceId, 'Topic 11 deployment')]) @@ -1374,7 +972,7 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu WHERE knowledge_base_id = ${ids.knowledgeBaseId}`) for (const topic of [0, 11, 23]) { const label = `workspace-kb.topic.${topic}` - await prepareOrganizationSample(label) + await prepareSample(label) const { result, plans, diagnostics } = await sample(label, () => searchWorkspaceKb(`Topic ${topic} deployment`) ) @@ -1441,7 +1039,7 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu for (const concurrency of [2, 8]) { const label = `workspace-kb.concurrent.${concurrency}` - await prepareOrganizationSample(label) + await prepareSample(label) diagnosticLog?.mockClear() const started = performance.now() const results = await Promise.all( @@ -1531,285 +1129,4 @@ describe.skipIf(!enabled)('Knowledge search latency on a realistic indexed corpu const restored = await sample('live.restored', () => search()) expect(restored.result.data.results).toHaveLength(15) }, 180_000) - - it('keeps organization searches complete with stale ACL estimates and concurrent requests', async () => { - await db.insert(member).values({ - id: generateId(), - organizationId: ids.organizationId, - userId: ids.aliceId, - role: 'owner', - }) - await db.insert(copilotChats).values({ - id: organizationChatId, - organizationId: ids.organizationId, - userId: ids.aliceId, - type: 'mothership', - }) - await db - .update(knowledgeBase) - .set({ workspaceId: null, organizationId: ids.organizationId }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - await db - .update(knowledgeConnector) - .set({ connectorType: 'google_drive', credentialId: null, sourceConfig: {} }) - .where(eq(knowledgeConnector.id, ids.connectorId)) - /** Keep the deliberate tenfold visibility underestimate until the measured requests finish. */ - await db.execute(sql`ALTER TABLE document SET (autovacuum_enabled = false)`) - try { - await db.execute(sql`UPDATE document - SET acl = ARRAY[CASE WHEN external_id::int % 10 = 0 - THEN ${`u:${ids.aliceId}@fixture.test`} ELSE ${`u:${ids.bobId}@fixture.test`} END] - WHERE knowledge_base_id = ${ids.knowledgeBaseId}`) - await db.execute(sql`ANALYZE document`) - await db.execute(sql`UPDATE document SET acl = ARRAY[${`u:${ids.aliceId}@fixture.test`}] - WHERE knowledge_base_id = ${ids.knowledgeBaseId}`) - report.organizationVisibility = { - analyzedVisibleDocuments: chunkCount / chunksPerDocument / 10, - actualVisibleDocuments: chunkCount / chunksPerDocument, - unrelatedChunks: unrelatedChunkCount, - } - /** Capture latency samples before EXPLAIN ANALYZE can warm the candidate paths. */ - for (const surface of ['dashboard', 'copilot'] as const) { - const label = `organization.${surface}` - await prepareOrganizationSample(label) - const { result, diagnostics } = await sample( - label, - () => - surface === 'dashboard' - ? searchDashboard('Orion deployment', ids.aliceId, true, 20) - : search(ids.aliceId, 'Orion deployment', {}, true, 20), - { explain: false } - ) - expectCompleteVectorSearch(diagnostics) - expect(result.data.results).toHaveLength(20) - expect( - result.data.results.every((row) => row.knowledgeBaseId === ids.knowledgeBaseId) - ).toBe(true) - } - await prepareOrganizationSample('organization.concurrent') - diagnosticLog?.mockClear() - const started = performance.now() - const results = await Promise.all([ - search(ids.aliceId, 'Orion deployment', {}, true, 20), - search(ids.aliceId, 'Topic 11 deployment', {}, true, 20), - ]) - const diagnostics = diagnosticLog!.mock.calls - .filter(([message]) => message === 'Knowledge search completed') - .map(([, metadata]) => diagnosticSchema.parse(metadata)) - report['organization.concurrent'] = { - milliseconds: performance.now() - started, - resultCounts: results.map((result) => result.data.results.length), - diagnostics, - } - saveReport() - expect(diagnostics).toHaveLength(2) - for (const item of diagnostics) expectCompleteVectorSearch(item) - for (const result of results) { - expect(result.data.results).toHaveLength(20) - expect( - result.data.results.every((row) => row.knowledgeBaseId === ids.knowledgeBaseId) - ).toBe(true) - } - const planned = await sample('organization.plans', () => - search(ids.aliceId, 'Orion deployment', {}, true, 20) - ) - expectCompleteVectorSearch(planned.diagnostics) - expect(planned.plans.filter((plan) => plan.kind === 'vector')).toHaveLength(1) - } finally { - await db.execute(sql`ALTER TABLE document RESET (autovacuum_enabled)`) - await db.execute(sql`ANALYZE document`) - } - }, 180_000) - /** Opt in with local Sim and Go URLs; uses the real configured provider, billing adapter, and async resume protocol. */ - it.skipIf(!process.env.KNOWLEDGE_SEARCH_ASSISTANT_URL)( - 'recovers quietly from incomplete search through local Go Assistant, then reads and cites evidence', - async () => { - const assistantUrl = new URL(process.env.KNOWLEDGE_SEARCH_ASSISTANT_URL!) - const simUrl = new URL(process.env.KNOWLEDGE_SEARCH_SIM_URL!) - for (const url of [assistantUrl, simUrl]) { - if (!['127.0.0.1', 'localhost'].includes(url.hostname)) - throw new Error( - 'Assistant integration requires local servers using the disposable test databases' - ) - } - const apiKey = process.env.KNOWLEDGE_SEARCH_ASSISTANT_API_KEY! - const internalKey = process.env.KNOWLEDGE_SEARCH_SIM_INTERNAL_KEY! - expect(apiKey).toBeTruthy() - expect(internalKey).toBeTruthy() - const chunkId = `${ids.workspaceId}-chunk-0` - const [original] = await db - .select({ - content: embedding.content, - contentLength: embedding.contentLength, - tokenCount: embedding.tokenCount, - }) - .from(embedding) - .where(eq(embedding.id, chunkId)) - .limit(1) - const longContent = - 'Orion deployment guide. The activation phrase and final checksum appear at the end.\n' + - 'Review the deployment stages in order. Preserve the rollback procedure.\n'.repeat(320) + - '\nActivation phrase: SILVER COMET\nFinal checksum: K7M2-84\n' - const registry = new ResolvedSecretTraceRegistry([], { userId: ids.aliceId }) - const calls: Array<{ - name: string - arguments: unknown - milliseconds: number - bytes: number - }> = [] - let answer = '' - let incompleteSearch = true - const query = SearchBudget.prototype.query - const delayed = vi.spyOn(SearchBudget.prototype, 'query').mockImplementation(function ( - this: SearchBudget, - stage: SearchStage, - run: (executor: SearchExecutor) => PromiseLike - ): Promise { - return query.call(this, stage, async (tx) => { - if (incompleteSearch) await tx.execute(sql`SELECT pg_sleep(9)`) - return run(tx) - }) as Promise - }) - const started = performance.now() - try { - await db - .update(embedding) - .set({ - content: longContent, - contentLength: longContent.length, - tokenCount: Math.ceil(longContent.length / 4), - }) - .where(eq(embedding.id, chunkId)) - const admission = await externalFetch(new URL('/api/copilot/api-keys/validate', simUrl), { - method: 'POST', - headers: { - 'content-type': 'application/json', - 'x-api-key': internalKey, - 'x-sim-billing-protocol': 'legacy-v0', - }, - body: JSON.stringify({ - userId: ids.aliceId, - organizationId: ids.organizationId, - chatId: organizationChatId, - }), - }) - expect(admission.status, await admission.text()).toBe(200) - let path = '/api/mothership' - let body: Record = { - message: - 'Search for the Orion deployment guide that mentions an activation phrase and final checksum. Read enough of that document to report both values and cite it.', - version: '3.0.0', - mode: 'assistant', - userId: ids.aliceId, - organizationId: ids.organizationId, - chatId: organizationChatId, - } - for (let round = 0; round < 8; round++) { - const response = await externalFetch(new URL(path, assistantUrl), { - method: 'POST', - headers: { - 'content-type': 'application/json', - 'x-api-key': apiKey, - 'x-sim-billing-protocol': 'legacy-v0', - }, - body: JSON.stringify(body), - signal: AbortSignal.timeout(120000), - }) - const wire = await response.text() - expect(response.status, wire.slice(0, 2000)).toBe(200) - expect(Buffer.byteLength(wire)).toBeLessThan(2 * 1024 * 1024) - const pending = new Map() - let checkpoint: MothershipStreamV1CheckpointPausePayload | undefined - let streamId = '' - for (const line of wire.split('\n')) { - if (!line.startsWith('data:')) continue - const raw = line.slice(5).trim() - if (!raw || raw === '[DONE]') continue - const event: unknown = JSON.parse(raw) - if (!isContractStreamEventEnvelope(event)) - throw new Error('Assistant returned an invalid generated stream envelope') - streamId = event.stream.streamId - if (event.type === 'error') throw new Error(JSON.stringify(event.payload)) - if (event.type === 'text') answer += event.payload.text - if ( - event.type === 'tool' && - event.payload.phase === 'call' && - !event.payload.partial && - event.payload.arguments - ) - pending.set(event.payload.toolCallId, event.payload) - if (event.type === 'run' && event.payload.kind === 'checkpoint_pause') - checkpoint = event.payload - } - if (!checkpoint) break - const results = await Promise.all( - checkpoint.pendingToolCallIds.map(async (callId) => { - const call = pending.get(callId) - if (!call) throw new Error('Checkpoint referenced an absent tool call') - const tool = - call.toolName === 'search_workspace' - ? searchWorkspaceServerTool - : call.toolName === 'read_document' - ? readDocumentServerTool - : undefined - if (!tool) throw new Error(`Unexpected Assistant tool: ${call.toolName}`) - const toolStarted = performance.now() - const result = await tool.execute(call.arguments, { - userId: ids.aliceId, - organizationId: ids.organizationId, - chatId: organizationChatId, - toolCallId: callId, - copilotToolExecution: true, - requestMode: 'assistant', - resolvedSecretTraceRegistry: registry, - }) - calls.push({ - name: call.toolName, - arguments: call.arguments, - milliseconds: performance.now() - toolStarted, - bytes: Buffer.byteLength(JSON.stringify(result)), - }) - const { success } = z.object({ success: z.boolean() }).parse(result) - expect(success).toBe(true) - if (incompleteSearch) { - expect(call.toolName).toBe('search_workspace') - expect(result).toMatchObject({ - data: { - retrieval: { status: 'partial', timedOutLegs: ['vector', 'keyword'] }, - results: [], - }, - }) - } - return { callId, name: call.toolName, success, data: result } - }) - ) - incompleteSearch = false - path = '/api/tools/resume' - body = { - checkpointId: checkpoint.checkpointId, - streamId, - userId: ids.aliceId, - organizationId: ids.organizationId, - chatId: organizationChatId, - results, - } - report['assistant.live.progress'] = { rounds: round + 1, calls } - saveReport() - } - report['assistant.live'] = { milliseconds: performance.now() - started, calls, answer } - saveReport() - expect(answer).toContain('SILVER COMET') - expect(answer).toContain('K7M2-84') - expect(answer).toContain('') - expect(answer).not.toMatch(/timed?\s*out|timeout|internal retr(?:y|ies)/i) - expect(calls.some((call) => call.name === 'read_document')).toBe(true) - expect(calls.filter((call) => call.name === 'search_workspace').length).toBeGreaterThan(1) - expect(calls.every((call) => call.bytes < 40000)).toBe(true) - } finally { - delayed.mockRestore() - await db.update(embedding).set(original).where(eq(embedding.id, chunkId)) - } - }, - 10 * 60_000 - ) }) diff --git a/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts b/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts index f1a99f53971..41edfd940a4 100644 --- a/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts @@ -1,6 +1,8 @@ import { db, runOutsideTransactionContext } from '@sim/db' import { + credential, credentialGroup, + credentialGroupEnrollment, mcpServers, member, organization, @@ -9,6 +11,7 @@ import { user, } from '@sim/db/schema' import * as dns from '@sim/security/dns' +import { sha256Hex } from '@sim/security/hash' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' import { createDeferred } from '@sim/testing/helpers/deferred' import { getPostgresErrorCode } from '@sim/utils/errors' @@ -18,6 +21,14 @@ import { eq, inArray, sql } from 'drizzle-orm' import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' import { listSearchIntegrationsContract } from '@/lib/api/contracts/knowledge/search-integrations' import { env } from '@/lib/core/config/env' +import { encryptSecret } from '@/lib/core/security/encryption' +import { credentialGroupScopePolicyVersion } from '@/lib/credential-groups/provider-adapter' +import { + emptyCredentialGroupProviderConfiguration, + encryptCredentialGroupProviderConfiguration, +} from '@/lib/credential-groups/provider-configuration' +import { getCredentialGroup } from '@/lib/credential-groups/service' +import { SLACK_MANAGED_USER_SCOPES } from '@/lib/credential-groups/slack-managed-user-scopes' import { createOrganizationAccountsGroup } from '@/lib/credential-groups/workspace-accounts' import { acquireAdvisoryXactLock, tryAcquireAdvisoryXactLock } from '@/lib/db/advisory-locks' import { @@ -25,6 +36,7 @@ import { listSearchIntegrations, } from '@/lib/knowledge/application/search-integrations' import { defaultLiveSearchPolicy } from '@/lib/sim-search/live/policy-schema' +import { SLACK_RTS_USER_SCOPES } from '@/lib/sim-search/live/scopes' /** * Real authorization, transactions, constraints and persistence; only DNS is a fixture. @@ -98,7 +110,7 @@ describe('atomic organization live Search MCP setup', () => { }) async function snapshot() { - const [groups, servers, approvals, policies, organizations] = await Promise.all([ + const [groups, servers, approvals, policies, organizations, credentials] = await Promise.all([ db.select().from(credentialGroup).where(eq(credentialGroup.organizationId, ids.organization)), db.select().from(mcpServers).where(eq(mcpServers.organizationId, ids.organization)), db @@ -110,10 +122,131 @@ describe('atomic organization live Search MCP setup', () => { .select({ metadata: organization.metadata }) .from(organization) .where(eq(organization.id, ids.organization)), + db.select().from(credential).where(eq(credential.organizationId, ids.organization)), ]) - return { groups, servers, approvals, policies, metadata: toRecord(organizations[0]?.metadata) } + return { + groups, + servers, + approvals, + policies, + credentials, + metadata: toRecord(organizations[0]?.metadata), + } } + async function seedWorkflowSlack(scopes: readonly string[] = SLACK_MANAGED_USER_SCOPES) { + const option = { + id: generateId(), + provider: 'slack', + label: 'Slack', + authorizationAppId: 'slack:fixture-app:fixture-team', + requiredScopes: [...scopes], + scopeVersion: credentialGroupScopePolicyVersion([...scopes]), + required: false, + status: 'active' as const, + } + const other = { ...option, id: generateId(), provider: 'gmail', label: 'Gmail' } + const group = await db.transaction((tx) => + createOrganizationAccountsGroup(tx, ids.organization, ids.owner, [option, other]) + ) + await db + .update(credentialGroup) + .set({ + encryptedProviderConfiguration: await encryptCredentialGroupProviderConfiguration({ + ...emptyCredentialGroupProviderConfiguration(), + slack: { + clientId: 'fixture-client', + clientSecret: 'fixture-secret', + appId: 'fixture-app', + teamId: 'fixture-team', + scopes: [...scopes], + verifiedAt: new Date().toISOString(), + }, + }), + }) + .where(eq(credentialGroup.id, group.id)) + const enrollmentId = generateId() + await db.insert(credentialGroupEnrollment).values({ + id: enrollmentId, + credentialGroupId: group.id, + userId: ids.owner, + email: `${ids.owner}@fixture.test`, + status: 'completed', + invitationTokenHash: sha256Hex(generateId()), + invitationExpiresAt: new Date(Date.now() + 60_000), + invitedAt: new Date(), + }) + const encrypted = (await encryptSecret('{"access_token":"fixture-token"}')).encrypted + await db.insert(credential).values( + [option, other].map((entry) => ({ + id: generateId(), + organizationId: ids.organization, + type: 'managed_oauth' as const, + providerId: entry.provider, + displayName: entry.label, + createdBy: ids.owner, + authorizationAppId: entry.authorizationAppId, + providerSubjectId: ids.owner, + credentialGroupEnrollmentId: enrollmentId, + credentialGroupOptionId: entry.id, + managedOauthStatus: 'active' as const, + managedOauthScopeVersion: entry.scopeVersion, + grantedScopes: [...scopes], + encryptedOauthTokenSet: encrypted, + grantedAt: new Date(), + })) + ) + return { groupId: group.id, optionId: option.id, otherOptionId: other.id } + } + + it.each([ + { name: 'workflow policy', scopes: SLACK_MANAGED_USER_SCOPES }, + { name: 'custom policy', scopes: ['chat:write', 'users:read', 'users:read.email'] }, + ])( + 'upgrades an existing Slack $name only through explicit Search approval', + async ({ scopes }) => { + const seeded = await seedWorkflowSlack(scopes) + const before = await snapshot() + await approve('slack') + const state = await snapshot() + const upgraded = state.groups[0].options.find((option) => option.id === seeded.optionId)! + expect(upgraded.requiredScopes).toEqual( + expect.arrayContaining([...scopes, ...SLACK_RTS_USER_SCOPES]) + ) + expect(upgraded.scopeVersion).not.toBe(before.groups[0].options[0].scopeVersion) + expect(state.groups[0].options.find((option) => option.id === seeded.otherOptionId)).toEqual( + before.groups[0].options[1] + ) + expect(state.policies).toEqual(before.policies) + expect(state.groups[0].encryptedProviderConfiguration).toBe( + before.groups[0].encryptedProviderConfiguration + ) + expect( + state.credentials.find((entry) => entry.credentialGroupOptionId === seeded.optionId) + ?.managedOauthStatus + ).toBe('needs_reauth') + expect( + state.credentials.find((entry) => entry.credentialGroupOptionId === seeded.otherOptionId) + ).toEqual( + before.credentials.find((entry) => entry.credentialGroupOptionId === seeded.otherOptionId) + ) + expect( + await getCredentialGroup( + { kind: 'organization', organizationId: ids.organization }, + seeded.groupId + ) + ).toMatchObject({ + options: expect.arrayContaining([ + expect.objectContaining({ id: seeded.optionId, configurationStatus: 'needs_update' }), + ]), + }) + await expect(approve('slack')).resolves.toMatchObject({ memberAccounts: { changed: false } }) + const repeated = await snapshot() + expect(repeated.groups).toEqual(state.groups) + expect(repeated.credentials).toEqual(state.credentials) + } + ) + it('keeps disabled Zoom approvals visible and removable without permitting reapproval', async () => { const connectorType = 'zoom' await db.insert(organizationSearchIntegration).values({ @@ -363,25 +496,29 @@ describe('atomic organization live Search MCP setup', () => { } ) - it('rolls back sign-in resources and policy metadata when the final approval write fails', async () => { - const constraint = `search_setup_${generateId().replace(/-/g, '')}` - await db.execute( - sql`ALTER TABLE organization_search_integration ADD CONSTRAINT ${sql.identifier(constraint)} CHECK (organization_id <> ${sql.raw(`'${ids.organization}'`)}) NOT VALID` - ) - const before = await snapshot() - try { - let failure: unknown - try { - await approve('fireflies') - } catch (error) { - failure = error - } - expect(getPostgresErrorCode(failure)).toBe('23514') - expect(await snapshot()).toEqual(before) - } finally { + it.each(['fireflies', 'slack'])( + 'rolls back %s sign-in policy and credentials when the final approval write fails', + async (provider) => { + if (provider === 'slack') await seedWorkflowSlack() + const constraint = `search_setup_${generateId().replace(/-/g, '')}` await db.execute( - sql`ALTER TABLE organization_search_integration DROP CONSTRAINT ${sql.identifier(constraint)}` + sql`ALTER TABLE organization_search_integration ADD CONSTRAINT ${sql.identifier(constraint)} CHECK (organization_id <> ${sql.raw(`'${ids.organization}'`)}) NOT VALID` ) + const before = await snapshot() + try { + let failure: unknown + try { + await approve(provider) + } catch (error) { + failure = error + } + expect(getPostgresErrorCode(failure)).toBe('23514') + expect(await snapshot()).toEqual(before) + } finally { + await db.execute( + sql`ALTER TABLE organization_search_integration DROP CONSTRAINT ${sql.identifier(constraint)}` + ) + } } - }) + ) }) diff --git a/apps/sim/lib/knowledge/__integration__/search-source-pagination.integration.ts b/apps/sim/lib/knowledge/__integration__/search-source-pagination.integration.ts index 4bc7a62c3f6..42e7b124f87 100644 --- a/apps/sim/lib/knowledge/__integration__/search-source-pagination.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-source-pagination.integration.ts @@ -1,7 +1,5 @@ import { db } from '@sim/db' import { - document, - embedding, knowledgeBase, knowledgeConnector, member, @@ -17,19 +15,14 @@ import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import { readSearchSourceOverview } from '@/lib/knowledge/application/search-source-overview' -import { readSearchSourceProgress } from '@/lib/knowledge/application/search-source-progress' import { listSearchSources } from '@/lib/knowledge/application/search-sources' const ids = createKnowledgeAclFixtureIds() const alice = { kind: 'session' as const, userId: ids.aliceId, sessionId: 'fixture-alice' } -const bob = { kind: 'session' as const, userId: ids.bobId, sessionId: 'fixture-bob' } const sourceIds = Array.from({ length: 105 }, () => generateId()) .sort() .reverse() const olderSourceId = generateId() -const documentId = generateId() -const embeddingId = generateId() const input = { workspaceId: ids.workspaceId } beforeAll(async () => { @@ -65,33 +58,6 @@ beforeAll(async () => { status: 'active', createdAt: new Date('2025-12-31T00:00:00Z'), }) - await db.insert(document).values({ - id: documentId, - connectorId: olderSourceId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: 'fixture', - filename: 'readable.txt', - fileUrl: 'https://fixture.test/readable', - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed', - acl: [`u:${ids.aliceId}@fixture.test`], - aclVerifiedAt: new Date(), - }) - await db.insert(embedding).values({ - id: embeddingId, - documentId, - knowledgeBaseId: ids.knowledgeBaseId, - chunkIndex: 0, - chunkHash: 'fixture-hash', - content: 'Fixture text', - contentLength: 12, - tokenCount: 3, - startOffset: 0, - endOffset: 12, - embeddingModel: 'text-embedding-3-small', - embedding: [1, ...Array(1535).fill(0)], - }) }) afterAll(async () => { @@ -101,7 +67,7 @@ afterAll(async () => { await db.delete(user).where(eq(user.id, ids.bobId)) }) -describe('bounded source pagination and provider overview', () => { +describe('bounded live Search source configuration pagination', () => { it('finds a provider beyond the unfiltered candidate bound on its first filtered page', async () => { const result = await listSearchSources.execute({ principal: alice, @@ -141,31 +107,6 @@ describe('bounded source pagination and provider overview', () => { expect(second.sources.map((source) => source.connectorId)).toEqual([olderSourceId]) expect(second.nextCursor).toBeNull() }) - it('includes providers beyond the loaded page and requires viewer-readable indexed content', async () => { - const first = await listSearchSources.execute({ principal: alice, input }) - expect(first.sources.every((source) => source.connectorType === 'confluence')).toBe(true) - const [aliceOverview, bobOverview] = await Promise.all( - [alice, bob].map((principal) => readSearchSourceOverview.execute({ principal, input })) - ) - expect(aliceOverview.providers).toEqual( - expect.arrayContaining([{ connectorType: 'google_drive', isSyncing: false }]) - ) - expect(aliceOverview.hasSearchableDocuments).toBe(true) - expect(bobOverview.hasSearchableDocuments).toBe(false) - }) - it('keeps paused but readable sources complete and ignores disabled chunks', async () => { - await db - .update(knowledgeConnector) - .set({ status: 'paused' }) - .where(eq(knowledgeConnector.id, olderSourceId)) - const paused = await readSearchSourceOverview.execute({ principal: alice, input }) - expect(paused.hasSearchableDocuments).toBe(true) - expect(paused.providers.every((provider) => !provider.isSyncing)).toBe(true) - await db.update(embedding).set({ enabled: false }).where(eq(embedding.id, embeddingId)) - expect( - (await readSearchSourceOverview.execute({ principal: alice, input })).hasSearchableDocuments - ).toBe(false) - }) it('shows a newly created source on the first-page refresh so enrollment can observe completion', async () => { const before = await listSearchSources.execute({ principal: alice, input }) expect(before.sources[0].connectorId).toBe(sourceIds[0]) @@ -179,7 +120,7 @@ describe('bounded source pagination and provider overview', () => { status: 'pending', }) const refreshed = await listSearchSources.execute({ principal: alice, input }) - expect(refreshed.sources[0]).toMatchObject({ connectorId: newSourceId, isSyncing: true }) + expect(refreshed.sources[0]).toMatchObject({ connectorId: newSourceId }) expect(refreshed.sources.map((source) => source.connectorId)).toEqual([ newSourceId, ...sourceIds.slice(0, 24), @@ -190,7 +131,7 @@ describe('bounded source pagination and provider overview', () => { }) expect(next.sources.map((source) => source.connectorId)).toEqual(sourceIds.slice(24, 49)) }) - it('does not report deactivated pending sources as indexing in any read model', async () => { + it('preserves organization deactivation when listing source configuration', async () => { await db.insert(member).values({ id: generateId(), organizationId: ids.organizationId, @@ -217,22 +158,10 @@ describe('bounded source pagination and provider overview', () => { approved: false, }) const owner = { organizationId: ids.organizationId } - const [progress, overview, summary] = await Promise.all([ - readSearchSourceProgress.execute({ - principal: alice, - input: { ...owner, connectorIds: [approvalSourceId] }, - }), - readSearchSourceOverview.execute({ principal: alice, input: owner }), - listSearchSources.execute({ principal: alice, input: owner }), - ]) - expect(progress.sources[0].isSyncing).toBe(false) - expect( - overview.providers.find((provider) => provider.connectorType === 'google_drive')?.isSyncing - ).toBe(false) + const summary = await listSearchSources.execute({ principal: alice, input: owner }) expect(summary.sources[0]).toMatchObject({ connectorId: approvalSourceId, approved: false, - isSyncing: false, }) }) }) diff --git a/apps/sim/lib/knowledge/__integration__/search-source-progress.integration.ts b/apps/sim/lib/knowledge/__integration__/search-source-progress.integration.ts index e47cb75cfc6..e4580b084e3 100644 --- a/apps/sim/lib/knowledge/__integration__/search-source-progress.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-source-progress.integration.ts @@ -22,14 +22,14 @@ import { import { deleteKnowledgeConnector, listKnowledgeConnectorDocuments, + updateKnowledgeConnectorDocuments, } from '@/lib/knowledge/application/connectors' import { + bulkUpdateKnowledgeDocuments, listKnowledgeDocuments, readKnowledgeDocument, updateKnowledgeDocument, } from '@/lib/knowledge/application/documents' -import { readSearchSourceProgress } from '@/lib/knowledge/application/search-source-progress' -import { listSearchSources } from '@/lib/knowledge/application/search-sources' import { KNOWLEDGE_CONNECTOR_DETACH_EVENT } from '@/lib/knowledge/connectors/detachment' import { createContentSyncLease } from '@/lib/knowledge/connectors/sync-lock' import { persistSkippedDocuments } from '@/lib/knowledge/connectors/sync-persistence' @@ -41,15 +41,10 @@ const alice = { kind: 'session' as const, userId: ids.aliceId, sessionId: 'fixtu const bob = { kind: 'session' as const, userId: ids.bobId, sessionId: 'fixture-bob' } const failedId = generateId() const pendingId = generateId() -const input = { workspaceId: ids.workspaceId, connectorIds: [ids.connectorId] } -/** Drive models the mirrored email grants exercised by these provider-independent progress tests. */ +/** Drive models the mirrored email grants exercised by these document recovery tests. */ beforeAll(async () => { await seedKnowledgeAclFixture(ids, { connectorType: 'google_drive' }) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) await db .update(knowledgeConnector) .set({ status: 'active', syncLockToken: null }) @@ -80,32 +75,7 @@ afterAll(async () => { await db.delete(user).where(eq(user.id, ids.bobId)) }) -describe('viewer-isolated indexing progress and recovery lists', () => { - it('keeps failed and pending state scoped to the viewer, including admins', async () => { - expect((await readSearchSourceProgress.execute({ principal: alice, input })).sources).toEqual([ - { - connectorId: ids.connectorId, - isSyncing: false, - hasSyncError: false, - hasIndexingError: true, - }, - ]) - expect((await readSearchSourceProgress.execute({ principal: bob, input })).sources).toEqual([ - { - connectorId: ids.connectorId, - isSyncing: true, - hasSyncError: false, - hasIndexingError: false, - }, - ]) - const [aliceSources, bobSources] = await Promise.all( - [alice, bob].map((principal) => - listSearchSources.execute({ principal, input: { workspaceId: ids.workspaceId } }) - ) - ) - expect(aliceSources.sources[0].viewerFailedDocumentCount).toBe(1) - expect(bobSources.sources[0].viewerFailedDocumentCount).toBe(0) - }) +describe('viewer-isolated knowledge-base recovery lists', () => { it('only lists accessible failures, with authoritative filtered pagination', async () => { const read = (principal: typeof alice) => listKnowledgeConnectorDocuments.execute({ @@ -123,28 +93,99 @@ describe('viewer-isolated indexing progress and recovery lists', () => { expect(result.hasMore).toBe(false) expect((await read(bob)).documents).toEqual([]) }) - it('does not report excluded or deleted failures as actionable', async () => { - await db.update(document).set({ userExcluded: true }).where(eq(document.id, failedId)) - expect( - (await readSearchSourceProgress.execute({ principal: alice, input })).sources[0] - .hasIndexingError - ).toBe(false) +}) + +describe('retired Search document admission', () => { + const fixture = createKnowledgeAclFixtureIds() + const viewer = { + kind: 'session' as const, + userId: fixture.aliceId, + sessionId: 'fixture-retirement', + } + const documentId = generateId() + + beforeAll(async () => { + await seedKnowledgeAclFixture(fixture, { connectorType: 'google_drive' }) + await db.insert(document).values({ + id: documentId, + knowledgeBaseId: fixture.knowledgeBaseId, + connectorId: fixture.connectorId, + externalId: documentId, + filename: 'Retirement fixture', + fileUrl: '', + fileSize: 0, + mimeType: 'text/plain', + processingStatus: 'completed', + acl: [`u:${fixture.aliceId}@fixture.test`], + aclVerifiedAt: new Date(), + }) + }) + afterAll(async () => { + await db.delete(workspace).where(eq(workspace.id, fixture.workspaceId)) + await db.delete(organization).where(eq(organization.id, fixture.organizationId)) + await db.delete(user).where(inArray(user.id, [fixture.aliceId, fixture.bobId])) + }) + + it.each([ + 'connector restore', + 'document enable', + 'document updates enable', + 'selected documents enable', + 'all documents enable', + ] as const)('refuses %s for retired Search while preserving ordinary KBs', async (operation) => { + const change = () => { + const input = { knowledgeBaseId: fixture.knowledgeBaseId, documentId } + if (operation === 'connector restore') + return updateKnowledgeConnectorDocuments.execute({ + principal: viewer, + input: { + ...input, + connectorId: fixture.connectorId, + operation: 'restore', + documentIds: [documentId], + }, + }) + if (operation === 'document enable') + return updateKnowledgeDocument.execute({ + principal: viewer, + input: { ...input, enabled: true }, + }) + if (operation === 'document updates enable') + return updateKnowledgeDocument.execute({ + principal: viewer, + input: { ...input, updates: { enabled: true } }, + }) + return bulkUpdateKnowledgeDocuments.execute({ + principal: viewer, + input: { + knowledgeBaseId: fixture.knowledgeBaseId, + operation: 'enable', + ...(operation === 'all documents enable' + ? { selectAll: true } + : { documentIds: [documentId] }), + }, + }) + } await db .update(document) - .set({ userExcluded: false, deletedAt: new Date() }) - .where(eq(document.id, failedId)) - expect( - (await readSearchSourceProgress.execute({ principal: alice, input })).sources[0] - .hasIndexingError - ).toBe(false) - }) - it('rechecks membership before showing progress', async () => { + .set({ enabled: false, userExcluded: operation === 'connector restore' }) + .where(eq(document.id, documentId)) await db - .delete(permissions) - .where(and(eq(permissions.entityId, ids.workspaceId), eq(permissions.userId, ids.bobId))) - await expect(readSearchSourceProgress.execute({ principal: bob, input })).rejects.toThrow( - 'Insufficient workspace permissions' - ) + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where(eq(knowledgeBase.id, fixture.knowledgeBaseId)) + const [before] = await db.select().from(document).where(eq(document.id, documentId)) + + await expect(change()).rejects.toMatchObject({ code: 'validation' }) + expect(await db.select().from(document).where(eq(document.id, documentId))).toEqual([before]) + + await db + .update(knowledgeBase) + .set({ isSearchIndex: false }) + .where(eq(knowledgeBase.id, fixture.knowledgeBaseId)) + await change() + const [restored] = await db.select().from(document).where(eq(document.id, documentId)) + expect(restored).toMatchObject({ enabled: true, userExcluded: false, acl: before.acl }) }) }) @@ -322,10 +363,6 @@ describe('intentional skips and genuine failures across document reads', () => { beforeAll(async () => { await seedKnowledgeAclFixture(fixture, { connectorType: 'google_drive' }) - await db - .update(knowledgeBase) - .set({ isSearchIndex: true }) - .where(eq(knowledgeBase.id, fixture.knowledgeBaseId)) const rows: Array & { id: string; filename: string }> = [ { id: legacySkipId, @@ -453,29 +490,7 @@ describe('intentional skips and genuine failures across document reads', () => { expect(legacyFailures.documents.map((row) => row.id)).toEqual(failureIds) }) - it('does not turn another viewer’s skips into indexing errors or expose their documents', async () => { - for (const [principal, failedCount] of [ - [viewer, 2], - [otherViewer, 0], - ] as const) { - const sources = await listSearchSources.execute({ - principal, - input: { workspaceId: fixture.workspaceId }, - }) - expect(sources.sources[0].viewerFailedDocumentCount).toBe(failedCount) - const progress = await readSearchSourceProgress.execute({ - principal, - input: { workspaceId: fixture.workspaceId, connectorIds: [fixture.connectorId] }, - }) - expect(progress.sources).toEqual([ - { - connectorId: fixture.connectorId, - isSyncing: false, - hasSyncError: false, - hasIndexingError: failedCount > 0, - }, - ]) - } + it('does not expose another viewer’s skipped documents or failures', async () => { for (const filter of ['active', 'skipped', 'failed'] as const) { const result = await listKnowledgeConnectorDocuments.execute({ principal: otherViewer, @@ -614,10 +629,6 @@ describe('intentional skips and genuine failures across document reads', () => { input: { ...scope, deleteDocuments: false }, }) ).rejects.toThrow('cannot be kept') - await db - .update(knowledgeBase) - .set({ isSearchIndex: false }) - .where(eq(knowledgeBase.id, fixture.knowledgeBaseId)) await db .update(knowledgeConnector) .set({ accessMode: 'workspace' }) diff --git a/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts b/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts index a339cd91cbe..eee347e4058 100644 --- a/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-source-setup.integration.ts @@ -14,13 +14,6 @@ import { generateId } from '@sim/utils/id' import { and, eq, isNull } from 'drizzle-orm' import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' -/** A Search source crawls into a search index, which only indexed organization search reads. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) - const fixture = vi.hoisted(() => ({ dispatch: vi.fn() })) vi.mock('@/lib/credential-groups/provider-registry', () => ({ getCredentialGroupProviderAdapter: (provider: string) => ({ @@ -36,13 +29,6 @@ vi.mock('@/lib/credential-groups/provider-registry', () => ({ }), })) vi.mock('@/lib/knowledge/connectors/member-queue', () => ({ dispatchMemberSync: fixture.dispatch })) -vi.mock('@/lib/knowledge/application/connector-access', () => ({ - startKnowledgeConnectorMemberEnrollment: { - execute: async ({ input }: { input: { connectorId: string } }) => ({ - url: `https://fixture.test/enroll/${input.connectorId}`, - }), - }, -})) vi.mock('@/connectors/registry.server', () => ({ CONNECTOR_REGISTRY: { google_drive: { @@ -67,11 +53,7 @@ import { seedKnowledgeMemberFixture, } from '@/lib/knowledge/__integration__/seed-source-access-fixture' import { createKnowledgeConnector } from '@/lib/knowledge/application/connectors' -import { - connectSimSearchConnector, - prepareSearchSource, - readSearchIndex, -} from '@/lib/knowledge/application/sim-search' +import { prepareSearchSource, readSearchIndex } from '@/lib/knowledge/application/sim-search' import { performCreateKnowledgeConnector } from '@/lib/knowledge/orchestration/connectors' import { createWorkspaceInTransaction } from '@/lib/workspaces/create' @@ -124,6 +106,7 @@ describe('Search source identity and concurrent creation', () => { id: ids.knowledgeBaseId, name: 'Renamed company index', workspaceId: ids.workspaceId, + isSearchIndex: true, }, connectorType: 'google_drive', sourceConfig: { folderId }, @@ -253,7 +236,7 @@ describe('Search source identity and concurrent creation', () => { ).toHaveLength(0) }) - it('adopts a legacy index only during admin setup and persists its canonical identity', async () => { + it('preserves ordinary KB name collisions and creates one explicit Search configuration', async () => { await db .update(knowledgeBase) .set({ name: 'Sim Search' }) @@ -275,18 +258,40 @@ describe('Search source identity and concurrent creation', () => { principal: { kind: 'session', userId: other.aliceId, sessionId: 'fixture-admin' }, input, }) - const results = await Promise.all([prepare(), prepare()]) - expect(results.map((result) => result.knowledgeBaseId)).toEqual([ - other.knowledgeBaseId, - other.knowledgeBaseId, - ]) + await expect(prepare()).rejects.toMatchObject({ code: 'conflict' }) + expect( + await db + .select({ name: knowledgeBase.name, isSearchIndex: knowledgeBase.isSearchIndex }) + .from(knowledgeBase) + .where(eq(knowledgeBase.id, other.knowledgeBaseId)) + ).toEqual([{ name: 'Sim Search', isSearchIndex: false }]) await db .update(knowledgeBase) - .set({ name: 'Renamed adopted index' }) + .set({ name: 'Ordinary knowledge' }) .where(eq(knowledgeBase.id, other.knowledgeBaseId)) + const results = await Promise.all([prepare(), prepare()]) + const searchId = results[0].knowledgeBaseId + expect(searchId).not.toBe(other.knowledgeBaseId) + expect(results[1].knowledgeBaseId).toBe(searchId) + expect( + await db + .select({ id: knowledgeBase.id }) + .from(knowledgeBase) + .where( + and( + eq(knowledgeBase.workspaceId, other.workspaceId), + eq(knowledgeBase.isSearchIndex, true) + ) + ) + ).toEqual([{ id: searchId }]) + await db + .update(knowledgeBase) + .set({ name: 'Renamed Search configuration' }) + .where(eq(knowledgeBase.id, searchId)) await expect(readSearchIndex.execute({ principal: reader, input })).resolves.toMatchObject({ - knowledgeBaseId: other.knowledgeBaseId, + knowledgeBaseId: searchId, }) + await expect(prepare()).resolves.toMatchObject({ knowledgeBaseId: searchId }) }) it('serializes matching creates across independent database transactions and keeps one grant', async () => { @@ -295,7 +300,7 @@ describe('Search source identity and concurrent creation', () => { const successful = results.filter((result) => result.success) expect(new Set(successful.map((result) => result.connector.id)).size).toBe(1) expect(successful.filter((result) => result.reused)).toHaveLength(1) - expect(fixture.dispatch).toHaveBeenCalledTimes(1) + expect(fixture.dispatch).not.toHaveBeenCalled() sourceIds.push(successful[0]!.connector.id) const [policy] = await db .select() @@ -316,57 +321,12 @@ describe('Search source identity and concurrent creation', () => { expect(rows.filter((row) => row.id !== member.connectorId)).toHaveLength(1) }) - it('keeps distinct source settings separate and allows readers to select the exact source', async () => { + it('keeps distinct source settings separate', async () => { const second = await createSource('other-folder') expect(second.success).toBe(true) if (!second.success) throw new Error(second.error) sourceIds.push(second.connector.id) expect(second.connector.id).not.toBe(sourceIds[0]) - const result = await connectSimSearchConnector.execute({ - principal: { kind: 'session', userId: ids.bobId, sessionId: 'fixture-reader' }, - input: { - workspaceId: ids.workspaceId, - connectorType: 'google_drive', - connectorId: second.connector.id, - }, - }) - expect(result.connectorId).toBe(second.connector.id) - expect(result.knowledgeBaseId).toBe(ids.knowledgeBaseId) - }) - - it('rejects stale settings, different providers, foreign sources, and noncanonical knowledge bases', async () => { - const noncanonical = generateId() - const extraSource = generateId() - await db.insert(knowledgeBase).values({ - id: noncanonical, - name: 'Ordinary base', - userId: ids.aliceId, - workspaceId: ids.workspaceId, - }) - await db.insert(knowledgeConnector).values({ - id: extraSource, - knowledgeBaseId: noncanonical, - connectorType: 'google_drive', - sourceConfig: {}, - accessMode: 'members', - }) - for (const input of [ - { - connectorType: 'google_drive', - connectorId: sourceIds[0], - sourceConfig: { folderId: 'changed-folder' }, - }, - { connectorType: 'confluence', connectorId: sourceIds[0] }, - { connectorType: 'google_drive', connectorId: other.connectorId }, - { connectorType: 'google_drive', connectorId: extraSource }, - ]) { - await expect( - connectSimSearchConnector.execute({ - principal: { kind: 'session', userId: ids.bobId, sessionId: 'fixture-reader' }, - input: { workspaceId: ids.workspaceId, ...input }, - }) - ).rejects.toMatchObject({ code: 'not_found' }) - } }) it('provisions one workspace container with optional provider options under concurrent setup', async () => { const [previousPolicy] = await db @@ -492,32 +452,6 @@ describe('Search source identity and concurrent creation', () => { }) }) - it('uses the same accounts option through actual first-source application setup', async () => { - const results = await Promise.all( - [1, 2].map(() => - connectSimSearchConnector.execute({ - principal: { kind: 'session', userId: ids.aliceId, sessionId: 'fixture-admin' }, - input: { - workspaceId: ids.workspaceId, - connectorType: 'google_drive', - sourceConfig: { folderId: 'search-account-folder' }, - }, - }) - ) - ) - expect(results[0]!.connectorId).toBe(results[1]!.connectorId) - const [source] = await db - .select() - .from(knowledgeConnector) - .where(eq(knowledgeConnector.id, results[0]!.connectorId)) - expect(source!.credentialGroupId).toBe(searchGroupId) - const groups = await db - .select() - .from(credentialGroup) - .where(eq(credentialGroup.workspaceId, ids.workspaceId)) - expect(groups).toHaveLength(1) - }) - it('refuses automatic provider expansion when a concurrent workflow grant commits first', async () => { let releaseGrant!: () => void let acquiredLock!: () => void diff --git a/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts b/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts index 4a50b7a74d0..50b29de0fc2 100644 --- a/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/stored-document-recovery.integration.ts @@ -9,7 +9,6 @@ import { embedding, knowledgeBase, knowledgeConnector, - member, organization, outboxEvent, user, @@ -26,12 +25,6 @@ const fixture = vi.hoisted(() => ({ listRuns: vi.fn(), batchTrigger: vi.fn(), })) -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) vi.mock('@/lib/core/config/trigger-runtime', () => ({ isInsideTriggerRun: () => fixture.useTrigger, })) @@ -105,7 +98,6 @@ import { retryDocumentProcessing, } from '@/lib/knowledge/documents/service' import { MAX_PROCESSING_ATTEMPTS, QUEUED_DISPATCH_GRACE_MS } from '@/lib/knowledge/documents/types' -import { searchScopedKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' import type { SyncResult } from '@/connectors/types' interface QueryPlan { @@ -132,10 +124,7 @@ async function eventsFor(ids: ReturnType) { ) ) } -async function failedFile( - ids: ReturnType, - organizationOwned = false -) { +async function failedFile(ids: ReturnType) { const file = await addDocument( ids.knowledgeBaseId, ids.connectorId, @@ -147,9 +136,7 @@ async function failedFile( mimeType: 'text/plain', contentHash: 'fixture-retained-v1', }, - organizationOwned - ? { userId: ids.aliceId, workspaceId: null, organizationId: ids.organizationId } - : { userId: ids.aliceId, workspaceId: ids.workspaceId }, + { userId: ids.aliceId, workspaceId: ids.workspaceId }, undefined, 'admin', createContentSyncLease(ids.connectorId, ids.lockId) @@ -628,45 +615,34 @@ describe('independent recovery of retained connector documents', () => { .where(eq(knowledgeConnector.id, ids.connectorId)) }) - it('uses the organization owner and preserves Search visibility during source backoff', async () => { + it('does not restart retired Search indexing when a source leaves provider backoff', async () => { const ids = await seed() - await db.insert(member).values({ - id: generateId(), - organizationId: ids.organizationId, - userId: ids.aliceId, - role: 'owner', - }) + const file = await failedFile(ids) await db .update(knowledgeBase) .set({ workspaceId: null, organizationId: ids.organizationId, isSearchIndex: true }) .where(eq(knowledgeBase.id, ids.knowledgeBaseId)) - const file = await failedFile(ids, true) - await db - .update(knowledgeConnector) - .set({ status: 'error', nextSyncAt: new Date(Date.now() + 3_600_000) }) - .where(eq(knowledgeConnector.id, ids.connectorId)) - expect(await recoverKnowledgeDocumentProcessing()).toBe(1) - const [event] = await eventsFor(ids) - expect(event.payload).toMatchObject({ - billingScope: 'organization', - workspaceId: null, - organizationId: ids.organizationId, + const before = fixture.embeddingCalls + for (const nextSyncAt of [new Date(Date.now() + 3_600_000), old()]) { + await db + .update(knowledgeConnector) + .set({ status: 'error', nextSyncAt }) + .where(eq(knowledgeConnector.id, ids.connectorId)) + await recoverKnowledgeDocumentProcessing() + expect(await eventsFor(ids)).toEqual([]) + } + const [retained] = await db.select().from(document).where(eq(document.id, file.documentId)) + expect(retained).toMatchObject({ + processingStatus: 'failed', + processingAttempts: 1, + processingQueueToken: 'old-fixture-generation', }) + expect(fixture.embeddingCalls).toBe(before) expect( - await outbox.processOutboxEventById(event.id, knowledgeDocumentProcessingOutboxHandlers) - ).toBe('completed') - const [indexed] = await db.select().from(document).where(eq(document.id, file.documentId)) - expect(indexed.processingStatus, indexed.processingError ?? undefined).toBe('completed') - const result = await searchScopedKnowledge.execute({ - principal: { kind: 'session', userId: ids.aliceId, sessionId: 'fixture-session' }, - input: { - organizationId: ids.organizationId, - query: 'Orion', - topK: 3, - }, - }) - expect(result.results.some((row) => row.documentId === file.documentId)).toBe(true) + await db.select().from(embedding).where(eq(embedding.documentId, file.documentId)) + ).toEqual([]) }) + it('recovers while the source is deferred, fences its old worker, and indexes exactly once', async () => { const ids = await seed() const file = await failedFile(ids) diff --git a/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts b/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts deleted file mode 100644 index 0cb20102915..00000000000 --- a/apps/sim/lib/knowledge/__integration__/unfilled-projection-source.integration.ts +++ /dev/null @@ -1,402 +0,0 @@ -/** - * A search candidate's source decides whether the caller's live source proof is resolved before - * its content is read. These fixtures put a GitHub installation source's chunks on projection rows - * the source and ACL fill has not reached (`acl` and `connector_id` NULL), and check that such a - * chunk still reaches a member who holds the installation grant, stays hidden from one who does - * not, and is left out of a page once its source is known to be denied. - */ -import { createHash } from 'node:crypto' -import { db } from '@sim/db' -import { - credential, - credentialGroup, - credentialGroupEnrollment, - document, - embedding, - embeddingKeywordTin, - embeddingSearch, - knowledgeConnector, - knowledgeConnectorMember, - knowledgeDocumentObservation, - organization, - user, - workspace, -} from '@sim/db/schema' -import { generateId } from '@sim/utils/id' -import { isRecordLike } from '@sim/utils/object' -import { eq, inArray, sql } from 'drizzle-orm' -import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' - -/** This suite covers indexed organization search, which is dormant unless Live Search is off. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags( - importOriginal - ) -) -/** The TINQL `resolveTinKeywordQuery` renders for `fixture`: its `english` stem, quoted. */ -vi.mock('@/lib/sim-search/indexed/retrieval/tin-keyword', () => ({ - resolveTinKeywordQuery: async () => '"fixtur"', -})) - -import { - createKnowledgeAclFixtureIds, - seedKnowledgeAclFixture, -} from '@/lib/knowledge/__integration__/seed-source-access-fixture' -import type { - GitHubInstallationReadGrant, - KnowledgeAccessProvider, - UserAccessScope, -} from '@/lib/knowledge/access/types' -import { liveSourceAccessForConnectors } from '@/lib/knowledge/search/candidates' -import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' -import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' -import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' -import type { IndexedRetrievalContext } from '@/lib/sim-search/indexed/retrieval/permitted' -import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' -import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' - -const ids = createKnowledgeAclFixtureIds() -const connectorId = generateId() -const contentCredentialId = generateId() -const groupId = generateId() -const optionId = generateId() -const documentId = generateId() -const embeddingId = generateId() -const repositoryId = '4242' -const members = { - alice: { id: generateId(), subject: 'alice-gh', credentialId: generateId() }, - bob: { id: generateId(), subject: 'bob-gh', credentialId: generateId() }, -} -const userIdOf = (who: 'alice' | 'bob') => (who === 'alice' ? ids.aliceId : ids.bobId) -const subjectToken = (subject: string) => `s:github-repositories:-:${subject}` -const queryVector = { - vector: JSON.stringify([1, ...Array(1535).fill(0)]), - dimensions: 1536 as const, - model: 'text-embedding-3-small', -} - -/** The shims stand in for the Tin extension, which the test database does not carry. */ -let createdTinShims = false - -const scopeFor = (who: 'alice' | 'bob'): UserAccessScope => ({ - kind: 'user', - userId: userIdOf(who), - tokens: [ - 'pub', - subjectToken(members[who].subject), - `u:${userIdOf(who)}@fixture.test`, - 'ws', - ].sort(), -}) - -const planFor = (who: 'alice' | 'bob'): SearchAccessPlan => ({ - connectors: { - workspace: [], - admin: [], - members: [connectorId], - liveProofRequired: [connectorId], - }, - observers: { confirmed: [{ id: members[who].id, connectorId }], observed: [] }, - memberSources: [connectorId], - connectorTypes: new Map([[connectorId, 'github']]), - uploads: false, -}) - -/** Alice's reader credential backs a real installation grant; Bob holds none. */ -const aliceGrant: GitHubInstallationReadGrant = { - connectorId, - contentCredentialId, - readerCredentialId: members.alice.credentialId, - readerSubjectToken: subjectToken(members.alice.subject), - repositoryId, -} - -const searchInputs = (who: 'alice' | 'bob') => ({ - knowledgeBaseIds: [ids.knowledgeBaseId], - topK: 5, - access: scopeFor(who), - queryVector, -}) - -/** A narrow reader of the search index, whose live installation grant is resolved on demand. */ -function searchContext(who: 'alice' | 'bob'): IndexedRetrievalContext { - const access = scopeFor(who) - const accessPlan = planFor(who) - const granted = who === 'alice' ? { ...access, githubInstallationGrants: [aliceGrant] } : access - const accessProvider: KnowledgeAccessProvider = { - get: async () => access, - getForConnectors: async () => granted, - getForDocuments: async () => granted, - liveSourceConnectorCondition: async () => null, - } - return { - access, - accessPlan, - filtered: false, - permitted: { kind: 'unbounded', broad: false }, - liveSourceAccess: liveSourceAccessForConnectors( - accessPlan.connectors.liveProofRequired, - accessProvider - ), - } -} - -const keywordIds = async (who: 'alice' | 'bob') => - ( - await executeIndexedKeywordSearch( - { ...searchInputs(who), query: 'fixture' }, - searchContext(who) - ) - ).map((row) => row.id) - -/** - * Ranked exactly on the row, under the same visibility predicate the graph walk applies. The walk - * is approximate: in a graph the other files sharing this database crowd with degenerate vectors, - * a chunk can be pruned from every neighbour list and never be reached, however far the walk goes. - */ -const vectorIds = async (who: 'alice' | 'bob') => - ( - await selectIndexedVectorResults( - { ...searchInputs(who), distanceThreshold: 2 }, - searchContext(who) - ) - ).map((row) => row.id) - -async function setProjection(state: 'filled' | 'unfilled') { - for (const table of [embeddingSearch, embeddingKeywordTin]) { - await db - .update(table) - .set( - state === 'filled' - ? { - connectorId, - acl: [subjectToken(members.alice.subject), subjectToken(members.bob.subject)].sort(), - } - : { connectorId: null, acl: null } - ) - .where(eq(table.id, embeddingId)) - } - forgetProjectionFilled() -} - -beforeAll(async () => { - await seedKnowledgeAclFixture(ids) - const now = new Date() - await db - .update(user) - .set({ emailVerified: true }) - .where(inArray(user.id, [ids.aliceId, ids.bobId])) - await db.insert(credential).values({ - id: contentCredentialId, - workspaceId: ids.workspaceId, - type: 'service_account', - displayName: 'Fixture GitHub installation', - createdBy: ids.aliceId, - providerId: GITHUB_INSTALLATION_PROVIDER_ID, - }) - await db.insert(credentialGroup).values({ - id: groupId, - workspaceId: ids.workspaceId, - publicId: generateId(), - name: 'GitHub readers', - options: [ - { - id: optionId, - provider: 'github-repositories', - label: 'GitHub fixture', - authorizationAppId: 'fixture-app', - requiredScopes: ['repo'], - scopeVersion: 1, - required: false, - status: 'active', - }, - ], - } as typeof credentialGroup.$inferInsert) - for (const who of ['alice', 'bob'] as const) { - const [enrollment] = await db - .insert(credentialGroupEnrollment) - .values({ - id: generateId(), - credentialGroupId: groupId, - userId: userIdOf(who), - email: `${userIdOf(who)}@fixture.test`, - status: 'completed', - invitationTokenHash: createHash('sha256').update(generateId()).digest('hex'), - invitationExpiresAt: new Date(Date.now() + 60 * 60 * 1000), - invitedAt: now, - }) - .returning({ id: credentialGroupEnrollment.id }) - await db.insert(credential).values({ - id: members[who].credentialId, - workspaceId: ids.workspaceId, - type: 'managed_oauth', - displayName: 'Fixture GitHub reader', - providerId: 'github-repositories', - authorizationAppId: 'fixture-app', - credentialGroupEnrollmentId: enrollment!.id, - credentialGroupOptionId: optionId, - managedOauthScopeVersion: 1, - providerSubjectId: members[who].subject, - providerTenantId: '', - managedOauthStatus: 'active', - grantedScopes: ['repo'], - encryptedOauthTokenSet: 'fixture-not-an-oauth-token', - grantedAt: now, - createdBy: userIdOf(who), - }) - } - await db.insert(knowledgeConnector).values({ - id: connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - connectorType: 'github', - sourceConfig: { githubRepositoryId: repositoryId }, - accessMode: 'members', - status: 'active', - credentialId: contentCredentialId, - credentialGroupId: groupId, - credentialGroupOptionId: optionId, - }) - await db.insert(knowledgeConnectorMember).values( - (['alice', 'bob'] as const).map((who) => ({ - id: members[who].id, - workspaceId: ids.workspaceId, - connectorId, - credentialId: members[who].credentialId, - subjectToken: subjectToken(members[who].subject), - status: 'active', - memberSyncedThrough: now, - })) - ) - await db.insert(document).values({ - id: documentId, - connectorId, - knowledgeBaseId: ids.knowledgeBaseId, - externalId: 'fixture-file', - filename: 'readme.md', - fileUrl: 'https://fixture.test/readme', - fileSize: 12, - mimeType: 'text/plain', - processingStatus: 'completed', - acl: [subjectToken(members.alice.subject), subjectToken(members.bob.subject)].sort(), - }) - await db.insert(knowledgeDocumentObservation).values( - (['alice', 'bob'] as const).map((who) => ({ - documentId, - memberId: members[who].id, - lastSeenAt: now, - runId: generateId(), - })) - ) - await db.insert(embedding).values({ - id: embeddingId, - documentId, - knowledgeBaseId: ids.knowledgeBaseId, - chunkIndex: 0, - chunkHash: 'fixture-hash', - content: 'fixture readme', - contentLength: 14, - tokenCount: 2, - startOffset: 0, - endOffset: 14, - embeddingModel: 'text-embedding-3-small', - embedding: [1, ...Array(1535).fill(0)], - }) - const [tin] = await db.execute<{ present: boolean }>( - sql`SELECT to_regnamespace('tin') IS NOT NULL AS present` - ) - if (!tin?.present) { - createdTinShims = true - await db.execute( - sql.raw(`CREATE SCHEMA tin; - CREATE FUNCTION tin.full_score(tid) RETURNS double precision LANGUAGE sql IMMUTABLE AS 'SELECT 1.0::float8'; - CREATE FUNCTION knowledge_tin_base_token(text) RETURNS text LANGUAGE sql IMMUTABLE AS $$SELECT 'kb'$$; - CREATE FUNCTION knowledge_tin_stream(vector tsvector) RETURNS text LANGUAGE sql IMMUTABLE AS $$ - SELECT coalesce(string_agg(entry.lexeme, ' ' ORDER BY position), '') - FROM unnest(vector) AS entry(lexeme, positions, weights), unnest(entry.positions) AS position - $$; - CREATE FUNCTION tin_fixture_match(text, text) RETURNS boolean LANGUAGE sql IMMUTABLE AS 'SELECT true'; - CREATE OPERATOR ==> (LEFTARG = text, RIGHTARG = text, FUNCTION = tin_fixture_match);`) - ) - } - /** Written as the projection trigger writes it, so real Tin scopes the row to its base. */ - await db.execute(sql` - INSERT INTO ${embeddingKeywordTin} (id, knowledge_base_id, document_id, enabled, content) - SELECT id, knowledge_base_id, document_id, enabled, - knowledge_tin_base_token(knowledge_base_id) || ' ' || knowledge_tin_stream(content_tsv) - FROM ${embedding} WHERE id = ${embeddingId} - ON CONFLICT (id) DO UPDATE SET content = EXCLUDED.content`) -}) - -afterAll(async () => { - if (createdTinShims) { - await db.execute( - sql.raw(`DROP OPERATOR IF EXISTS ==> (text, text); - DROP FUNCTION IF EXISTS tin_fixture_match(text, text); - DROP FUNCTION IF EXISTS knowledge_tin_base_token(text); - DROP FUNCTION IF EXISTS knowledge_tin_stream(tsvector); - DROP SCHEMA IF EXISTS tin CASCADE;`) - ) - } - await db.delete(embeddingKeywordTin).where(eq(embeddingKeywordTin.id, embeddingId)) - await db.delete(workspace).where(eq(workspace.id, ids.workspaceId)) - await db.delete(credentialGroup).where(eq(credentialGroup.id, groupId)) - await db.delete(organization).where(eq(organization.id, ids.organizationId)) - await db.delete(user).where(inArray(user.id, [ids.aliceId, ids.bobId])) - forgetProjectionFilled() -}) - -describe('a chunk whose projection row the fill has not reached', () => { - beforeEach(() => setProjection('unfilled')) - - it('reaches the member holding the installation grant through the keyword ranking', async () => { - expect(await keywordIds('alice')).toEqual([embeddingId]) - }) - - it('reaches the member holding the installation grant through the vector ranking', async () => { - expect(await vectorIds('alice')).toEqual([embeddingId]) - }) - - it('stays hidden from a member without the grant', async () => { - expect(await keywordIds('bob')).toEqual([]) - expect(await vectorIds('bob')).toEqual([]) - }) - - describe('once its source is known to be denied', () => { - /** Each Tin ranking statement's page of candidates, in the order the search read them. */ - const pages: Array<{ candidates: unknown[] }> = [] - beforeEach(() => { - pages.length = 0 - const execute = db.execute.bind(db) - vi.spyOn(db, 'execute').mockImplementation((async (query: Parameters[0]) => { - const rows = await execute(query) - const [row] = Array.from(rows) - if (isRecordLike(row) && 'ranked' in row && Array.isArray(row.candidates)) - pages.push({ candidates: row.candidates }) - return rows - }) as typeof db.execute) - }) - afterEach(() => vi.restoreAllMocks()) - - it('carries the source read from its document and is left out of the rebuilt keyword page', async () => { - expect(await keywordIds('bob')).toEqual([]) - expect(pages.length).toBeGreaterThanOrEqual(2) - expect(pages[0]!.candidates).toEqual([{ id: embeddingId, documentId, connectorId }]) - expect(pages.at(-1)!.candidates).toEqual([]) - }) - }) -}) - -describe('a chunk whose projection row is filled', () => { - beforeEach(() => setProjection('filled')) - - it('ranks on the row as before and is read only by the member holding the grant', async () => { - const [{ unfilled }] = await db.execute<{ unfilled: boolean }>( - sql`SELECT EXISTS (SELECT 1 FROM ${embeddingKeywordTin} WHERE ${embeddingKeywordTin.acl} IS NULL) AS unfilled` - ) - expect(unfilled).toBe(false) - expect(await keywordIds('alice')).toEqual([embeddingId]) - expect(await keywordIds('bob')).toEqual([]) - expect(await vectorIds('alice')).toEqual([embeddingId]) - expect(await vectorIds('bob')).toEqual([]) - }) -}) diff --git a/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts b/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts index c0fc6cfd66b..85be5ca3113 100644 --- a/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/workspace-kb-document-access.integration.ts @@ -5,8 +5,8 @@ * the keyword projection — so projection rows that are stale, unfilled, or missing change * nothing about what a workspace search returns. A signed-in reader's own grants widen what they * read, and a source whose reader must be proven live admits a candidate only once the proof - * holds, with the readable rows below it filling the page when it does not. A search index named - * by id while indexed organization search is dormant is searched the same way. + * holds, with the readable rows below it filling the page when it does not. A Search-marked knowledge base + * named explicitly by id is searched the same way. */ import { createHash } from 'node:crypto' import type { Principal } from '@sim/auth/principal' @@ -29,14 +29,7 @@ import { } from '@sim/db/schema' import { generateId } from '@sim/utils/id' import { eq, inArray } from 'drizzle-orm' -import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' - -/** Pinned to Live Search, so indexed organization search is dormant whatever the run's environment. */ -vi.mock('@/lib/core/config/env-flags', async (importOriginal) => ({ - ...(await importOriginal>()), - isLiveEnterpriseSearchEnabled: true, -})) - +import { afterAll, beforeAll, describe, expect, it } from 'vitest' import { createKnowledgeAclFixtureIds, seedKnowledgeAclFixture, @@ -50,7 +43,6 @@ import type { import { type KnowledgeSearchMode, retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' import { embeddingVectorValues } from '@/lib/knowledge/vector-columns' import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' -import { usesIndexedRetrieval } from '@/lib/sim-search/indexed/gate' afterAll(async () => { await db.$client.end() @@ -165,7 +157,6 @@ describe.each([ access: await accessProvider.get(), accessProvider, searchMode, - indexedRetrieval: usesIndexedRetrieval([{ isSearchIndex }]), query: 'handbook onboarding', queryVector, }) diff --git a/apps/sim/lib/knowledge/access/predicate.integration.ts b/apps/sim/lib/knowledge/access/predicate.integration.ts index c42d9108c8b..d00e3f08dcf 100644 --- a/apps/sim/lib/knowledge/access/predicate.integration.ts +++ b/apps/sim/lib/knowledge/access/predicate.integration.ts @@ -1,6 +1,6 @@ import { readFile } from 'node:fs/promises' import { readTestDatabaseUrl } from '@sim/db/testing/test-infrastructure' -import { type SQL, sql } from 'drizzle-orm' +import type { SQL } from 'drizzle-orm' import type postgres from 'postgres' import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' import { createEnterpriseSearchMigrationFixture } from '@/lib/knowledge/__integration__/migration-fixture' @@ -26,9 +26,6 @@ const { PgDialect } = await import('drizzle-orm/pg-core') const { knowledgeAccessCondition, knowledgeMetadataCandidateAccessCondition } = await import( '@/lib/knowledge/access/predicate' ) -const { restrictSearchAccessPlan } = await import('@/lib/sim-search/indexed/retrieval/access-plan') -const { knowledgeCandidateAccessConditionForConnectors, projectionCandidateAccessCondition } = - await import('@/lib/sim-search/indexed/retrieval/projection-access') const { confluencePageAcl } = await import('@/lib/knowledge/access/confluence-permissions') /** Every table and index belongs to an isolated disposable schema. */ @@ -511,209 +508,6 @@ describe('knowledge ACLs in PostgreSQL', () => { expect(await readable(['ws'], 'upload')).toBe(true) }) - it('admits the same documents whether connector state is proven per row or resolved per query', async () => { - const scope = { kind: 'user' as const, userId: 'reader', tokens: [alice, 'u:alice@corp.com'] } - await connection.unsafe( - `INSERT INTO knowledge_connector(id, access_mode, deleted_at, archived_at) VALUES - ('gone', 'admin', statement_timestamp(), NULL), ('shelved', 'admin', NULL, statement_timestamp())` - ) - await connection.unsafe( - "INSERT INTO knowledge_connector(id, access_mode) VALUES ('ws-mode', 'workspace')" - ) - await connection.unsafe( - `INSERT INTO knowledge_connector_member(id, workspace_id, connector_id, subject_token, status) - VALUES ('m-alice', 'workspace', 'members', $1, 'active')`, - [alice] - ) - const cases: Array<[string, string, string[]]> = [ - ['admin-current', 'admin', ['u:alice@corp.com']], - ['members-current', 'members', [alice]], - ['workspace-doc', 'ws-mode', ['ws']], - ['deleted-connector', 'gone', ['u:alice@corp.com']], - ['archived-connector', 'shelved', ['u:alice@corp.com']], - ] - for (const [id, connectorId, acl] of cases) { - await connection.unsafe( - `INSERT INTO document(id, connector_id, acl, acl_verified_at) - VALUES ($1, $2, string_to_array($3, E'\n'), statement_timestamp())`, - [id, connectorId, acl.join('\n')] - ) - } - await connection.unsafe( - "INSERT INTO knowledge_document_observation VALUES ('members-current', 'm-alice', statement_timestamp())" - ) - await connection.unsafe("INSERT INTO document(id) VALUES ('upload-doc')") - /** What `resolveConnectorEligibility` returns for this base: the connectors it admits, by mode. */ - const eligibility = { - workspace: ['ws-mode'], - admin: ['admin'], - members: ['members'], - liveProofRequired: [], - } - /** What `resolveSearchAccessPlan` resolves for this caller: their member identity, confirmed. */ - const observers = { confirmed: [{ id: 'm-alice', connectorId: 'members' }], observed: [] } - const perRow = knowledgeMetadataCandidateAccessCondition(scope) - const plan = { - connectors: eligibility, - observers, - memberSources: ['members'], - connectorTypes: new Map([ - ['ws-mode', 'slack'], - ['admin', 'google_drive'], - ['members', 'slack'], - ]), - uploads: true, - } - const perQuery = knowledgeCandidateAccessConditionForConnectors(scope, plan) - for (const id of [...cases.map(([documentId]) => documentId), 'upload-doc']) { - expect([id, await admits(perQuery, id)]).toEqual([id, await admits(perRow, id)]) - } - /** - * A document that changed hands keeps its old observations. The caller's member observed it - * while its old connector held it; under its new connector, of which the caller is no member, - * neither predicate may carry it on that observation. - */ - await connection.unsafe( - "INSERT INTO knowledge_connector(id, access_mode) VALUES ('members-elsewhere', 'members')" - ) - await connection.unsafe( - `INSERT INTO document(id, connector_id, acl, acl_verified_at) - VALUES ('moved-doc', 'members-elsewhere', ARRAY[$1], statement_timestamp())`, - [alice] - ) - await connection.unsafe( - "INSERT INTO knowledge_document_observation VALUES ('moved-doc', 'm-alice', statement_timestamp())" - ) - const movedPlan = { - ...plan, - connectors: { ...eligibility, members: [...eligibility.members, 'members-elsewhere'] }, - } - expect(await admits(perRow, 'moved-doc')).toBe(false) - expect( - await admits(knowledgeCandidateAccessConditionForConnectors(scope, movedPlan), 'moved-doc') - ).toBe(false) - /** - * The projection predicate decides on the ranking row alone, from the source and ACL mirrored - * there. It must never refuse a document the per-row predicate admits: whatever it admits beyond - * that is refused at hydration, under the full predicate, before content is returned. - */ - await connection.unsafe(`CREATE TABLE IF NOT EXISTS embedding_search( - id text PRIMARY KEY, document_id text, connector_id text, acl text[])`) - await connection.unsafe( - 'CREATE TABLE IF NOT EXISTS knowledge_projection_dirty(document_id text PRIMARY KEY)' - ) - await connection.unsafe('DELETE FROM embedding_search') - await connection.unsafe( - "INSERT INTO embedding_search SELECT id || '-chunk', id, connector_id, acl FROM document" - ) - const onRow = new PgDialect().sqlToQuery( - projectionCandidateAccessCondition(schema.embeddingSearch, scope, plan) - ) - const onRowAdmits = async (id: string) => { - const rows = await connection.unsafe( - `SELECT 1 FROM embedding_search WHERE ${onRow.sql} AND document_id = $${onRow.params.length + 1}`, - [...(onRow.params as string[]), id] - ) - return rows.length > 0 - } - for (const id of [...cases.map(([documentId]) => documentId), 'upload-doc']) { - if (await admits(perRow, id)) expect([id, await onRowAdmits(id)]).toEqual([id, true]) - } - /** - * A source the caller is a member of still holds documents that name other members only; the - * mirrored ACL keeps those out of the ranking rather than leaving them for hydration to drop. - */ - await connection.unsafe( - "INSERT INTO document(id, connector_id, acl, acl_verified_at) VALUES ('members-other', 'members', ARRAY['s:slack:-:bob'], statement_timestamp())" - ) - await connection.unsafe( - "INSERT INTO embedding_search SELECT id || '-chunk', id, connector_id, acl FROM document WHERE id = 'members-other'" - ) - expect(await admits(perRow, 'members-other')).toBe(false) - expect(await onRowAdmits('members-other')).toBe(false) - /** - * A chunk the backfill has not reached carries no source or ACL yet and is decided on its - * document, as every candidate was before the columns existed: search must not depend on the - * backfill, must not lose a document to it, and must not rank an unreadable one because of it. - */ - await connection.unsafe( - "INSERT INTO document(id, connector_id, acl, acl_verified_at) VALUES ('unfilled-other', 'members', ARRAY['s:slack:-:bob'], statement_timestamp())" - ) - await connection.unsafe(`INSERT INTO embedding_search(id, document_id) VALUES - ('unfilled-other-chunk', 'unfilled-other'), ('unfilled-admin-chunk', 'admin-current'), - ('unfilled-members-chunk', 'members-current'), ('unfilled-gone-chunk', 'deleted-connector')`) - await connection.unsafe( - "DELETE FROM embedding_search WHERE id IN ('admin-current-chunk', 'members-current-chunk', 'deleted-connector-chunk')" - ) - expect(await onRowAdmits('unfilled-other')).toBe(false) - expect(await onRowAdmits('admin-current')).toBe(true) - expect(await onRowAdmits('members-current')).toBe(true) - expect(await onRowAdmits('deleted-connector')).toBe(false) - /** - * Candidate ranking defers the live source proof, as the per-row candidate predicate does: a - * caller holds those grants only after authorization, so applying the clause during ranking - * would drop every candidate of a gated source before it could be proven. - */ - const gated = { ...plan, connectors: { ...eligibility, liveProofRequired: ['admin'] } } - expect( - await admits(knowledgeCandidateAccessConditionForConnectors(scope, gated), 'admin-current') - ).toBe(true) - expect( - await admits( - knowledgeCandidateAccessConditionForConnectors(scope, gated, sql`false`), - 'admin-current' - ) - ).toBe(false) - /** - * A plan confined to one kind of source admits that kind alone, on the document and on the - * row, and a plan confined to uploads admits only documents without a source. - */ - const drive = restrictSearchAccessPlan(plan, 'google_drive') - const driveOnRow = new PgDialect().sqlToQuery( - projectionCandidateAccessCondition(schema.embeddingSearch, scope, drive) - ) - const driveOnRowAdmits = async (id: string) => { - const rows = await connection.unsafe( - `SELECT 1 FROM embedding_search WHERE ${driveOnRow.sql} AND document_id = $${driveOnRow.params.length + 1}`, - [...(driveOnRow.params as string[]), id] - ) - return rows.length > 0 - } - const drivePerQuery = knowledgeCandidateAccessConditionForConnectors(scope, drive) - for (const [id, expected] of [ - ['admin-current', true], - ['workspace-doc', false], - ['members-current', false], - ['upload-doc', false], - ] as const) { - expect([id, await admits(drivePerQuery, id)]).toEqual([id, expected]) - expect([id, await driveOnRowAdmits(id)]).toEqual([id, expected]) - } - /** Uploads carry the workspace ACL, so a caller with that token reads them and nothing sourced. */ - await connection.unsafe("INSERT INTO document(id, acl) VALUES ('upload-mine', ARRAY['ws'])") - const wsScope = { ...scope, tokens: [...scope.tokens, 'ws'] } - const uploadsPerQuery = knowledgeCandidateAccessConditionForConnectors( - wsScope, - restrictSearchAccessPlan(plan, 'upload') - ) - expect(await admits(knowledgeMetadataCandidateAccessCondition(wsScope), 'upload-mine')).toBe( - true - ) - expect(await admits(uploadsPerQuery, 'upload-mine')).toBe(true) - expect(await admits(uploadsPerQuery, 'workspace-doc')).toBe(false) - expect(await admits(uploadsPerQuery, 'admin-current')).toBe(false) - /** A connector left out of the resolution is refused, however current its documents are. */ - expect( - await admits( - knowledgeCandidateAccessConditionForConnectors(scope, { - ...plan, - connectors: { ...eligibility, admin: [] }, - }), - 'admin-current' - ) - ).toBe(false) - }) - it('does not let one member refresh another member’s stale observation', async () => { await connection.unsafe( "INSERT INTO document(id, connector_id, acl) VALUES ('shared', 'members', string_to_array($1, ','))", diff --git a/apps/sim/lib/knowledge/access/predicate.test.ts b/apps/sim/lib/knowledge/access/predicate.test.ts index 06b4e248481..b2ef8ba4d37 100644 --- a/apps/sim/lib/knowledge/access/predicate.test.ts +++ b/apps/sim/lib/knowledge/access/predicate.test.ts @@ -13,80 +13,13 @@ vi.unmock('@sim/db/schema') process.env.DATABASE_URL ??= 'postgresql://user:pass@localhost:5432/test' const { PgDialect } = await import('drizzle-orm/pg-core') -const { embeddingSearch } = await import('@sim/db/schema') -const { sql: rawSql } = await import('drizzle-orm') -const sqlColumn = (name: string) => rawSql.raw(`"row"."${name}"`) const { knowledgeAccessCondition } = await import('@/lib/knowledge/access/predicate') -const { restrictSearchAccessPlan } = await import('@/lib/sim-search/indexed/retrieval/access-plan') -const { knowledgeCandidateAccessConditionForConnectors, projectionCandidateAccessCondition } = - await import('@/lib/sim-search/indexed/retrieval/projection-access') const { SYSTEM_ACCESS_SCOPE } = await import('@/lib/knowledge/access/types') function render(condition: ReturnType) { return new PgDialect().sqlToQuery(condition) } -describe('projectionCandidateAccessCondition', () => { - const plan = { - connectors: { workspace: ['ws-src'], admin: [], members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - } - - const pending = - '(EXISTS (SELECT 1 FROM "knowledge_projection_dirty" WHERE "knowledge_projection_dirty"."document_id" = "embedding_search"."document_id"))' - const onDocument = - 'AND (SELECT "document"."id" FROM "document"\n WHERE "document"."id" = "embedding_search"."document_id"\n AND (' - - it('decides a filled row on its mirrored columns and an unfilled or marked row on its document', () => { - const { sql, params } = render( - projectionCandidateAccessCondition( - embeddingSearch, - { kind: 'user', userId: 'user-1', tokens: ['ws', 'u:alice'] }, - plan - ) - ) - expect(sql).toContain(`(("embedding_search"."acl" IS NULL OR ${pending}) ${onDocument}`) - expect(sql).toContain('"document"."acl" && ARRAY[$1, $2]::text[]') - expect(sql).toContain('OR (("embedding_search"."acl" && ARRAY[') - expect(sql).toContain( - 'AND ("embedding_search"."connector_id" IS NULL OR "embedding_search"."connector_id" = ANY(ARRAY[' - ) - expect(sql).toContain(`AND NOT ${pending})`) - expect(params.slice(0, 2)).toEqual(['ws', 'u:alice']) - expect(params.slice(-3)).toEqual(['ws', 'u:alice', 'ws-src']) - for (const param of params) expect(Array.isArray(param)).toBe(false) - }) - - it('still decides a marked row on its document once the projection is filled', () => { - const { sql } = render( - projectionCandidateAccessCondition( - embeddingSearch, - { kind: 'user', userId: 'user-1', tokens: ['ws', 'u:alice'] }, - plan, - { filled: true } - ) - ) - expect(sql).toContain(`((${pending} ${onDocument}`) - expect(sql).not.toContain('"embedding_search"."acl" IS NULL') - expect(sql).toContain(`AND NOT ${pending})`) - }) - - it('still denies everything for an empty token set', () => { - expect( - render( - projectionCandidateAccessCondition( - embeddingSearch, - { kind: 'user', userId: 'user-1', tokens: [] }, - plan - ) - ).sql - ).toBe('false') - }) -}) - describe('knowledgeAccessCondition', () => { it('overlaps the ACL with the tokens as a literal array of scalar binds', () => { const { sql, params } = render( @@ -134,52 +67,3 @@ describe('knowledgeAccessCondition', () => { expect(sql).not.toContain('"document"."acl"') }) }) - -describe('restrictSearchAccessPlan', () => { - const plan = { - connectors: { - workspace: ['slack-ws'], - admin: ['drive-admin', 'confluence-admin'], - members: ['slack-members'], - liveProofRequired: ['confluence-admin'], - }, - observers: { - confirmed: [{ id: 'm-1', connectorId: 'slack-members' }], - observed: [{ id: 'm-2', connectorId: 'drive-admin' }], - }, - memberSources: ['slack-members'], - connectorTypes: new Map([ - ['slack-ws', 'slack'], - ['slack-members', 'slack'], - ['drive-admin', 'google_drive'], - ['confluence-admin', 'confluence'], - ]), - uploads: true, - } - const reader = { kind: 'user' as const, userId: 'u', tokens: ['u:reader@example.com'] } - - it('drops source-less rows from both predicates once uploads are out of scope', () => { - const rowSql = (restricted: typeof plan) => - render( - projectionCandidateAccessCondition( - { - connectorId: sqlColumn('connector_id'), - acl: sqlColumn('acl'), - documentId: sqlColumn('document_id'), - }, - reader, - restricted - ) - ).sql - expect(rowSql(plan)).toContain('IS NULL OR') - expect(rowSql(restrictSearchAccessPlan(plan, 'slack'))).not.toContain( - '"row"."connector_id" IS NULL' - ) - const documentSql = (restricted: typeof plan) => - render(knowledgeCandidateAccessConditionForConnectors(reader, restricted)).sql - expect(documentSql(plan)).toContain('"document"."connector_id" IS NULL OR') - expect(documentSql(restrictSearchAccessPlan(plan, 'slack'))).not.toContain( - '"document"."connector_id" IS NULL' - ) - }) -}) diff --git a/apps/sim/lib/knowledge/api/route-policies.ts b/apps/sim/lib/knowledge/api/route-policies.ts index 27f3b4c36f8..80102c19bec 100644 --- a/apps/sim/lib/knowledge/api/route-policies.ts +++ b/apps/sim/lib/knowledge/api/route-policies.ts @@ -22,7 +22,6 @@ import { KnowledgeDocumentNotReadyError } from '@/lib/knowledge/application/chun import { KnowledgeSearchProvenanceUnavailableError } from '@/lib/knowledge/application/search' import { KnowledgeDocumentUnsupportedMediaTypeError } from '@/lib/knowledge/application/upload-sessions' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' -import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' import { v2Error } from '@/app/api/v2/lib/response' function internalKnowledgeErrorPolicy(unhandledMessage: string): InternalErrorPolicy { @@ -89,18 +88,6 @@ export const internalKnowledgeSessionOrExecutorAuth = createInternalSessionOrExe export const KNOWLEDGE_BASE_NOT_FOUND_MESSAGE = 'Knowledge base not found' -/** - * Answers an indexed-only surface refused while indexed organization search is dormant with a - * `409`: the request is well formed and authorized, and the deployment's state is what refuses it. - */ -function refuseDormantSearchIndex(base: InternalErrorPolicy): InternalErrorPolicy { - return extendInternalErrorPolicy(base, (error) => - error instanceof SearchIndexDormantError - ? internalErrorResponse(409, { error: error.message }) - : null - ) -} - /** * Conceals a knowledge-base-scoped internal policy the way the v2 knowledge * routes conceal theirs. The workspace-level `list` and `create` policies are @@ -154,12 +141,8 @@ export const internalKnowledgeErrorPolicies = { tags: concealKnowledgeBase( internalKnowledgeErrorPolicy('Failed to process knowledge tag request') ), - connectors: concealKnowledgeBase( - refuseDormantSearchIndex(internalKnowledgeErrorPolicy('Internal server error')) - ), - connectAccount: concealKnowledgeBase( - refuseDormantSearchIndex(internalPersonalCredentialConnectionErrorPolicy) - ), + connectors: concealKnowledgeBase(internalKnowledgeErrorPolicy('Internal server error')), + connectAccount: concealKnowledgeBase(internalPersonalCredentialConnectionErrorPolicy), uploads: concealKnowledgeBase(internalKnowledgeUploadErrorPolicy), } as const diff --git a/apps/sim/lib/knowledge/application/connect-personal-search-integration.test.ts b/apps/sim/lib/knowledge/application/connect-personal-search-integration.test.ts index fc2c7726945..67f5b55117e 100644 --- a/apps/sim/lib/knowledge/application/connect-personal-search-integration.test.ts +++ b/apps/sim/lib/knowledge/application/connect-personal-search-integration.test.ts @@ -17,7 +17,6 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' const hoisted = vi.hoisted(() => ({ resolve: vi.fn(), - indexed: vi.fn(), oauth: vi.fn(), })) vi.mock('@/lib/core/application/organization-authorization', () => organizationAuthorizationMock) @@ -25,9 +24,6 @@ vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) vi.mock('@/lib/knowledge/application/personal-search-integrations', () => ({ resolvePersonalSearchConnection: { execute: hoisted.resolve }, })) -vi.mock('@/lib/knowledge/application/sim-search', () => ({ - connectSimSearchConnector: { execute: hoisted.indexed }, -})) vi.mock('@/lib/credential-groups/service', () => credentialGroupsServiceMock) vi.mock('@/lib/credential-groups/scoped-availability', () => credentialGroupsAvailabilityMock) vi.mock('@/lib/credential-groups/credentials', () => credentialGroupsCredentialsMock) @@ -90,7 +86,6 @@ describe('personal live Search connection', () => { input: { ...input, target: selected }, }) expect(result.url).toBe('https://provider.test/oauth') - expect(result.connectorId).toBeUndefined() expect( connectPersonalSearchIntegrationContract.response.schema.safeParse({ success: true, @@ -105,7 +100,6 @@ describe('personal live Search connection', () => { completionId: input.oauthCompletionId, connectionIntent: credentialId ? { kind: 'reconnect', credentialId } : { kind: 'create' }, }) - expect(m.indexed).not.toHaveBeenCalled() expect( organizationAuthorizationMockFns.mockAuthorizeOrganizationOperation ).toHaveBeenCalledWith( @@ -125,7 +119,6 @@ describe('personal live Search connection', () => { 'no longer available' ) expect(m.oauth).not.toHaveBeenCalled() - expect(m.indexed).not.toHaveBeenCalled() }) it('rechecks disabled account options after resolving the card', async () => { @@ -139,33 +132,4 @@ describe('personal live Search connection', () => { ) expect(m.oauth).not.toHaveBeenCalled() }) - - it('continues to use indexed enrollment for indexed controls', async () => { - const indexed = { - type: 'link', - provider: 'slack', - connectorType: 'slack', - connectorId: 'source', - } as const - m.resolve.mockResolvedValue({ target: indexed }) - m.indexed.mockResolvedValue({ - url: 'https://provider.test/oauth', - connectorId: 'source', - knowledgeBaseId: 'kb', - }) - await connectPersonalSearchIntegration.execute({ - principal, - input: { ...input, target: indexed }, - }) - expect(m.indexed).toHaveBeenCalledWith( - expect.objectContaining({ - principal, - input: expect.objectContaining({ - connectorId: 'source', - oauthCompletionId: input.oauthCompletionId, - }), - }) - ) - expect(m.oauth).not.toHaveBeenCalled() - }) }) diff --git a/apps/sim/lib/knowledge/application/connect-personal-search-integration.ts b/apps/sim/lib/knowledge/application/connect-personal-search-integration.ts index 8dbfaf8051a..bf24747149d 100644 --- a/apps/sim/lib/knowledge/application/connect-personal-search-integration.ts +++ b/apps/sim/lib/knowledge/application/connect-personal-search-integration.ts @@ -3,14 +3,12 @@ import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/au import { resolveKnowledgeOrganizationContext } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' import { resolvePersonalSearchConnection } from '@/lib/knowledge/application/personal-search-integrations' -import { connectSimSearchConnector } from '@/lib/knowledge/application/sim-search' import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' interface ConnectPersonalSearchIntegrationInput { organizationId: string target: SearchConnectionTarget oauthCompletionId: string - sourceConfig?: Record } /** A user click validates the requested control before starting the ordinary source enrollment. */ @@ -20,38 +18,20 @@ export const connectPersonalSearchIntegration = defineAuthorizedKnowledgeUseCase resolveKnowledgeOrganizationContext(input), async execute({ principal, input, request }) { const { target } = await resolvePersonalSearchConnection.execute({ principal, input }) - if (target.connectionMode === 'live' && target.optionId) { - const result = await startOrganizationAccountConnection.execute({ - principal, - request, - input: { - organizationId: input.organizationId, - optionId: target.optionId, - oauthCompletionId: input.oauthCompletionId, - connectionIntent: target.credentialId - ? { kind: 'reconnect', credentialId: target.credentialId } - : { kind: 'create' }, - }, - }) - return { - url: result.authorizationUrl ?? result.invitationLink, - connectorId: undefined, - knowledgeBaseId: undefined, - } - } - return connectSimSearchConnector.execute({ + const result = await startOrganizationAccountConnection.execute({ principal, request, input: { organizationId: input.organizationId, - connectorType: target.connectorType, - connectorId: target.connectorId, - sourceConfig: input.sourceConfig, + optionId: target.optionId, oauthCompletionId: input.oauthCompletionId, connectionIntent: target.credentialId ? { kind: 'reconnect', credentialId: target.credentialId } : { kind: 'create' }, }, }) + return { + url: result.authorizationUrl ?? result.invitationLink, + } }, }) diff --git a/apps/sim/lib/knowledge/application/connectors.test.ts b/apps/sim/lib/knowledge/application/connectors.test.ts index 0763cf0c38f..630893c5e51 100644 --- a/apps/sim/lib/knowledge/application/connectors.test.ts +++ b/apps/sim/lib/knowledge/application/connectors.test.ts @@ -17,10 +17,6 @@ import { knowledgeContextsMock, knowledgeContextsMockFns, } from '@sim/testing/mocks/knowledge-contexts.mock' -import { - knowledgeSearchIntegrationPolicyMock, - knowledgeSearchIntegrationPolicyMockFns, -} from '@sim/testing/mocks/knowledge-search-integration-policy.mock' import { permissionGroupsResolveMock, permissionGroupsResolveMockFns, @@ -44,8 +40,6 @@ const hoisted = vi.hoisted(() => ({ vi.mock('@sim/audit', () => auditMock) -vi.mock('@/lib/knowledge/search/integration-policy', () => knowledgeSearchIntegrationPolicyMock) - vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) @@ -120,7 +114,6 @@ import { internalOrchestrationErrorPolicy } from '@/lib/api/server/routes/intern import { OrchestrationError } from '@/lib/core/orchestration/types' import * as encryption from '@/lib/core/security/encryption' import { - createApprovedSearchSource, createKnowledgeConnector, deleteKnowledgeConnector, resolveConnectorCredentialAccessToken, @@ -140,7 +133,6 @@ import { googleDriveConnectorMeta } from '@/connectors/google-drive/meta' const mocks = { ...hoisted, - requireApproval: knowledgeSearchIntegrationPolicyMockFns.mockRequireOrganizationSearchApproval, getCredentialActorContext: credentialsAccessMockFns.mockGetCredentialActorContext, canUseCredential: credentialsAccessMockFns.mockCanUseCredential, resolveTokenIdentity: credentialsAccessMockFns.mockResolveCredentialTokenIdentity, @@ -937,89 +929,6 @@ describe('members-mode connector creation', () => { }) }) -describe('approved organization member source creation', () => { - const principal = createSessionPrincipal({ userId: 'actor', sessionId: 'session' }) - const input = { - knowledgeBaseId: 'org-index', - assertedOrganizationId: 'org', - connectorType: 'google_drive', - sourceConfig: {}, - } - beforeEach(() => { - resetDbChainMock() - queueTableRows(member, [{ role: 'member' }]) - permissionGroupsResolveMockFns.mockGetUserPermissionConfig.mockResolvedValue(null) - mocks.requireApproval.mockResolvedValue(undefined) - knowledgeContextsMockFns.mockResolveActiveKnowledgeBaseContext.mockResolvedValue({ - organizationId: 'org', - knowledgeBaseId: 'org-index', - knowledgeBase: { id: 'org-index', name: 'Search', isSearchIndex: true }, - }) - mocks.resolveMembersBinding.mockResolvedValue({ - organizationId: 'org', - credentialGroupId: 'group', - credentialGroupOptionId: 'option', - }) - mocks.createConnector.mockResolvedValue({ - success: true, - connector: { id: 'connector', connectorType: 'google_drive', accessMode: 'members' }, - }) - }) - - it('uses the member actor and ignores attempts to supply credentials or broader access', async () => { - const maliciousInput = { - ...input, - credentialId: 'other-person', - apiKey: 'injected', - accessMode: 'admin', - } - await createApprovedSearchSource.execute({ principal, input: maliciousInput }) - expect(mocks.requireApproval).toHaveBeenCalledWith('org', 'google_drive') - expect(mocks.resolveMembersBinding).toHaveBeenCalledWith( - expect.objectContaining({ actingUserId: 'actor', organizationId: 'org' }) - ) - expect(mocks.createConnector).toHaveBeenCalledWith( - expect.objectContaining({ - userId: 'actor', - accessMode: 'members', - credentialId: undefined, - apiKey: undefined, - }) - ) - }) - - it('refuses deactivated integrations before provisioning a credential group', async () => { - mocks.requireApproval.mockRejectedValue(new Error('Integration is deactivated')) - await expect(createApprovedSearchSource.execute({ principal, input })).rejects.toThrow( - 'deactivated' - ) - expect(mocks.resolveMembersBinding).not.toHaveBeenCalled() - expect(mocks.createConnector).not.toHaveBeenCalled() - }) - - it('refuses custom configuration outside the personal setup fields', async () => { - await expect( - createApprovedSearchSource.execute({ - principal, - input: { ...input, sourceConfig: { adminEmail: 'other-person@fixture.test' } }, - }) - ).rejects.toMatchObject({ code: 'validation' }) - expect(mocks.createConnector).not.toHaveBeenCalled() - }) - - it('refuses a knowledge base that is not the organization Search index', async () => { - knowledgeContextsMockFns.mockResolveActiveKnowledgeBaseContext.mockResolvedValue({ - organizationId: 'org', - knowledgeBaseId: 'org-index', - knowledgeBase: { isSearchIndex: false }, - }) - await expect(createApprovedSearchSource.execute({ principal, input })).rejects.toMatchObject({ - code: 'forbidden', - }) - expect(mocks.createConnector).not.toHaveBeenCalled() - }) -}) - describe('organization connector credential authorization', () => { const principal = createSessionPrincipal({ userId: 'org-admin', sessionId: 'session' }) const credential = { diff --git a/apps/sim/lib/knowledge/application/connectors.ts b/apps/sim/lib/knowledge/application/connectors.ts index c9ef81eb794..45994904047 100644 --- a/apps/sim/lib/knowledge/application/connectors.ts +++ b/apps/sim/lib/knowledge/application/connectors.ts @@ -105,19 +105,12 @@ import type { KnowledgeOperationSource, KnowledgeOrchestrationResult, } from '@/lib/knowledge/orchestration/shared' -import { requireOrganizationSearchApproval } from '@/lib/knowledge/search/integration-policy' import { escapeLikePattern } from '@/lib/knowledge/tags/utils' import { credentialProviderMatchesService, type ServiceProviderIdentity } from '@/lib/oauth' import { ServiceAccountTokenError } from '@/lib/oauth/credential-service' import { CAPABILITY_RULES, refuseCapability } from '@/lib/permission-groups/capabilities' import { resolvePermissionGroupConfig } from '@/lib/permission-groups/config-scope.server' import { getUserPermissionConfigForOrganization } from '@/lib/permission-groups/resolve.server' -import { - canConnectPersonally, - personalSourceConfigFieldIds, - withSearchSourceDefaults, -} from '@/lib/sim-search/connectors' -import { SIM_SEARCH_SYNC_INTERVAL_MINUTES } from '@/lib/sim-search/constants' import { getConnectorApiKeyConfig, isConnectorCredentialTypeAllowed } from '@/connectors/auth' import { getConnectorMeta } from '@/connectors/registry' import type { ConnectorAuthConfig } from '@/connectors/types' @@ -698,20 +691,17 @@ async function resolveConnectorApiKey( return value } -async function executeCreateKnowledgeConnector( - { - principal, - input, - context, - request, - }: { - principal: Principal - input: CreateKnowledgeConnectorInput - context: ActiveKnowledgeResourceBaseContext - request?: OrchestrationRequestContext - }, - approvedMemberSetup = false -) { +async function executeCreateKnowledgeConnector({ + principal, + input, + context, + request, +}: { + principal: Principal + input: CreateKnowledgeConnectorInput + context: ActiveKnowledgeResourceBaseContext + request?: OrchestrationRequestContext +}) { const requestId = generateRequestId() const scope = resourceScopeFromOwner(context) const owner = resourceScopeFields(scope) @@ -769,9 +759,7 @@ async function executeCreateKnowledgeConnector( if (!subjectUserId) { throw new OrchestrationError('forbidden', 'Permission-scoped access needs a signed-in admin') } - if (approvedMemberSetup && context.organizationId) { - await requireOrganizationSearchApproval(context.organizationId, input.connectorType) - } else if (context.organizationId) + if (context.organizationId) await requireOrganizationMembership( principal, context.organizationId, @@ -899,76 +887,6 @@ export const createKnowledgeConnector = defineAuthorizedKnowledgeUseCase({ }, }) -/** Members may create only a personal-account source in an approved organization index. */ -export const createApprovedSearchSource = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.createApprovedSearchSource, - resolveContext: ({ - principal, - input, - }: { - principal: Principal - input: { - knowledgeBaseId: string - assertedOrganizationId: string - connectorType: string - sourceConfig: Record - } - }) => resolveActiveKnowledgeResourceContext(input, principal), - async execute({ principal, input, context, request }) { - const meta = getConnectorMeta(input.connectorType) - if ( - !context.organizationId || - !context.knowledgeBase.isSearchIndex || - !meta || - !canConnectPersonally(meta) - ) { - throw new OrchestrationError( - 'forbidden', - 'Only approved personal Search sources may be connected' - ) - } - const allowedFields = personalSourceConfigFieldIds(meta) - if (Object.keys(input.sourceConfig).some((field) => !allowedFields.has(field))) { - throw new OrchestrationError( - 'validation', - 'Only the personal connection settings may be supplied' - ) - } - return executeCreateKnowledgeConnector( - { - principal, - context, - request, - input: { - knowledgeBaseId: input.knowledgeBaseId, - assertedOrganizationId: input.assertedOrganizationId, - connectorType: input.connectorType, - sourceConfig: withSearchSourceDefaults(meta, input.sourceConfig), - accessMode: 'members', - syncIntervalMinutes: SIM_SEARCH_SYNC_INTERVAL_MINUTES, - reuseSearchSource: true, - source: 'ui', - }, - }, - true - ) - }, - projectAudit: ({ result, context }) => - result.reused - ? [] - : { - action: AuditAction.CONNECTOR_CREATED, - resourceType: AuditResourceType.CONNECTOR, - resourceId: result.connector.id, - resourceName: result.connector.connectorType, - metadata: { - knowledgeBaseId: context.knowledgeBaseId, - accessMode: 'members', - approvedMemberSetup: true, - }, - }, -}) - /** * Deliberately not gated by `knowledge.connectors`: an update may change the * source config, sync interval or status, never the connector type. The @@ -1371,6 +1289,9 @@ export const updateKnowledgeConnectorDocuments = defineAuthorizedKnowledgeUseCas } const documentIds = [...new Set(input.documentIds)] const restoring = input.operation === 'restore' + if (restoring && !requiresConnectorIndexing(context.knowledgeBase.isSearchIndex)) { + throw new OrchestrationError('validation', 'This search index is inactive; use Sim Search.') + } const updated = await db .update(document) .set({ userExcluded: !restoring, enabled: restoring }) diff --git a/apps/sim/lib/knowledge/application/documents.ts b/apps/sim/lib/knowledge/application/documents.ts index c2de94c0d5b..6a29c50085f 100644 --- a/apps/sim/lib/knowledge/application/documents.ts +++ b/apps/sim/lib/knowledge/application/documents.ts @@ -27,6 +27,7 @@ import { resolveCanonicalActiveKnowledgeDocumentContext, } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' +import { requiresConnectorIndexing } from '@/lib/knowledge/connectors/indexing-policy' import { ALL_TAG_SLOTS, type AllTagSlot, @@ -837,6 +838,9 @@ export const updateKnowledgeDocument = defineAuthorizedKnowledgeUseCase({ const updates: KnowledgeDocumentUpdates = input.updates ? { ...input.updates } : { filename: input.filename, enabled: input.enabled } + if (updates.enabled && !requiresConnectorIndexing(context.knowledgeBase.isSearchIndex)) { + throw new OrchestrationError('validation', 'This search index is inactive; use Sim Search.') + } if (input.tagValues !== undefined) { Object.assign( updates, @@ -889,6 +893,12 @@ export const bulkUpdateKnowledgeDocuments = defineAuthorizedKnowledgeUseCase({ input: BulkKnowledgeDocumentsInput }) => resolveActiveKnowledgeResourceContext(input, principal), async execute({ input, context }) { + if ( + input.operation === 'enable' && + !requiresConnectorIndexing(context.knowledgeBase.isSearchIndex) + ) { + throw new OrchestrationError('validation', 'This search index is inactive; use Sim Search.') + } const result = input.selectAll ? await bulkDocumentOperationByFilter( context.knowledgeBaseId, diff --git a/apps/sim/lib/knowledge/application/operations.test.ts b/apps/sim/lib/knowledge/application/operations.test.ts index 0f91cbc3c8d..1f700e47137 100644 --- a/apps/sim/lib/knowledge/application/operations.test.ts +++ b/apps/sim/lib/knowledge/application/operations.test.ts @@ -42,26 +42,11 @@ describe('knowledge operation registry', () => { knowledgeOperations.search, knowledgeOperations.readSearchIndex, knowledgeOperations.enrollConnectorMember, - knowledgeOperations.simSearchConnect, - knowledgeOperations.listPersonalSourceSetupAccounts, - knowledgeOperations.personalSourceSetup, ]) { expect(operation.organizationOperation.minimumRole).toBe('member') } }) - it('limits personal source setup to the signed-in member without delegating credential discovery', () => { - for (const operation of [ - knowledgeOperations.listPersonalSourceSetupAccounts, - knowledgeOperations.personalSourceSetup, - ]) { - expect(operation.principalKinds).toEqual(['session']) - expect(operation.organizationOperation.principalKinds).not.toContain('organization_delegated') - expect(operation.workspaceApiKey).toBe('deny') - expect(operation.capability).toBe('knowledge.use') - } - }) - it('keeps human-delegated tag, connector, and composed document operations off workspace keys', () => { const operations = [ knowledgeOperations.updateDocument, diff --git a/apps/sim/lib/knowledge/application/operations.ts b/apps/sim/lib/knowledge/application/operations.ts index c341f2cb6be..c8f1e7768d6 100644 --- a/apps/sim/lib/knowledge/application/operations.ts +++ b/apps/sim/lib/knowledge/application/operations.ts @@ -767,25 +767,6 @@ export const knowledgeOperations = { delegatedServices: ['copilot'], }) ), - readSearchSourceOverview: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.sources.overview', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session', 'delegated'], - delegatedServices: ['copilot'], - }) - ), - readSearchSourceProgress: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.sources.progress', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), listSearchIntegrations: defineKnowledgeOperation( defineWorkspaceOperation({ id: 'knowledge.search.integrations.list', @@ -796,24 +777,6 @@ export const knowledgeOperations = { }), { organizationDelegation: 'allow' } ), - readOrganizationSearchOverview: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.integrations.overview', - minimumRole: 'admin', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), - readOrganizationSearchStats: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.stats.read', - minimumRole: 'admin', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), approveSearchIntegration: defineKnowledgeOperation( defineWorkspaceOperation({ id: 'knowledge.search.integrations.approve', @@ -837,48 +800,6 @@ export const knowledgeOperations = { principalKinds: ['session'], }) ), - /** - * Connecting a Sim Search source: any reader may connect their own account. - * The first connect of a source also creates its knowledge base and - * connector, which the use case reserves for an admin and refuses to anyone - * else with the way forward (ask an admin to connect the source first). - */ - simSearchConnect: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.simSearch.connect', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), - listPersonalSourceSetupAccounts: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.personalSetup.accounts.list', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), - personalSourceSetup: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.personalSetup', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), - createApprovedSearchSource: defineKnowledgeOperation( - defineWorkspaceOperation({ - id: 'knowledge.search.sources.connectApproved', - minimumRole: 'read', - workspaceApiKey: 'deny', - capability: 'knowledge.use', - principalKinds: ['session'], - }) - ), readSearchIndex: defineKnowledgeOperation( defineWorkspaceOperation({ id: 'knowledge.search.index.read', diff --git a/apps/sim/lib/knowledge/application/organization-search-overview.test.ts b/apps/sim/lib/knowledge/application/organization-search-overview.test.ts deleted file mode 100644 index bdf895c32cb..00000000000 --- a/apps/sim/lib/knowledge/application/organization-search-overview.test.ts +++ /dev/null @@ -1,181 +0,0 @@ -import { knowledgeConnector, member, organizationSearchIntegration } from '@sim/db/schema' -import { dbChainMockFns, queueTableRows, resetDbChainMock } from '@sim/testing' -import { - createSessionPrincipal, - createWorkspaceApiKeyPrincipal, -} from '@sim/testing/factories/principal.factory' -import { - knowledgeAvailabilityMock, - knowledgeAvailabilityMockFns, -} from '@sim/testing/mocks/knowledge-availability.mock' -import { - knowledgeContextsMock, - knowledgeContextsMockFns, -} from '@sim/testing/mocks/knowledge-contexts.mock' -import { - permissionGroupsResolveMock, - permissionGroupsResolveMockFns, -} from '@sim/testing/mocks/permission-groups-resolve.mock' -import { workspaceAuthzMock } from '@sim/testing/mocks/workspace-authz.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) -vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) -vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) -vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) -vi.mock('@/lib/sim-search/connectors', () => ({ - canConnectWithDefaults: (meta: { id: string }) => ['google_drive', 'gmail'].includes(meta.id), - SEARCH_SOURCE_TYPES: [ - ['google_drive', { id: 'google_drive', mirrorsSourceAcls: true, permissionScopedListing: {} }], - ['gmail', { id: 'gmail', permissionScopedListing: {} }], - ['github', { id: 'github', permissionScopedListing: {} }], - ['gitlab', { id: 'gitlab', mirrorsSourceAcls: true }], - ], -})) - -import { readOrganizationSearchOverview } from '@/lib/knowledge/application/organization-search-overview' -import { SOURCE_CONTENT_ERROR } from '@/lib/knowledge/connectors/sync-limits' - -/** The global drizzle mock nests fragments as params; flatten one for inspection. */ -function renderFragment(fragment: unknown): { sql: string; params: unknown[] } { - if (!fragment || typeof fragment !== 'object') return { sql: '', params: [] } - if ('conditions' in fragment && Array.isArray(fragment.conditions)) { - const parts = fragment.conditions.map(renderFragment) - return { - sql: parts.map((part) => part.sql).join(' '), - params: parts.flatMap((part) => part.params), - } - } - const rendered = (fragment as { toSQL?: () => { sql: string; params: unknown[] } }).toSQL?.() - if (!rendered) return { sql: '', params: [] } - const params: unknown[] = [] - let sqlText = rendered.sql - for (const param of rendered.params) { - if (param && typeof param === 'object' && 'toSQL' in param) { - const nested = renderFragment(param) - sqlText += ` ${nested.sql}` - params.push(...nested.params) - } else params.push(param) - } - return { sql: sqlText, params } -} - -const principal = createSessionPrincipal({ userId: 'admin', sessionId: 'session' }) -const input = { organizationId: 'organization' } -const health = { - connectorType: 'google_drive', - sourceCount: 4, - pausedCount: 0, - hasError: false, - hasAccountError: false, - hasDocumentError: false, - hasPermissionError: false, - hasIndexing: false, - hasPendingSync: false, - hasWaiting: false, - hasUnstarted: false, -} - -beforeEach(() => { - resetDbChainMock() - knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue(input) - permissionGroupsResolveMockFns.mockGetUserPermissionConfigForOrganization.mockResolvedValue(null) - knowledgeAvailabilityMockFns.mockResolveKnowledgeAccessAvailability.mockResolvedValue({ - memberScoped: true, - sourceMirrored: true, - }) -}) - -describe('organization Search administration overview', () => { - it.each([ - { - hasPermissionError: true, - hasAccountError: false, - hasDocumentError: false, - issue: 'permission_sync_incomplete', - }, - { hasAccountError: true, hasDocumentError: false, issue: 'account_sync_incomplete' }, - { hasAccountError: false, hasDocumentError: true, issue: 'document_indexing_failed' }, - { hasAccountError: false, hasDocumentError: false, issue: 'sync_failed' }, - ])('identifies $issue without exposing error details', async ({ issue, ...errors }) => { - queueTableRows(member, [{ role: 'admin' }]) - queueTableRows(knowledgeConnector, [ - { ...health, ...errors, hasError: true, rawError: 'private provider response' }, - ]) - const result = await readOrganizationSearchOverview.execute({ principal, input }) - expect(result.providers[0]).toMatchObject({ status: 'needs_attention', issue }) - expect(JSON.stringify(result)).not.toContain('private provider response') - }) - it('does not count the per-document relisting marker as a member account error', async () => { - queueTableRows(member, [{ role: 'admin' }]) - queueTableRows(knowledgeConnector, [{ ...health }]) - await readOrganizationSearchOverview.execute({ principal, input }) - const rendered = dbChainMockFns.where.mock.calls.flatMap((call) => call.map(renderFragment)) - const memberErrorClause = rendered.find((fragment) => fragment.sql.includes("'suspended'")) - expect(memberErrorClause).toBeDefined() - expect(memberErrorClause?.sql).toContain('IS NOT NULL AND ? <> ?') - expect(memberErrorClause?.params).toContain(SOURCE_CONTENT_ERROR) - }) - it.each([ - { rows: [{ role: 'member' }], code: 'forbidden' }, - { rows: [], code: 'not_found' }, - ])('refuses unauthorized reads with $code before health queries', async ({ rows, code }) => { - queueTableRows(member, rows) - await expect( - readOrganizationSearchOverview.execute({ principal, input }) - ).rejects.toMatchObject({ code }) - expect(dbChainMockFns.from).not.toHaveBeenCalledWith(knowledgeConnector) - expect(dbChainMockFns.from).not.toHaveBeenCalledWith(organizationSearchIntegration) - }) - it('rejects a workspace key before canonical loading', async () => { - await expect( - readOrganizationSearchOverview.execute({ - principal: createWorkspaceApiKeyPrincipal({ workspaceId: 'workspace', keyId: 'key' }), - input, - }) - ).rejects.toThrow() - expect(knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext).not.toHaveBeenCalled() - }) - it('does not mistake infrastructure failure for an empty integration list', async () => { - queueTableRows(member, [{ role: 'admin' }]) - permissionGroupsResolveMockFns.mockGetUserPermissionConfigForOrganization.mockRejectedValue( - new Error('Database unavailable') - ) - await expect(readOrganizationSearchOverview.execute({ principal, input })).rejects.toThrow( - 'Database unavailable' - ) - }) - it('does not report indexing when the owner-scoped Search gate is disabled', async () => { - queueTableRows(member, [{ role: 'admin' }]) - queueTableRows(knowledgeConnector, [{ ...health, hasIndexing: true }]) - queueTableRows(organizationSearchIntegration, [{ connectorType: 'gmail', approved: true }]) - knowledgeAvailabilityMockFns.mockResolveKnowledgeAccessAvailability.mockResolvedValue({ - memberScoped: false, - sourceMirrored: false, - }) - const result = await readOrganizationSearchOverview.execute({ principal, input }) - expect( - knowledgeAvailabilityMockFns.mockResolveKnowledgeAccessAvailability - ).toHaveBeenCalledWith(input) - expect(result.providers).toEqual([ - { - connectorType: 'google_drive', - sourceCount: 4, - approved: true, - status: 'paused', - issue: null, - isSyncing: false, - hasPendingSync: false, - }, - { - connectorType: 'gmail', - sourceCount: 0, - approved: true, - status: 'paused', - issue: null, - isSyncing: false, - hasPendingSync: false, - }, - ]) - }) -}) diff --git a/apps/sim/lib/knowledge/application/organization-search-overview.ts b/apps/sim/lib/knowledge/application/organization-search-overview.ts deleted file mode 100644 index f1d0b114158..00000000000 --- a/apps/sim/lib/knowledge/application/organization-search-overview.ts +++ /dev/null @@ -1,307 +0,0 @@ -import { db } from '@sim/db' -import { - document, - knowledgeBase, - knowledgeConnector, - knowledgeConnectorMember, - knowledgeConnectorMemberSyncLog, - knowledgeConnectorSyncLog, - organizationSearchIntegration, -} from '@sim/db/schema' -import { and, eq, exists, inArray, isNotNull, isNull, type SQL, sql } from 'drizzle-orm' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' -import { sourceAclFreshnessCutoff } from '@/lib/knowledge/access/predicate' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { - SOURCE_CONTENT_ERROR, - SOURCE_PERMISSION_ERROR, -} from '@/lib/knowledge/connectors/sync-limits' -import { MAX_SEARCH_SOURCE_PROVIDER_TYPES } from '@/lib/knowledge/constants' -import { failedDocumentCondition } from '@/lib/knowledge/documents/processing-status' -import { canConnectWithDefaults, SEARCH_SOURCE_TYPES } from '@/lib/sim-search/connectors' - -interface OrganizationSearchOverviewInput { - organizationId: string -} - -interface ProviderHealth { - sourceCount: number - pausedCount: number - hasError: boolean - hasAccountError: boolean - hasDocumentError: boolean - hasPermissionError: boolean - hasIndexing: boolean - hasPendingSync: boolean - hasWaiting: boolean - hasUnstarted: boolean -} - -/** A successful empty crawl is active; neither this state nor its count describes readable documents. */ -function organizationSearchProviderStatus( - health: ProviderHealth | undefined, - approved: boolean, - automaticSetup: boolean, - available: boolean -) { - if (!approved || !available) return 'paused' as const - if (!health?.sourceCount) - return automaticSetup ? ('waiting_for_connections' as const) : ('needs_setup' as const) - if (health.pausedCount === health.sourceCount) return 'paused' as const - if (health.hasError) return 'needs_attention' as const - if (health.hasIndexing) return 'indexing' as const - if (health.hasWaiting) return 'waiting_for_connections' as const - if (health.hasUnstarted) return 'needs_setup' as const - return 'active' as const -} - -/** Organization admins receive aggregate operational facts, never documents, account identities or raw errors. */ -export const readOrganizationSearchOverview = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.readOrganizationSearchOverview, - resolveContext: ({ input }: { input: OrganizationSearchOverviewInput }) => - resolveKnowledgeOwnerContext({ organizationId: input.organizationId }), - async execute({ context }) { - if (!context.organizationId) - throw new OrchestrationError('validation', 'Organization is required') - const providerTypes = SEARCH_SOURCE_TYPES.map(([connectorType]) => connectorType) - if (providerTypes.length > MAX_SEARCH_SOURCE_PROVIDER_TYPES) { - throw new Error('Search provider catalog exceeds the overview bound') - } - const availability = await resolveKnowledgeAccessAvailability(context) - const identityRequiredTypes = SEARCH_SOURCE_TYPES.filter( - ([, meta]) => meta.requiresMemberIdentity - ).map(([connectorType]) => connectorType) - - const hasActiveMembers = exists( - db - .select({ id: knowledgeConnectorMember.id }) - .from(knowledgeConnectorMember) - .where( - and( - eq(knowledgeConnectorMember.connectorId, knowledgeConnector.id), - eq(knowledgeConnectorMember.status, 'active') - ) - ) - ) - const hasMemberContinuation = exists( - db - .select({ id: knowledgeConnectorMember.id }) - .from(knowledgeConnectorMember) - .where( - and( - eq(knowledgeConnectorMember.connectorId, knowledgeConnector.id), - eq(knowledgeConnectorMember.status, 'active'), - isNotNull(knowledgeConnectorMember.listingCheckpoint) - ) - ) - ) - const continuing = sql`( - ${knowledgeConnector.listingCheckpoint} IS NOT NULL - OR (${knowledgeConnector.accessMode} = 'members' AND ( - ${knowledgeConnector.directoryCheckpoint} IS NOT NULL - OR ${hasMemberContinuation} - OR coalesce(${knowledgeConnector.nextMemberSyncAt} <= statement_timestamp(), false) - )) - )` - const cutoff = sourceAclFreshnessCutoff() - /** - * A member whose last run only had per-document content failures carries - * {@link SOURCE_CONTENT_ERROR} as a marker so its next run lists fully; the - * member itself is healthy and the documents are reported separately. - */ - const hasMemberError = exists( - db - .select({ id: knowledgeConnectorMember.id }) - .from(knowledgeConnectorMember) - .where( - and( - eq(knowledgeConnectorMember.connectorId, knowledgeConnector.id), - sql`( - ${knowledgeConnectorMember.status} = 'suspended' - OR (${knowledgeConnectorMember.status} = 'active' AND ( - (${knowledgeConnectorMember.lastError} IS NOT NULL AND ${knowledgeConnectorMember.lastError} <> ${SOURCE_CONTENT_ERROR}) - OR ${knowledgeConnectorMember.consecutiveFailures} > 0 - OR coalesce(${knowledgeConnectorMember.memberSyncedThrough}, ${knowledgeConnectorMember.lastCompleteListingAt}, ${knowledgeConnectorMember.createdAt}) < ${cutoff} - )) - )` - ) - ) - ) - const hasMemberFirstListing = exists( - db - .select({ id: knowledgeConnectorMember.id }) - .from(knowledgeConnectorMember) - .where( - and( - eq(knowledgeConnectorMember.connectorId, knowledgeConnector.id), - eq(knowledgeConnectorMember.status, 'active'), - isNull(knowledgeConnectorMember.lastCompleteListingAt) - ) - ) - ) - const hasDocumentsInState = (condition: SQL) => - exists( - db - .select({ id: document.id }) - .from(document) - .where( - and( - eq(document.connectorId, knowledgeConnector.id), - eq(document.knowledgeBaseId, knowledgeConnector.knowledgeBaseId), - eq(document.enabled, true), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - condition - ) - ) - ) - /** Partial runs can be normal continuations; completed timestamps alone do not prove a complete crawl. */ - const latestMemberRunHasError = sql`coalesce(( - SELECT ${knowledgeConnectorMemberSyncLog.status} = 'failed' - OR (${knowledgeConnectorMemberSyncLog.status} = 'partial' AND ( - ${knowledgeConnectorMemberSyncLog.membersFailed} > 0 - OR ${knowledgeConnectorMemberSyncLog.docsFailed} > 0 - OR ${knowledgeConnectorMemberSyncLog.processingDispatchFailed} > 0 - OR NOT ${continuing} - )) - FROM ${knowledgeConnectorMemberSyncLog} - WHERE ${knowledgeConnectorMemberSyncLog.connectorId} = ${knowledgeConnector.id} - AND ${knowledgeConnectorMemberSyncLog.status} <> 'started' - ORDER BY ${knowledgeConnectorMemberSyncLog.startedAt} DESC, ${knowledgeConnectorMemberSyncLog.id} DESC - LIMIT 1 - ), false)` - const latestCentralRunHasError = sql`coalesce(( - SELECT ${knowledgeConnectorSyncLog.status} = 'failed' - OR (${knowledgeConnectorSyncLog.status} = 'partial' AND ( - ${knowledgeConnectorSyncLog.docsFailed} > 0 OR NOT ${continuing} - )) - FROM ${knowledgeConnectorSyncLog} - WHERE ${knowledgeConnectorSyncLog.connectorId} = ${knowledgeConnector.id} - AND ${knowledgeConnectorSyncLog.status} <> 'started' - ORDER BY ${knowledgeConnectorSyncLog.startedAt} DESC, ${knowledgeConnectorSyncLog.id} DESC - LIMIT 1 - ), false)` - const paused = sql`( - ${knowledgeConnector.status} IN ('paused', 'disabled') - OR (${knowledgeConnector.accessMode} = 'members' AND ${knowledgeConnector.memberSyncStatus} = 'disabled') - OR (${knowledgeConnector.accessMode} = 'members' AND ${!availability.memberScoped}) - OR (${knowledgeConnector.accessMode} = 'admin' AND ( - ${!availability.sourceMirrored} - OR (${!availability.memberScoped} AND ${inArray(knowledgeConnector.connectorType, identityRequiredTypes)}) - )) - )` - const [health, decisions] = await Promise.all([ - db - .select({ - connectorType: knowledgeConnector.connectorType, - sourceCount: sql`count(*)::int`, - pausedCount: sql`count(*) FILTER (WHERE ${paused})::int`, - hasError: sql`bool_or(NOT ${paused} AND ( - ${knowledgeConnector.status} = 'error' - OR ${knowledgeConnector.lastSyncError} IS NOT NULL - OR ${hasDocumentsInState(failedDocumentCondition())} - OR (${knowledgeConnector.accessMode} = 'admin' AND ${latestCentralRunHasError}) - OR (${knowledgeConnector.accessMode} = 'members' AND ( - ${knowledgeConnector.memberSyncStatus} = 'error' - OR ${knowledgeConnector.lastMemberSyncError} IS NOT NULL - OR ${hasMemberError} OR ${latestMemberRunHasError} - )) - ))`, - hasAccountError: sql`bool_or(NOT ${paused} AND ${knowledgeConnector.accessMode} = 'members' AND ${hasMemberError})`, - hasDocumentError: sql`bool_or(NOT ${paused} AND ${hasDocumentsInState(failedDocumentCondition())})`, - hasPermissionError: sql`bool_or(NOT ${paused} AND ${SOURCE_PERMISSION_ERROR} = ANY(string_to_array(${knowledgeConnector.lastSyncError}, ${'\n'})))`, - hasIndexing: sql`bool_or(NOT ${paused} - AND (${knowledgeConnector.accessMode} <> 'members' OR ${hasActiveMembers} OR ${knowledgeConnector.credentialId} IS NOT NULL) - AND ( - ${knowledgeConnector.status} IN ('pending', 'syncing') - OR ${hasDocumentsInState(inArray(document.processingStatus, ['pending', 'processing']))} - OR (${knowledgeConnector.accessMode} = 'members' AND ( - ${knowledgeConnector.memberSyncStatus} IN ('pending', 'running') - )) - ))`, - hasWaiting: sql`bool_or(NOT ${paused} AND ${knowledgeConnector.accessMode} = 'members' AND NOT ${hasActiveMembers})`, - hasPendingSync: sql`bool_or(NOT ${paused} AND ( - ${continuing} OR (${knowledgeConnector.accessMode} = 'members' AND ${hasMemberFirstListing}) - ))`, - hasUnstarted: sql`bool_or(NOT ${paused} AND ( - (${knowledgeConnector.accessMode} = 'admin' AND ${knowledgeConnector.lastSyncAt} IS NULL) - OR (${knowledgeConnector.accessMode} = 'members' AND ${hasMemberFirstListing}) - ))`, - }) - .from(knowledgeConnector) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) - .where( - and( - eq(knowledgeBase.organizationId, context.organizationId), - eq(knowledgeBase.isSearchIndex, true), - isNull(knowledgeBase.deletedAt), - inArray(knowledgeConnector.connectorType, providerTypes), - inArray(knowledgeConnector.accessMode, ['admin', 'members']), - isNull(knowledgeConnector.archivedAt), - isNull(knowledgeConnector.deletedAt) - ) - ) - .groupBy(knowledgeConnector.connectorType) - .limit(MAX_SEARCH_SOURCE_PROVIDER_TYPES), - db - .select({ - connectorType: organizationSearchIntegration.connectorType, - approved: organizationSearchIntegration.approved, - }) - .from(organizationSearchIntegration) - .where( - and( - eq(organizationSearchIntegration.organizationId, context.organizationId), - inArray(organizationSearchIntegration.connectorType, providerTypes) - ) - ) - .limit(MAX_SEARCH_SOURCE_PROVIDER_TYPES), - ]) - const healthByType = new Map(health.map((provider) => [provider.connectorType, provider])) - const approvals = new Map( - decisions.map((decision) => [decision.connectorType, decision.approved]) - ) - return { - providers: SEARCH_SOURCE_TYPES.flatMap(([connectorType, meta]) => { - const state = healthByType.get(connectorType) - if (!state && !approvals.has(connectorType)) return [] - const approved = approvals.get(connectorType) ?? Boolean(state?.sourceCount) - const status = organizationSearchProviderStatus( - state, - approved, - canConnectWithDefaults(meta) && availability.memberScoped, - Boolean( - (meta.permissionScopedListing && availability.memberScoped) || - (meta.mirrorsSourceAcls && - availability.sourceMirrored && - (!meta.requiresMemberIdentity || availability.memberScoped)) - ) - ) - return [ - { - connectorType, - approved, - sourceCount: state?.sourceCount ?? 0, - status, - issue: - status === 'needs_attention' - ? state?.hasAccountError - ? ('account_sync_incomplete' as const) - : state?.hasPermissionError - ? ('permission_sync_incomplete' as const) - : state?.hasDocumentError - ? ('document_indexing_failed' as const) - : ('sync_failed' as const) - : null, - isSyncing: status !== 'paused' && Boolean(state?.hasIndexing), - hasPendingSync: status !== 'paused' && Boolean(state?.hasPendingSync), - }, - ] - }), - } - }, -}) diff --git a/apps/sim/lib/knowledge/application/organization-search-stats.test.ts b/apps/sim/lib/knowledge/application/organization-search-stats.test.ts deleted file mode 100644 index 050cf7e7bf3..00000000000 --- a/apps/sim/lib/knowledge/application/organization-search-stats.test.ts +++ /dev/null @@ -1,94 +0,0 @@ -import { member } from '@sim/db/schema' -import { queueTableRows, resetDbChainMock } from '@sim/testing' -import { - createPersonalApiKeyPrincipal, - createSessionPrincipal, -} from '@sim/testing/factories/principal.factory' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { - knowledgeAvailabilityMock, - knowledgeAvailabilityMockFns, -} from '@sim/testing/mocks/knowledge-availability.mock' -import { - knowledgeContextsMock, - knowledgeContextsMockFns, -} from '@sim/testing/mocks/knowledge-contexts.mock' -import { - permissionGroupsResolveMock, - permissionGroupsResolveMockFns, -} from '@sim/testing/mocks/permission-groups-resolve.mock' -import { workspaceAuthzMock } from '@sim/testing/mocks/workspace-authz.mock' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const mocks = vi.hoisted(() => ({ - load: vi.fn(), -})) -vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) -vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) -vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) -vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) -vi.mock('@/lib/knowledge/search/activity-stats', () => ({ - loadOrganizationSearchStats: mocks.load, -})) - -import { readOrganizationSearchStats } from '@/lib/knowledge/application/organization-search-stats' -import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' - -const principal = createSessionPrincipal({ userId: 'admin', sessionId: 'session' }) -const input = { organizationId: 'organization', period: '7d', surface: 'mcp' } as const - -afterEach(resetEnvFlagsMock) - -beforeEach(() => { - resetDbChainMock() - /** The Stats tab reports indexed organization search, so these cases run with it on. */ - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue({ - organizationId: 'organization', - }) - permissionGroupsResolveMockFns.mockGetUserPermissionConfigForOrganization.mockResolvedValue(null) - knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable.mockResolvedValue(undefined) - mocks.load.mockResolvedValue({ totals: { invocations: 3 } }) -}) - -describe('organization Search stats authorization', () => { - it.each([ - { rows: [{ role: 'member' }], code: 'forbidden' }, - { rows: [], code: 'not_found' }, - ])('rejects unauthorized access with $code before aggregation', async ({ rows, code }) => { - queueTableRows(member, rows) - await expect(readOrganizationSearchStats.execute({ principal, input })).rejects.toMatchObject({ - code, - }) - expect(mocks.load).not.toHaveBeenCalled() - }) - it('rejects API keys before protected loading', async () => { - await expect( - readOrganizationSearchStats.execute({ - principal: createPersonalApiKeyPrincipal({ userId: 'admin', keyId: 'key' }), - input, - }) - ).rejects.toThrow() - expect(knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext).not.toHaveBeenCalled() - expect(mocks.load).not.toHaveBeenCalled() - }) - it('refuses while indexed organization search is dormant, before aggregation', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) - queueTableRows(member, [{ role: 'admin' }]) - await expect(readOrganizationSearchStats.execute({ principal, input })).rejects.toBeInstanceOf( - SearchIndexDormantError - ) - expect(mocks.load).not.toHaveBeenCalled() - }) - - it('fails closed when Search is disabled', async () => { - queueTableRows(member, [{ role: 'admin' }]) - knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable.mockRejectedValueOnce( - new Error('Search is disabled') - ) - await expect(readOrganizationSearchStats.execute({ principal, input })).rejects.toThrow( - 'Search is disabled' - ) - expect(mocks.load).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/knowledge/application/organization-search-stats.ts b/apps/sim/lib/knowledge/application/organization-search-stats.ts deleted file mode 100644 index 6b79404d65b..00000000000 --- a/apps/sim/lib/knowledge/application/organization-search-stats.ts +++ /dev/null @@ -1,24 +0,0 @@ -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { requireOrganizationSearchAvailable } from '@/lib/knowledge/access/availability' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { - loadOrganizationSearchStats, - type SearchStatsInput, -} from '@/lib/knowledge/search/activity-stats' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' - -/** The indexed Stats tab's activity report; refused while indexed organization search is dormant. */ -export const readOrganizationSearchStats = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.readOrganizationSearchStats, - resolveContext: ({ input }: { input: SearchStatsInput }) => - resolveKnowledgeOwnerContext({ organizationId: input.organizationId }), - async execute({ context, input }) { - assertIndexedOrgSearchEnabled() - if (!context.organizationId) - throw new OrchestrationError('validation', 'Organization is required') - await requireOrganizationSearchAvailable(context.organizationId) - return loadOrganizationSearchStats({ ...input, organizationId: context.organizationId }) - }, -}) diff --git a/apps/sim/lib/knowledge/application/personal-search-account.test.ts b/apps/sim/lib/knowledge/application/personal-search-account.test.ts deleted file mode 100644 index cfbcd2aa52d..00000000000 --- a/apps/sim/lib/knowledge/application/personal-search-account.test.ts +++ /dev/null @@ -1,125 +0,0 @@ -import { user } from '@sim/db/schema' -import { dbChainMockFns, queueTableRows, resetDbChainMock } from '@sim/testing' -import { - createPersonalApiKeyPrincipal, - createSessionPrincipal, -} from '@sim/testing/factories/principal.factory' -import { - integrationsAvailabilityMock, - integrationsAvailabilityMockFns, -} from '@sim/testing/mocks/integrations-availability.mock' -import { knowledgeAvailabilityMock } from '@sim/testing/mocks/knowledge-availability.mock' -import { - knowledgeSearchIntegrationPolicyMock, - knowledgeSearchIntegrationPolicyMockFns, -} from '@sim/testing/mocks/knowledge-search-integration-policy.mock' -import { - organizationAuthorizationMock, - organizationAuthorizationMockFns, -} from '@sim/testing/mocks/organization-authorization.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -const hoisted = vi.hoisted(() => ({ - accounts: vi.fn(), -})) -vi.mock('@/lib/core/application/organization-authorization', () => organizationAuthorizationMock) -vi.mock('@/lib/knowledge/search/integration-policy', () => knowledgeSearchIntegrationPolicyMock) -vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) -vi.mock('@/lib/integrations/availability.server', () => integrationsAvailabilityMock) -vi.mock('@/lib/credentials/organization-managed', () => ({ - getOwnOrganizationManagedOAuthCredentials: hoisted.accounts, -})) - -import { - authorizePersonalSearchSetup, - authorizePersonalSearchSetupCredential, -} from '@/lib/knowledge/application/personal-search-account' - -const mocks = { - ...hoisted, - approval: knowledgeSearchIntegrationPolicyMockFns.mockRequireOrganizationSearchApproval, - deployed: integrationsAvailabilityMockFns.mockIsOAuthServiceDeploymentAvailable, -} - -const principal = createSessionPrincipal({ userId: 'member-1' }) -const input = { - organizationId: 'organization-1', - connectorType: 'jira', - credentialId: 'own-account', -} as const - -describe('personal Search setup authorization', () => { - beforeEach(() => { - vi.resetAllMocks() - resetDbChainMock() - queueTableRows(user, [{ emailVerified: true }]) - mocks.deployed.mockReturnValue(true) - mocks.accounts.mockResolvedValue([ - { - id: 'own-account', - providerId: 'jira', - displayName: 'Work account', - scopes: ['read:jira-work'], - }, - ]) - }) - - it('permits a verified member and scopes the account lookup to that person and provider', async () => { - await expect(authorizePersonalSearchSetupCredential(principal, input)).resolves.toMatchObject({ - id: 'own-account', - }) - expect(organizationAuthorizationMockFns.mockRequireOrganizationMembership).toHaveBeenCalledWith( - principal, - input.organizationId, - 'member', - 'knowledge.use' - ) - expect(mocks.approval).toHaveBeenCalledWith(input.organizationId, 'jira') - expect(mocks.accounts).toHaveBeenCalledWith({ - organizationId: input.organizationId, - userId: principal.userId, - providerId: 'jira', - credentialId: input.credentialId, - }) - }) - - it('rejects non-session callers before membership or protected account reads', async () => { - await expect( - authorizePersonalSearchSetup(createPersonalApiKeyPrincipal({ userId: 'member-1' }), input) - ).rejects.toThrow('Sign in') - expect( - organizationAuthorizationMockFns.mockRequireOrganizationMembership - ).not.toHaveBeenCalled() - expect(dbChainMockFns.select).not.toHaveBeenCalled() - }) - - it('rejects non-members before looking up credentials', async () => { - organizationAuthorizationMockFns.mockRequireOrganizationMembership.mockRejectedValue( - new Error('Membership ended') - ) - await expect(authorizePersonalSearchSetupCredential(principal, input)).rejects.toThrow( - 'Membership ended' - ) - expect(mocks.accounts).not.toHaveBeenCalled() - }) - - it('requires a verified email before approving or provisioning setup', async () => { - resetDbChainMock() - queueTableRows(user, [{ emailVerified: false }]) - await expect(authorizePersonalSearchSetup(principal, input)).rejects.toThrow( - 'Verify your email' - ) - expect(mocks.approval).not.toHaveBeenCalled() - }) - - it.each([ - { accounts: [] }, - { accounts: [{ id: 'another-account', providerId: 'jira' }] }, - { accounts: [{ id: 'own-account', providerId: 'confluence' }] }, - ])('rejects revoked, foreign or mismatched grants %#', async ({ accounts }) => { - mocks.accounts.mockResolvedValue(accounts) - await expect(authorizePersonalSearchSetupCredential(principal, input)).rejects.toThrow( - 'Connect your account again' - ) - }) -}) diff --git a/apps/sim/lib/knowledge/application/personal-search-account.ts b/apps/sim/lib/knowledge/application/personal-search-account.ts deleted file mode 100644 index 04a25443acb..00000000000 --- a/apps/sim/lib/knowledge/application/personal-search-account.ts +++ /dev/null @@ -1,68 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import { db } from '@sim/db' -import { user } from '@sim/db/schema' -import { eq } from 'drizzle-orm' -import { requireOrganizationMembership } from '@/lib/core/application/organization-authorization' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { getOwnOrganizationManagedOAuthCredentials } from '@/lib/credentials/organization-managed' -import { isOAuthServiceDeploymentAvailable } from '@/lib/integrations/availability.server' -import { requireKnowledgeMemberAccessAvailable } from '@/lib/knowledge/access/availability' -import { requireOrganizationSearchApproval } from '@/lib/knowledge/search/integration-policy' - -export type PersonalSearchSetupConnector = 'jira' | 'confluence' - -/** Rechecks the caller's organization and approved integration before personal account discovery. */ -export async function authorizePersonalSearchSetup( - principal: Principal, - input: { organizationId: string; connectorType: PersonalSearchSetupConnector } -) { - if ( - principal.kind !== 'session' || - (input.connectorType !== 'jira' && input.connectorType !== 'confluence') - ) { - throw new OrchestrationError('forbidden', 'Sign in to connect this Search source') - } - await requireOrganizationMembership(principal, input.organizationId, 'member', 'knowledge.use') - const [viewer] = await db - .select({ emailVerified: user.emailVerified }) - .from(user) - .where(eq(user.id, principal.userId)) - .limit(1) - if (!viewer?.emailVerified) { - throw new OrchestrationError( - 'validation', - 'Verify your email address before connecting an account' - ) - } - await requireOrganizationSearchApproval(input.organizationId, input.connectorType) - await requireKnowledgeMemberAccessAvailable({ organizationId: input.organizationId }) - if (!isOAuthServiceDeploymentAvailable(input.connectorType)) { - throw new OrchestrationError('validation', 'This account connection is unavailable') - } - return principal.userId -} - -/** Personal setup can browse only a currently live grant owned by the signed-in member. */ -export async function authorizePersonalSearchSetupCredential( - principal: Principal, - input: { - organizationId: string - connectorType: PersonalSearchSetupConnector - credentialId: string - } -) { - const userId = await authorizePersonalSearchSetup(principal, input) - const accounts = await getOwnOrganizationManagedOAuthCredentials({ - organizationId: input.organizationId, - userId, - providerId: input.connectorType, - credentialId: input.credentialId, - }) - const account = accounts.find( - (entry) => entry.id === input.credentialId && entry.providerId === input.connectorType - ) - if (!account) { - throw new OrchestrationError('not_found', 'Connect your account again before choosing sources') - } - return account -} diff --git a/apps/sim/lib/knowledge/application/personal-search-integration-pages.ts b/apps/sim/lib/knowledge/application/personal-search-integration-pages.ts deleted file mode 100644 index 709ee5f81b2..00000000000 --- a/apps/sim/lib/knowledge/application/personal-search-integration-pages.ts +++ /dev/null @@ -1,49 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import { - type ListPersonalSearchIntegrationsInput, - listPersonalSearchIntegrations, -} from '@/lib/knowledge/application/personal-search-integrations' - -/** One page of the viewer's personal Search integrations. */ -export type PersonalSearchIntegrationsPage = Awaited< - ReturnType -> - -/** The most pages one walk of the personal inventory reads before it stops. */ -const MAX_PERSONAL_SEARCH_INTEGRATION_PAGES = 100 - -/** - * Every page of the viewer's personal Search integrations, in order: Live Search answers in one - * page, indexed search in one per batch of sources. A cursor that repeats, or a walk past - * {@link MAX_PERSONAL_SEARCH_INTEGRATION_PAGES}, is a defect and throws rather than answering from - * part of the inventory. - */ -export async function* personalSearchIntegrationPages({ - principal, - input, - signal, -}: { - principal: Principal - input: Omit - signal?: AbortSignal -}): AsyncGenerator { - const seen = new Set() - let cursor: string | undefined - for (let page = 0; page < MAX_PERSONAL_SEARCH_INTEGRATION_PAGES; page++) { - signal?.throwIfAborted() - const inventory = await listPersonalSearchIntegrations.execute({ - principal, - input: { ...input, ...(cursor ? { cursor } : {}) }, - }) - signal?.throwIfAborted() - yield inventory - if (inventory.nextCursor === null) return - if (seen.has(inventory.nextCursor)) - throw new Error('Personal Search integration pagination did not advance') - seen.add(inventory.nextCursor) - cursor = inventory.nextCursor - } - throw new Error( - `Personal Search integration pagination exceeded ${MAX_PERSONAL_SEARCH_INTEGRATION_PAGES} pages` - ) -} diff --git a/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts b/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts index aeaf4296001..92bc535c87b 100644 --- a/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts +++ b/apps/sim/lib/knowledge/application/personal-search-integrations.test.ts @@ -1,5 +1,5 @@ import { user } from '@sim/db/schema' -import { queueTableRows, resetDbChainMock, resetEnvFlagsMock, setEnvFlags } from '@sim/testing' +import { queueTableRows, resetDbChainMock, resetEnvFlagsMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' import { credentialGroupsAvailabilityMock, @@ -127,108 +127,6 @@ beforeEach(() => { ) mockGetConnectorAccessAvailability.mockReturnValue({ members: true }) }) -describe('personal Search inventory', () => { - beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) - - it.each([ - [{}, 'connected', 'not_indexed'], - [{ isSyncing: true }, 'connected', 'indexing'], - [{ hasViewerDocuments: true }, 'connected', 'indexed'], - [{ hasSyncError: true }, 'connected', 'sync_failed'], - [ - { - viewerAccounts: [{ credentialId: 'mine', displayName: 'My mail', status: 'needs_reauth' }], - }, - 'reconnect_needed', - 'not_indexed', - ], - ])( - 'separates account state from index state %#', - async (changes, connectionStatus, indexingStatus) => { - m.sources.mockResolvedValue({ sources: [{ ...source, ...changes }], nextCursor: null }) - const result = await listPersonalSearchIntegrations.execute({ principal, input }) - expect(result.connections[0]).toMatchObject({ connectionStatus, indexingStatus }) - expect(personalSearchIntegrationPageSchema.safeParse(result).success).toBe(true) - expect(m.sources).toHaveBeenCalledWith({ principal, input }) - expect(m.configuredTypes).toHaveBeenCalledWith({ organizationId: 'org' }) - expect(JSON.stringify(result)).not.toMatch(/accessToken|authorizationUrl|other-person/) - } - ) - it('offers approved ready providers with no index, excluding app setup and unapproved providers', async () => { - m.sources.mockResolvedValue({ sources: [], nextCursor: null }) - m.configuredTypes.mockResolvedValue([]) - const result = await listPersonalSearchIntegrations.execute({ principal, input }) - expect(result.available.map((entry) => entry.target.connectorType)).toEqual(['gmail']) - expect(result.connections).toEqual([]) - }) - it('omits every connection action when the person has an unverified email', async () => { - resetDbChainMock() - queueTableRows(user, [{ emailVerified: false }]) - m.sources.mockResolvedValue({ sources: [], nextCursor: null }) - m.configuredTypes.mockResolvedValue([]) - expect((await listPersonalSearchIntegrations.execute({ principal, input })).available).toEqual( - [] - ) - }) - it('forwards the bounded cursor and provider filter without exposing other people’s accounts', async () => { - m.sources.mockResolvedValue({ - sources: [{ ...source, viewerAccounts: [], viewerMembership: 'not_enrolled' }], - nextCursor: 'next', - }) - const filtered = { ...input, connectorType: 'gmail', cursor: 'page' } - const result = await listPersonalSearchIntegrations.execute({ principal, input: filtered }) - expect(result.connections).toEqual([]) - expect(result.nextCursor).toBe('next') - expect(result.available).toEqual([{ name: 'gmail', description: '', target }]) - expect(m.sources).toHaveBeenCalledWith({ principal, input: filtered }) - }) - it('rechecks authorization before every read and returns nothing after membership is revoked', async () => { - organizationAuthorizationMockFns.mockAuthorizeOrganizationOperation.mockRejectedValue( - new Error('Membership revoked') - ) - await expect(listPersonalSearchIntegrations.execute({ principal, input })).rejects.toThrow( - 'Membership revoked' - ) - expect(m.sources).not.toHaveBeenCalled() - }) - it.each([ - { ...target, credentialId: 'another-person' }, - { ...target, connectorId: 'other-source' }, - { ...target, provider: 'notion' }, - ])('rejects a forged or stale reconnect target', async (forged) => { - m.sources.mockResolvedValue({ - sources: [ - { - ...source, - viewerAccounts: [ - { credentialId: 'mine', displayName: 'My mail', status: 'needs_reauth' }, - ], - }, - ], - nextCursor: null, - }) - await expect( - resolvePersonalSearchConnection.execute({ principal, input: { ...input, target: forged } }) - ).rejects.toThrow('no longer available') - }) - it('returns the precise owned reconnect target', async () => { - m.sources.mockResolvedValue({ - sources: [ - { - ...source, - viewerAccounts: [ - { credentialId: 'mine', displayName: 'My mail', status: 'needs_reauth' }, - ], - }, - ], - nextCursor: null, - }) - const selected = { ...target, credentialId: 'mine' } - await expect( - resolvePersonalSearchConnection.execute({ principal, input: { ...input, target: selected } }) - ).resolves.toEqual({ name: 'gmail', target: selected }) - }) -}) describe('live Search connection controls', () => { const liveTarget = { @@ -238,7 +136,6 @@ describe('live Search connection controls', () => { connectionMode: 'live', optionId: 'slack-option', } as const - beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: true })) it('offers Slack without a knowledge base or indexed connector and round-trips the response', async () => { const result = await listPersonalSearchIntegrations.execute({ principal, input }) @@ -298,7 +195,6 @@ describe('live Search connection controls', () => { ...liveTarget, credentialId: 'mine', }) - expect(result.connections[0].indexingStatus).toBeUndefined() expect(personalSearchIntegrationPageSchema.safeParse(result).success).toBe(true) }) @@ -306,7 +202,6 @@ describe('live Search connection controls', () => { { ...liveTarget, credentialId: 'another-person' }, { ...liveTarget, optionId: 'another-option' }, { ...liveTarget, provider: 'gmail' }, - { ...liveTarget, connectionMode: undefined, optionId: undefined }, ])('rejects a forged or stale live target: %j', async (target) => { await expect( resolvePersonalSearchConnection.execute({ principal, input: { ...input, target } }) @@ -367,14 +262,4 @@ describe('live Search connection controls', () => { completionId: 'attempt', }) }) - - it('rejects an indexed source target after switching to live search', async () => { - await expect( - listPersonalSearchIntegrations.execute({ - principal, - input: { ...input, connectorId: 'stale-source' }, - }) - ).rejects.toThrow('Refresh your live account connections') - expect(m.group).not.toHaveBeenCalled() - }) }) diff --git a/apps/sim/lib/knowledge/application/personal-search-integrations.ts b/apps/sim/lib/knowledge/application/personal-search-integrations.ts index 5121c564f0c..3a910921cb5 100644 --- a/apps/sim/lib/knowledge/application/personal-search-integrations.ts +++ b/apps/sim/lib/knowledge/application/personal-search-integrations.ts @@ -14,15 +14,11 @@ import { knowledgeOperations } from '@/lib/knowledge/application/operations' import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' import { SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import { listIndexedPersonalSearchIntegrations } from '@/lib/sim-search/indexed/integrations/personal-search-integrations' import { LIVE_SEARCH_SCOPE_FIELDS } from '@/lib/sim-search/live/policy-schema' export interface ListPersonalSearchIntegrationsInput { organizationId: string connectorType?: string - connectorId?: string - cursor?: string completionId?: string } @@ -39,10 +35,6 @@ export const listPersonalSearchIntegrations = defineAuthorizedKnowledgeUseCase({ .where(eq(user.id, userId)) .limit(1) if (!viewer) throw new OrchestrationError('forbidden', 'The current person is unavailable') - if (isIndexedOrgSearchEnabled()) - return listIndexedPersonalSearchIntegrations({ principal, input, context, userId, viewer }) - if (input.connectorId || input.cursor) - throw new OrchestrationError('validation', 'Refresh your live account connections') const scope = { kind: 'organization', organizationId: context.organizationId } as const if (!(await isScopedCredentialGroupsAvailable(scope))) return { completedCredentialId: null, connections: [], available: [], nextCursor: null } @@ -95,9 +87,6 @@ export const listPersonalSearchIntegrations = defineAuthorizedKnowledgeUseCase({ name: connector.meta.name, providerId: option.provider, connectorType: connector.type, - connectorId: undefined, - knowledgeBaseId: undefined, - indexingStatus: undefined, description: '', accounts: own, connectionStatus: !ready @@ -144,7 +133,6 @@ export const resolvePersonalSearchConnection = defineAuthorizedKnowledgeUseCase( input: { organizationId: input.organizationId, connectorType: input.target.connectorType, - connectorId: input.target.connectorId, }, }) const targets = [ @@ -159,7 +147,6 @@ export const resolvePersonalSearchConnection = defineAuthorizedKnowledgeUseCase( ({ target }) => target.provider === input.target.provider && target.connectorType === input.target.connectorType && - target.connectorId === input.target.connectorId && target.credentialId === input.target.credentialId && target.connectionMode === input.target.connectionMode && target.optionId === input.target.optionId diff --git a/apps/sim/lib/knowledge/application/personal-source-setup.test.ts b/apps/sim/lib/knowledge/application/personal-source-setup.test.ts deleted file mode 100644 index 506037b2245..00000000000 --- a/apps/sim/lib/knowledge/application/personal-source-setup.test.ts +++ /dev/null @@ -1,271 +0,0 @@ -import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' -import { - credentialGroupsCredentialsMock, - credentialGroupsCredentialsMockFns, -} from '@sim/testing/mocks/credential-groups-credentials.mock' -import { - credentialGroupsEnrollmentsMock, - credentialGroupsEnrollmentsMockFns, -} from '@sim/testing/mocks/credential-groups-enrollments.mock' -import { - credentialGroupsSelfEnrollmentMock, - credentialGroupsSelfEnrollmentMockFns, -} from '@sim/testing/mocks/credential-groups-self-enrollment.mock' -import { knowledgeContextsMock } from '@sim/testing/mocks/knowledge-contexts.mock' -import { - knowledgeMemberQueueMock, - knowledgeMemberQueueMockFns, -} from '@sim/testing/mocks/knowledge-member-queue.mock' -import { - organizationAuthorizationMock, - organizationAuthorizationMockFns, -} from '@sim/testing/mocks/organization-authorization.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -const hoisted = vi.hoisted(() => ({ - authorize: vi.fn(), - ownAccount: vi.fn(), - listAccounts: vi.fn(), - completion: vi.fn(), - provision: vi.fn(), - oauth: vi.fn(), - selector: vi.fn(), - configure: vi.fn(), - billing: vi.fn(), -})) -vi.mock('@/lib/core/application/organization-authorization', () => organizationAuthorizationMock) -vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) -vi.mock('@/lib/knowledge/application/personal-search-account', () => ({ - authorizePersonalSearchSetup: hoisted.authorize, - authorizePersonalSearchSetupCredential: hoisted.ownAccount, -})) -vi.mock('@/lib/credentials/organization-managed', () => ({ - getOwnOrganizationManagedOAuthCredentials: hoisted.listAccounts, -})) -vi.mock('@/lib/credential-groups/search-connection-completion', () => ({ - readSearchConnectionCompletion: hoisted.completion, -})) -vi.mock('@/lib/knowledge/connectors/member-provisioning', () => ({ - provisionKnowledgeConnectorMembersBinding: hoisted.provision, -})) -vi.mock('@/lib/credential-groups/self-enrollment', () => credentialGroupsSelfEnrollmentMock) -vi.mock('@/lib/credential-groups/enrollments', () => credentialGroupsEnrollmentsMock) -vi.mock('@/lib/credential-groups/oauth', () => ({ startCredentialGroupOAuth: hoisted.oauth })) -vi.mock('@/lib/selectors/application/execute-selector', () => ({ - executeSelector: { execute: hoisted.selector }, -})) -vi.mock('@/lib/knowledge/application/sim-search', () => ({ - configureSimSearchConnector: { execute: hoisted.configure }, -})) -vi.mock('@/lib/credential-groups/credentials', () => credentialGroupsCredentialsMock) -vi.mock('@/lib/knowledge/connectors/member-queue', () => knowledgeMemberQueueMock) -vi.mock('@/lib/knowledge/application/billing', () => ({ - resolveKnowledgeBillingAttribution: hoisted.billing, -})) -vi.mock('@/connectors/registry', () => ({ - CONNECTOR_META_REGISTRY: { jira: { name: 'Jira' }, confluence: { name: 'Confluence' } }, -})) - -import { - listPersonalSourceSetupAccounts, - personalSourceSetup, -} from '@/lib/knowledge/application/personal-source-setup' -import type { SelectorRequest } from '@/lib/selectors/types' - -const mocks = { - ...hoisted, - enrollment: credentialGroupsSelfEnrollmentMockFns.mockCreateViewerCredentialGroupEnrollment, - oauthContext: credentialGroupsEnrollmentsMockFns.mockGetCredentialGroupOAuthContextForEnrollment, - binding: credentialGroupsCredentialsMockFns.mockLoadManagedCredentialGroupBinding, - group: credentialGroupsCredentialsMockFns.mockLoadScopedAccountsCredentialListContext, - dispatch: knowledgeMemberQueueMockFns.mockDispatchMemberSync, -} - -interface ValidationSelectorCall { - input: { request: SelectorRequest; signal: AbortSignal } -} - -const principal = createSessionPrincipal({ userId: 'member-1' }) -const owner = { organizationId: 'organization-1', connectorType: 'jira' } as const -const credential = { credentialId: 'own-account', domain: 'example.atlassian.net' } -const connect = { ...owner, ...credential, action: 'connect', keys: ['PROJECT'] } as const -const account = { - id: 'own-account', - displayName: 'Work account', - providerId: 'jira', - scopes: ['read:jira-work'], -} -const runConnect = (changes = {}) => - personalSourceSetup.execute({ principal, input: { ...connect, keys: ['PROJECT'], ...changes } }) - -describe('personal source setup', () => { - it.each(['confluence', 'jira'] as const)( - 'rejects mixed All and explicit %s keys before discovery or saving', - async (connectorType) => { - await expect(runConnect({ connectorType, keys: ['*', 'ENG'] })).rejects.toThrow( - 'Use "*" by itself for All, or remove it to select individual items.' - ) - expect(mocks.selector).not.toHaveBeenCalled() - expect(mocks.configure).not.toHaveBeenCalled() - } - ) - - it('does not let All bypass a failed site or credential check', async () => { - mocks.selector.mockRejectedValue(new Error('Site unavailable')) - await expect(runConnect({ keys: ['*'] })).rejects.toThrow('Site unavailable') - expect(mocks.configure).not.toHaveBeenCalled() - }) - - beforeEach(() => { - vi.resetAllMocks() - mocks.authorize.mockResolvedValue(principal.userId) - mocks.ownAccount.mockResolvedValue(account) - mocks.listAccounts.mockResolvedValue([account]) - mocks.completion.mockResolvedValue(account.id) - mocks.provision.mockResolvedValue({ - credentialGroupId: 'group-1', - credentialGroupOptionId: 'option-1', - }) - mocks.enrollment.mockResolvedValue({ - enrollment: { id: 'enrollment-1', email: 'member@example.com' }, - invitationLink: 'https://example.com/credential-groups/enroll/invitation-token', - }) - mocks.oauthContext.mockResolvedValue({ option: { id: 'option-1' } }) - mocks.oauth.mockResolvedValue('https://auth.atlassian.com/authorize') - mocks.selector.mockResolvedValue({ kind: 'list', items: [{ id: 'PROJECT', label: 'Project' }] }) - mocks.configure.mockResolvedValue({ knowledgeBaseId: 'kb-1', connectorId: 'source-1' }) - mocks.binding.mockResolvedValue({ - organizationId: owner.organizationId, - credentialGroupId: 'group-1', - credentialGroupOptionId: 'option-1', - }) - mocks.group.mockResolvedValue({ credentialGroupId: 'group-1' }) - mocks.billing.mockResolvedValue({ actorUserId: principal.userId }) - mocks.dispatch.mockResolvedValue({ queued: true }) - }) - - it('does not return a receipt for a foreign or no longer active credential', async () => { - mocks.completion.mockResolvedValue('another-account') - expect( - ( - await listPersonalSourceSetupAccounts.execute({ - principal, - input: { ...owner, completionId: 'completion-1' }, - }) - ).completedCredentialId - ).toBeNull() - }) - - it('does not start OAuth if enrollment was revoked', async () => { - mocks.enrollment.mockRejectedValue(new Error('An admin removed your access')) - await expect( - personalSourceSetup.execute({ - principal, - input: { ...owner, action: 'authorize', oauthCompletionId: 'completion-1' }, - }) - ).rejects.toThrow('admin removed') - expect(mocks.oauth).not.toHaveBeenCalled() - }) - - it.each([ - { kind: 'list', items: [], truncated: true, nextCursor: '50' }, - { kind: 'list', items: [] }, - ])('rejects unavailable manually entered keys before source creation %#', async (page) => { - mocks.selector.mockResolvedValueOnce(page).mockResolvedValue({ kind: 'detail', item: null }) - await expect(runConnect({ keys: ['MISSING'] })).rejects.toThrow('could not be found') - expect(mocks.configure).not.toHaveBeenCalled() - }) - - it('fails before mutation when direct validation also exceeds its deadline', async () => { - const listing = new AbortController() - const details = new AbortController() - vi.spyOn(AbortSignal, 'timeout') - .mockReturnValueOnce(listing.signal) - .mockReturnValueOnce(details.signal) - mocks.selector - .mockResolvedValueOnce({ kind: 'list', items: [] }) - .mockImplementationOnce(({ input }: ValidationSelectorCall) => { - details.abort(new DOMException('Validation timed out', 'TimeoutError')) - input.signal.throwIfAborted() - }) - await expect(runConnect()).rejects.toThrow('took too long') - expect(mocks.configure).not.toHaveBeenCalled() - }) - - it('rejects a detail response for a different key', async () => { - mocks.selector - .mockResolvedValueOnce({ kind: 'list', items: [] }) - .mockResolvedValueOnce({ kind: 'detail', item: { id: 'OTHER', label: 'Other project' } }) - await expect(runConnect()).rejects.toThrow('could not be found') - expect(mocks.configure).not.toHaveBeenCalled() - }) - - it('bounds direct validation concurrency and validates each unresolved key only once', async () => { - let release = () => {} - let allStarted = () => {} - const gate = new Promise((resolve) => { - release = resolve - }) - const started = new Promise((resolve) => { - allStarted = resolve - }) - let active = 0 - let maximumActive = 0 - mocks.selector.mockImplementation(async ({ input }: ValidationSelectorCall) => { - if (input.request.kind === 'list') return { kind: 'list', items: [] } - active++ - maximumActive = Math.max(maximumActive, active) - if (active === 5) allStarted() - await gate - active-- - return { kind: 'detail', item: { id: input.request.id, label: input.request.id } } - }) - const keys = Array.from({ length: 12 }, (_, index) => `PROJECT${index}`) - const connection = runConnect({ keys: [...keys, ...keys] }) - await started - try { - expect(mocks.selector).toHaveBeenCalledTimes(6) - expect(mocks.configure).not.toHaveBeenCalled() - } finally { - release() - } - await expect(connection).resolves.toMatchObject({ kind: 'connected' }) - expect(maximumActive).toBe(5) - expect(mocks.selector).toHaveBeenCalledTimes(keys.length + 1) - }) - - it('rejects stale or wrong-user credentials before provider calls', async () => { - mocks.ownAccount.mockRejectedValue(new Error('Account unavailable')) - await expect(runConnect()).rejects.toThrow('Account unavailable') - expect(mocks.selector).not.toHaveBeenCalled() - expect(mocks.configure).not.toHaveBeenCalled() - }) - - it('rejects an account whose enrollment group differs from the organization setup', async () => { - mocks.binding.mockResolvedValue({ - organizationId: owner.organizationId, - credentialGroupId: 'other-group', - credentialGroupOptionId: 'option-1', - }) - await expect(runConnect()).rejects.toThrow('Connect your account again') - expect(mocks.configure).not.toHaveBeenCalled() - }) - - it('rejects a current operation authorization denial before any setup effects', async () => { - organizationAuthorizationMockFns.mockAuthorizeOrganizationOperation.mockRejectedValue( - new Error('Membership ended') - ) - await expect(runConnect()).rejects.toThrow('Membership ended') - expect(mocks.authorize).not.toHaveBeenCalled() - expect(mocks.configure).not.toHaveBeenCalled() - }) - - it.each([ - 'example.atlassian.net/wiki/spaces', - 'user:password@example.atlassian.net', - 'example.atlassian.net?site=other', - ])('rejects an invalid site value before discovery: %s', async (domain) => { - await expect(runConnect({ domain })).rejects.toThrow('hostname') - expect(mocks.selector).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/knowledge/application/personal-source-setup.ts b/apps/sim/lib/knowledge/application/personal-source-setup.ts deleted file mode 100644 index e38870f5103..00000000000 --- a/apps/sim/lib/knowledge/application/personal-source-setup.ts +++ /dev/null @@ -1,305 +0,0 @@ -import { createLogger } from '@sim/logger' -import { getErrorMessage } from '@sim/utils/errors' -import { normalizeAtlassianSiteUrl } from '@/lib/atlassian/discovery' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { mapWithConcurrency } from '@/lib/core/utils/concurrency' -import { - loadManagedCredentialGroupBinding, - loadScopedAccountsCredentialListContext, -} from '@/lib/credential-groups/credentials' -import { getCredentialGroupOAuthContextForEnrollment } from '@/lib/credential-groups/enrollments' -import { startCredentialGroupOAuth } from '@/lib/credential-groups/oauth' -import { readSearchConnectionCompletion } from '@/lib/credential-groups/search-connection-completion' -import { createViewerCredentialGroupEnrollment } from '@/lib/credential-groups/self-enrollment' -import { getOwnOrganizationManagedOAuthCredentials } from '@/lib/credentials/organization-managed' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { resolveKnowledgeBillingAttribution } from '@/lib/knowledge/application/billing' -import { resolveKnowledgeOrganizationContext } from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { - authorizePersonalSearchSetup, - authorizePersonalSearchSetupCredential, - type PersonalSearchSetupConnector, -} from '@/lib/knowledge/application/personal-search-account' -import { configureSimSearchConnector } from '@/lib/knowledge/application/sim-search' -import { provisionKnowledgeConnectorMembersBinding } from '@/lib/knowledge/connectors/member-provisioning' -import { executeSelector } from '@/lib/selectors/application/execute-selector' -import { MAX_SELECTOR_PAGES } from '@/lib/selectors/limits' -import type { SelectorExecutionResult, SelectorRequest } from '@/lib/selectors/types' -import { MAX_PERSONAL_SOURCE_SETUP_KEYS } from '@/lib/sim-search/personal-source-setup' -import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' -import { getSourceSelectionError, isAllSourceItems } from '@/connectors/selection' - -const logger = createLogger('PersonalSourceSetup') -const VALIDATION_PHASE_TIMEOUT_MS = 30_000 -const DETAIL_VALIDATION_CONCURRENCY = 5 - -interface PersonalSourceSetupOwner { - organizationId: string - connectorType: PersonalSearchSetupConnector -} - -type PersonalSourceSetupInput = PersonalSourceSetupOwner & - ( - | { action: 'authorize'; oauthCompletionId: string } - | { action: 'options'; credentialId: string; domain: string; request: SelectorRequest } - | { action: 'connect'; credentialId: string; domain: string; keys: string[] } - ) - -type PersonalSourceSetupResult = - | { kind: 'authorization'; url: string } - | { kind: 'connected'; knowledgeBaseId: string; connectorId: string } - | SelectorExecutionResult - -function normalizeSetupDomain(value: string) { - let url: URL - try { - url = new URL(normalizeAtlassianSiteUrl(value)) - } catch { - throw new OrchestrationError('validation', 'Enter your Atlassian site hostname') - } - if ( - url.username || - url.password || - url.port || - url.pathname !== '/' || - url.search || - url.hash || - !url.hostname.includes('.') - ) { - throw new OrchestrationError( - 'validation', - 'Enter your Atlassian site hostname without a page path' - ) - } - return url.hostname -} - -/** Lists the viewer's live accounts independently of whether any source exists yet. */ -export const listPersonalSourceSetupAccounts = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.listPersonalSourceSetupAccounts, - resolveContext: ({ input }: { input: PersonalSourceSetupOwner & { completionId?: string } }) => - resolveKnowledgeOrganizationContext(input), - async execute({ principal, input }) { - const userId = await authorizePersonalSearchSetup(principal, input) - const accounts = await getOwnOrganizationManagedOAuthCredentials({ - organizationId: input.organizationId, - userId, - providerId: input.connectorType, - }) - const completedCredentialId = input.completionId - ? await readSearchConnectionCompletion({ - organizationId: input.organizationId, - userId, - completionId: input.completionId, - }) - : null - return { - accounts: accounts.map((account) => ({ - id: account.id, - name: account.displayName, - provider: input.connectorType, - type: 'managed_oauth' as const, - scopes: account.scopes, - })), - completedCredentialId: accounts.some((account) => account.id === completedCredentialId) - ? completedCredentialId - : null, - } - }, -}) - -/** Connects the viewer first, then configures only an approved personal-account Search source. */ -export const personalSourceSetup = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.personalSourceSetup, - resolveContext: ({ input }: { input: PersonalSourceSetupInput }) => - resolveKnowledgeOrganizationContext(input), - async execute({ principal, input, context, request }): Promise { - const userId = await authorizePersonalSearchSetup(principal, input) - if (input.action === 'authorize') { - const binding = await provisionKnowledgeConnectorMembersBinding({ - organizationId: input.organizationId, - connectorMeta: CONNECTOR_META_REGISTRY[input.connectorType]!, - userId, - }) - const { enrollment, invitationLink } = await createViewerCredentialGroupEnrollment({ - organizationId: input.organizationId, - userId, - credentialGroupId: binding.credentialGroupId, - }) - const token = new URL(invitationLink).pathname.split('/').at(-1) - if (!token) throw new Error('Account enrollment did not return an invitation token') - const oauth = await getCredentialGroupOAuthContextForEnrollment( - { - organizationId: input.organizationId, - credentialGroupId: binding.credentialGroupId, - enrollmentId: enrollment.id, - email: enrollment.email, - userId, - }, - binding.credentialGroupOptionId - ) - if (!oauth) - throw new OrchestrationError('forbidden', 'This account connection is no longer available') - return { - kind: 'authorization', - url: await startCredentialGroupOAuth(oauth, token, { - completionRedirect: true, - returnTo: 'search', - completionId: input.oauthCompletionId, - connectionIntent: { kind: 'create' }, - }), - } - } - - await authorizePersonalSearchSetupCredential(principal, input) - const domain = normalizeSetupDomain(input.domain) - const selectorInput = { - selectorKey: - input.connectorType === 'jira' - ? ('jira.projectKeys' as const) - : ('confluence.spaces' as const), - scope: { kind: 'organization' as const, organizationId: input.organizationId }, - context: { oauthCredential: input.credentialId, domain }, - personalSearchSetup: input.connectorType, - } - if (input.action === 'options') { - return executeSelector.execute({ - principal, - request, - input: { ...selectorInput, request: input.request }, - }) - } - if ( - !input.keys.length || - input.keys.length > MAX_PERSONAL_SOURCE_SETUP_KEYS || - input.keys.some((key) => !key.trim() || key.length > 255) - ) { - throw new OrchestrationError('validation', 'Select between 1 and 1,000 projects or spaces') - } - const keys = [...new Set(input.keys.map((key) => key.trim()))] - const selectionError = getSourceSelectionError(keys) - if (selectionError) throw new OrchestrationError('validation', selectionError) - const remaining = new Set(keys) - const cursors = new Set() - const timeout = AbortSignal.timeout(VALIDATION_PHASE_TIMEOUT_MS) - const signal = request?.signal ? AbortSignal.any([request.signal, timeout]) : timeout - let cursor: string | undefined - for (let page = 0; page < MAX_SELECTOR_PAGES; page++) { - let result: SelectorExecutionResult - try { - signal.throwIfAborted() - result = await executeSelector.execute({ - principal, - request, - input: { - ...selectorInput, - signal, - request: { kind: 'list', ...(cursor ? { cursor } : {}) }, - }, - }) - signal.throwIfAborted() - } catch (error) { - if (timeout.aborted && !request?.signal?.aborted) break - throw error - } - if (result.kind !== 'list') throw new Error('Source discovery returned an unexpected result') - if (isAllSourceItems(keys)) { - remaining.clear() - break - } - for (const option of result.items) remaining.delete(option.id) - if (remaining.size === 0) break - if (!result.nextCursor || result.truncated || cursors.has(result.nextCursor)) break - cursor = result.nextCursor - cursors.add(cursor) - } - request?.signal?.throwIfAborted() - if (remaining.size > 0) { - const detailTimeout = AbortSignal.timeout(VALIDATION_PHASE_TIMEOUT_MS) - const details = new AbortController() - const detailSignal = AbortSignal.any([ - detailTimeout, - details.signal, - ...(request?.signal ? [request.signal] : []), - ]) - try { - await mapWithConcurrency([...remaining], DETAIL_VALIDATION_CONCURRENCY, async (key) => { - detailSignal.throwIfAborted() - const result = await executeSelector.execute({ - principal, - request, - input: { - ...selectorInput, - signal: detailSignal, - request: { kind: 'detail', id: key }, - }, - }) - detailSignal.throwIfAborted() - if (result.kind !== 'detail' || result.item?.id !== key) { - throw new OrchestrationError( - 'validation', - 'Some selected projects or spaces could not be found with this account. Refresh the choices and try again.' - ) - } - }) - } catch (error) { - details.abort(error) - if (detailTimeout.aborted && !request?.signal?.aborted) { - throw new OrchestrationError( - 'validation', - 'Checking the selected projects or spaces took too long. Try fewer selections.' - ) - } - throw error - } - } - const [binding, group] = await Promise.all([ - loadManagedCredentialGroupBinding(input.credentialId), - loadScopedAccountsCredentialListContext({ - kind: 'organization', - organizationId: input.organizationId, - }), - ]) - if ( - !binding || - !group || - binding.credentialGroupId !== group.credentialGroupId || - binding.organizationId !== input.organizationId - ) { - throw new OrchestrationError( - 'not_found', - 'Connect your account again before choosing sources' - ) - } - await authorizePersonalSearchSetupCredential(principal, input) - request?.signal?.throwIfAborted() - const result = await configureSimSearchConnector.execute({ - principal, - request, - input: { - organizationId: input.organizationId, - connectorType: input.connectorType, - memberCredentialBinding: { - credentialGroupId: binding.credentialGroupId, - credentialGroupOptionId: binding.credentialGroupOptionId, - }, - sourceConfig: { - domain, - [input.connectorType === 'jira' ? 'projectKey' : 'spaceKey']: [...keys].join(','), - }, - }, - }) - try { - const { dispatchMemberSync } = await import('@/lib/knowledge/connectors/member-queue') - await dispatchMemberSync(result.connectorId, { - billingAttribution: await resolveKnowledgeBillingAttribution(principal, context), - }) - } catch (error) { - logger.warn('Initial personal source sync will retry on its schedule', { - error: getErrorMessage(error), - }) - } - return { kind: 'connected', ...result } - }, -}) diff --git a/apps/sim/lib/knowledge/application/search-diagnostics.ts b/apps/sim/lib/knowledge/application/search-diagnostics.ts index 7643ad14836..da404865994 100644 --- a/apps/sim/lib/knowledge/application/search-diagnostics.ts +++ b/apps/sim/lib/knowledge/application/search-diagnostics.ts @@ -1,5 +1,4 @@ import type { AuthorizingUseCase } from '@/lib/core/application' -import type { ResourceOwner } from '@/lib/core/resource-scope' import type { knowledgeOperations } from '@/lib/knowledge/application/operations' import type { SearchKnowledgeInput } from '@/lib/knowledge/application/search' import { @@ -44,29 +43,3 @@ export function instrumentSearchUseCase< ), } } - -/** - * Times the source overview, whose cost is access batching and live source proof rather than - * retrieval. It shares the search trace so one log line explains a slow Sim Search surface. - */ -export function instrumentSourceOverviewUseCase( - useCase: AuthorizingUseCase -): AuthorizingUseCase { - return { - ...useCase, - execute: (args) => - withSearchDiagnostics( - { - operation: 'read_search_source_overview', - principalKind: args.principal.kind, - /** Derived without `resourceScopeFromOwner`, which throws before authorization runs. */ - scopeKind: args.input.organizationId - ? 'organization' - : args.input.workspaceId - ? 'workspace' - : undefined, - }, - () => measureSearchStage('source_overview', () => useCase.execute(args)) - ), - } -} diff --git a/apps/sim/lib/knowledge/application/search-integrations.test.ts b/apps/sim/lib/knowledge/application/search-integrations.test.ts index 1f00c17269a..e353084631e 100644 --- a/apps/sim/lib/knowledge/application/search-integrations.test.ts +++ b/apps/sim/lib/knowledge/application/search-integrations.test.ts @@ -4,13 +4,7 @@ import { organization, organizationSearchIntegration, } from '@sim/db/schema' -import { - dbChainMockFns, - queueTableRows, - resetDbChainMock, - resetEnvFlagsMock, - setEnvFlags, -} from '@sim/testing' +import { dbChainMockFns, queueTableRows, resetDbChainMock, resetEnvFlagsMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' import { auditMock, auditMockFns } from '@sim/testing/mocks/audit.mock' import { @@ -153,7 +147,7 @@ describe('organization Search approval', () => { }) it('preserves existing sources while an explicit deactivation overrides them', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + queueTableRows(organization, [{ metadata: {} }]) queueTableRows(member, [{ role: 'member' }]) queueTableRows(organizationSearchIntegration, [{ connectorType: 'gmail', approved: false }]) queueTableRows(knowledgeConnector, [ @@ -162,12 +156,12 @@ describe('organization Search approval', () => { ]) await expect( listSearchIntegrations.execute({ principal, input: { organizationId: 'organization' } }) - ).resolves.toEqual([ - { connectorType: 'gmail', approved: false }, - { connectorType: 'google_drive', approved: true }, - { connectorType: 'github', approved: false }, - { connectorType: 'jira', approved: false }, - ]) + ).resolves.toEqual( + expect.arrayContaining([ + expect.objectContaining({ connectorType: 'gmail', approved: false }), + expect.objectContaining({ connectorType: 'google_drive', approved: true }), + ]) + ) }) }) @@ -224,7 +218,7 @@ describe('organization Search controls through Mothership', () => { }) it('allows delegated members to read approval state without granting writes', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + queueTableRows(organization, [{ metadata: {} }]) queueTableRows(member, [{ role: 'member' }]) queueTableRows(organizationSearchIntegration, [{ connectorType: 'gmail', approved: true }]) queueTableRows(knowledgeConnector, []) @@ -233,7 +227,7 @@ describe('organization Search controls through Mothership', () => { principal: delegatedPrincipal, input: { organizationId: 'organization' }, }) - ).resolves.toContainEqual({ connectorType: 'gmail', approved: true }) + ).resolves.toContainEqual(expect.objectContaining({ connectorType: 'gmail', approved: true })) expect(dbChainMockFns.insert).not.toHaveBeenCalled() }) }) @@ -254,7 +248,6 @@ describe('live organization search policies', () => { it.each([principal, delegatedPrincipal])( 'repairs an already approved member source through the same atomic admin action', async (actor) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) mocks.memberSetup.mockResolvedValueOnce({ groupId: 'group', changed: true }) const result = await approveSearchIntegration.execute({ @@ -282,7 +275,6 @@ describe('live organization search policies', () => { it.each([principal, delegatedPrincipal])( 'requires integrations capability before provisioning sign-in for an authorized admin', async (actor) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) permissionGroupsResolveMockFns.mockGetUserPermissionConfigForOrganization.mockResolvedValue({ hideIntegrationsTab: true, @@ -299,7 +291,6 @@ describe('live organization search policies', () => { ) it('does not approve a source when member sign-in setup fails', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) mocks.memberSetup.mockRejectedValueOnce(new Error('Provider configuration is unavailable')) await expect(approveSearchIntegration.execute({ principal, input })).rejects.toThrow( @@ -310,7 +301,6 @@ describe('live organization search policies', () => { }) it('explains missing OAuth app setup instead of approving a source with an unusable connection', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) mocks.memberSetup.mockRejectedValueOnce( new CredentialGroupProviderConfigurationError('Managed Jira authorization is not configured') @@ -326,13 +316,11 @@ describe('live organization search policies', () => { }) it('does not provision sign-in while removing a source', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) await approveSearchIntegration.execute({ principal, input: { ...input, approved: false } }) expect(mocks.memberSetup).not.toHaveBeenCalled() }) it('clears source settings when switching to member accounts', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) const result = await approveSearchIntegration.execute({ principal, @@ -357,7 +345,6 @@ describe('live organization search policies', () => { ])( 'rejects unavailable sources before writing without masking infrastructure errors', async ({ error, code }) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) mocks.source.mockRejectedValueOnce(error) const attempt = approveSearchIntegration.execute({ @@ -377,21 +364,7 @@ describe('live organization search policies', () => { expect(dbChainMockFns.insert).not.toHaveBeenCalled() } ) - - it('rejects policy writes when the rollout flag is off', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - queueTableRows(member, [{ role: 'owner' }]) - await expect( - approveSearchIntegration.execute({ - principal, - input: { ...input, policy: defaultLiveSearchPolicy() }, - }) - ).rejects.toMatchObject({ code: 'validation' }) - expect(dbChainMockFns.insert).not.toHaveBeenCalled() - expect(dbChainMockFns.update).not.toHaveBeenCalled() - }) it('saves a validated service source and its approval in one transaction', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) const result = await approveSearchIntegration.execute({ principal, @@ -424,7 +397,6 @@ describe('live organization search policies', () => { expect(result.changed).toBe(true) }) it('allows GitHub App mode without a single source ID', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) const result = await approveSearchIntegration.execute({ principal, @@ -440,7 +412,6 @@ describe('live organization search policies', () => { expect(dbChainMockFns.transaction).toHaveBeenCalledOnce() }) it('saves an unfinished service-account source without exposing member-mode search', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) const result = await approveSearchIntegration.execute({ principal, @@ -454,7 +425,6 @@ describe('live organization search policies', () => { expect(mocks.source).not.toHaveBeenCalled() }) it('rejects invalid scope data before any protected write', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'admin' }]) await expect( approveSearchIntegration.execute({ @@ -466,7 +436,6 @@ describe('live organization search policies', () => { expect(dbChainMockFns.insert).not.toHaveBeenCalled() }) it('prevents members from changing live search scopes', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'member' }]) await expect( approveSearchIntegration.execute({ @@ -477,7 +446,6 @@ describe('live organization search policies', () => { expect(dbChainMockFns.transaction).not.toHaveBeenCalled() }) it('reads saved policies alongside inherited approval without requiring an index', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) queueTableRows(member, [{ role: 'member' }]) queueTableRows(organizationSearchIntegration, [{ connectorType: 'gmail', approved: true }]) queueTableRows(organization, [ diff --git a/apps/sim/lib/knowledge/application/search-integrations.ts b/apps/sim/lib/knowledge/application/search-integrations.ts index 9456a2de42f..2e5b97ca5db 100644 --- a/apps/sim/lib/knowledge/application/search-integrations.ts +++ b/apps/sim/lib/knowledge/application/search-integrations.ts @@ -3,18 +3,17 @@ import { requirePrincipalSubjectUserId } from '@sim/auth/principal' import { db } from '@sim/db' import { organization, organizationSearchIntegration } from '@sim/db/schema' import { eq, sql } from 'drizzle-orm' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { OrchestrationError } from '@/lib/core/orchestration/types' import { CredentialGroupProviderConfigurationError } from '@/lib/credential-groups/provider-adapter' import { isScopedCredentialGroupsAvailable } from '@/lib/credential-groups/scoped-availability' import { addOrganizationAccountProvider } from '@/lib/credential-groups/service' +import { SLACK_SEARCH_USER_SCOPES } from '@/lib/credential-groups/slack-managed-user-scopes' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' import { refuseCapability } from '@/lib/permission-groups/capabilities' import { isOrganizationCapabilityWithheld } from '@/lib/permission-groups/capability-assertions' -import { SEARCH_SOURCE_TYPES } from '@/lib/sim-search/connectors' import { NativeSearchError } from '@/lib/sim-search/live/http' import { addOrganizationSearchMcpProvider, @@ -52,24 +51,19 @@ export const listSearchIntegrations = defineAuthorizedKnowledgeUseCase({ if (!context.organizationId) throw new OrchestrationError('validation', 'Organization is required') const approvals = await listOrganizationSearchApprovals(context.organizationId) - const policies = isLiveEnterpriseSearchEnabled - ? await loadLiveSearchPolicies({ organizationId: context.organizationId }) - : undefined + const policies = await loadLiveSearchPolicies({ organizationId: context.organizationId }) const scope = { kind: 'organization', organizationId: context.organizationId } as const - const zoomEnabled = - !isLiveEnterpriseSearchEnabled || (await isSearchProviderEnabled('zoom', scope)) - return (isLiveEnterpriseSearchEnabled ? LIVE_SEARCH_SOURCE_TYPES : SEARCH_SOURCE_TYPES).map( - ([connectorType]) => ({ - connectorType, - approved: approvals.get(connectorType) ?? false, - ...(policies - ? { - policy: livePolicyFor(policies, connectorType), - available: connectorType !== 'zoom' || zoomEnabled, - } - : {}), - }) - ) + const zoomEnabled = await isSearchProviderEnabled('zoom', scope) + return LIVE_SEARCH_SOURCE_TYPES.map(([connectorType]) => ({ + connectorType, + approved: approvals.get(connectorType) ?? false, + ...(policies + ? { + policy: livePolicyFor(policies, connectorType), + available: connectorType !== 'zoom' || zoomEnabled, + } + : {}), + })) }, }) @@ -81,14 +75,11 @@ export const approveSearchIntegration = defineAuthorizedKnowledgeUseCase({ async execute({ input, context, principal }) { if (!context.organizationId) throw new OrchestrationError('validation', 'Organization is required') - const source = ( - isLiveEnterpriseSearchEnabled ? LIVE_SEARCH_SOURCE_TYPES : SEARCH_SOURCE_TYPES - ).find(([type]) => type === input.connectorType) + const source = LIVE_SEARCH_SOURCE_TYPES.find(([type]) => type === input.connectorType) if (!source) { throw new OrchestrationError('validation', 'This integration is not supported by Sim Search') } if ( - isLiveEnterpriseSearchEnabled && input.approved && !(await isSearchProviderEnabled(input.connectorType, { kind: 'organization', @@ -99,16 +90,12 @@ export const approveSearchIntegration = defineAuthorizedKnowledgeUseCase({ 'forbidden', 'Zoom Search is not available for this organization' ) - if (input.policy && !isLiveEnterpriseSearchEnabled) - throw new OrchestrationError('validation', 'Live search settings are not enabled') - const memberProvider = - isLiveEnterpriseSearchEnabled && input.approved - ? liveSearchMemberAccountProvider(input.connectorType) - : null - const mcpProvider = - isLiveEnterpriseSearchEnabled && input.approved - ? liveSearchMcpConnector(input.connectorType) - : null + const memberProvider = input.approved + ? input.connectorType === 'slack' + ? 'slack' + : liveSearchMemberAccountProvider(input.connectorType) + : null + const mcpProvider = input.approved ? liveSearchMcpConnector(input.connectorType) : null if (memberProvider || mcpProvider) { /** permission-group-enforced: integrations.manage — adding sign-in is part of this explicit source action. */ if (await isOrganizationCapabilityWithheld(context.organizationId, 'integrations.manage')) @@ -185,7 +172,13 @@ export const approveSearchIntegration = defineAuthorizedKnowledgeUseCase({ memberAccounts = await addOrganizationAccountProvider( context.organizationId!, requirePrincipalSubjectUserId(principal), - { provider: memberProvider, label: source[1].name }, + { + provider: memberProvider, + label: source[1].name, + ...(memberProvider === 'slack' + ? { requiredScopes: [...SLACK_SEARCH_USER_SCOPES] } + : {}), + }, tx ).catch((error: unknown) => { if (error instanceof CredentialGroupProviderConfigurationError) @@ -233,7 +226,7 @@ export const approveSearchIntegration = defineAuthorizedKnowledgeUseCase({ action: AuditAction.CREDENTIAL_GROUP_UPDATED, resourceType: AuditResourceType.CREDENTIAL_GROUP, resourceId: result.memberAccounts.groupId, - description: `Added ${result.connectorType} member sign-in for Sim Search`, + description: `Configured ${result.connectorType} member sign-in for Sim Search`, metadata: { connectorType: result.connectorType }, }, ] diff --git a/apps/sim/lib/knowledge/application/search-source-overview.ts b/apps/sim/lib/knowledge/application/search-source-overview.ts deleted file mode 100644 index dbc4f92ec0a..00000000000 --- a/apps/sim/lib/knowledge/application/search-source-overview.ts +++ /dev/null @@ -1,220 +0,0 @@ -import { db } from '@sim/db' -import { document, embedding, knowledgeBase, knowledgeConnector } from '@sim/db/schema' -import { and, eq, exists, inArray, isNull, notInArray, or } from 'drizzle-orm' -import type { SearchSourceOverview } from '@/lib/api/contracts/knowledge/connectors' -import { type ResourceOwner, resourceScopeFromOwner } from '@/lib/core/resource-scope' -import { resourceScopeCondition } from '@/lib/core/resource-scope.server' -import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' -import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { instrumentSourceOverviewUseCase } from '@/lib/knowledge/application/search-diagnostics' -import { MAX_SEARCH_SOURCE_PROVIDER_TYPES } from '@/lib/knowledge/constants' -import { knowledgeReadAccessBatches } from '@/lib/knowledge/read-access' -import { annotateSearchDiagnostics, measureSearchStage } from '@/lib/knowledge/search/diagnostics' -import { searchIntegrationAccessCondition } from '@/lib/knowledge/search/integration-policy' -import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' - -function searchProviderTypes() { - const providerTypes = Object.keys(CONNECTOR_META_REGISTRY) - if (providerTypes.length > MAX_SEARCH_SOURCE_PROVIDER_TYPES) { - throw new Error('Search provider catalog exceeds the overview bound') - } - return providerTypes -} - -/** Live search-index sources of a known provider, over `knowledge_connector` joined to its base. */ -function configuredSearchSourceCondition(owner: ResourceOwner, providerTypes: string[]) { - return and( - resourceScopeCondition(knowledgeBase, resourceScopeFromOwner(owner)), - eq(knowledgeBase.isSearchIndex, true), - isNull(knowledgeBase.deletedAt), - inArray(knowledgeConnector.connectorType, providerTypes), - inArray(knowledgeConnector.accessMode, ['admin', 'members']), - isNull(knowledgeConnector.archivedAt), - isNull(knowledgeConnector.deletedAt) - ) -} - -function configuredProvidersQuery() { - return db - .selectDistinct({ connectorType: knowledgeConnector.connectorType }) - .from(knowledgeConnector) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) -} - -/** - * Provider types with at least one configured search source in an already authorized owner. - * Reads no documents, so callers needing only setup state avoid the overview's access probes. - */ -export async function listConfiguredSearchProviderTypes(owner: ResourceOwner): Promise { - const rows = await configuredProvidersQuery() - .where(configuredSearchSourceCondition(owner, searchProviderTypes())) - .limit(MAX_SEARCH_SOURCE_PROVIDER_TYPES) - return rows.map(({ connectorType }) => connectorType) -} - -/** Provider-level existence probes keep setup cards independent of the loaded source pages. */ -export const readSearchSourceOverview = instrumentSourceOverviewUseCase( - defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.readSearchSourceOverview, - resolveContext: ({ input }: { input: ResourceOwner }) => resolveKnowledgeOwnerContext(input), - async execute({ principal, context }): Promise { - const availability = await measureSearchStage('source_overview.availability', () => - resolveKnowledgeAccessAvailability(context) - ) - const access = createKnowledgeAccessProvider(principal, context) - const providerTypes = searchProviderTypes() - const configured = configuredSearchSourceCondition(context, providerTypes) - const available = or( - availability.memberScoped ? eq(knowledgeConnector.accessMode, 'members') : undefined, - availability.sourceMirrored - ? and( - eq(knowledgeConnector.accessMode, 'admin'), - inArray( - knowledgeConnector.connectorType, - providerTypes.filter( - (type) => - availability.memberScoped || - !CONNECTOR_META_REGISTRY[type].requiresMemberIdentity - ) - ) - ) - : undefined - ) - const syncingEnabled = and( - available, - searchIntegrationAccessCondition(), - notInArray(knowledgeConnector.status, ['paused', 'disabled']), - or( - eq(knowledgeConnector.accessMode, 'admin'), - notInArray(knowledgeConnector.memberSyncStatus, ['disabled']) - ) - ) - const documentConditions = and( - eq(document.connectorId, knowledgeConnector.id), - eq(document.knowledgeBaseId, knowledgeConnector.knowledgeBaseId), - eq(document.enabled, true), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt) - ) - const providers = await measureSearchStage('source_overview.providers', () => - configuredProvidersQuery().where(configured).limit(MAX_SEARCH_SOURCE_PROVIDER_TYPES) - ) - annotateSearchDiagnostics({ configuredProviderCount: providers.length }) - const indexingTypes = new Set() - let searchableProbes = 0 - let hasSearchableDocuments = false - const probesSources: boolean = availability.memberScoped || availability.sourceMirrored - /** One searchable document is the whole answer, so later batches skip the probe entirely. */ - const probesSearchable = (): boolean => probesSources && !hasSearchableDocuments - /** - * A provider type is only read back as membership of `indexingTypes`, so once every - * configured type is in the set no later batch can change the answer. - */ - const probesIndexing = (): boolean => - probesSources && providers.some(({ connectorType }) => !indexingTypes.has(connectorType)) - for await (const accessCondition of knowledgeReadAccessBatches(access, [ - configured, - available, - documentConditions, - ])) { - const readableDocument = and(documentConditions, accessCondition) - const probesSearchableNow = probesSearchable() - if (probesSearchableNow) searchableProbes += 1 - /** Annotated so the searchable probe's guard does not infer through its own result. */ - const [indexing, searchable]: [{ connectorType: string }[], { id: string }[]] = - await Promise.all([ - probesIndexing() - ? measureSearchStage('source_overview.indexing', () => - configuredProvidersQuery() - .where( - and( - configured, - syncingEnabled, - /** - * The probe narrows the configured set the provider list came from, so a - * type already found stays found; excluding it only drops repeated work. - * An empty set adds no predicate rather than a no-op one. - */ - indexingTypes.size > 0 - ? notInArray(knowledgeConnector.connectorType, [...indexingTypes]) - : undefined, - or( - inArray(knowledgeConnector.status, ['pending', 'syncing']), - and( - eq(knowledgeConnector.accessMode, 'members'), - inArray(knowledgeConnector.memberSyncStatus, ['pending', 'running']) - ), - exists( - db - .select({ id: document.id }) - .from(document) - .where( - and( - readableDocument, - inArray(document.processingStatus, ['pending', 'processing']) - ) - ) - ) - ) - ) - ) - .limit(MAX_SEARCH_SOURCE_PROVIDER_TYPES) - ) - : [], - probesSearchableNow - ? measureSearchStage('source_overview.searchable', () => - db - .select({ id: document.id }) - .from(document) - .innerJoin(knowledgeConnector, eq(knowledgeConnector.id, document.connectorId)) - .innerJoin( - knowledgeBase, - eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId) - ) - .where( - and( - configured, - available, - readableDocument, - eq(document.processingStatus, 'completed'), - exists( - db - .select({ id: embedding.id }) - .from(embedding) - .where( - and( - eq(embedding.documentId, document.id), - eq(embedding.enabled, true) - ) - ) - ) - ) - ) - .limit(1) - ) - : [], - ]) - for (const provider of indexing) indexingTypes.add(provider.connectorType) - hasSearchableDocuments ||= searchable.length > 0 - /** - * Both probes are saturated, so every remaining batch would be discovered and live-proved - * for no probe. `accessBatchCount` and `liveProofConnectorCount` stay what they document: - * the batches and proofs this read actually spent, not the batches the owner could produce. - */ - if (!probesSearchable() && !probesIndexing()) break - } - annotateSearchDiagnostics({ searchableProbeCount: searchableProbes }) - return { - providers: providers.map(({ connectorType }) => ({ - connectorType, - isSyncing: indexingTypes.has(connectorType), - })), - hasSearchableDocuments, - } - }, - }) -) diff --git a/apps/sim/lib/knowledge/application/search-source-progress.ts b/apps/sim/lib/knowledge/application/search-source-progress.ts deleted file mode 100644 index 89ad3efa680..00000000000 --- a/apps/sim/lib/knowledge/application/search-source-progress.ts +++ /dev/null @@ -1,110 +0,0 @@ -import { requirePrincipalSubjectUserId } from '@sim/auth/principal' -import { db } from '@sim/db' -import { document, knowledgeBase, knowledgeConnector } from '@sim/db/schema' -import { and, eq, exists, inArray, isNull, type SQL, sql } from 'drizzle-orm' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { type ResourceOwner, resourceScopeFromOwner } from '@/lib/core/resource-scope' -import { resourceScopeCondition } from '@/lib/core/resource-scope.server' -import { knowledgeAccessCondition } from '@/lib/knowledge/access/predicate' -import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { hasViewerMemberSyncError } from '@/lib/knowledge/connectors/viewer-member-sync-error' -import { MAX_SEARCH_SOURCE_PROGRESS_ITEMS } from '@/lib/knowledge/constants' -import { failedDocumentCondition } from '@/lib/knowledge/documents/processing-status' -import { searchIntegrationAccessCondition } from '@/lib/knowledge/search/integration-policy' - -interface ReadSearchSourceProgressInput extends ResourceOwner { - connectorIds: string[] -} - -/** Progress probes stop at the first visible document, without recounting completed chunks. */ -export const readSearchSourceProgress = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.readSearchSourceProgress, - resolveContext: ({ input }: { input: ReadSearchSourceProgressInput }) => - resolveKnowledgeOwnerContext(input), - async execute({ principal, input, context }) { - if ( - input.connectorIds.length === 0 || - input.connectorIds.length > MAX_SEARCH_SOURCE_PROGRESS_ITEMS - ) { - throw new OrchestrationError( - 'validation', - `Provide between 1 and ${MAX_SEARCH_SOURCE_PROGRESS_ITEMS} sources` - ) - } - const access = await createKnowledgeAccessProvider(principal, context).getForConnectors( - input.connectorIds - ) - const hasDocumentsInState = (condition: SQL) => - sql`${exists( - db - .select({ id: document.id }) - .from(document) - .where( - and( - eq(document.knowledgeBaseId, knowledgeConnector.knowledgeBaseId), - eq(document.connectorId, knowledgeConnector.id), - condition, - eq(document.enabled, true), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - knowledgeAccessCondition(access) - ) - ) - )}` - const rows = await db - .select({ - connectorId: knowledgeConnector.id, - status: knowledgeConnector.status, - accessMode: knowledgeConnector.accessMode, - memberSyncStatus: knowledgeConnector.memberSyncStatus, - hasRetainedSyncError: sql`${knowledgeConnector.lastSyncError} IS NOT NULL`, - hasViewerMemberSyncError: hasViewerMemberSyncError( - requirePrincipalSubjectUserId(principal) - ), - approved: sql`${searchIntegrationAccessCondition()}`, - isIndexing: hasDocumentsInState( - inArray(document.processingStatus, ['pending', 'processing']) - ), - hasIndexingError: hasDocumentsInState(failedDocumentCondition()), - }) - .from(knowledgeConnector) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) - .where( - and( - resourceScopeCondition(knowledgeBase, resourceScopeFromOwner(context)), - eq(knowledgeBase.isSearchIndex, true), - isNull(knowledgeBase.deletedAt), - inArray(knowledgeConnector.id, [...new Set(input.connectorIds)]), - inArray(knowledgeConnector.accessMode, ['admin', 'members']), - isNull(knowledgeConnector.archivedAt), - isNull(knowledgeConnector.deletedAt) - ) - ) - .limit(MAX_SEARCH_SOURCE_PROGRESS_ITEMS) - return { - sources: rows.map((row) => ({ - connectorId: row.connectorId, - isSyncing: - row.approved && - row.status !== 'paused' && - row.status !== 'disabled' && - !(row.accessMode === 'members' && row.memberSyncStatus === 'disabled') && - (row.status === 'pending' || - row.status === 'syncing' || - (row.accessMode === 'members' && - (row.memberSyncStatus === 'pending' || row.memberSyncStatus === 'running')) || - row.isIndexing), - hasSyncError: - row.status === 'error' || - row.hasRetainedSyncError === true || - row.hasViewerMemberSyncError === true || - (row.accessMode === 'members' && row.memberSyncStatus === 'error'), - hasIndexingError: row.hasIndexingError, - })), - } - }, -}) diff --git a/apps/sim/lib/knowledge/application/search-sources.test.ts b/apps/sim/lib/knowledge/application/search-sources.test.ts index e4c0154fc18..7792aeeac6e 100644 --- a/apps/sim/lib/knowledge/application/search-sources.test.ts +++ b/apps/sim/lib/knowledge/application/search-sources.test.ts @@ -1,14 +1,10 @@ -import { knowledgeBase, knowledgeConnector, member, user } from '@sim/db/schema' +import { knowledgeConnector, member } from '@sim/db/schema' import { dbChainMockFns, queueTableRows, resetDbChainMock } from '@sim/testing' import { createPersonalApiKeyPrincipal, createSessionPrincipal, createWorkspaceApiKeyPrincipal, } from '@sim/testing/factories/principal.factory' -import { - knowledgeAccessScopeMock, - knowledgeAccessScopeMockFns, -} from '@sim/testing/mocks/knowledge-access-scope.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -21,26 +17,10 @@ import { permissionGroupsResolveMock } from '@sim/testing/mocks/permission-group import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' import { beforeEach, describe, expect, it, vi } from 'vitest' -const hoisted = vi.hoisted(() => ({ - memberships: vi.fn(), - accounts: vi.fn(), - predicate: vi.fn(), -})) vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) -vi.mock('@/lib/knowledge/connectors/member-provisioning', () => ({ - resolveViewerConnectorMemberships: hoisted.memberships, -})) -vi.mock('@/lib/knowledge/connectors/viewer-source-accounts', () => ({ - resolveViewerSourceAccounts: hoisted.accounts, -})) -vi.mock('@/lib/knowledge/access/scope', () => knowledgeAccessScopeMock) -vi.mock('@/lib/knowledge/access/predicate', () => ({ - knowledgeAccessCondition: hoisted.predicate, - knowledgeMetadataCandidateAccessCondition: hoisted.predicate, -})) vi.mock('@/connectors/registry', () => { const registry = { google_drive: { id: 'google_drive', search: true, configFields: [{ id: 'folderId' }] }, @@ -60,24 +40,14 @@ vi.mock('@/connectors/registry', () => { } }) -import { searchSourceSummarySchema } from '@/lib/api/contracts/knowledge/connectors' -import { readSearchSourceOverview } from '@/lib/knowledge/application/search-source-overview' -import { readSearchSourceProgress } from '@/lib/knowledge/application/search-source-progress' import { listSearchSources } from '@/lib/knowledge/application/search-sources' -const mocks = { - ...hoisted, - access: knowledgeAccessScopeMockFns.mockCreateKnowledgeAccessProvider, -} - workspaceAuthzMockFns.mockPermissionSatisfies.mockImplementation( (actual: string | null) => actual !== null ) const principal = createSessionPrincipal({ userId: 'reader', sessionId: 'session' }) const input = { workspaceId: 'workspace' } -const access = { kind: 'user', userId: principal.userId, tokens: ['u:reader@example.test'] } -const ACL = { type: 'viewer-acl' } const LAST_SYNC = new Date('2026-09-05T12:00:00.000Z') function source(id: string, connectorType = 'google_drive', accessMode = 'admin') { @@ -103,9 +73,8 @@ function source(id: string, connectorType = 'google_drive', accessMode = 'admin' } } -function seed(rows: ReturnType[], emailVerified = true) { +function seed(rows: ReturnType[]) { queueTableRows(knowledgeConnector, rows) - queueTableRows(user, [{ emailVerified }]) } beforeEach(() => { @@ -120,15 +89,6 @@ beforeEach(() => { sourceMirrored: true, memberScoped: true, }) - mocks.memberships.mockResolvedValue(new Map()) - mocks.accounts.mockResolvedValue(new Map()) - mocks.access.mockReturnValue({ - get: async () => access, - getForConnectors: async () => access, - getForDocuments: async () => access, - liveSourceConnectorCondition: async () => null, - }) - mocks.predicate.mockReturnValue(ACL) }) describe('Search source summaries', () => { @@ -137,9 +97,6 @@ describe('Search source summaries', () => { async (role) => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(role) seed([source('drive')]) - queueTableRows(knowledgeConnector, [ - { connectorId: 'drive', hasDocuments: true, failedCount: 0, isIndexing: false }, - ]) const result = await listSearchSources.execute({ principal, input }) expect(result.sources).toEqual([ { @@ -151,71 +108,21 @@ describe('Search source summaries', () => { isGitHubInstallation: false, availability: 'available', enabled: true, - isSyncing: false, - lastSyncAt: LAST_SYNC.toISOString(), - hasSyncError: false, - hasViewerDocuments: true, - viewerFailedDocumentCount: 0, - viewerEmailVerified: true, - viewerAccounts: [], - connectionRequired: false, - viewerMembership: null, }, ]) - expect(searchSourceSummarySchema.parse(result.sources[0])).toEqual(result.sources[0]) expect(JSON.stringify(result)).not.toMatch( /secret-fixture|admin@example|group-secret|option-secret|sourceConfig/ ) expect(knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext).toHaveBeenCalledWith(input) - expect(mocks.access).toHaveBeenCalledWith(principal, { - workspaceId: 'workspace', - workspaceOrganizationId: null, - allowPersonalApiKeys: true, - }) - expect(mocks.predicate).toHaveBeenCalledWith(access) } ) - it('surfaces retained partial sync errors without returning the private error message', async () => { - seed([{ ...source('drive'), hasRetainedSyncError: true }]) - const result = await listSearchSources.execute({ principal, input }) - expect(result.sources[0].hasSyncError).toBe(true) - expect(result.sources[0]).not.toHaveProperty('lastSyncError') - }) - - it('restricts the source query to this workspace, the Search index, and live configured sources', async () => { - seed([]) - await expect(listSearchSources.execute({ principal, input })).resolves.toEqual({ - sources: [], - nextCursor: null, - }) - expect(dbChainMockFns.where).toHaveBeenCalledWith({ - type: 'and', - conditions: expect.arrayContaining([ - { - type: 'and', - conditions: [ - { type: 'eq', left: knowledgeBase.workspaceId, right: 'workspace' }, - { type: 'isNull', column: knowledgeBase.organizationId }, - ], - }, - { type: 'eq', left: knowledgeBase.isSearchIndex, right: true }, - { type: 'isNull', column: knowledgeBase.deletedAt }, - { type: 'isNull', column: knowledgeConnector.archivedAt }, - { type: 'isNull', column: knowledgeConnector.deletedAt }, - { type: 'inArray', column: knowledgeConnector.accessMode, values: ['admin', 'members'] }, - ]), - }) - expect(mocks.memberships).not.toHaveBeenCalled() - }) - it('rejects a former workspace member before querying source data', async () => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(null) await expect(listSearchSources.execute({ principal, input })).rejects.toMatchObject({ code: 'forbidden', }) expect(dbChainMockFns.select).not.toHaveBeenCalled() - expect(mocks.memberships).not.toHaveBeenCalled() }) it.each([ @@ -240,31 +147,20 @@ describe('Search source summaries', () => { describe('organization Search source summaries', () => { it.each(['member', 'admin'])( - 'returns only the current %s viewer ACL counts without a workspace membership', + 'returns configured sources to a current organization %s without workspace membership', async (role) => { knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue({ organizationId: 'org-1', }) queueTableRows(member, [{ role }]) seed([source('drive')]) - queueTableRows(knowledgeConnector, []) - queueTableRows(knowledgeConnector, [ - { connectorId: 'drive', hasDocuments: true, failedCount: 0, isIndexing: false }, - ]) const result = await listSearchSources.execute({ principal, input: { organizationId: 'org-1' }, }) expect(result.sources[0]).toMatchObject({ connectorId: 'drive', - hasViewerDocuments: true, - viewerEmailVerified: true, }) - expect(mocks.access).toHaveBeenCalledWith(principal, { organizationId: 'org-1' }) - expect(mocks.predicate).toHaveBeenCalledWith(access) - expect(mocks.memberships).toHaveBeenCalledWith( - expect.objectContaining({ organizationId: 'org-1', userId: 'reader' }) - ) expect(workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission).not.toHaveBeenCalled() expect(JSON.stringify(result)).not.toMatch( /secret-fixture|admin@example|group-secret|option-secret/ @@ -280,28 +176,6 @@ describe('organization Search source summaries', () => { await expect( listSearchSources.execute({ principal, input: { organizationId: 'org-1' } }) ).rejects.toThrow('Organization not found') - expect(mocks.memberships).not.toHaveBeenCalled() - expect(mocks.access).not.toHaveBeenCalled() - }) -}) - -describe('bounded Search progress', () => { - it('does not read progress for a former member', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(null) - await expect( - readSearchSourceProgress.execute({ principal, input: { ...input, connectorIds: ['drive'] } }) - ).rejects.toMatchObject({ code: 'forbidden' }) - expect(mocks.access).not.toHaveBeenCalled() - }) - - it('rejects an oversized progress request before document work', async () => { - await expect( - readSearchSourceProgress.execute({ - principal, - input: { ...input, connectorIds: Array(101).fill('drive') }, - }) - ).rejects.toMatchObject({ code: 'validation' }) - expect(mocks.access).not.toHaveBeenCalled() }) }) @@ -327,7 +201,7 @@ describe('bounded Search source pagination', () => { input: { ...input, cursor: first.nextCursor!, - ...(change === 'filter' ? { mine: true } : {}), + ...(change === 'filter' ? { search: 'new' } : {}), ...(change === 'provider' ? { connectorType: 'gmail' } : {}), ...(change === 'excluded-provider' ? { excludeConnectorType: 'github' } : {}), }, @@ -337,14 +211,3 @@ describe('bounded Search source pagination', () => { } ) }) - -describe('Search source overview', () => { - it('rechecks current membership before reading the overview', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(null) - await expect(readSearchSourceOverview.execute({ principal, input })).rejects.toMatchObject({ - code: 'forbidden', - }) - expect(mocks.access).not.toHaveBeenCalled() - expect(dbChainMockFns.select).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/knowledge/application/search-sources.ts b/apps/sim/lib/knowledge/application/search-sources.ts index 637ccc9145e..3a725f8c53f 100644 --- a/apps/sim/lib/knowledge/application/search-sources.ts +++ b/apps/sim/lib/knowledge/application/search-sources.ts @@ -1,8 +1,8 @@ import { requirePrincipalSubjectUserId } from '@sim/auth/principal' import { db } from '@sim/db' -import { document, embedding, knowledgeBase, knowledgeConnector, user } from '@sim/db/schema' +import { knowledgeBase, knowledgeConnector } from '@sim/db/schema' import { toRecord } from '@sim/utils/object' -import { and, desc, eq, exists, inArray, isNull, lt, ne, or, type SQL, sql } from 'drizzle-orm' +import { and, desc, eq, inArray, isNull, lt, ne, or, sql } from 'drizzle-orm' import { listSearchSourcesContract, searchSourceCursorSchema, @@ -12,19 +12,13 @@ import { OrchestrationError } from '@/lib/core/orchestration/types' import { type ResourceOwner, resourceScopeFromOwner } from '@/lib/core/resource-scope' import { resourceScopeCondition } from '@/lib/core/resource-scope.server' import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' -import { knowledgeAccessCondition } from '@/lib/knowledge/access/predicate' -import { createKnowledgeAccessProvider } from '@/lib/knowledge/access/scope' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { resolveViewerConnectorMemberships } from '@/lib/knowledge/connectors/member-provisioning' -import { hasViewerMemberSyncError } from '@/lib/knowledge/connectors/viewer-member-sync-error' -import { resolveViewerSourceAccounts } from '@/lib/knowledge/connectors/viewer-source-accounts' import { SEARCH_SOURCE_CANDIDATE_PAGE_SIZE, SEARCH_SOURCE_PAGE_SIZE, } from '@/lib/knowledge/constants' -import { failedDocumentCondition } from '@/lib/knowledge/documents/processing-status' import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' import { describeSearchSource } from '@/lib/sim-search/source-identity' import { getConnectorMeta } from '@/connectors/registry' @@ -35,10 +29,9 @@ export interface ListSearchSourcesInput extends ResourceOwner { connectorType?: string excludeConnectorType?: string search?: string - mine?: boolean } -/** Viewer-safe setup and indexing state; source credentials and other members never leave this use case. */ +/** Bounded live source configuration; credentials and account identities remain private. */ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.listSearchSources, resolveContext: ({ input }: { input: ListSearchSourcesInput }) => @@ -56,7 +49,6 @@ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ connectorType: connectorType ?? '', ...(excludeConnectorType ? { excludeConnectorType } : {}), connectorId: input.connectorId ?? '', - mine: input.mine === true, order: 'newest', }) const cursor = (() => { @@ -84,12 +76,6 @@ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ accessMode: knowledgeConnector.accessMode, status: knowledgeConnector.status, memberSyncStatus: knowledgeConnector.memberSyncStatus, - lastSyncAt: knowledgeConnector.lastSyncAt, - hasRetainedSyncError: sql`${knowledgeConnector.lastSyncError} IS NOT NULL`, - hasViewerMemberSyncError: hasViewerMemberSyncError(userId), - lastMemberSyncAt: knowledgeConnector.lastMemberSyncAt, - credentialGroupId: knowledgeConnector.credentialGroupId, - credentialGroupOptionId: knowledgeConnector.credentialGroupOptionId, }) .from(knowledgeConnector) .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) @@ -122,38 +108,11 @@ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ if (candidates.length === 0) return { sources: [], nextCursor: null } const scanned = candidates.slice(0, SEARCH_SOURCE_CANDIDATE_PAGE_SIZE) - const [availability, memberships, viewers, approvals, accounts] = await Promise.all([ + const [availability, approvals] = await Promise.all([ resolveKnowledgeAccessAvailability(context), - resolveViewerConnectorMemberships({ - userId: userId, - workspaceId: context.workspaceId, - organizationId: context.organizationId, - connectors: scanned, - }), - db - .select({ emailVerified: user.emailVerified }) - .from(user) - .where(eq(user.id, userId)) - .limit(1), context.organizationId ? listOrganizationSearchApprovals(context.organizationId) : null, - context.organizationId - ? resolveViewerSourceAccounts({ - organizationId: context.organizationId, - userId: userId, - connectors: scanned, - }) - : new Map(), ]) - /** Owned grants stay manageable even when the source can no longer authorize Search. */ const matches = scanned.filter((row) => { - const membership = memberships.get(row.id) - if ( - input.mine && - (context.organizationId - ? !accounts.has(row.id) - : membership !== 'connected' && membership !== 'needs_reauth') - ) - return false const meta = getConnectorMeta(row.connectorType) const label = meta ? `${meta.name ?? row.connectorType} ${describeSearchSource(meta, row.sourceConfig)}` @@ -174,67 +133,6 @@ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ ).toString('base64url') : null if (rows.length === 0) return { sources: [], nextCursor } - const access = await createKnowledgeAccessProvider(principal, context).getForConnectors( - rows.map((row) => row.id) - ) - /** - * Per-source probes rather than one aggregate: the existence checks stop at the first - * visible document, and the failed and in-progress rows are few and indexed, so the cost no - * longer grows with every document the viewer can read. - */ - const readable = knowledgeAccessCondition(access) - const viewerDocument = (condition: SQL | undefined) => - and( - eq(document.connectorId, knowledgeConnector.id), - eq(document.enabled, true), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - condition, - readable - ) - const documentStates = await db - .select({ - connectorId: knowledgeConnector.id, - hasDocuments: sql`${exists( - db - .select({ id: document.id }) - .from(document) - .where( - viewerDocument( - and( - eq(document.processingStatus, 'completed'), - exists( - db - .select({ id: embedding.id }) - .from(embedding) - .where( - and(eq(embedding.documentId, document.id), eq(embedding.enabled, true)) - ) - ) - ) - ) - ) - )}`, - failedCount: sql`${db - .select({ count: sql`count(*)::int` }) - .from(document) - .where(viewerDocument(failedDocumentCondition()))}`, - isIndexing: sql`${exists( - db - .select({ id: document.id }) - .from(document) - .where(viewerDocument(inArray(document.processingStatus, ['pending', 'processing']))) - )}`, - }) - .from(knowledgeConnector) - .where( - inArray( - knowledgeConnector.id, - rows.map((row) => row.id) - ) - ) - const states = new Map(documentStates.map((state) => [state.connectorId, state])) return { nextCursor, @@ -252,7 +150,6 @@ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ row.status !== 'paused' && row.status !== 'disabled' && (row.accessMode !== 'members' || row.memberSyncStatus !== 'disabled') - const state = states.get(row.id) const source = { knowledgeBaseId: row.knowledgeBaseId, connectorId: row.id, @@ -266,39 +163,8 @@ export const listSearchSources = defineAuthorizedKnowledgeUseCase({ availability: available ? ('available' as const) : ('unavailable' as const), enabled, ...(approvals ? { approved: approvals.get(row.connectorType) ?? true } : {}), - isSyncing: - available && - enabled && - approvals?.get(row.connectorType) !== false && - (row.status === 'pending' || - row.status === 'syncing' || - (row.accessMode === 'members' && - (row.memberSyncStatus === 'pending' || row.memberSyncStatus === 'running')) || - state?.isIndexing === true), - lastSyncAt: - (row.accessMode === 'members' ? row.lastMemberSyncAt : row.lastSyncAt)?.toISOString() ?? - null, - hasSyncError: - row.status === 'error' || - row.hasRetainedSyncError === true || - row.hasViewerMemberSyncError === true || - (row.accessMode === 'members' && row.memberSyncStatus === 'error'), - hasViewerDocuments: available && state?.hasDocuments === true, - viewerFailedDocumentCount: available ? (state?.failedCount ?? 0) : 0, - viewerEmailVerified: viewers[0]?.emailVerified === true, - viewerAccounts: accounts.get(row.id) ?? [], } as const - return [ - { - ...source, - ...(connectionRequired - ? { - connectionRequired: true as const, - viewerMembership: available ? (memberships.get(row.id) ?? null) : null, - } - : { connectionRequired: false as const, viewerMembership: null }), - }, - ] + return [source] }), } }, diff --git a/apps/sim/lib/knowledge/application/search.test.ts b/apps/sim/lib/knowledge/application/search.test.ts index 48f1b6e06ee..4134e49312d 100644 --- a/apps/sim/lib/knowledge/application/search.test.ts +++ b/apps/sim/lib/knowledge/application/search.test.ts @@ -17,7 +17,6 @@ import { billingUsageMonitorMock, billingUsageMonitorMockFns, } from '@sim/testing/mocks/billing-usage-monitor.mock' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -41,7 +40,7 @@ import { import { permissionGroupsResolveMock } from '@sim/testing/mocks/permission-groups-resolve.mock' import { getMockPlatformEvent, telemetryMock } from '@sim/testing/mocks/telemetry.mock' import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' import { OrchestrationError } from '@/lib/core/orchestration/types' const hoisted = vi.hoisted(() => ({ @@ -52,15 +51,10 @@ const hoisted = vi.hoisted(() => ({ getTagDefinitions: vi.fn(), importProvenance: vi.fn(), rerank: vi.fn(), - recordActivity: vi.fn(), })) vi.mock('@/lib/core/telemetry', () => telemetryMock) -vi.mock('@/lib/knowledge/search/activity', () => ({ - recordOrganizationSearchActivity: hoisted.recordActivity, -})) - vi.mock('@/lib/knowledge/reranker', () => ({ hasRerankerCredential: hoisted.hasRerankerCredential, rerank: hoisted.rerank, @@ -236,42 +230,6 @@ describe('knowledge search application use case', () => { } ) - describe.each(['workspace', 'organization'] as const)('%s ranking policy', (scope) => { - beforeEach(() => { - if (scope === 'organization') { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - mocks.getKnowledgeBase.mockResolvedValue({ - ...knowledgeBase, - workspaceId: null, - organizationId: 'org-canonical', - isSearchIndex: true, - }) - queueTableRows(member, [{ role: 'member' }]) - } - }) - afterEach(resetEnvFlagsMock) - - it('meters only successful organization calls under the acting person', async () => { - await searchKnowledge.execute({ - principal: createSessionPrincipal(), - input: { knowledgeBaseIds: ['knowledge-1'], query: 'answer', topK: 10, surface: 'mcp' }, - }) - if (scope === 'organization') { - expect(mocks.recordActivity).toHaveBeenCalledExactlyOnceWith({ - organizationId: 'org-canonical', - userId: 'user-1', - surface: 'mcp', - results: expect.any(Array), - }) - } else { - expect(mocks.recordActivity).not.toHaveBeenCalled() - } - }) - - const _principal = createSessionPrincipal() - const _input = { knowledgeBaseIds: ['knowledge-1'], query: 'answer', topK: 10 } - }) - it('gates organization search using the persisted owner even when the request omits it', async () => { mocks.getKnowledgeBase.mockResolvedValue({ ...knowledgeBase, @@ -288,7 +246,6 @@ describe('knowledge search application use case', () => { input: { knowledgeBaseIds: ['knowledge-1'], query: 'answer', topK: 5 }, }) ).rejects.toThrow('Search is not enabled for this organization') - expect(mocks.recordActivity).not.toHaveBeenCalled() expect( knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable ).toHaveBeenCalledExactlyOnceWith('org-canonical') @@ -297,10 +254,8 @@ describe('knowledge search application use case', () => { expect(mocks.executeSearch).not.toHaveBeenCalled() }) - describe('while indexed organization search is dormant', () => { + describe('Search-marked knowledge bases named explicitly', () => { const principal = createSessionPrincipal() - beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: true })) - afterEach(resetEnvFlagsMock) it.each([ ['an organization', { workspaceId: null, organizationId: 'org-canonical' }], @@ -318,7 +273,7 @@ describe('knowledge search application use case', () => { }) expect(result.results).toHaveLength(1) expect(mocks.executeSearch).toHaveBeenCalledWith( - expect.objectContaining({ knowledgeBaseIds: ['knowledge-1'], indexedRetrieval: false }) + expect.objectContaining({ knowledgeBaseIds: ['knowledge-1'] }) ) }) }) diff --git a/apps/sim/lib/knowledge/application/search.ts b/apps/sim/lib/knowledge/application/search.ts index 319e2622ee1..515b3f73826 100644 --- a/apps/sim/lib/knowledge/application/search.ts +++ b/apps/sim/lib/knowledge/application/search.ts @@ -37,7 +37,6 @@ import type { ActiveKnowledgeBaseReference } from '@/lib/knowledge/knowledge-bas import { runWithKnowledgeModelInputProvenance } from '@/lib/knowledge/model-input-provenance' import { hasRerankerCredential, rerank } from '@/lib/knowledge/reranker' import type { RerankerStatus } from '@/lib/knowledge/reranker-models' -import { recordOrganizationSearchActivity } from '@/lib/knowledge/search/activity' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' import type { SearchResult } from '@/lib/knowledge/search/candidates' import { resolveKnowledgeSearchDefaults } from '@/lib/knowledge/search/defaults' @@ -53,7 +52,6 @@ import { import { getDocumentTagDefinitionsByKnowledgeBaseIds } from '@/lib/knowledge/tags/service' import type { DocumentTagDefinition } from '@/lib/knowledge/tags/types' import type { StructuredFilter } from '@/lib/knowledge/types' -import { usesIndexedRetrieval } from '@/lib/sim-search/indexed/gate' import { estimateTokenCount } from '@/lib/tokenization/estimators' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' import { getRerankModelPricing } from '@/providers/models' @@ -473,7 +471,6 @@ export async function runKnowledgeSearch({ } : undefined, structuredFilters: structuredFilters.length > 0 ? structuredFilters : undefined, - indexedRetrieval: usesIndexedRetrieval(context.knowledgeBases), }) ) @@ -770,7 +767,7 @@ export async function runKnowledgeSearch({ } } -/** What follows a completed search on every surface: the organization's activity record and the platform event. */ +/** Records the platform event after an authorized knowledge search. */ export async function afterKnowledgeSearch({ principal, context, @@ -782,17 +779,6 @@ export async function afterKnowledgeSearch({ input: Pick result: SearchKnowledgeResult }): Promise { - const actorUserId = resolvePrincipalSubjectUserId(principal) - if (context.organizationId && actorUserId) { - await measureSearchStage('activity_recording', () => - recordOrganizationSearchActivity({ - organizationId: context.organizationId, - userId: actorUserId, - surface: input.surface ?? 'other', - results: result.results, - }) - ) - } PlatformEvents.knowledgeBaseSearched({ knowledgeBaseId: result.knowledgeBaseId, knowledgeBaseIds: result.knowledgeBaseIds, diff --git a/apps/sim/lib/knowledge/application/sim-search.test.ts b/apps/sim/lib/knowledge/application/sim-search.test.ts index b06c4ef7087..d9bca6e9378 100644 --- a/apps/sim/lib/knowledge/application/sim-search.test.ts +++ b/apps/sim/lib/knowledge/application/sim-search.test.ts @@ -1,4 +1,4 @@ -import { knowledgeBase, knowledgeConnector, member } from '@sim/db/schema' +import { member } from '@sim/db/schema' import { queueTableRows, resetDbChainMock } from '@sim/testing' import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' import { auditMock } from '@sim/testing/mocks/audit.mock' @@ -6,7 +6,6 @@ import { credentialGroupsServiceMock, credentialGroupsServiceMockFns, } from '@sim/testing/mocks/credential-groups-service.mock' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' import { knowledgeAvailabilityMock, knowledgeAvailabilityMockFns, @@ -23,10 +22,6 @@ import { knowledgeEmbeddingsMock, knowledgeEmbeddingsMockFns, } from '@sim/testing/mocks/knowledge-embeddings.mock' -import { - knowledgeSearchIntegrationPolicyMock, - knowledgeSearchIntegrationPolicyMockFns, -} from '@sim/testing/mocks/knowledge-search-integration-policy.mock' import { knowledgeServiceMock, knowledgeServiceMockFns, @@ -36,14 +31,8 @@ import { permissionGroupsResolveMockFns, } from '@sim/testing/mocks/permission-groups-resolve.mock' import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { afterAll, afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' -const hoisted = vi.hoisted(() => ({ - createConnector: vi.fn(), - createApprovedSource: vi.fn(), - deleteConnector: vi.fn(), - enroll: vi.fn(), -})) vi.mock('@/lib/credential-groups/service', () => credentialGroupsServiceMock) vi.mock('@sim/audit', () => auditMock) @@ -58,35 +47,12 @@ vi.mock('@/lib/knowledge/service', () => knowledgeServiceMock) vi.mock('@/lib/knowledge/embeddings', () => knowledgeEmbeddingsMock) vi.mock('@/lib/knowledge/application/knowledge-bases', () => knowledgeBaseUseCasesMock) -vi.mock('@/lib/knowledge/search/integration-policy', () => knowledgeSearchIntegrationPolicyMock) - -vi.mock('@/lib/knowledge/application/connectors', () => ({ - createApprovedSearchSource: { execute: hoisted.createApprovedSource }, - createKnowledgeConnector: { execute: hoisted.createConnector }, - deleteKnowledgeConnector: { execute: hoisted.deleteConnector }, -})) - -vi.mock('@/lib/knowledge/application/connector-access', () => ({ - startKnowledgeConnectorMemberEnrollment: { execute: hoisted.enroll }, -})) - vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) vi.mock('@/lib/sim-search/connectors', () => ({ SIM_SEARCH_KNOWLEDGE_BASE_NAME: 'Sim Search', canConnectPersonally: (meta: { permissionScopedListing?: unknown }) => Boolean(meta.permissionScopedListing), - withSearchSourceDefaults: ( - meta: { searchDefaultSourceConfig?: Record }, - sourceConfig: Record = {} - ) => ({ ...(meta.searchDefaultSourceConfig ?? {}), ...sourceConfig }), - missingSetupFields: ( - meta: { configFields: Array<{ id: string; title: string; required?: boolean }> }, - sourceConfig: Record - ) => - meta.configFields.filter( - (field) => field.required && typeof sourceConfig[field.id] !== 'string' - ), })) vi.mock('@/connectors/registry', () => ({ @@ -124,21 +90,13 @@ vi.mock('@/connectors/registry', () => ({ })) import { OrchestrationError } from '@/lib/core/orchestration/types' -import { - configureSimSearchConnector, - connectSimSearchConnector, - prepareSearchSource, -} from '@/lib/knowledge/application/sim-search' +import { prepareSearchSource } from '@/lib/knowledge/application/sim-search' import { DEFAULT_PERMISSION_GROUP_CONFIG } from '@/lib/permission-groups/fields' -import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' const mocks = { - ...hoisted, ensureAccounts: credentialGroupsServiceMockFns.mockEnsureWorkspaceAccountsGroup, createOrganizationKnowledgeBase: knowledgeServiceMockFns.mockCreateAuthorizedKnowledgeBase, createKnowledgeBase: knowledgeBaseUseCasesMockFns.mockCreateKnowledgeBaseExecute, - deleteKnowledgeBase: knowledgeBaseUseCasesMockFns.mockDeleteKnowledgeBaseOperationExecute, - requireApproval: knowledgeSearchIntegrationPolicyMockFns.mockRequireOrganizationSearchApproval, } knowledgeEmbeddingsMockFns.mockGetConfiguredKbEmbedding.mockResolvedValue({ @@ -164,81 +122,20 @@ const workspaceContext = { } const principal = createSessionPrincipal() -const existingConnector = { knowledgeBaseId: 'kb-search', connectorId: 'connector-drive' } - -/** The first lookup runs before the coalesced creation and the second inside it. */ -function queueConnectorLookups(...results: Array) { - for (const result of results) { - queueTableRows(knowledgeConnector, result ? [result] : []) - } -} - -describe('connectSimSearchConnector', () => { +describe('prepareSearchSource', () => { afterAll(resetDbChainMock) - afterEach(resetEnvFlagsMock) beforeEach(() => { resetDbChainMock() - /** A Search source crawls into the search index, which only indexed organization search reads. */ - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue(workspaceContext) permissionGroupsResolveMockFns.mockGetUserPermissionConfig.mockResolvedValue( DEFAULT_PERMISSION_GROUP_CONFIG ) knowledgeAvailabilityMockFns.mockIsKnowledgeMemberAccessAvailable.mockResolvedValue(true) mocks.createKnowledgeBase.mockResolvedValue({ knowledgeBase: { id: 'kb-new' } }) - mocks.createConnector.mockResolvedValue({ connector: { id: 'connector-new' } }) - mocks.enroll.mockResolvedValue({ url: 'https://sim.test/enroll/token' }) mocks.ensureAccounts.mockResolvedValue({ id: 'accounts-group' }) }) - it('reuses a prepared account only when the source enrollment group and option match', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - queueTableRows(knowledgeConnector, [ - { ...existingConnector, credentialGroupId: 'group-1', credentialGroupOptionId: 'option-1' }, - ]) - await expect( - configureSimSearchConnector.execute({ - principal, - input: { - workspaceId: 'workspace-1', - connectorType: 'google_drive', - memberCredentialBinding: { - credentialGroupId: 'group-1', - credentialGroupOptionId: 'option-1', - }, - }, - }) - ).resolves.toEqual(existingConnector) - expect(mocks.enroll).not.toHaveBeenCalled() - }) - - it.each([ - { credentialGroupId: 'other-group', credentialGroupOptionId: 'option-1' }, - { credentialGroupId: 'group-1', credentialGroupOptionId: 'other-option' }, - ])( - 'refuses an existing source with a mismatched prepared account binding %#', - async (sourceBinding) => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - queueTableRows(knowledgeConnector, [{ ...existingConnector, ...sourceBinding }]) - await expect( - configureSimSearchConnector.execute({ - principal, - input: { - workspaceId: 'workspace-1', - connectorType: 'google_drive', - memberCredentialBinding: { - credentialGroupId: 'group-1', - credentialGroupOptionId: 'option-1', - }, - }, - }) - ).rejects.toMatchObject({ code: 'conflict' }) - expect(mocks.enroll).not.toHaveBeenCalled() - expect(mocks.createConnector).not.toHaveBeenCalled() - } - ) - it('requires an administrator before preparing a managed source', async () => { workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') await expect( @@ -260,153 +157,20 @@ describe('connectSimSearchConnector', () => { ).rejects.toMatchObject({ code: 'validation' }) expect(mocks.createKnowledgeBase).not.toHaveBeenCalled() }) - - it('refuses while indexed organization search is dormant, before creating anything', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('admin') - queueConnectorLookups(null) - - await expect( - connectSimSearchConnector.execute({ - principal, - input: { workspaceId: 'workspace-1', connectorType: 'google_drive' }, - }) - ).rejects.toBeInstanceOf(SearchIndexDormantError) - expect(mocks.createKnowledgeBase).not.toHaveBeenCalled() - expect(mocks.createConnector).not.toHaveBeenCalled() - expect(mocks.enroll).not.toHaveBeenCalled() - }) - - it('refuses before creating anything when per-member access is unavailable', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('admin') - knowledgeAvailabilityMockFns.mockIsKnowledgeMemberAccessAvailable.mockResolvedValue(false) - queueConnectorLookups(null) - - await expect( - connectSimSearchConnector.execute({ - principal, - input: { workspaceId: 'workspace-1', connectorType: 'google_drive' }, - }) - ).rejects.toMatchObject({ code: 'validation' }) - expect(mocks.createKnowledgeBase).not.toHaveBeenCalled() - expect(mocks.createConnector).not.toHaveBeenCalled() - }) - - it('uses the source returned by transaction-level creation reuse without deleting another source', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('admin') - mocks.createKnowledgeBase.mockRejectedValueOnce(new Error('Duplicate knowledge base name')) - mocks.createConnector.mockResolvedValueOnce({ - connector: { id: existingConnector.connectorId }, - reused: true, - }) - queueTableRows(knowledgeBase, []) - queueTableRows(knowledgeBase, [{ id: existingConnector.knowledgeBaseId }]) - queueConnectorLookups(null, null) - const result = await connectSimSearchConnector.execute({ - principal, - input: { workspaceId: 'workspace-1', connectorType: 'google_drive' }, - }) - expect(mocks.deleteConnector).not.toHaveBeenCalled() - expect(mocks.createConnector).toHaveBeenCalledWith( - expect.objectContaining({ - input: expect.objectContaining({ reuseSearchSource: true }), - }) - ) - expect(result).toEqual({ ...existingConnector, url: 'https://sim.test/enroll/token' }) - }) - - it('requires an explicit source when legacy duplicate settings make selection ambiguous', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - queueTableRows(knowledgeConnector, [ - { ...existingConnector, connectorId: 'one', sourceConfig: {} }, - { ...existingConnector, connectorId: 'two', sourceConfig: {} }, - ]) - await expect( - connectSimSearchConnector.execute({ - principal, - input: { workspaceId: 'workspace-1', connectorType: 'google_drive' }, - }) - ).rejects.toMatchObject({ code: 'conflict' }) - expect(mocks.enroll).not.toHaveBeenCalled() - }) - - it.each([ - { name: 'missing or outside the canonical index', rows: [], config: undefined }, - { - name: 'different settings', - rows: [{ ...existingConnector, sourceConfig: { spaceKey: 'OPS' } }], - config: { spaceKey: 'ENG' }, - }, - ])('rejects an explicitly selected source that is $name', async ({ rows, config }) => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - queueTableRows(knowledgeConnector, rows) - await expect( - connectSimSearchConnector.execute({ - principal, - input: { - workspaceId: 'workspace-1', - connectorType: 'confluence', - connectorId: existingConnector.connectorId, - sourceConfig: config, - }, - }) - ).rejects.toMatchObject({ code: 'not_found' }) - expect(mocks.enroll).not.toHaveBeenCalled() - expect(mocks.createConnector).not.toHaveBeenCalled() - }) }) describe('organization Search setup', () => { const owner = { organizationId: 'org-1' } - afterEach(resetEnvFlagsMock) beforeEach(() => { resetDbChainMock() - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockResolvedValue(owner) knowledgeAvailabilityMockFns.mockIsKnowledgeMemberAccessAvailable.mockResolvedValue(true) mocks.ensureAccounts.mockResolvedValue({ id: 'org-accounts' }) mocks.createOrganizationKnowledgeBase.mockResolvedValue({ id: 'org-index' }) - mocks.enroll.mockResolvedValue({ url: 'https://fixture.test/enroll' }) }) function asRole(role: string) { for (let i = 0; i < 4; i++) queueTableRows(member, [{ role }]) } - it('lets an approved member create a personal source without impersonating an admin', async () => { - asRole('member') - queueTableRows(knowledgeBase, []) - mocks.createApprovedSource.mockResolvedValue({ connector: { id: 'approved-source' } }) - await expect( - connectSimSearchConnector.execute({ - principal, - input: { ...owner, connectorType: 'google_drive' }, - }) - ).resolves.toMatchObject({ knowledgeBaseId: 'org-index', connectorId: 'approved-source' }) - expect(mocks.requireApproval).toHaveBeenCalledWith('org-1', 'google_drive') - expect(mocks.createApprovedSource).toHaveBeenCalledWith( - expect.objectContaining({ - principal, - input: { - knowledgeBaseId: 'org-index', - assertedOrganizationId: 'org-1', - connectorType: 'google_drive', - sourceConfig: {}, - }, - }) - ) - expect(mocks.createConnector).not.toHaveBeenCalled() - }) - it('refuses unapproved members before provisioning anything', async () => { - asRole('member') - mocks.requireApproval.mockRejectedValueOnce(new Error('Approval required')) - await expect( - connectSimSearchConnector.execute({ - principal, - input: { ...owner, connectorType: 'google_drive' }, - }) - ).rejects.toThrow('Approval required') - expect(mocks.createOrganizationKnowledgeBase).not.toHaveBeenCalled() - expect(mocks.enroll).not.toHaveBeenCalled() - }) it('refuses organization source setup by a member before provisioning accounts or an index', async () => { asRole('member') await expect( @@ -418,17 +182,6 @@ describe('organization Search setup', () => { expect(mocks.ensureAccounts).not.toHaveBeenCalled() expect(mocks.createOrganizationKnowledgeBase).not.toHaveBeenCalled() }) - it('refuses a former organization member before looking up any configured source', async () => { - queueTableRows(member, []) - await expect( - connectSimSearchConnector.execute({ - principal, - input: { ...owner, connectorType: 'google_drive' }, - }) - ).rejects.toThrow('Organization not found') - expect(mocks.enroll).not.toHaveBeenCalled() - expect(mocks.createOrganizationKnowledgeBase).not.toHaveBeenCalled() - }) }) describe('Mothership Search setup authorization', () => { @@ -461,7 +214,6 @@ describe('Mothership Search setup authorization', () => { else await expect(action).resolves.toBeUndefined() expect(mocks.createOrganizationKnowledgeBase).not.toHaveBeenCalled() expect(mocks.ensureAccounts).not.toHaveBeenCalled() - expect(mocks.enroll).not.toHaveBeenCalled() } ) }) diff --git a/apps/sim/lib/knowledge/application/sim-search.ts b/apps/sim/lib/knowledge/application/sim-search.ts index 75c10b00e4c..90e3f839efe 100644 --- a/apps/sim/lib/knowledge/application/sim-search.ts +++ b/apps/sim/lib/knowledge/application/sim-search.ts @@ -1,13 +1,6 @@ import { type Principal, resolvePrincipalSubjectUserId } from '@sim/auth/principal' -import { db } from '@sim/db' -import { knowledgeBase, knowledgeConnector } from '@sim/db/schema' -import { and, eq, isNull } from 'drizzle-orm' import { coalesceLocally } from '@/lib/concurrency/singleflight' import { requireOrganizationMembership } from '@/lib/core/application/organization-authorization' -import { - InsufficientWorkspacePermissionsError, - requireCurrentHumanRole, -} from '@/lib/core/application/workspace-authorization' import { OrchestrationError, type OrchestrationRequestContext, @@ -15,13 +8,10 @@ import { import { type ResourceOwner, type ResourceScope, - resourceScopeFields, resourceScopeFromOwner, resourceScopeKey, } from '@/lib/core/resource-scope' -import { resourceScopeCondition } from '@/lib/core/resource-scope.server' import { generateRequestId } from '@/lib/core/utils/request' -import type { CredentialGroupConnectionIntent } from '@/lib/credential-groups/oauth-intent' import { ensureWorkspaceAccountsGroup } from '@/lib/credential-groups/service' import { requireKnowledgeMemberAccessAvailable, @@ -29,116 +19,20 @@ import { requireSourceMirroredAccessAvailable, } from '@/lib/knowledge/access/availability' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { startKnowledgeConnectorMemberEnrollment } from '@/lib/knowledge/application/connector-access' -import { - createApprovedSearchSource, - createKnowledgeConnector, -} from '@/lib/knowledge/application/connectors' -import { - type KnowledgeOrganizationContext, - type KnowledgeWorkspaceContext, - resolveKnowledgeOwnerContext, -} from '@/lib/knowledge/application/contexts' +import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' import { createKnowledgeBase } from '@/lib/knowledge/application/knowledge-bases' import { knowledgeOperations } from '@/lib/knowledge/application/operations' import { DEFAULT_CHUNKING_CONFIG } from '@/lib/knowledge/constants' import { getConfiguredKbEmbedding } from '@/lib/knowledge/embeddings' -import { requireOrganizationSearchApproval } from '@/lib/knowledge/search/integration-policy' import { findSearchIndex } from '@/lib/knowledge/search/search-index' import { createAuthorizedKnowledgeBase } from '@/lib/knowledge/service' -import { - canConnectPersonally, - missingSetupFields, - SIM_SEARCH_KNOWLEDGE_BASE_NAME, - withSearchSourceDefaults, -} from '@/lib/sim-search/connectors' -import { SIM_SEARCH_SYNC_INTERVAL_MINUTES } from '@/lib/sim-search/constants' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import { searchSourceIdentity } from '@/lib/sim-search/source-identity' +import { canConnectPersonally, SIM_SEARCH_KNOWLEDGE_BASE_NAME } from '@/lib/sim-search/connectors' import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' const SIM_SEARCH_KNOWLEDGE_BASE_DESCRIPTION = 'What each person can open in the sources they connected, searched as them.' -export interface ConnectSimSearchConnectorInput extends ResourceOwner { - /** `CONNECTOR_META_REGISTRY` key of the source to connect. */ - connectorType: string - /** An existing source selected by the person, scoped to the canonical workspace index. */ - connectorId?: string - /** Source settings identify a compatible configuration when creating or reusing a source. */ - sourceConfig?: Record - /** Correlates a direct provider authorization with the initiating Integrations tab. */ - connectionIntent?: CredentialGroupConnectionIntent - oauthCompletionId?: string - /** Internal setup assertion; a reused source must use the viewer's prepared enrollment option. */ - memberCredentialBinding?: { credentialGroupId: string; credentialGroupOptionId: string } -} - -export interface ConnectSimSearchConnectorResult { - knowledgeBaseId: string - connectorId: string - /** The invitation link or provider authorization URL for the caller's own account. */ - url: string -} - -async function findSimSearchConnector(input: ConnectSimSearchConnectorInput) { - const rows = await db - .select({ - knowledgeBaseId: knowledgeBase.id, - connectorId: knowledgeConnector.id, - sourceConfig: knowledgeConnector.sourceConfig, - credentialGroupId: knowledgeConnector.credentialGroupId, - credentialGroupOptionId: knowledgeConnector.credentialGroupOptionId, - }) - .from(knowledgeConnector) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) - .where( - and( - resourceScopeCondition(knowledgeBase, resourceScopeFromOwner(input)), - eq(knowledgeBase.isSearchIndex, true), - isNull(knowledgeBase.deletedAt), - eq(knowledgeConnector.connectorType, input.connectorType), - input.connectorId ? eq(knowledgeConnector.id, input.connectorId) : undefined, - eq(knowledgeConnector.accessMode, 'members'), - isNull(knowledgeConnector.archivedAt), - isNull(knowledgeConnector.deletedAt) - ) - ) - const meta = CONNECTOR_META_REGISTRY[input.connectorType]! - const identity = searchSourceIdentity(meta, input.sourceConfig ?? {}) - const matches = rows.filter((row) => - input.connectorId && input.sourceConfig === undefined - ? true - : searchSourceIdentity(meta, row.sourceConfig) === identity - ) - if (input.connectorId && matches.length === 0) { - throw new OrchestrationError( - 'not_found', - 'This Search source is unavailable or its settings have changed' - ) - } - if (matches.length > 1) { - throw new OrchestrationError( - 'conflict', - 'Several Search sources use these settings. Choose the source you want to connect.' - ) - } - const match = matches[0] - if ( - match && - input.memberCredentialBinding && - (match.credentialGroupId !== input.memberCredentialBinding.credentialGroupId || - match.credentialGroupOptionId !== input.memberCredentialBinding.credentialGroupOptionId) - ) { - throw new OrchestrationError( - 'conflict', - 'This source uses a different account configuration. Ask an admin to review its settings.' - ) - } - return match ? { knowledgeBaseId: match.knowledgeBaseId, connectorId: match.connectorId } : null -} - -/** Resolves the current workspace search index without creating one. */ +/** Resolves the current Search source configuration without creating it. */ export const readSearchIndex = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.readSearchIndex, resolveContext: ({ input }: { input: ResourceOwner }) => resolveKnowledgeOwnerContext(input), @@ -151,12 +45,11 @@ export const readSearchIndex = defineAuthorizedKnowledgeUseCase({ }, }) -/** Admin setup adopts legacy indexes; reads and deletion always use the persisted index marker. */ +/** Search service sources share a marked knowledge base that stores their configuration. */ async function ensureSearchKnowledgeBase( scope: ResourceScope, principal: Principal, - request?: OrchestrationRequestContext, - approvedConnectorType?: string + request?: OrchestrationRequestContext ): Promise { return coalesceLocally(`sim-search:base:${resourceScopeKey(scope)}`, async () => { const existing = await findSearchIndex(scope) @@ -165,12 +58,10 @@ async function ensureSearchKnowledgeBase( if (scope.kind === 'organization') { const userId = resolvePrincipalSubjectUserId(principal) if (!userId) throw new OrchestrationError('forbidden', 'Sign in to configure sources') - if (approvedConnectorType) - await requireOrganizationSearchApproval(scope.organizationId, approvedConnectorType) await requireOrganizationMembership( principal, scope.organizationId, - approvedConnectorType ? 'member' : 'admin', + 'admin', 'knowledge.create' ) const embedding = await getConfiguredKbEmbedding() @@ -190,19 +81,6 @@ async function ensureSearchKnowledgeBase( return created.id } const workspaceId = scope.workspaceId - const [legacy] = await db - .update(knowledgeBase) - .set({ isSearchIndex: true, updatedAt: new Date() }) - .where( - and( - eq(knowledgeBase.workspaceId, workspaceId), - eq(knowledgeBase.name, SIM_SEARCH_KNOWLEDGE_BASE_NAME), - eq(knowledgeBase.isSearchIndex, false), - isNull(knowledgeBase.deletedAt) - ) - ) - .returning({ id: knowledgeBase.id }) - if (legacy) return legacy.id const created = await createKnowledgeBase.execute({ principal, input: { @@ -223,7 +101,7 @@ async function ensureSearchKnowledgeBase( }) } -/** Prepares the shared search index before the existing connector setup modal collects credentials. */ +/** Prepares shared Search source configuration before collecting connector credentials. */ export const prepareSearchSource = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.prepareSearchSource, resolveContext: ({ @@ -255,166 +133,3 @@ export const prepareSearchSource = defineAuthorizedKnowledgeUseCase({ } }, }) - -/** - * The first connect of a source turns it on for the whole workspace, which is - * an admin decision the same way a members-mode connector is. Refused with - * the way forward rather than the nested operations' generic role error, so a - * reader learns whom to ask and for what. - */ -async function requireSimSearchSetupAdmin( - principal: Principal, - context: KnowledgeWorkspaceContext | KnowledgeOrganizationContext, - sourceName: string -): Promise { - try { - if (context.organizationId) - await requireOrganizationMembership( - principal, - context.organizationId, - 'admin', - 'knowledge.use' - ) - else if (context.workspaceId) { - const userId = resolvePrincipalSubjectUserId(principal) - if (!userId) throw new OrchestrationError('forbidden', 'Sign in to configure sources') - await requireCurrentHumanRole(userId, context, 'admin') - } - } catch (error) { - if (!(error instanceof InsufficientWorkspacePermissionsError)) throw error - throw new OrchestrationError( - 'forbidden', - `${sourceName} is not connected in this workspace yet. Ask a workspace admin to connect ${sourceName} first; after that everyone connects their own account.` - ) - } -} - -/** - * Connects the caller's account, creating the owner's index and personal source - * when needed. Organization members may set up approved integrations; workspace - * setup requires an admin. OAuth completion queues indexing for the member. - * - * The database identifies one active search index per owner. Local - * singleflight also coalesces repeated setup clicks for each source; concurrent - * source creation is serialized by the connector insert transaction before enrollment. - * The source crawls into the owner's search index, so it is refused while indexed organization - * search is dormant. - */ -export const configureSimSearchConnector = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.simSearchConnect, - resolveContext: ({ input }: { input: ConnectSimSearchConnectorInput }) => - resolveKnowledgeOwnerContext(input), - async execute({ principal, input, context, request }) { - assertIndexedOrgSearchEnabled() - const meta = CONNECTOR_META_REGISTRY[input.connectorType] - if (!meta || !canConnectPersonally(meta)) { - throw new OrchestrationError( - 'validation', - 'This source cannot be connected per person; a workspace admin sets it up from a knowledge base' - ) - } - const scope = resourceScopeFromOwner(context) - const owner = resourceScopeFields(scope) - const workspaceId = context.workspaceId - if (context.organizationId) { - await requireOrganizationSearchApproval(context.organizationId, input.connectorType) - } - /** - * Defaults are applied before any lookup so a second person connecting with - * an untouched form lands on the source the first connection created. - */ - const sourceConfig = - input.connectorId && input.sourceConfig === undefined - ? undefined - : withSearchSourceDefaults(meta, input.sourceConfig) - let target = await findSimSearchConnector({ ...input, sourceConfig, ...owner }) - if (!target) { - const userId = resolvePrincipalSubjectUserId(principal) - if (!userId) throw new OrchestrationError('forbidden', 'Sign in to connect your account') - const sourceConfig = withSearchSourceDefaults(meta, input.sourceConfig) - const missing = missingSetupFields(meta, sourceConfig) - if (missing.length > 0) { - throw new OrchestrationError( - 'validation', - `${meta.name} needs ${missing.map((field) => field.title).join(' and ')} to connect` - ) - } - /** - * Judged before anything is created: the connector creation below checks - * the same availability, but only after the knowledge base exists. - */ - await Promise.all([ - requireKnowledgeMemberAccessAvailable(owner), - context.organizationId - ? Promise.resolve() - : requireSimSearchSetupAdmin(principal, context, meta.name), - ]) - const knowledgeBaseId = await ensureSearchKnowledgeBase( - scope, - principal, - request, - context.organizationId ? input.connectorType : undefined - ) - target = await coalesceLocally( - `sim-search:connect:${resourceScopeKey(scope)}:${input.connectorType}:${searchSourceIdentity(meta, sourceConfig)}`, - async () => { - const existing = await findSimSearchConnector({ ...input, ...owner }) - if (existing) return existing - if (context.organizationId) { - const created = await createApprovedSearchSource.execute({ - principal, - input: { - knowledgeBaseId, - assertedOrganizationId: context.organizationId, - connectorType: input.connectorType, - sourceConfig, - }, - request, - }) - return { knowledgeBaseId, connectorId: created.connector.id } - } - const created = await createKnowledgeConnector.execute({ - principal, - input: { - knowledgeBaseId, - assertedWorkspaceId: workspaceId, - assertedOrganizationId: context.organizationId, - connectorType: input.connectorType, - sourceConfig, - syncIntervalMinutes: SIM_SEARCH_SYNC_INTERVAL_MINUTES, - accessMode: 'members', - reuseSearchSource: true, - source: 'ui', - }, - request, - }) - return { knowledgeBaseId, connectorId: created.connector.id } - } - ) - } - return target - }, -}) - -/** Creates or reuses the source, then authorizes the member when setup has not already done so. */ -export const connectSimSearchConnector = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.simSearchConnect, - resolveContext: ({ input }: { input: ConnectSimSearchConnectorInput }) => - resolveKnowledgeOwnerContext(input), - async execute({ principal, input, context, request }): Promise { - const target = await configureSimSearchConnector.execute({ principal, input, request }) - const { url } = await startKnowledgeConnectorMemberEnrollment.execute({ - principal, - input: { - knowledgeBaseId: target.knowledgeBaseId, - connectorId: target.connectorId, - assertedWorkspaceId: context.workspaceId, - assertedOrganizationId: context.organizationId, - oauthCompletionId: input.oauthCompletionId, - connectionIntent: input.connectionIntent, - }, - request, - }) - return { ...target, url } - }, -}) diff --git a/apps/sim/lib/knowledge/connectors/indexing-policy.test.ts b/apps/sim/lib/knowledge/connectors/indexing-policy.test.ts deleted file mode 100644 index b30a175ce85..00000000000 --- a/apps/sim/lib/knowledge/connectors/indexing-policy.test.ts +++ /dev/null @@ -1,27 +0,0 @@ -import { knowledgeBase } from '@sim/db/schema' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing' -import { eq } from 'drizzle-orm' -import { beforeEach, describe, expect, it, vi } from 'vitest' -import { - connectorIndexingCondition, - requiresConnectorIndexing, -} from '@/lib/knowledge/connectors/indexing-policy' - -describe('connector indexing policy', () => { - beforeEach(resetEnvFlagsMock) - - it('preserves ordinary knowledge-base indexing when Search uses live APIs', () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) - expect(requiresConnectorIndexing(false)).toBe(true) - expect(requiresConnectorIndexing(true)).toBe(false) - connectorIndexingCondition() - expect(vi.mocked(eq)).toHaveBeenCalledWith(knowledgeBase.isSearchIndex, false) - }) - - it('preserves indexed Search and existing schedules when live search is disabled', () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - expect(requiresConnectorIndexing(true)).toBe(true) - expect(requiresConnectorIndexing(false)).toBe(true) - expect(connectorIndexingCondition()).toBeUndefined() - }) -}) diff --git a/apps/sim/lib/knowledge/connectors/indexing-policy.ts b/apps/sim/lib/knowledge/connectors/indexing-policy.ts index bbb61df298f..ac2ee44afcf 100644 --- a/apps/sim/lib/knowledge/connectors/indexing-policy.ts +++ b/apps/sim/lib/knowledge/connectors/indexing-policy.ts @@ -1,13 +1,12 @@ import { knowledgeBase } from '@sim/db/schema' import { eq } from 'drizzle-orm' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' /** Federated Search keeps source configuration but does not crawl content into a knowledge base. */ export function requiresConnectorIndexing(isSearchIndex?: boolean | null): boolean { - return isIndexedOrgSearchEnabled() || isSearchIndex !== true + return isSearchIndex !== true } /** Keeps federated sources out of bounded indexing scheduler pages. */ export function connectorIndexingCondition() { - return isIndexedOrgSearchEnabled() ? undefined : eq(knowledgeBase.isSearchIndex, false) + return eq(knowledgeBase.isSearchIndex, false) } diff --git a/apps/sim/lib/knowledge/connectors/organization-account-indexing.test.ts b/apps/sim/lib/knowledge/connectors/organization-account-indexing.test.ts deleted file mode 100644 index 0882dc75a6f..00000000000 --- a/apps/sim/lib/knowledge/connectors/organization-account-indexing.test.ts +++ /dev/null @@ -1,110 +0,0 @@ -import { credentialGroup, knowledgeBase, knowledgeConnector } from '@sim/db/schema' -import { - dbChainMockFns, - flattenMockConditions, - queueTableRows, - resetDbChainMock, -} from '@sim/testing' -import { - knowledgeMemberAccessMock, - knowledgeMemberAccessMockFns, -} from '@sim/testing/mocks/knowledge-member-access.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -vi.mock('@/lib/knowledge/connectors/member-access', () => knowledgeMemberAccessMock) - -import { setOrganizationAccountIndexing } from '@/lib/knowledge/connectors/organization-account-indexing' - -const validateBinding = knowledgeMemberAccessMockFns.mockValidateKnowledgeConnectorMembersBinding - -const input = { - organizationId: 'org-1', - credentialGroupId: 'group-1', - optionId: 'gmail-option', - enabled: false, -} -const group = { - status: 'active', - options: [{ id: 'gmail-option', status: 'active', provider: 'gmail' }], -} -const source = { - id: 'source-1', - knowledgeBaseId: 'kb-1', - status: 'active', - memberSyncStatus: 'idle', - sourceConfig: {}, -} - -describe('organization provider indexing changes', () => { - beforeEach(() => { - resetDbChainMock() - validateBinding.mockReturnValue({ ok: true }) - queueTableRows(credentialGroup, [group]) - }) - - it('pauses all bound sources in one transaction, cancels queued work and retains documents', async () => { - queueTableRows(knowledgeConnector, [ - source, - { ...source, id: 'source-2', memberSyncStatus: 'pending' }, - ]) - await expect(setOrganizationAccountIndexing(input)).resolves.toMatchObject({ - enabled: false, - changed: true, - knowledgeBaseIds: ['kb-1'], - }) - expect(dbChainMockFns.set).toHaveBeenCalledWith( - expect.objectContaining({ - status: 'paused', - nextMemberSyncAt: null, - memberSyncLockToken: null, - memberSyncStatus: 'idle', - }) - ) - expect(dbChainMockFns.update).toHaveBeenCalledExactlyOnceWith(knowledgeConnector) - expect(dbChainMockFns.delete).not.toHaveBeenCalled() - const predicates = dbChainMockFns.where.mock.calls.flatMap(([condition]) => - flattenMockConditions(condition) - ) - for (const expected of [ - { type: 'eq', left: knowledgeBase.organizationId, right: 'org-1' }, - { type: 'eq', left: knowledgeBase.isSearchIndex, right: true }, - { type: 'eq', left: knowledgeConnector.credentialGroupId, right: 'group-1' }, - { type: 'eq', left: knowledgeConnector.credentialGroupOptionId, right: 'gmail-option' }, - { type: 'eq', left: knowledgeConnector.accessMode, right: 'members' }, - ]) - expect(predicates).toContainEqual(expected) - }) - - it('refuses the entire change when one source has an active run', async () => { - queueTableRows(knowledgeConnector, [ - source, - { ...source, id: 'source-2', memberSyncStatus: 'running' }, - ]) - await expect(setOrganizationAccountIndexing(input)).rejects.toMatchObject({ code: 'conflict' }) - expect(dbChainMockFns.update).not.toHaveBeenCalled() - }) - - it('refuses a stale or foreign option before touching any source', async () => { - await expect( - setOrganizationAccountIndexing({ ...input, optionId: 'foreign-option' }) - ).rejects.toMatchObject({ code: 'not_found' }) - expect(dbChainMockFns.update).not.toHaveBeenCalled() - }) - - it('requires setup instead of reporting indexing enabled without a source', async () => { - queueTableRows(knowledgeConnector, []) - await expect(setOrganizationAccountIndexing({ ...input, enabled: true })).rejects.toMatchObject( - { code: 'not_found' } - ) - expect(dbChainMockFns.update).not.toHaveBeenCalled() - }) - - it('rejects re-enabling a source whose scopes no longer meet the ingestion requirements', async () => { - queueTableRows(knowledgeConnector, [{ ...source, status: 'paused' }]) - validateBinding.mockReturnValue({ ok: false, message: 'Reconnect with the required scopes' }) - await expect(setOrganizationAccountIndexing({ ...input, enabled: true })).rejects.toThrow( - 'Reconnect with the required scopes' - ) - expect(dbChainMockFns.update).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/knowledge/connectors/organization-account-indexing.ts b/apps/sim/lib/knowledge/connectors/organization-account-indexing.ts deleted file mode 100644 index a1e34897ede..00000000000 --- a/apps/sim/lib/knowledge/connectors/organization-account-indexing.ts +++ /dev/null @@ -1,136 +0,0 @@ -import { db } from '@sim/db' -import { credentialGroup, knowledgeBase, knowledgeConnector } from '@sim/db/schema' -import { isPlainRecord } from '@sim/utils/object' -import { and, asc, eq, inArray, isNull } from 'drizzle-orm' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { resourceScopeCondition } from '@/lib/core/resource-scope.server' -import { getCredentialGroupIndexingConnector } from '@/lib/credential-groups/indexing' -import { ORGANIZATION_ACCOUNT_INDEXING_SOURCE_LIMIT } from '@/lib/credential-groups/limits' -import { isCredentialGroupProvider } from '@/lib/credential-groups/providers' -import { validateKnowledgeConnectorMembersBinding } from '@/lib/knowledge/connectors/member-access' - -export interface SetOrganizationAccountIndexingInput { - organizationId: string - credentialGroupId: string - optionId: string - enabled: boolean -} - -/** Changes every Search source bound to this org option atomically, respecting running sync leases. */ -export async function setOrganizationAccountIndexing(input: SetOrganizationAccountIndexingInput) { - const scope = { kind: 'organization' as const, organizationId: input.organizationId } - return db.transaction(async (tx) => { - const [group] = await tx - .select({ options: credentialGroup.options, status: credentialGroup.status }) - .from(credentialGroup) - .where( - and( - eq(credentialGroup.id, input.credentialGroupId), - resourceScopeCondition(credentialGroup, scope) - ) - ) - .limit(1) - .for('update') - if (!group) - throw new OrchestrationError('not_found', 'Organization connected accounts were not found') - const option = group.options.find( - (candidate) => candidate.id === input.optionId && candidate.status === 'active' - ) - if (!option || !isCredentialGroupProvider(option.provider)) - throw new OrchestrationError('not_found', 'Connected account provider was not found') - const connector = getCredentialGroupIndexingConnector(option.provider) - if (!connector) - throw new OrchestrationError('validation', 'Indexing is not supported for this provider') - const sources = await tx - .select({ - id: knowledgeConnector.id, - knowledgeBaseId: knowledgeBase.id, - sourceConfig: knowledgeConnector.sourceConfig, - status: knowledgeConnector.status, - memberSyncStatus: knowledgeConnector.memberSyncStatus, - }) - .from(knowledgeConnector) - .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) - .where( - and( - resourceScopeCondition(knowledgeBase, scope), - eq(knowledgeBase.isSearchIndex, true), - isNull(knowledgeBase.deletedAt), - eq(knowledgeConnector.credentialGroupId, input.credentialGroupId), - eq(knowledgeConnector.credentialGroupOptionId, option.id), - eq(knowledgeConnector.connectorType, connector.type), - eq(knowledgeConnector.accessMode, 'members'), - isNull(knowledgeConnector.archivedAt), - isNull(knowledgeConnector.deletedAt) - ) - ) - .orderBy(asc(knowledgeConnector.id)) - .limit(ORGANIZATION_ACCOUNT_INDEXING_SOURCE_LIMIT + 1) - .for('update') - if (sources.length > ORGANIZATION_ACCOUNT_INDEXING_SOURCE_LIMIT) - throw new OrchestrationError('validation', 'Too many indexing sources for one provider') - if (!sources.length) - throw new OrchestrationError('not_found', 'Set up an indexing source for this provider first') - const changed = sources.filter((source) => - input.enabled - ? source.status === 'paused' || - source.status === 'disabled' || - source.memberSyncStatus === 'disabled' - : source.status !== 'paused' - ) - if ( - changed.some((source) => source.status === 'syncing' || source.memberSyncStatus === 'running') - ) - throw new OrchestrationError( - 'conflict', - 'Indexing is running. Wait for the current sync to finish, then try again.' - ) - if (input.enabled) { - for (const source of changed) { - if (!isPlainRecord(source.sourceConfig)) - throw new OrchestrationError('validation', 'Indexing source settings are invalid') - const validation = validateKnowledgeConnectorMembersBinding({ - connectorMeta: connector.meta, - group, - credentialGroupOptionId: option.id, - sourceConfig: source.sourceConfig, - }) - if (!validation.ok) throw new OrchestrationError('validation', validation.message) - } - } - if (changed.length) { - const now = new Date() - await tx - .update(knowledgeConnector) - .set({ - status: input.enabled ? 'active' : 'paused', - memberSyncStatus: 'idle', - memberSyncLockToken: null, - memberSyncLockLeaseAt: null, - syncLockToken: null, - syncLockLeaseAt: null, - nextMemberSyncAt: input.enabled ? now : null, - updatedAt: now, - ...(input.enabled - ? { consecutiveFailures: 0, lastSyncError: null, lastMemberSyncError: null } - : {}), - }) - .where( - and( - inArray( - knowledgeConnector.id, - changed.map((source) => source.id) - ), - eq(knowledgeConnector.credentialGroupId, input.credentialGroupId), - eq(knowledgeConnector.credentialGroupOptionId, option.id) - ) - ) - } - return { - enabled: input.enabled, - changed: changed.length > 0, - providerName: connector.meta.name, - knowledgeBaseIds: [...new Set(sources.map((source) => source.knowledgeBaseId))], - } - }) -} diff --git a/apps/sim/lib/knowledge/connectors/viewer-member-sync-error.ts b/apps/sim/lib/knowledge/connectors/viewer-member-sync-error.ts deleted file mode 100644 index 1d8e1919cde..00000000000 --- a/apps/sim/lib/knowledge/connectors/viewer-member-sync-error.ts +++ /dev/null @@ -1,33 +0,0 @@ -import { db } from '@sim/db' -import { - credential, - credentialGroupEnrollment, - knowledgeConnector, - knowledgeConnectorMember, -} from '@sim/db/schema' -import { and, eq, exists, inArray, isNotNull, isNull, sql } from 'drizzle-orm' - -/** Retains the viewer's account-level failure even after a run claims no due members. */ -export function hasViewerMemberSyncError(userId: string) { - return sql`${exists( - db - .select({ id: knowledgeConnectorMember.id }) - .from(knowledgeConnectorMember) - .innerJoin(credential, eq(credential.id, knowledgeConnectorMember.credentialId)) - .innerJoin( - credentialGroupEnrollment, - eq(credentialGroupEnrollment.id, credential.credentialGroupEnrollmentId) - ) - .where( - and( - eq(knowledgeConnector.accessMode, 'members'), - eq(knowledgeConnectorMember.connectorId, knowledgeConnector.id), - eq(credentialGroupEnrollment.userId, userId), - inArray(knowledgeConnectorMember.status, ['active', 'suspended']), - inArray(credential.managedOauthStatus, ['active', 'needs_reauth']), - isNull(credential.revokedAt), - isNotNull(knowledgeConnectorMember.lastError) - ) - ) - )}` -} diff --git a/apps/sim/lib/knowledge/connectors/viewer-source-accounts.test.ts b/apps/sim/lib/knowledge/connectors/viewer-source-accounts.test.ts deleted file mode 100644 index d2337681577..00000000000 --- a/apps/sim/lib/knowledge/connectors/viewer-source-accounts.test.ts +++ /dev/null @@ -1,76 +0,0 @@ -import { credential, credentialGroup, credentialGroupEnrollment } from '@sim/db/schema' -import { dbChainMockFns, queueTableRows, resetDbChainMock } from '@sim/testing' -import { eq, inArray, isNull } from 'drizzle-orm' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -vi.mock('@/connectors/registry', () => ({ - getConnectorMeta: (id: string) => ({ - requiresMemberIdentity: id === 'slack', - auth: { mode: 'oauth', provider: id }, - }), -})) - -import { resolveViewerSourceAccounts } from '@/lib/knowledge/connectors/viewer-source-accounts' -import { SEARCH_SOURCE_CANDIDATE_PAGE_SIZE } from '@/lib/knowledge/constants' - -const source = { - id: 'gmail-source', - connectorType: 'gmail', - accessMode: 'members', - credentialGroupId: 'group-1', - credentialGroupOptionId: 'gmail-option', -} -const input = { organizationId: 'org-1', userId: 'viewer', connectors: [source] } -const account = { - credentialId: 'mine', - displayName: 'My Gmail', - groupId: 'group-1', - optionId: 'gmail-option', - providerId: 'gmail', - status: 'active', -} - -describe('personal source account projection', () => { - beforeEach(() => { - resetDbChainMock() - }) - - it('binds the current contributor and organization on both credentials and groups', async () => { - queueTableRows(credential, [account]) - const result = await resolveViewerSourceAccounts(input) - expect(eq).toHaveBeenCalledWith(credentialGroupEnrollment.userId, 'viewer') - expect(eq).toHaveBeenCalledWith(credential.organizationId, 'org-1') - expect(eq).toHaveBeenCalledWith(credentialGroup.organizationId, 'org-1') - expect(isNull).toHaveBeenCalledWith(credential.workspaceId) - expect(isNull).toHaveBeenCalledWith(credentialGroup.workspaceId) - expect(isNull).toHaveBeenCalledWith(credential.revokedAt) - expect(inArray).toHaveBeenCalledWith(credential.managedOauthStatus, ['active', 'needs_reauth']) - expect(result.get(source.id)).toEqual([ - { credentialId: 'mine', displayName: 'My Gmail', status: 'active' }, - ]) - expect(dbChainMockFns.select).toHaveBeenCalledWith({ - credentialId: credential.id, - displayName: credential.displayName, - status: credential.managedOauthStatus, - groupId: credentialGroup.id, - optionId: credential.credentialGroupOptionId, - providerId: credential.providerId, - }) - }) - - it('does not attach an account from another source option or group', async () => { - queueTableRows(credential, [ - { ...account, groupId: 'other' }, - { ...account, optionId: 'other' }, - ]) - expect(await resolveViewerSourceAccounts(input)).toEqual(new Map()) - }) - - it('fails instead of silently truncating too many accounts', async () => { - queueTableRows( - credential, - Array.from({ length: SEARCH_SOURCE_CANDIDATE_PAGE_SIZE + 1 }, () => account) - ) - await expect(resolveViewerSourceAccounts(input)).rejects.toThrow('Too many personal accounts') - }) -}) diff --git a/apps/sim/lib/knowledge/connectors/viewer-source-accounts.ts b/apps/sim/lib/knowledge/connectors/viewer-source-accounts.ts deleted file mode 100644 index 924dd0f386b..00000000000 --- a/apps/sim/lib/knowledge/connectors/viewer-source-accounts.ts +++ /dev/null @@ -1,76 +0,0 @@ -import { credential, credentialGroup } from '@sim/db/schema' -import { and, eq, or } from 'drizzle-orm' -import { listViewerOrganizationAccounts } from '@/lib/credential-groups/viewer-accounts' -import { getConnectorMeta } from '@/connectors/registry' - -interface ViewerSourceAccount { - credentialId: string - displayName: string - status: 'active' | 'needs_reauth' -} - -interface SourceAccountBinding { - id: string - connectorType: string - accessMode: string - credentialGroupId: string | null - credentialGroupOptionId: string | null -} - -/** - * Own account controls remain available when provider setup, enrollment, or sync is disabled. - * Called inside the authorized source read; selects no token material or other contributors. - */ -export async function resolveViewerSourceAccounts(input: { - organizationId: string - userId: string - connectors: ReadonlyArray -}): Promise> { - const bindings = input.connectors.map((source) => { - const meta = getConnectorMeta(source.connectorType) - const providerId = - source.accessMode === 'admin' && meta?.requiresMemberIdentity && meta.auth.mode === 'oauth' - ? meta.auth.provider - : null - return { source, providerId } - }) - const matches = bindings.flatMap(({ source, providerId }) => { - if ( - source.accessMode === 'members' && - source.credentialGroupId && - source.credentialGroupOptionId - ) - return [ - and( - eq(credentialGroup.id, source.credentialGroupId), - eq(credential.credentialGroupOptionId, source.credentialGroupOptionId) - ), - ] - return providerId ? [eq(credential.providerId, providerId)] : [] - }) - const result = new Map() - if (!matches.length) return result - const accounts = await listViewerOrganizationAccounts({ - organizationId: input.organizationId, - userId: input.userId, - matching: or(...matches)!, - }) - for (const { source, providerId } of bindings) { - const own = accounts.filter((account) => - source.accessMode === 'members' - ? account.groupId === source.credentialGroupId && - account.optionId === source.credentialGroupOptionId - : providerId !== null && account.providerId === providerId - ) - if (own.length) - result.set( - source.id, - own.map(({ credentialId, displayName, status }) => { - if (status !== 'active' && status !== 'needs_reauth') - throw new Error('Invalid personal account status') - return { credentialId, displayName, status } - }) - ) - } - return result -} diff --git a/apps/sim/lib/knowledge/constants.ts b/apps/sim/lib/knowledge/constants.ts index f6de98cf0a0..3f9093870d8 100644 --- a/apps/sim/lib/knowledge/constants.ts +++ b/apps/sim/lib/knowledge/constants.ts @@ -29,8 +29,6 @@ export const DEFAULT_KNOWLEDGE_CONNECTOR_DOCUMENT_PAGE_SIZE = 100 export const MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_PAGE_SIZE = 200 export const MAX_KNOWLEDGE_CONNECTOR_DOCUMENT_SEARCH_LENGTH = 200 -/** Maximum source IDs in one viewer-authorized progress request. */ -export const MAX_SEARCH_SOURCE_PROGRESS_ITEMS = 100 /** Bound viewer-specific source resolution and document counts to a single page. */ export const SEARCH_SOURCE_PAGE_SIZE = 25 export const SEARCH_SOURCE_CANDIDATE_PAGE_SIZE = 100 diff --git a/apps/sim/lib/knowledge/documents/ocr-recovery.md b/apps/sim/lib/knowledge/documents/ocr-recovery.md index 91723bbf352..caaca290016 100644 --- a/apps/sim/lib/knowledge/documents/ocr-recovery.md +++ b/apps/sim/lib/knowledge/documents/ocr-recovery.md @@ -1,6 +1,6 @@ # OCR capacity and indexing recovery -Regular knowledge bases and legacy indexed Sim Search (`SIM_SEARCH_LIVE=false`) use the connector content pass, document processor, embeddings, and processing continuations described here. Live Search, the default, calls provider APIs and does not run this OCR/indexing recovery path. Authorization and source visibility remain specific to each access mode. +Regular knowledge bases use the connector content pass, document processor, embeddings, and processing continuations described here. Enterprise Search calls provider APIs and does not run this OCR/indexing recovery path. Authorization and source visibility remain specific to each access mode. ## Operating budgets diff --git a/apps/sim/lib/knowledge/mcp/route-handler.test.ts b/apps/sim/lib/knowledge/mcp/route-handler.test.ts index 074656fc040..41daf2f7ba2 100644 --- a/apps/sim/lib/knowledge/mcp/route-handler.test.ts +++ b/apps/sim/lib/knowledge/mcp/route-handler.test.ts @@ -63,7 +63,6 @@ vi.mock('@/connectors/registry', () => ({ CONNECTOR_META_REGISTRY: {} })) vi.mock('@/lib/sim-search/connectors', () => ({ SIM_SEARCH_KNOWLEDGE_BASE_NAME: 'Sim Search', canConnectPersonally: vi.fn(), - missingSetupFields: vi.fn(), })) import { V2ApiKeyUnauthenticatedError } from '@/lib/api/server/routes/v2-api-key-auth' @@ -160,7 +159,7 @@ describe('organization MCP request admission', () => { expect(mocks.index).not.toHaveBeenCalled() }) - it('admits a current personal-key member and binds the canonical organization index', async () => { + it('admits a current personal-key member for the canonical organization', async () => { const result = await post() expect(result.status).toBe(200) expect(result.headers.get('Cache-Control')).toBe('private, no-store') @@ -169,7 +168,7 @@ describe('organization MCP request admission', () => { knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable ).toHaveBeenCalledExactlyOnceWith('org-1') expect(mocks.createServer).toHaveBeenCalledWith( - expect.objectContaining({ auth, organizationId: 'org-1', searchIndexId: 'index-1' }) + expect.objectContaining({ auth, organizationId: 'org-1' }) ) expect(mocks.close).toHaveBeenCalledOnce() expect(v2RouteMocks.authenticate).toHaveBeenCalledWith( diff --git a/apps/sim/lib/knowledge/mcp/route-handler.ts b/apps/sim/lib/knowledge/mcp/route-handler.ts index 006b69ad43e..980672d468b 100644 --- a/apps/sim/lib/knowledge/mcp/route-handler.ts +++ b/apps/sim/lib/knowledge/mcp/route-handler.ts @@ -42,7 +42,7 @@ export function createKnowledgeMcpHandlers() { maxBodyBytes: 64 * 1024, }) if (!parsed.success) return parsed.response - const index = await readSearchIndex.execute({ + await readSearchIndex.execute({ principal: admission.auth.principal, input: parsed.data.params, request, @@ -51,7 +51,6 @@ export function createKnowledgeMcpHandlers() { request, auth: admission.auth, ...parsed.data.params, - searchIndexId: index.knowledgeBaseId, }) return await serveStatelessMcp(server, request, parsed.data.body) } catch (error) { diff --git a/apps/sim/lib/knowledge/mcp/server.protocol.test.ts b/apps/sim/lib/knowledge/mcp/server.protocol.test.ts index a0ea9244d10..594af0f2070 100644 --- a/apps/sim/lib/knowledge/mcp/server.protocol.test.ts +++ b/apps/sim/lib/knowledge/mcp/server.protocol.test.ts @@ -1,33 +1,19 @@ import { Client } from '@modelcontextprotocol/sdk/client/index.js' import { InMemoryTransport } from '@modelcontextprotocol/sdk/inMemory.js' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing' import { createPersonalApiKeyPrincipal } from '@sim/testing/factories/principal.factory' -import { - knowledgeSearchUseCaseMock, - knowledgeSearchUseCaseMockFns, -} from '@sim/testing/mocks/knowledge-search-use-case.mock' import { urlsMockFns } from '@sim/testing/mocks/urls.mock' import { NextRequest } from 'next/server' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { describe, expect, it, vi } from 'vitest' const hoisted = vi.hoisted(() => ({ liveSearch: vi.fn(), liveRead: vi.fn(), - indexedRead: vi.fn(), })) vi.mock('@/lib/core/utils/after-response', () => ({ afterResponse: vi.fn() })) vi.mock('@/lib/knowledge/mcp/activity', () => ({ recordOrganizationSearchMcpActivity: vi.fn() })) vi.mock('@/lib/api/server/routes/v2-json-route', () => ({ v2RateLimits: { publicApi: { enforce: vi.fn().mockResolvedValue(null) } }, })) -vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) -vi.mock('@/lib/sim-search/indexed/documents/read-indexed-document', () => ({ - readIndexedKnowledgeDocument: { execute: hoisted.indexedRead }, -})) -vi.mock('@/lib/sim-search/indexed', async () => ({ - registerIndexedKnowledgeMcpTools: (await import('@/lib/sim-search/indexed/mcp/register-tools')) - .registerIndexedKnowledgeMcpTools, -})) vi.mock('@/lib/sim-search/live/application', () => ({ searchLiveKnowledge: { execute: hoisted.liveSearch }, readLiveDocument: { execute: hoisted.liveRead }, @@ -35,99 +21,68 @@ vi.mock('@/lib/sim-search/live/application', () => ({ import { createKnowledgeMcpServer } from '@/lib/knowledge/mcp/server' -const mocks = { - ...hoisted, - indexedSearch: knowledgeSearchUseCaseMockFns.mockSearchKnowledgeExecute, -} +const mocks = hoisted urlsMockFns.mockGetBaseUrl.mockReturnValue('http://localhost') describe('Search MCP protocol', () => { - beforeEach(() => { - resetEnvFlagsMock() - }) - - it.each([true, false])( - 'lists and calls the selected backend through the SDK (live: %s)', - async (live) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: live }) - const documentId = live ? `live:${'a'.repeat(500)}` : 'doc-1' - const row = { - documentId, - knowledgeBaseId: live ? '' : 'index-1', - documentName: 'Release notes', - sourceUrl: 'https://example.com/notes', - connectorType: 'google_drive', - content: 'Evidence', - chunkIndex: 0, - similarity: 0.5, - } - mocks.liveSearch.mockResolvedValue({ - results: [row], - live: { backend: 'live', accounts: [], guidance: '' }, - }) - mocks.indexedSearch.mockResolvedValue({ results: [row] }) - mocks.liveRead.mockResolvedValue({ - ...row, - knowledgeBaseName: 'Drive', - chunks: [{ chunkIndex: 0, content: 'Read evidence' }], - hasMore: false, - next: null, - }) - mocks.indexedRead.mockResolvedValue({ - ...row, - title: row.documentName, - chunks: [{ chunkIndex: 0, content: 'Read evidence' }], - }) - - const server = createKnowledgeMcpServer({ - organizationId: 'org-1', - searchIndexId: live ? null : 'index-1', - request: new NextRequest('http://localhost/api/mcp/search/organizations/org-1'), - auth: { - principal: createPersonalApiKeyPrincipal(), - keyType: 'personal', - keyExpiresAt: null, - rateLimitSubjectIds: ['user-1'], - rateLimitSubscription: null, - }, - }) - const client = new Client({ name: 'Search test', version: '1.0.0' }) - const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair() - try { - await server.connect(serverTransport) - await client.connect(clientTransport) - const { tools } = await client.listTools() - expect(tools.map((tool) => tool.name)).toEqual(['search', 'read_document', 'chat']) - const searchSchema = tools.find((tool) => tool.name === 'search')?.inputSchema - const readSchema = tools.find((tool) => tool.name === 'read_document')?.inputSchema - if (live) { - expect(searchSchema?.properties).toHaveProperty('nativeQueries') - expect(readSchema?.properties).toHaveProperty('startOffset') - expect(readSchema?.properties).not.toHaveProperty('url') - } else { - expect(searchSchema?.properties).not.toHaveProperty('nativeQueries') - expect(readSchema?.properties).toHaveProperty('offset') - expect(readSchema?.properties).toHaveProperty('url') - } - const search = await client.callTool({ name: 'search', arguments: { query: 'release' } }) - expect(search.isError).not.toBe(true) - expect(search.content).toEqual([ - { type: 'text', text: expect.stringContaining(documentId) }, - ]) - const read = await client.callTool({ name: 'read_document', arguments: { documentId } }) - expect(read.isError).not.toBe(true) - expect(read.content).toEqual([ - { type: 'text', text: expect.stringContaining('Read evidence') }, - ]) - expect(live ? mocks.liveSearch : mocks.indexedSearch).toHaveBeenCalledOnce() - expect(live ? mocks.liveRead : mocks.indexedRead).toHaveBeenCalledOnce() - expect(live ? mocks.indexedSearch : mocks.liveSearch).not.toHaveBeenCalled() - expect(live ? mocks.indexedRead : mocks.liveRead).not.toHaveBeenCalled() - } finally { - await client.close() - await server.close() - } + it('lists and calls live Search through the SDK', async () => { + const documentId = `live:${'a'.repeat(500)}` + const row = { + documentId, + knowledgeBaseId: '', + documentName: 'Release notes', + sourceUrl: 'https://example.com/notes', + connectorType: 'google_drive', + content: 'Evidence', + chunkIndex: 0, + similarity: 0.5, + } + mocks.liveSearch.mockResolvedValue({ + results: [row], + live: { backend: 'live', accounts: [], guidance: '' }, + }) + mocks.liveRead.mockResolvedValue({ + ...row, + knowledgeBaseName: 'Drive', + chunks: [{ chunkIndex: 0, content: 'Read evidence' }], + hasMore: false, + next: null, + }) + const server = createKnowledgeMcpServer({ + organizationId: 'org-1', + request: new NextRequest('http://localhost/api/mcp/search/organizations/org-1'), + auth: { + principal: createPersonalApiKeyPrincipal(), + keyType: 'personal', + keyExpiresAt: null, + rateLimitSubjectIds: ['user-1'], + rateLimitSubscription: null, + }, + }) + const client = new Client({ name: 'Search test', version: '1.0.0' }) + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair() + try { + await server.connect(serverTransport) + await client.connect(clientTransport) + const { tools } = await client.listTools() + expect(tools.map((tool) => tool.name)).toEqual(['search', 'read_document', 'chat']) + const searchSchema = tools.find((tool) => tool.name === 'search')?.inputSchema + const readSchema = tools.find((tool) => tool.name === 'read_document')?.inputSchema + expect(searchSchema?.properties).toHaveProperty('nativeQueries') + expect(readSchema?.properties).toHaveProperty('startOffset') + expect(readSchema?.properties).not.toHaveProperty('url') + const search = await client.callTool({ name: 'search', arguments: { query: 'release' } }) + expect(search.isError).not.toBe(true) + expect(search.content).toEqual([{ type: 'text', text: expect.stringContaining(documentId) }]) + const read = await client.callTool({ name: 'read_document', arguments: { documentId } }) + expect(read.isError).not.toBe(true) + expect(read.content).toEqual([ + { type: 'text', text: expect.stringContaining('Read evidence') }, + ]) + } finally { + await client.close() + await server.close() } - ) + }) }) diff --git a/apps/sim/lib/knowledge/mcp/server.test.ts b/apps/sim/lib/knowledge/mcp/server.test.ts index aca1577fd0d..4617bf078af 100644 --- a/apps/sim/lib/knowledge/mcp/server.test.ts +++ b/apps/sim/lib/knowledge/mcp/server.test.ts @@ -1,10 +1,5 @@ import type { CallToolResult } from '@modelcontextprotocol/sdk/types.js' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing' import { createPersonalApiKeyPrincipal } from '@sim/testing/factories/principal.factory' -import { - knowledgeSearchUseCaseMock, - knowledgeSearchUseCaseMockFns, -} from '@sim/testing/mocks/knowledge-search-use-case.mock' import { getMockLogger } from '@sim/testing/mocks/logger.mock' import { urlsMockFns } from '@sim/testing/mocks/urls.mock' import { NextRequest } from 'next/server' @@ -19,7 +14,6 @@ const hoisted = vi.hoisted(() => ({ string, { description: string; inputSchema: { parse: (input: unknown) => unknown } } >(), - read: vi.fn(), liveSearch: vi.fn(), liveRead: vi.fn(), chat: vi.fn(), @@ -49,14 +43,6 @@ vi.mock('@modelcontextprotocol/sdk/server/mcp.js', () => ({ vi.mock('@/lib/api/server/routes/v2-json-route', () => ({ v2RateLimits: { publicApi: { enforce: hoisted.rateLimit } }, })) -vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) -vi.mock('@/lib/sim-search/indexed/documents/read-indexed-document', () => ({ - readIndexedKnowledgeDocument: { execute: hoisted.read }, -})) -vi.mock('@/lib/sim-search/indexed', async () => ({ - registerIndexedKnowledgeMcpTools: (await import('@/lib/sim-search/indexed/mcp/register-tools')) - .registerIndexedKnowledgeMcpTools, -})) vi.mock('@/lib/sim-search/live/application', () => ({ searchLiveKnowledge: { execute: hoisted.liveSearch }, readLiveDocument: { execute: hoisted.liveRead }, @@ -69,10 +55,7 @@ import { OrchestrationError } from '@/lib/core/orchestration/types' import { createKnowledgeMcpServer } from '@/lib/knowledge/mcp/server' import type { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' -const mocks = { - ...hoisted, - search: knowledgeSearchUseCaseMockFns.mockSearchKnowledgeExecute, -} +const mocks = hoisted urlsMockFns.mockGetBaseUrl.mockReturnValue('https://sim.example') @@ -85,8 +68,8 @@ const auth = { rateLimitSubscription: null, } const request = new NextRequest('http://localhost/api/mcp/search/organizations/org-1') -function create(searchIndexId: string | null = 'index-1') { - createKnowledgeMcpServer({ organizationId: 'org-1', searchIndexId, request, auth }) +function create() { + createKnowledgeMcpServer({ organizationId: 'org-1', request, auth }) } function call(tool: string, input: Record, signal = new AbortController().signal) { const run = mocks.tools.get(tool) @@ -94,28 +77,11 @@ function call(tool: string, input: Record, signal = new AbortCo return run(input, { signal }) } -function _payload(result: CallToolResult): unknown { - const first = result.content[0] - if (first.type !== 'text') throw new Error('Expected a text result') - return JSON.parse(first.text) -} - beforeEach(() => { - resetEnvFlagsMock() - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) mocks.tools.clear() mocks.configs.clear() mocks.rateLimit.mockReset().mockResolvedValue(null) - mocks.search.mockResolvedValue({ results: [] }) - mocks.read.mockResolvedValue({ - knowledgeBaseId: 'index-1', - documentId: 'doc-1', - title: 'A source', - sourceUrl: 'https://example.com/source', - processingStatus: 'completed', - chunks: [{ id: 'chunk-1', chunkIndex: 0, content: 'Indexed text' }], - pagination: { total: 1, offset: 0, limit: 20, hasMore: false }, - }) + mocks.liveSearch.mockResolvedValue({ results: [] }) mocks.chat.mockResolvedValue({ content: 'An answer', citations: [] }) }) @@ -139,7 +105,6 @@ describe('live organization Search MCP', () => { guidance: 'Use nativeQueries to continue.', } beforeEach(() => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) mocks.liveSearch.mockResolvedValue({ results: [document], live: coverage, @@ -163,15 +128,13 @@ describe('live organization Search MCP', () => { }) it.each(['search', 'read_document'])( - 'does not fall back to indexed data after a live %s denial', + 'does not return live %s content after an authorization denial', async (tool) => { create() const backend = tool === 'search' ? mocks.liveSearch : mocks.liveRead backend.mockRejectedValueOnce(new OrchestrationError('forbidden', 'Access denied')) const result = await call(tool, tool === 'search' ? { query: 'release' } : { documentId }) expect(result).toEqual({ isError: true, content: [{ type: 'text', text: 'Access denied' }] }) - expect(mocks.search).not.toHaveBeenCalled() - expect(mocks.read).not.toHaveBeenCalled() } ) @@ -198,40 +161,10 @@ describe('live organization Search MCP', () => { ) }) -describe('search', () => { - const tool = 'search' - it('searches only the canonical index using the actual personal key principal', async () => { - create() - expect((await call(tool, { query: 'find it', topK: 10 })).isError).toBeUndefined() - expect(mocks.search).toHaveBeenCalledWith({ - principal, - request, - input: expect.objectContaining({ - organizationId: 'org-1', - knowledgeBaseIds: ['index-1'], - query: 'find it', - surface: 'mcp', - }), - }) - }) -}) - describe('organization Search MCP tools', () => { - it('delegates URL resolution without fetching the URL in the adapter', async () => { - create() - await call('read_document', { url: 'https://example.com/source', limit: 20 }) - expect(mocks.read).toHaveBeenCalledWith({ - principal, - request, - input: expect.objectContaining({ - organizationId: 'org-1', - target: { kind: 'url', url: 'https://example.com/source' }, - }), - }) - }) it('does not return content denied by the canonical document ACL operation', async () => { create() - mocks.read.mockRejectedValueOnce(new OrchestrationError('not_found', 'Document not found')) + mocks.liveRead.mockRejectedValueOnce(new OrchestrationError('not_found', 'Document not found')) const result = await call('read_document', { documentId: 'foreign-document', }) @@ -243,16 +176,16 @@ describe('organization Search MCP tools', () => { it('does not expose cached success after a later membership or policy denial', async () => { create() await call('search', { query: 'first', topK: 10 }) - mocks.search.mockRejectedValueOnce( + mocks.liveSearch.mockRejectedValueOnce( new OrchestrationError('forbidden', 'Knowledge access is disabled') ) expect((await call('search', { query: 'second', topK: 10 })).isError).toBe(true) - expect(mocks.search).toHaveBeenCalledTimes(2) + expect(mocks.liveSearch).toHaveBeenCalledTimes(2) }) it('does not return partial metadata when the chunk read is denied', async () => { create() - mocks.read.mockRejectedValueOnce(new OrchestrationError('not_found', 'Document not found')) + mocks.liveRead.mockRejectedValueOnce(new OrchestrationError('not_found', 'Document not found')) expect(await call('read_document', { documentId: 'doc-1' })).toEqual({ isError: true, content: [{ type: 'text', text: 'Document not found' }], @@ -295,7 +228,9 @@ describe('MCP tool completion records', () => { it('records an authorization failure without including the query or error message', async () => { create() - mocks.search.mockRejectedValueOnce(new OrchestrationError('forbidden', 'Private denial reason')) + mocks.liveSearch.mockRejectedValueOnce( + new OrchestrationError('forbidden', 'Private denial reason') + ) await call('search', { query: 'private query' }) expect(getMockLogger('KnowledgeMcp').info).toHaveBeenCalledExactlyOnceWith( 'Knowledge MCP tool completed', diff --git a/apps/sim/lib/knowledge/mcp/server.ts b/apps/sim/lib/knowledge/mcp/server.ts index e3a9c600186..996e352885a 100644 --- a/apps/sim/lib/knowledge/mcp/server.ts +++ b/apps/sim/lib/knowledge/mcp/server.ts @@ -25,8 +25,6 @@ import { } from '@/lib/knowledge/mcp/tool-runner' import { liveCitationId } from '@/lib/knowledge/search/citation' import { toolError } from '@/lib/mcp/tool-result' -import { registerIndexedKnowledgeMcpTools } from '@/lib/sim-search/indexed' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { readLiveDocument, searchLiveKnowledge } from '@/lib/sim-search/live/application' import { v2CaughtOrchestrationError } from '@/app/api/v2/lib/response' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' @@ -36,12 +34,11 @@ interface KnowledgeMcpContext { organizationId: string request: NextRequest auth: V2ApiKeyAuthContext - searchIndexId: string | null } /** A request owns its server; no credential or principal survives into another HTTP request. */ export function createKnowledgeMcpServer(context: KnowledgeMcpContext): McpServer { - const { request, auth, searchIndexId, organizationId } = context + const { request, auth, organizationId } = context const principal = auth.principal const server = new McpServer({ name: 'Sim Search', version: '1.0.0' }) @@ -107,103 +104,92 @@ export function createKnowledgeMcpServer(context: KnowledgeMcpContext): McpServe } } - if (isIndexedOrgSearchEnabled()) { - registerIndexedKnowledgeMcpTools({ - server, - principal, - request, - organizationId, - searchIndexId, - execute, - }) - } else { - server.registerTool( - 'search', - { - title: 'Search', - description: - 'Search this organization’s sources through their live APIs, within your access and the admin’s source settings. Use source and date filters to narrow results, or nativeQueries for provider queries and pagination. Inspect live.accounts for provider status and continuation cursors, and live.guidance for query syntax. Results are candidates, not proof of complete coverage. Use read_document with the exact returned documentId for context and cite citationUrl when available.', - inputSchema: liveSearchMcpSchema, - annotations: KNOWLEDGE_MCP_READ_ONLY, - }, - async (input: unknown, extra: { signal: AbortSignal }) => - execute('search', knowledgeOperations.search, extra.signal, async (registry, signal) => { - const { query, topK, nativeQueries, ...filters } = liveSearchMcpSchema.parse(input) - const result = await searchLiveKnowledge.execute({ + server.registerTool( + 'search', + { + title: 'Search', + description: + 'Search this organization’s sources through their live APIs, within your access and the admin’s source settings. Use source and date filters to narrow results, or nativeQueries for provider queries and pagination. Inspect live.accounts for provider status and continuation cursors, and live.guidance for query syntax. Results are candidates, not proof of complete coverage. Use read_document with the exact returned documentId for context and cite citationUrl when available.', + inputSchema: liveSearchMcpSchema, + annotations: KNOWLEDGE_MCP_READ_ONLY, + }, + async (input: unknown, extra: { signal: AbortSignal }) => + execute('search', knowledgeOperations.search, extra.signal, async (registry, signal) => { + const { query, topK, nativeQueries, ...filters } = liveSearchMcpSchema.parse(input) + const result = await searchLiveKnowledge.execute({ + principal, + input: { + organizationId, + query, + topK, + nativeQueries, + filters, + resultSecretRegistry: registry, + signal, + }, + request, + }) + return projectResult( + { + results: result.results.map((row) => ({ + documentId: row.documentId, + title: row.documentName, + sourceUrl: row.sourceUrl, + citationId: liveCitationId(row.documentId), + citationUrl: row.sourceUrl, + sourceModifiedAt: row.sourceModifiedAt ?? null, + sourceDate: row.sourceDate, + sourceContainerName: row.sourceContainerName, + sourceContainerUrl: row.sourceContainerUrl, + connectorType: row.connectorType, + content: row.content, + chunkIndex: row.chunkIndex, + })), + retrieval: result.retrieval, + live: result.live, + }, + registry + ) + }) + ) + server.registerTool( + 'read_document', + { + title: 'Read document', + description: + 'Read a live document using the exact documentId returned by search. Access and the admin’s source settings are checked again on every read. When hasMore is true, pass next.startChunkIndex and next.startOffset with the same documentId to continue. Cite citationUrl when available.', + inputSchema: readLiveDocumentMcpSchema, + annotations: KNOWLEDGE_MCP_READ_ONLY, + }, + async (raw: unknown, extra: { signal: AbortSignal }) => + execute( + 'read_document', + knowledgeOperations.readDocument, + extra.signal, + async (registry, signal) => { + const input = readLiveDocumentMcpSchema.parse(raw) + const result = await readLiveDocument.execute({ principal, - input: { - organizationId, - query, - topK, - nativeQueries, - filters, - resultSecretRegistry: registry, - signal, - }, + input: { ...input, organizationId, resultSecretRegistry: registry, signal }, request, }) + const { + knowledgeBaseId: _knowledgeBaseId, + knowledgeBaseName: _name, + ...document + } = result return projectResult( { - results: result.results.map((row) => ({ - documentId: row.documentId, - title: row.documentName, - sourceUrl: row.sourceUrl, - citationId: liveCitationId(row.documentId), - citationUrl: row.sourceUrl, - sourceModifiedAt: row.sourceModifiedAt ?? null, - sourceDate: row.sourceDate, - sourceContainerName: row.sourceContainerName, - sourceContainerUrl: row.sourceContainerUrl, - connectorType: row.connectorType, - content: row.content, - chunkIndex: row.chunkIndex, - })), - retrieval: result.retrieval, - live: result.live, + ...document, + title: result.documentName, + citationId: liveCitationId(result.documentId), + citationUrl: result.sourceUrl, }, registry ) - }) - ) - server.registerTool( - 'read_document', - { - title: 'Read document', - description: - 'Read a live document using the exact documentId returned by search. Access and the admin’s source settings are checked again on every read. When hasMore is true, pass next.startChunkIndex and next.startOffset with the same documentId to continue. Cite citationUrl when available.', - inputSchema: readLiveDocumentMcpSchema, - annotations: KNOWLEDGE_MCP_READ_ONLY, - }, - async (raw: unknown, extra: { signal: AbortSignal }) => - execute( - 'read_document', - knowledgeOperations.readDocument, - extra.signal, - async (registry, signal) => { - const input = readLiveDocumentMcpSchema.parse(raw) - const result = await readLiveDocument.execute({ - principal, - input: { ...input, organizationId, resultSecretRegistry: registry, signal }, - request, - }) - const { - knowledgeBaseId: _knowledgeBaseId, - knowledgeBaseName: _name, - ...document - } = result - return projectResult( - { - ...document, - title: result.documentName, - citationId: liveCitationId(result.documentId), - citationUrl: result.sourceUrl, - }, - registry - ) - } - ) - ) - } + } + ) + ) server.registerTool( 'chat', diff --git a/apps/sim/lib/knowledge/orchestration/connectors.ts b/apps/sim/lib/knowledge/orchestration/connectors.ts index 7a3170a6934..af72cbeac37 100644 --- a/apps/sim/lib/knowledge/orchestration/connectors.ts +++ b/apps/sim/lib/knowledge/orchestration/connectors.ts @@ -9,7 +9,7 @@ import { knowledgeConnectorMember, } from '@sim/db/schema' import { createLogger } from '@sim/logger' -import { getErrorMessage, getPostgresErrorCode, toError } from '@sim/utils/errors' +import { getPostgresErrorCode, toError } from '@sim/utils/errors' import { generateId } from '@sim/utils/id' import { and, eq, isNull, sql } from 'drizzle-orm' import { encryptApiKey } from '@/lib/api-key/crypto' @@ -67,7 +67,6 @@ import { type KnowledgeOperationContext, type KnowledgeOrchestrationResult, } from '@/lib/knowledge/orchestration/shared' -import { dropSourceVectorIndex } from '@/lib/knowledge/search/source-vector-indexes' import { createTagDefinition } from '@/lib/knowledge/tags/service' import { captureServerEvent } from '@/lib/posthog/server' import { searchSourceIdentity } from '@/lib/sim-search/source-identity' @@ -1414,14 +1413,6 @@ export async function performDeleteKnowledgeConnector( }) } - /** The source is gone, so its vector index is too; ranking falls back to the exact path. */ - await dropSourceVectorIndex(connectorId).catch((error: unknown) => { - logger.warn('Could not drop the source vector index', { - connectorId, - error: getErrorMessage(error), - }) - }) - return { success: true, documentsDeleted: deleteDocuments ? docCount : 0, diff --git a/apps/sim/lib/knowledge/projection/enqueue.ts b/apps/sim/lib/knowledge/projection/enqueue.ts index 93b8887c845..69592a29e2a 100644 --- a/apps/sim/lib/knowledge/projection/enqueue.ts +++ b/apps/sim/lib/knowledge/projection/enqueue.ts @@ -7,7 +7,6 @@ import { import { createLogger } from '@sim/logger' import { getErrorMessage } from '@sim/utils/errors' import { isTriggerAvailable } from '@/lib/core/config/trigger-availability' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' const logger = createLogger('KnowledgeProjectionEnqueue') @@ -69,20 +68,19 @@ export interface KnowledgeProjectionSweepResult { * The knowledge projector's only trigger: one pass per window while marks need one, so marks are * settled within about a minute of their write, a document a pass gave up is retried by the next, * and an idle deployment starts no pass at all. The sweep first releases, on the pooled database, - * the marks no pass is owed, so the always-on marking of writes never starts one. While indexed - * organization search is off that is every mark without content to project, search-index ones - * included, since nothing reads their mirrored source and ACL. + * the marks no pass is owed, so the always-on marking of writes never starts one. + * Search marks and permission-only marks are released without copying their rows. Only deferred + * ordinary-KB content admits a repair pass. */ export async function enqueueKnowledgeProjectionSweep(): Promise { - const scope = { searchIndexes: isIndexedOrgSearchEnabled() } let release = { drained: false, empty: false } try { - release = await releaseSettledMarks(db.$client, Date.now() + MARK_RELEASE_BUDGET_MS, scope) + release = await releaseSettledMarks(db.$client, Date.now() + MARK_RELEASE_BUDGET_MS) } catch (error) { /** A release that failed leaves its marks for the next sweep; whether a pass is owed still stands. */ logger.warn('Releasing settled projection marks failed', { error: getErrorMessage(error) }) } - if (release.empty || !(await hasKnowledgeProjectionWork(db.$client, release, scope))) { + if (release.empty || !(await hasKnowledgeProjectionWork(db.$client))) { return { triggered: false, backend: null, jobId: null } } if (!isTriggerAvailable()) { diff --git a/apps/sim/lib/knowledge/projection/run.ts b/apps/sim/lib/knowledge/projection/run.ts index b0f2c6a5cad..2cc64fe56e7 100644 --- a/apps/sim/lib/knowledge/projection/run.ts +++ b/apps/sim/lib/knowledge/projection/run.ts @@ -9,21 +9,20 @@ import { withUtcTimestamps } from '@sim/db/timestamps' import { createLogger } from '@sim/logger' import postgres, { type Sql } from 'postgres' import { env, envNumber } from '@/lib/core/config/env' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' const logger = createLogger('KnowledgeProjectionPass') /** * Most documents one pass projects at once, each worker on a connection of its own. Measured * locally on content-heavy load, projection throughput kept rising to eight workers without - * deadlocks, so eight is the default; `KB_CONFIG_PROJECTION_CONCURRENCY` raises or lowers it per + * deadlocks, so eight is the ceiling; `KB_CONFIG_PROJECTION_CONCURRENCY` can lower it per * deployment. A round opens only as many workers as there are marks, so a pass over a few * documents holds a few connections. */ -const PROJECTION_CONCURRENCY = envNumber(env.KB_CONFIG_PROJECTION_CONCURRENCY, 8, { - min: 1, - integer: true, -}) +const PROJECTION_CONCURRENCY = Math.min( + 8, + envNumber(env.KB_CONFIG_PROJECTION_CONCURRENCY, 8, { min: 1, integer: true }) +) /** How the projector's connections name themselves in `pg_stat_activity`. */ const PROJECTOR_APPLICATION_NAME = 'sim-knowledge-projector' @@ -45,17 +44,14 @@ export interface KnowledgeProjectionPassResult extends KnowledgeProjectionProgre * One pass of the knowledge projector: settles every marked document. Each round first releases, * for no longer than a release's own short budget, the marks no pass is owed (see * `releaseSettledMarks`), so a backlog of them shrinks every round without holding up the content - * behind it. What is left is projected: content a writer deferred, and, while indexed organization - * search is on, search-index documents, whose rows mirror their source and ACL. + * behind it. Only ordinary-KB vector content that older writers deferred is repaired; retired + * Search content and copied ACLs are never projected. * * Workers project documents in parallel, each on a connection of its own that holds its * per-document advisory locks; a pass this long should not hold the pool's connections. Workers * read the same oldest marks and split them at those locks. A round ends when every worker found * nothing more it could take; the pass goes on while rounds settle documents, and `remaining` * reports marks it left for the next sweep. - * - * The pass writes Tin keyword rows only while indexed organization search is enabled: only that - * search reads them. */ export async function runKnowledgeProjectionPass(options: { budgetMs: number @@ -75,7 +71,6 @@ export async function runKnowledgeProjectionPass(options: { ) ) const deadline = Date.now() + options.budgetMs - const scope = { searchIndexes: isIndexedOrgSearchEnabled() } const result: KnowledgeProjectionPassResult = { settled: 0, deferred: 0, @@ -88,8 +83,7 @@ export async function runKnowledgeProjectionPass(options: { while (Date.now() < deadline) { const release = await releaseSettledMarks( sessions[0], - Math.min(deadline, Date.now() + MARK_RELEASE_BUDGET_MS), - scope + Math.min(deadline, Date.now() + MARK_RELEASE_BUDGET_MS) ) result.released += release.released const workers = sessions.slice(0, await workersFor(sessions[0])) @@ -98,7 +92,6 @@ export async function runKnowledgeProjectionPass(options: { workers.map((session) => runKnowledgeProjection(session, { budgetMs: Math.max(0, deadline - Date.now()), - ...scope, }) ) ) diff --git a/apps/sim/lib/knowledge/search/activity-stats.ts b/apps/sim/lib/knowledge/search/activity-stats.ts deleted file mode 100644 index 29108baa858..00000000000 --- a/apps/sim/lib/knowledge/search/activity-stats.ts +++ /dev/null @@ -1,100 +0,0 @@ -import { dbReplica } from '@sim/db' -import { organizationSearchInvocation, user } from '@sim/db/schema' -import { sql } from 'drizzle-orm' -import { - getSearchStatsWindow, - SEARCH_STATS_PEOPLE_LIMIT, - type SEARCH_STATS_PERIODS, - SEARCH_STATS_SOURCE_LIMIT, - type SEARCH_STATS_SURFACES, - type SearchStatsDateRange, -} from '@/lib/knowledge/search/stats' - -export interface SearchStatsInput extends SearchStatsDateRange { - organizationId: string - period: (typeof SEARCH_STATS_PERIODS)[number] - surface?: (typeof SEARCH_STATS_SURFACES)[number] -} - -type StatsRow = { - totals: { invocations: number; activePeople: number; results: number } - series: { timestamp: string; invocations: number }[] - surfaces: { surface: (typeof SEARCH_STATS_SURFACES)[number]; invocations: number }[] - sources: { sourceType: string; invocations: number }[] - people: { - userId: string | null - name: string | null - email: string | null - invocations: number - sourceTypes: string[] - }[] -} - -/** One statement snapshot; only bounded aggregates leave Postgres, never individual search records. */ -export async function loadOrganizationSearchStats(input: SearchStatsInput, now = new Date()) { - const { start, end, days } = getSearchStatsWindow(input.period, now, input) - const rows = await dbReplica.transaction( - async (tx) => { - await tx.execute(sql`SET LOCAL statement_timeout = '10s'`) - return tx.execute(sql` - WITH activity AS ( - SELECT id, user_id, surface, source_types, result_count, created_at - FROM ${organizationSearchInvocation} - WHERE organization_id = ${input.organizationId} - AND created_at >= ${start.toISOString()}::timestamptz - AND created_at < ${end.toISOString()}::timestamptz - ${input.surface ? sql`AND surface = ${input.surface}` : sql``} - ), people AS ( - SELECT user_id, count(*)::float8 AS invocations - FROM activity GROUP BY user_id - ORDER BY invocations DESC, user_id NULLS LAST - LIMIT ${SEARCH_STATS_PEOPLE_LIMIT} - ) - SELECT - (SELECT jsonb_build_object( - 'invocations', count(*)::float8, - 'activePeople', count(DISTINCT user_id)::float8, - 'results', coalesce(sum(result_count), 0)::float8 - ) FROM activity) AS totals, - (SELECT coalesce(jsonb_agg(row ORDER BY row.timestamp), '[]'::jsonb) FROM ( - SELECT to_char(created_at AT TIME ZONE 'UTC', 'YYYY-MM-DD') || 'T00:00:00.000Z' AS timestamp, - count(*)::float8 AS invocations - FROM activity GROUP BY 1 - ) row) AS series, - (SELECT coalesce(jsonb_agg(row ORDER BY row.invocations DESC, row.surface), '[]'::jsonb) FROM ( - SELECT surface, count(*)::float8 AS invocations FROM activity GROUP BY surface - ) row) AS surfaces, - (SELECT coalesce(jsonb_agg(row ORDER BY row.invocations DESC, row."sourceType"), '[]'::jsonb) FROM ( - SELECT source_type AS "sourceType", count(*)::float8 AS invocations - FROM activity CROSS JOIN LATERAL unnest(source_types) AS source_type - GROUP BY source_type ORDER BY invocations DESC, source_type - LIMIT ${SEARCH_STATS_SOURCE_LIMIT} - ) row) AS sources, - (SELECT coalesce(jsonb_agg(row ORDER BY row.invocations DESC, row."userId" NULLS LAST), '[]'::jsonb) FROM ( - SELECT people.user_id AS "userId", ${user.name} AS name, ${user.email} AS email, - people.invocations, - ARRAY( - SELECT DISTINCT source_type FROM activity - CROSS JOIN LATERAL unnest(source_types) AS source_type - WHERE activity.user_id IS NOT DISTINCT FROM people.user_id - ORDER BY source_type LIMIT ${SEARCH_STATS_SOURCE_LIMIT} - ) AS "sourceTypes" - FROM people LEFT JOIN ${user} ON ${user.id} = people.user_id - ) row) AS people - `) - }, - { accessMode: 'read only' } - ) - const row = rows[0] - if (!row) throw new Error('Search activity aggregate returned no result') - const byDay = new Map(row.series.map((point) => [point.timestamp, point.invocations])) - return { - ...row, - start: start.toISOString(), - end: end.toISOString(), - series: Array.from({ length: days }, (_, index) => { - const timestamp = new Date(start.getTime() + index * 86_400_000).toISOString() - return { timestamp, invocations: byDay.get(timestamp) ?? 0 } - }), - } -} diff --git a/apps/sim/lib/knowledge/search/activity.test.ts b/apps/sim/lib/knowledge/search/activity.test.ts deleted file mode 100644 index f2bdb78ca85..00000000000 --- a/apps/sim/lib/knowledge/search/activity.test.ts +++ /dev/null @@ -1,67 +0,0 @@ -import { dbChainMockFns } from '@sim/testing/mocks/database.mock' -import { beforeEach, describe, expect, it, vi } from 'vitest' - -const mocks = vi.hoisted(() => ({ - values: vi.fn(), - insert: vi.fn(), - execute: vi.fn(), -})) - -import { recordOrganizationSearchActivity } from '@/lib/knowledge/search/activity' - -beforeEach(() => { - mocks.insert.mockReturnValue({ values: mocks.values }) - mocks.values.mockResolvedValue(undefined) - mocks.execute.mockResolvedValue(undefined) - dbChainMockFns.transaction.mockImplementation((callback) => - callback({ execute: mocks.execute, insert: mocks.insert }) - ) -}) - -describe('Search activity metering', () => { - it('counts documents once per invocation and stores no document identity or content', async () => { - await recordOrganizationSearchActivity({ - organizationId: 'org', - userId: 'actor', - surface: 'mcp', - results: [ - { documentId: 'private-document-1', connectorType: 'confluence' }, - { documentId: 'private-document-1', connectorType: 'confluence' }, - { documentId: 'private-document-2', connectorType: 'jira' }, - { documentId: 'private-document-3', connectorType: null }, - ], - }) - expect(mocks.values).toHaveBeenCalledWith({ - id: expect.any(String), - organizationId: 'org', - userId: 'actor', - surface: 'mcp', - sourceTypes: ['confluence', 'jira', 'uploads'], - resultCount: 3, - }) - expect(JSON.stringify(mocks.values.mock.calls)).not.toContain('private-document') - }) - it('does not attempt an unbounded insert if setting the deadline fails', async () => { - mocks.execute.mockRejectedValueOnce(new Error('Could not set statement timeout')) - await expect( - recordOrganizationSearchActivity({ - organizationId: 'org', - userId: 'actor', - surface: 'slack', - results: [], - }) - ).resolves.toBeUndefined() - expect(mocks.insert).not.toHaveBeenCalled() - }) - it('does not fail Search when activity storage is unavailable', async () => { - mocks.values.mockRejectedValueOnce(new Error('offline')) - await expect( - recordOrganizationSearchActivity({ - organizationId: 'org', - userId: 'actor', - surface: 'dashboard', - results: [], - }) - ).resolves.toBeUndefined() - }) -}) diff --git a/apps/sim/lib/knowledge/search/activity.ts b/apps/sim/lib/knowledge/search/activity.ts deleted file mode 100644 index f3a68393de7..00000000000 --- a/apps/sim/lib/knowledge/search/activity.ts +++ /dev/null @@ -1,38 +0,0 @@ -import { db } from '@sim/db' -import { organizationSearchInvocation } from '@sim/db/schema' -import { createLogger } from '@sim/logger' -import { getErrorMessage } from '@sim/utils/errors' -import { generateId } from '@sim/utils/id' -import { sql } from 'drizzle-orm' -import type { SEARCH_STATS_SURFACES } from '@/lib/knowledge/search/stats' - -const logger = createLogger('OrganizationSearchActivity') - -interface SearchActivityInput { - organizationId: string - userId: string - surface: (typeof SEARCH_STATS_SURFACES)[number] - results: ReadonlyArray<{ documentId: string; connectorType: string | null }> -} - -/** Records completed, authorized searches without query text, document IDs, or content. */ -export async function recordOrganizationSearchActivity(input: SearchActivityInput): Promise { - const sourceTypes = [...new Set(input.results.map((result) => result.connectorType ?? 'uploads'))] - try { - await db.transaction(async (tx) => { - await tx.execute(sql`SET LOCAL statement_timeout = '2s'`) - await tx.insert(organizationSearchInvocation).values({ - id: generateId(), - organizationId: input.organizationId, - userId: input.userId, - surface: input.surface, - sourceTypes, - resultCount: new Set(input.results.map((result) => result.documentId)).size, - }) - }) - } catch (error) { - logger.warn('Failed to record organization Search activity', { - error: getErrorMessage(error), - }) - } -} diff --git a/apps/sim/lib/knowledge/search/author.ts b/apps/sim/lib/knowledge/search/author.ts deleted file mode 100644 index bd6bff80d5e..00000000000 --- a/apps/sim/lib/knowledge/search/author.ts +++ /dev/null @@ -1,34 +0,0 @@ -/** - * The tag names connectors give the person behind a document, in the order - * they are tried. Connectors were never asked to agree on a name, so the - * result's author is derived here rather than in each of them. - */ -const AUTHOR_TAG_NAMES = [ - 'From', - 'Author', - 'Sender', - 'Owner', - 'Organizer', - 'Creator', - 'Reporter', - 'Assignee', -] as const - -/** - * The person a search result shows beside its source: the first author-like - * tag the document carries, reduced to a display name when the connector - * stored an address form such as `Name `. - */ -export function sourceAuthor(metadata: Record): string | null { - for (const name of AUTHOR_TAG_NAMES) { - const value = metadata[name] - if (typeof value !== 'string') continue - const display = value - .replace(/<[^>]*>/g, '') - .trim() - .replace(/^"|"$/g, '') - .trim() - if (display) return display - } - return null -} diff --git a/apps/sim/lib/knowledge/search/connection-attempt.ts b/apps/sim/lib/knowledge/search/connection-attempt.ts index 3eec8ee04c5..a820a04afee 100644 --- a/apps/sim/lib/knowledge/search/connection-attempt.ts +++ b/apps/sim/lib/knowledge/search/connection-attempt.ts @@ -5,7 +5,6 @@ export const SEARCH_CONNECTION_ATTEMPT_EVENT = 'sim:search-connection-attempt' const attemptSchema = z.object({ completionId: z.string().uuid(), requestedAt: z.number(), - connectorId: z.string().optional(), credentialId: z.string().optional(), status: z.enum(['pending', 'connected', 'failed']), error: z.string().nullable(), diff --git a/apps/sim/lib/knowledge/search/connection-target.test.ts b/apps/sim/lib/knowledge/search/connection-target.test.ts index 42ccb94de8b..fbadb5ae518 100644 --- a/apps/sim/lib/knowledge/search/connection-target.test.ts +++ b/apps/sim/lib/knowledge/search/connection-target.test.ts @@ -5,13 +5,14 @@ const target = { type: 'link', provider: 'google-email', connectorType: 'gmail', - connectorId: 'source', + connectionMode: 'live', + optionId: 'option', } describe('shared Search connection tags', () => { it.each([ { ...target, value: 'https://evil.test' }, { ...target, organizationId: 'other' }, - { ...target, connectorId: undefined, credentialId: 'account' }, + { ...target, optionId: undefined, credentialId: 'account' }, ])('rejects model URLs, scope and incomplete reconnects', (forged) => { expect( parseSearchConnectionTargets(`${JSON.stringify(forged)}`) diff --git a/apps/sim/lib/knowledge/search/connection-target.ts b/apps/sim/lib/knowledge/search/connection-target.ts index 003a8234d98..040224f912a 100644 --- a/apps/sim/lib/knowledge/search/connection-target.ts +++ b/apps/sim/lib/knowledge/search/connection-target.ts @@ -7,28 +7,11 @@ export const searchConnectionTargetSchema = z type: z.literal('link'), provider: z.string().trim().min(1).max(100), connectorType: z.string().trim().min(1).max(100), - connectorId: z.string().min(1).max(200).optional(), credentialId: z.string().min(1).max(128).optional(), - connectionMode: z.literal('live').optional(), - optionId: z.string().min(1).max(128).optional(), + connectionMode: z.literal('live'), + optionId: z.string().min(1).max(128), }) .strict() - .refine( - (target) => - !target.credentialId || Boolean(target.connectorId) || target.connectionMode === 'live', - { - message: 'A reconnect requires a configured source', - } - ) - .refine( - (target) => - target.connectionMode === 'live' - ? Boolean(target.optionId) && !target.connectorId - : !target.optionId, - { - message: 'Live account connections require an account option instead of an indexed source', - } - ) export type SearchConnectionTarget = z.infer @@ -63,5 +46,11 @@ export function parseSearchConnectionTargets(text: string): SearchConnectionTarg /** Carries an untrusted selection to the existing authenticated Integrations page, without OAuth state. */ export function searchConnectionPath(organizationId: string, target: SearchConnectionTarget) { - return `${organizationRoutes(organizationId).integrations}?${new URLSearchParams({ connectorType: target.connectorType, ...(target.connectorId ? { connectorId: target.connectorId } : {}), ...(target.credentialId ? { credentialId: target.credentialId } : {}), ...(target.connectionMode ? { connectionMode: target.connectionMode, ...(target.optionId ? { optionId: target.optionId } : {}), provider: target.provider } : {}) })}` + return `${organizationRoutes(organizationId).integrations}?${new URLSearchParams({ + connectorType: target.connectorType, + ...(target.credentialId ? { credentialId: target.credentialId } : {}), + connectionMode: target.connectionMode, + optionId: target.optionId, + provider: target.provider, + })}` } diff --git a/apps/sim/lib/knowledge/search/diagnostics.ts b/apps/sim/lib/knowledge/search/diagnostics.ts index 95ab1a6505b..2468262b182 100644 --- a/apps/sim/lib/knowledge/search/diagnostics.ts +++ b/apps/sim/lib/knowledge/search/diagnostics.ts @@ -14,11 +14,9 @@ export type SearchStage = | 'tool_presentation' | 'workspace_application' | 'organization_application' - | 'scoped_application' | 'knowledge_application' | 'scope_resolution' | 'knowledge_context' - | 'index_resolution' | 'availability' | 'billing_attribution' | 'usage_admission' @@ -28,10 +26,7 @@ export type SearchStage = | 'access_scope' | 'defaults' | 'retrieval' - | 'access_plan' | 'live_source_grants' - | 'vector.source_exact' - | 'vector.source_walk' | 'permitted_documents' | 'result_provenance' | 'reranking' @@ -39,7 +34,6 @@ export type SearchStage = | 'overage_billing' | 'tag_definitions' | 'metadata_provenance' - | 'activity_recording' | RetrievalLeg | `${RetrievalLeg}.candidates` | `${RetrievalLeg}.hydration` @@ -48,20 +42,9 @@ export type SearchStage = | 'vector.settings' | 'vector.probe' | 'vector.page' - | 'vector.projection_filled' - | 'vector.source_indexes' - | 'keyword.projection_filled' | 'vector.exact_candidates' | 'vector.exact' | 'vector.candidate_search' - | 'keyword.tin' - | 'keyword.tin_readiness' - | 'keyword.tin_query' - | 'source_overview' - | 'source_overview.availability' - | 'source_overview.providers' - | 'source_overview.indexing' - | 'source_overview.searchable' | 'access_batch.connectors' | 'access_batch.live_proof' | 'live.policies' @@ -74,7 +57,7 @@ export type SearchStage = /** Fixed, content-free fields. Never pass queries, filters, document identities, SQL, or errors. */ export interface SearchDiagnosticMetadata { - operation?: 'search_workspace' | 'read_document' | 'read_search_source_overview' + operation?: 'search_workspace' | 'read_document' surface?: 'dashboard' | 'mcp' | 'copilot' | 'workflow' | 'api' | 'slack' | 'other' toolCallId?: string executionId?: string @@ -91,7 +74,7 @@ export interface SearchDiagnosticMetadata { searchMode?: 'hybrid' | 'vector' boostRecency?: boolean embeddingDimensions?: number - vectorRanking?: 'exact' | 'exact-candidates' | 'projection-walk' | 'per-source' + vectorRanking?: 'exact' | 'exact-candidates' | 'projection-walk' /** * Whether the bounded traversal filled its candidate limit. `underfilled` means visibility * removed enough neighbours that the rerank pool is smaller than requested, which lowers recall @@ -102,25 +85,6 @@ export interface SearchDiagnosticMetadata { vectorCandidateLimit?: number /** Visible documents the tractability probe enumerated, capped at its own document limit. */ vectorProbeDocumentCount?: number - /** - * Whether a user-scoped search resolved its permitted documents before retrieval: `bounded` - * ranks inside that set, `unbounded` means it exceeded the probe's limit and both legs search - * the index with the access predicate applied per candidate. - */ - permittedDocuments?: 'bounded' | 'unbounded' - /** Documents in a bounded permitted set. */ - permittedDocumentCount?: number - vectorSourcesSliced?: number - /** The sliced sources held more readable documents than one exact ranking may enumerate. */ - vectorSlicedSaturated?: boolean - vectorSourcesWalked?: number - /** - * Which index ranked an unbounded keyword leg: `tin` ranks by BM25 and checks access on the top - * of that ranking; `gin` ranks every match. Absent when the leg ranked inside a bounded set. - */ - keywordRanking?: 'tin' | 'gin' - /** Candidates Tin ranked before access was checked on the last keyword page. */ - keywordTinWindow?: number vectorCandidateCount?: number vectorCandidateDimensions?: number resultCount?: number @@ -139,10 +103,6 @@ export interface SearchDiagnosticMetadata { accessBatchCount?: number /** Connector identities sent for live proof, summed over every batch after the first. */ liveProofConnectorCount?: number - /** Provider types with a configured search source, before any access probe. */ - configuredProviderCount?: number - /** Searchable-document probes actually issued; one per batch until the answer is known. */ - searchableProbeCount?: number } interface StageTiming { diff --git a/apps/sim/lib/knowledge/search/keyword-ranking.ts b/apps/sim/lib/knowledge/search/keyword-ranking.ts index ed9edf76869..0859183b1ab 100644 --- a/apps/sim/lib/knowledge/search/keyword-ranking.ts +++ b/apps/sim/lib/knowledge/search/keyword-ranking.ts @@ -1,4 +1,4 @@ -import { document, type embedding, type embeddingKeywordSearch } from '@sim/db/schema' +import { document, type embedding } from '@sim/db/schema' import { and, type SQL, sql } from 'drizzle-orm' /** @@ -18,7 +18,7 @@ export function keywordCandidateRankingQuery(input: { /** What a matched document must satisfy to be readable. */ documentConditions: (SQL | undefined)[] /** The table whose text-search vector `rank` reads, joined to the matches on `id`. */ - rankTable: typeof embedding | typeof embeddingKeywordSearch + rankTable: typeof embedding rank: SQL limit: number offset: number diff --git a/apps/sim/lib/knowledge/search/prewarm.test.ts b/apps/sim/lib/knowledge/search/prewarm.test.ts index 07790f39e3f..55dd5339544 100644 --- a/apps/sim/lib/knowledge/search/prewarm.test.ts +++ b/apps/sim/lib/knowledge/search/prewarm.test.ts @@ -1,9 +1,4 @@ import { describe, expect, it, vi } from 'vitest' - -vi.mock('@sim/db/knowledge-projection', () => ({ - SOURCE_ACL_PROJECTIONS: ['embedding_search', 'embedding_keyword_tin'], -})) - import { prewarmSearchProjection } from '@/lib/knowledge/search/prewarm' interface Statement { @@ -60,7 +55,7 @@ describe('prewarmSearchProjection', () => { installed: true, relations: [ 'embedding_search', - 'embedding_keyword_tin', + 'embedding_search_cosine_hnsw_idx', 'embedding_search_512_cosine_hnsw_idx', ], }) @@ -74,7 +69,7 @@ describe('prewarmSearchProjection', () => { const warmed = await prewarmSearchProjection(fake, { budgetMs: 1000 }) expect(warmed.map((item) => item.relation)).toEqual([ 'embedding_search', - 'embedding_keyword_tin', + 'embedding_search_cosine_hnsw_idx', 'embedding_search_512_cosine_hnsw_idx', ]) const timeouts = fake.statements diff --git a/apps/sim/lib/knowledge/search/prewarm.ts b/apps/sim/lib/knowledge/search/prewarm.ts index 239907f1b13..5713d58d5bf 100644 --- a/apps/sim/lib/knowledge/search/prewarm.ts +++ b/apps/sim/lib/knowledge/search/prewarm.ts @@ -1,15 +1,13 @@ -import { SOURCE_ACL_PROJECTIONS } from '@sim/db/knowledge-projection' import { createLogger } from '@sim/logger' import { getErrorMessage } from '@sim/utils/errors' const logger = createLogger('SearchProjectionPrewarm') /** - * The access methods a ranking touches at random: the vector graphs, the Tin keyword index, and - * the GIN index the on-row permission test reads. The remaining b-trees serve hydration, which + * Vector ranking touches graph pages at random. The remaining b-trees serve hydration, which * reads a handful of rows by key and is fast cold. */ -const RANKING_ACCESS_METHODS = ['hnsw', 'tin', 'gin'] as const +const RANKING_ACCESS_METHODS = ['hnsw'] as const /** The one call the helper needs from a `postgres` connection or a reserved session. */ export interface PrewarmSession { @@ -143,7 +141,7 @@ async function rankingRelations(session: PrewarmSession): Promise { AND am.amname = ANY($2::text[]) ) ORDER BY c.relkind = 'r' DESC, pg_relation_size(c.oid)`, - [toArrayLiteral(SOURCE_ACL_PROJECTIONS), toArrayLiteral(RANKING_ACCESS_METHODS)] + [toArrayLiteral(['embedding_search']), toArrayLiteral(RANKING_ACCESS_METHODS)] ) return Array.from(rows, (row) => String(row.relation)) } diff --git a/apps/sim/lib/knowledge/search/queries.ts b/apps/sim/lib/knowledge/search/queries.ts index 0002fd9bb6b..0e55d49ad21 100644 --- a/apps/sim/lib/knowledge/search/queries.ts +++ b/apps/sim/lib/knowledge/search/queries.ts @@ -646,8 +646,6 @@ interface ExecuteKnowledgeSearchParams { queryVector?: KnowledgeQueryVector structuredFilters?: StructuredFilter[] filters?: WorkspaceSearchFilters - /** Runs the search-index retrieval legs; see `usesIndexedRetrieval`. */ - indexedRetrieval?: boolean } export interface RetrievalStatus { @@ -704,27 +702,9 @@ export async function retrieveKnowledgeSearch( if (hasQuery && !queryVector) { throw new Error('Query vector is required when searching with a query') } - /** - * The one seam between the two retrieval strategies. A signed-in reader's search over search - * indexes, while indexed organization search is on, binds the reader's resolved plan and ranks - * on the projection rows; the dormant module loads only then. Everything else decides - * readability on each candidate's document, under the caller's own tokens and any live source - * proof they hold. - */ - const legs: RetrievalLegs = - access.kind === 'user' && accessProvider && params.indexedRetrieval === true - ? await (await import('@/lib/sim-search/indexed/retrieval')).prepareIndexedRetrieval({ - knowledgeBaseIds, - access, - accessProvider, - filters: params.filters, - signal: params.signal, - ranked: hasQuery, - budget: budgets.vector, - }) - : documentRetrievalLegs( - await resolveDocumentReadAccess(knowledgeBaseIds, access, accessProvider, params.signal) - ) + const legs = documentRetrievalLegs( + await resolveDocumentReadAccess(knowledgeBaseIds, access, accessProvider, params.signal) + ) const common = { knowledgeBaseIds, access, diff --git a/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts b/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts deleted file mode 100644 index 31c4e68ca10..00000000000 --- a/apps/sim/lib/knowledge/search/source-vector-indexes.test.ts +++ /dev/null @@ -1,21 +0,0 @@ -import { dbChainMockFns, resetDbChainMock } from '@sim/testing' -import { beforeEach, describe, expect, it } from 'vitest' -import { dropSourceVectorIndex } from '@/lib/knowledge/search/source-vector-indexes' - -describe('source vector indexes', () => { - let statements: string[] - - beforeEach(() => { - resetDbChainMock() - statements = [] - dbChainMockFns.execute.mockImplementation(async (query: unknown) => { - statements.push(typeof query === 'string' ? query : JSON.stringify(query)) - return [] - }) - }) - - it('never spells an unexpected identifier into DDL', async () => { - await dropSourceVectorIndex("x'; DROP TABLE document; --") - expect(statements.some((text) => text.includes('DROP TABLE'))).toBe(false) - }) -}) diff --git a/apps/sim/lib/knowledge/search/source-vector-indexes.ts b/apps/sim/lib/knowledge/search/source-vector-indexes.ts deleted file mode 100644 index 4a12dea2038..00000000000 --- a/apps/sim/lib/knowledge/search/source-vector-indexes.ts +++ /dev/null @@ -1,24 +0,0 @@ -import { db } from '@sim/db' -import { sql } from 'drizzle-orm' - -/** - * An index's name and predicate are spelled into DDL, which takes no parameters, so a connector id - * is only ever used after it matches the shape connectors carry. - */ -const CONNECTOR_ID = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/ - -/** Derived, never stored, so a dropped source leaves nothing to reconcile. */ -function indexName(connectorId: string): string { - return `embedding_search_src_${connectorId.replaceAll('-', '')}_hnsw` -} - -/** - * Drops a deleted source's per-source vector index, where an earlier build left one. Nothing - * builds these any more; the dormant indexed search (`lib/sim-search/indexed/retrieval/`) still - * reads the ones that exist, and its brief cache of them only names sources a search can no - * longer reach once their connector is deleted. - */ -export async function dropSourceVectorIndex(connectorId: string): Promise { - if (!CONNECTOR_ID.test(connectorId)) return - await db.execute(sql.raw(`DROP INDEX CONCURRENTLY IF EXISTS "${indexName(connectorId)}"`)) -} diff --git a/apps/sim/lib/knowledge/search/stats.test.ts b/apps/sim/lib/knowledge/search/stats.test.ts deleted file mode 100644 index 14f2cd5146b..00000000000 --- a/apps/sim/lib/knowledge/search/stats.test.ts +++ /dev/null @@ -1,63 +0,0 @@ -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { organizationSearchStatsQuerySchema } from '@/lib/api/contracts/knowledge/search-stats' -import { getSearchStatsRangeError, getSearchStatsWindow } from '@/lib/knowledge/search/stats' - -const now = new Date('2026-09-10T20:00:00.000Z') -beforeEach(() => { - vi.useFakeTimers() - vi.setSystemTime(now) -}) -afterEach(() => vi.useRealTimers()) - -describe('Search stats window and contract', () => { - it('includes the last custom day with an exclusive next-midnight bound', () => { - expect( - getSearchStatsWindow('custom', now, { startDate: '2026-08-31', endDate: '2026-09-02' }) - ).toEqual({ - start: new Date('2026-08-31T00:00:00.000Z'), - end: new Date('2026-09-03T00:00:00.000Z'), - days: 3, - }) - expect( - getSearchStatsWindow('custom', now, { startDate: '2026-09-10', endDate: '2026-09-10' }) - ).toEqual({ - start: new Date('2026-09-10T00:00:00.000Z'), - end: now, - days: 1, - }) - }) - it.each([ - { startDate: '2026-09-01' }, - { endDate: '2026-09-01' }, - { startDate: '2026-02-29', endDate: '2026-03-01' }, - { startDate: '2026-04-31', endDate: '2026-05-01' }, - { startDate: '2026-9-01', endDate: '2026-09-02' }, - { startDate: '2026-09-03', endDate: '2026-09-02' }, - { startDate: '2026-09-10', endDate: '2026-09-11' }, - { startDate: '2026-06-12', endDate: '2026-09-10' }, - ])('rejects invalid or unbounded custom ranges: %j', (range) => { - expect(getSearchStatsRangeError(range, now)).not.toBeNull() - expect( - organizationSearchStatsQuerySchema.safeParse({ - organizationId: 'org', - period: 'custom', - ...range, - }).success - ).toBe(false) - expect(() => getSearchStatsWindow('custom', now, range)).toThrow() - }) - it('bounds scans and rejects caller-supplied surfaces', () => { - expect( - organizationSearchStatsQuerySchema.safeParse({ organizationId: 'org', period: '365d' }) - .success - ).toBe(false) - expect( - organizationSearchStatsQuerySchema.safeParse({ organizationId: 'org', surface: 'forged' }) - .success - ).toBe(false) - expect(organizationSearchStatsQuerySchema.parse({ organizationId: 'org' })).toEqual({ - organizationId: 'org', - period: '30d', - }) - }) -}) diff --git a/apps/sim/lib/knowledge/search/stats.ts b/apps/sim/lib/knowledge/search/stats.ts deleted file mode 100644 index 35dbf4ded42..00000000000 --- a/apps/sim/lib/knowledge/search/stats.ts +++ /dev/null @@ -1,74 +0,0 @@ -export const SEARCH_STATS_PERIODS = ['today', '3d', '7d', '14d', '30d', '90d', 'custom'] as const -export const SEARCH_STATS_SURFACES = [ - 'dashboard', - 'copilot', - 'mcp', - 'slack', - 'api', - 'workflow', - 'other', -] as const -export const SEARCH_STATS_MAX_DAYS = 90 -const DAY_MS = 86_400_000 - -export const SEARCH_STATS_PEOPLE_LIMIT = 20 -export const SEARCH_STATS_SOURCE_LIMIT = 100 - -export const SEARCH_STATS_SURFACE_LABELS: Record<(typeof SEARCH_STATS_SURFACES)[number], string> = { - dashboard: 'Search', - copilot: 'Assistant', - mcp: 'MCP', - slack: 'Slack', - api: 'API', - workflow: 'Workflows', - other: 'Other', -} - -export interface SearchStatsDateRange { - startDate?: string - endDate?: string -} - -/** Checks the date-only UTC range shared by the picker, API, and aggregate query. */ -export function getSearchStatsRangeError( - range: SearchStatsDateRange, - now = new Date() -): string | null { - const { startDate, endDate } = range - if (!startDate || !endDate) return 'Select both a start date and an end date.' - for (const date of [startDate, endDate]) { - if (!/^\d{4}-\d{2}-\d{2}$/.test(date)) return 'Use dates in YYYY-MM-DD format.' - const parsed = new Date(`${date}T00:00:00.000Z`) - if (!Number.isFinite(parsed.getTime()) || parsed.toISOString().slice(0, 10) !== date) { - return 'Select valid calendar dates.' - } - } - if (endDate < startDate) return 'The end date must be on or after the start date.' - if (endDate > now.toISOString().slice(0, 10)) return 'Select dates on or before today (UTC).' - const days = (Date.parse(endDate) - Date.parse(startDate)) / DAY_MS + 1 - if (days > SEARCH_STATS_MAX_DAYS) - return `Select a range of ${SEARCH_STATS_MAX_DAYS} days or fewer.` - return null -} - -/** Daily UTC buckets, with the selected end date included and today capped at the current time. */ -export function getSearchStatsWindow( - period: (typeof SEARCH_STATS_PERIODS)[number], - now = new Date(), - range: SearchStatsDateRange = {} -) { - if (period === 'custom') { - const error = getSearchStatsRangeError(range, now) - if (error) throw new Error(error) - const start = new Date(`${range.startDate}T00:00:00.000Z`) - const endDay = new Date(`${range.endDate}T00:00:00.000Z`) - const end = new Date(Math.min(endDay.getTime() + DAY_MS, now.getTime())) - const days = (endDay.getTime() - start.getTime()) / DAY_MS + 1 - return { start, end, days } - } - const daysByPeriod = { today: 1, '3d': 3, '7d': 7, '14d': 14, '30d': 30, '90d': 90 } as const - const days = daysByPeriod[period] - const start = new Date(Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), now.getUTCDate())) - start.setUTCDate(start.getUTCDate() - days + 1) - return { start, end: now, days } -} diff --git a/apps/sim/lib/mothership/application/load-search-integrations.test.ts b/apps/sim/lib/mothership/application/load-search-integrations.test.ts index e1d38de0db7..e2bf1887645 100644 --- a/apps/sim/lib/mothership/application/load-search-integrations.test.ts +++ b/apps/sim/lib/mothership/application/load-search-integrations.test.ts @@ -1,4 +1,4 @@ -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing' +import { resetEnvFlagsMock } from '@sim/testing' import { mothershipOrganizationChatsMock, mothershipOrganizationChatsMockFns, @@ -42,14 +42,17 @@ const emptyPage: InventoryPage = { describe('loadCopilotSearchIntegrations', () => { beforeEach(() => { resetEnvFlagsMock() - /** The paged inventory below is the indexed arm; the live test opts back in. */ - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) + authorizeChat.mockResolvedValue(undefined) listIntegrations.mockResolvedValue(emptyPage) + liveAccounts.mockResolvedValue({ backend: 'live', accounts: [] }) }) it('authorizes the private chat and reads only for the authenticated person and organization', async () => { - expect(await loadCopilotSearchIntegrations(context)).toBe('{"connections":[],"available":[]}') + expect(JSON.parse(await loadCopilotSearchIntegrations(context))).toMatchObject({ + connections: [], + available: [], + }) const principal = authorizeChat.mock.calls[0][0].principal expect(principal).toMatchObject({ kind: 'organization_delegated', @@ -70,7 +73,6 @@ describe('loadCopilotSearchIntegrations', () => { }) it('includes exact live connection targets alongside current provider search accounts', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) const target = { type: 'link', provider: 'slack', @@ -97,45 +99,6 @@ describe('loadCopilotSearchIntegrations', () => { }) }) - it('loads every page and preserves account status and exact connection controls', async () => { - const available: InventoryPage['available'][number] = { - name: 'Gmail', - description: 'Personal mail', - target: { type: 'link', provider: 'google-email', connectorType: 'gmail' }, - } - const connection: InventoryPage['connections'][number] = { - name: 'Gmail', - providerId: 'google-email', - connectorType: 'gmail', - connectorId: 'source-1', - knowledgeBaseId: 'kb-1', - description: 'Personal mail', - accounts: [ - { - credentialId: 'account-1', - displayName: 'me@example.com', - status: 'reconnect_needed', - action: { ...available.target, connectorId: 'source-1', credentialId: 'account-1' }, - }, - ], - connectionStatus: 'reconnect_needed', - indexingStatus: 'indexed', - action: null, - } - listIntegrations - .mockResolvedValueOnce({ ...emptyPage, available: [available], nextCursor: 'page-2' }) - .mockResolvedValueOnce({ ...emptyPage, connections: [connection], available: [available] }) - - expect(JSON.parse(await loadCopilotSearchIntegrations(context))).toEqual({ - connections: [connection], - available: [available], - }) - expect(listIntegrations.mock.calls[1][0]).toEqual({ - principal: authorizeChat.mock.calls[0][0].principal, - input: { organizationId: 'org-1', cursor: 'page-2' }, - }) - }) - it('does not read inventory when private-chat authorization fails', async () => { authorizeChat.mockRejectedValueOnce(new Error('Chat belongs to another person')) await expect(loadCopilotSearchIntegrations(context)).rejects.toThrow( @@ -144,30 +107,6 @@ describe('loadCopilotSearchIntegrations', () => { expect(listIntegrations).not.toHaveBeenCalled() }) - it('fails the turn if a later page cannot be read', async () => { - listIntegrations - .mockResolvedValueOnce({ ...emptyPage, nextCursor: 'page-2' }) - .mockRejectedValueOnce(new Error('Inventory unavailable')) - await expect(loadCopilotSearchIntegrations(context)).rejects.toThrow('Inventory unavailable') - }) - - it('rejects a repeated cursor instead of looping or returning partial inventory', async () => { - listIntegrations.mockResolvedValue({ ...emptyPage, nextCursor: 'page-2' }) - await expect(loadCopilotSearchIntegrations(context)).rejects.toThrow( - 'pagination did not advance' - ) - expect(listIntegrations).toHaveBeenCalledTimes(2) - }) - - it('bounds page loading when the inventory never ends', async () => { - listIntegrations.mockImplementation(async () => ({ - ...emptyPage, - nextCursor: `page-${listIntegrations.mock.calls.length + 1}`, - })) - await expect(loadCopilotSearchIntegrations(context)).rejects.toThrow('exceeded 100 pages') - expect(listIntegrations).toHaveBeenCalledTimes(100) - }) - it('rejects oversized prompt content instead of silently truncating it', async () => { listIntegrations.mockResolvedValueOnce({ ...emptyPage, diff --git a/apps/sim/lib/mothership/application/load-search-integrations.ts b/apps/sim/lib/mothership/application/load-search-integrations.ts index 68c85837496..6a728fbef7e 100644 --- a/apps/sim/lib/mothership/application/load-search-integrations.ts +++ b/apps/sim/lib/mothership/application/load-search-integrations.ts @@ -5,8 +5,6 @@ import { createTrustedOrganizationCopilotPrincipal, } from '@/lib/mothership/auth/application-delegation' import { authorizeOrganizationChatDelegation } from '@/lib/mothership/chat/organization-chats' -import { loadIndexedSearchIntegrationInventory } from '@/lib/sim-search/indexed' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { listLiveSearchAccounts } from '@/lib/sim-search/live/application' const MAX_INVENTORY_BYTES = 256 * 1024 @@ -33,14 +31,6 @@ export async function loadCopilotSearchIntegrations( ) await authorizeOrganizationChatDelegation.execute({ principal }) - if (isIndexedOrgSearchEnabled()) - return loadIndexedSearchIntegrationInventory({ - principal, - organizationId: context.organizationId, - signal: context.signal, - maxBytes: MAX_INVENTORY_BYTES, - }) - const [inventory, connections] = await Promise.all([ listLiveSearchAccounts.execute({ principal, diff --git a/apps/sim/lib/mothership/assistant/connected-account-tool.test.ts b/apps/sim/lib/mothership/assistant/connected-account-tool.test.ts index d3e7b8e58c3..360f6356c3d 100644 --- a/apps/sim/lib/mothership/assistant/connected-account-tool.test.ts +++ b/apps/sim/lib/mothership/assistant/connected-account-tool.test.ts @@ -7,9 +7,8 @@ import { getIssueV2Tool } from '@/tools/github/get_issue' import { searchIssuesV2Tool } from '@/tools/github/search_issues' describe('GitHub Assistant connected-account adapter', () => { - it('preserves Build and flag-off tool configuration without mutating the registry', () => { - expect(projectAssistantConnectedAccountTool(getIssueV2Tool, false)).toBe(getIssueV2Tool) - const adapted = projectAssistantConnectedAccountTool(getIssueV2Tool, true) + it('preserves Build tool configuration without mutating the registry', () => { + const adapted = projectAssistantConnectedAccountTool(getIssueV2Tool) expect(adapted).not.toBe(getIssueV2Tool) expect(getIssueV2Tool.params.apiKey.required).toBe(true) expect(getIssueV2Tool.oauth).toBeUndefined() @@ -23,7 +22,7 @@ describe('GitHub Assistant connected-account adapter', () => { expect(assistantConnectedAccountTokenParam(adapted)).toBe('apiKey') }) it('uses the same adapter for the existing issue/PR search tool with total_count', () => { - const adapted = projectAssistantConnectedAccountTool(searchIssuesV2Tool, true) + const adapted = projectAssistantConnectedAccountTool(searchIssuesV2Tool) expect(adapted.request).toBe(searchIssuesV2Tool.request) expect(adapted.transformResponse).toBe(searchIssuesV2Tool.transformResponse) expect(adapted.outputs).toBe(searchIssuesV2Tool.outputs) @@ -31,7 +30,7 @@ describe('GitHub Assistant connected-account adapter', () => { }) it('does not turn unrelated API-key tools or GitLab admin sources into personal credentials', () => { const tool = { ...getIssueV2Tool, id: 'gitlab_get_project' } - expect(projectAssistantConnectedAccountTool(tool, true)).toBe(tool) + expect(projectAssistantConnectedAccountTool(tool)).toBe(tool) expect(assistantConnectedAccountTokenParam(tool)).toBeUndefined() }) }) diff --git a/apps/sim/lib/mothership/assistant/connected-account-tool.ts b/apps/sim/lib/mothership/assistant/connected-account-tool.ts index 3569a8d70dd..910b40bf8ce 100644 --- a/apps/sim/lib/mothership/assistant/connected-account-tool.ts +++ b/apps/sim/lib/mothership/assistant/connected-account-tool.ts @@ -1,12 +1,8 @@ import type { ToolMetadata } from '@/tools/metadata' /** Adapt legacy GitHub API-token operations only at the Assistant boundary; Build schemas stay unchanged. */ -export function projectAssistantConnectedAccountTool( - tool: T, - liveSearch: boolean -): T { - if (!liveSearch || !/^github_[a-z0-9_]+$/.test(tool.id) || !tool.params.apiKey || tool.oauth) - return tool +export function projectAssistantConnectedAccountTool(tool: T): T { + if (!/^github_[a-z0-9_]+$/.test(tool.id) || !tool.params.apiKey || tool.oauth) return tool return { ...tool, oauth: { diff --git a/apps/sim/lib/mothership/assistant/tool-policy.ts b/apps/sim/lib/mothership/assistant/tool-policy.ts index 3ccb8b824e6..0c8bcf5251e 100644 --- a/apps/sim/lib/mothership/assistant/tool-policy.ts +++ b/apps/sim/lib/mothership/assistant/tool-policy.ts @@ -1,4 +1,3 @@ -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { projectAssistantConnectedAccountTool } from '@/lib/mothership/assistant/connected-account-tool' import type { ToolMetadata } from '@/tools/metadata' @@ -14,7 +13,7 @@ const CREDENTIAL_PARAMS = new Set(['credential', 'credentialId', 'oauthCredentia /** Assistant uses the regular integration registry, with authentication supplied by the caller's account. */ export function isAssistantIntegrationTool(tool: ToolMetadata | undefined): boolean { if (!tool) return false - tool = projectAssistantConnectedAccountTool(tool, isLiveEnterpriseSearchEnabled) + tool = projectAssistantConnectedAccountTool(tool) const tokenBinding = tool.personalToken const supportsToken = tokenBinding && tool.params[tokenBinding.tokenParam] && tool.params[tokenBinding.hostParam] @@ -35,7 +34,7 @@ export function isAssistantIntegrationTool(tool: ToolMetadata | undefined): bool } export function isAssistantIntegrationParameter(tool: ToolMetadata, name: string): boolean { - tool = projectAssistantConnectedAccountTool(tool, isLiveEnterpriseSearchEnabled) + tool = projectAssistantConnectedAccountTool(tool) if (CREDENTIAL_PARAMS.has(name)) return true if (name === tool.personalToken?.tokenParam || name === tool.personalToken?.hostParam) return false diff --git a/apps/sim/lib/mothership/chat/payload.ts b/apps/sim/lib/mothership/chat/payload.ts index ed649d21e75..7ca78e3d76f 100644 --- a/apps/sim/lib/mothership/chat/payload.ts +++ b/apps/sim/lib/mothership/chat/payload.ts @@ -7,7 +7,7 @@ import { LRUCache } from 'lru-cache' import { getHighestPrioritySubscription } from '@/lib/billing/core/subscription' import { isPaid } from '@/lib/billing/plan-helpers' import type { BlockVisibilityState } from '@/lib/core/config/block-visibility' -import { isHosted, isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' +import { isHosted } from '@/lib/core/config/env-flags' import { isOAuthServiceDeploymentAvailable } from '@/lib/integrations/availability.server' import { type IntegrationGateConfig, @@ -218,9 +218,7 @@ export async function buildIntegrationToolSchemas( return structuredClone( schemas.filter((schema) => { const original = getToolMetadata(schema.name) - const metadata = original - ? projectAssistantConnectedAccountTool(original, isLiveEnterpriseSearchEnabled) - : undefined + const metadata = original ? projectAssistantConnectedAccountTool(original) : undefined return ( !metadata?.personalToken && metadata?.oauth?.required && @@ -250,7 +248,7 @@ async function buildIntegrationToolSchemasUncached({ const metadata = getToolMetadata(toolId) if (options.personalAccountsOnly && !isAssistantIntegrationTool(metadata)) continue const projectedTool = options.personalAccountsOnly - ? projectAssistantConnectedAccountTool(toolConfig, isLiveEnterpriseSearchEnabled) + ? projectAssistantConnectedAccountTool(toolConfig) : toolConfig const userSchema = createUserToolSchema(projectedTool, { surface: options.schemaSurface, diff --git a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts index 39ff46cf777..1b78b765008 100644 --- a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts +++ b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts @@ -1,4 +1,4 @@ -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' +import { resetEnvFlagsMock } from '@sim/testing/mocks/env-flags.mock' import { getMockLogger } from '@sim/testing/mocks/logger.mock' import { mothershipOrganizationChatsMock, @@ -11,20 +11,14 @@ const hoisted = vi.hoisted(() => ({ read: vi.fn(), })) vi.mock('@/lib/mothership/chat/organization-chats', () => mothershipOrganizationChatsMock) -vi.mock('@/lib/sim-search/indexed', () => ({ - searchOrganizationKnowledge: { +vi.mock('@/lib/sim-search/live/application', () => ({ + searchLiveKnowledge: { get operation() { return knowledgeOperations.search }, execute: hoisted.search, }, - searchWorkspaceKnowledge: { - get operation() { - return knowledgeOperations.search - }, - execute: hoisted.search, - }, - readSearchDocument: { + readLiveDocument: { get operation() { return knowledgeOperations.readDocument }, @@ -32,9 +26,7 @@ vi.mock('@/lib/sim-search/indexed', () => ({ }, })) -import { EmbeddingConfigurationError } from '@/lib/embeddings/configuration-error' import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' import { readDocumentServerTool, searchWorkspaceServerTool, @@ -61,9 +53,8 @@ const context = { describe('Assistant retrieval tools', () => { afterEach(resetEnvFlagsMock) beforeEach(() => { - /** These cover the indexed arm of Sim's search and read tools. */ - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) - mocks.search.mockResolvedValue({ + mocks.search.mockImplementation(async ({ input }: { input: { query: string } }) => ({ + query: input.query, retrieval: { status: 'complete', timedOutLegs: [] }, knowledgeBases: [{ id: 'index', name: 'Enterprise Search' }], results: [ @@ -79,7 +70,7 @@ describe('Assistant retrieval tools', () => { similarity: 1, }, ], - }) + })) mocks.read.mockResolvedValue({ knowledgeBaseId: 'index', documentId: 'doc', @@ -93,7 +84,6 @@ describe('Assistant retrieval tools', () => { it.each([{ startDate: '2026-09-01T00:00:00Z' }, { sortBy: 'newest' }, { sortBy: 'oldest' }])( 'returns actionable validation for empty Notion native queries with %j', async (bound) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) const result = await searchWorkspaceServerTool.execute( { ...bound, @@ -113,7 +103,6 @@ describe('Assistant retrieval tools', () => { { provider: 'slack', kind: 'meeting' }, { provider: 'google_drive', kind: 'transcript' }, ])('rejects $provider searches with an unsupported $kind selector', async (selection) => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) const result = await searchWorkspaceServerTool.execute( { query: 'release', nativeQueries: [{ ...selection, query: 'release' }] }, { ...context, assistantSearch: undefined } @@ -124,89 +113,6 @@ describe('Assistant retrieval tools', () => { }) expect(result).not.toHaveProperty('data') }) - it('returns a safe permanent configuration failure instead of empty results or opaque error', async () => { - mocks.search.mockRejectedValue(new EmbeddingConfigurationError()) - const result = await searchWorkspaceServerTool.execute({ query: 'policy' }, context) - expect(result).toMatchObject({ - success: false, - retryable: false, - capability: 'semantic_retrieval', - reason: 'provider_not_configured', - recovery: expect.stringContaining('authorized original'), - message: expect.stringContaining('embedding provider is not configured'), - }) - expect(result).not.toHaveProperty('data') - expect(result).not.toHaveProperty('results') - }) - it('returns empty incomplete retrieval as a recoverable search outcome and logs coverage', async () => { - mocks.search.mockImplementation(async () => { - annotateSearchDiagnostics({ - retrievalStatus: 'partial', - timedOutLegs: ['vector', 'keyword'], - }) - return { - retrieval: { status: 'partial', timedOutLegs: ['vector', 'keyword'] }, - knowledgeBases: [{ id: 'index', name: 'Enterprise Search' }], - results: [], - } - }) - - const result = await searchWorkspaceServerTool.execute({ query: 'canaries' }, context) - - expect(result).toMatchObject({ - success: true, - message: expect.stringContaining('cannot establish absence or completeness'), - data: { - retrieval: { status: 'partial', timedOutLegs: ['vector', 'keyword'] }, - results: [], - }, - }) - expect(result).not.toHaveProperty('error') - expect(mocks.info).toHaveBeenCalledWith( - 'Knowledge search completed', - expect.objectContaining({ passageBytes: 0, originalPassageBytes: 0, outcome: 'partial' }) - ) - }) - it('pins organization and private chat while reusing the canonical search index and citations', async () => { - const orgContext = { - ...context, - workspaceId: undefined, - organizationId: 'org-1', - chatId: 'chat-1', - requestMode: 'assistant', - } - const result = await searchWorkspaceServerTool.execute( - { query: 'policy', organizationId: 'forged' }, - orgContext - ) - expect(result).toMatchObject({ - success: true, - data: { - results: [ - expect.objectContaining({ - citationUrl: expect.stringContaining('/o/org-1/knowledge/index/doc'), - }), - ], - }, - }) - expect(mocks.authorizeChat).toHaveBeenCalledWith({ - principal: expect.objectContaining({ - kind: 'organization_delegated', - subjectUserId: 'reader', - organizationId: 'org-1', - resourceScope: { chatId: 'chat-1' }, - }), - }) - expect(mocks.search).toHaveBeenCalledWith({ - principal: expect.objectContaining({ organizationId: 'org-1' }), - input: expect.objectContaining({ organizationId: 'org-1', filters: context.assistantSearch }), - }) - await readDocumentServerTool.execute({ documentId: 'doc' }, orgContext) - expect(mocks.read).toHaveBeenCalledWith({ - principal: expect.objectContaining({ organizationId: 'org-1' }), - input: expect.objectContaining({ assertedOrganizationId: 'org-1' }), - }) - }) it.each([ { searchSurface: undefined, expected: 'copilot' }, @@ -284,97 +190,6 @@ describe('Assistant retrieval tools', () => { ) expect(JSON.stringify(result)).not.toContain(secret) }) - - it.each([0, 20, 50])( - 'measures UTF-8 bytes for %i passages without logging their content', - async (count) => { - const content = 'Confidential passage é🔎'.repeat(100) - mocks.search.mockResolvedValueOnce({ - retrieval: { status: 'complete', timedOutLegs: [] }, - knowledgeBases: [{ id: 'index', name: 'Enterprise Search' }], - results: Array.from({ length: count }, (_, index) => ({ - knowledgeBaseId: 'index', - documentId: `doc-${index % 4}`, - documentName: 'Private title', - sourceUrl: null, - sourceModifiedAt: null, - metadata: {}, - content, - chunkIndex: index, - similarity: 1, - })), - }) - - const output = await searchWorkspaceServerTool.execute( - { query: 'Private query', ...(count === 50 ? { topK: 50 } : {}) }, - context - ) - - expect(output.success).toBe(true) - expect(mocks.info).toHaveBeenCalledWith( - 'Knowledge search completed', - expect.objectContaining({ - toolCallId: 'call', - toolResultBytes: Buffer.byteLength(JSON.stringify(output)), - passageBytes: count * Buffer.byteLength(content.slice(0, 1200)), - originalPassageBytes: count * Buffer.byteLength(content), - maxPassageBytes: count ? Buffer.byteLength(content.slice(0, 1200)) : 0, - uniqueDocumentCount: Math.min(count, 4), - }) - ) - const logged = JSON.stringify(mocks.info.mock.calls) - expect(logged).not.toContain('Confidential passage') - expect(logged).not.toContain('Private title') - expect(logged).not.toContain('Private query') - } - ) - - it('returns stable citation IDs with internal links for uploaded documents', async () => { - const result = await searchWorkspaceServerTool.execute({ query: 'orion' }, context) - expect(result).toMatchObject({ - success: true, - data: { - results: [ - expect.objectContaining({ - citationId: 'document:doc', - citationUrl: expect.stringContaining('/workspace/workspace/knowledge/index/doc'), - }), - ], - }, - }) - }) - it('projects the provider name for connected-source citations instead of the index name', async () => { - mocks.search.mockResolvedValueOnce({ - retrieval: { status: 'complete', timedOutLegs: [] }, - knowledgeBases: [{ id: 'index', name: 'Sim Search' }], - results: [ - { - knowledgeBaseId: 'index', - documentId: 'doc', - documentName: 'Launch checklist', - sourceUrl: 'https://mail.google.com/thread', - connectorType: 'gmail', - sourceModifiedAt: null, - metadata: {}, - content: 'body', - chunkIndex: 0, - similarity: 1, - }, - ], - }) - expect(await searchWorkspaceServerTool.execute({ query: 'launch' }, context)).toMatchObject({ - success: true, - data: { - results: [ - expect.objectContaining({ - documentName: 'Launch checklist', - siteName: 'Gmail', - knowledgeBaseName: 'Sim Search', - }), - ], - }, - }) - }) it('rejects untrusted contexts, incompatible sources and out-of-scope document reads', async () => { expect( await searchWorkspaceServerTool.execute( @@ -391,34 +206,4 @@ describe('Assistant retrieval tools', () => { expect(mocks.search).not.toHaveBeenCalled() expect(mocks.read).not.toHaveBeenCalled() }) - it('reads a selected document through the shared use case and caps oversized pages', async () => { - expect( - await readDocumentServerTool.execute({ documentId: 'doc', startChunkIndex: 20 }, context) - ).toMatchObject({ success: true }) - expect(mocks.read).toHaveBeenCalledWith( - expect.objectContaining({ - input: expect.objectContaining({ - assertedWorkspaceId: 'workspace', - filters: context.assistantSearch, - startChunkIndex: 20, - limit: 3, - }), - }) - ) - mocks.read.mockImplementationOnce(async ({ input }: { input: { limit: number } }) => ({ - knowledgeBaseId: 'index', - documentId: 'doc', - documentName: 'Title', - sourceUrl: 'https://source.test/doc', - chunks: Array.from({ length: input.limit }, (_, chunkIndex) => ({ - content: 'body', - chunkIndex, - })), - hasMore: true, - next: null, - })) - const capped = await readDocumentServerTool.execute({ documentId: 'doc', limit: 9 }, context) - expect(capped).toMatchObject({ success: true }) - expect((capped as { data: { chunks: unknown[] } }).data.chunks).toHaveLength(8) - }) }) diff --git a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts index e50144f149a..ce9957cfec1 100644 --- a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts +++ b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts @@ -7,17 +7,10 @@ import { import { getValidationErrorMessage } from '@/lib/api/server/validation' import { getBaseUrl } from '@/lib/core/utils/urls' import { EmbeddingConfigurationError } from '@/lib/embeddings/configuration-error' -import { sourceAuthor } from '@/lib/knowledge/search/author' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' import { createKnowledgeDocumentCitation, liveCitationId } from '@/lib/knowledge/search/citation' -import { - annotateSearchDiagnostics, - measureSearchStage, - recordSearchStageDuration, - withSearchDiagnostics, -} from '@/lib/knowledge/search/diagnostics' +import { withSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' import { intersectWorkspaceSearchFilters } from '@/lib/knowledge/search/filters' -import { matchPassage } from '@/lib/knowledge/search/snippet' import { executeCopilotKnowledgeUseCase, executeCopilotOrganizationKnowledgeUseCase, @@ -26,12 +19,6 @@ import { } from '@/lib/mothership/application/execute-knowledge-use-case' import type { BaseServerTool, ServerToolContext } from '@/lib/mothership/tools/server/base-tool' import { connectorDisplayName } from '@/lib/sim-search/connectors' -import { - readSearchDocument, - searchOrganizationKnowledge, - searchWorkspaceKnowledge, -} from '@/lib/sim-search/indexed' -import { isIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' import { readLiveDocument, searchLiveKnowledge } from '@/lib/sim-search/live/application' import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' @@ -39,9 +26,7 @@ const logger = createLogger('WorkspaceSearchTool') const CITATION_INSTRUCTION = 'Cite the evidence you use as {"id":""}. Use only IDs returned by these tools.' + - (isIndexedOrgSearchEnabled() - ? '' - : ' When referring to a Slack conversation, link the returned sourceContainerName to its sourceContainerUrl when available.') + ' When referring to a Slack conversation, link the returned sourceContainerName to its sourceContainerUrl when available.' export const searchWorkspaceServerTool: BaseServerTool = { name: 'search_workspace', @@ -56,7 +41,6 @@ export const searchWorkspaceServerTool: BaseServerTool = { }, async () => { try { - const inputStarted = performance.now() const scope = requireCopilotKnowledgeScope(context) const { query, topK, nativeQueries, ...requestedFilters } = searchWorkspaceInputSchema.parse(raw) @@ -79,127 +63,47 @@ export const searchWorkspaceServerTool: BaseServerTool = { resultSecretRegistry: registry, signal: context?.abortSignal, } as const - if (!isIndexedOrgSearchEnabled()) { - const nativeProjection = projectResolvedSecretModelContent( - nativeQueries ?? [], - registry - ) - if (!nativeProjection.safe) - return { - success: false, - message: 'Native queries contain protected content. Rephrase them.', - } - const liveInput = { - ...input, - nativeQueries: searchWorkspaceInputSchema.shape.nativeQueries.parse( - nativeQueries ? nativeProjection.value : undefined - ), - } - const data = - scope.kind === 'organization' - ? await executeCopilotOrganizationKnowledgeUseCase(context, searchLiveKnowledge, { - ...liveInput, - organizationId: scope.organizationId, - }) - : await executeCopilotKnowledgeUseCase(context, searchLiveKnowledge, { - ...liveInput, - workspaceId: scope.workspaceId, - }) - return { - success: true, - message: `Found ${data.results.length} live results. Read a documentId when its passage does not answer the question or more context is needed. ${CITATION_INSTRUCTION}`, - data: { - ...data, - results: data.results.map((item) => ({ - ...item, - siteName: connectorDisplayName(item.connectorType ?? ''), - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: '', - documentId: item.documentId, - sourceUrl: item.sourceUrl, - baseUrl: getBaseUrl(), - }), - citationId: liveCitationId(item.documentId), - })), - }, - } - } - if ( - nativeQueries || - !safeQuery.trim() || - requestedFilters.startDate || - requestedFilters.endDate || - requestedFilters.sortBy - ) + const nativeProjection = projectResolvedSecretModelContent(nativeQueries ?? [], registry) + if (!nativeProjection.safe) return { success: false, - message: - 'Native queries, date-only search, startDate/endDate and sorting require live search. Use modifiedAfter/modifiedBefore with a text query for indexed search.', + message: 'Native queries contain protected content. Rephrase them.', } - recordSearchStageDuration('tool_input', performance.now() - inputStarted) - const result = await measureSearchStage('tool_application', () => + const liveInput = { + ...input, + nativeQueries: searchWorkspaceInputSchema.shape.nativeQueries.parse( + nativeQueries ? nativeProjection.value : undefined + ), + } + const data = scope.kind === 'organization' - ? executeCopilotOrganizationKnowledgeUseCase(context, searchOrganizationKnowledge, { - ...input, + ? await executeCopilotOrganizationKnowledgeUseCase(context, searchLiveKnowledge, { + ...liveInput, organizationId: scope.organizationId, }) - : executeCopilotKnowledgeUseCase(context, searchWorkspaceKnowledge, { - ...input, + : await executeCopilotKnowledgeUseCase(context, searchLiveKnowledge, { + ...liveInput, workspaceId: scope.workspaceId, }) - ) - return await measureSearchStage('tool_presentation', () => { - const names = new Map(result.knowledgeBases.map((base) => [base.id, base.name])) - const output = { - success: true, - message: `${result.retrieval.status === 'partial' ? 'Search coverage is incomplete. Continue with a more specific query or source filter; these results cannot establish absence or completeness. ' : ''}Found ${result.results.length} passage previews. Read a document at its chunkIndex for more context. ${CITATION_INSTRUCTION}`, - data: { - query: safeQuery, - retrieval: result.retrieval, - results: result.results.map((item) => { - const content = projectResolvedSecretModelContent(item.content, registry) - if (!content.safe || typeof content.value !== 'string') - throw new Error('Knowledge result provenance is unavailable') - return { - documentId: item.documentId, - knowledgeBaseId: item.knowledgeBaseId, - knowledgeBaseName: names.get(item.knowledgeBaseId) ?? '', - siteName: item.connectorType - ? connectorDisplayName(item.connectorType) - : names.get(item.knowledgeBaseId), - documentName: item.documentName, - sourceUrl: item.sourceUrl, - connectorType: item.connectorType, - sourceModifiedAt: item.sourceModifiedAt?.toISOString() ?? null, - author: sourceAuthor(item.metadata), - ...matchPassage(content.value, safeQuery, 1200), - chunkIndex: item.chunkIndex, - similarity: item.similarity, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: item.knowledgeBaseId, - documentId: item.documentId, - sourceUrl: item.sourceUrl, - baseUrl: getBaseUrl(), - }), - } + return { + success: true, + message: `${data.retrieval.status === 'partial' ? 'Search coverage is incomplete. Continue with a more specific query or source filter; these results cannot establish absence or completeness. ' : ''}Found ${data.results.length} live results. Read a documentId when its passage does not answer the question or more context is needed. ${CITATION_INSTRUCTION}`, + data: { + ...data, + results: data.results.map((item) => ({ + ...item, + siteName: connectorDisplayName(item.connectorType ?? ''), + ...createKnowledgeDocumentCitation({ + scope, + knowledgeBaseId: '', + documentId: item.documentId, + sourceUrl: item.sourceUrl, + baseUrl: getBaseUrl(), }), - }, - } - const passageBytes = output.data.results.map((item) => Buffer.byteLength(item.content)) - annotateSearchDiagnostics({ - toolResultBytes: Buffer.byteLength(JSON.stringify(output)), - passageBytes: passageBytes.reduce((total, bytes) => total + bytes, 0), - originalPassageBytes: result.results.reduce( - (total, item) => total + Buffer.byteLength(item.content), - 0 - ), - maxPassageBytes: Math.max(0, ...passageBytes), - uniqueDocumentCount: new Set(output.data.results.map((item) => item.documentId)).size, - }) - return output - }) + citationId: liveCitationId(item.documentId), + })), + }, + } } catch (error) { logger.error('Workspace search failed', { error }) return { @@ -238,47 +142,8 @@ export const readDocumentServerTool: BaseServerTool = { const input = readDocumentInputSchema.parse(raw) const registry = context?.resolvedSecretTraceRegistry if (!registry) throw new Error('Knowledge result provenance is unavailable') - if (!isIndexedOrgSearchEnabled()) { - const liveInput = { - ...input, - filters: intersectWorkspaceSearchFilters( - { documentIds: [input.documentId] }, - context?.assistantSearch - ), - resultSecretRegistry: registry, - signal: context?.abortSignal, - } - const data = - scope.kind === 'organization' - ? await executeCopilotOrganizationKnowledgeUseCase(context, readLiveDocument, { - ...liveInput, - organizationId: scope.organizationId, - }) - : await executeCopilotKnowledgeUseCase(context, readLiveDocument, { - ...liveInput, - workspaceId: scope.workspaceId, - }) - return { - success: true, - message: CITATION_INSTRUCTION, - data: { - ...data, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: '', - documentId: data.documentId, - sourceUrl: data.sourceUrl, - baseUrl: getBaseUrl(), - }), - citationId: liveCitationId(data.documentId), - }, - } - } - const readInput = { + const liveInput = { ...input, - ...(scope.kind === 'organization' - ? { assertedOrganizationId: scope.organizationId } - : { assertedWorkspaceId: scope.workspaceId }), filters: intersectWorkspaceSearchFilters( { documentIds: [input.documentId] }, context?.assistantSearch @@ -286,33 +151,31 @@ export const readDocumentServerTool: BaseServerTool = { resultSecretRegistry: registry, signal: context?.abortSignal, } - const result = await measureSearchStage('document_read', () => + const data = scope.kind === 'organization' - ? executeCopilotOrganizationKnowledgeUseCase(context, readSearchDocument, readInput) - : executeCopilotKnowledgeUseCase(context, readSearchDocument, readInput) - ) - const output = { + ? await executeCopilotOrganizationKnowledgeUseCase(context, readLiveDocument, { + ...liveInput, + organizationId: scope.organizationId, + }) + : await executeCopilotKnowledgeUseCase(context, readLiveDocument, { + ...liveInput, + workspaceId: scope.workspaceId, + }) + return { success: true, message: CITATION_INSTRUCTION, data: { - ...result, + ...data, ...createKnowledgeDocumentCitation({ scope, - knowledgeBaseId: result.knowledgeBaseId, - documentId: result.documentId, - sourceUrl: result.sourceUrl, + knowledgeBaseId: '', + documentId: data.documentId, + sourceUrl: data.sourceUrl, baseUrl: getBaseUrl(), }), + citationId: liveCitationId(data.documentId), }, } - annotateSearchDiagnostics({ - toolResultBytes: Buffer.byteLength(JSON.stringify(output)), - passageBytes: result.chunks.reduce( - (total, chunk) => total + Buffer.byteLength(chunk.content), - 0 - ), - }) - return output } catch (error) { logger.error('Document read failed', { error }) return { diff --git a/apps/sim/lib/mothership/tools/server/search-sources.test.ts b/apps/sim/lib/mothership/tools/server/search-sources.test.ts index e16971bcf44..4af61f7fd45 100644 --- a/apps/sim/lib/mothership/tools/server/search-sources.test.ts +++ b/apps/sim/lib/mothership/tools/server/search-sources.test.ts @@ -53,9 +53,11 @@ beforeEach(() => { }) describe('Search source direct tool', () => { it('uses authenticated actor and org and forwards viewer-safe pagination', async () => { - expect(await tool.execute({ action: 'list', cursor: 'previous', mine: true }, context)).toEqual( - { action: 'list', sources: [], nextCursor: 'next' } - ) + expect(await tool.execute({ action: 'list', cursor: 'previous' }, context)).toEqual({ + action: 'list', + sources: [], + nextCursor: 'next', + }) expect(mocks.list).toHaveBeenCalledWith({ principal: expect.objectContaining({ kind: 'organization_delegated', @@ -64,7 +66,7 @@ describe('Search source direct tool', () => { audience: 'sim:knowledge', resourceScope: { chatId: 'actual-chat' }, }), - input: { organizationId: 'actual-org', cursor: 'previous', mine: true }, + input: { organizationId: 'actual-org', cursor: 'previous' }, }) expect(mocks.chat).toHaveBeenCalledBefore(mocks.list) }) diff --git a/apps/sim/lib/mothership/tools/server/settings-connected-accounts.ts b/apps/sim/lib/mothership/tools/server/settings-connected-accounts.ts index 24baafe6418..30bf4b31c4e 100644 --- a/apps/sim/lib/mothership/tools/server/settings-connected-accounts.ts +++ b/apps/sim/lib/mothership/tools/server/settings-connected-accounts.ts @@ -9,14 +9,12 @@ import { inviteOrganizationAccountPeopleBodySchema, listOrganizationAccountPeopleQuerySchema, startOrganizationAccountConnectionBodySchema, - updateOrganizationAccountIndexingBodySchema, updateOrganizationAccountWorkspaceAccessBodySchema, } from '@/lib/api/contracts/organization-accounts' import { getOrganizationAccountWorkspaceAccess, updateOrganizationAccountWorkspaceAccess, } from '@/lib/credential-groups/application/organization-access' -import { updateOrganizationAccountIndexing } from '@/lib/credential-groups/application/organization-account-indexing' import { addOrganizationAccountMcpProvider, inviteOrganizationAccountPeople, @@ -152,17 +150,6 @@ export const connectedAccountSettingsActions = { return { revision: result.revision, grants: result.grants } } ), - set_indexing: settingsOperation( - 'write', - updateOrganizationAccountIndexingBodySchema, - async (context, input) => { - const result = await updateOrganizationAccountIndexing.execute({ - principal: context.principal, - input: { ...input, organizationId: settingsOrganizationId(context) }, - }) - return { enabled: result.enabled, knowledgeBaseIds: result.knowledgeBaseIds } - } - ), add_mcp_provider: settingsOperation( 'write', z.union([ diff --git a/apps/sim/lib/organizations/surface.test.ts b/apps/sim/lib/organizations/surface.test.ts index 469a4b660a0..474b6ba9991 100644 --- a/apps/sim/lib/organizations/surface.test.ts +++ b/apps/sim/lib/organizations/surface.test.ts @@ -49,6 +49,27 @@ describe('getOrganizationSurfaceContext', () => { setEnvFlags({ isInvitationsDisabled: false, isHosted: true, isBillingEnabled: true }) }) + it.each([ + { integrationsDenied: false, knowledgeDenied: false, allowed: true }, + { integrationsDenied: true, knowledgeDenied: false, allowed: false }, + { integrationsDenied: false, knowledgeDenied: true, allowed: false }, + ])( + 'respects Search connection permissions: %o', + async ({ integrationsDenied, knowledgeDenied, allowed }) => { + queueTableRows(member, [{ role: 'member' }]) + queueTableRows(organization, [{ id: 'org-1', name: 'Acme', slug: 'acme', logo: null }]) + queueTableRows(member, [{ memberCount: 1 }]) + mockPermissionConfig.mockResolvedValue({ + ...DEFAULT_PERMISSION_GROUP_CONFIG, + hideIntegrationsTab: integrationsDenied, + hideKnowledgeBaseTab: knowledgeDenied, + }) + await expect(getOrganizationSurfaceContext('org-1', 'viewer')).resolves.toMatchObject({ + viewer: { canConnectSearchIntegrations: allowed }, + }) + } + ) + it.each([ { role: 'owner', billing: true, denied: false, expected: true }, { role: 'admin', billing: true, denied: true, expected: false }, diff --git a/apps/sim/lib/organizations/surface.ts b/apps/sim/lib/organizations/surface.ts index 39237785860..91eb4a34a23 100644 --- a/apps/sim/lib/organizations/surface.ts +++ b/apps/sim/lib/organizations/surface.ts @@ -36,6 +36,7 @@ interface OrganizationSurfaceViewer { role: OrganizationRole isAdmin: boolean canInviteMembers: boolean + canConnectSearchIntegrations: boolean canUsePersonalApiKeys: boolean canUseSearchMcp: boolean } @@ -126,6 +127,9 @@ async function resolveOrganizationSurfaceContext( isAdmin: access.isAdmin, canInviteMembers: access.isAdmin && !isInvitationsDisabled && !capabilityDeniedBy('invitations.send', config), + canConnectSearchIntegrations: + !capabilityDeniedBy('integrations.manage', config) && + !capabilityDeniedBy('knowledge.use', config), canUsePersonalApiKeys: !capabilityDeniedBy('personal_api_key.use', config) && !capabilityDeniedBy('api_keys.manage', config), diff --git a/apps/sim/lib/selectors/application/execute-selector.test.ts b/apps/sim/lib/selectors/application/execute-selector.test.ts index 71563f35f56..f3611cd2469 100644 --- a/apps/sim/lib/selectors/application/execute-selector.test.ts +++ b/apps/sim/lib/selectors/application/execute-selector.test.ts @@ -21,12 +21,8 @@ const hoisted = vi.hoisted(() => ({ resolveReferences: vi.fn(), resolveScope: vi.fn(), sanitize: vi.fn(), - authorizePersonalSearch: vi.fn(), })) -vi.mock('@/lib/knowledge/application/personal-search-account', () => ({ - authorizePersonalSearchSetup: hoisted.authorizePersonalSearch, -})) vi.mock('@/lib/core/application/organization-authorization', () => organizationAuthorizationMock) vi.mock('@sim/audit', () => auditMock) @@ -169,7 +165,6 @@ describe('executeSelector', () => { 'admin', 'knowledge.use' ) - expect(mocks.authorizePersonalSearch).not.toHaveBeenCalled() expect(mocks.getAttachment).not.toHaveBeenCalled() } ) @@ -190,39 +185,6 @@ describe('executeSelector', () => { expect(mocks.executeAttachment).not.toHaveBeenCalled() }) - it('rejects a personal setup marker outside its approved provider selector and organization scope', async () => { - await expect(execute({ personalSearchSetup: 'jira' })).rejects.toBeInstanceOf( - SelectorContextUnavailableError - ) - await expect( - execute({ - scope: { kind: 'organization', organizationId: 'org-1' }, - selectorKey: 'jira.issues', - personalSearchSetup: 'jira', - }) - ).rejects.toBeInstanceOf(SelectorContextUnavailableError) - expect(mocks.authorizeCredential).not.toHaveBeenCalled() - expect(mocks.executeAttachment).not.toHaveBeenCalled() - }) - - it('requires the personal setup authorization before canonical discovery and provider calls', async () => { - mocks.authorizePersonalSearch.mockRejectedValueOnce(new Error('Integration unapproved')) - await expect( - execute({ - scope: { kind: 'organization', organizationId: 'org-1' }, - selectorKey: 'jira.projectKeys', - context: { oauthCredential: 'managed-1', domain: 'example.atlassian.net' }, - personalSearchSetup: 'jira', - }) - ).rejects.toThrow('Integration unapproved') - expect(mocks.authorizePersonalSearch).toHaveBeenCalledWith(principal, { - organizationId: 'org-1', - connectorType: 'jira', - }) - expect(mocks.resolveScope).not.toHaveBeenCalled() - expect(mocks.executeAttachment).not.toHaveBeenCalled() - }) - /** * The picker is a use of the integration, not a neutral list: it reaches the * provider's API with the caller's credential. The authorization funnel never diff --git a/apps/sim/lib/selectors/application/execute-selector.ts b/apps/sim/lib/selectors/application/execute-selector.ts index df881f217fb..a1b2c38fe8d 100644 --- a/apps/sim/lib/selectors/application/execute-selector.ts +++ b/apps/sim/lib/selectors/application/execute-selector.ts @@ -10,7 +10,6 @@ import type { OperationUseCase } from '@/lib/core/application/operation' import { requireOrganizationMembership } from '@/lib/core/application/organization-authorization' import { withResourceOutboundScope } from '@/lib/core/network/resource-scope.server' import { OrchestrationError } from '@/lib/core/orchestration/types' -import { authorizePersonalSearchSetup } from '@/lib/knowledge/application/personal-search-account' import { type CredentialAuditRequest, recordCredentialAccess } from '@/lib/oauth/token-resolution' import { SELECTOR_DELEGATION_AUDIENCE, @@ -45,8 +44,6 @@ const logger = createLogger('ExecuteSelector') export interface ExecuteSelectorInput extends ExecuteSelectorRequest { signal?: AbortSignal auditRequest?: CredentialAuditRequest - /** Set only by the personal Search setup use case; excluded from the public selector contract. */ - personalSearchSetup?: 'jira' | 'confluence' } function validateAuthorizedInput( @@ -172,7 +169,6 @@ async function executeAuthorizedSelector(args: { scope: args.input.scope, workspaceId: args.context.workspaceId, organizationId, - personalSearchSetup: args.input.personalSearchSetup, policy: attachment.credential, protectedValues, references: resolved.references, @@ -203,9 +199,7 @@ async function executeAuthorizedSelector(args: { }) const credentialAccess = credential?.access - const credentialResourceId = credential?.personalSearchSetup - ? credential.suppliedId - : credentialAccess?.resolvedCredentialId + const credentialResourceId = credentialAccess?.resolvedCredentialId let credentialUseRecorded = false const recordCredentialUse = attachment.auditCredentialUse && credentialResourceId @@ -355,25 +349,15 @@ export const executeSelector: OperationUseCase< }, } if (args.input.scope.kind !== 'organization') { - if (args.input.personalSearchSetup) throw new SelectorContextUnavailableError() return executeWorkspaceSelector.execute(args) } if (args.principal.kind !== 'session') throw new SelectorContextUnavailableError() - if (args.input.personalSearchSetup) { - const selectorKey = - args.input.personalSearchSetup === 'jira' ? 'jira.projectKeys' : 'confluence.spaces' - if (args.input.selectorKey !== selectorKey) throw new SelectorContextUnavailableError() - await authorizePersonalSearchSetup(args.principal, { - organizationId: args.input.scope.organizationId, - connectorType: args.input.personalSearchSetup, - }) - } else - await requireOrganizationMembership( - args.principal, - args.input.scope.organizationId, - 'admin', - 'knowledge.use' - ) + await requireOrganizationMembership( + args.principal, + args.input.scope.organizationId, + 'admin', + 'knowledge.use' + ) const context = await resolveSelectorApplicationContext({ selectorKey: args.input.selectorKey as ServerSelectorKey, scope: args.input.scope, diff --git a/apps/sim/lib/selectors/server/credentials.test.ts b/apps/sim/lib/selectors/server/credentials.test.ts index 46ebb61fd5f..ef2d76f9514 100644 --- a/apps/sim/lib/selectors/server/credentials.test.ts +++ b/apps/sim/lib/selectors/server/credentials.test.ts @@ -1,10 +1,6 @@ import { credential } from '@sim/db/schema' import { queueTableRows, resetDbChainMock } from '@sim/testing' import { authOAuthUtilsMock, authOAuthUtilsMockFns } from '@sim/testing/mocks/auth-oauth-utils.mock' -import { - credentialsManagedOauthMock, - credentialsManagedOauthMockFns, -} from '@sim/testing/mocks/credentials-managed-oauth.mock' import { oauthUtilsMock, oauthUtilsMockFns } from '@sim/testing/mocks/oauth-utils.mock' import { beforeEach, describe, expect, it, vi } from 'vitest' @@ -12,7 +8,6 @@ const mocksHoisted = vi.hoisted(() => ({ authorizeCredentialUse: vi.fn(), authorizeOrganizationCredentialUse: vi.fn(), resolveOrganizationCredentialTokenBundle: vi.fn(), - authorizePersonalSearch: vi.fn(), })) vi.mock('@/lib/auth/credential-access', () => ({ @@ -26,11 +21,6 @@ vi.mock('@/lib/credentials/application/organization-credentials', () => ({ vi.mock('@/lib/oauth/credential-service', () => authOAuthUtilsMock) -vi.mock('@/lib/knowledge/application/personal-search-account', () => ({ - authorizePersonalSearchSetupCredential: mocksHoisted.authorizePersonalSearch, -})) -vi.mock('@/lib/credentials/managed-oauth', () => credentialsManagedOauthMock) - vi.mock('@/lib/oauth/utils', () => oauthUtilsMock) import { @@ -45,7 +35,6 @@ const mocks = { resolveCredentialTokenBundle: authOAuthUtilsMockFns.mockResolveCredentialTokenBundle, credentialProviderMatchesService: oauthUtilsMockFns.mockCredentialProviderMatchesService, getServiceConfig: oauthUtilsMockFns.mockGetServiceConfigByServiceId, - resolveManagedOAuthToken: credentialsManagedOauthMockFns.mockResolveManagedOAuthToken, } const principal = { kind: 'session' as const, userId: 'user-1', sessionId: 'session-1' } @@ -124,83 +113,6 @@ describe('authorizeSelectorCredential', () => { }) }) - it('browses with only the member’s own prepared account and refreshes it without the admin credential path', async () => { - mocks.authorizePersonalSearch.mockResolvedValue({ id: 'managed-1', providerId: 'jira' }) - mocks.resolveManagedOAuthToken.mockResolvedValue({ accessToken: 'private-token' }) - const selected = await authorizeSelectorCredential({ - principal, - context: { oauthCredential: 'managed-1' }, - scope: { kind: 'organization', organizationId: 'org-1' }, - organizationId: 'org-1', - personalSearchSetup: 'jira', - policy: { kind: 'stored', field: 'oauthCredential', serviceIds: ['jira'] }, - protectedValues: createSelectorProtectedValues(), - references: new Map(), - }) - const protectedValues = createSelectorProtectedValues() - const recordCredentialUse = vi.fn() - await expect( - resolveSelectorOAuthAccessToken({ - credential: selected, - serviceId: 'jira', - scopes: ['read:jira-work'], - protectedValues, - recordCredentialUse, - }) - ).resolves.toBe('private-token') - expect(mocks.authorizePersonalSearch).toHaveBeenCalledTimes(2) - expect(mocks.resolveManagedOAuthToken).toHaveBeenCalledWith({ - organizationId: 'org-1', - credentialId: 'managed-1', - expectedProviderId: 'jira', - requiredScopes: ['read:jira-work'], - }) - expect(protectedValues.contains('private-token')).toBe(true) - expect(recordCredentialUse).toHaveBeenCalledWith('jira') - expect(mocks.authorizeOrganizationCredentialUse).not.toHaveBeenCalled() - expect(mocks.resolveOrganizationCredentialTokenBundle).not.toHaveBeenCalled() - }) - - it('rejects a revoked prepared credential before refreshing its token', async () => { - mocks.authorizePersonalSearch.mockRejectedValue(new Error('Credential revoked')) - await expect( - resolveSelectorOAuthAccessToken({ - credential: { - suppliedId: 'managed-1', - providerId: 'jira', - personalSearchSetup: { principal, organizationId: 'org-1', connectorType: 'jira' }, - }, - serviceId: 'jira', - protectedValues: createSelectorProtectedValues(), - }) - ).rejects.toThrow('Credential revoked') - expect(mocks.resolveManagedOAuthToken).not.toHaveBeenCalled() - }) - - it('refuses using a prepared Jira credential for another provider or impersonation', async () => { - const selected = { - suppliedId: 'managed-1', - providerId: 'jira', - personalSearchSetup: { principal, organizationId: 'org-1', connectorType: 'jira' as const }, - } - await expect( - resolveSelectorOAuthAccessToken({ - credential: selected, - serviceId: 'confluence', - protectedValues: createSelectorProtectedValues(), - }) - ).rejects.toBeInstanceOf(SelectorConnectionUnavailableError) - await expect( - resolveSelectorOAuthAccessToken({ - credential: selected, - serviceId: 'jira', - impersonateEmail: 'other@example.com', - protectedValues: createSelectorProtectedValues(), - }) - ).rejects.toBeInstanceOf(SelectorConnectionUnavailableError) - expect(mocks.resolveManagedOAuthToken).not.toHaveBeenCalled() - }) - it('rejects an organization connection for the wrong selector provider', async () => { mocks.authorizeOrganizationCredentialUse.mockResolvedValue({ credential: { id: 'managed-1', providerId: 'jira', type: 'managed_oauth' }, diff --git a/apps/sim/lib/selectors/server/credentials.ts b/apps/sim/lib/selectors/server/credentials.ts index aa4eb7ef2ed..1167e61306d 100644 --- a/apps/sim/lib/selectors/server/credentials.ts +++ b/apps/sim/lib/selectors/server/credentials.ts @@ -10,8 +10,6 @@ import { authorizeOrganizationCredentialUse, resolveOrganizationCredentialTokenBundle, } from '@/lib/credentials/application/organization-credentials' -import { resolveManagedOAuthToken } from '@/lib/credentials/managed-oauth' -import { authorizePersonalSearchSetupCredential } from '@/lib/knowledge/application/personal-search-account' import { resolveCredentialTokenBundle } from '@/lib/oauth/credential-service' import { credentialProviderMatchesService, getServiceConfigByServiceId } from '@/lib/oauth/utils' import { SelectorConnectionUnavailableError } from '@/lib/selectors/server/errors' @@ -107,7 +105,6 @@ export async function authorizeSelectorCredential(input: { scope: SelectorScope workspaceId?: string organizationId?: string - personalSearchSetup?: 'jira' | 'confluence' policy: SelectorCredentialPolicy protectedValues: SelectorProtectedValues references: ReadonlyMap @@ -115,32 +112,6 @@ export async function authorizeSelectorCredential(input: { const suppliedId = input.context[input.policy.field] if (!suppliedId) throw new SelectorConnectionUnavailableError() - if (input.personalSearchSetup) { - if ( - input.scope.kind !== 'organization' || - input.principal.kind !== 'session' || - input.organizationId !== input.scope.organizationId || - input.workspaceId || - !input.policy.serviceIds.includes(input.personalSearchSetup) - ) - throw new SelectorConnectionUnavailableError() - const row = await authorizePersonalSearchSetupCredential(input.principal, { - organizationId: input.scope.organizationId, - connectorType: input.personalSearchSetup, - credentialId: suppliedId, - }) - input.protectedValues.add(suppliedId, 'reference') - return { - suppliedId, - providerId: row.providerId, - personalSearchSetup: { - principal: input.principal, - organizationId: input.scope.organizationId, - connectorType: input.personalSearchSetup, - }, - } - } - if (input.scope.kind === 'organization') { if ( input.principal.kind !== 'session' || @@ -217,31 +188,6 @@ export async function resolveSelectorOAuthAccessToken(input: { input.credential.signal?.throwIfAborted() if (input.credential.fixedToken) return input.credential.fixedToken - if (input.credential.personalSearchSetup) { - const setup = input.credential.personalSearchSetup - if (input.impersonateEmail || input.serviceId !== setup.connectorType) { - throw new SelectorConnectionUnavailableError() - } - await authorizePersonalSearchSetupCredential(setup.principal, { - ...setup, - credentialId: input.credential.suppliedId, - }) - const result = await waitForSelectorCredentialResolution( - resolveManagedOAuthToken({ - organizationId: setup.organizationId, - credentialId: input.credential.suppliedId, - expectedProviderId: setup.connectorType, - requiredScopes: input.scopes ? [...input.scopes] : [], - }), - input.credential.signal - ) - input.credential.signal?.throwIfAborted() - if (!result?.accessToken) throw new SelectorConnectionUnavailableError() - input.protectedValues.add(result.accessToken) - input.recordCredentialUse?.(setup.connectorType) - return result.accessToken - } - if (input.credential.organization) { const result = await waitForSelectorCredentialResolution( resolveOrganizationCredentialTokenBundle({ diff --git a/apps/sim/lib/selectors/server/providers/credential-bundle.test.ts b/apps/sim/lib/selectors/server/providers/credential-bundle.test.ts index bbac1e777e9..a98da4a940e 100644 --- a/apps/sim/lib/selectors/server/providers/credential-bundle.test.ts +++ b/apps/sim/lib/selectors/server/providers/credential-bundle.test.ts @@ -1,18 +1,7 @@ import { authOAuthUtilsMock, authOAuthUtilsMockFns } from '@sim/testing/mocks/auth-oauth-utils.mock' -import { - credentialsManagedOauthMock, - credentialsManagedOauthMockFns, -} from '@sim/testing/mocks/credentials-managed-oauth.mock' import { describe, expect, it, vi } from 'vitest' const mockResolveOrganizationToken = vi.hoisted(() => vi.fn()) -const mockOwnAccount = vi.hoisted(() => vi.fn()) - -vi.mock('@/lib/knowledge/application/personal-search-account', () => ({ - authorizePersonalSearchSetupCredential: mockOwnAccount, -})) -vi.mock('@/lib/credentials/managed-oauth', () => credentialsManagedOauthMock) - vi.mock('@/lib/credentials/application/organization-credentials', () => ({ resolveOrganizationCredentialTokenBundle: mockResolveOrganizationToken, })) @@ -24,45 +13,7 @@ import { resolveSelectorCredentialBundle } from '@/lib/selectors/server/provider const mockResolveCredentialAccessToken = authOAuthUtilsMockFns.mockResolveCredentialTokenBundle -const mockResolveManagedToken = credentialsManagedOauthMockFns.mockResolveManagedOAuthToken - describe('selector credential bundles', () => { - it('resolves a personal Atlassian grant through the owned managed-account path', async () => { - mockOwnAccount.mockResolvedValue({ id: 'managed-1', providerId: 'jira' }) - mockResolveManagedToken.mockResolvedValue({ accessToken: 'own-managed-token' }) - const principal = { kind: 'session', userId: 'member-1', sessionId: 'session-1' } as const - const protectedValues = createSelectorProtectedValues() - await expect( - resolveSelectorCredentialBundle({ - credential: { - suppliedId: 'managed-1', - providerId: 'jira', - personalSearchSetup: { principal, organizationId: 'org-1', connectorType: 'jira' }, - }, - providerId: 'jira', - scopes: ['read:jira-work'], - protectedValues, - }) - ).resolves.toEqual({ accessToken: 'own-managed-token' }) - expect(mockOwnAccount).toHaveBeenCalledWith( - principal, - expect.objectContaining({ - credentialId: 'managed-1', - organizationId: 'org-1', - connectorType: 'jira', - }) - ) - expect(mockResolveManagedToken).toHaveBeenCalledWith({ - organizationId: 'org-1', - credentialId: 'managed-1', - expectedProviderId: 'jira', - requiredScopes: ['read:jira-work'], - }) - expect(mockResolveOrganizationToken).not.toHaveBeenCalled() - expect(mockResolveCredentialAccessToken).not.toHaveBeenCalled() - expect(protectedValues.contains('own-managed-token')).toBe(true) - }) - it('protects short credential-bound cloud ids as exact identifiers', async () => { mockResolveCredentialAccessToken.mockResolvedValue({ accessToken: 'server-only-token', diff --git a/apps/sim/lib/selectors/server/providers/credential-bundle.ts b/apps/sim/lib/selectors/server/providers/credential-bundle.ts index b134cf89047..97669ff8a8f 100644 --- a/apps/sim/lib/selectors/server/providers/credential-bundle.ts +++ b/apps/sim/lib/selectors/server/providers/credential-bundle.ts @@ -3,10 +3,7 @@ import { resolveCredentialTokenBundle, type ServiceAccountTokenResult, } from '@/lib/oauth/credential-service' -import { - resolveSelectorOAuthAccessToken, - waitForSelectorCredentialResolution, -} from '@/lib/selectors/server/credentials' +import { waitForSelectorCredentialResolution } from '@/lib/selectors/server/credentials' import { SelectorConnectionUnavailableError } from '@/lib/selectors/server/errors' import type { AuthorizedSelectorCredential, @@ -29,16 +26,6 @@ export async function resolveSelectorCredentialBundle(input: { if (!credential) throw new SelectorConnectionUnavailableError() credential.signal?.throwIfAborted() - if (credential.personalSearchSetup) { - if (!input.providerId) throw new SelectorConnectionUnavailableError() - return { - accessToken: await resolveSelectorOAuthAccessToken({ - ...input, - credential, - serviceId: input.providerId, - }), - } - } if (credential.fixedToken) { if (input.providerId) { input.recordCredentialUse?.(credential.providerId ?? input.providerId) diff --git a/apps/sim/lib/selectors/server/types.ts b/apps/sim/lib/selectors/server/types.ts index 113ac2e2174..b361c675c26 100644 --- a/apps/sim/lib/selectors/server/types.ts +++ b/apps/sim/lib/selectors/server/types.ts @@ -78,11 +78,6 @@ export type SelectorCredentialPolicy = export interface AuthorizedSelectorCredential { suppliedId: string organization?: { principal: SessionPrincipal; organizationId: string } - personalSearchSetup?: { - principal: SessionPrincipal - organizationId: string - connectorType: 'jira' | 'confluence' - } access?: CredentialAccessResult fixedToken?: string /** Trusted provider id loaded during server-side credential binding. */ diff --git a/apps/sim/lib/selectors/types.ts b/apps/sim/lib/selectors/types.ts index 2584863072c..67e9256d5cc 100644 --- a/apps/sim/lib/selectors/types.ts +++ b/apps/sim/lib/selectors/types.ts @@ -109,13 +109,6 @@ export type SelectorScope = workspaceId: string } -/** Chooses a dedicated client transport without granting access through the generic selector API. */ -export interface SelectorSurface { - kind: 'personal-search-setup' - organizationId: string - connectorType: 'jira' | 'confluence' -} - export type SelectorRequest = | { kind: 'list' diff --git a/apps/sim/lib/sim-search/connectors.ts b/apps/sim/lib/sim-search/connectors.ts index b8e892ebf8c..0e7da56295a 100644 --- a/apps/sim/lib/sim-search/connectors.ts +++ b/apps/sim/lib/sim-search/connectors.ts @@ -13,7 +13,7 @@ import { import { CONNECTOR_META_REGISTRY } from '@/connectors/registry' import type { ConnectorConfigField, ConnectorMeta } from '@/connectors/types' -/** The workspace knowledge base Sim Search indexes into, one per workspace, created on first connect. */ +/** The knowledge-base shell that holds live Search source configuration. */ export const SIM_SEARCH_KNOWLEDGE_BASE_NAME = 'Sim Search' /** @@ -130,42 +130,6 @@ export function canConnectWithDefaults(meta: ConnectorMeta): boolean { return canConnectPersonally(meta) && meta.id !== 'slack' && personalSetupFields(meta).length === 0 } -/** - * The settings a person may supply when a source is created from Sim Search: - * its setup fields plus anything the connector's Search defaults cover. - */ -export function personalSourceConfigFieldIds(meta: ConnectorMeta): Set { - return new Set([ - ...personalSetupFields(meta).map((field) => field.id), - ...Object.keys(meta.searchDefaultSourceConfig ?? {}), - ]) -} - -/** - * A Search source's settings, starting from the connector's Search defaults. - * A supplied value replaces its default; a blank one leaves the default in - * place, so an untouched form field never widens the source. - */ -export function withSearchSourceDefaults( - meta: Pick, - sourceConfig: Record = {} -): Record { - const merged: Record = { ...(meta.searchDefaultSourceConfig ?? {}) } - for (const [field, value] of Object.entries(sourceConfig)) { - if (typeof value === 'string' && value.trim() === '' && field in merged) continue - merged[field] = value - } - return merged -} - -/** The setup fields a source config leaves empty. */ -export function missingSetupFields( - meta: ConnectorMeta, - sourceConfig: Record -): ConnectorConfigField[] { - return personalSetupFields(meta).filter((field) => !sourceConfig[field.id]?.trim()) -} - /** The name a connector shows, from its registry entry. */ export function connectorDisplayName(connectorType: string): string { return CONNECTOR_META_REGISTRY[connectorType]?.name ?? connectorType diff --git a/apps/sim/lib/sim-search/indexed/README.md b/apps/sim/lib/sim-search/indexed/README.md deleted file mode 100644 index 07440bedb8e..00000000000 --- a/apps/sim/lib/sim-search/indexed/README.md +++ /dev/null @@ -1,39 +0,0 @@ -# Indexed organization search (dormant) - -The indexed backend for Sim Search: retrieval over `is_search_index` knowledge bases that organization and workspace connectors crawl into, ranked from the embedding projections. Live Search (`../live/`) replaced it. **This code is dormant**: it stays in the tree so it can be switched back on, but no request reaches it in a default deployment. - -Ordinary workspace knowledge bases, the Knowledge block, the embedding projections, the projector, and the document access predicate (`lib/knowledge/access/predicate.ts`) live outside this directory and behave the same whichever backend is selected. - -## The gate - -`isIndexedOrgSearchEnabled()` in `gate.ts` is the only switch. It is the inverse of `SIM_SEARCH_LIVE`, which defaults to `true`, so indexed search is on only where a deployment sets `SIM_SEARCH_LIVE=false`. Every use case in this directory calls `assertIndexedOrgSearchEnabled()` itself, so dormancy holds even for a caller that skipped the gate. - -While the gate is off: - -- Search, the MCP tools, and Sim's `search_workspace` and `read_document` tools serve Live Search, and personal Search integrations are read from live accounts. -- Indexed-only surfaces refuse with `SearchIndexDormantError` (a `409`): the Stats report and connecting a source that crawls into a search index. The indexed document page is not found. -- A knowledge search that names a search-index knowledge base (the Knowledge block, v1, v2, Sim's knowledge tool) still answers from the documents it already holds, decided on each document exactly as a workspace knowledge base is. -- Nothing crawls into search indexes: content syncs, member syncs, processing recovery, document dispatch, and queued document workers skip them (`lib/knowledge/connectors/indexing-policy.ts`). -- The projector owes search-index documents nothing: their marks are released with the rest, and it writes no Tin keyword rows. The GIN keyword projection follows `is_search_index` alone, so it keeps its search-index rows either way. - -## Layout - -- `gate.ts`: the switch, `SearchIndexDormantError`, and the search-index helpers. Anything may import it. -- `index.ts`: the use-case barrel. Its callers branch on the gate first. -- `search/`, `documents/`, `mcp/`, `integrations/`: indexed search, document reads, the indexed MCP tools, and the indexed arms of the personal Search integration inventory. -- `retrieval/`: the search-index retrieval legs behind one entry, `prepareIndexedRetrieval`, which `lib/knowledge/search/queries.ts` loads with a dynamic import only for a signed-in reader whose every base is a search index while the gate is on. Every other search, including every workspace knowledge base search, decides readability on the document and reads none of it. - -The dormant UI sits in `indexed/` folders next to the component that picks it from `features.liveEnterpriseSearch` (`useDeploymentShape()`), so each can be deleted in one step: `app/o/[organizationId]/integrations/indexed/`, `app/o/[organizationId]/settings/components/integrations/indexed/`, and `app/workspace/[workspaceId]/home/components/knowledge-search-results/indexed/`. - -## Re-enabling - -1. Set `SIM_SEARCH_LIVE=false` in both the app and the Trigger.dev environment, and deploy. The container entrypoint (`apps/sim/bootstrap.ts`) mirrors it to `NEXT_PUBLIC_SIM_SEARCH_LIVE` for the client; crawling, processing, and projection read it in whichever process runs them. -2. Confirm the keyword projection objects exist (`0019_tin_keyword_projection`, `0024_knowledge_projection_async`, `0025_scope_keyword_projections`). Both keyword projections, `embedding_keyword_search` and `embedding_keyword_tin`, hold only search-index rows, written by the chunk triggers and by the trigger on `knowledge_base.is_search_index`. Backfill both for every search-index knowledge base whose rows were removed while dormant, and build the Tin index. -3. If the `0027_retire_search_embeddings` cleanup ran for a base, deliberately restore its retired documents' eligibility before resyncing. Its deleted embeddings cannot be recovered by changing the backend flag alone. See `packages/db/script-migrations/search-embedding-retirement.md` for the cleanup lifecycle. -4. Resume and fully resync the connectors of search-index knowledge bases, so content that went stale while dormant is indexed again. - -Projection rows written before projections carried their document's source and ACL are decided on their document until they are rewritten. - -## Database objects it depends on - -`knowledge_base.is_search_index`, `document.acl`, `document.connector_id`, `knowledge_connector`, `embedding_search`, `embedding_keyword_search`, `embedding_keyword_tin`, `knowledge_projection_dirty`, and the Tin extension objects. All of them are owned by `packages/db`; no schema or migration belongs to this directory. diff --git a/apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts b/apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts deleted file mode 100644 index 30b2fa5e947..00000000000 --- a/apps/sim/lib/sim-search/indexed/documents/read-indexed-document.ts +++ /dev/null @@ -1,230 +0,0 @@ -import { db } from '@sim/db' -import { document, embedding } from '@sim/db/schema' -import { and, eq, isNull, lte, sql } from 'drizzle-orm' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { knowledgeAccessCondition } from '@/lib/knowledge/access/predicate' -import type { KnowledgeAccessScope } from '@/lib/knowledge/access/types' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { listKnowledgeChunks } from '@/lib/knowledge/application/chunks' -import { - resolveActiveKnowledgeResourceContext, - resolveKnowledgeOrganizationContext, -} from '@/lib/knowledge/application/contexts' -import { readKnowledgeDocument } from '@/lib/knowledge/application/documents' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import type { ChunkQueryResult } from '@/lib/knowledge/chunks/types' -import { knowledgeReadAccessBatches } from '@/lib/knowledge/read-access' -import { findSearchIndex } from '@/lib/knowledge/search/search-index' -import { isKnowledgeSourceUrl } from '@/lib/knowledge/search/source-url' -import { - createKnowledgeDocumentSourceValue, - importKnowledgePersistedResponseSecretProvenance, -} from '@/lib/knowledge/secret-provenance' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import type { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -type IndexedKnowledgeDocumentTarget = - | { kind: 'id'; documentId: string } - | { kind: 'url'; url: string } - -export interface ReadIndexedKnowledgeDocumentInput { - organizationId: string - target: IndexedKnowledgeDocumentTarget - limit: number - offset?: number - aroundChunkIndex?: number - resultSecretRegistry: ResolvedSecretTraceRegistry - signal?: AbortSignal -} - -export interface ReadIndexedKnowledgeDocumentResult { - knowledgeBaseId: string - documentId: string - title: string - sourceUrl: string | null - sourceModifiedAt: string | null - connectorType: string | null - processingStatus: string - chunks?: { id: string; chunkIndex: number; content: string }[] - pagination?: ChunkQueryResult['pagination'] -} - -function activeDocumentConditions(knowledgeBaseId: string, access?: KnowledgeAccessScope) { - return [ - eq(document.knowledgeBaseId, knowledgeBaseId), - eq(document.enabled, true), - eq(document.userExcluded, false), - isNull(document.archivedAt), - isNull(document.deletedAt), - access ? knowledgeAccessCondition(access) : undefined, - ] -} - -function validateReadInput(input: ReadIndexedKnowledgeDocumentInput) { - if ( - !Number.isInteger(input.limit) || - input.limit < 1 || - input.limit > 50 || - (input.offset !== undefined && - (!Number.isInteger(input.offset) || input.offset < 0 || input.offset > 1_000_000)) || - (input.aroundChunkIndex !== undefined && - (!Number.isInteger(input.aroundChunkIndex) || - input.aroundChunkIndex < 0 || - input.aroundChunkIndex > 1_000_000)) - ) { - throw new OrchestrationError('validation', 'Invalid document page bounds') - } - if (input.offset !== undefined && input.aroundChunkIndex !== undefined) { - throw new OrchestrationError('validation', 'Use offset or aroundChunkIndex, not both') - } - if ( - input.target.kind === 'url' && - (input.target.url.length > 8192 || !isKnowledgeSourceUrl(input.target.url.trim())) - ) { - throw new OrchestrationError('validation', 'url must be an HTTP or HTTPS document URL') - } -} - -/** Resolves only indexed references; URL lookup never contacts the provider or fetches content. */ -export const readIndexedKnowledgeDocument = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.readDocument, - resolveContext: ({ input }: { input: ReadIndexedKnowledgeDocumentInput }) => { - assertIndexedOrgSearchEnabled() - input.signal?.throwIfAborted() - validateReadInput(input) - return resolveKnowledgeOrganizationContext({ organizationId: input.organizationId }) - }, - async execute({ - principal, - input, - context, - request, - }): Promise { - input.signal?.throwIfAborted() - const assertions = { - assertedOrganizationId: context.organizationId, - } - const index = await findSearchIndex({ - kind: 'organization', - organizationId: context.organizationId, - }) - if (!index) throw new OrchestrationError('not_found', 'Document not found') - const knowledgeBaseId = index.id - const knowledgeContext = await resolveActiveKnowledgeResourceContext( - { knowledgeBaseId, ...assertions }, - principal - ) - let documentId: string - if (input.target.kind === 'id') { - documentId = input.target.documentId - } else { - const conditions = [ - ...activeDocumentConditions(knowledgeBaseId), - eq(document.sourceUrl, input.target.url.trim()), - ] - const matches: { id: string }[] = [] - for await (const accessCondition of knowledgeReadAccessBatches( - knowledgeContext.access, - conditions, - input.signal - )) { - matches.push( - ...(await db - .select({ id: document.id }) - .from(document) - .where(and(...conditions, accessCondition)) - .limit(2 - matches.length)) - ) - if (matches.length > 1) break - } - if (!matches.length) throw new OrchestrationError('not_found', 'Document not found') - if (matches.length > 1) { - throw new OrchestrationError( - 'validation', - 'Multiple accessible documents use this URL. Use documentId from search.' - ) - } - documentId = matches[0].id - } - const access = await knowledgeContext.access.getForDocuments([documentId], input.signal) - input.signal?.throwIfAborted() - const { document: doc } = await readKnowledgeDocument.execute({ - principal, - input: { knowledgeBaseId, documentId, ...assertions, requireEnabledDocument: true }, - request, - }) - const value: ReadIndexedKnowledgeDocumentResult = { - knowledgeBaseId, - documentId: doc.id, - title: doc.filename, - sourceUrl: doc.sourceUrl, - sourceModifiedAt: doc.sourceModifiedAt?.toISOString() ?? null, - connectorType: doc.connectorType, - processingStatus: doc.processingStatus, - } - if ( - !(await importKnowledgePersistedResponseSecretProvenance({ - registry: input.resultSecretRegistry, - documents: [{ id: doc.id, source: createKnowledgeDocumentSourceValue(doc), value }], - })) - ) { - throw new Error('Knowledge document provenance is unavailable') - } - input.signal?.throwIfAborted() - if (doc.processingStatus !== 'completed') return value - - let offset = input.offset ?? 0 - if (input.aroundChunkIndex !== undefined) { - const [position] = await db - .select({ - matched: - sql`count(*) filter (where ${embedding.chunkIndex} = ${input.aroundChunkIndex})`.mapWith( - Number - ), - preceding: - sql`count(*) filter (where ${embedding.chunkIndex} < ${input.aroundChunkIndex})`.mapWith( - Number - ), - }) - .from(embedding) - .innerJoin(document, eq(document.id, embedding.documentId)) - .where( - and( - ...activeDocumentConditions(knowledgeBaseId, access), - eq(document.id, documentId), - eq(embedding.enabled, true), - lte(embedding.chunkIndex, input.aroundChunkIndex) - ) - ) - if (!position?.matched) throw new OrchestrationError('not_found', 'Document chunk not found') - offset = Math.max(0, position.preceding - Math.min(2, input.limit - 1)) - } - input.signal?.throwIfAborted() - const page = await listKnowledgeChunks.execute({ - principal, - input: { - knowledgeBaseId, - documentId, - ...assertions, - requireEnabledDocument: true, - enabled: 'true', - sortBy: 'chunkIndex', - sortOrder: 'asc', - limit: input.limit, - offset, - }, - request, - }) - const chunks = page.chunks.map(({ id, chunkIndex, content }) => ({ id, chunkIndex, content })) - if ( - !(await importKnowledgePersistedResponseSecretProvenance({ - registry: input.resultSecretRegistry, - chunks: chunks.map((chunk) => ({ ...chunk, documentId, value: chunk })), - })) - ) { - throw new Error('Knowledge chunk provenance is unavailable') - } - input.signal?.throwIfAborted() - return { ...value, chunks, pagination: page.pagination } - }, -}) diff --git a/apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts b/apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts deleted file mode 100644 index ce6a0052598..00000000000 --- a/apps/sim/lib/sim-search/indexed/documents/read-search-document.test.ts +++ /dev/null @@ -1,267 +0,0 @@ -import { member } from '@sim/db/schema' -import { queueTableRows, resetDbChainMock } from '@sim/testing' -import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { - knowledgeContextsMock, - knowledgeContextsMockFns, -} from '@sim/testing/mocks/knowledge-contexts.mock' -import { permissionGroupsResolveMock } from '@sim/testing/mocks/permission-groups-resolve.mock' -import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -vi.mock('@/lib/knowledge/search/search-index', () => ({ - findSearchIndex: async () => ({ id: 'index' }), -})) -const mocks = vi.hoisted(() => ({ - chunks: vi.fn(), - provenance: vi.fn(), - importProvenance: vi.fn(), -})) -vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) -vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) -vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) -vi.mock('@/lib/knowledge/chunks/service', () => ({ queryChunks: mocks.chunks })) -vi.mock('@/lib/knowledge/secret-provenance', () => ({ - importKnowledgeSearchResultSecretProvenance: mocks.provenance, -})) -vi.mock('@/lib/execution/durable-secret-provenance', () => ({ - importDurableSecretProvenance: mocks.importProvenance, -})) - -import { readSearchDocument } from '@/lib/sim-search/indexed/documents/read-search-document' -import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -workspaceAuthzMockFns.mockPermissionSatisfies.mockImplementation( - (actual: string | null) => actual !== null -) - -const principal = createSessionPrincipal({ userId: 'reader', sessionId: 'session' }) -const access = { kind: 'user', workspaceId: 'workspace', tokens: ['u:reader'] } as const -const context = { - workspaceId: 'workspace', - workspaceOrganizationId: null, - allowPersonalApiKeys: true, - billedAccountUserId: 'payer', - knowledgeBaseId: 'index', - knowledgeBase: { isSearchIndex: true }, - documentId: 'doc', - document: { - enabled: true, - processingStatus: 'completed', - filename: 'Title', - sourceUrl: 'https://source.test/doc', - }, - access: { get: async () => access }, -} -const input = { - documentId: 'doc', - assertedWorkspaceId: 'workspace', - limit: 3, - filters: { source: 'slack', documentIds: ['doc'] }, - resultSecretRegistry: new ResolvedSecretTraceRegistry([], { - userId: 'reader', - workspaceId: 'workspace', - }), -} - -/** These use cases run only while indexed organization search is on. */ -beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) -afterEach(resetEnvFlagsMock) - -describe('Assistant document read', () => { - beforeEach(() => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockResolvedValue( - context - ) - mocks.chunks.mockResolvedValue({ - chunks: [{ id: 'chunk', chunkIndex: 0, content: 'body' }], - pagination: { total: 1, hasMore: false }, - }) - mocks.provenance.mockResolvedValue({ - imported: true, - documentMetadata: { - doc: { - filename: 'Title', - sourceUrl: 'https://source.test/doc', - provenance: { status: 'exact', entries: [] }, - }, - }, - }) - mocks.importProvenance.mockResolvedValue(true) - }) - it('uses canonical authorization and filters enabled chunks by the same scope', async () => { - await expect(readSearchDocument.execute({ principal, input })).resolves.toMatchObject({ - documentId: 'doc', - chunks: [{ content: 'body', chunkIndex: 0 }], - next: null, - }) - expect( - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext - ).toHaveBeenCalledWith({ ...input, knowledgeBaseId: 'index' }, principal) - expect(mocks.chunks).toHaveBeenCalledWith( - 'doc', - expect.objectContaining({ - documentFilters: input.filters, - enabled: 'true', - limit: 3, - }), - expect.any(String), - access - ) - }) - it('rejects ordinary KBs and disabled documents', async () => { - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockResolvedValueOnce( - { ...context, knowledgeBase: { isSearchIndex: false } } - ) - await expect(readSearchDocument.execute({ principal, input })).rejects.toThrow( - 'Document not found' - ) - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockResolvedValueOnce( - { - ...context, - document: { ...context.document, enabled: false }, - } - ) - await expect(readSearchDocument.execute({ principal, input })).rejects.toThrow( - 'Document not found' - ) - expect(mocks.chunks).not.toHaveBeenCalled() - }) - it('fails closed on absent filtered documents or unverified provenance', async () => { - mocks.chunks.mockResolvedValueOnce({ chunks: [], pagination: { total: 0, hasMore: false } }) - await expect(readSearchDocument.execute({ principal, input })).rejects.toThrow( - 'Document not found' - ) - mocks.provenance.mockResolvedValueOnce({ imported: false }) - await expect(readSearchDocument.execute({ principal, input })).rejects.toThrow('provenance') - }) - it('rechecks the current workspace role', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(null) - await expect(readSearchDocument.execute({ principal, input })).rejects.toThrow( - 'Insufficient workspace' - ) - expect(mocks.chunks).not.toHaveBeenCalled() - }) -}) - -describe('organization Search document reads', () => { - beforeEach(() => { - resetDbChainMock() - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockResolvedValue({ - ...context, - workspaceId: undefined, - organizationId: 'org-1', - knowledgeBase: { isSearchIndex: true, organizationId: 'org-1' }, - }) - mocks.chunks.mockResolvedValue({ - chunks: [{ id: 'chunk', chunkIndex: 0, content: 'body' }], - pagination: { total: 1, hasMore: false }, - }) - mocks.provenance.mockResolvedValue({ - imported: true, - documentMetadata: { - doc: { - filename: 'Title', - sourceUrl: 'https://source.test/doc', - provenance: { status: 'exact', entries: [] }, - }, - }, - }) - mocks.importProvenance.mockResolvedValue(true) - }) - const orgInput = { ...input, assertedWorkspaceId: undefined, assertedOrganizationId: 'org-1' } - it('uses the current org member and existing ACL-filtered document read pipeline', async () => { - queueTableRows(member, [{ role: 'member' }]) - await expect(readSearchDocument.execute({ principal, input: orgInput })).resolves.toMatchObject( - { documentId: 'doc', chunks: [{ content: 'body' }] } - ) - expect(mocks.chunks).toHaveBeenCalledWith( - 'doc', - expect.objectContaining({ enabled: 'true', documentFilters: orgInput.filters }), - expect.any(String), - access - ) - expect(workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission).not.toHaveBeenCalled() - }) - it('refuses a removed member before reading document content or importing provenance', async () => { - queueTableRows(member, []) - await expect(readSearchDocument.execute({ principal, input: orgInput })).rejects.toThrow( - 'Organization not found' - ) - expect(mocks.chunks).not.toHaveBeenCalled() - expect(mocks.provenance).not.toHaveBeenCalled() - }) - it('rejects ambiguous workspace and organization scope before canonical lookup', async () => { - await expect( - readSearchDocument.execute({ - principal, - input: { ...orgInput, assertedWorkspaceId: 'workspace' }, - }) - ).rejects.toThrow('exactly one') - expect( - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext - ).not.toHaveBeenCalled() - expect(mocks.chunks).not.toHaveBeenCalled() - }) -}) - -describe('precise bounded passage expansion', () => { - beforeEach(() => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - knowledgeContextsMockFns.mockResolveCanonicalActiveKnowledgeDocumentContext.mockResolvedValue( - context - ) - mocks.provenance.mockResolvedValue({ imported: true, documentMetadata: {} }) - }) - - it('projects a secret spanning the page boundary before slicing it', async () => { - const secret = 'private-token-that-crosses-the-boundary' - const registry = new ResolvedSecretTraceRegistry([ - { name: 'TOKEN', plaintext: secret, encryptedValue: 'ciphertext' }, - ]) - mocks.provenance.mockImplementationOnce(async () => { - registry.recordResolved('TOKEN', secret) - return { imported: true, documentMetadata: {} } - }) - mocks.chunks.mockResolvedValue({ - chunks: [{ id: 'secret-chunk', chunkIndex: 0, content: `${'x'.repeat(7990) + secret}tail` }], - pagination: { total: 1, hasMore: false }, - }) - const result = await readSearchDocument.execute({ - principal, - input: { ...input, resultSecretRegistry: registry }, - }) - expect(result.chunks[0].content).toContain('{{TOKEN}}') - expect(JSON.stringify(result)).not.toContain('private-token') - }) - - it('rejects a continuation when all remaining chunks have disappeared', async () => { - mocks.chunks.mockResolvedValue({ - chunks: [], - pagination: { total: 2, hasMore: false }, - }) - await expect( - readSearchDocument.execute({ principal, input: { ...input, startChunkIndex: 7 } }) - ).rejects.toThrow('no longer available') - expect(mocks.provenance).not.toHaveBeenCalled() - }) - - it('rejects positions without an anchor and stale within-chunk continuation', async () => { - await expect( - readSearchDocument.execute({ principal, input: { ...input, startOffset: 2 } }) - ).rejects.toThrow('startOffset requires startChunkIndex') - expect(mocks.chunks).not.toHaveBeenCalled() - mocks.chunks.mockResolvedValue({ - chunks: [{ id: 'c8', chunkIndex: 8, content: 'replacement' }], - pagination: { total: 1, hasMore: false }, - }) - await expect( - readSearchDocument.execute({ - principal, - input: { ...input, startChunkIndex: 7, startOffset: 100 }, - }) - ).rejects.toThrow('no longer available') - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/documents/read-search-document.ts b/apps/sim/lib/sim-search/indexed/documents/read-search-document.ts deleted file mode 100644 index 309754a0482..00000000000 --- a/apps/sim/lib/sim-search/indexed/documents/read-search-document.ts +++ /dev/null @@ -1,173 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import type { ReadSearchDocumentResult } from '@/lib/api/contracts/knowledge/documents' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { generateRequestId } from '@/lib/core/utils/request' -import { importDurableSecretProvenance } from '@/lib/execution/durable-secret-provenance' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { KnowledgeDocumentNotReadyError } from '@/lib/knowledge/application/chunk-errors' -import { resolveCanonicalActiveKnowledgeDocumentContext } from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { queryChunks } from '@/lib/knowledge/chunks/service' -import { measureSearchStage } from '@/lib/knowledge/search/diagnostics' -import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' -import { findSearchIndex } from '@/lib/knowledge/search/search-index' -import { passageWindow } from '@/lib/knowledge/search/snippet' -import { importKnowledgeSearchResultSecretProvenance } from '@/lib/knowledge/secret-provenance' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-secret-content-projection' -import type { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' - -export interface ReadSearchDocumentInput { - documentId: string - assertedWorkspaceId?: string - assertedOrganizationId?: string - filters?: WorkspaceSearchFilters - limit: number - startChunkIndex?: number - startOffset?: number - resultSecretRegistry: ResolvedSecretTraceRegistry - signal?: AbortSignal -} - -/** Bounds model text to at most 24KB of UTF-8, with continuation even inside a large chunk. */ -const READ_PAGE_CHARACTERS = 8000 -const READ_PAGE_CHUNKS = 8 -const STALE_POSITION_MESSAGE = - 'The passage position is no longer available; search again or read from the chunk start' - -/** Reads enabled indexed passages with the same document scope and ACLs as search. */ -export const readSearchDocument = defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.readDocument, - resolveContext: async ({ - principal, - input, - }: { - principal: Principal - input: ReadSearchDocumentInput - }) => { - assertIndexedOrgSearchEnabled() - if (Boolean(input.assertedWorkspaceId) === Boolean(input.assertedOrganizationId)) - throw new OrchestrationError('validation', 'Document reads require exactly one search owner') - const index = await findSearchIndex( - input.assertedOrganizationId - ? { kind: 'organization', organizationId: input.assertedOrganizationId } - : { kind: 'workspace', workspaceId: input.assertedWorkspaceId! } - ) - if (!index) throw new OrchestrationError('not_found', 'Document not found') - return resolveCanonicalActiveKnowledgeDocumentContext( - { ...input, knowledgeBaseId: index.id }, - principal - ) - }, - async execute({ input, context }): Promise { - input.signal?.throwIfAborted() - if ( - !Number.isInteger(input.limit) || - input.limit < 1 || - input.limit > READ_PAGE_CHUNKS || - (input.startChunkIndex !== undefined && - (!Number.isSafeInteger(input.startChunkIndex) || - input.startChunkIndex < 0 || - input.startChunkIndex > 2147483647)) || - (input.startOffset !== undefined && - (!Number.isSafeInteger(input.startOffset) || - input.startOffset < 0 || - input.startOffset > 2147483647 || - input.startChunkIndex === undefined)) - ) { - throw new OrchestrationError( - 'validation', - 'Document reads require limit 1–8 and nonnegative chunk positions; startOffset requires startChunkIndex' - ) - } - if (context.document.processingStatus !== 'completed') { - throw new KnowledgeDocumentNotReadyError(context.document.processingStatus) - } - if (!context.knowledgeBase.isSearchIndex) - throw new OrchestrationError('not_found', 'Document not found') - if (!context.document.enabled) throw new OrchestrationError('not_found', 'Document not found') - const access = await measureSearchStage('access_scope', () => context.access.get()) - const page = await measureSearchStage('document_read.sql', () => - queryChunks( - context.documentId, - { - limit: input.limit, - startChunkIndex: input.startChunkIndex, - requireEnabledDocument: true, - enabled: 'true', - sortBy: 'chunkIndex', - sortOrder: 'asc', - documentFilters: input.filters, - }, - generateRequestId(), - access - ) - ) - if (page.pagination.total === 0) throw new OrchestrationError('not_found', 'Document not found') - if (input.startChunkIndex !== undefined && page.chunks.length === 0) { - throw new OrchestrationError('validation', STALE_POSITION_MESSAGE) - } - const provenance = await measureSearchStage('result_provenance', () => - importKnowledgeSearchResultSecretProvenance({ - registry: input.resultSecretRegistry, - results: page.chunks.map((chunk) => ({ ...chunk, documentId: context.documentId })), - }) - ) - if (!provenance.imported) throw new Error('Knowledge result provenance is unavailable') - const metadata = provenance.documentMetadata[context.documentId] - if ( - metadata && - !(await importDurableSecretProvenance(input.resultSecretRegistry, metadata.provenance, { - documentName: metadata.filename, - sourceUrl: metadata.sourceUrl, - })) - ) { - throw new Error('Knowledge document provenance is unavailable') - } - /** Project complete strings before slicing, so a window cannot expose part of a secret. */ - const projectedChunks = page.chunks.map((chunk) => { - const projected = projectResolvedSecretModelContent(chunk.content, input.resultSecretRegistry) - if (!projected.safe || typeof projected.value !== 'string') - throw new Error('Knowledge result provenance is unavailable') - return { ...chunk, content: projected.value } - }) - if ( - input.startOffset && - (projectedChunks[0]?.chunkIndex !== input.startChunkIndex || - input.startOffset >= projectedChunks[0].content.length) - ) { - throw new OrchestrationError('validation', STALE_POSITION_MESSAGE) - } - let remaining = READ_PAGE_CHARACTERS - const chunks: ReadSearchDocumentResult['chunks'] = [] - let next: ReadSearchDocumentResult['next'] = null - for (const chunk of projectedChunks) { - if (remaining < 2) { - next = { startChunkIndex: chunk.chunkIndex, startOffset: 0 } - break - } - const start = chunk.chunkIndex === input.startChunkIndex ? (input.startOffset ?? 0) : 0 - const excerpt = passageWindow(chunk.content, start, remaining) - chunks.push({ chunkIndex: chunk.chunkIndex, ...excerpt }) - remaining -= excerpt.content.length - if (excerpt.endOffset < chunk.content.length) { - next = { startChunkIndex: chunk.chunkIndex, startOffset: excerpt.endOffset } - break - } - } - const last = chunks.at(-1) - if (!next && page.pagination.hasMore && last) { - next = { startChunkIndex: last.chunkIndex + 1, startOffset: 0 } - } - input.signal?.throwIfAborted() - return { - documentId: context.documentId, - knowledgeBaseId: context.knowledgeBaseId, - documentName: metadata?.filename ?? null, - sourceUrl: metadata?.sourceUrl ?? null, - chunks, - hasMore: next !== null, - next, - } - }, -}) diff --git a/apps/sim/lib/sim-search/indexed/gate.ts b/apps/sim/lib/sim-search/indexed/gate.ts deleted file mode 100644 index 9b5771ad870..00000000000 --- a/apps/sim/lib/sim-search/indexed/gate.ts +++ /dev/null @@ -1,42 +0,0 @@ -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' - -/** - * The single switch for indexed organization search: retrieval over `is_search_index` knowledge - * bases and the crawling that fills them. It is the inverse of the Live Search backend selector - * (`SIM_SEARCH_LIVE`, on by default), so indexed search is dormant unless a deployment sets - * `SIM_SEARCH_LIVE=false`. The selector is read once at startup, so the answer is constant for the - * life of the process. - */ -export function isIndexedOrgSearchEnabled(): boolean { - return !isLiveEnterpriseSearchEnabled -} - -/** An indexed-only surface was reached while indexed organization search is dormant. */ -export class SearchIndexDormantError extends Error { - constructor() { - super('This search index is inactive; use Sim Search.') - this.name = 'SearchIndexDormantError' - } -} - -/** - * Refuses entry to dormant indexed organization search. Every indexed use case and entry calls it - * itself, so dormancy holds even for a caller that forgot to ask the gate. - */ -export function assertIndexedOrgSearchEnabled(): void { - if (!isIndexedOrgSearchEnabled()) throw new SearchIndexDormantError() -} - -/** - * Whether a search runs the search-index retrieval legs: indexed organization search is on and - * every knowledge base it names is a search index. Every other search decides readability on - * each candidate's document. - */ -export function usesIndexedRetrieval( - knowledgeBases: ReadonlyArray<{ isSearchIndex?: boolean | null }> -): boolean { - return ( - isIndexedOrgSearchEnabled() && - knowledgeBases.every((knowledgeBase) => knowledgeBase.isSearchIndex === true) - ) -} diff --git a/apps/sim/lib/sim-search/indexed/index.ts b/apps/sim/lib/sim-search/indexed/index.ts deleted file mode 100644 index 65d2f4e08a9..00000000000 --- a/apps/sim/lib/sim-search/indexed/index.ts +++ /dev/null @@ -1,18 +0,0 @@ -/** - * Dormant indexed organization search: the use cases that search and read `is_search_index` - * knowledge bases, and the indexed arms of the personal Search integration inventory. Callers - * check `isIndexedOrgSearchEnabled()` before reaching these, and each refuses on its own while the - * gate is off; see this directory's README. - */ - -export { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' -export { readSearchDocument } from '@/lib/sim-search/indexed/documents/read-search-document' -export { ownsIndexedPersonalSearchAccount } from '@/lib/sim-search/indexed/integrations/personal-account-ownership' -export { loadIndexedSearchIntegrationInventory } from '@/lib/sim-search/indexed/integrations/personal-inventory' -export { listIndexedPersonalSearchIntegrations } from '@/lib/sim-search/indexed/integrations/personal-search-integrations' -export { registerIndexedKnowledgeMcpTools } from '@/lib/sim-search/indexed/mcp/register-tools' -export { - searchOrganizationKnowledge, - searchScopedKnowledge, - searchWorkspaceKnowledge, -} from '@/lib/sim-search/indexed/search/scoped-search' diff --git a/apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts b/apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts deleted file mode 100644 index 561e7cbb039..00000000000 --- a/apps/sim/lib/sim-search/indexed/integrations/personal-account-ownership.ts +++ /dev/null @@ -1,29 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import { personalSearchIntegrationPages } from '@/lib/knowledge/application/personal-search-integration-pages' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' - -/** - * The indexed arm of organization personal-token ownership: whether the viewer's personal Search - * inventory lists `credentialId` as a connected account on this connector type. Indexed search - * answers ownership from its per-source inventory, where Live Search reads the live accounts. - */ -export async function ownsIndexedPersonalSearchAccount( - principal: Principal, - input: { organizationId: string; connectorType: string; credentialId: string } -): Promise { - assertIndexedOrgSearchEnabled() - for await (const page of personalSearchIntegrationPages({ - principal, - input: { organizationId: input.organizationId, connectorType: input.connectorType }, - })) { - if ( - page.connections.some((connection) => - connection.accounts.some( - (account) => account.credentialId === input.credentialId && account.status === 'connected' - ) - ) - ) - return true - } - return false -} diff --git a/apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts b/apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts deleted file mode 100644 index 6b7309b7055..00000000000 --- a/apps/sim/lib/sim-search/indexed/integrations/personal-inventory.ts +++ /dev/null @@ -1,43 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import { - type PersonalSearchIntegrationsPage, - personalSearchIntegrationPages, -} from '@/lib/knowledge/application/personal-search-integration-pages' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' - -/** - * The indexed arm of Sim's Search inventory for one chat turn: every page of the viewer's - * personal connections on the indexed sources, merged and serialized for the prompt. Indexed - * inventory pages by source, where Live Search returns a single page. - */ -export async function loadIndexedSearchIntegrationInventory({ - principal, - organizationId, - signal, - maxBytes, -}: { - principal: Principal - organizationId: string - signal?: AbortSignal - maxBytes: number -}): Promise { - assertIndexedOrgSearchEnabled() - const connections: Array = [] - const available = new Map() - let inventory = JSON.stringify({ connections, available: [] }) - for await (const page of personalSearchIntegrationPages({ - principal, - input: { organizationId }, - signal, - })) { - connections.push(...page.connections) - for (const entry of page.available) { - available.set(JSON.stringify(entry.target), entry) - } - inventory = JSON.stringify({ connections, available: [...available.values()] }) - if (Buffer.byteLength(inventory) > maxBytes) { - throw new Error('Search integration inventory exceeds the prompt size limit') - } - } - return inventory -} diff --git a/apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts b/apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts deleted file mode 100644 index 8762e38f8e9..00000000000 --- a/apps/sim/lib/sim-search/indexed/integrations/personal-search-integrations.ts +++ /dev/null @@ -1,164 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import { readSearchConnectionCompletion } from '@/lib/credential-groups/search-connection-completion' -import { - getIntegrationAvailability, - isOAuthServiceDeploymentAvailable, -} from '@/lib/integrations/availability.server' -import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' -import type { KnowledgeOrganizationContext } from '@/lib/knowledge/application/contexts' -import type { ListPersonalSearchIntegrationsInput } from '@/lib/knowledge/application/personal-search-integrations' -import { listConfiguredSearchProviderTypes } from '@/lib/knowledge/application/search-source-overview' -import { listSearchSources } from '@/lib/knowledge/application/search-sources' -import type { SearchConnectionTarget } from '@/lib/knowledge/search/connection-target' -import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' -import { getConnectorAccessAvailability, SEARCH_CONNECTORS } from '@/lib/sim-search/connectors' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import { findSharedSlackSearchInstallation } from '@/lib/slack-search/shared-app' - -/** - * The indexed arm of `listPersonalSearchIntegrations`: the viewer's accounts on the connectors - * that crawl the organization search index, with each source's indexing state. The caller has - * authorized the principal and loaded the viewer; it reaches this only while - * `isIndexedOrgSearchEnabled()` is on. - */ -export async function listIndexedPersonalSearchIntegrations({ - principal, - input, - context, - userId, - viewer, -}: { - principal: Principal - input: ListPersonalSearchIntegrationsInput - context: KnowledgeOrganizationContext - userId: string - viewer: { emailVerified: boolean } -}) { - assertIndexedOrgSearchEnabled() - const [page, configuredTypes, approvals, access, sharedSlack] = await Promise.all([ - listSearchSources.execute({ principal, input }), - listConfiguredSearchProviderTypes({ organizationId: context.organizationId }), - listOrganizationSearchApprovals(context.organizationId), - resolveKnowledgeAccessAvailability(context), - findSharedSlackSearchInstallation(context.organizationId), - ]) - const deployment = new Map( - getIntegrationAvailability().map((entry) => [entry.type.toLowerCase(), entry]) - ) - const oauth = new Map( - SEARCH_CONNECTORS.map((entry) => [ - entry.providerId, - isOAuthServiceDeploymentAvailable(entry.providerId), - ]) - ) - const configured = new Set(configuredTypes) - const eligible = (connectorType: string) => { - const connector = SEARCH_CONNECTORS.find((entry) => entry.type === connectorType) - return Boolean( - viewer.emailVerified && - connector && - approvals.get(connectorType) && - getConnectorAccessAvailability(connector.meta, deployment, { - memberAccessAvailable: access.memberScoped, - mirroredAccessAvailable: access.sourceMirrored, - oauthServiceAvailability: oauth, - isIntegrationAvailabilityReady: true, - }).members - ) - } - const projected = page.sources.flatMap((source) => { - const connector = SEARCH_CONNECTORS.find((entry) => entry.type === source.connectorType) - if (!connector) return [] - const target: SearchConnectionTarget = { - type: 'link', - provider: connector.providerId, - connectorType: source.connectorType, - connectorId: source.connectorId, - } - const canConnect = - eligible(source.connectorType) && - source.enabled && - source.availability === 'available' && - source.viewerEmailVerified && - source.connectionRequired && - source.viewerMembership !== null && - !['revoked', 'unverified_email'].includes(source.viewerMembership) - const accounts = source.viewerAccounts.map((account) => { - if (!account.status) throw new Error('Personal Search account status is missing') - return { - credentialId: account.credentialId, - displayName: account.displayName, - status: - account.status === 'active' ? ('connected' as const) : ('reconnect_needed' as const), - action: - canConnect && account.status === 'needs_reauth' - ? { ...target, credentialId: account.credentialId } - : null, - } - }) - return [ - { - name: connector.meta.name, - providerId: connector.providerId, - connectorType: connector.type, - connectorId: source.connectorId, - knowledgeBaseId: source.knowledgeBaseId, - description: source.sourceDescription, - accounts, - connectionStatus: accounts.some((account) => account.status === 'reconnect_needed') - ? ('reconnect_needed' as const) - : accounts.length - ? ('connected' as const) - : canConnect - ? ('not_connected' as const) - : ('unavailable' as const), - indexingStatus: - !source.enabled || source.availability !== 'available' || source.approved === false - ? ('paused' as const) - : source.isSyncing - ? ('indexing' as const) - : source.hasSyncError || source.viewerFailedDocumentCount > 0 - ? ('sync_failed' as const) - : source.hasViewerDocuments - ? ('indexed' as const) - : ('not_indexed' as const), - action: canConnect && !accounts.length ? target : null, - }, - ] - }) - const available: Array<{ name: string; description: string; target: SearchConnectionTarget }> = [ - ...projected.flatMap((entry) => - entry.action - ? [{ name: entry.name, description: entry.description, target: entry.action }] - : [] - ), - ...SEARCH_CONNECTORS.filter( - (connector) => - !input.connectorId && - (!input.connectorType || connector.type === input.connectorType) && - (connector.type !== 'slack' || sharedSlack !== null) && - (!configured.has(connector.type) || connector.setupFields.length > 0) && - eligible(connector.type) - ).map((connector) => ({ - name: connector.meta.name, - description: '', - target: { - type: 'link' as const, - provider: connector.providerId, - connectorType: connector.type, - }, - })), - ] - return { - completedCredentialId: input.completionId - ? await readSearchConnectionCompletion({ - organizationId: context.organizationId, - userId, - completionId: input.completionId, - }) - : null, - connections: projected.filter((entry) => entry.accounts.length > 0), - available, - nextCursor: page.nextCursor, - } -} diff --git a/apps/sim/lib/sim-search/indexed/mcp/register-tools.ts b/apps/sim/lib/sim-search/indexed/mcp/register-tools.ts deleted file mode 100644 index 174defef89b..00000000000 --- a/apps/sim/lib/sim-search/indexed/mcp/register-tools.ts +++ /dev/null @@ -1,148 +0,0 @@ -import type { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js' -import type { Principal } from '@sim/auth/principal' -import type { NextRequest } from 'next/server' -import { readDocumentMcpSchema, searchMcpSchema } from '@/lib/api/contracts/knowledge/mcp' -import type { ResourceScope } from '@/lib/core/resource-scope' -import { getBaseUrl } from '@/lib/core/utils/urls' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { searchKnowledge } from '@/lib/knowledge/application/search' -import { - KNOWLEDGE_MCP_READ_ONLY, - type KnowledgeMcpToolRunner, - projectResult, -} from '@/lib/knowledge/mcp/tool-runner' -import { createKnowledgeDocumentCitation } from '@/lib/knowledge/search/citation' -import { toolError } from '@/lib/mcp/tool-result' -import { readIndexedKnowledgeDocument } from '@/lib/sim-search/indexed/documents/read-indexed-document' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' - -interface IndexedKnowledgeMcpToolsContext { - server: McpServer - principal: Principal - request: NextRequest - organizationId: string - /** The organization's search index, or null when no source is connected yet. */ - searchIndexId: string | null - execute: KnowledgeMcpToolRunner -} - -/** - * Registers the indexed `search` and `read_document` Search MCP tools: passages ranked from the - * organization's search index, and indexed documents read by id or source URL. Called only while - * indexed organization search is on, and refuses otherwise; both tools run through use cases that - * refuse a dormant search index on their own. - */ -export function registerIndexedKnowledgeMcpTools(context: IndexedKnowledgeMcpToolsContext): void { - assertIndexedOrgSearchEnabled() - const { server, principal, request, organizationId, searchIndexId, execute } = context - const scope: ResourceScope = { kind: 'organization', organizationId } - - server.registerTool( - 'search', - { - title: 'Search', - description: - 'Search accessible passages in this organization’s Search index. Use source (for example, jira), modifiedAfter (an ISO timestamp), or documentIds to narrow results. Results are candidates; score is similarity, not answer confidence. Use read_document for context and cite citationUrl.', - inputSchema: searchMcpSchema, - annotations: KNOWLEDGE_MCP_READ_ONLY, - }, - async (input: unknown, extra: { signal: AbortSignal }) => - execute('search', knowledgeOperations.search, extra.signal, async (registry, signal) => { - const { query, topK, ...filters } = searchMcpSchema.parse(input) - if (!searchIndexId) { - return projectResult( - { - results: [], - message: 'No Search index is configured. Ask an admin to connect a source.', - }, - registry - ) - } - const result = await searchKnowledge.execute({ - principal, - input: { - organizationId, - knowledgeBaseIds: [searchIndexId], - query, - topK, - filters, - resultSecretRegistry: registry, - surface: 'mcp', - signal, - }, - request, - }) - return projectResult( - { - results: result.results.map((row) => ({ - documentId: row.documentId, - title: row.documentName, - sourceUrl: row.sourceUrl, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: row.knowledgeBaseId, - documentId: row.documentId, - sourceUrl: row.sourceUrl, - baseUrl: getBaseUrl(), - }), - sourceModifiedAt: row.sourceModifiedAt?.toISOString() ?? null, - connectorType: row.connectorType, - content: row.content, - chunkIndex: row.chunkIndex, - score: row.similarity, - })), - }, - result.resultSecretRegistry ?? registry - ) - }) - ) - server.registerTool( - 'read_document', - { - title: 'Read document', - description: - 'Read an indexed document by documentId from search or its original URL. URLs must match an accessible indexed source; this tool does not browse the web. Set aroundChunkIndex to a search hit’s chunkIndex for nearby context, or use offset for sequential pages. When pagination.hasMore is true, continue with pagination.offset + pagination.limit. Cite citationUrl. Documents still indexing return metadata only.', - inputSchema: readDocumentMcpSchema, - annotations: KNOWLEDGE_MCP_READ_ONLY, - }, - async (raw: unknown, extra: { signal: AbortSignal }) => - execute( - 'read_document', - knowledgeOperations.readDocument, - extra.signal, - async (registry, signal) => { - const input = readDocumentMcpSchema.parse(raw) - if (!input.url && !input.documentId) return toolError('Document not found') - const result = await readIndexedKnowledgeDocument.execute({ - principal, - input: { - organizationId, - target: input.url - ? { kind: 'url', url: input.url } - : { kind: 'id', documentId: input.documentId! }, - limit: input.limit, - offset: input.offset, - aroundChunkIndex: input.aroundChunkIndex, - resultSecretRegistry: registry, - signal, - }, - request, - }) - const { knowledgeBaseId, ...document } = result - return projectResult( - { - ...document, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId, - documentId: result.documentId, - sourceUrl: result.sourceUrl, - baseUrl: getBaseUrl(), - }), - }, - registry - ) - } - ) - ) -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts b/apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts deleted file mode 100644 index eba0bc5aa4f..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/access-plan.ts +++ /dev/null @@ -1,188 +0,0 @@ -import { db } from '@sim/db' -import { knowledgeConnector, knowledgeConnectorMember } from '@sim/db/schema' -import { and, eq, inArray, isNull, sql } from 'drizzle-orm' -import { SOURCE_ACL_MAX_AGE_MS } from '@/lib/knowledge/access/freshness' -import { textArrayLiteral } from '@/lib/knowledge/access/predicate' -import type { KnowledgeAccessScope } from '@/lib/knowledge/access/types' -import { searchIntegrationAccessCondition } from '@/lib/knowledge/search/integration-policy' - -/** - * The connectors a search may read from, resolved once per query: their ids grouped by the shape - * their documents' ACLs take, and separately those whose reader access is proven live per request. - */ -export interface KnowledgeConnectorEligibility { - /** Documents carry the workspace ACL. */ - workspace: readonly string[] - /** Documents carry mirrored source permissions verified as a whole. */ - admin: readonly string[] - /** Documents carry the subject tokens of the members who observe them. */ - members: readonly string[] - /** Of the above, those that additionally require this request's live source proof. */ - liveProofRequired: readonly string[] -} - -/** One of the caller's member identities and the connector it belongs to. */ -export interface KnowledgeMemberObserver { - id: string - connectorId: string -} - -/** - * The caller's active member identities on the connectors a search reads, by what makes their - * observations current: `confirmed` members drained their change feed inside the freshness window, - * so every observation they hold stands; `observed` members are trusted only where the observation - * itself is recent. - */ -export interface KnowledgeMemberObservers { - confirmed: readonly KnowledgeMemberObserver[] - observed: readonly KnowledgeMemberObserver[] -} - -/** What a search resolves once about its sources and the caller's standing in them. */ -export interface SearchAccessPlan { - connectors: KnowledgeConnectorEligibility - observers: KnowledgeMemberObservers - /** Connectors the caller is an active member of, whose documents they read broadly. */ - memberSources: readonly string[] - /** Each eligible connector's type, so a search may be confined to one kind of source. */ - connectorTypes: ReadonlyMap - /** Whether documents without a source — uploads — are in scope. */ - uploads: boolean -} - -/** - * The plan confined to one kind of source: the connectors of that type keep their eligibility and - * the rest lose it, so every predicate built from the plan — on the row and on the document — and - * every source the legs walk or rank are that kind alone. `upload` keeps only source-less documents. - */ -export function restrictSearchAccessPlan(plan: SearchAccessPlan, source: string): SearchAccessPlan { - const keep = (id: string) => source !== 'upload' && plan.connectorTypes.get(id) === source - const kept = (ids: readonly string[]) => ids.filter(keep) - return { - connectors: { - workspace: kept(plan.connectors.workspace), - admin: kept(plan.connectors.admin), - members: kept(plan.connectors.members), - liveProofRequired: kept(plan.connectors.liveProofRequired), - }, - observers: { - confirmed: plan.observers.confirmed.filter((observer) => keep(observer.connectorId)), - observed: plan.observers.observed.filter((observer) => keep(observer.connectorId)), - }, - memberSources: kept(plan.memberSources), - connectorTypes: plan.connectorTypes, - uploads: source === 'upload', - } -} - -/** - * The connectors a search may read from, grouped by access mode, with the ones whose reader access - * must be proven live marked. - * - * Deletion, archival, a pending access rewrite and the organization's integration approval are - * facts about a connector. Resolving them once per query — there are tens of connectors against - * hundreds of thousands of documents — leaves each candidate its own columns to check. - */ -async function resolveConnectorEligibility( - knowledgeBaseIds: readonly string[] -): Promise<{ eligibility: KnowledgeConnectorEligibility; types: Map }> { - const eligibility: { - workspace: string[] - admin: string[] - members: string[] - liveProofRequired: string[] - } = { workspace: [], admin: [], members: [], liveProofRequired: [] } - const types = new Map() - if (knowledgeBaseIds.length === 0) return { eligibility, types } - const rows = await db - .select({ - id: knowledgeConnector.id, - accessMode: knowledgeConnector.accessMode, - connectorType: knowledgeConnector.connectorType, - /** A GitHub connector is gated only where it names the immutable repository behind a grant. */ - githubRepository: sql`${knowledgeConnector.sourceConfig}::jsonb ? 'githubRepositoryId'`, - }) - .from(knowledgeConnector) - .where( - and( - inArray(knowledgeConnector.knowledgeBaseId, [...knowledgeBaseIds]), - isNull(knowledgeConnector.deletedAt), - isNull(knowledgeConnector.archivedAt), - eq(knowledgeConnector.accessRewritePending, false), - searchIntegrationAccessCondition() - ) - ) - for (const row of rows) { - if (row.accessMode === 'workspace') eligibility.workspace.push(row.id) - else if (row.accessMode === 'admin') eligibility.admin.push(row.id) - else if (row.accessMode === 'members') eligibility.members.push(row.id) - else continue - types.set(row.id, row.connectorType) - const live = - (row.connectorType === 'github' && row.githubRepository) || - (row.connectorType === 'confluence' && row.accessMode === 'admin') - if (live) eligibility.liveProofRequired.push(row.id) - } - return { eligibility, types } -} - -/** - * The caller's own member identities on these connectors, split by whether the member's change - * feed is itself current. - * - * A members-mode document is readable while one of the caller's active members observes it, - * freshly — and which members those are is a fact about the caller, not about any document. A - * member whose feed drained recently confirms every observation it holds, so its observations need - * no age check at all; the rest are checked against the age of the observation itself. Resolved - * once, the per-document check becomes one lookup on the observation key, with no join to the - * member behind it. - */ -async function resolveMemberObservers( - access: KnowledgeAccessScope, - connectorIds: readonly string[] -): Promise<{ observers: KnowledgeMemberObservers; memberSources: string[] }> { - if (access.kind !== 'user' || connectorIds.length === 0 || access.tokens.length === 0) { - return { observers: { confirmed: [], observed: [] }, memberSources: [] } - } - const rows = await db - .select({ - id: knowledgeConnectorMember.id, - connectorId: knowledgeConnectorMember.connectorId, - syncedThrough: knowledgeConnectorMember.memberSyncedThrough, - }) - .from(knowledgeConnectorMember) - .where( - and( - inArray(knowledgeConnectorMember.connectorId, [...connectorIds]), - eq(knowledgeConnectorMember.status, 'active'), - sql`${knowledgeConnectorMember.subjectToken} = ANY(${textArrayLiteral([...access.tokens])})` - ) - ) - const cutoff = Date.now() - SOURCE_ACL_MAX_AGE_MS - const confirmed: KnowledgeMemberObserver[] = [] - const observed: KnowledgeMemberObserver[] = [] - const memberSources = new Set() - for (const row of rows) { - const member = { id: row.id, connectorId: row.connectorId } - if (row.syncedThrough !== null && row.syncedThrough.getTime() > cutoff) confirmed.push(member) - else observed.push(member) - memberSources.add(row.connectorId) - } - return { observers: { confirmed, observed }, memberSources: [...memberSources] } -} - -/** - * Everything a search needs to know about its sources and the caller's standing in them, resolved - * once: which connectors it may read, the caller's member identities there, and the sources they - * are a member of. Each is a fact about a connector or a caller, so deriving them per candidate - * document is what made retrieval cost grow with the size of what someone may read. - */ -export async function resolveSearchAccessPlan( - knowledgeBaseIds: readonly string[], - access: KnowledgeAccessScope -): Promise { - const { eligibility: connectors, types: connectorTypes } = - await resolveConnectorEligibility(knowledgeBaseIds) - const { observers, memberSources } = await resolveMemberObservers(access, connectors.members) - return { connectors, observers, memberSources, connectorTypes, uploads: true } -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/index.ts b/apps/sim/lib/sim-search/indexed/retrieval/index.ts deleted file mode 100644 index 79691075482..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/index.ts +++ /dev/null @@ -1,10 +0,0 @@ -/** - * The search-index retrieval legs shared retrieval (`lib/knowledge/search/queries.ts`) runs for a - * user-scoped search over search indexes while indexed organization search is on. Everything that - * decides readability on the projection row lives behind this barrel: the resolved access plan - * and the projection-row predicates built from it, reach and permitted sets, per-source vector - * walks, the projection-fill probe, live source proof, and keyword ranking over the GIN and Tin - * projections. Kept apart from the use-case barrel because the use cases depend on that retrieval - * layer, which depends on these. - */ -export { prepareIndexedRetrieval } from '@/lib/sim-search/indexed/retrieval/legs' diff --git a/apps/sim/lib/sim-search/indexed/retrieval/keyword.ts b/apps/sim/lib/sim-search/indexed/retrieval/keyword.ts deleted file mode 100644 index c0016e77518..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/keyword.ts +++ /dev/null @@ -1,325 +0,0 @@ -import { document, embedding, embeddingKeywordSearch, embeddingKeywordTin } from '@sim/db/schema' -import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' -import { knowledgeAccessCondition, textArrayLiteral } from '@/lib/knowledge/access/predicate' -import { runSearchQuery } from '@/lib/knowledge/search/budget' -import { - candidateDocumentConditions, - excludeSearchSources, - FTS_CONFIG, - hydrateSearchCandidates, - type KeywordSearchParams, - type SearchReadCandidate, - type SearchReadCandidatePage, - type SearchResult, - selectAuthorizedSearchResults, -} from '@/lib/knowledge/search/candidates' -import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' -import { searchDateFilterCondition } from '@/lib/knowledge/search/filter-conditions' -import { keywordCandidateRankingQuery } from '@/lib/knowledge/search/keyword-ranking' -import { getStructuredTagFilters } from '@/lib/knowledge/search/tag-filters' -import { embeddingDistance } from '@/lib/knowledge/vector-columns' -import { - documentSatisfies, - type IndexedRetrievalContext, - PERMITTED_EXACT_DOCUMENT_LIMIT, -} from '@/lib/sim-search/indexed/retrieval/permitted' -import { - excludeSearchSourcesOnRow, - knowledgeCandidateAccessConditionForConnectors, - projectionCandidateAccessCondition, - projectionDecidedOnDocument, -} from '@/lib/sim-search/indexed/retrieval/projection-access' -import { isProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' -import { resolveTinKeywordQuery } from '@/lib/sim-search/indexed/retrieval/tin-keyword' - -/** - * Chunks Tin ranks before access is checked, widening while too few are readable to fill a page. - * A caller past the permitted-set limit reads a large share of the index, so the first window - * almost always fills; the widest bounds the work before the GIN ranking takes over. - */ -const TIN_KEYWORD_WINDOWS = [2000, 10_000, 50_000] as const - -/** Readable rows one wide window returns for a narrow reader: several pages' worth, ranked once. */ -const NARROW_KEYWORD_PAGE = 1000 - -/** - * The widest window a narrow reader ranks: wide enough that a few percent of it fills their page - * several times over, and less than half the cost of the widest window the broad readers reach. - * It is tried only after the first window came back short: ranking costs grow with the window, - * and a term that is common where the reader can read fills the page from the narrowest one. - */ -const NARROW_KEYWORD_WINDOWS = [TIN_KEYWORD_WINDOWS[0], 20_000] as const - -/** - * The keyword leg of a user-scoped search-index search. - * - * A bounded permitted set confines matching to the chunks the caller may read. A caller reaching - * past the permitted-set limit reads much of the index, so ranking every match before checking - * access is the leg's whole cost for a common term: where the Tin projection is complete, BM25 - * ranks inside the bases first and access is checked, on the row, only on the top of that - * ranking. Otherwise the GIN projection (`embedding_keyword_search`) ranks in three stages: - * match, authorize, rank. - * - * The visibility predicate carries correlated subqueries — one per connector, one per - * search-integration decision — so evaluating it across a base ahead of the query costs a table - * pass priced by how many documents the base holds rather than by how many the query matched. - * Matching first restricts that predicate to the documents the query actually matched. - * - * Two details keep that ordering from paying the saving back. Restricting the predicate with - * `document.id = ANY (...)` rather than a subquery keeps the narrowed lookup on a bitmap scan, - * which prefetches, where a plain `IN (SELECT ...)` plans as an index walk that does not. And - * the match stage carries identifiers only: ranking every match rather than every *visible* - * match would detoast one text-search vector per match, which on a mid-frequency term costs - * more than the pass it replaces. - */ -export async function executeIndexedKeywordSearch( - params: KeywordSearchParams, - context: IndexedRetrievalContext -): Promise { - const { knowledgeBaseIds, topK, query, queryVector, structuredFilters } = params - if (!query.trim()) return [] - const { access, accessPlan, permitted } = context - const tsQuery = sql`websearch_to_tsquery(${FTS_CONFIG}, ${query})` - const tagFilterConditions = structuredFilters?.length - ? getStructuredTagFilters(structuredFilters, embedding) - : [] - const candidateRank = sql`ts_rank_cd(${embeddingKeywordSearch.contentTsv}, ${tsQuery})` - /** - * A bounded set past the exact-ranking size is read on the row like an unbounded one: the - * bounded read materializes every chunk of the set before it matches a term, where a ranking - * decided on the row costs what the term matches. - */ - const largePermittedSet = - permitted?.kind === 'bounded' && permitted.documents.length >= PERMITTED_EXACT_DOCUMENT_LIMIT - const onRowReader = permitted?.kind === 'unbounded' || largePermittedSet - let tinQuery: Awaited> = null - if (onRowReader && tagFilterConditions.length === 0) { - try { - tinQuery = await resolveTinKeywordQuery(query, FTS_CONFIG, params.budget) - } catch (error) { - /** A leg whose deadline passed before it ranked anything is short, not failed. */ - if (!params.budget?.isTimeout(error)) throw error - return [] - } - } - if (onRowReader) annotateSearchDiagnostics({ keywordRanking: tinQuery ? 'tin' : 'gin' }) - /** A filled projection decides readability on the ranked row alone; none of its rows needs the document. */ - const tinFilled = tinQuery - ? await isProjectionFilled('embedding_keyword_tin', 'keyword.projection_filled', params.budget) - : false - /** The ranked CTE's mirrored columns, which the on-row predicates read. */ - const rankedTinRow = { - connectorId: sql`ranked_tin_chunks.connector_id`, - acl: sql`ranked_tin_chunks.acl`, - documentId: sql`ranked_tin_chunks.document_id`, - } - /** The projection predicate over the ranked CTE's mirrored columns, plus any excluded source. */ - const onRowKeywordVisibility = (excludedSources: readonly string[]) => - and( - projectionCandidateAccessCondition(rankedTinRow, access, accessPlan, { - filled: tinFilled, - }), - documentSatisfies( - sql`ranked_tin_chunks.document_id`, - searchDateFilterCondition(params.filters) - ), - excludeSearchSourcesOnRow(rankedTinRow, tinFilled, excludedSources) - ) - const documentConditions = (excludedSources: readonly string[]) => - and( - ...candidateDocumentConditions( - knowledgeBaseIds, - params.filters, - knowledgeCandidateAccessConditionForConnectors(access, accessPlan) - ), - excludeSearchSources(excludedSources) - ) - /** - * A page read on the row takes each candidate's source from the row, which is what decides - * whether its live source proof is asked for. A row decided on its document — not yet filled, - * or its document marked for the projector — takes it from the document: one primary-key read - * per such row of the page, after its limit, never per ranked row. - */ - const onRowPage = (ranked: SQL) => sql` - SELECT paged.id, paged."documentId", - CASE WHEN paged.decided_on_document - THEN (SELECT ${document.connectorId} FROM ${document} WHERE ${document.id} = paged."documentId") - ELSE paged."connectorId" - END AS "connectorId", - paged.keyword_rank - FROM (${ranked}) AS paged` - /** - * One page from the top of Tin's ranking. Readability is decided on the ranked row. The windows - * widen while the page is short, a narrow reader's to a wide one sooner and no further, and what - * the widest cannot fill is left short rather than handed to a ranking over every match. A large - * bounded set is the exception on its first page: its bounded read was exhaustive, so the widest - * window that still falls short hands that page to the GIN ranking, which covers every match. A - * later page stays with Tin: the two rankers order differently, so an offset advanced through - * one cannot resume the other. - */ - const selectTinPage = async ( - scopedQuery: SQL, - limit: number, - offset: number, - excludedSources: readonly string[] - ): Promise => { - const narrow = (permitted?.kind === 'unbounded' && !permitted.broad) || largePermittedSet - const windows: readonly number[] = narrow ? NARROW_KEYWORD_WINDOWS : TIN_KEYWORD_WINDOWS - /** - * A narrow reader's page is the readable remainder of a wide ranking, and that ranking is - * the cost: each page would rank the window again to find the next few readable rows, so one - * statement returns as many as several pages could ask for. - */ - const pageLimit = narrow ? Math.max(limit, NARROW_KEYWORD_PAGE) : limit - for (const window of windows) { - if (window < offset + limit) continue - const [page] = await runSearchQuery(params.budget, 'keyword.tin', (executor) => - executor.execute<{ ranked: number; candidates: SearchReadCandidate[] }>(sql` - WITH ranked_tin_chunks AS MATERIALIZED ( - SELECT ${embeddingKeywordTin.id} AS id, ${embeddingKeywordTin.documentId} AS document_id, - ${embeddingKeywordTin.enabled} AS enabled, ${embeddingKeywordTin.connectorId} AS connector_id, - ${embeddingKeywordTin.acl} AS acl, - tin.full_score(${embeddingKeywordTin}.ctid) AS keyword_rank - FROM ${embeddingKeywordTin} - WHERE ${embeddingKeywordTin.content} ==> (${scopedQuery}) - ORDER BY keyword_rank DESC - LIMIT ${window} - ), page AS ( - ${ - /** - * Readability decided on the ranked row: its source and ACL are mirrored there, so a - * window of mostly unreadable chunks costs an array test per row, not a document - * lookup. The full predicate follows at hydration. - */ - onRowPage( - sql` - SELECT ranked_tin_chunks.id, ranked_tin_chunks.document_id AS "documentId", - ranked_tin_chunks.connector_id AS "connectorId", ranked_tin_chunks.keyword_rank, - ${projectionDecidedOnDocument(rankedTinRow, tinFilled)} AS decided_on_document - FROM ranked_tin_chunks /* on-row visibility */ - WHERE ranked_tin_chunks.enabled AND ${onRowKeywordVisibility(excludedSources)} - ORDER BY ranked_tin_chunks.keyword_rank DESC, ranked_tin_chunks.id - LIMIT ${pageLimit} OFFSET ${offset}` - ) - } - ) - SELECT (SELECT count(*)::int FROM ranked_tin_chunks) AS ranked, - coalesce(( - SELECT json_agg(json_build_object( - 'id', page.id, 'documentId', page."documentId", 'connectorId', page."connectorId" - ) ORDER BY page.keyword_rank DESC, page.id) - FROM page - ), '[]'::json) AS candidates - `) - ) - annotateSearchDiagnostics({ keywordTinWindow: window }) - if ( - page.candidates.length >= limit || - page.ranked < window || - (!(largePermittedSet && offset === 0) && window === windows[windows.length - 1]) - ) { - return { candidates: page.candidates, nextOffset: offset + page.candidates.length } - } - } - return null - } - /** Parenthesized where used: `==>` binds tighter than `||`. */ - const tinScope = tinQuery - ? sql`'(' || ${sql.join( - knowledgeBaseIds.map((id) => sql`knowledge_tin_base_token(${id}) || '^0'`), - sql` || ' OR ' || ` - )} || ') AND (' || ${tinQuery} || ')'` - : undefined - /** - * Tin and GIN order candidates differently, so a search that once handed a page to GIN stays - * with GIN: an offset advanced through one ranking cannot resume the other. - */ - let handedToGin = false - /** Keep readable identities and rank scalars separate so sorts never carry full text-search vectors. */ - return selectAuthorizedSearchResults({ - leg: 'keyword', - access, - liveSourceAccess: context.liveSourceAccess, - signal: params.signal, - budget: params.budget, - topK, - selectPage: async (limit, offset, excludedSources) => { - /** - * A bounded permitted set confines matching to the chunks the caller may read, so a term - * common across the index is ranked only where it can surface. The visibility CTE below - * still re-applies the candidate predicate, so the restriction can only narrow. - */ - const permittedIds = - permitted?.kind === 'bounded' && !largePermittedSet - ? permitted.documents.map((entry) => entry.id) - : undefined - if (permittedIds?.length === 0) return { candidates: [], nextOffset: offset } - if (tinScope && !handedToGin) { - const tinPage = await selectTinPage(tinScope, limit, offset, excludedSources) - if (tinPage) return tinPage - handedToGin = true - annotateSearchDiagnostics({ keywordRanking: 'gin' }) - } - const baseScope = and( - inArray(embeddingKeywordSearch.knowledgeBaseId, knowledgeBaseIds), - eq(embeddingKeywordSearch.enabled, true) - ) - const chunkMatch = and( - sql`${embeddingKeywordSearch.contentTsv} @@ ${tsQuery}`, - tagFilterConditions.length - ? sql`EXISTS ( - SELECT 1 FROM ${embedding} WHERE ${embedding.id} = ${embeddingKeywordSearch.id} - AND ${and(...tagFilterConditions)} - )` - : undefined - ) - /** - * A bounded permitted set is read through its documents alone and matched row by row, at a - * cost linear in the permitted chunks. Offered the text or base indexes alongside, - * PostgreSQL may intersect the permitted chunks with every chunk in the base that holds the - * term or sits in the base; measured on an organization index that plan cost several - * times the direct read, and the direct read is never materially slower. The permitted - * documents were resolved inside these bases; the base check still applies to the rows - * read, so the read can never widen the scope. `OFFSET 0` keeps the read from being - * flattened back into an intersection; the alias lets the shared conditions bind to it. - */ - const matchedChunks = permittedIds - ? sql` - SELECT ${embeddingKeywordSearch.id} AS id, ${embeddingKeywordSearch.documentId} AS document_id - FROM ( - SELECT * FROM ${embeddingKeywordSearch} - WHERE ${embeddingKeywordSearch.documentId} = ANY(${textArrayLiteral(permittedIds)}) - OFFSET 0 - ) AS ${embeddingKeywordSearch} - WHERE ${and(baseScope, chunkMatch)}` - : sql` - SELECT ${embeddingKeywordSearch.id} AS id, ${embeddingKeywordSearch.documentId} AS document_id - FROM ${embeddingKeywordSearch} - WHERE ${and(baseScope, chunkMatch)}` - const candidates = await runSearchQuery(params.budget, 'keyword.sql', (executor) => - executor.execute( - keywordCandidateRankingQuery({ - matchedChunks, - documentConditions: [documentConditions(excludedSources)], - rankTable: embeddingKeywordSearch, - rank: candidateRank, - limit, - offset, - }) - ) - ) - return { candidates, nextOffset: offset + candidates.length } - }, - /** Every candidate already matched the query where it was ranked; matching it again here would detoast one text-search vector per result. */ - hydrate: (ids, authorized) => - hydrateSearchCandidates( - ids, - knowledgeAccessCondition(authorized), - embeddingDistance(queryVector.dimensions, queryVector.vector).as('distance'), - params.filters, - [inArray(embedding.knowledgeBaseId, knowledgeBaseIds), ...tagFilterConditions], - 'keyword', - params.budget - ), - }) -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts deleted file mode 100644 index ed313c45f8c..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/legs.test.ts +++ /dev/null @@ -1,995 +0,0 @@ -import { - dbChainMockFns, - hasMockCondition, - queueTableRows, - resetDbChainMock, - resetEnvFlagsMock, - schemaMock, - setEnvFlags, -} from '@sim/testing' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const { mockResolveTinKeywordQuery } = vi.hoisted(() => ({ - mockResolveTinKeywordQuery: vi.fn<() => Promise>(async () => null), -})) - -vi.mock('@/lib/sim-search/indexed/retrieval/tin-keyword', () => ({ - resolveTinKeywordQuery: mockResolveTinKeywordQuery, -})) - -import type { KnowledgeAccessProvider, UserAccessScope } from '@/lib/knowledge/access/types' -import { SearchBudget } from '@/lib/knowledge/search/budget' -import type { KeywordSearchParams, SearchParams } from '@/lib/knowledge/search/candidates' -import { retrieveKnowledgeSearch } from '@/lib/knowledge/search/queries' -import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' -import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' -import { selectIndexedTagResults } from '@/lib/sim-search/indexed/retrieval/legs' -import { - forgetSearchReach, - type IndexedRetrievalContext, - isSearchFiltered, - PERMITTED_EXACT_DOCUMENT_LIMIT, - type PermittedDocuments, - resolvePermittedDocuments, - resolveReach, -} from '@/lib/sim-search/indexed/retrieval/permitted' -import { forgetProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' -import { forgetIndexedVectorSources } from '@/lib/sim-search/indexed/retrieval/source-vector-indexes' -import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' - -/** - * The projection-fill memo outlives a test; every case starts without one. Shared retrieval runs - * these legs only while indexed organization search is on. - */ -beforeEach(() => { - forgetProjectionFilled() - setEnvFlags({ isLiveEnterpriseSearchEnabled: false }) -}) -afterEach(resetEnvFlagsMock) - -/** A plan that admits every source and resolves no membership: what a test leaves unsaid. */ -const openPlan = (): SearchAccessPlan => ({ - connectors: { workspace: [], admin: [], members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, -}) - -type Resolved = Partial> & { - accessPlan?: Omit & { - connectors: Omit & { - liveProofRequired?: readonly string[] - } - } -} - -/** The caller's resolved state, split from the leg's own parameters. */ -function context( - { accessPlan, permitted, liveSourceAccess }: Resolved, - params: SearchParams -): IndexedRetrievalContext { - return { - access: params.access as UserAccessScope, - filtered: isSearchFiltered(params.filters), - accessPlan: accessPlan - ? { - ...accessPlan, - connectors: { liveProofRequired: [], ...accessPlan.connectors }, - } - : openPlan(), - permitted, - liveSourceAccess, - } -} - -const vectorSearch = ({ - accessPlan, - permitted, - liveSourceAccess, - ...params -}: SearchParams & Resolved) => - selectIndexedVectorResults(params, context({ accessPlan, permitted, liveSourceAccess }, params)) - -const tagSearch = ({ - accessPlan, - permitted, - liveSourceAccess, - ...params -}: SearchParams & Resolved) => - selectIndexedTagResults(params, context({ accessPlan, permitted, liveSourceAccess }, params)) - -const keywordSearch = ({ - accessPlan, - permitted, - liveSourceAccess, - ...params -}: KeywordSearchParams & Resolved) => - executeIndexedKeywordSearch(params, context({ accessPlan, permitted, liveSourceAccess }, params)) - -/** - * The global `drizzle-orm` mock renders `sql` fragments to a `?`-placeholder - * string via `toSQL()`, so we can assert the exact predicate each statement builds. - */ -function render(condition: unknown) { - return (condition as { toSQL: () => { sql: string; params: unknown[] } }).toSQL() -} - -/** The permitted-documents probe: the reach count and the saturation sentinel, never a slice. */ -function isProbeStatement(sql: string) { - return sql.includes('AS saturated') && !sql.includes('readable_chunks') -} - -/** A graph walk decided on the row it visits. */ -function isWalk(sql: string) { - return sql.includes('on-row visibility') && !sql.includes('ranked_tin_chunks') -} - -/** `+ 0` is what keeps the exact ranking off the ANN index, so it also identifies the statement. */ -function isExactRanking(sql: string) { - return sql.includes(') + 0 LIMIT') -} - -/** The page read: a slice of the pool's identities, from the projection and its documents. */ -function isPageStatement(sql: string) { - return ( - sql.includes('AS "connectorId"') && sql.includes('= ANY(') && !sql.includes('ranked_tin_chunks') - ) -} - -const statements = () => dbChainMockFns.execute.mock.calls.map(([query]) => render(query)) - -const bounded = ( - ...documents: Array<{ id: string; connectorId: string | null }> -): PermittedDocuments => ({ kind: 'bounded', documents }) - -describe('search-index legs rank identifiers before verification', () => { - const identity: UserAccessScope = { - kind: 'user', - userId: 'reader', - tokens: ['org', 's:github-repositories:-:42'], - } - const candidate = (id: string, connectorId: string) => ({ - id, - documentId: `doc-${id}`, - connectorId, - distance: 0.1, - }) - const provider: KnowledgeAccessProvider = { - get: async () => identity, - getForConnectors: async () => identity, - getForDocuments: async () => identity, - liveSourceConnectorCondition: async () => null, - } - const params: SearchParams = { - knowledgeBaseIds: ['org-index'], - topK: 1, - access: identity, - queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, - distanceThreshold: 0.8, - structuredFilters: [{ tagSlot: 'tag1', fieldType: 'text', operator: 'eq', value: 'release' }], - } - - const probePages: Array> = [] - const exactPages: Array> = [] - const candidatePages: Array> = [] - const rerankPages: Array>> = [] - const keywordPages: Array>> = [] - function queueRerank(rows: Array>) { - rerankPages.push(rows) - } - function queueCandidates(rows: Array<{ id: string }>, initialCount = rows.length) { - candidatePages.push(rows.map(({ id }) => ({ id, initial_count: initialCount }))) - } - - beforeEach(() => { - resetDbChainMock() - probePages.length = 0 - exactPages.length = 0 - candidatePages.length = 0 - rerankPages.length = 0 - keywordPages.length = 0 - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('AS visible')) return candidatePages.shift() ?? [] - if (isPageStatement(statement)) return rerankPages.shift() ?? [] - if (statement.includes('WITH matched_keyword_chunks')) return keywordPages.shift() ?? [] - if (isExactRanking(statement)) return exactPages.shift() ?? [] - if (isProbeStatement(statement)) return probePages.shift() ?? [] - return [] - }) - }) - - it.each(['vector', 'tag-vector', 'tags', 'keyword'] as const)( - '%s ranks identifiers before verification and loads content under the full predicate', - async (mode) => { - const candidates = [candidate('selected', 'allowed-source')] - if (mode === 'vector' || mode === 'tag-vector') { - exactPages.push([{ id: 'selected' }]) - queueRerank(candidates) - } - if (mode === 'keyword') keywordPages.push(candidates) - if (mode === 'tags') queueTableRows(schemaMock.embedding, candidates) - queueTableRows(schemaMock.embedding, [{ id: 'selected', content: 'verified result' }]) - const permitted = bounded({ id: 'doc-selected', connectorId: 'allowed-source' }) - const rows = - mode === 'vector' - ? await vectorSearch({ ...params, structuredFilters: undefined, permitted }) - : mode === 'tag-vector' - ? await vectorSearch({ ...params, permitted }) - : mode === 'tags' - ? await tagSearch(params) - : await keywordSearch({ - ...params, - query: 'release', - queryVector: params.queryVector!, - }) - expect(rows).toEqual([{ id: 'selected', content: 'verified result' }]) - if (mode === 'keyword') { - const ranking = render(dbChainMockFns.execute.mock.calls[0][0]).sql - expect(ranking).toContain('matched_keyword_chunks AS MATERIALIZED') - expect(ranking).toContain('ORDER BY keyword_rank DESC, matched_keyword_chunks.id') - expect(ranking).not.toContain('<=>') - expect(ranking).not.toContain('"content"') - } else if (mode === 'tags') { - expect(Object.keys(dbChainMockFns.select.mock.calls[0][0]).sort()).toEqual( - ['id', 'documentId', 'connectorId'].sort() - ) - } else { - const ranking = statements().find((query) => isExactRanking(query.sql))! - /** The identities are one nested fragment; the mock renders it into the parameters. */ - expect(JSON.stringify(ranking)).toContain('connectorId') - expect(ranking.sql).not.toContain('"content"') - } - const rankingOrder = - mode === 'tags' - ? dbChainMockFns.select.mock.invocationCallOrder[0] - : dbChainMockFns.execute.mock.invocationCallOrder[0] - expect(rankingOrder).toBeLessThan(dbChainMockFns.select.mock.invocationCallOrder.at(-1)!) - const fullPredicate = dbChainMockFns.where.mock.calls.at(-1)![0] - expect(JSON.stringify(fullPredicate)).toContain('acl') - expect( - hasMockCondition( - fullPredicate, - (node) => - node.type === 'inArray' && - node.column === schemaMock.embedding.id && - Array.isArray(node.values) && - node.values.length === 1 && - node.values[0] === 'selected' - ) - ).toBe(true) - } - ) - - it('matches keyword chunks before the visibility predicate and ranks only what survives it', async () => { - keywordPages.push([candidate('selected', 'allowed-source')]) - queueTableRows(schemaMock.embedding, [{ id: 'selected', content: 'verified result' }]) - await keywordSearch({ ...params, query: 'release', queryVector: params.queryVector! }) - const ranking = render(dbChainMockFns.execute.mock.calls[0][0]).sql - const matched = ranking.indexOf('matched_keyword_chunks AS MATERIALIZED') - const visible = ranking.indexOf('visible_keyword_documents AS MATERIALIZED') - expect(matched).toBeGreaterThanOrEqual(0) - expect(visible).toBeGreaterThan(matched) - expect(ranking.slice(matched, visible)).not.toContain('keyword_rank') - expect(ranking.slice(visible)).toContain('FROM matched_keyword_chunks INNER JOIN') - /** The predicate fragments are parameterized, so the restriction is read off the query tree. */ - const fragments = JSON.stringify(dbChainMockFns.execute.mock.calls[0][0]) - expect(fragments).toContain('= ANY (ARRAY(SELECT document_id FROM matched_keyword_chunks))') - }) -}) - -describe('permitted-document planner', () => { - const reader: UserAccessScope = { - kind: 'user', - userId: 'reader', - tokens: ['u:reader@example.com'], - } - const provider: KnowledgeAccessProvider = { - get: async () => reader, - getForConnectors: async () => reader, - getForDocuments: async () => reader, - liveSourceConnectorCondition: async () => null, - } - const params: SearchParams = { - knowledgeBaseIds: ['org-index'], - topK: 1, - access: reader, - queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, - distanceThreshold: 1, - } - const hit = (id: string, connectorId: string | null) => ({ - id, - documentId: `doc-${id}`, - connectorId, - distance: 0.1, - }) - let probeRows: Array<{ id: string | null; connectorId: string | null; saturated: boolean }> - let exactRows: Array<{ id: string }> - let traversedRows: Array<{ id: string; distance?: number }> - let rerankRows: Array> - let indexedSourceRows: Array<{ name: string; connectorId: string }> - let sourceExactRows: Array<{ id: string; distance: number }> - - beforeEach(() => { - resetDbChainMock() - probeRows = [] - exactRows = [] - traversedRows = [] - rerankRows = [] - sourceExactRows = [] - indexedSourceRows = [] - forgetIndexedVectorSources() - forgetSearchReach() - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('pg_index')) return indexedSourceRows - if (isWalk(statement)) return traversedRows - if (isPageStatement(statement)) return rerankRows - if (statement.includes('WITH readable_chunks')) return sourceExactRows - if (isExactRanking(statement)) return exactRows - if (isProbeStatement(statement)) return probeRows - return [] - }) - }) - - it('ranks a bounded permitted set exactly without walking the graph', async () => { - exactRows = [{ id: 'a' }] - rerankRows = [hit('a', null)] - queueTableRows(schemaMock.embedding, [hit('a', null)]) - const results = await vectorSearch({ - ...params, - permitted: bounded({ id: 'doc-a', connectorId: null }, { id: 'doc-b', connectorId: 'src' }), - }) - expect(results.map((row) => row.id)).toEqual(['a']) - const sqls = statements().map((query) => query.sql) - expect(sqls.some((sql) => sql.includes('hnsw.iterative_scan'))).toBe(false) - expect(sqls.some((sql) => sql.includes('AS visible'))).toBe(false) - expect(sqls.some(isProbeStatement)).toBe(false) - const exact = JSON.stringify(statements().find((query) => isExactRanking(query.sql))) - expect(exact).toContain('doc-a') - expect(exact).toContain('doc-b') - }) - - it('walks the whole graph once for a caller whose reach is broad', async () => { - const eligibility = { workspace: [], admin: ['other-src'], members: ['member-src'] } - indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] - /** A full pool: the walk found as many readable neighbours as it was asked for. */ - traversedRows = Array.from({ length: 400 }, (_, i) => ({ id: `walked-${i}`, distance: 0.2 })) - rerankRows = [hit('walked-0', 'member-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await vectorSearch({ - ...params, - permitted: { kind: 'unbounded', broad: true }, - accessPlan: { - connectors: eligibility, - observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, - memberSources: ['member-src'], - connectorTypes: new Map(), - uploads: true, - }, - }) - /** One walk over every source, scoped to the bases alone — no source is singled out. */ - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(JSON.stringify(walks[0])).not.toContain('"right":"member-src"') - expect(statements().some((query) => query.sql.includes('WITH readable_chunks'))).toBe(false) - }) - - it('walks an indexed source a bounded caller is a member of instead of ranking it exactly', async () => { - const eligibility = { workspace: [], admin: ['small-src'], members: ['member-src'] } - indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] - sourceExactRows = [{ id: 'small-hit', distance: 0.3, saturated: false }] - traversedRows = [{ id: 'walked-hit', distance: 0.2 }] - rerankRows = [hit('walked-hit', 'member-src'), hit('small-hit', 'small-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await vectorSearch({ - ...params, - topK: 2, - permitted: bounded( - { id: 'doc-a', connectorId: 'member-src' }, - { id: 'doc-b', connectorId: 'small-src' } - ), - accessPlan: { - connectors: eligibility, - observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, - memberSources: ['member-src'], - connectorTypes: new Map(), - uploads: true, - }, - }) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(JSON.stringify(walks[0])).toContain('"right":"member-src"') - expect(statements().some((q) => isExactRanking(q.sql))).toBe(false) - }) - - it('walks the sliced sources when more documents are readable than one ranking may enumerate', async () => { - const eligibility = { workspace: [], admin: ['sliced-src'], members: [] } - /** The slice enumerates in no order, so a saturated one would rank an arbitrary subset. */ - sourceExactRows = [{ id: 'arbitrary-hit', distance: 0.4, saturated: true }] - traversedRows = [{ id: 'walked-hit', distance: 0.2 }] - rerankRows = [hit('walked-hit', 'sliced-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await vectorSearch({ - ...params, - permitted: { kind: 'unbounded', broad: false }, - accessPlan: { - connectors: eligibility, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(1) - expect(JSON.stringify(walks[0])).toContain('sliced-src') - const reranked = JSON.stringify(statements().find((query) => isPageStatement(query.sql))) - expect(reranked).toContain('walked-hit') - expect(reranked).not.toContain('arbitrary-hit') - }) - - it('ranks uploaded documents even when every connector source is walked', async () => { - const eligibility = { workspace: [], admin: [], members: ['member-src'] } - indexedSourceRows = [{ name: 'idx', connectorId: 'member-src' }] - sourceExactRows = [{ id: 'upload-hit', distance: 0.05, saturated: false }] - traversedRows = [{ id: 'walked-hit', distance: 0.2 }] - rerankRows = [hit('upload-hit', null), hit('walked-hit', 'member-src')] - queueTableRows(schemaMock.embedding, rerankRows) - await vectorSearch({ - ...params, - topK: 2, - permitted: { kind: 'unbounded', broad: false }, - accessPlan: { - connectors: eligibility, - observers: { confirmed: [{ id: 'm-1', connectorId: 'member-src' }], observed: [] }, - memberSources: ['member-src'], - connectorTypes: new Map(), - uploads: true, - }, - }) - /** Uploads carry no connector, so their slice runs even with no sliced source beside them. */ - const exact = statements().filter((query) => query.sql.includes('WITH readable_chunks')) - expect(exact).toHaveLength(1) - expect(JSON.stringify(statements().find((q) => isPageStatement(q.sql)))).toContain('upload-hit') - }) - - it('confines keyword matching to the bounded permitted set', async () => { - await keywordSearch({ - ...params, - topK: 1, - query: 'release', - queryVector: params.queryVector!, - permitted: bounded({ id: 'doc-a', connectorId: null }), - }) - const keyword = statements().find((query) => query.sql.includes('WITH matched_keyword_chunks'))! - /** The mock renders the whole WHERE as one parameter, so the restriction shows up in it. */ - expect(JSON.stringify(keyword)).toContain('doc-a') - }) - - describe('Tin keyword ranking for an unbounded caller', () => { - const unbounded: PermittedDocuments = { kind: 'unbounded' } - const keyword = (overrides: Partial[0]> = {}) => - keywordSearch({ - ...params, - topK: 1, - query: 'release', - queryVector: params.queryVector!, - permitted: unbounded, - ...overrides, - }) - const tinStatements = () => - statements().filter((query) => query.sql.includes('ranked_tin_chunks')) - const ginStatements = () => - statements().filter((query) => query.sql.includes('WITH matched_keyword_chunks')) - let tinPages: Array<{ ranked: number; candidates: ReturnType[] }> - - beforeEach(() => { - mockResolveTinKeywordQuery.mockReset() - mockResolveTinKeywordQuery.mockResolvedValue('"releas"') - tinPages = [] - dbChainMockFns.execute.mockImplementation(async (query) => - render(query).sql.includes('ranked_tin_chunks') - ? [tinPages.shift() ?? { ranked: 0, candidates: [] }] - : [] - ) - }) - - it('ranks with Tin and checks access only on the top of that ranking', async () => { - tinPages = [{ ranked: 1500, candidates: [hit('a', null)] }] - queueTableRows(schemaMock.embedding, [{ ...hit('a', null), content: 'release notes' }]) - const results = await keyword() - expect(results.map((row) => row.id)).toEqual(['a']) - expect(mockResolveTinKeywordQuery).toHaveBeenCalledWith('release', 'english', undefined) - expect(ginStatements()).toHaveLength(0) - expect(JSON.stringify(tinStatements()[0])).toContain('2000') - /** `==>` binds tighter than `||`, so the concatenated query must be parenthesized. */ - expect(tinStatements()[0].sql).toContain('==> (?)') - }) - - it('decides a row the fill has not reached on its document while the fill runs', async () => { - tinPages.push({ - ranked: 1, - candidates: [{ id: 'a', documentId: 'doc-a', connectorId: 'src-a' }], - }) - const execute = dbChainMockFns.execute.getMockImplementation()! - dbChainMockFns.execute.mockImplementation(async (query) => - render(query).sql.includes('AS unfilled') ? [{ unfilled: true }] : execute(query) - ) - await keyword({ - accessPlan: { - connectors: { workspace: [], admin: ['src-a'], members: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - const statement = JSON.stringify(tinStatements()[0]) - /** A row the fill has not reached (`acl IS NULL`), or a marked document's row, is decided on its document. */ - expect(statement).toContain(' IS NULL OR ') - expect(statement).toContain('knowledgeProjectionDirty.documentId') - expect(statement).toContain('EXISTS (') - expect(statement).toContain('ranked_tin_chunks.document_id') - }) - - it('widens the window for a broad resolved scope whose first page came back short', async () => { - tinPages = [ - { ranked: 2000, candidates: [] }, - { ranked: 4000, candidates: [hit('b', 'src-a')] }, - ] - queueTableRows(schemaMock.embedding, [{ ...hit('b', 'src-a'), content: 'release notes' }]) - const results = await keyword({ - permitted: { kind: 'unbounded', broad: true }, - accessPlan: { - connectors: { workspace: [], admin: ['src-a'], members: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - expect(results.map((row) => row.id)).toEqual(['b']) - const windows = tinStatements().map((query) => JSON.stringify(query)) - expect(windows).toHaveLength(2) - expect(windows[0]).toContain('2000') - expect(windows[1]).toContain('10000') - }) - - it('hydrates an oversized keyword page in slices and stops at the results it needs', async () => { - const ranked = Array.from({ length: 1000 }, (_, i) => hit(`k-${i}`, 'src-a')) - tinPages = [{ ranked: 20_000, candidates: ranked }] - /** The first slice — as many candidates as results are wanted — fills the page of results. */ - queueTableRows( - schemaMock.embedding, - ranked.slice(0, 20).map((row) => ({ ...row, content: 'release notes' })) - ) - const results = await keyword({ - topK: 20, - permitted: { kind: 'unbounded', broad: false }, - accessPlan: { - connectors: { workspace: [], admin: ['src-a'], members: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - }, - }) - expect(results).toHaveLength(20) - expect(tinStatements()).toHaveLength(1) - /** One hydration, of one slice — never the whole page. */ - const hydrations = dbChainMockFns.where.mock.calls.filter(([condition]) => - hasMockCondition( - condition, - (node) => node.type === 'inArray' && node.column === schemaMock.embedding.id - ) - ) - expect(hydrations).toHaveLength(1) - expect( - hasMockCondition( - hydrations[0][0], - (node) => - node.type === 'inArray' && - node.column === schemaMock.embedding.id && - Array.isArray(node.values) && - node.values.length === 20 - ) - ).toBe(true) - }) - - describe('a bounded set past the exact-ranking size', () => { - const large = Array.from({ length: PERMITTED_EXACT_DOCUMENT_LIMIT }, (_, index) => ({ - id: `doc-${index}`, - connectorId: 'src-a', - })) - const accessPlan = { - connectors: { workspace: [], admin: ['src-a'], members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - } - - it('ranks with Tin as a narrow reader, decided on the row', async () => { - tinPages = [{ ranked: 1500, candidates: [hit('a', 'src-a')] }] - queueTableRows(schemaMock.embedding, [{ ...hit('a', 'src-a'), content: 'release notes' }]) - const results = await keyword({ - permitted: { kind: 'bounded', documents: large }, - accessPlan, - }) - expect(results.map((row) => row.id)).toEqual(['a']) - expect(mockResolveTinKeywordQuery).toHaveBeenCalledTimes(1) - expect(tinStatements()).toHaveLength(1) - expect(JSON.stringify(tinStatements()[0])).toContain('2000') - expect(JSON.stringify(tinStatements()[0])).not.toContain('doc-4999') - expect(ginStatements()).toHaveLength(0) - }) - - it('leaves a later page short rather than resuming a different ranking at its offset', async () => { - /** The first page fills from Tin; hydration keeps half, so a second page is asked for. */ - const first = Array.from({ length: 40 }, (_, index) => hit(`t-${index}`, 'src-a')) - tinPages = [ - { ranked: 2000, candidates: first }, - { ranked: 2000, candidates: [] }, - { ranked: 20_000, candidates: [] }, - ] - queueTableRows( - schemaMock.embedding, - first.slice(0, 20).map((row) => ({ ...row, content: 'release notes' })) - ) - const results = await keyword({ - topK: 40, - permitted: { kind: 'bounded', documents: large }, - accessPlan, - }) - expect(results).toHaveLength(20) - expect(tinStatements()).toHaveLength(3) - expect(ginStatements()).toHaveLength(0) - }) - }) - - it('keeps GIN ranking when Tin is not ready or cannot express the query', async () => { - mockResolveTinKeywordQuery.mockResolvedValue(null) - await keyword() - expect(tinStatements()).toHaveLength(0) - expect(ginStatements()).toHaveLength(1) - }) - }) - - it('skips keyword SQL entirely when nothing is permitted', async () => { - expect( - await keywordSearch({ - ...params, - query: 'release', - queryVector: params.queryVector!, - permitted: bounded(), - }) - ).toEqual([]) - expect(dbChainMockFns.execute).not.toHaveBeenCalled() - }) - - it('reads a user scope through its reachable documents and reports saturation', async () => { - probeRows = [{ id: 'doc-a', connectorId: null, saturated: false }] - await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: { ...reader, tokens: ['u:reachable-documents@example.com'] }, - accessPlan: openPlan(), - filtered: false, - }) - const user = statements().find((query) => isProbeStatement(query.sql))! - const userSql = user.sql - expect(userSql).toContain('WITH reach AS MATERIALIZED') - expect(userSql).toContain('reachable AS MATERIALIZED') - expect(userSql).toContain('FROM reachable AS') - expect(userSql).toContain('AS saturated') - /** - * Baseline tokens reach every tenant's org-wide, public, and uploaded documents, so both the - * count and the rows are confined to the requested bases, outside the fence around the index. - */ - const [reach, reachable] = userSql.split('reachable AS MATERIALIZED') - for (const cte of [reach, reachable.split('FROM reachable AS')[0]]) { - expect(cte).toMatch(/OFFSET 0\s*\) AS \?\s*WHERE \?/) - } - expect(JSON.stringify(user.params)).toContain('org-index') - }) - - it.each([ - [[{ id: null, connectorId: null, saturated: true }], 'unbounded'], - [[{ id: 'doc-a', connectorId: null, saturated: false }], 'bounded'], - ] as const)('resolves %j as %s', async (rows, kind) => { - probeRows = [...rows] - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: { ...reader, tokens: [`u:resolves-${kind}@example.com`] }, - accessPlan: openPlan(), - filtered: false, - }) - expect(permitted.kind).toBe(kind) - if (permitted.kind === 'bounded') - expect(permitted.documents).toEqual([{ id: 'doc-a', connectorId: null }]) - }) - - describe('saturated reach', () => { - const scope = (name: string): UserAccessScope => ({ - ...reader, - tokens: [`u:${name}@example.com`], - }) - const resolve = (access: UserAccessScope, knowledgeBaseIds = ['org-index']) => - resolvePermittedDocuments({ - knowledgeBaseIds, - access, - accessPlan: openPlan(), - filtered: false, - }) - const probes = () => statements().filter((query) => isProbeStatement(query.sql)).length - - it('counts a saturated reach against the broad bound once, and remembers the answer', async () => { - /** The index holds a million documents; the bound is a quarter of them. */ - const counts = { index: 1_000_000, reached: 250_000 } - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (isProbeStatement(statement)) return [{ id: null, connectorId: null, saturated: true }] - if (statement.includes(') reached')) return [{ n: counts.reached }] - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': counts.index } }] }] - return [] - }) - const reachCounts = () => statements().filter((query) => query.sql.includes(') reached')) - const broad = await resolve(scope('broad-reach')) - expect(broad).toEqual({ kind: 'unbounded', broad: true }) - expect(reachCounts()).toHaveLength(1) - expect(JSON.stringify(reachCounts()[0])).toContain('250000') - await resolve(scope('broad-reach')) - expect(reachCounts()).toHaveLength(1) - counts.reached = 120_000 - const narrow = await resolve(scope('narrow-reach')) - expect(narrow).toEqual({ kind: 'unbounded', broad: false }) - expect(reachCounts()).toHaveLength(2) - }) - - it('does not remember a saturated reach whose count ran out of time', async () => { - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (isProbeStatement(statement)) return [{ id: null, connectorId: null, saturated: true }] - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] - if (statement.includes(') reached')) - throw Object.assign(new Error('canceling statement due to statement timeout'), { - code: '57014', - }) - return [] - }) - const reachCounts = () => statements().filter((query) => query.sql.includes(') reached')) - const budget = () => new SearchBudget('vector', performance.now() + 10_000) - expect( - await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: scope('timed-saturated'), - budget: budget(), - accessPlan: openPlan(), - filtered: false, - }) - ).toEqual({ kind: 'unbounded', broad: true }) - expect(reachCounts()).toHaveLength(1) - /** The next search probes and counts again rather than trusting a reach that was never measured. */ - await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: scope('timed-saturated'), - budget: budget(), - accessPlan: openPlan(), - filtered: false, - }) - expect(probes()).toBe(2) - expect(reachCounts()).toHaveLength(2) - }) - - it('does not read an unanalyzed index as a reach of nothing', async () => { - /** The planner knows no rows yet, so the bound is zero and the count looked at nothing. */ - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (statement.includes('EXPLAIN')) return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 0 } }] }] - if (statement.includes(') reached')) return [{ n: 0 }] - return [] - }) - await expect( - resolveReach( - ['org-index'], - scope('unanalyzed'), - new SearchBudget('vector', performance.now() + 10_000), - { - connectors: { workspace: [], admin: [], members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(), - uploads: true, - } - ) - ).resolves.toEqual({ kind: 'unbounded', broad: true }) - }) - - it('is remembered per set of bases and tokens', async () => { - probeRows = [{ id: null, connectorId: null, saturated: true }] - await resolve(scope('per-key')) - probeRows = [{ id: 'doc-a', connectorId: null, saturated: false }] - expect((await resolve(scope('per-key'), ['other-index'])).kind).toBe('bounded') - expect((await resolve(scope('per-key-other'))).kind).toBe('bounded') - expect(probes()).toBe(3) - }) - }) - - it('reports an exhausted vector budget as unbounded instead of failing both legs', async () => { - const budget = new SearchBudget('vector', performance.now() - 1) - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: reader, - budget, - accessPlan: openPlan(), - filtered: false, - }) - expect(permitted.kind).toBe('unbounded') - expect(budget.timedOut).toBe(true) - }) - - const liveSearch = { - knowledgeBaseIds: ['org-index'], - indexedRetrieval: true, - topK: 1, - searchMode: 'hybrid' as const, - query: 'release', - queryVector: params.queryVector!, - } - - it('never asks a source for live grants when the scope reads none', async () => { - const getForConnectors = vi.fn(async () => reader) - await retrieveKnowledgeSearch({ - ...liveSearch, - access: reader, - accessProvider: { ...provider, getForConnectors }, - }) - expect(getForConnectors).not.toHaveBeenCalled() - }) - - it('rebuilds the pool without a gated source the caller turns out not to hold', async () => { - queueTableRows(schemaMock.knowledgeConnector, [ - { - id: 'gated-src', - accessMode: 'admin', - connectorType: 'confluence', - githubRepository: false, - }, - ]) - /** - * The first pool is filled by the gated source alone; only a pool built without it — the - * exclusion carries the source id into the walk — reaches the accessible candidate. The - * projection is filled, so the walk carries each candidate's source and no page is read. - */ - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - if (statement.includes('AS unfilled')) return [{ unfilled: false }] - const rebuilt = JSON.stringify(query).includes('/* excluded sources */') - if (isWalk(statement)) - return Array.from({ length: 400 }, (_, i) => - i === 0 - ? rebuilt - ? hit('b', 'other-src') - : hit('a', 'gated-src') - : hit(`w-${i}`, rebuilt ? 'other-src' : 'gated-src') - ) - return [] - }) - queueTableRows(schemaMock.embedding, []) - queueTableRows(schemaMock.embedding, [hit('b', 'other-src')]) - /** No grants come back, so the gated source is denied. */ - const getForConnectors = vi.fn(async () => reader) - const result = await retrieveKnowledgeSearch({ - ...liveSearch, - searchMode: 'vector', - access: reader, - accessProvider: { ...provider, getForConnectors }, - }) - expect(getForConnectors).toHaveBeenCalledOnce() - expect(result.rows.map((row) => row.id)).toEqual(['b']) - const walks = statements().filter((query) => isWalk(query.sql)) - expect(walks).toHaveLength(2) - expect(JSON.stringify(walks[0])).not.toContain('/* excluded sources */') - expect(JSON.stringify(walks[1])).toContain('/* excluded sources */') - expect(JSON.stringify(walks[1])).toContain('OR NOT (') - expect(statements().some((query) => isPageStatement(query.sql))).toBe(false) - }) -}) - -describe('filters on a resolved scope', () => { - const reader: UserAccessScope = { - kind: 'user', - userId: 'reader', - tokens: ['u:reader@example.com'], - } - const provider: KnowledgeAccessProvider = { - get: async () => reader, - getForConnectors: async () => reader, - getForDocuments: async () => reader, - liveSourceConnectorCondition: async () => null, - } - const params: SearchParams = { - knowledgeBaseIds: ['org-index'], - topK: 1, - access: reader, - queryVector: { vector: '[0.1,0.2]', dimensions: 1536, model: 'text-embedding-3-small' }, - distanceThreshold: 1, - } - const plan = (sources: string[] = ['src-a']) => ({ - connectors: { workspace: [], admin: sources, members: [], liveProofRequired: [] }, - observers: { confirmed: [], observed: [] }, - memberSources: [], - connectorTypes: new Map(sources.map((id) => [id, 'slack'])), - uploads: true, - }) - const hit = (id: string, connectorId: string | null) => ({ - id, - documentId: `doc-${id}`, - connectorId, - distance: 0.1, - }) - let probeRows: Array<{ id: string | null; connectorId: string | null; saturated: boolean }> - let traversedRows: Array<{ id: string; distance?: number }> - let rerankRows: Array> - let exactRows: Array<{ id: string }> - let indexedSourceRows: Array<{ name: string; connectorId: string }> - - beforeEach(() => { - resetDbChainMock() - forgetIndexedVectorSources() - forgetSearchReach() - probeRows = [] - traversedRows = [] - rerankRows = [] - exactRows = [] - indexedSourceRows = [] - dbChainMockFns.execute.mockImplementation(async (query) => { - const statement = render(query).sql - /** The fixtures model the page read, which only an unfilled projection makes. */ - if (statement.includes('AS unfilled')) return [{ unfilled: true }] - if (statement.includes('EXPLAIN')) - return [{ 'QUERY PLAN': [{ Plan: { 'Plan Rows': 1_000_000 } }] }] - if (statement.includes('pg_index')) return indexedSourceRows - if (isExactRanking(statement)) return exactRows - if (statement.includes(') reached')) return [{ n: 250_000 }] - if (isProbeStatement(statement)) return probeRows - if (isWalk(statement)) return traversedRows - if (isPageStatement(statement)) return rerankRows - if (statement.includes('ranked_tin_chunks')) return [{ ranked: 0, candidates: [] }] - return [] - }) - }) - - it('enumerates the documents a date filter admits even when the reach is remembered', async () => { - probeRows = [{ id: 'doc-recent', connectorId: 'src-a', saturated: false }] - const budget = () => new SearchBudget('vector', performance.now() + 10_000) - await resolveReach(['org-index'], reader, budget(), plan()) - const permitted = await resolvePermittedDocuments({ - knowledgeBaseIds: ['org-index'], - access: reader, - filters: { modifiedAfter: '2026-09-13T00:00:00.000Z' }, - budget: budget(), - accessPlan: plan(), - filtered: true, - }) - expect(permitted).toEqual({ - kind: 'bounded', - documents: [{ id: 'doc-recent', connectorId: 'src-a' }], - }) - const probes = statements().filter((query) => query.sql.includes('AS saturated')) - expect(probes).toHaveLength(1) - /** Filter first, over the date index: never the reach count that reports a broad reader saturated. */ - expect(probes[0].sql).not.toContain('WITH reach') - /** An index-driven probe earns its own budget: a window at the document limit fits inside it. */ - const deadlines = statements().filter((query) => query.sql.includes('statement_timeout')) - expect(deadlines.at(-1)?.params[0]).toBe('1500') - expect(JSON.stringify(probes[0])).toContain('"type":"gte"') - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/legs.ts b/apps/sim/lib/sim-search/indexed/retrieval/legs.ts deleted file mode 100644 index 2eb4a9aa2c6..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/legs.ts +++ /dev/null @@ -1,128 +0,0 @@ -import type { KnowledgeAccessProvider, UserAccessScope } from '@/lib/knowledge/access/types' -import type { SearchBudget } from '@/lib/knowledge/search/budget' -import { - liveSourceAccessForConnectors, - type RetrievalLegs, - type SearchParams, - type SearchResult, - VECTOR_PROBE_BUDGET_MS, - VECTOR_PROBE_DOCUMENT_LIMIT, -} from '@/lib/knowledge/search/candidates' -import { measureSearchStage } from '@/lib/knowledge/search/diagnostics' -import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' -import { selectAuthorizedTagResults } from '@/lib/knowledge/search/tag-filters' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' -import { - resolveSearchAccessPlan, - restrictSearchAccessPlan, -} from '@/lib/sim-search/indexed/retrieval/access-plan' -import { executeIndexedKeywordSearch } from '@/lib/sim-search/indexed/retrieval/keyword' -import { - estimateFilteredDocuments, - type IndexedRetrievalContext, - isSearchFiltered, - type PermittedDocuments, - resolvePermittedDocuments, - resolveReach, -} from '@/lib/sim-search/indexed/retrieval/permitted' -import { knowledgeCandidateAccessConditionForConnectors } from '@/lib/sim-search/indexed/retrieval/projection-access' -import { selectIndexedVectorResults } from '@/lib/sim-search/indexed/retrieval/vector' - -/** - * The tag-only leg of a user-scoped search-index search: candidate identities under the caller's - * resolved plan, in id order, hydrated under the full read predicate once live source proof is - * known. - */ -export function selectIndexedTagResults( - params: SearchParams, - context: IndexedRetrievalContext -): Promise { - if (!params.structuredFilters || params.structuredFilters.length === 0) { - throw new Error('Tag filters are required for tag-only search') - } - return selectAuthorizedTagResults( - params, - knowledgeCandidateAccessConditionForConnectors(context.access, context.accessPlan), - context.liveSourceAccess - ) -} - -/** - * Resolves what a user-scoped search over search indexes needs before any leg ranks, and binds it - * to the search-index legs. Connector state is the same for every document a connector owns, so - * every leg reads it from one resolution instead of proving it per candidate. A source filter - * confines the plan rather than the rows: with only that kind of source eligible, every predicate - * the plan builds and every source the legs walk is that kind. - * - * A ranked search also resolves, once and on the vector leg's budget, what the caller may read. - * A filter that leaves few documents is enumerated and ranked exactly inside them, both legs: the - * row does not carry the document's date, and a keyword ranking of the whole base may hold few of - * a small source's matches. A filter that leaves many is ranked as the scope is — the source - * confined on the row, the date tested through the document — since a set that large holds most - * of the query's neighbours anyway. The planner's estimate decides which. Otherwise readability is - * decided on the projection row, so the caller's reach alone chooses between one walk over the - * whole graph and a search of each source. Explicit documents are already a bounded scope with - * their own exhaustive ordering. - */ -export async function prepareIndexedRetrieval(input: { - knowledgeBaseIds: string[] - access: UserAccessScope - accessProvider: KnowledgeAccessProvider - filters?: WorkspaceSearchFilters - signal?: AbortSignal - /** Whether the search ranks a query; a tag-only search needs no permitted set. */ - ranked: boolean - /** The vector leg's budget, which resolving the permitted set spends. */ - budget: SearchBudget -}): Promise { - assertIndexedOrgSearchEnabled() - const { knowledgeBaseIds, access, filters } = input - const resolvedPlan = await measureSearchStage('access_plan', () => - resolveSearchAccessPlan(knowledgeBaseIds, access) - ) - const accessPlan = filters?.source - ? restrictSearchAccessPlan(resolvedPlan, filters.source) - : resolvedPlan - const liveSourceAccess = liveSourceAccessForConnectors( - accessPlan.connectors.liveProofRequired, - input.accessProvider, - input.signal - ) - const filtered = isSearchFiltered(filters) - let permitted: PermittedDocuments | undefined - if (input.ranked && !filters?.documentIds?.length) { - /** Planning only, so a short cap of its own: running past it answers as the wide window it may be. */ - const estimateBudget = input.budget.capped(VECTOR_PROBE_BUDGET_MS) - const enumerateFiltered = - filters && filtered - ? await estimateFilteredDocuments(knowledgeBaseIds, filters, accessPlan, estimateBudget) - .then((estimate) => estimate <= VECTOR_PROBE_DOCUMENT_LIMIT) - .catch((error) => { - if (!estimateBudget.isTimeout(error)) throw error - return false - }) - : false - permitted = enumerateFiltered - ? await resolvePermittedDocuments({ - knowledgeBaseIds, - access, - filters, - budget: input.budget, - accessPlan, - filtered, - }) - : await resolveReach(knowledgeBaseIds, access, input.budget, accessPlan) - } - const context: IndexedRetrievalContext = { - access, - accessPlan, - filtered, - permitted, - liveSourceAccess, - } - return { - tags: (params) => selectIndexedTagResults(params, context), - vector: (params) => selectIndexedVectorResults(params, context), - keyword: (params) => executeIndexedKeywordSearch(params, context), - } -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/permitted.ts b/apps/sim/lib/sim-search/indexed/retrieval/permitted.ts deleted file mode 100644 index 70518416745..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/permitted.ts +++ /dev/null @@ -1,424 +0,0 @@ -import { document } from '@sim/db/schema' -import { sha256Hex } from '@sim/security/hash' -import { and, inArray, isNull, type SQL, sql } from 'drizzle-orm' -import type { AnyPgColumn } from 'drizzle-orm/pg-core' -import { LRUCache } from 'lru-cache' -import { textArrayLiteral } from '@/lib/knowledge/access/predicate' -import type { UserAccessScope } from '@/lib/knowledge/access/types' -import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' -import { - candidateDocumentConditions, - directVisibleDocumentsQuery, - type LiveSourceAccess, - type PermittedDocument, - type ProbeOutcome, - probeVisibleDocuments, - VECTOR_PROBE_BUDGET_MS, - VECTOR_PROBE_DOCUMENT_LIMIT, -} from '@/lib/knowledge/search/candidates' -import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' -import { searchDateFilterCondition } from '@/lib/knowledge/search/filter-conditions' -import type { WorkspaceSearchFilters } from '@/lib/knowledge/search/filters' -import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' -import { - knowledgeAclOverlapCondition, - knowledgeCandidateAccessConditionForConnectors, -} from '@/lib/sim-search/indexed/retrieval/projection-access' - -/** - * What a filter-first probe may spend: it reads the filtered documents off their own index and - * tests each one's access, bounded by the same document limit, and measures around 2 µs per - * document to enumerate plus the access test — a window at the limit fits with room. Its result - * is ranked exactly, at a cost that is predictable where a walk through a mostly-excluded - * neighbourhood is not. - */ -const FILTERED_PROBE_BUDGET_MS = 1500 - -/** - * Whether a filter narrows the documents in a way the projection row cannot see — a date window, - * or a source kind, which confines the plan — so the filtered set is worth estimating and, when - * small, enumerating directly. - */ -export function isSearchFiltered(filters: WorkspaceSearchFilters | undefined): boolean { - return Boolean(searchDateFilterCondition(filters) || filters?.source) -} - -/** A row whose document satisfies `condition`; nothing when there is nothing to ask the document. */ -export function documentSatisfies( - documentId: AnyPgColumn | SQL, - condition: SQL | undefined -): SQL | undefined { - if (condition === undefined) return undefined - return sql`EXISTS (SELECT 1 FROM ${document} WHERE ${and(sql`${document.id} = ${documentId}`, condition)})` -} - -/** - * Documents a bounded permitted set may hold before ranking it exactly costs more than walking - * the graph on the row. Exact ranking reads every chunk of the set, a few per document, where an - * on-row walk reads at most a capped number of tuples; at this size the two meet. A set past it - * is walked first and ranked exactly only if the walk cannot fill its pool, so its recall is never - * below the exact ranking's and its usual cost is the walk's. The same size turns the keyword leg - * from a read of the set's every chunk into a ranking decided on the row. - */ -export const PERMITTED_EXACT_DOCUMENT_LIMIT = 5_000 - -/** - * The documents a user-scoped search-index search may rank, resolved once before either leg runs. - * - * Organization search indexes grant most documents to a single mailbox, channel, or file owner, - * so a member typically reads a vanishing share of the index. Ranking the whole index and - * checking access afterwards then scans thousands of candidates to find none; ranking inside the - * permitted set finds every eligible chunk at a cost proportional to what the member can read. - * `unbounded` means the set exceeded the probe's limit, where post-filtered index search fills - * quickly because most candidates are readable. - */ -export type PermittedDocuments = - | { kind: 'bounded'; documents: readonly PermittedDocument[] } - | { kind: 'unbounded'; broad: boolean } - -/** - * The share of the index a caller must reach before the whole graph is walked for them. pgvector - * post-filters, so a walk returns a caller's own neighbours in proportion to their reach: above - * this share almost every neighbour the graph visits is theirs and one walk is the cheapest exact - * answer there is; below it the walk spends its budget on chunks they cannot read, and each - * readable source is searched on its own instead. - */ -const BROAD_REACH_SHARE = 0.25 - -/** - * How long a caller's saturated reach is remembered. Reach counts the documents a caller's tokens - * touch in the bases, which moves slowly, and an unbounded set only means the legs search the - * index with the full access predicate, so a stale answer costs speed, never access. - */ -const SATURATED_REACH_TTL_MS = 5 * 60 * 1000 - -/** - * A counted reach: whether it is broad enough to walk the whole graph for, or empty, in which - * case the caller reads nothing in these bases and no leg has anything to rank. - */ -interface CountedReach { - broad: boolean - empty: boolean -} - -/** - * Only breadth is remembered. Emptiness decides completeness, not strategy, so it is counted on - * every search: the count of a reach of nothing finds nothing and costs almost nothing. - */ -const saturatedReach = new LRUCache({ - max: 10_000, - ttl: SATURATED_REACH_TTL_MS, -}) - -/** How many documents the bases hold: the denominator of a reach share, and it moves slowly. */ -const indexDocumentCounts = new LRUCache({ - max: 1000, - ttl: SATURATED_REACH_TTL_MS, - /** - * The planner's estimate of the bases' documents, from the statistics it already keeps: a share - * threshold needs the order of magnitude, and counting every row to learn it costs more than the - * search it serves. The read that misses is the search's own, under its deadline. - */ - fetchMethod: async (key, _stale, { context: budget }) => { - const [row] = await runSearchQuery(budget, 'permitted_documents', (executor) => - executor.execute<{ 'QUERY PLAN': Array<{ Plan: { 'Plan Rows': number } }> }>(sql` - EXPLAIN (FORMAT JSON) SELECT 1 FROM ${document} - WHERE ${document.knowledgeBaseId} = ANY(${textArrayLiteral(key.split(','))}) - AND ${document.deletedAt} IS NULL`) - ) - /** An empty answer is not remembered; the bases may simply not have been analyzed yet. */ - return Number(row?.['QUERY PLAN']?.[0]?.Plan?.['Plan Rows'] ?? 0) || undefined - }, -}) - -/** - * The permitted-set probe's SQL, returning at most one row past the document limit. - * - * A user scope first materializes the documents its tokens reach in these bases, read through - * `doc_acl_gin_idx` alone, then applies the state and full access conditions to those rows in - * memory; the set is aliased as `document` so the shared conditions bind to it unchanged. Handed - * the combined predicate instead, PostgreSQL misjudges the token overlap as unselective and - * intersects it with base-wide indexes that read the whole search index. - * - * The index is global and every caller holds the baseline tokens every tenant's org-wide, public, - * and uploaded documents carry, so the reach must be counted inside these bases or those - * documents alone would saturate it. The base check is applied outside an `OFFSET 0` fence so it - * filters the index's rows instead of replacing the index with a base-wide scan. The reach is - * counted before any row is materialized, so a caller whose tokens reach past the limit pays only - * for the count, and a `saturated` sentinel row then reports the set as unbounded. Resolved scopes - * hold base-wide tokens, so they filter directly. - */ -function visibleDocumentsQuery( - knowledgeBaseIds: string[], - conditions: (SQL | undefined)[], - access: UserAccessScope, - shape: 'reach-first' | 'direct' = 'reach-first' -): SQL { - const limit = VECTOR_PROBE_DOCUMENT_LIMIT + 1 - /** - * `direct` applies the conditions as they are: a date filter is selective on its own and has - * its own index, so counting the reach first would only report a broad caller as saturated - * before the filter was consulted. - */ - if (shape === 'direct') return directVisibleDocumentsQuery(conditions) - /** Exactly `doc_acl_gin_idx`'s predicate, so both the count and the rows read that index alone. */ - const reached = sql`${document.deletedAt} IS NULL AND ${knowledgeAclOverlapCondition(access)}` - const underLimit = sql`(SELECT n FROM reach) < ${limit}` - const inBases = inArray(document.knowledgeBaseId, knowledgeBaseIds) - return sql` - WITH reach AS MATERIALIZED ( - SELECT count(*) AS n FROM ( - SELECT 1 FROM ( - SELECT ${document.knowledgeBaseId} FROM ${document} WHERE ${reached} OFFSET 0 - ) AS ${document} - WHERE ${inBases} - LIMIT ${limit} - ) AS reached - ), reachable AS MATERIALIZED ( - SELECT * FROM ( - SELECT * FROM ${document} WHERE ${underLimit} AND ${reached} OFFSET 0 - ) AS ${document} - WHERE ${inBases} - ) - ( - SELECT ${document.id} AS id, ${document.connectorId} AS "connectorId", false AS saturated - FROM reachable AS ${document} - WHERE ${underLimit} AND ${and(...conditions)} - LIMIT ${limit} - ) - UNION ALL - SELECT NULL, NULL, true WHERE (SELECT n FROM reach) >= ${limit} - ` -} - -/** - * The planner's estimate of the documents a filter leaves in the bases — a date filter from the - * statistics on its index, a source filter from its connectors' — so whether the filtered set is - * worth enumerating is decided from its order of magnitude, without reading a row. - */ -export async function estimateFilteredDocuments( - knowledgeBaseIds: string[], - filters: WorkspaceSearchFilters, - plan: SearchAccessPlan, - budget: SearchBudget | undefined -): Promise { - const [row] = await runSearchQuery(budget, 'permitted_documents', (executor) => - executor.execute<{ 'QUERY PLAN': Array<{ Plan: { 'Plan Rows': number } }> }>(sql` - EXPLAIN (FORMAT JSON) SELECT 1 FROM ${document} - WHERE ${and( - inArray(document.knowledgeBaseId, knowledgeBaseIds), - isNull(document.deletedAt), - searchDateFilterCondition(filters), - filters.source ? planSourceCondition(plan) : undefined - )}`) - ) - return Number(row?.['QUERY PLAN']?.[0]?.Plan?.['Plan Rows'] ?? 0) -} - -/** - * How far a caller reaches: broad when they reach at least {@link BROAD_REACH_SHARE} of the - * bases' documents, empty when they reach none. A reach of nothing is a bounded set of nothing: a - * caller who reads no document in these bases, such as a member with no source of their own yet, - * has nothing for any leg to rank, where an unbounded set would have each leg scan to its - * deadline for rows it cannot find. Breadth is counted once against the bound and remembered, so - * the first search after the window pays for it and the rest do not. A caller whose probe already - * saturated is known to reach past the probe's limit, so a bound inside that limit is met without - * counting. - * - * The count reads as many index entries as the caller reaches, so on a large index it can cost - * more than the leg it serves; it gets the probe's share of the deadline, never the whole leg's. - * A count that runs out of that share answers `null`: the leg keeps its time and its deadline - * intact, and the caller decides this search alone without remembering anything. - */ -async function countReach( - knowledgeBaseIds: string[], - access: UserAccessScope, - budget: SearchBudget | undefined, - plan: SearchAccessPlan, - saturated: boolean -): Promise { - const countBudget = budget?.capped(VECTOR_PROBE_BUDGET_MS) - try { - const total = - (await indexDocumentCounts.fetch([...knowledgeBaseIds].sort().join(','), { - context: countBudget, - })) ?? 0 - const bound = Math.ceil(total * BROAD_REACH_SHARE) - if (saturated && bound <= VECTOR_PROBE_DOCUMENT_LIMIT) return { broad: true, empty: false } - const [row] = await runSearchQuery(countBudget, 'permitted_documents', (executor) => - executor.execute<{ n: number }>(sql` - SELECT count(*) AS n FROM ( - SELECT 1 FROM ${document} - WHERE ${and( - isNull(document.deletedAt), - knowledgeAclOverlapCondition(access), - inArray(document.knowledgeBaseId, knowledgeBaseIds), - planSourceCondition(plan) - )} - LIMIT ${bound} - ) reached`) - ) - const reached = Number(row?.n ?? 0) - /** A count that looked and found nothing: only a bound of zero looks at nothing. */ - return { broad: reached >= bound, empty: bound > 0 && reached === 0 } - } catch (error) { - if (!budget || !countBudget?.isTimeout(error)) throw error - /** Only the count's share was spent; the leg's own deadline still governs. */ - budget.remaining() - return null - } -} - -/** - * Reach depends on the bases, the caller's tokens and, when the plan is confined to one kind of - * source, which sources those are; a date filter narrows the set, not the reach. - */ -function reachKey( - knowledgeBaseIds: readonly string[], - access: UserAccessScope, - plan: SearchAccessPlan -): string { - const sources = `:${sha256Hex([...planSources(plan)].sort().join('\n'))}:${plan.uploads}` - return `${[...knowledgeBaseIds].sort().join(',')}:${sha256Hex([...access.tokens].sort().join('\n'))}${sources}` -} - -/** Every connector the plan admits, whatever its access mode. */ -function planSources(plan: SearchAccessPlan): readonly string[] { - return [...plan.connectors.workspace, ...plan.connectors.admin, ...plan.connectors.members] -} - -/** The documents a plan's sources own, on the document row; every source when unconfined. */ -function planSourceCondition(plan: SearchAccessPlan): SQL { - const owned = planSources(plan) - const inSources = owned.length - ? sql`${document.connectorId} = ANY(${textArrayLiteral([...owned])})` - : sql`false` - return plan.uploads ? sql`(${document.connectorId} IS NULL OR ${inSources})` : inSources -} - -/** Forgets every remembered reach, after the bases' documents or a caller's tokens changed. */ -export function forgetSearchReach(): void { - saturatedReach.clear() - indexDocumentCounts.clear() -} - -/** A resolved scope's reach, remembered per bases and tokens, with no document enumerated. */ -export async function resolveReach( - knowledgeBaseIds: string[], - access: UserAccessScope, - budget: SearchBudget | undefined, - plan: SearchAccessPlan -): Promise { - const key = reachKey(knowledgeBaseIds, access, plan) - const remembered = saturatedReach.get(key) - if (remembered) return { kind: 'unbounded', broad: remembered.broad } - try { - const reach = await countReach(knowledgeBaseIds, access, budget, plan, false) - /** A count that ran out of time decides this search only; the next one counts again. */ - if (reach === null) return { kind: 'unbounded', broad: true } - if (reach.empty) return { kind: 'bounded', documents: [] } - saturatedReach.set(key, { broad: reach.broad }) - return { kind: 'unbounded', broad: reach.broad } - } catch (error) { - /** The leg's own deadline passed during the count: the leg is short, the search is not failed. */ - if (!budget?.isTimeout(error)) throw error - return { kind: 'unbounded', broad: true } - } -} - -/** - * Resolve the permitted set with the candidate predicate both legs apply, so restricting a leg - * to it never admits a document the leg would otherwise refuse. Tag filters stay chunk-level in - * each leg; the set is the document-level superset they narrow. - * - * It runs ahead of both legs on the vector leg's budget, so exhausting that budget here reports - * `unbounded` and marks the vector leg timed out rather than failing the keyword leg with it. - */ -export async function resolvePermittedDocuments(params: { - knowledgeBaseIds: string[] - access: UserAccessScope - filters?: WorkspaceSearchFilters - budget?: SearchBudget - accessPlan: SearchAccessPlan - /** Whether a date or source filter narrows the set, which then is enumerated directly. */ - filtered: boolean -}): Promise { - const key = reachKey(params.knowledgeBaseIds, params.access, params.accessPlan) - let probe: ProbeOutcome - let broad = true - /** - * A remembered reach says how much of the bases the caller reads, which a date filter does not - * change; the filtered set still has to be enumerated, so under one the probe always runs. A plan - * under a date or source filter enumerates the filtered set directly; reach cannot stand in for it. - */ - const remembered = params.filtered ? undefined : saturatedReach.get(key) - if (remembered) { - probe = { kind: 'saturated' } - broad = remembered.broad - } else { - try { - probe = await probeVisibleDocuments( - visibleDocumentsQuery( - params.knowledgeBaseIds, - candidateDocumentConditions( - params.knowledgeBaseIds, - params.filters, - knowledgeCandidateAccessConditionForConnectors(params.access, params.accessPlan) - ), - params.access, - params.filtered ? 'direct' : 'reach-first' - ), - params.budget, - 'permitted_documents', - params.filtered ? FILTERED_PROBE_BUDGET_MS : VECTOR_PROBE_BUDGET_MS - ) - } catch (error) { - if (!params.budget?.isTimeout(error)) throw error - probe = { kind: 'timed_out' } - } - if (probe.kind === 'saturated') { - try { - const reach = await countReach( - params.knowledgeBaseIds, - params.access, - params.budget, - params.accessPlan, - true - ) - /** A count that ran out of time decides this search only; the next one counts again. */ - if (reach?.empty) probe = { kind: 'documents', documents: [] } - else if (reach !== null) { - broad = reach.broad - saturatedReach.set(key, { broad }) - } - } catch (error) { - /** The leg's own deadline passed during the count: the leg is short, the search is not failed. */ - if (!params.budget?.isTimeout(error)) throw error - } - } - } - const permitted: PermittedDocuments = - probe.kind === 'documents' - ? { kind: 'bounded', documents: probe.documents } - : { kind: 'unbounded', broad } - annotateSearchDiagnostics({ - permittedDocuments: permitted.kind, - ...(probe.kind === 'documents' ? { permittedDocumentCount: probe.documents.length } : {}), - }) - return permitted -} - -/** - * What a user-scoped search-index search resolves about the caller once, before any leg ranks: - * the connectors it may read and the caller's standing in them, the permitted set or reach, and - * the proof the gated sources still need. - */ -export interface IndexedRetrievalContext { - access: UserAccessScope - accessPlan: SearchAccessPlan - /** Whether a date or source filter narrows the documents; see {@link isSearchFiltered}. */ - filtered: boolean - /** Absent for a tag-only search and for explicit documents, which are already a bounded scope. */ - permitted?: PermittedDocuments - liveSourceAccess?: LiveSourceAccess -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts b/apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts deleted file mode 100644 index 284b08e3f61..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/projection-access.ts +++ /dev/null @@ -1,234 +0,0 @@ -import { - document, - knowledgeConnector, - knowledgeDocumentObservation, - knowledgeProjectionDirty, -} from '@sim/db/schema' -import { type SQL, sql } from 'drizzle-orm' -import type { AnyPgColumn } from 'drizzle-orm/pg-core' -import { - aclOverlap, - aclRequirementsSatisfied, - documentHasMirroredAcl, - documentHasWorkspaceAcl, - sourceAclFreshnessCutoff, - textArrayLiteral, -} from '@/lib/knowledge/access/predicate' -import type { UserAccessScope } from '@/lib/knowledge/access/types' -import type { - KnowledgeMemberObserver, - KnowledgeMemberObservers, - SearchAccessPlan, -} from '@/lib/sim-search/indexed/retrieval/access-plan' - -/** - * The candidate predicate with connector state resolved ahead of the query instead of per row. - * - * Deletion, archival, a pending access rewrite, the organization's integration approval and the - * access mode are facts about a connector, not a document, so checking them once per query leaves - * each candidate an id comparison plus its own columns. - * - * `liveSourceAccess` is the caller's live source proof, and defaults to admitting everything: - * candidate ranking defers that proof until after ranking, exactly as - * `knowledgeMetadataCandidateAccessCondition` does, and only a reader that already holds the - * grants — content hydration — passes it. The connectors it would gate are listed separately so - * that clause is applied to those alone. - * - * Either way it narrows exactly as the predicate it stands in for: the eligible ids are the - * connectors that predicate's `EXISTS` would admit, and every document-level clause is carried - * over unchanged. - */ -export function knowledgeCandidateAccessConditionForConnectors( - scope: UserAccessScope, - plan: SearchAccessPlan, - liveSourceAccess: SQL = sql`true` -): SQL { - const eligibility = plan.connectors - if (scope.tokens.length === 0) return sql`false` - const tokens = textArrayLiteral(scope.tokens) - const cutoff = sourceAclFreshnessCutoff() - const liveProof = new Set(eligibility.liveProofRequired) - const inConnectors = (ids: readonly string[]): SQL => - ids.length === 0 - ? sql`false` - : sql`${document.connectorId} = ANY(${textArrayLiteral([...ids])})` - const mirrored = (ids: readonly string[], current: SQL): SQL => { - const direct = ids.filter((id) => !liveProof.has(id)) - const gated = ids.filter((id) => liveProof.has(id)) - const currentAndMirrored = sql`${documentHasMirroredAcl()} AND ${current}` - return sql`( - (${inConnectors(direct)} AND ${currentAndMirrored}) - OR (${inConnectors(gated)} AND ${currentAndMirrored} AND EXISTS ( - SELECT 1 FROM ${knowledgeConnector} - WHERE ${knowledgeConnector.id} = ${document.connectorId} - AND ${liveSourceAccess} - )) - )` - } - const workspaceOwned = plan.uploads - ? sql`(${document.connectorId} IS NULL OR ${inConnectors(eligibility.workspace)})` - : inConnectors(eligibility.workspace) - return sql`( - ${aclOverlap(tokens)} - AND ${aclRequirementsSatisfied(tokens)} - AND ( - (${workspaceOwned} AND ${documentHasWorkspaceAcl()}) - OR ${mirrored(eligibility.admin, sql`${document.aclVerifiedAt} > ${cutoff}`)} - OR ${mirrored(eligibility.members, resolvedObservationCondition(plan.observers, cutoff))} - ) - )` -} - -/** - * Whether a projection row belongs to a document marked for the knowledge projector: its source, - * ACL, or chunks changed and its rows may not show it yet. A probe of the marks' primary key: the - * planner may instead hash the whole set once per statement, which is as cheap while the marks are - * few, and an `IN` would risk re-reading them per row once they outgrow the hash. - */ -export function projectionPending(documentId: AnyPgColumn | SQL): SQL { - return sql`(EXISTS (SELECT 1 FROM ${knowledgeProjectionDirty} WHERE ${knowledgeProjectionDirty.documentId} = ${documentId}))` -} - -/** - * Whether a projection row is decided on its document rather than on its own columns: its - * document is marked for the projector, or, while the source and ACL fill runs, the row has not - * been filled. - */ -export function projectionDecidedOnDocument( - projection: { acl: AnyPgColumn | SQL; documentId: AnyPgColumn | SQL }, - filled: boolean -): SQL { - const pending = projectionPending(projection.documentId) - return filled ? pending : sql`(${projection.acl} IS NULL OR ${pending})` -} - -/** - * The candidate predicate on a ranking projection's own row, for a scope whose connectors were - * resolved: `connectorId` and `acl` are mirrored there from the document, so a walk or a keyword - * window decides readability on the row it scores instead of joining `document` per candidate. - * - * It admits a superset of the document predicate, never a subset: a mirrored ACL names the members - * who observe a document, so overlap with the caller's tokens is the per-row test without the - * observation's freshness, and requirement clauses live on the document. Both are refused there, - * under the full predicate, before content is returned — this predicate only decides what is worth - * ranking. - * - * A row whose columns may be behind its document is decided on the document instead, under - * {@link knowledgeCandidateAccessConditionForConnectors} — the join per candidate that every row - * paid before the columns existed: a row the source and ACL fill has not reached (`acl IS NULL`), - * and every row of a document marked for the knowledge projector. A revoked grant still on such a - * row never admits it, and a new grant not yet on it never hides it from a statement that reaches - * the row. A source-scoped walk or slice reaches rows by the source on the row, though, so a - * document that moved to another source joins that source's ranking once the projector has - * rewritten its rows; until then it can be missing there, never shown where it is not readable. - * The projector and the fill run in the background, so search never waits on either. - */ -export function projectionCandidateAccessCondition( - projection: { - connectorId: AnyPgColumn | SQL - acl: AnyPgColumn | SQL - documentId: AnyPgColumn | SQL - }, - scope: UserAccessScope, - plan: SearchAccessPlan, - options: { - /** - * Whether every row of the projection carries its mirrored source and ACL. While the fill - * is under way, a row it has not reached is decided on its document; once it is complete only - * a marked document's rows are. - */ - filled?: boolean - } = {} -): SQL { - if (scope.tokens.length === 0) return sql`false` - const tokens = textArrayLiteral(scope.tokens) - const inSources = (ids: readonly string[]): SQL => - ids.length === 0 - ? sql`false` - : sql`${projection.connectorId} = ANY(${textArrayLiteral([...ids])})` - const mirrored = [ - ...plan.connectors.workspace, - ...plan.connectors.admin, - ...plan.connectors.members, - ] - const owned = plan.uploads - ? sql`(${projection.connectorId} IS NULL OR ${inSources(mirrored)})` - : inSources(mirrored) - const onRow = sql`(${projection.acl} && ${tokens} AND ${owned})` - /** - * A scalar subquery rather than `EXISTS`: the planner may turn an `EXISTS` into one hash of every - * readable document, a sequential scan of `document` for a statement that only needs a few rows - * decided. A scalar subquery is only ever a primary-key probe per row that needs it. - */ - const onDocument = sql`(SELECT ${document.id} FROM ${document} - WHERE ${document.id} = ${projection.documentId} - AND ${knowledgeCandidateAccessConditionForConnectors(scope, plan)} - LIMIT 1) IS NOT NULL` - return sql`((${projectionDecidedOnDocument(projection, options.filled ?? false)} AND ${onDocument}) - OR (${onRow} AND NOT ${projectionPending(projection.documentId)}))` -} - -/** - * The same membership, resolved ahead of the query: each candidate costs one lookup on the - * observation key instead of a join to the member behind it. Equivalent by construction — the ids - * are the members that join would have matched, and each one's freshness rule is carried over. - */ -function resolvedObservationCondition(observers: KnowledgeMemberObservers, cutoff: SQL): SQL { - if (observers.confirmed.length === 0 && observers.observed.length === 0) return sql`false` - /** - * An observation vouches for a document only from a member of the document's own connector: a - * document that changed hands keeps its old observations, which must not carry it. - */ - const byMember = (members: readonly KnowledgeMemberObserver[]): SQL => - sql`(${knowledgeDocumentObservation.memberId}, ${document.connectorId}) IN (${sql.join( - members.map((member) => sql`(${member.id}, ${member.connectorId})`), - sql`, ` - )})` - const current = - observers.confirmed.length === 0 - ? sql`${byMember(observers.observed)} AND ${knowledgeDocumentObservation.lastSeenAt} > ${cutoff}` - : observers.observed.length === 0 - ? byMember(observers.confirmed) - : sql`(${byMember(observers.confirmed)} - OR (${byMember(observers.observed)} AND ${knowledgeDocumentObservation.lastSeenAt} > ${cutoff}))` - return sql`EXISTS ( - SELECT 1 FROM ${knowledgeDocumentObservation} - WHERE ${knowledgeDocumentObservation.documentId} = ${document.id} - AND ${current} - )` -} - -/** - * The token half of the stored access predicate: the documents a caller's tokens reach before - * any source, freshness, or requirement check narrows them. It is a necessary condition of - * `knowledgeAccessCondition`, never a substitute for it. - * - * Paired with `deleted_at IS NULL` it matches `doc_acl_gin_idx` exactly, so a query can enumerate - * a member's reachable documents from that index alone. PostgreSQL cannot estimate array-overlap - * selectivity, so left to itself it intersects this highly selective bitmap with base-wide ones. - */ -export function knowledgeAclOverlapCondition(scope: UserAccessScope): SQL { - if (scope.tokens.length === 0) return sql`false` - return aclOverlap(textArrayLiteral(scope.tokens)) -} - -/** - * Keeps the rows of sources the caller turned out not to hold out of a ranking decided on the - * row. A row decided on its document — not yet filled, or its document marked for the projector, - * so its own source may be stale — asks the document instead. - */ -export function excludeSearchSourcesOnRow( - projection: { - connectorId: AnyPgColumn | SQL - acl: AnyPgColumn | SQL - documentId: AnyPgColumn | SQL - }, - filled: boolean, - excludedSources: readonly string[] -): SQL | undefined { - if (!excludedSources.length) return undefined - const excluded = textArrayLiteral([...excludedSources]) - const decided = projectionDecidedOnDocument(projection, filled) - return sql`((${decided} AND NOT EXISTS (SELECT 1 FROM ${document} WHERE ${document.id} = ${projection.documentId} AND ${document.connectorId} = ANY(${excluded}))) - OR (NOT ${decided} AND (${projection.connectorId} IS NULL OR NOT (${projection.connectorId} = ANY(${excluded}))))) /* excluded sources */` -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts deleted file mode 100644 index 98f83e7d799..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.test.ts +++ /dev/null @@ -1,78 +0,0 @@ -/** - * @vitest-environment node - */ -import { beforeEach, describe, expect, it, vi } from 'vitest' -import { SearchBudget } from '@/lib/knowledge/search/budget' -import { - forgetProjectionFilled, - isProjectionFilled, -} from '@/lib/sim-search/indexed/retrieval/projection-fill' - -const LEG_BUDGET_MS = 8000 - -describe('projection fill probe', () => { - beforeEach(() => forgetProjectionFilled()) - - it('remembers a failed probe as unfilled rather than probing again on every search', async () => { - const query = vi - .spyOn(SearchBudget.prototype, 'query') - .mockRejectedValue(new Error('connection reset')) - const budget = new SearchBudget('keyword', performance.now() + LEG_BUDGET_MS) - await expect( - isProjectionFilled('embedding_keyword_tin', 'keyword.projection_filled', budget) - ).resolves.toBe(false) - await expect( - isProjectionFilled('embedding_keyword_tin', 'keyword.projection_filled', budget) - ).resolves.toBe(false) - expect(query).toHaveBeenCalledOnce() - }) - - it('does not hold a later search past its own share while another search probes', async () => { - vi.useFakeTimers() - try { - let answerFirst: (rows: Array<{ unfilled: boolean }>) => void = () => {} - vi.spyOn(SearchBudget.prototype, 'query').mockImplementation( - () => - new Promise((resolve) => { - answerFirst = resolve as typeof answerFirst - }) as ReturnType - ) - const first = isProjectionFilled( - 'embedding_search', - 'vector.projection_filled', - new SearchBudget('vector', performance.now() + LEG_BUDGET_MS) - ) - let secondSettled = false - const second = isProjectionFilled( - 'embedding_search', - 'vector.projection_filled', - new SearchBudget('vector', performance.now() + 20) - ).finally(() => { - secondSettled = true - }) - await vi.advanceTimersByTimeAsync(20) - expect(secondSettled).toBe(true) - await expect(second).resolves.toBe(false) - answerFirst([{ unfilled: false }]) - await expect(first).resolves.toBe(true) - } finally { - vi.useRealTimers() - } - }) - - it('does not remember a probe its own search cancelled', async () => { - const query = vi - .spyOn(SearchBudget.prototype, 'query') - .mockRejectedValue(new DOMException('aborted', 'AbortError')) - const controller = new AbortController() - controller.abort() - const budget = new SearchBudget('vector', performance.now() + LEG_BUDGET_MS, controller.signal) - await expect( - isProjectionFilled('embedding_search', 'vector.projection_filled', budget) - ).resolves.toBe(false) - await expect( - isProjectionFilled('embedding_search', 'vector.projection_filled', budget) - ).resolves.toBe(false) - expect(query).toHaveBeenCalledTimes(2) - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts b/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts deleted file mode 100644 index 04b8a1592ff..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/projection-fill.ts +++ /dev/null @@ -1,102 +0,0 @@ -import { SOURCE_ACL_PROJECTIONS, type SourceAclProjection } from '@sim/db/knowledge-projection' -import { embeddingKeywordTin, embeddingSearch } from '@sim/db/schema' -import { sleep } from '@sim/utils/helpers' -import { sql } from 'drizzle-orm' -import { LRUCache } from 'lru-cache' -import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' -import type { SearchStage } from '@/lib/knowledge/search/diagnostics' - -/** How long a fully filled projection is taken on trust before its unfilled rows are looked for again. */ -const PROJECTION_FILLED_TTL_MS = 60_000 - -/** - * The most of a leg's budget the probe may spend. The probe is one index read that answers in - * milliseconds when the partial index serves it; a read slower than this is fighting a cold cache - * or a busy database, and waiting longer would spend the leg's ranking time on an optimization. - * An unanswered probe costs only the slower plan, so a small cap loses nothing. - */ -export const PROJECTION_FILLED_PROBE_BUDGET_MS = 250 - -/** - * How long an unanswered probe is remembered as unfilled. Short enough that a recovered database - * is asked again within seconds, long enough that searches arriving during an outage do not each - * spend their own probe budget rediscovering it. - */ -const PROJECTION_FILLED_UNKNOWN_TTL_MS = 5_000 - -/** - * Whether the ranking projection still holds rows the source and ACL fill has not reached. Read off the - * unfilled-rows index in milliseconds and remembered briefly: the answer only ever changes once. - * - * The read asks for the last unfilled row by id, not whether one exists: an `EXISTS` drops its - * order and limit, and while most rows are unfilled the planner expects a sequential scan to - * meet one at once, then walks the whole projection when the unfilled rows sit past the filled - * ones. Ordered by id and capped at one row, the read can only be the partial index, whose - * last entry is the row the fill reaches last. - */ -const projectionFilled = new LRUCache< - SourceAclProjection, - boolean, - { budget: SearchBudget | undefined; stage: SearchStage } ->({ - max: SOURCE_ACL_PROJECTIONS.length, - ttl: PROJECTION_FILLED_TTL_MS, - /** - * The read that misses the cache is the search's own, capped to a small share of its budget, - * and the searches that miss together share it. A read that fails or runs out of that share - * answers unfilled, the slower and safe form, and that answer is remembered briefly so the - * searches behind it do not each pay for the same failure. A read cut short by its own - * search's cancellation learned nothing about the projection and is not remembered. - */ - fetchMethod: async (projection, _stale, { context, options }) => { - const table = projection === 'embedding_search' ? embeddingSearch : embeddingKeywordTin - try { - const [row] = await runSearchQuery( - context.budget?.capped(PROJECTION_FILLED_PROBE_BUDGET_MS), - context.stage, - (executor) => - executor.execute<{ unfilled: boolean }>(sql` - SELECT ( - SELECT ${table.id} FROM ${table} WHERE ${table.acl} IS NULL - ORDER BY ${table.id} DESC LIMIT 1 - ) IS NOT NULL AS unfilled`) - ) - return !row?.unfilled - } catch { - if (context.budget?.signal?.aborted) return undefined - options.ttl = PROJECTION_FILLED_UNKNOWN_TTL_MS - return false - } - }, -}) - -/** - * Whether every row of the projection carries its mirrored source and ACL; unknown counts as not yet. - * - * Searches that miss the cache together share the first one's read, which is capped to that - * search's share. Each caller still waits no longer than its own share, or its own deadline if - * nearer, and reads an unanswered probe as unfilled: a caller that joined late, with less of its - * leg left, never waits on another search's timetable. A remembered answer is returned at once, - * so only a search that missed the memo starts a wait. - */ -export async function isProjectionFilled( - projection: SourceAclProjection, - stage: SearchStage, - budget: SearchBudget | undefined -): Promise { - const remembered = projectionFilled.get(projection) - if (remembered !== undefined) return remembered - const answer = projectionFilled.fetch(projection, { context: { budget, stage } }) - if (!budget) return (await answer) ?? false - const waitMs = Math.max( - 0, - Math.min(PROJECTION_FILLED_PROBE_BUDGET_MS, budget.deadline - performance.now()) - ) - const unanswered = sleep(waitMs).then(() => undefined) - return (await Promise.race([answer.catch(() => undefined), unanswered])) ?? false -} - -/** Forgets whether the projections were filled; the memo is per process and otherwise expires on its own. */ -export function forgetProjectionFilled(): void { - projectionFilled.clear() -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts b/apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts deleted file mode 100644 index 165757779fe..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/source-vector-indexes.ts +++ /dev/null @@ -1,33 +0,0 @@ -import { sql } from 'drizzle-orm' -import { LRUCache } from 'lru-cache' -import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' - -/** - * The sources that have their own index, cached briefly: every unbounded ranking asks, and the - * answer changes only when a connector deletion drops one. - */ -const indexedSources = new LRUCache<'sources', ReadonlySet>({ max: 1, ttl: 60 * 1000 }) - -/** Forgets the cached answer. */ -export function forgetIndexedVectorSources(): void { - indexedSources.clear() -} - -/** The sources with a graph of their own; a search that misses the memo reads under its own deadline. */ -export async function indexedVectorSources(budget?: SearchBudget): Promise> { - const cached = indexedSources.get('sources') - if (cached) return cached - const rows = await runSearchQuery(budget, 'vector.source_indexes', (executor) => - executor.execute<{ connectorId: string | null }>(sql` - SELECT substring(pg_get_expr(i.indpred, i.indrelid) from '''([0-9a-f-]+)''') AS "connectorId" - FROM pg_index i - JOIN pg_class c ON c.oid = i.indexrelid - WHERE i.indrelid = 'embedding_search'::regclass - AND c.relname LIKE 'embedding_search_src_%' AND i.indisvalid AND i.indisready`) - ) - const sources = new Set( - rows.map((row) => row.connectorId).filter((id): id is string => id !== null) - ) - indexedSources.set('sources', sources) - return sources -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts deleted file mode 100644 index d6b1fbece87..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword-readiness.test.ts +++ /dev/null @@ -1,22 +0,0 @@ -import { dbChainMockFns, resetDbChainMock } from '@sim/testing' -import { expect, it } from 'vitest' -import { resolveTinKeywordQuery } from '@/lib/sim-search/indexed/retrieval/tin-keyword' - -/** Its own file, so the process-wide readiness cache starts empty. */ -it('stays on the GIN projection while the Tin index is incomplete, and remembers that', async () => { - resetDbChainMock() - let indexValid = false - dbChainMockFns.execute.mockImplementation(async (query) => { - const text = JSON.stringify(query) - if (text.includes('indisvalid')) return [{ valid: indexValid }] - if (text.includes('websearch_to_tsquery')) return [{ rendered: "'releas'" }] - return [] - }) - expect(await resolveTinKeywordQuery('release', 'english', undefined)).toBeNull() - indexValid = true - expect(await resolveTinKeywordQuery('release', 'english', undefined)).toBeNull() - const readinessReads = dbChainMockFns.execute.mock.calls.filter(([query]) => - JSON.stringify(query).includes('indisvalid') - ) - expect(readinessReads).toHaveLength(1) -}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts deleted file mode 100644 index 4d6558042c4..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.test.ts +++ /dev/null @@ -1,41 +0,0 @@ -import { dbChainMockFns, resetDbChainMock } from '@sim/testing' -import { beforeEach, describe, expect, it } from 'vitest' -import { SearchBudget, SearchDeadlineError } from '@/lib/knowledge/search/budget' -import { resolveTinKeywordQuery } from '@/lib/sim-search/indexed/retrieval/tin-keyword' - -/** Readiness is cached per process; the incomplete-index case lives in its own file, where the cache starts empty. */ -describe('resolveTinKeywordQuery', () => { - let indexValid: boolean - let rendered: string - - beforeEach(() => { - resetDbChainMock() - indexValid = true - rendered = "'releas' & 'note'" - dbChainMockFns.execute.mockImplementation(async (query) => { - const text = JSON.stringify(query) - if (text.includes('indisvalid')) return indexValid ? [{ valid: true }] : [{ valid: false }] - if (text.includes('websearch_to_tsquery')) return [{ rendered }] - return [] - }) - }) - - it('translates the analyzed query', async () => { - expect(await resolveTinKeywordQuery('release notes', 'english', undefined)).toBe( - '("releas" AND "note")' - ) - }) - - it('reads under the keyword budget, so an expired deadline ends the leg instead of querying', async () => { - const expired = new SearchBudget('keyword', performance.now() - 1) - await expect( - resolveTinKeywordQuery('release notes', 'english', expired) - ).rejects.toBeInstanceOf(SearchDeadlineError) - expect(dbChainMockFns.execute).not.toHaveBeenCalled() - }) - - it('falls back to GIN instead of failing the search when the query cannot be analyzed', async () => { - dbChainMockFns.execute.mockRejectedValue(new Error('connection reset')) - expect(await resolveTinKeywordQuery('release notes', 'english', undefined)).toBeNull() - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts deleted file mode 100644 index 37360aae927..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/tin-keyword.ts +++ /dev/null @@ -1,65 +0,0 @@ -import { EMBEDDING_KEYWORD_TIN_INDEX } from '@sim/db/schema' -import { createLogger } from '@sim/logger' -import { getErrorMessage } from '@sim/utils/errors' -import { sql } from 'drizzle-orm' -import { LRUCache } from 'lru-cache' -import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' -import { tinQueryFromTsquery } from '@/lib/sim-search/indexed/retrieval/tin-query' - -const logger = createLogger('TinKeywordSearch') - -/** - * How long a readiness answer holds. The index only becomes valid once the projection is fully - * backfilled, and a newly valid or dropped index is noticed within this window. - */ -const READINESS_TTL_MS = 60 * 1000 - -/** - * Whether the Tin index exists and finished building, i.e. the projection is complete. The read - * spends the budget of the search that missed the cache. - */ -const indexReadiness = new LRUCache<'index', boolean, SearchBudget | undefined>({ - max: 1, - ttl: READINESS_TTL_MS, - fetchMethod: async (_key, _stale, { context }) => { - const [row] = await runSearchQuery(context, 'keyword.tin_readiness', (executor) => - executor.execute<{ valid: boolean }>(sql` - SELECT i.indisvalid AS valid FROM pg_index i - WHERE i.indexrelid = to_regclass(${EMBEDDING_KEYWORD_TIN_INDEX})`) - ) - return row?.valid === true - }, -}) - -/** - * The TINQL query that ranks `query` inside a search index's bases, or null when keyword search - * must keep the GIN projection: the database has no complete Tin index, or the query uses a shape - * TINQL cannot express. The text is analyzed by the same `websearch_to_tsquery` the GIN path - * uses, so both engines match the same stemmed terms. - * - * Every read runs under the keyword leg's `budget`, so deciding the engine cannot outlast the leg's - * deadline. A read that fails for another reason, including a shared cache read cut short by - * another search's deadline, keeps the GIN projection; this search's own expired deadline or - * cancellation propagates like any other keyword query's. - */ -export async function resolveTinKeywordQuery( - query: string, - ftsConfig: string, - budget: SearchBudget | undefined -): Promise { - try { - if (!(await indexReadiness.fetch('index', { context: budget }))) return null - const [{ rendered }] = await runSearchQuery(budget, 'keyword.tin_query', (executor) => - executor.execute<{ rendered: string }>( - sql`SELECT websearch_to_tsquery(${ftsConfig}::regconfig, ${query})::text AS rendered` - ) - ) - return tinQueryFromTsquery(rendered) - } catch (error) { - budget?.remaining() - logger.warn('Tin keyword readiness check failed; using the GIN projection', { - error: getErrorMessage(error), - }) - return null - } -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts deleted file mode 100644 index 472a52d9665..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/tin-query.test.ts +++ /dev/null @@ -1,23 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { tinQueryFromTsquery } from '@/lib/sim-search/indexed/retrieval/tin-query' - -/** Inputs are `websearch_to_tsquery('english', …)::text` exactly as PostgreSQL renders them. */ -describe('tinQueryFromTsquery', () => { - it('keeps punctuation and reserved words literal', () => { - expect(tinQueryFromTsquery("'user@example.com' & 'https' & '/sim.ai/docs'")).toBe( - '("user@example.com" AND "https" AND "/sim.ai/docs")' - ) - expect(tinQueryFromTsquery("'near' & 'and'")).toBe('("near" AND "and")') - expect(tinQueryFromTsquery("'snake_case' & 'it''s'")).toBe('("snake\\_case" AND "it\'s")') - }) - - it.each([ - ['', 'a query of stopwords only'], - ["!'onlyneg'", 'a lone negation'], - ["!'a' & !'b'", 'a conjunction of negations'], - ["'a' | !'b'", 'a negated disjunct'], - ["'appl':*", 'a prefix match'], - ])('declines %j (%s)', (rendered) => { - expect(tinQueryFromTsquery(rendered)).toBeNull() - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/retrieval/tin-query.ts b/apps/sim/lib/sim-search/indexed/retrieval/tin-query.ts deleted file mode 100644 index aa7b88d5063..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/tin-query.ts +++ /dev/null @@ -1,195 +0,0 @@ -/** - * Translates PostgreSQL's text rendering of a `tsquery` into TINQL, the query language of the Tin - * text index. The keyword leg already derives its `tsquery` with `websearch_to_tsquery`, so the - * lexemes arrive stemmed by the same configuration that stemmed the indexed text; translating - * them keeps Tin matching the same documents the GIN path matches, without re-implementing - * stemming. - * - * Supported shapes are the ones `websearch_to_tsquery` produces: `&`, `|`, `!` applied to a - * lexeme, phrases of lexemes joined by `<->` or ``, and parentheses. TINQL has no standalone - * negation, so a query that is only negated, or negates inside a disjunction, has no translation - * and returns `null`; the caller then keeps the GIN path. - */ - -type Node = - | { kind: 'term'; lexeme: string } - | { kind: 'phrase'; terms: string[]; gaps: number[] } - | { kind: 'not'; operand: Node } - | { kind: 'and'; operands: Node[] } - | { kind: 'or'; operands: Node[] } - -type Token = - | { kind: 'lexeme'; value: string } - | { kind: 'and' | 'or' | 'not' | 'open' | 'close' } - | { kind: 'follow'; distance: number } - -class UntranslatableQuery extends Error {} - -function tokenize(text: string): Token[] { - const tokens: Token[] = [] - let index = 0 - while (index < text.length) { - const char = text[index] - if (char === ' ') { - index++ - } else if (char === "'") { - let value = '' - index++ - for (;;) { - if (index >= text.length) throw new UntranslatableQuery('unterminated lexeme') - if (text[index] === "'" && text[index + 1] === "'") { - value += "'" - index += 2 - } else if (text[index] === "'") { - index++ - break - } else { - value += text[index++] - } - } - /** Weight and prefix suffixes (`:A`, `:*`) change matching and are never emitted by websearch. */ - if (text[index] === ':') throw new UntranslatableQuery('lexeme modifiers') - tokens.push({ kind: 'lexeme', value }) - } else if (char === '&') { - tokens.push({ kind: 'and' }) - index++ - } else if (char === '|') { - tokens.push({ kind: 'or' }) - index++ - } else if (char === '!') { - tokens.push({ kind: 'not' }) - index++ - } else if (char === '(') { - tokens.push({ kind: 'open' }) - index++ - } else if (char === ')') { - tokens.push({ kind: 'close' }) - index++ - } else if (char === '<') { - const end = text.indexOf('>', index) - if (end < 0) throw new UntranslatableQuery('unterminated distance') - const body = text.slice(index + 1, end) - const distance = body === '-' ? 1 : Number(body) - if (!Number.isInteger(distance) || distance < 1) { - throw new UntranslatableQuery('unsupported distance') - } - tokens.push({ kind: 'follow', distance }) - index = end + 1 - } else { - throw new UntranslatableQuery(`unexpected character ${char}`) - } - } - return tokens -} - -/** Recursive descent over tsquery precedence: `|` binds loosest, then `&`, then ``, then `!`. */ -function parse(tokens: Token[]): Node { - let position = 0 - const peek = () => tokens[position] - - function parseOr(): Node { - const operands = [parseAnd()] - while (peek()?.kind === 'or') { - position++ - operands.push(parseAnd()) - } - return operands.length === 1 ? operands[0] : { kind: 'or', operands } - } - - function parseAnd(): Node { - const operands = [parseFollow()] - while (peek()?.kind === 'and') { - position++ - operands.push(parseFollow()) - } - return operands.length === 1 ? operands[0] : { kind: 'and', operands } - } - - function parseFollow(): Node { - const first = parseUnary() - if (peek()?.kind !== 'follow') return first - if (first.kind !== 'term') throw new UntranslatableQuery('phrase over a compound operand') - const terms = [first.lexeme] - const gaps: number[] = [] - for (let token = peek(); token?.kind === 'follow'; token = peek()) { - position++ - const next = parseUnary() - if (next.kind !== 'term') throw new UntranslatableQuery('phrase over a compound operand') - gaps.push(token.distance) - terms.push(next.lexeme) - } - return { kind: 'phrase', terms, gaps } - } - - function parseUnary(): Node { - const token = tokens[position++] - if (!token) throw new UntranslatableQuery('unexpected end') - if (token.kind === 'not') return { kind: 'not', operand: parseUnary() } - if (token.kind === 'lexeme') return { kind: 'term', lexeme: token.value } - if (token.kind === 'open') { - const inner = parseOr() - if (tokens[position++]?.kind !== 'close') throw new UntranslatableQuery('unbalanced group') - return inner - } - throw new UntranslatableQuery(`unexpected ${token.kind}`) - } - - const root = parseOr() - if (position !== tokens.length) throw new UntranslatableQuery('trailing input') - return root -} - -/** A lexeme is always quoted, so reserved words and punctuation stay literal terms. */ -function quote(lexeme: string): string { - return `"${lexeme.replace(/[\\"_[\]]/g, (char) => `\\${char}`)}"` -} - -function render(node: Node): string { - switch (node.kind) { - case 'term': - return quote(node.lexeme) - case 'phrase': - /** - * Adjacent lexemes form a phrase. A wider gap marks stopwords the analyzer removed; the - * indexed stream omits them too, so the gap becomes ordered proximity within that distance. - */ - if (node.gaps.every((gap) => gap === 1)) { - return `"${node.terms.map((term) => quote(term).slice(1, -1)).join(' ')}"` - } - return `(${node.terms - .map((term, index) => - index === 0 ? quote(term) : `THEN/${node.gaps[index - 1]} ${quote(term)}` - ) - .join(' ')})` - case 'not': - throw new UntranslatableQuery('negation outside a conjunction') - case 'and': { - const positive = node.operands.filter((operand) => operand.kind !== 'not') - const negative = node.operands.filter( - (operand): operand is Extract => operand.kind === 'not' - ) - if (positive.length === 0) throw new UntranslatableQuery('conjunction of negations only') - return `(${[ - positive.map(render).join(' AND '), - ...negative.map((operand) => `NOT ${render(operand.operand)}`), - ].join(' AND ')})` - } - case 'or': - return `(${node.operands.map(render).join(' OR ')})` - } -} - -/** - * The TINQL equivalent of a rendered `tsquery`, or `null` when the query has no lexemes or uses a - * shape TINQL cannot express. - */ -export function tinQueryFromTsquery(rendered: string): string | null { - const text = rendered.trim() - if (!text) return null - try { - return render(parse(tokenize(text))) - } catch (error) { - if (error instanceof UntranslatableQuery) return null - throw error - } -} diff --git a/apps/sim/lib/sim-search/indexed/retrieval/vector.ts b/apps/sim/lib/sim-search/indexed/retrieval/vector.ts deleted file mode 100644 index 9ccb72b1fd9..00000000000 --- a/apps/sim/lib/sim-search/indexed/retrieval/vector.ts +++ /dev/null @@ -1,491 +0,0 @@ -import { document, embeddingSearch } from '@sim/db/schema' -import { and, eq, inArray, type SQL, sql } from 'drizzle-orm' -import type { AnyPgColumn } from 'drizzle-orm/pg-core' -import { mapWithConcurrency } from '@/lib/core/utils/concurrency' -import { knowledgeAccessCondition, textArrayLiteral } from '@/lib/knowledge/access/predicate' -import type { UserAccessScope } from '@/lib/knowledge/access/types' -import { runSearchQuery, type SearchBudget } from '@/lib/knowledge/search/budget' -import { - excludeSearchSources, - getVisibilityConditions, - type SearchParams, - type SearchReadCandidate, - type SearchResult, - SOURCE_RANKING_CONCURRENCY, - selectAuthorizedSearchResults, -} from '@/lib/knowledge/search/candidates' -import { annotateSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' -import { searchDateFilterCondition } from '@/lib/knowledge/search/filter-conditions' -import { - annotateVectorPoolPlanned, - annotateVectorPoolSelected, - CANDIDATE_HNSW_MAX_SCAN_TUPLES, - gatheredVectorCandidatePool, - hydrateVectorCandidates, - prepareVectorLeg, - rankVectorCandidatesExactly, - readVectorCandidatePool, - selectExactVectorPage, - sliceVectorCandidatePool, - type VectorCandidatePool, - withVectorScanSettings, -} from '@/lib/knowledge/search/vector-leg' -import type { SearchAccessPlan } from '@/lib/sim-search/indexed/retrieval/access-plan' -import { - documentSatisfies, - type IndexedRetrievalContext, - PERMITTED_EXACT_DOCUMENT_LIMIT, -} from '@/lib/sim-search/indexed/retrieval/permitted' -import { - excludeSearchSourcesOnRow, - knowledgeCandidateAccessConditionForConnectors, - projectionCandidateAccessCondition, - projectionPending, -} from '@/lib/sim-search/indexed/retrieval/projection-access' -import { isProjectionFilled } from '@/lib/sim-search/indexed/retrieval/projection-fill' -import { indexedVectorSources } from '@/lib/sim-search/indexed/retrieval/source-vector-indexes' - -/** - * How far a walk that decides readability on the row may go before giving up: a cap, not a target, - * since the scan stops as soon as the limit is met. The default cap was sized for a walk that looked - * a document up per visited tuple; on the row a tuple costs a fraction of that, so a caller whose - * neighbourhood is mostly unreadable can be carried past it for tens of milliseconds rather than - * left with what the neighbourhood happened to hold. - */ -const ON_ROW_WALK_SCAN_TUPLES = 100_000 - -/** - * How far a walk may go when readability is on the row: the on-row cap, unless the walk still - * has to ask the document about tuples — a tag or date filter, or rows the source and ACL fill - * has not reached yet — in which case such a tuple costs what it did before the columns were mirrored, - * and the default cap keeps a walk through a mostly-excluded neighbourhood at a short answer - * rather than a missed deadline. - */ -function onRowWalkScanTuples( - documentCondition: SQL | undefined, - projectionFilled: boolean -): number { - return documentCondition === undefined && projectionFilled - ? ON_ROW_WALK_SCAN_TUPLES - : CANDIDATE_HNSW_MAX_SCAN_TUPLES -} - -/** - * A candidate's source: the row's, unless the row's document is marked for the projector, whose - * source may have moved since the row was written — then the document's, read for that row only. - */ -function projectionCandidateSource(projection: { - connectorId: AnyPgColumn | SQL - documentId: AnyPgColumn | SQL -}): SQL { - return sql`CASE WHEN ${projectionPending(projection.documentId)} - THEN (SELECT ${document.connectorId} FROM ${document} WHERE ${document.id} = ${projection.documentId}) - ELSE ${projection.connectorId} END` -} - -/** - * The same identities read off a projection row in raw SQL: the aliases are what - * `SearchReadCandidate` deserializes, so every walk reads them from one place. - */ -const PROJECTION_CANDIDATE_COLUMNS = sql`${embeddingSearch.id} AS id, ${embeddingSearch.documentId} AS "documentId", ${projectionCandidateSource(embeddingSearch)} AS "connectorId"` - -/** - * How many chunks the sliced sources contribute to exact ranking. A caller's slice of mirrored - * sources — their mail, their files, the spaces they belong to — sits below this, and ranking that - * many exactly, on the projection's half-precision vectors, measures in tens of milliseconds. - */ -const SOURCE_EXACT_CHUNK_LIMIT = 150_000 - -/** Sources whose own index a caller's ranking walks, and whether anything is left to rank exactly. */ -interface SourceVectorPlan { - walked: readonly string[] - sliced: readonly string[] -} - -/** - * How each readable source contributes its nearest chunks. - * - * Membership decides it, not a count: a member of a source reads essentially all of it, so its own - * index is walked and the graph's neighbours are chunks they can read. Every other source is - * sliced — mirrored permissions give a caller their own mail, their own files — and those slices - * are ranked exactly together, which is cheaper than a walk and exact by construction. A source - * the caller is a member of but which has no index of its own is sliced too. - */ -function planSourceVectorCandidates(input: { - plan: SearchAccessPlan - indexedSources: ReadonlySet -}): SourceVectorPlan { - const eligible = [ - ...new Set([ - ...input.plan.connectors.workspace, - ...input.plan.connectors.admin, - ...input.plan.connectors.members, - ]), - ] - const walked = input.plan.memberSources.filter((id) => input.indexedSources.has(id)) - const walking = new Set(walked) - return { walked, sliced: eligible.filter((id) => !walking.has(id)) } -} - -/** - * The nearest readable chunks, gathered per source and merged by distance. - * - * Walking one source at a time is what keeps recall: pgvector post-filters, so a walk over every - * source spends its scan budget on the sources this caller cannot read and returns few of their - * true neighbours. Inside one source they read, almost every neighbour qualifies. - * - * Nothing but the merged identities crosses the wire — each source's readable documents are - * resolved inside its own statement. - */ -async function selectSourceVectorCandidates(input: { - access: UserAccessScope - knowledgeBaseIds: string[] - plan: SearchAccessPlan - tagCondition: SQL | undefined - documentCondition: SQL | undefined - /** Sources the caller turned out not to hold, kept out of every source's ranking. */ - exclusion: SQL | undefined - /** Whether every projection row carries its mirrored columns, so a walk needs no document. */ - projectionFilled: boolean - candidateDistance: SQL - candidateLimit: number - budget?: SearchBudget -}): Promise { - const sources = planSourceVectorCandidates({ - plan: input.plan, - indexedSources: await indexedVectorSources(input.budget), - }) - annotateSearchDiagnostics({ - vectorRanking: 'per-source', - vectorSourcesWalked: sources.walked.length, - vectorSourcesSliced: sources.sliced.length, - }) - const base = and( - inArray(embeddingSearch.knowledgeBaseId, input.knowledgeBaseIds), - eq(embeddingSearch.enabled, true), - input.tagCondition, - input.exclusion - ) - type RankedChunks = Promise> - /** - * Walks one source's own index, or the sliced sources together when their slice saturated. - * Readability is decided on the row the walk visits — the source and ACL are mirrored there — - * so the graph is not stalled by a document lookup per candidate; the tag filter, which lives on - * the chunk, still joins. - */ - const onRow = projectionCandidateAccessCondition(embeddingSearch, input.access, input.plan, { - filled: input.projectionFilled, - }) - const walk = - (scope: SQL): (() => RankedChunks) => - () => - withVectorScanSettings( - (executor) => - executor.execute(sql` - SELECT ${PROJECTION_CANDIDATE_COLUMNS}, ${input.candidateDistance} AS distance - FROM ${embeddingSearch} /* on-row visibility */ - WHERE ${and( - base, - scope, - onRow, - documentSatisfies(embeddingSearch.documentId, input.documentCondition) - )} - ORDER BY ${input.candidateDistance} LIMIT ${input.candidateLimit}`), - input.budget, - 'vector.source_walk', - onRowWalkScanTuples(input.documentCondition, input.projectionFilled) - ) - const walks: Array<() => RankedChunks> = sources.walked.map((connectorId) => - walk(eq(embeddingSearch.connectorId, connectorId)) - ) - const slicedScope = sql`(${embeddingSearch.connectorId} IS NULL - OR ${embeddingSearch.connectorId} = ANY(${textArrayLiteral([...sources.sliced])}))` - /** - * One statement for the sliced sources and, with them, every uploaded document: uploads carry no - * connector, so a caller who is a member of all the indexed sources would otherwise rank none. - */ - const slice: Array<() => RankedChunks> = [ - async () => { - /** - * The sliced sources' readable chunks, decided on the row, ranked exactly: the ACL index - * enumerates them and `+ 0` keeps the planner off the graph. The chunks are counted one past - * the bound in the same statement, so a set too large to rank exactly is known before it is. - */ - const readableChunks = and( - base, - slicedScope, - onRow, - documentSatisfies(embeddingSearch.documentId, input.documentCondition) - ) - const rows = await runSearchQuery(input.budget, 'vector.source_exact', (executor) => - executor.execute(sql` - WITH readable_chunks AS MATERIALIZED ( - SELECT ${PROJECTION_CANDIDATE_COLUMNS}, ${input.candidateDistance} AS distance - FROM ${embeddingSearch} - WHERE ${readableChunks} - LIMIT ${SOURCE_EXACT_CHUNK_LIMIT + 1} - ) - SELECT id, "documentId", "connectorId", distance + 0 AS distance, - (SELECT count(*) FROM readable_chunks) > ${SOURCE_EXACT_CHUNK_LIMIT} AS saturated - FROM readable_chunks - ORDER BY distance LIMIT ${input.candidateLimit}`) - ) - /** - * The slice enumerates readable chunks in no particular order, so a set past its bound - * would rank an arbitrary subset and could miss the nearest chunks entirely. Walk those - * sources instead: approximate, but drawn from the whole of them. - */ - if (!rows.some((row) => row.saturated)) return rows - annotateSearchDiagnostics({ vectorSlicedSaturated: true }) - return walk(slicedScope)() - }, - ] - /** A source whose search runs out of budget marks the leg partial; the others' results stand. */ - const scored = await mapWithConcurrency( - [...walks, ...slice], - SOURCE_RANKING_CONCURRENCY, - async (run) => { - try { - return await run() - } catch (error) { - if (!input.budget?.isTimeout(error)) throw error - return [] - } - } - ) - const ranked: Array = scored.flat() - return ranked - .sort((a, b) => Number(a.distance) - Number(b.distance)) - .slice(0, input.candidateLimit) -} - -/** - * The vector leg of a user-scoped search-index search: a bounded candidate pool decided on the - * projection row through the caller's resolved plan, then hydrated under the full read predicate. - * - * A bounded permitted set is ranked exactly; a reader who is a member of indexed sources has each - * walked on its own; a broad reader walks the whole graph once. Live source authorization and - * content hydration still run after candidate ranking. - */ -export async function selectIndexedVectorResults( - params: SearchParams, - context: IndexedRetrievalContext -): Promise { - const setup = prepareVectorLeg(params) - const { access, accessPlan: plan, permitted } = context - /** Only live-verified readers may defer source authorization until after candidate ranking. */ - const candidateAccess = knowledgeCandidateAccessConditionForConnectors(access, plan) - /** - * What an on-row walk still has to ask the document: the tags, which live on chunks, and the - * date filter, which the row does not carry. A bounded set never walks, so this only runs when - * the filtered documents were too many to enumerate. - */ - const dateCondition = searchDateFilterCondition(params.filters) - const documentCondition = - setup.documentTagCondition || dateCondition - ? and(setup.documentTagCondition, dateCondition) - : undefined - /** `filled`: whether the pool's rows carry their source, so a page needs no read of its own. */ - let candidatePool: (VectorCandidatePool & { filled: boolean }) | undefined - return selectAuthorizedSearchResults({ - leg: 'vector', - access: params.access, - liveSourceAccess: context.liveSourceAccess, - signal: params.signal, - budget: params.budget, - topK: params.topK, - compareResults: (a, b) => a.distance - b.distance, - selectPage: async (limit, offset, excludedSources) => { - if (params.filters?.documentIds?.length) { - return selectExactVectorPage( - setup, - params.budget, - [ - ...getVisibilityConditions(params.filters, candidateAccess), - excludeSearchSources(excludedSources), - ], - limit, - offset - ) - } - if (permitted?.kind === 'bounded' && permitted.documents.length === 0) - return { candidates: [], nextOffset: offset } - candidatePool = await readVectorCandidatePool( - candidatePool, - excludedSources, - offset, - limit, - async ({ excludedKey, candidateLimit, previous: previousPool }) => { - /** Two remembered facts, read together when neither is remembered. */ - const [filled, plannedIndexedSources] = await Promise.all([ - isProjectionFilled('embedding_search', 'vector.projection_filled', params.budget), - plan.memberSources.length ? indexedVectorSources(params.budget) : undefined, - ]) - /** - * A source the caller turned out not to hold is left out where the pool is built: the - * pool is the page's order now, so a denied source's chunks would otherwise keep their - * slots. The row's mirrored source decides it, unless the row is decided on its document. - */ - const excludedOnRow = excludeSearchSourcesOnRow(embeddingSearch, filled, excludedSources) - annotateVectorPoolPlanned(setup, candidateLimit) - /** - * Exact ranking reads what the permitted set costs rather than re-deriving permission - * across the whole index, and honours `statement_timeout`, which a traversal cannot. - * `read` are chunks a pool already holds, ranked past. - */ - const rankPermittedExactly = (documentIds: string[], read?: readonly string[]) => - rankVectorCandidatesExactly({ - setup, - knowledgeBaseIds: params.knowledgeBaseIds, - documentIds, - columns: PROJECTION_CANDIDATE_COLUMNS, - conditions: [ - read?.length - ? sql`NOT (${embeddingSearch.id} = ANY(${textArrayLiteral([...read])}))` - : undefined, - excludedOnRow, - ], - candidateLimit, - budget: params.budget, - }) - let selected: SearchReadCandidate[] - /** Set where a pool's end is known better than by its length. */ - let exhausted: boolean | undefined - const walkGraph = () => - withVectorScanSettings( - (executor) => - executor.execute(sql` - SELECT ${PROJECTION_CANDIDATE_COLUMNS} - FROM ${embeddingSearch} /* on-row visibility */ - WHERE ${and( - inArray(embeddingSearch.knowledgeBaseId, params.knowledgeBaseIds), - eq(embeddingSearch.enabled, true), - excludedOnRow, - projectionCandidateAccessCondition(embeddingSearch, access, plan, { filled }), - documentSatisfies(embeddingSearch.documentId, documentCondition) - )} - ORDER BY ${setup.distance} LIMIT ${candidateLimit} - `), - params.budget, - 'vector.candidate_search', - onRowWalkScanTuples(documentCondition, filled) - ) - /** - * A source the caller is a member of that has its own index is walked on its own, which - * beats ranking it exactly once it is large enough to have earned that index. - */ - const walksASource = plan.memberSources.some( - (id) => plannedIndexedSources?.has(id) ?? false - ) - if ( - permitted?.kind === 'bounded' && - filled && - permitted.documents.length >= PERMITTED_EXACT_DOCUMENT_LIMIT - ) { - /** - * A set this large costs more to rank exactly than to walk: exact ranking reads every - * chunk of every document in it, while the walk decides readability on the rows it - * visits and stops at its tuple cap. The walk answers whenever the set is a fair share - * of the graph; where it is not, the walk underfills and the exact ranking that was - * always complete takes over, so nothing is lost but the walk's bounded cost. - * - * The walk decides readability on the projection row, which is broader than the - * document predicate hydration applies, so a pool it filled can still run short of - * readable rows. That shortfall is what refills a pool: the refill is the exact ranking, - * complete over the set, ranked past the rows already read and placed behind them, so - * the pages keep their offsets and every refill is a full window of fresh rows. - */ - const permittedIds = permitted.documents.map((entry) => entry.id) - const previous = previousPool?.ids - if (previous) { - const exact = await rankPermittedExactly( - permittedIds, - previous.map((candidate) => candidate.id) - ) - selected = [...previous, ...exact] - exhausted = exact.length < candidateLimit - } else { - selected = await walkGraph() - if (selected.length < candidateLimit) - selected = await rankPermittedExactly(permittedIds) - } - } else if (permitted?.kind === 'bounded' && (!walksASource || context.filtered)) { - /** - * A bounded permitted set is ranked exactly without walking the graph first: the walk - * post-filters, so when the caller reads a small share of the index it spends its whole - * uninterruptible tuple budget and still returns almost none of their neighbours. A - * member's indexed source is otherwise walked instead, but not under a filter: the walk - * cannot see the date, and a filtered set is small by construction. - */ - selected = await rankPermittedExactly(permitted.documents.map((entry) => entry.id)) - } else if (!(permitted?.kind === 'unbounded' && permitted.broad)) { - /** - * Readability follows sources, so each readable source is searched in its own index and - * the results merged. A member reads a source whole or barely at all: walking one source - * spends its budget among chunks they can read, where a walk over every source spends it - * on the sources they cannot. A caller whose reach is broad skips this: for them the - * whole graph's neighbours are mostly theirs already, and one walk is the cheaper answer. - */ - selected = await selectSourceVectorCandidates({ - access, - knowledgeBaseIds: params.knowledgeBaseIds, - plan, - exclusion: excludedOnRow, - projectionFilled: filled, - tagCondition: setup.candidateTagCondition, - documentCondition, - candidateDistance: setup.distance, - candidateLimit, - budget: params.budget, - }) - } else { - /** - * A broad reader's nearest chunks are mostly theirs, so one walk over the whole graph, - * deciding readability on the row, is the cheapest exact answer there is. - */ - selected = await walkGraph() - } - const pool = { - ...gatheredVectorCandidatePool(excludedKey, selected, candidateLimit, exhausted), - filled, - } - annotateVectorPoolSelected(selected.length, candidateLimit) - return pool - } - ) - /** - * The walk carries each candidate's document and source, so a page is a slice of the pool. - * Rescoring the pool against the original vectors here read one out-of-line vector per - * candidate from storage no cache holds, seconds on a query nobody had run before; the page - * is scored at hydration instead. - */ - if (candidatePool.filled) return sliceVectorCandidatePool(candidatePool, offset, limit) - /** - * While the source and ACL fill runs, a row it has not reached carries no source, so the - * page's identities are read off the documents; a slice whose documents all went away since - * the walk is passed over, not mistaken for the pool's end. - */ - for (let start = offset; start < candidatePool.ids.length; start += limit) { - const slice = candidatePool.ids.slice(start, start + limit) - const ranked = new Map(slice.map((candidate, index) => [candidate.id, index])) - const identities = await runSearchQuery(params.budget, 'vector.page', (executor) => - executor.execute(sql` - SELECT ${embeddingSearch.id} AS id, ${document.id} AS "documentId", - ${document.connectorId} AS "connectorId" - FROM ${embeddingSearch} - INNER JOIN ${document} ON ${document.id} = ${embeddingSearch.documentId} - WHERE ${embeddingSearch.id} = ANY(${textArrayLiteral(slice.map((candidate) => candidate.id))}) - `) - ) - if (!identities.length) continue - const page = [...identities].sort( - (a, b) => (ranked.get(a.id) ?? 0) - (ranked.get(b.id) ?? 0) - ) - return { candidates: page, nextOffset: start + slice.length } - } - return { candidates: [], nextOffset: candidatePool.ids.length } - }, - hydrate: (ids, authorized) => - hydrateVectorCandidates(ids, knowledgeAccessCondition(authorized), setup, params), - }) -} diff --git a/apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts b/apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts deleted file mode 100644 index 612f5617545..00000000000 --- a/apps/sim/lib/sim-search/indexed/search/scoped-search.activity.test.ts +++ /dev/null @@ -1,135 +0,0 @@ -import { member } from '@sim/db/schema' -import { queueTableRows, resetDbChainMock } from '@sim/testing' -import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { - knowledgeAvailabilityMock, - knowledgeAvailabilityMockFns, -} from '@sim/testing/mocks/knowledge-availability.mock' -import { - knowledgeContextsMock, - knowledgeContextsMockFns, -} from '@sim/testing/mocks/knowledge-contexts.mock' -import { - knowledgeSearchUseCaseMock, - knowledgeSearchUseCaseMockFns, -} from '@sim/testing/mocks/knowledge-search-use-case.mock' -import { - permissionGroupsResolveMock, - permissionGroupsResolveMockFns, -} from '@sim/testing/mocks/permission-groups-resolve.mock' -import { workspaceAuthzMock } from '@sim/testing/mocks/workspace-authz.mock' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -const hoisted = vi.hoisted(() => ({ - findIndex: vi.fn(), - activity: vi.fn(), -})) -vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) -vi.mock('@/lib/permission-groups/resolve.server', () => permissionGroupsResolveMock) -vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) -vi.mock('@/lib/knowledge/search/search-index', () => ({ - findSearchIndex: hoisted.findIndex, -})) -vi.mock('@/lib/knowledge/access/availability', () => knowledgeAvailabilityMock) -vi.mock('@/lib/knowledge/search/activity', () => ({ - recordOrganizationSearchActivity: hoisted.activity, -})) -vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) - -import { - searchOrganizationKnowledge, - searchScopedKnowledge, -} from '@/lib/sim-search/indexed/search/scoped-search' - -const mocks = { - ...hoisted, - afterSearch: knowledgeSearchUseCaseMockFns.mockAfterKnowledgeSearch, - search: knowledgeSearchUseCaseMockFns.mockRunKnowledgeSearch, -} -mocks.afterSearch.mockImplementation(async () => undefined) - -knowledgeContextsMockFns.mockResolveKnowledgeOwnerContext.mockImplementation((...args: unknown[]) => - knowledgeContextsMockFns.mockResolveKnowledgeOrganizationContext(...args) -) -knowledgeContextsMockFns.mockResolveKnowledgeWorkspaceContext.mockImplementation( - (...args: unknown[]) => knowledgeContextsMockFns.mockResolveKnowledgeOrganizationContext(...args) -) - -const principal = createSessionPrincipal({ userId: 'reader', sessionId: 'session' }) -const input = { organizationId: 'org', query: 'policy', topK: 20, surface: 'slack' } as const - -/** These use cases run only while indexed organization search is on. */ -beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) -afterEach(resetEnvFlagsMock) - -beforeEach(() => { - resetDbChainMock() - knowledgeContextsMockFns.mockResolveKnowledgeOrganizationContext.mockResolvedValue({ - organizationId: 'org', - }) - permissionGroupsResolveMockFns.mockGetUserPermissionConfigForOrganization.mockResolvedValue(null) - mocks.findIndex.mockResolvedValue(null) - knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable.mockResolvedValue(undefined) - mocks.activity.mockResolvedValue(undefined) - mocks.search.mockResolvedValue({ - results: [], - knowledgeBases: [{ id: 'index' }], - knowledgeBaseId: 'index', - }) -}) - -describe.each([ - { name: 'organization Assistant', operation: searchOrganizationKnowledge }, - { name: 'scoped Search', operation: searchScopedKnowledge }, -])('$name activity before an index exists', ({ operation }) => { - it('records an authorized empty invocation for the acting member', async () => { - queueTableRows(member, [{ role: 'member' }]) - expect(await operation.execute({ principal, input })).toEqual({ - results: [], - retrieval: { status: 'complete', timedOutLegs: [] }, - query: 'policy', - knowledgeBases: [], - }) - expect( - knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable - ).toHaveBeenCalledExactlyOnceWith('org') - expect(mocks.activity).toHaveBeenCalledExactlyOnceWith({ - organizationId: 'org', - userId: 'reader', - surface: 'slack', - results: [], - }) - expect(mocks.search).not.toHaveBeenCalled() - }) - - it('does not meter an unavailable Search request', async () => { - queueTableRows(member, [{ role: 'member' }]) - knowledgeAvailabilityMockFns.mockRequireOrganizationSearchAvailable.mockRejectedValueOnce( - new Error('Search is disabled') - ) - await expect(operation.execute({ principal, input })).rejects.toThrow('Search is disabled') - expect(mocks.activity).not.toHaveBeenCalled() - expect(mocks.search).not.toHaveBeenCalled() - }) - - it('does not discover the index or meter a nonmember request', async () => { - queueTableRows(member, []) - await expect(operation.execute({ principal, input })).rejects.toMatchObject({ - code: 'not_found', - }) - expect(mocks.findIndex).not.toHaveBeenCalled() - expect(mocks.activity).not.toHaveBeenCalled() - }) - - it('does not meter a request that was already cancelled', async () => { - queueTableRows(member, [{ role: 'member' }]) - const controller = new AbortController() - controller.abort(new Error('Search cancelled')) - await expect( - operation.execute({ principal, input: { ...input, signal: controller.signal } }) - ).rejects.toThrow('Search cancelled') - expect(mocks.activity).not.toHaveBeenCalled() - expect(mocks.search).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts b/apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts deleted file mode 100644 index 70891390be9..00000000000 --- a/apps/sim/lib/sim-search/indexed/search/scoped-search.test.ts +++ /dev/null @@ -1,100 +0,0 @@ -import { - dbChainMockFns, - hasMockCondition, - queueTableRows, - resetDbChainMock, - schemaMock, -} from '@sim/testing' -import { createSessionPrincipal } from '@sim/testing/factories/principal.factory' -import { resetEnvFlagsMock, setEnvFlags } from '@sim/testing/mocks/env-flags.mock' -import { - knowledgeContextsMock, - knowledgeContextsMockFns, -} from '@sim/testing/mocks/knowledge-contexts.mock' -import { - knowledgeSearchUseCaseMock, - knowledgeSearchUseCaseMockFns, -} from '@sim/testing/mocks/knowledge-search-use-case.mock' -import { workspaceAuthzMock, workspaceAuthzMockFns } from '@sim/testing/mocks/workspace-authz.mock' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' - -vi.mock('@sim/platform-authz/workspace', () => workspaceAuthzMock) -vi.mock('@/lib/knowledge/application/contexts', () => knowledgeContextsMock) -vi.mock('@/lib/knowledge/application/search', () => knowledgeSearchUseCaseMock) - -import { SearchIndexDormantError } from '@/lib/sim-search/indexed/gate' -import { searchWorkspaceKnowledge } from '@/lib/sim-search/indexed/search/scoped-search' - -const mocks = { - afterSearch: knowledgeSearchUseCaseMockFns.mockAfterKnowledgeSearch, - search: knowledgeSearchUseCaseMockFns.mockRunKnowledgeSearch, -} -mocks.afterSearch.mockImplementation(async () => undefined) - -workspaceAuthzMockFns.mockPermissionSatisfies.mockImplementation( - (actual: string | null) => actual !== null -) - -const principal = createSessionPrincipal({ userId: 'reader', sessionId: 'session' }) -const input = { workspaceId: 'workspace', query: 'orion', topK: 20, filters: { source: 'slack' } } - -/** These use cases run only while indexed organization search is on. */ -beforeEach(() => setEnvFlags({ isLiveEnterpriseSearchEnabled: false })) -afterEach(resetEnvFlagsMock) - -describe('canonical workspace search', () => { - beforeEach(() => { - resetDbChainMock() - knowledgeContextsMockFns.mockResolveKnowledgeWorkspaceContext.mockResolvedValue({ - workspaceId: 'workspace', - workspaceOrganizationId: null, - allowPersonalApiKeys: true, - billedAccountUserId: 'payer', - }) - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue('read') - mocks.search.mockResolvedValue({ - results: [], - knowledgeBases: [{ id: 'index', name: 'Enterprise Search' }], - }) - }) - it('authorizes the person before selecting the canonical active index and passes the same principal and filters', async () => { - queueTableRows(schemaMock.knowledgeBase, [{ id: 'index' }]) - await searchWorkspaceKnowledge.execute({ principal, input }) - /** The search runs under the context this use case resolved; the index is its one base. */ - expect(mocks.search).toHaveBeenCalledWith( - expect.objectContaining({ - principal, - input: { ...input, knowledgeBaseIds: ['index'] }, - context: expect.objectContaining({ - workspaceId: 'workspace', - knowledgeBases: [expect.objectContaining({ id: 'index' })], - }), - }) - ) - expect( - hasMockCondition( - dbChainMockFns.where.mock.calls[0][0], - (node) => - node.type === 'eq' && - node.left === schemaMock.knowledgeBase.isSearchIndex && - node.right === true - ) - ).toBe(true) - expect(dbChainMockFns.limit).toHaveBeenCalledWith(1) - }) - it('refuses on its own while indexed organization search is dormant', async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) - await expect(searchWorkspaceKnowledge.execute({ principal, input })).rejects.toBeInstanceOf( - SearchIndexDormantError - ) - expect(knowledgeContextsMockFns.mockResolveKnowledgeWorkspaceContext).not.toHaveBeenCalled() - expect(dbChainMockFns.where).not.toHaveBeenCalled() - }) - it('refuses a nonmember before querying the protected index', async () => { - workspaceAuthzMockFns.mockResolveEffectiveWorkspacePermission.mockResolvedValue(null) - await expect(searchWorkspaceKnowledge.execute({ principal, input })).rejects.toThrow( - 'Insufficient workspace' - ) - expect(dbChainMockFns.where).not.toHaveBeenCalled() - }) -}) diff --git a/apps/sim/lib/sim-search/indexed/search/scoped-search.ts b/apps/sim/lib/sim-search/indexed/search/scoped-search.ts deleted file mode 100644 index 63c038f4c75..00000000000 --- a/apps/sim/lib/sim-search/indexed/search/scoped-search.ts +++ /dev/null @@ -1,184 +0,0 @@ -import type { Principal } from '@sim/auth/principal' -import { resolvePrincipalSubjectUserId } from '@sim/auth/principal' -import { OrchestrationError } from '@/lib/core/orchestration/types' -import { type ResourceOwner, resourceScopeFromOwner } from '@/lib/core/resource-scope' -import { requireOrganizationSearchAvailable } from '@/lib/knowledge/access/availability' -import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' -import { - type KnowledgeResourceContext, - resolveKnowledgeOrganizationContext, - resolveKnowledgeOwnerContext, - resolveKnowledgeWorkspaceContext, -} from '@/lib/knowledge/application/contexts' -import { knowledgeOperations } from '@/lib/knowledge/application/operations' -import { - afterKnowledgeSearch, - buildKnowledgeSearchContext, - runKnowledgeSearch, - type SearchKnowledgeInput, - type SearchKnowledgeResult, - validateKnowledgeSearchInput, -} from '@/lib/knowledge/application/search' -import { instrumentSearchUseCase } from '@/lib/knowledge/application/search-diagnostics' -import type { ActiveKnowledgeBaseReference } from '@/lib/knowledge/knowledge-base-reference' -import { recordOrganizationSearchActivity } from '@/lib/knowledge/search/activity' -import { measureSearchStage } from '@/lib/knowledge/search/diagnostics' -import { findSearchIndex } from '@/lib/knowledge/search/search-index' -import { assertIndexedOrgSearchEnabled } from '@/lib/sim-search/indexed/gate' - -export type SearchWorkspaceKnowledgeInput = Omit< - SearchKnowledgeInput, - 'knowledgeBaseIds' | 'workspaceId' -> & { - workspaceId: string -} - -export type SearchOrganizationKnowledgeInput = Omit< - SearchWorkspaceKnowledgeInput, - 'workspaceId' -> & { organizationId: string } - -export type SearchScopedKnowledgeInput = Omit< - SearchKnowledgeInput, - 'knowledgeBaseIds' | 'workspaceId' | 'organizationId' -> & - ResourceOwner - -/** What an owner without an index answers: nothing, completely. */ -interface SearchWithoutIndex { - results: [] - query: string - knowledgeBases: [] - retrieval: { status: 'complete'; timedOutLegs: [] } -} - -type ScopedSearchResult = SearchKnowledgeResult | SearchWithoutIndex - -/** Whether an index was searched, which is what the follow-up to a search is for. */ -function searchedAnIndex(result: ScopedSearchResult): result is SearchKnowledgeResult { - return 'knowledgeBaseId' in result -} - -/** - * An owner without an index answers empty. An organization still has to be allowed to search, - * and its empty search is recorded like any other, so the activity view shows the attempt. - */ -async function searchWithoutIndex( - principal: Principal, - context: KnowledgeResourceContext, - input: Pick -): Promise { - if (context.organizationId) { - await requireOrganizationSearchAvailable(context.organizationId) - input.signal?.throwIfAborted() - const userId = resolvePrincipalSubjectUserId(principal) - if (userId) - await recordOrganizationSearchActivity({ - organizationId: context.organizationId, - userId, - surface: input.surface ?? 'other', - results: [], - }) - } - return { - results: [], - query: input.query ?? '', - knowledgeBases: [], - retrieval: { status: 'complete', timedOutLegs: [] }, - } -} - -type ScopedSearchInput = Omit - -/** - * A search surface that resolves an owner, finds the owner's index and searches it. The owner is - * resolved and authorized once, here; the search runs under that context, and nothing about the - * index is read twice. Surfaces differ only in how they name the owner and find the index. - */ -function defineScopedSearchUseCase< - I extends Pick, ->(surface: { - resolveContext: (input: I) => Promise - findIndex: (context: KnowledgeResourceContext) => Promise - searchInput: (input: I, context: KnowledgeResourceContext) => ScopedSearchInput -}) { - return defineAuthorizedKnowledgeUseCase({ - operation: knowledgeOperations.search, - resolveContext: ({ input }: { input: I }) => { - assertIndexedOrgSearchEnabled() - return measureSearchStage('scope_resolution', () => surface.resolveContext(input)) - }, - async execute({ principal, input, context }): Promise { - input.signal?.throwIfAborted() - if ( - !input.query?.trim() || - input.filters?.startDate || - input.filters?.endDate || - input.filters?.sortBy - ) - throw new OrchestrationError( - 'validation', - 'Date-only search, startDate/endDate and sorting require live search. Use a text query and modification filters for indexed search.' - ) - const index = await measureSearchStage('index_resolution', () => surface.findIndex(context)) - if (!index) return searchWithoutIndex(principal, context, input) - const searchInput: SearchKnowledgeInput = { - ...surface.searchInput(input, context), - knowledgeBaseIds: [index.id], - } - /** An owner the request asserts is the one that was resolved, or the request names none. */ - if ( - (searchInput.organizationId && searchInput.organizationId !== context.organizationId) || - (searchInput.workspaceId && searchInput.workspaceId !== context.workspaceId) - ) { - throw new OrchestrationError('not_found', 'Knowledge base not found') - } - validateKnowledgeSearchInput(searchInput) - return runKnowledgeSearch({ - principal, - input: searchInput, - context: buildKnowledgeSearchContext(principal, context, [index], searchInput), - }) - }, - afterSuccess: ({ principal, context, input, result }) => - searchedAnIndex(result) - ? afterKnowledgeSearch({ principal, context, input, result }) - : undefined, - }) -} - -/** Search and Assistant share the workspace's canonical Enterprise Search index. */ -export const searchWorkspaceKnowledge = instrumentSearchUseCase( - 'workspace_application', - defineScopedSearchUseCase({ - resolveContext: (input) => resolveKnowledgeWorkspaceContext(input), - findIndex: (context) => - findSearchIndex({ kind: 'workspace', workspaceId: context.workspaceId! }), - searchInput: (input, context) => ({ ...input, workspaceId: context.workspaceId }), - }) -) - -/** Organization Search and Assistant resolve the same index and provider ACLs. */ -export const searchOrganizationKnowledge = instrumentSearchUseCase( - 'organization_application', - defineScopedSearchUseCase({ - resolveContext: (input) => resolveKnowledgeOrganizationContext(input), - findIndex: (context) => - findSearchIndex({ kind: 'organization', organizationId: context.organizationId! }), - searchInput: (input) => input, - }) -) - -/** The routed owner selects the index; current membership and provider ACLs select its documents. */ -export const searchScopedKnowledge = instrumentSearchUseCase( - 'scoped_application', - defineScopedSearchUseCase({ - resolveContext: (input) => resolveKnowledgeOwnerContext(input), - findIndex: (context) => findSearchIndex(resourceScopeFromOwner(context)), - searchInput: (input) => ({ - ...input, - workspaceId: input.workspaceId ?? undefined, - organizationId: input.organizationId ?? undefined, - }), - }) -) diff --git a/apps/sim/lib/sim-search/live/README.md b/apps/sim/lib/sim-search/live/README.md index ccde37b1e19..4e90300c998 100644 --- a/apps/sim/lib/sim-search/live/README.md +++ b/apps/sim/lib/sim-search/live/README.md @@ -1,6 +1,6 @@ # Federated Search access and connector behavior -This describes the live enterprise-search path. Credential Groups and ordinary knowledge-base indexing retain their existing behavior. Live Search is enabled by default; only an explicit `SIM_SEARCH_LIVE=false` selects the legacy indexed backend, which is kept dormant in `../indexed/` (see its README). Live Search sources do not create content-indexing jobs, and queued content or persisted-directory jobs stop before crawling, embedding, or building ACL snapshots. Ordinary KB jobs remain enabled. Administrators can still maintain GitLab CSV grants, and request-time source permission checks remain required. +This describes the live enterprise-search path. Credential Groups and ordinary knowledge-base indexing retain their existing behavior. Enterprise Search uses this live backend. The legacy indexed backend and its runtime toggle have been removed. Live Search sources do not create content-indexing jobs, and queued content or persisted-directory jobs stop before crawling, embedding, or building ACL snapshots. Ordinary KB jobs remain enabled. Administrators can still maintain GitLab CSV grants, and request-time source permission checks remain required. ## Admin and member surfaces diff --git a/apps/sim/lib/sim-search/live/application.ts b/apps/sim/lib/sim-search/live/application.ts index 5e61b8cd208..08304f4748b 100644 --- a/apps/sim/lib/sim-search/live/application.ts +++ b/apps/sim/lib/sim-search/live/application.ts @@ -18,7 +18,6 @@ import { } from '@/lib/api/contracts/mothership-assistant-tools' import { canonicalJson, fingerprint, instantScopePart } from '@/lib/api/cursor-binding' import { env } from '@/lib/core/config/env' -import { isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' import { OrchestrationError } from '@/lib/core/orchestration/types' import { type ResourceOwner, @@ -160,10 +159,6 @@ interface LiveSearchOptions { } export type LiveSearchInput = ResourceOwner & LiveSearchOptions -function requireLiveSearch() { - if (!isLiveEnterpriseSearchEnabled) - throw new OrchestrationError('not_found', 'Live search is not enabled') -} function safeContent(content: string, registry?: ResolvedSecretTraceRegistry): string { if (!registry) return content const projected = projectResolvedSecretModelContent(content, registry) @@ -345,7 +340,6 @@ export const searchLiveKnowledge = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.search, resolveContext: ({ input }: { input: LiveSearchInput }) => resolveKnowledgeOwnerContext(input), async execute({ principal, input }): Promise { - requireLiveSearch() const userId = requirePrincipalSubjectUserId(principal) if (input.organizationId) await requireOrganizationSearchAvailable(input.organizationId) input.signal?.throwIfAborted() @@ -740,7 +734,6 @@ export const readLiveDocument = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.readDocument, resolveContext: ({ input }: { input: LiveReadInput }) => resolveKnowledgeOwnerContext(input), async execute({ principal, input }) { - requireLiveSearch() const userId = requirePrincipalSubjectUserId(principal) if (input.organizationId) await requireOrganizationSearchAvailable(input.organizationId) if (!Number.isInteger(input.limit) || input.limit < 1 || input.limit > 8) @@ -845,7 +838,6 @@ export const listLiveSearchAccounts = defineAuthorizedKnowledgeUseCase({ operation: knowledgeOperations.listPersonalSearchIntegrations, resolveContext: ({ input }: { input: ResourceOwner }) => resolveKnowledgeOwnerContext(input), async execute({ principal, input }) { - requireLiveSearch() if (input.organizationId) await requireOrganizationSearchAvailable(input.organizationId) const accounts = await listLiveAccounts(input, requirePrincipalSubjectUserId(principal)) return { diff --git a/apps/sim/lib/sim-search/live/scopes.ts b/apps/sim/lib/sim-search/live/scopes.ts index a1d88b949b8..4de22a12977 100644 --- a/apps/sim/lib/sim-search/live/scopes.ts +++ b/apps/sim/lib/sim-search/live/scopes.ts @@ -1,4 +1,4 @@ -/** User-token RTS permissions. Existing grant policies remain valid when switching back to indexed search. */ +/** User-token permissions required by Slack's live retrieval API. */ export const SLACK_RTS_USER_SCOPES = [ 'search:read.files', 'files:read', diff --git a/apps/sim/lib/sim-search/personal-source-setup.ts b/apps/sim/lib/sim-search/personal-source-setup.ts deleted file mode 100644 index 7155fd26ed9..00000000000 --- a/apps/sim/lib/sim-search/personal-source-setup.ts +++ /dev/null @@ -1,2 +0,0 @@ -/** Personal Atlassian setup bounds explicitly selected project or space keys. */ -export const MAX_PERSONAL_SOURCE_SETUP_KEYS = 1000 diff --git a/apps/sim/lib/slack-search/connections.test.ts b/apps/sim/lib/slack-search/connections.test.ts index 3b2ff89c94b..c07699c071d 100644 --- a/apps/sim/lib/slack-search/connections.test.ts +++ b/apps/sim/lib/slack-search/connections.test.ts @@ -16,7 +16,8 @@ const target = { type: 'link', provider: 'google-email', connectorType: 'gmail', - connectorId: 'source', + connectionMode: 'live', + optionId: 'gmail-option', } as const const input = { targets: [target], diff --git a/apps/sim/lib/workspaces/__integration__/fork-sync.integration.ts b/apps/sim/lib/workspaces/__integration__/fork-sync.integration.ts index 619a4068957..bfd53dbe599 100644 --- a/apps/sim/lib/workspaces/__integration__/fork-sync.integration.ts +++ b/apps/sim/lib/workspaces/__integration__/fork-sync.integration.ts @@ -2,7 +2,10 @@ import { AuditAction } from '@sim/audit' import { db } from '@sim/db' import { auditLog, + document, + embedding, folder, + knowledgeBase, outboxEvent, permissions, user, @@ -37,7 +40,13 @@ import { } from '@/ee/workspace-forking/application/create-and-sync' import { assertForkSourceVersions } from '@/ee/workspace-forking/application/revision' import { setForkSyncDefault } from '@/ee/workspace-forking/application/sync-default' +import { + copyForkResourceContainers, + copyForkResourceContent, + planForkMappedKbDocumentCopies, +} from '@/ee/workspace-forking/lib/copy/copy-resources' import { loadSourceDeployedStates } from '@/ee/workspace-forking/lib/copy/deploy-bridge' +import type { ForkCopyProgress } from '@/ee/workspace-forking/lib/copy/progress' import type { WorkflowState } from '@/stores/workflows/workflow/types' const userId = generateId() @@ -917,4 +926,198 @@ describe('authorized fork and sync against PostgreSQL', () => { await setDefault(sourceWorkspaceId, false) } }) + async function seedKnowledgeCopy() { + const childWorkspaceId = generateId() + const sourceWorkspaceId = generateId() + createdWorkspaceIds.push(sourceWorkspaceId, childWorkspaceId) + await db.insert(workspace).values( + [sourceWorkspaceId, childWorkspaceId].map((id) => ({ + id, + name: 'Knowledge copy fixture', + ownerId: userId, + billedAccountUserId: userId, + })) + ) + const sourceId = generateId() + const childId = generateId() + await db.insert(knowledgeBase).values([ + { id: sourceId, workspaceId: sourceWorkspaceId, userId, name: `Source ${sourceId}` }, + { id: childId, workspaceId: childWorkspaceId, userId, name: 'Target fixture' }, + ]) + const [source] = await db + .insert(document) + .values({ + id: generateId(), + knowledgeBaseId: sourceId, + filename: 'Copy fixture', + fileUrl: '', + fileSize: 0, + mimeType: 'text/plain', + processingStatus: 'completed', + }) + .returning() + return { sourceWorkspaceId, childWorkspaceId, sourceId, childId, source } + } + + it('copies ordinary knowledge containers but excludes retired Search containers', async () => { + const fixture = await seedKnowledgeCopy() + const retiredId = generateId() + await db.insert(knowledgeBase).values({ + id: retiredId, + workspaceId: fixture.sourceWorkspaceId, + userId, + name: 'Retired Search fixture', + isSearchIndex: true, + }) + const copied = await db.transaction((tx) => + copyForkResourceContainers({ + tx, + sourceWorkspaceId: fixture.sourceWorkspaceId, + childWorkspaceId: fixture.childWorkspaceId, + userId, + now: new Date(), + selection: { + customTools: [], + skills: [], + mcpServers: [], + workflowMcpServers: [], + tables: [], + knowledgeBases: [fixture.sourceId, retiredId], + }, + workflowIdMap: new Map(), + documentMappingContext: { + edgeChildWorkspaceId: fixture.childWorkspaceId, + sourceIsParent: true, + }, + }) + ) + expect(copied.contentPlan.knowledgeBases.map((entry) => entry.sourceId)).toEqual([ + fixture.sourceId, + ]) + expect( + await db + .select() + .from(knowledgeBase) + .where( + and( + eq(knowledgeBase.workspaceId, fixture.childWorkspaceId), + eq(knowledgeBase.isSearchIndex, true) + ) + ) + ).toEqual([]) + }) + + it.each(['source', 'target'] as const)( + 'does not plan document copies for a retired Search %s', + async (retiredSide) => { + const fixture = await seedKnowledgeCopy() + const retiredId = retiredSide === 'source' ? fixture.sourceId : fixture.childId + await db + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where(eq(knowledgeBase.id, retiredId)) + const plan = () => + db.transaction((tx) => + planForkMappedKbDocumentCopies({ + tx, + resolver: (kind, id) => + kind === 'knowledge-base' && id === fixture.sourceId ? fixture.childId : null, + referencedDocumentIds: [fixture.source.id], + alreadyCopiedSourceDocIds: new Set(), + now: new Date(), + }) + ) + const refused = await plan() + expect(refused.documents).toEqual([]) + expect(refused.mappingEntries).toEqual([]) + expect( + await db.select().from(document).where(eq(document.knowledgeBaseId, fixture.childId)) + ).toEqual([]) + + await db + .update(knowledgeBase) + .set({ isSearchIndex: false }) + .where(eq(knowledgeBase.id, retiredId)) + const allowed = await plan() + expect(allowed.documents).toHaveLength(1) + const [placeholder] = await db + .select() + .from(document) + .where(eq(document.knowledgeBaseId, fixture.childId)) + expect(placeholder.archivedAt).not.toBeNull() + expect(allowed.docIdMap.get(fixture.source.id)).toBe(placeholder.id) + } + ) + + it.each(['source', 'target', 'target during copy', 'ordinary'] as const)( + 'checks retired Search admission for queued content with %s', + async (retiredSide) => { + const fixture = await seedKnowledgeCopy() + const childDocId = generateId() + await db.insert(document).values({ + ...fixture.source, + id: childDocId, + knowledgeBaseId: fixture.childId, + archivedAt: new Date(), + }) + if (retiredSide === 'source' || retiredSide === 'target') { + await db + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where( + eq(knowledgeBase.id, retiredSide === 'source' ? fixture.sourceId : fixture.childId) + ) + } + const progress: ForkCopyProgress = { completed: [], tables: {}, embeddings: {} } + const result = await copyForkResourceContent({ + contentPlan: { + sourceWorkspaceId: fixture.sourceWorkspaceId, + childWorkspaceId: fixture.childWorkspaceId, + userId, + tables: [], + knowledgeBases: [], + skills: [], + documents: [ + { + sourceDocId: fixture.source.id, + childDocId, + childKnowledgeBaseId: fixture.childId, + storageKey: null, + fileUrl: '', + fileSize: 0, + filename: fixture.source.filename, + mimeType: fixture.source.mimeType, + }, + ], + }, + control: { + progress, + checkpoint: async () => { + if (retiredSide === 'target during copy') + await db + .update(knowledgeBase) + .set({ isSearchIndex: true }) + .where(eq(knowledgeBase.id, fixture.childId)) + }, + }, + }) + const [copied] = await db.select().from(document).where(eq(document.id, childDocId)) + if (retiredSide === 'ordinary') { + expect(result).toMatchObject({ copied: 1, failed: 0 }) + expect(copied.archivedAt).toBeNull() + } else { + expect(result).toMatchObject({ copied: 0, failed: 1 }) + expect(copied.archivedAt).not.toBeNull() + expect( + await db.select().from(embedding).where(eq(embedding.documentId, childDocId)) + ).toEqual([]) + if (retiredSide !== 'target during copy') expect(progress.embeddings).toEqual({}) + } + const [targetWorkspace] = await db + .select() + .from(workspace) + .where(eq(workspace.id, fixture.childWorkspaceId)) + expect(targetWorkspace.storageUsedBytes).toBe(0) + } + ) }) diff --git a/apps/sim/scripts/fixtures/desktop-source-connect.tsx b/apps/sim/scripts/fixtures/desktop-source-connect.tsx index 0e4dfe86046..521fe50ade9 100644 --- a/apps/sim/scripts/fixtures/desktop-source-connect.tsx +++ b/apps/sim/scripts/fixtures/desktop-source-connect.tsx @@ -1,13 +1,13 @@ import { StrictMode, useEffect, useRef, useState } from 'react' import { ToastProvider } from '@sim/emcn' -import { QueryClient, QueryClientProvider } from '@tanstack/react-query' +import { QueryClient, QueryClientProvider, useMutation } from '@tanstack/react-query' import { createRoot } from 'react-dom/client' import { isCredentialGroupOAuthFailure } from '@/lib/credential-groups/oauth-completion' import { startDesktopSourceBrowser } from '@/lib/desktop/source-browser' +import { connectDesktopSource } from '@/lib/desktop/source-connect' import { CredentialGroupCompletionHandoff } from '@/app/credential-groups/complete/completion-handoff' import { SlackCompletion } from '@/app/credential-groups/slack-complete/slack-completion' import { SourceCompletion } from '@/app/desktop/connect/source-completion' -import { useMemberEnrollment } from '@/app/o/[organizationId]/integrations/indexed/use-member-enrollment' import { useConnectOrganizationAccount, useOrganizationAccounts, @@ -17,16 +17,16 @@ import { useSlackSearchInstallations, useStartSlackSearchOAuth } from '@/hooks/q import { useGitHubInstallationSetup } from '@/hooks/use-github-installation-setup' import { useSearchIntegrationConnection } from '@/hooks/use-search-integration-connection' -const NO_CONNECTIONS = new Set() -const MEMBERSHIP_KEYS: readonly (readonly string[])[] = [] - function SourceConnectFixture() { const accountConnection = useConnectOrganizationAccount() const reconnect = useReconnectPersonalOrganizationAccount() const accounts = useOrganizationAccounts('fixture-organization') - const enrollment = useMemberEnrollment({ - membershipQueryKeys: MEMBERSHIP_KEYS, - connectedConnectorIds: NO_CONNECTIONS, + const enrollment = useMutation({ + mutationFn: () => + connectDesktopSource({ + kind: 'member-enrollment', + params: { id: '00000000-0000-4000-8000-000000000001', connectorId: 'fixture-connector' }, + }), }) const [githubCredential, setGithubCredential] = useState('') const github = useGitHubInstallationSetup({ @@ -120,16 +120,11 @@ function SourceConnectFixture() { {String(github.pending)} {githubCredential} {github.error} - {String(enrollment.isPending)} - {enrollment.error} + {enrollment.error?.message} {connection.status} {inventory.data?.installations.length ?? 0} {connection.error &&

{connection.error.message}

} diff --git a/apps/sim/tools/index.test.ts b/apps/sim/tools/index.test.ts index b5ce9ee0134..c00162deb52 100644 --- a/apps/sim/tools/index.test.ts +++ b/apps/sim/tools/index.test.ts @@ -7163,7 +7163,6 @@ describe('organization scratch internal entrance', () => { describe('Live Search Assistant GitHub OAuth binding', () => { beforeEach(async () => { - setEnvFlags({ isLiveEnterpriseSearchEnabled: true }) const metadata = await import('@/tools/metadata') const actual = await vi.importActual('@/tools/metadata') vi.mocked(metadata.getToolMetadata).mockImplementation(actual.getToolMetadata) @@ -7199,8 +7198,6 @@ describe('Live Search Assistant GitHub OAuth binding', () => { }) it('executes the existing issue/PR counting tool using the selected personal OAuth account', async () => { const { getToolMetadata } = await import('@/tools/metadata') - const { isLiveEnterpriseSearchEnabled } = await import('@/lib/core/config/env-flags') - expect(isLiveEnterpriseSearchEnabled).toBe(true) expect(getToolMetadata('github_search_issues_v2')).toMatchObject({ id: 'github_search_issues_v2', params: { apiKey: { required: true } }, diff --git a/apps/sim/tools/index.ts b/apps/sim/tools/index.ts index b5710fe348b..d7fa5f42bce 100644 --- a/apps/sim/tools/index.ts +++ b/apps/sim/tools/index.ts @@ -15,7 +15,7 @@ import { type BillingAttributionSnapshot, serializeBillingAttributionHeader, } from '@/lib/billing/core/billing-attribution' -import { isHosted, isLiveEnterpriseSearchEnabled } from '@/lib/core/config/env-flags' +import { isHosted } from '@/lib/core/config/env-flags' import { findDatabaseQueryError } from '@/lib/core/errors/database-query-error' import { createTimeoutAbortController, @@ -1829,7 +1829,7 @@ async function executeToolImplementation( } if (operationContext?.requestMode === 'assistant' && tool) { - tool = projectAssistantConnectedAccountTool(tool, isLiveEnterpriseSearchEnabled) + tool = projectAssistantConnectedAccountTool(tool) } // Ensure context is preserved if it exists diff --git a/docker/app.Dockerfile b/docker/app.Dockerfile index 142be6253c8..126c1efe6b1 100644 --- a/docker/app.Dockerfile +++ b/docker/app.Dockerfile @@ -134,8 +134,6 @@ WORKDIR /app # Runtime flags override these image defaults; dev images opt into Plan. ARG MSHIP_PLAN_MODE_DEFAULT=false ENV MSHIP_PLAN_MODE_DEFAULT=$MSHIP_PLAN_MODE_DEFAULT -ARG SIM_SEARCH_LIVE_DEFAULT=true -ENV SIM_SEARCH_LIVE_DEFAULT=$SIM_SEARCH_LIVE_DEFAULT # Node.js 24, Python, ffmpeg, etc. are already installed in base stage ENV NODE_ENV=production diff --git a/packages/db/knowledge-projection.test.ts b/packages/db/knowledge-projection.test.ts deleted file mode 100644 index 22e9d4b86f6..00000000000 --- a/packages/db/knowledge-projection.test.ts +++ /dev/null @@ -1,245 +0,0 @@ -import { runKnowledgeProjection } from '@sim/db/knowledge-projection' -import type { Sql } from 'postgres' -import { describe, expect, it, vi } from 'vitest' - -interface Mark { - generation: number - content: boolean -} - -interface FakeDatabase { - marks: Map - /** Documents whose advisory lock another pass holds. */ - lockedElsewhere: Set - tin: boolean - /** Chunk count per document, paged by chunk index. */ - chunks: Map - /** Called before each page statement; may throw to fail the page or change marks. */ - beforePage?: (page: { documentId: string; projection: string; after: number }) => void - /** Called before a settle. */ - beforeSettle?: (documentId: string) => void -} - -/** The page statements a pass ran, in order. */ -interface Trace { - pages: Array<{ documentId: string; projection: string; after: number; mode: 'content' | 'acl' }> - locks: string[] - unlocks: string[] - claims: string[][] -} - -function postgresError(code: string, message: string): Error { - return Object.assign(new Error(message), { code }) -} - -/** - * A stand-in for a postgres.js session that answers the projector's statements from `state`, by - * the statement's text: enough of the database to drive the pass's control flow. - */ -function fakeSql(state: FakeDatabase): { sql: Sql; trace: Trace } { - const trace: Trace = { pages: [], locks: [], unlocks: [], claims: [] } - const answer = async (text: string, values: unknown[]): Promise => { - if (text.includes('to_regprocedure')) return [{ installed: state.tin }] - if (text.includes('ORDER BY marked_at')) { - const skipped = new Set(values[0] as string[]) - const claimed = [...state.marks.keys()].filter((id) => !skipped.has(id)).slice(0, 50) - trace.claims.push(claimed) - return claimed.map((document_id) => ({ document_id })) - } - if (text.includes('pg_try_advisory_lock')) { - const documentId = String(values[0]) - trace.locks.push(documentId) - return [{ acquired: !state.lockedElsewhere.has(documentId) }] - } - if (text.includes('pg_advisory_unlock')) { - trace.unlocks.push(String(values[0])) - return [] - } - if (text.includes('SELECT m.generation')) { - const mark = state.marks.get(String(values[0])) - return mark - ? [{ generation: String(mark.generation), content: mark.content, knowledge_base_id: 'kb' }] - : [] - } - if (text.includes('WITH removed AS')) { - const documentId = String(values[0]) - state.beforeSettle?.(documentId) - const mark = state.marks.get(documentId) - if (mark && mark.generation === values[1]) { - state.marks.delete(documentId) - return [{ removed: 1, marked: true }] - } - return [{ removed: 0, marked: Boolean(mark) }] - } - if (text.includes('SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty')) { - return [{ marked: state.marks.has(String(values[0])) }] - } - return [] - } - const tagged = (strings: TemplateStringsArray, ...values: unknown[]) => - answer(strings.join('?'), values) - const unsafe = async (text: string, values: unknown[] = []) => { - if (!text.includes('WITH page AS')) return answer(text, values) - const [documentId, after, pageSize] = values as [string, number, number] - const projection = - /INSERT INTO (\w+)|UPDATE (\w+) s SET/.exec(text)?.slice(1).find(Boolean) ?? 'unknown' - const mode = text.includes('INSERT INTO') ? 'content' : 'acl' - state.beforePage?.({ documentId, projection, after }) - trace.pages.push({ documentId, projection, after, mode }) - const total = state.chunks.get(documentId) ?? 1 - const first = after + 1 - const scanned = Math.max(0, Math.min(pageSize, total - first)) - return [ - { - scanned, - written: scanned, - last_chunk: scanned === 0 ? null : first + scanned - 1, - }, - ] - } - const session = Object.assign(tagged, { - unsafe, - begin: async (work: (tx: unknown) => Promise) => - work(Object.assign(tagged, { unsafe })), - }) - return { sql: session as unknown as Sql, trace } -} - -function database(overrides: Partial = {}): FakeDatabase { - return { - marks: new Map(), - lockedElsewhere: new Set(), - tin: true, - chunks: new Map(), - ...overrides, - } -} - -describe('runKnowledgeProjection', () => { - it('projects every marked document under its lock and removes each mark it settled', async () => { - const state = database({ - marks: new Map([ - ['doc-a', { generation: 1, content: true }], - ['doc-b', { generation: 3, content: false }], - ]), - }) - const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) - expect(progress).toMatchObject({ settled: 2, deferred: 0, remaining: false }) - expect(state.marks.size).toBe(0) - expect(trace.locks).toEqual(['doc-a', 'doc-b']) - expect(trace.unlocks).toEqual(['doc-a', 'doc-b']) - /** A content mark rewrites every projection; a source and ACL mark only the mirrored ones. */ - expect(trace.pages.filter((page) => page.documentId === 'doc-a')).toEqual([ - { documentId: 'doc-a', projection: 'embedding_search', after: -1, mode: 'content' }, - { documentId: 'doc-a', projection: 'embedding_keyword_search', after: -1, mode: 'content' }, - { documentId: 'doc-a', projection: 'embedding_keyword_tin', after: -1, mode: 'content' }, - ]) - expect(trace.pages.filter((page) => page.documentId === 'doc-b')).toEqual([ - { documentId: 'doc-b', projection: 'embedding_search', after: -1, mode: 'acl' }, - { documentId: 'doc-b', projection: 'embedding_keyword_tin', after: -1, mode: 'acl' }, - ]) - }) - - it('keeps a mark whose generation moved while the pass ran', async () => { - const state = database({ marks: new Map([['doc', { generation: 1, content: false }]]) }) - state.beforeSettle = () => state.marks.set('doc', { generation: 2, content: false }) - const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) - expect(progress).toMatchObject({ settled: 0, deferred: 1, remaining: true }) - expect(state.marks.get('doc')?.generation).toBe(2) - /** A document given up is not claimed again by the same pass. */ - expect(trace.claims).toEqual([['doc'], []]) - }) - - it('leaves a document another pass holds to that pass', async () => { - const state = database({ - marks: new Map([ - ['held', { generation: 1, content: false }], - ['free', { generation: 1, content: false }], - ]), - lockedElsewhere: new Set(['held']), - }) - const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) - expect(progress).toMatchObject({ settled: 1, deferred: 1, remaining: true }) - expect(state.marks.has('held')).toBe(true) - expect(trace.pages.some((page) => page.documentId === 'held')).toBe(false) - expect(trace.unlocks).toEqual(['free']) - }) - - it.each([ - ['a lock timeout', postgresError('55P03', 'canceling statement due to lock timeout')], - ['a statement timeout', postgresError('57014', 'canceling statement due to statement timeout')], - ['a deadlock', postgresError('40P01', 'deadlock detected')], - ['a chunk deleted under the page', postgresError('23503', 'insert violates foreign key')], - ])('gives a document up on %s and carries on with the next', async (_, failure) => { - const state = database({ - marks: new Map([ - ['slow', { generation: 1, content: false }], - ['fine', { generation: 1, content: false }], - ]), - }) - state.beforePage = ({ documentId }) => { - if (documentId === 'slow') throw failure - } - const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql, { searchIndexes: true }) - expect(progress).toMatchObject({ settled: 1, deferred: 1, remaining: true }) - expect(state.marks.has('slow')).toBe(true) - expect(trace.unlocks).toEqual(['slow', 'fine']) - }) - - it('counts a document deleted during its pass as gone rather than left marked', async () => { - const state = database({ marks: new Map([['doc', { generation: 1, content: false }]]) }) - state.beforePage = () => { - state.marks.delete('doc') - throw postgresError('23503', 'insert violates foreign key') - } - const { sql } = fakeSql(state) - await expect(runKnowledgeProjection(sql, { searchIndexes: true })).resolves.toMatchObject({ - settled: 0, - deferred: 0, - remaining: false, - }) - }) - - it('fails the pass on any other error, releasing the document it held', async () => { - const state = database({ marks: new Map([['doc', { generation: 1, content: false }]]) }) - state.beforePage = () => { - throw postgresError('42P01', 'relation does not exist') - } - const { sql, trace } = fakeSql(state) - await expect(runKnowledgeProjection(sql, { searchIndexes: true })).rejects.toThrow( - 'relation does not exist' - ) - expect(trace.unlocks).toEqual(['doc']) - expect(state.marks.has('doc')).toBe(true) - }) - - it('stops between pages of a long document at its deadline and keeps that mark', async () => { - vi.useFakeTimers() - try { - const state = database({ - marks: new Map([['long', { generation: 1, content: false }]]), - chunks: new Map([['long', 10]]), - tin: false, - }) - state.beforePage = () => { - vi.advanceTimersByTime(40) - } - const { sql, trace } = fakeSql(state) - const progress = await runKnowledgeProjection(sql, { - budgetMs: 100, - pageSize: 2, - searchIndexes: true, - }) - expect(trace.pages).toHaveLength(3) - expect(progress).toMatchObject({ settled: 0, deferred: 1, remaining: true }) - expect(state.marks.has('long')).toBe(true) - expect(trace.unlocks).toEqual(['long']) - } finally { - vi.useRealTimers() - } - }) -}) diff --git a/packages/db/knowledge-projection.ts b/packages/db/knowledge-projection.ts index fa55824873f..75511f56fe8 100644 --- a/packages/db/knowledge-projection.ts +++ b/packages/db/knowledge-projection.ts @@ -14,9 +14,8 @@ const logger = createLogger('KnowledgeProjection') const KNOWLEDGE_PROJECTION_MODE_SETTING = 'sim.projection_mode' /** - * Selected by every projector transaction. The projector writes each projection row's source and - * ACL itself, so the projection tables' own source and ACL triggers, which would re-read the - * document for every row, are skipped. + * Selected by every repair transaction so still-installed legacy source/ACL triggers do not copy + * permissions onto repaired vectors. Ordinary KB reads authorize against the parent document. */ const SKIP_SYNCHRONOUS_PROJECTION = `set_config('${KNOWLEDGE_PROJECTION_MODE_SETTING}', 'async', true)` @@ -53,6 +52,9 @@ const SETTLE_LOCK_TIMEOUT_MS = 2_000 /** Marks claimed per round; each is then projected under its own advisory lock. */ const CLAIM_BATCH_SIZE = 50 +/** Caps the IDs retained and resent to each claim query when documents cannot be repaired. */ +const MAX_DEFERRED_DOCUMENTS = 1_000 + /** The embedding models trained for prefix retrieval, whose 512 projection is a prefix. */ const SHORTENED_EMBEDDING_MODELS = `('text-embedding-3-small', 'text-embedding-3-large')` @@ -92,23 +94,10 @@ function searchVectorShortened(model: string, prefix: string): string { return `${model} IN ${SHORTENED_EMBEDDING_MODELS} AND ${prefix}.embedding_384 IS NULL` } -/** The projections the projector keeps; the Tin projection exists only where `tin` is installed. */ -const KNOWLEDGE_PROJECTIONS = [ - 'embedding_search', - 'embedding_keyword_search', - 'embedding_keyword_tin', -] as const -export type KnowledgeProjection = (typeof KNOWLEDGE_PROJECTIONS)[number] +export type KnowledgeProjection = 'embedding_search' -/** The projections that mirror their document's source and ACL, and so can be unfilled. */ +/** Historical trigger installers retain these immutable names until their contract migration. */ export const SOURCE_ACL_PROJECTIONS = ['embedding_search', 'embedding_keyword_tin'] as const -export type SourceAclProjection = (typeof SOURCE_ACL_PROJECTIONS)[number] -const mirrorsSourceAcl = (projection: KnowledgeProjection): projection is SourceAclProjection => - (SOURCE_ACL_PROJECTIONS as readonly string[]).includes(projection) - -/** The keyword projections, which hold only the rows of search-index knowledge bases. */ -const holdsSearchIndexesOnly = (projection: KnowledgeProjection) => - projection === 'embedding_keyword_search' || projection === 'embedding_keyword_tin' /** * The chunks one page covers: the next {@link PROJECTION_ROW_BATCH_SIZE} of the document in @@ -127,96 +116,31 @@ const PAGE_RESULT = `SELECT (SELECT count(*)::int FROM page) AS scanned, (SELECT max(chunk_index) FROM page) AS last_chunk` /** - * Rewrites a page's projection rows from their chunks and the document, for a mark whose chunk - * write skipped the synchronous triggers (only releases that deferred projection wrote those), - * writing only the rows that differ, so a document whose rows are already current costs reads and no index writes. The - * source and ACL are the document's as this statement reads it; a change that commits after it - * marks the document again, so the projector's settle leaves the mark for the next pass. + * Repairs only ordinary-KB vectors left by older deferred writers. The parent KB is checked and + * share-locked on every page so a concurrent Search-marker change cannot admit retired content. + * Binary columns remain compatibility writes until their database width constraint is replaced. */ -function contentPageStatement(projection: KnowledgeProjection): string { - if (projection === 'embedding_search') { - const vectors = SEARCH_VECTOR_COLUMNS.join(', ') - const compared = [ - 'knowledge_base_id', - 'document_id', - 'enabled', - 'connector_id', - 'acl', - ...SEARCH_VECTOR_COLUMNS, - ] - return `WITH ${PAGE}, source AS MATERIALIZED ( - SELECT e.id, e.knowledge_base_id, e.document_id, e.enabled, e.embedding, e.embedding_384, - e.embedding_768, e.embedding_1024, e.embedding_3072, - ${searchVectorShortened('k.embedding_model', 'e')} AS shortened, d.connector_id, d.acl - FROM page p JOIN embedding e ON e.id = p.id - JOIN knowledge_base k ON k.id = e.knowledge_base_id - JOIN document d ON d.id = e.document_id - ), written AS ( - INSERT INTO embedding_search AS s - (id, knowledge_base_id, document_id, enabled, ${SEARCH_BINARY_COLUMNS.join(', ')}, ${vectors}, - connector_id, acl) - SELECT id, knowledge_base_id, document_id, enabled, ${searchBinaryProjections('source')}, - ${searchVectorProjections('source', 'shortened')}, connector_id, acl - FROM source - ON CONFLICT (id) DO UPDATE SET - ${[...compared, ...SEARCH_BINARY_COLUMNS].map((column) => `${column} = EXCLUDED.${column}`).join(', ')} - WHERE (${compared.map((column) => `s.${column}`).join(', ')}) - IS DISTINCT FROM (${compared.map((column) => `EXCLUDED.${column}`).join(', ')}) - RETURNING s.id - ) ${PAGE_RESULT}` - } - if (projection === 'embedding_keyword_search') { - const compared = ['knowledge_base_id', 'document_id', 'enabled', 'content_tsv'] - return `WITH ${PAGE}, source AS MATERIALIZED ( - SELECT e.id, ${compared.map((column) => `e.${column}`).join(', ')}, k.is_search_index - FROM page p JOIN embedding e ON e.id = p.id - JOIN knowledge_base k ON k.id = e.knowledge_base_id - ), removed AS ( - DELETE FROM embedding_keyword_search s USING source - WHERE s.id = source.id AND NOT source.is_search_index - ), written AS ( - INSERT INTO embedding_keyword_search AS s (${['id', ...compared].join(', ')}) - SELECT id, ${compared.join(', ')} FROM source WHERE is_search_index - ON CONFLICT (id) DO UPDATE SET - ${compared.map((column) => `${column} = EXCLUDED.${column}`).join(', ')} - WHERE (${compared.map((column) => `s.${column}`).join(', ')}) - IS DISTINCT FROM (${compared.map((column) => `EXCLUDED.${column}`).join(', ')}) - RETURNING s.id - ) ${PAGE_RESULT}` - } - const compared = ['knowledge_base_id', 'document_id', 'enabled', 'content', 'connector_id', 'acl'] +function contentPageStatement(): string { + const vectors = SEARCH_VECTOR_COLUMNS.join(', ') + const compared = ['knowledge_base_id', 'document_id', 'enabled', ...SEARCH_VECTOR_COLUMNS] return `WITH ${PAGE}, source AS MATERIALIZED ( - SELECT e.id, e.knowledge_base_id, e.document_id, e.enabled, - knowledge_tin_base_token(e.knowledge_base_id) || ' ' || knowledge_tin_stream(e.content_tsv) AS content, - d.connector_id, d.acl, k.is_search_index + SELECT e.id, e.knowledge_base_id, e.document_id, e.enabled, e.embedding, e.embedding_384, + e.embedding_768, e.embedding_1024, e.embedding_3072, + ${searchVectorShortened('k.embedding_model', 'e')} AS shortened FROM page p JOIN embedding e ON e.id = p.id JOIN knowledge_base k ON k.id = e.knowledge_base_id - JOIN document d ON d.id = e.document_id - ), removed AS ( - DELETE FROM embedding_keyword_tin t USING source - WHERE t.id = source.id AND NOT source.is_search_index + WHERE NOT k.is_search_index + FOR SHARE OF k ), written AS ( - INSERT INTO embedding_keyword_tin AS t (${['id', ...compared].join(', ')}) - SELECT id, ${compared.join(', ')} FROM source WHERE is_search_index + INSERT INTO embedding_search AS s + (id, knowledge_base_id, document_id, enabled, ${SEARCH_BINARY_COLUMNS.join(', ')}, ${vectors}) + SELECT id, knowledge_base_id, document_id, enabled, ${searchBinaryProjections('source')}, + ${searchVectorProjections('source', 'shortened')} + FROM source ON CONFLICT (id) DO UPDATE SET - ${compared.map((column) => `${column} = EXCLUDED.${column}`).join(', ')} - WHERE (${compared.map((column) => `t.${column}`).join(', ')}) + ${[...compared, ...SEARCH_BINARY_COLUMNS].map((column) => `${column} = EXCLUDED.${column}`).join(', ')} + WHERE (${compared.map((column) => `s.${column}`).join(', ')}) IS DISTINCT FROM (${compared.map((column) => `EXCLUDED.${column}`).join(', ')}) - RETURNING t.id - ) ${PAGE_RESULT}` -} - -/** - * Copies the document's source and ACL onto a page's existing projection rows that differ, - * including rows written before projections carried a source and ACL. Chunks are untouched, so their vectors - * are never read; a row a chunk change has not projected yet is left to that change's own mark. - */ -function sourceAclPageStatement(projection: KnowledgeProjection): string { - return `WITH ${PAGE}, written AS ( - UPDATE ${projection} s SET connector_id = d.connector_id, acl = d.acl - FROM page p, document d - WHERE s.id = p.id AND d.id = $1 - AND (s.connector_id IS DISTINCT FROM d.connector_id OR s.acl IS DISTINCT FROM d.acl) RETURNING s.id ) ${PAGE_RESULT}` } @@ -245,7 +169,7 @@ async function enterProjectorTransaction(tx: TransactionSql, lockTimeoutMs: numb interface ProjectionMark { generation: number content: boolean - knowledgeBaseId: string + isSearchIndex: boolean } export interface KnowledgeProjectionOptions { @@ -256,14 +180,6 @@ export interface KnowledgeProjectionOptions { */ budgetMs?: number pageSize?: number - /** - * Whether search-index rows are read, that is whether indexed organization search is on (see - * {@link MarkScope}). Only then is the Tin keyword projection written, since it holds only - * search-index rows; the other projections are written either way, the GIN keyword projection - * for search-index bases alone whatever this says, and Tin is still skipped where it is not - * installed. - */ - searchIndexes: boolean /** Called after each page commits, for tests that interleave writes with a run. */ onPage?: (page: { documentId: string @@ -283,13 +199,6 @@ export interface KnowledgeProjectionProgress { remaining: boolean } -/** Whether the Tin keyword projection is maintained here: its functions exist only where installed. */ -async function tinInstalled(sql: Sql): Promise { - const [row] = await sql>` - SELECT to_regprocedure('knowledge_tin_stream(tsvector)') IS NOT NULL AS installed` - return Boolean(row?.installed) -} - /** * The oldest marks, past the ones this run has already passed over. A read without row locks: a * pass owns a document through its advisory lock, and a row lock on the mark would either block @@ -311,16 +220,17 @@ async function claimMarks(sql: Sql, skipped: readonly string[]): Promise { const [row] = await sql< - Array<{ generation: string; content: boolean; knowledge_base_id: string }> + Array<{ generation: string; content: boolean; is_search_index: boolean }> >` - SELECT m.generation, m.content, d.knowledge_base_id + SELECT m.generation, m.content, k.is_search_index FROM knowledge_projection_dirty m JOIN document d ON d.id = m.document_id + JOIN knowledge_base k ON k.id = d.knowledge_base_id WHERE m.document_id = ${documentId}` return row ? { generation: Number(row.generation), content: row.content, - knowledgeBaseId: row.knowledge_base_id, + isSearchIndex: row.is_search_index, } : null } @@ -370,15 +280,11 @@ async function stillMarked(sql: Sql, documentId: string): Promise { async function projectDocumentRows( sql: Sql, documentId: string, - projection: KnowledgeProjection, - mark: ProjectionMark, options: KnowledgeProjectionOptions, deadline: number ): Promise<{ pages: number; written: number; finished: boolean }> { const pageSize = options.pageSize ?? PROJECTION_ROW_BATCH_SIZE - const statement = mark.content - ? contentPageStatement(projection) - : sourceAclPageStatement(projection) + const statement = contentPageStatement() let after = -1 let pages = 0 let written = 0 @@ -386,9 +292,6 @@ async function projectDocumentRows( if (Date.now() >= deadline) return { pages, written, finished: false } const page = await sql.begin(async (tx) => { await enterProjectorTransaction(tx, PROJECTION_PAGE_LOCK_TIMEOUT_MS) - /** Shares the base's membership lock, as the embedding triggers do, so a flip of its marker waits. */ - if (holdsSearchIndexesOnly(projection)) - await tx`SELECT pg_advisory_xact_lock_shared(knowledge_tin_membership_key(${mark.knowledgeBaseId}))` const [row] = await tx.unsafe< Array<{ scanned: number; written: number; last_chunk: number | null }> >(statement, [documentId, after, pageSize]) @@ -396,7 +299,7 @@ async function projectDocumentRows( }) pages += 1 written += page.written - await options.onPage?.({ documentId, projection, written: page.written }) + await options.onPage?.({ documentId, projection: 'embedding_search', written: page.written }) if (page.last_chunk === null || page.scanned < pageSize) break after = page.last_chunk } @@ -412,7 +315,6 @@ type DocumentOutcome = 'settled' | 'deferred' | 'gone' async function projectMarkedDocument( sql: Sql, documentId: string, - projections: readonly KnowledgeProjection[], options: KnowledgeProjectionOptions, totals: { pages: number; written: number }, deadline: number @@ -423,9 +325,8 @@ async function projectMarkedDocument( try { const mark = await readMark(sql, documentId) if (!mark) return 'gone' - for (const projection of projections) { - if (!mark.content && !mirrorsSourceAcl(projection)) continue - const done = await projectDocumentRows(sql, documentId, projection, mark, options, deadline) + if (mark.content && !mark.isSearchIndex) { + const done = await projectDocumentRows(sql, documentId, options, deadline) totals.pages += done.pages totals.written += done.written if (!done.finished) return 'deferred' @@ -445,7 +346,7 @@ async function projectMarkedDocument( } /** - * Converges the search projections of every marked document, oldest mark first, until none is + * Repairs ordinary-KB vectors and releases obsolete marks, oldest mark first, until none is * left or the budget runs out. Runs on a connection of its own: each document is projected under * a session advisory lock, so concurrent runs never project the same document at once and a run * that dies releases its locks with its connection. A document keeps its mark, left to a later @@ -459,10 +360,6 @@ export async function runKnowledgeProjection( ): Promise { const deadline = options.budgetMs === undefined ? Number.POSITIVE_INFINITY : Date.now() + options.budgetMs - const projections = - options.searchIndexes && (await tinInstalled(sql)) - ? KNOWLEDGE_PROJECTIONS - : KNOWLEDGE_PROJECTIONS.filter((projection) => projection !== 'embedding_keyword_tin') const totals = { pages: 0, written: 0 } const skipped: string[] = [] let settled = 0 @@ -473,16 +370,12 @@ export async function runKnowledgeProjection( } for (const documentId of claimed) { if (Date.now() >= deadline) break - const outcome = await projectMarkedDocument( - sql, - documentId, - projections, - options, - totals, - deadline - ) + const outcome = await projectMarkedDocument(sql, documentId, options, totals, deadline) if (outcome === 'settled') settled += 1 else if (outcome === 'deferred') skipped.push(documentId) + if (skipped.length >= MAX_DEFERRED_DOCUMENTS) { + return { settled, deferred: skipped.length, ...totals, remaining: true } + } } } return { settled, deferred: skipped.length, ...totals, remaining: true } @@ -498,42 +391,15 @@ export const MARK_RELEASE_BUDGET_MS = 10_000 /** Marks one release statement removes; a release repeats it while statements come back full. */ const RELEASE_BATCH_SIZE = 1_000 -/** - * Which marks a pass is owed besides content: search-index documents, whose rows mirror their - * source and ACL, while indexed organization search reads them. With it off, a mark with no content - * to project is owed nothing, whatever its knowledge base. - */ -export interface MarkScope { - searchIndexes: boolean -} - -/** - * Whether any mark needs a projector pass: one carrying content to project, or, when `scope` owes - * search-index documents a pass, one on a search-index document. Every other mark is released by - * {@link releaseSettledMarks} without a pass. - * - * Asked right after a release. One that drained its marks left only the marks a pass is owed - * (and the few a writer held), so the join that tells a search-index mark apart walks a handful of - * rows. One cut short left a backlog the join would walk in full, so only the content marks are - * asked about: a search-index mark behind that backlog waits for the sweep whose release drains it. - */ -export async function hasKnowledgeProjectionWork( - sql: Sql | TransactionSql, - release: { drained: boolean }, - scope: MarkScope -): Promise { - const [row] = - release.drained && scope.searchIndexes - ? await sql>` - SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty WHERE content) - OR EXISTS ( - SELECT 1 FROM knowledge_projection_dirty d - JOIN document doc ON doc.id = d.document_id - JOIN knowledge_base k ON k.id = doc.knowledge_base_id - WHERE k.is_search_index - ) AS pending` - : await sql>` - SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty WHERE content) AS pending` +/** Only deferred ordinary-KB content requires projection work after indexed Search retirement. */ +export async function hasKnowledgeProjectionWork(sql: Sql | TransactionSql): Promise { + const [row] = await sql>` + SELECT EXISTS ( + SELECT 1 FROM knowledge_projection_dirty m + JOIN document d ON d.id = m.document_id + JOIN knowledge_base k ON k.id = d.knowledge_base_id + WHERE m.content AND NOT k.is_search_index + ) AS pending` return Boolean(row?.pending) } @@ -547,22 +413,11 @@ export interface SettledMarkRelease { } /** - * Transitional: removes the marks that carry no content to project and that `scope` owes no pass, - * until a follow-up migration scopes the mark triggers to search-index knowledge bases. Their - * synchronous writers already wrote every projection row a workspace search reads, and those - * searches decide nothing on a projection row's source, ACL, or mark, so a pass over them would - * only re-read rows it then leaves as they are. While indexed organization search is on, the marks - * of search-index documents, whose rows it reads by source and ACL, stay for a pass. - * - * An empty mark table costs one probe, and reports `empty` so its caller asks nothing further. Marks a writer holds are skipped rather than waited on, - * and a mark whose writer skipped the synchronous triggers carries content and stays for a pass. - * Stops once a statement comes back short or `deadline` passes. + * Releases obsolete ACL-only and Search marks in bounded transactions. Ordinary-KB content marks + * survive for vector repair. Locked marks and bases are skipped; locks prevent a content upgrade + * or a Search-marker change from making the deleted mark necessary before this commit. */ -export async function releaseSettledMarks( - sql: Sql, - deadline: number, - scope: MarkScope -): Promise { +export async function releaseSettledMarks(sql: Sql, deadline: number): Promise { const [marked] = await sql>` SELECT EXISTS (SELECT 1 FROM knowledge_projection_dirty) AS any` if (!marked?.any) return { released: 0, drained: true, empty: true } @@ -570,23 +425,18 @@ export async function releaseSettledMarks( while (Date.now() < deadline) { const count = await sql.begin(async (tx) => { await enterProjectorTransaction(tx, SETTLE_LOCK_TIMEOUT_MS) - const settled = scope.searchIndexes - ? tx` - SELECT d.document_id FROM knowledge_projection_dirty d - JOIN document doc ON doc.id = d.document_id - JOIN knowledge_base k ON k.id = doc.knowledge_base_id - WHERE NOT d.content AND NOT k.is_search_index - LIMIT ${RELEASE_BATCH_SIZE} - FOR UPDATE OF d SKIP LOCKED` - : tx` - SELECT d.document_id FROM knowledge_projection_dirty d - WHERE NOT d.content - LIMIT ${RELEASE_BATCH_SIZE} - FOR UPDATE SKIP LOCKED` + const settled = tx` + SELECT m.document_id FROM knowledge_projection_dirty m + JOIN document d ON d.id = m.document_id + JOIN knowledge_base k ON k.id = d.knowledge_base_id + WHERE NOT m.content OR k.is_search_index + LIMIT ${RELEASE_BATCH_SIZE} + FOR UPDATE OF m SKIP LOCKED + FOR SHARE OF k SKIP LOCKED` const [row] = await tx>` WITH released AS ( DELETE FROM knowledge_projection_dirty m - WHERE m.document_id IN (${settled}) AND NOT m.content + WHERE m.document_id IN (${settled}) RETURNING 1 ) SELECT count(*)::int AS released FROM released` diff --git a/packages/db/schema.ts b/packages/db/schema.ts index ac6a68ea216..0a47dffac6e 100644 --- a/packages/db/schema.ts +++ b/packages/db/schema.ts @@ -3749,6 +3749,7 @@ export const embedding = pgTable( ) /** Keyword ranking reads text-search vectors independently of chunk content and semantic vectors. */ +// contract-pending(after the indexed-search retirement release and all legacy projection writers have drained): drop embedding_keyword_search — regular KB keyword queries read embedding.content_tsv. export const embeddingKeywordSearch = pgTable( 'embedding_keyword_search', { @@ -3779,6 +3780,7 @@ export const EMBEDDING_KEYWORD_TIN_INDEX = 'embedding_keyword_tin_content_idx' * the index, and the embedding and knowledge base triggers that own these rows, and only where * `tin` exists; elsewhere the table stays empty and keyword search keeps the GIN projection. */ +// contract-pending(after the indexed-search retirement release and all legacy projection writers have drained): drop embedding_keyword_tin — only retired indexed Search ranks this projection. export const embeddingKeywordTin = pgTable( 'embedding_keyword_tin', { @@ -3825,10 +3827,12 @@ export const embeddingSearch = pgTable( * source spends its scan budget on chunks the graph reached but the member cannot read. * NULL for uploads. */ + // contract-pending(after the indexed-search retirement release and source/ACL projection writers have drained): drop connector_id — regular KB retrieval checks the parent document. connectorId: text('connector_id'), /** The document's ACL, mirrored by trigger, so a walk can test readability on the row it visits. */ + // contract-pending(after the indexed-search retirement release and source/ACL projection writers have drained): drop acl — regular KB retrieval retains document-level access checks. acl: text('acl').array(), - /** contract-pending(after the projection sync trigger stops writing them): drop the binary columns; their ANN indexes were dropped in 0372, and nothing reads them. */ + // contract-pending(after vector writers stop computing binary projections and embedding_search_width_check is replaced): drop binary and all binary_* columns — their ANN indexes were dropped in 0372 and no reader uses them. binary: bit('binary', { dimensions: 1536 }), binary384: bit('binary_384', { dimensions: 384 }), binary768: bit('binary_768', { dimensions: 768 }), @@ -3885,6 +3889,7 @@ export const embeddingSearch = pgTable( * holds the document row, but clearing it would otherwise take that row again, and readers probe * this small table instead of joining `document` per ranked row. */ +// contract-pending(after deferred vector content is repaired and projection mark writers/workers are retired): drop knowledge_projection_dirty — legacy ACL copies need no repair, but unfinished vector repairs must survive retirement. export const knowledgeProjectionDirty = pgTable( 'knowledge_projection_dirty', { @@ -5222,6 +5227,7 @@ export const usageLogSourceEnum = pgEnum('usage_log_source', [ ]) /** Content-free organization Search activity, independent of billable model usage. */ +// contract-pending(after the indexed-search retirement release and old activity writers have drained): drop organization_search_invocation — live Search does not record indexed result activity. export const organizationSearchInvocation = pgTable( 'organization_search_invocation', { diff --git a/packages/db/script-migrations/search-embedding-retirement.md b/packages/db/script-migrations/search-embedding-retirement.md index 86d48536e5b..7b8864c4499 100644 --- a/packages/db/script-migrations/search-embedding-retirement.md +++ b/packages/db/script-migrations/search-embedding-retirement.md @@ -15,8 +15,8 @@ KBs' live source/credential configuration, document metadata, and backing files ## Before running The app and workers must already use live Search, and older indexing jobs must be drained. -`SIM_SEARCH_LIVE=true` (the default) makes `isIndexedOrgSearchEnabled()` false. **`SIM_SEARCH_LIVE=false` -enables indexed Search again.** The cleanup does not inspect this flag. Live source setup may still +Enterprise Search no longer has an indexed backend or an environment toggle to re-enable it. +Older releases could re-enable indexed Search, so their workers must be drained before retirement. Live source setup may still create a Search KB for configuration; it does not index content. Document uploads, dispatch and queued processing also honor the indexed-search gate. diff --git a/packages/testing/src/mocks/deployment-shape.mock.ts b/packages/testing/src/mocks/deployment-shape.mock.ts index 443f6ba8082..0b61e4aaf26 100644 --- a/packages/testing/src/mocks/deployment-shape.mock.ts +++ b/packages/testing/src/mocks/deployment-shape.mock.ts @@ -8,7 +8,6 @@ export interface MockDeploymentShape { azureConfigured: boolean cohereConfigured: boolean features: { - liveEnterpriseSearch?: boolean accessControl: boolean auditLogs: boolean customBlocks: boolean @@ -43,7 +42,6 @@ export function createMockDeploymentShape( cohereConfigured: false, ...rest, features: { - liveEnterpriseSearch: true, accessControl: false, auditLogs: false, customBlocks: false, diff --git a/packages/testing/src/mocks/env-flags.mock.ts b/packages/testing/src/mocks/env-flags.mock.ts index 98666e661ec..cba6ac6fc4c 100644 --- a/packages/testing/src/mocks/env-flags.mock.ts +++ b/packages/testing/src/mocks/env-flags.mock.ts @@ -32,7 +32,6 @@ interface EnvFlagsMockState { isUsageMonitoringEnabled: boolean isAccessControlEnabled: boolean isOrganizationsEnabled: boolean - isLiveEnterpriseSearchEnabled: boolean isInboxEnabled: boolean isSandboxDeploymentEntitled: boolean isSandboxesEnabled: boolean @@ -87,7 +86,6 @@ const defaultEnvFlagsState: EnvFlagsMockState = { isAccessControlEnabled: false, isOrganizationsEnabled: false, /** Live Search is the default Sim Search backend; indexed search is dormant. */ - isLiveEnterpriseSearchEnabled: true, // True with billing off and no flags set — these carry a legacy default of // `true` so upgrades do not remove a feature. See // ENTERPRISE_FEATURE_LEGACY_DEFAULTS. diff --git a/packages/testing/src/mocks/indexed-org-search.mock.ts b/packages/testing/src/mocks/indexed-org-search.mock.ts deleted file mode 100644 index d13b8db4a63..00000000000 --- a/packages/testing/src/mocks/indexed-org-search.mock.ts +++ /dev/null @@ -1,22 +0,0 @@ -/** - * The `vi.mock` factory a real-infrastructure suite of indexed organization search passes for - * `@/lib/core/config/env-flags`: the real module, with Live Search off so the dormant indexed - * backend is the one selected. Unit suites use the shared env-flags mock's `setEnvFlags` instead. - * - * Loaded inside the factory, which runs before the test file's own imports are initialized. - * - * @example - * ```ts - * vi.mock('@/lib/core/config/env-flags', async (importOriginal) => - * (await import('@sim/testing/mocks/indexed-org-search.mock')).indexedOrgSearchEnvFlags(importOriginal) - * ) - * ``` - */ -export async function indexedOrgSearchEnvFlags( - importOriginal: () => Promise -): Promise> { - return { - ...(await importOriginal>()), - isLiveEnterpriseSearchEnabled: false, - } -} diff --git a/packages/testing/src/mocks/kb-connectors-queries.mock.ts b/packages/testing/src/mocks/kb-connectors-queries.mock.ts index 107e821d42d..f5e3ef8fe05 100644 --- a/packages/testing/src/mocks/kb-connectors-queries.mock.ts +++ b/packages/testing/src/mocks/kb-connectors-queries.mock.ts @@ -21,13 +21,6 @@ const connectorKeys = { details: (knowledgeBaseId?: string) => [...connectorKeys.all(knowledgeBaseId), 'detail'] as const, detail: (knowledgeBaseId?: string, connectorId?: string) => [...connectorKeys.details(knowledgeBaseId), connectorId ?? ''] as const, - progress: (knowledgeBaseId?: string, connectorId?: string, scope?: MockResourceScope) => - [ - ...connectorKeys.progresses(knowledgeBaseId, connectorId), - scope ? resourceScopeKey(scope) : '', - ] as const, - progresses: (knowledgeBaseId?: string, connectorId?: string) => - [...connectorKeys.detail(knowledgeBaseId, connectorId), 'progress'] as const, } const searchIndexKeys = { @@ -55,7 +48,7 @@ const mutationHook = () => vi.fn((..._args: unknown[]): unknown => createMutatio * - `mockIsConnectorSyncingOrPending` is the real predicate (`status` `pending`/`syncing`, or * `memberSyncStatus` `pending`/`running`). * - Query hooks (`useConnectorList`, `useConnectorDetail`, `useSearchIndex`, - * `useSearchSourceOverview`, `useOrganizationSearchOverview`, `useSearchSources`, + * `useSearchSources`, * `useConnectorDocuments`) return a fresh {@link createQueryResultMock} (`data: undefined`, * `isPending: true`). * - Every mutation hook returns a fresh {@link createMutationResultMock} (`idle`, no-op @@ -82,8 +75,6 @@ export const kbConnectorsQueriesMockFns = { mockUseCreateConnector: mutationHook(), mockUseUpdateConnector: mutationHook(), mockUseSearchIndex: queryHook(), - mockUseSearchSourceOverview: queryHook(), - mockUseOrganizationSearchOverview: queryHook(), mockUseSearchSources: queryHook(), mockUseStartConnectorMemberEnrollment: mutationHook(), mockUseUpdateConnectorAccess: mutationHook(), @@ -92,7 +83,6 @@ export const kbConnectorsQueriesMockFns = { mockUseConnectorDocuments: queryHook(), mockUseExcludeConnectorDocument: mutationHook(), mockUseRestoreConnectorDocument: mutationHook(), - mockUseConnectSimSearchConnector: mutationHook(), mockUsePrepareSearchSource: mutationHook(), } @@ -121,8 +111,6 @@ export const kbConnectorsQueriesMock = { useCreateConnector: kbConnectorsQueriesMockFns.mockUseCreateConnector, useUpdateConnector: kbConnectorsQueriesMockFns.mockUseUpdateConnector, useSearchIndex: kbConnectorsQueriesMockFns.mockUseSearchIndex, - useSearchSourceOverview: kbConnectorsQueriesMockFns.mockUseSearchSourceOverview, - useOrganizationSearchOverview: kbConnectorsQueriesMockFns.mockUseOrganizationSearchOverview, useSearchSources: kbConnectorsQueriesMockFns.mockUseSearchSources, useStartConnectorMemberEnrollment: kbConnectorsQueriesMockFns.mockUseStartConnectorMemberEnrollment, @@ -132,6 +120,5 @@ export const kbConnectorsQueriesMock = { useConnectorDocuments: kbConnectorsQueriesMockFns.mockUseConnectorDocuments, useExcludeConnectorDocument: kbConnectorsQueriesMockFns.mockUseExcludeConnectorDocument, useRestoreConnectorDocument: kbConnectorsQueriesMockFns.mockUseRestoreConnectorDocument, - useConnectSimSearchConnector: kbConnectorsQueriesMockFns.mockUseConnectSimSearchConnector, usePrepareSearchSource: kbConnectorsQueriesMockFns.mockUsePrepareSearchSource, } diff --git a/scripts/test-patterns-baseline.json b/scripts/test-patterns-baseline.json index 0702a5bd1e4..dc44ea44c49 100644 --- a/scripts/test-patterns-baseline.json +++ b/scripts/test-patterns-baseline.json @@ -59,7 +59,6 @@ "local-factory\tapps/sim/executor/utils/output-filter.test.ts\t@/blocks", "local-factory\tapps/sim/hooks/use-table-undo.test.ts\t@/lib/table/constants", "local-factory\tapps/sim/lib/credentials/application/resolve-organization-personal-token.test.ts\t@/lib/sim-search/connectors", - "local-factory\tapps/sim/lib/knowledge/application/organization-search-overview.test.ts\t@/lib/sim-search/connectors", "local-factory\tapps/sim/lib/knowledge/application/search-integrations.test.ts\t@/lib/sim-search/connectors", "local-factory\tapps/sim/lib/knowledge/application/sim-search.test.ts\t@/lib/sim-search/connectors", "local-factory\tapps/sim/lib/knowledge/mcp/route-handler.test.ts\t@/lib/sim-search/connectors", From 064cd410303826f8e477da375117703d6a079a13 Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 13:11:41 -0700 Subject: [PATCH 11/31] fix(sandbox): preserve workbench availability for unrecorded API responses (#8536) * fix(sandbox): preserve workbench availability for unrecorded API responses * fix(sandbox): preserve completed table mutation outcomes --- .../workbench-confidentiality.live.test.ts | 164 ++++++++++++++++++ .../tools/sandbox-resource-transport.test.ts | 49 ------ .../tools/sandbox-resource-transport.ts | 89 +++++----- 3 files changed, 214 insertions(+), 88 deletions(-) diff --git a/apps/sim/lib/mothership/tools/handlers/workbench-confidentiality.live.test.ts b/apps/sim/lib/mothership/tools/handlers/workbench-confidentiality.live.test.ts index e1457dc6259..83042c5f896 100644 --- a/apps/sim/lib/mothership/tools/handlers/workbench-confidentiality.live.test.ts +++ b/apps/sim/lib/mothership/tools/handlers/workbench-confidentiality.live.test.ts @@ -18,6 +18,10 @@ import { mothershipGoFetchMock, mothershipGoFetchMockFns, } from '@sim/testing/mocks/mothership-go-fetch.mock' +import { + mothershipWorkspaceTargetMock, + mothershipWorkspaceTargetMockFns, +} from '@sim/testing/mocks/mothership-workspace-target.mock' import { redisConfigMockFns } from '@sim/testing/mocks/redis-config.mock' import { remoteSandboxProviderMock, @@ -30,6 +34,7 @@ import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' const io = vi.hoisted(() => ({ mount: vi.fn(), find: vi.fn(), write: vi.fn() })) vi.mock('@/tools', () => toolsMock) +vi.mock('@/lib/mothership/application/workspace-target', () => mothershipWorkspaceTargetMock) vi.mock('@/lib/mothership/tools/secret-mount-materializer.server', () => ({ materializeCopilotCodeSecrets: io.mount, CopilotCodeSecretAccessError: class extends Error {}, @@ -54,6 +59,7 @@ vi.mock('@/lib/mothership/vfs/resource-writer', () => ({ })) import { functionExecuteBodySchema } from '@/lib/api/contracts' +import * as inProcessTransport from '@/lib/api/server/routes/in-process-transport' import { encryptSecret } from '@/lib/core/security/encryption' import { PRIVATE_TOOL_METADATA_REQUEST_HEADER, @@ -73,12 +79,15 @@ import { inspectToolResultForCopilot } from '@/lib/mothership/request/tools/reso import type { ToolExecutionContext } from '@/lib/mothership/tool-executor/types' import { executeFunctionExecute } from '@/lib/mothership/tools/handlers/function-execute' import { executeRunCode } from '@/lib/mothership/tools/handlers/run-code' +import { proxySandboxResourceRequest } from '@/lib/mothership/tools/sandbox-resource-transport' import { readSandboxResourceScope, withSandboxResourceScope, } from '@/lib/mothership/tools/sandbox-resources' import { buildMothershipSandboxSession } from '@/lib/mothership/tools/sandbox-session' import { chatSandboxSessionKey } from '@/lib/mothership/tools/sandbox-session-key' +import { reportTableRowDelivery } from '@/lib/table/application/row-delivery-observer' +import { reportWorkspaceFileDelivery } from '@/lib/workspace-files/application/file-delivery-observer' import { ResolvedSecretTraceRegistry } from '@/executor/utils/resolved-secret-trace-registry' import { buildFunctionExecuteBody, functionExecuteTool } from '@/tools/function/execute' import type { CodeExecutionInput } from '@/tools/function/types' @@ -320,6 +329,161 @@ function inResourceScope(action: () => Promise) { ) } +async function sandboxApi(path: string, handler: () => Promise, method = 'GET') { + mothershipWorkspaceTargetMockFns.mockResolveInvocationWorkspace.mockResolvedValue(scope) + vi.spyOn(inProcessTransport, 'matchV2Route').mockReturnValue({ + pattern: path, + params: { fileId: 'fixture', tableId: 'fixture' }, + literals: 3, + load: async () => ({ GET: handler, POST: handler }), + }) + return inResourceScope(async () => { + const session = await buildMothershipSandboxSession({ + ...scope, + sessionKey: chatSandboxSessionKey(chatId), + }) + const endpoint = session.envs!.SIM_ENDPOINT + return proxySandboxResourceRequest( + new Request(`${endpoint}${path}`, { + method, + headers: { 'x-api-key': session.envs!.SIM_API_KEY }, + }), + endpoint.split('/').at(-1)! + ) + }) +} + +describe('sandbox API provenance admission', () => { + it('keeps ordinary API mutations usable for later code output and generated CLI input', async () => { + const response = await sandboxApi( + '/api/v2/custom-tools', + async () => Response.json({ data: { id: 'fixture-tool', title: 'fixture' } }), + 'POST' + ) + expect(response.status).toBe(200) + const result = await run('printf "[]" > operations.json; printf "ready"') + expect(result.projected.safe).toBe(true) + expect(result.projected.result).toMatchObject({ success: true, output: { stdout: 'ready' } }) + expect( + (await readCliInputFile(chatSandboxSessionKey(chatId), 'operations.json')).toString() + ).toBe('[]') + }) + + it('retains earlier secret protection after an API response without provenance', async () => { + await run('printf "%s" "$TOKEN" > saved.txt', ['TOKEN']) + await sandboxApi('/api/v2/custom-tools', async () => Response.json({ data: [] })) + const result = await run('cat saved.txt') + expect(result.projected.safe).toBe(true) + expect(result.projected.result).toMatchObject({ + success: true, + output: { stdout: '{{TOKEN}}' }, + }) + await expect(readCliInputFile(chatSandboxSessionKey(chatId), 'saved.txt')).rejects.toThrow( + 'protected workbench values' + ) + }) + + it.each(['file', 'table'] as const)( + 'imports explicit %s delivery evidence before later output', + async (source) => { + const response = await sandboxApi( + `/api/v2/${source === 'file' ? 'files/fixture' : 'tables/fixture/rows'}`, + async () => { + if (source === 'file') { + await reportWorkspaceFileDelivery({ + status: 'exact', + entries: [ + { + name: 'TOKEN', + encryptedValue: catalog[0].encryptedValue, + sourceUserId: scope.userId, + sourceWorkspaceId: scope.workspaceId, + }, + ], + }) + } else { + await reportTableRowDelivery( + { + version: 1, + complete: true, + scope, + entries: [{ name: 'TOKEN', encryptedValue: catalog[0].encryptedValue }], + }, + [{ value: canary }] + ) + } + return new Response(canary) + } + ) + await machine.writeFile('/home/user/delivered.txt', await response.text()) + const result = await run('cat delivered.txt') + expect(result.projected.safe).toBe(true) + expect(result.projected.result).toMatchObject({ + success: true, + output: { stdout: '{{TOKEN}}' }, + }) + } + ) + + it('preserves mutation completion and withholds its body when provenance storage fails', async () => { + const mutationPath = join(root, 'mutation.json') + const response = await sandboxApi( + '/api/v2/tables/fixture/rows', + async () => { + await writeFile(mutationPath, JSON.stringify({ committed: true, completed: false })) + const evalCommand = redis.eval.bind(redis) + vi.spyOn(redis, 'eval').mockImplementation((...args) => { + if (String(args[2]).startsWith('mothership:workbench-provenance:v2:')) { + return Promise.reject(new Error('Synthetic provenance storage failure')) + } + return evalCommand(...args) + }) + await reportTableRowDelivery( + { + version: 1, + complete: true, + scope, + entries: [{ name: 'TOKEN', encryptedValue: catalog[0].encryptedValue }], + }, + [{ value: canary }] + ) + await writeFile(mutationPath, JSON.stringify({ committed: true, completed: true })) + return Response.json({ data: { value: canary } }, { status: 201 }) + }, + 'POST' + ) + expect(JSON.parse(await readFile(mutationPath, 'utf8'))).toEqual({ + committed: true, + completed: true, + }) + expect(response.status).toBe(502) + const body = await response.text() + expect(body).not.toContain(canary) + expect(body).toContain('completed with HTTP 201') + expect(body).toContain('Do not retry a mutation automatically') + }) + + it.each(['file', 'table'] as const)( + 'preserves an explicit unknown %s delivery as unknown', + async (source) => { + await sandboxApi( + `/api/v2/${source === 'file' ? 'files/fixture' : 'tables/fixture/rows'}`, + async () => { + if (source === 'file') await reportWorkspaceFileDelivery({ status: 'unknown' }) + else + await reportTableRowDelivery({ version: 1, complete: false, entries: [] }, [ + { value: 'unknown' }, + ]) + return new Response('unknown') + } + ) + const result = await run('printf "ready"') + expect(result.projected.safe).toBe(false) + expect(JSON.stringify(result.projected.result)).not.toContain('ready') + } + ) +}) + describe('persistent workbench output confidentiality', () => { it.each(['javascript', 'shell'] as const)( 'redacts session credentials in %s output while preserving routing metadata', diff --git a/apps/sim/lib/mothership/tools/sandbox-resource-transport.test.ts b/apps/sim/lib/mothership/tools/sandbox-resource-transport.test.ts index 952b40b5f3b..2f0659341dc 100644 --- a/apps/sim/lib/mothership/tools/sandbox-resource-transport.test.ts +++ b/apps/sim/lib/mothership/tools/sandbox-resource-transport.test.ts @@ -90,10 +90,6 @@ describe('private sandbox v2 resource transport', () => { token ) expect(await response.json()).toEqual({ data: { inserted: 1 } }) - expect(recordInput).toHaveBeenCalledWith('mothership-chat:chat', false) - expect(recordInput.mock.invocationCallOrder[0]).toBeLessThan( - fetcher.mock.invocationCallOrder[0]! - ) expect(fetcher).toHaveBeenCalledTimes(1) expect(recordEffects).toHaveBeenCalledWith(token, scope, [ { @@ -261,14 +257,6 @@ vi.mock('@/lib/execution/remote-sandbox/session-file-provenance', () => ({ recordExistingSessionFileInput: recordInput, })) -it('refuses delivery before dispatch when provenance cannot be recorded', async () => { - recordInput.mockRejectedValueOnce(new Error('storage unavailable')) - await expect(proxySandboxResourceRequest(request('/api/v2/tables/table'), token)).rejects.toThrow( - 'storage unavailable' - ) - expect(fetcher).not.toHaveBeenCalled() -}) - it('does not poison public scratch after authenticated static catalog discovery', async () => { routeMatcher.mockReturnValue({ params: { toolId: 'function_execute' }, @@ -333,40 +321,3 @@ it('keeps the server identity out of callback response headers and body', async expect(JSON.stringify([...response.headers])).not.toContain('server-only-identity') expect(await response.text()).not.toContain('server-only-identity') }) - -it.each(['builtin', 'custom'] as const)( - 'preserves public research scratch only for producer-classified %s block catalog content', - async (source) => { - routeMatcher.mockReturnValue({ params: {}, load: async () => ({ GET: fetcher }) }) - fetcher.mockResolvedValue( - Response.json({ - data: [ - { - id: 'agent', - name: 'Agent', - description: 'Build an agent', - category: 'blocks', - source, - triggerAllowed: false, - triggerCapable: false, - triggerIds: [], - toolIds: [], - operationIds: [], - preview: false, - tags: [], - }, - ], - nextCursor: null, - }) - ) - expect((await proxySandboxResourceRequest(request('/api/v2/blocks'), token)).status).toBe(200) - if (source === 'builtin') expect(recordInput).not.toHaveBeenCalled() - else expect(recordInput).toHaveBeenCalledWith('mothership-chat:chat', false) - } -) -it('does not trust a source label in a malformed catalog result', async () => { - routeMatcher.mockReturnValue({ params: {}, load: async () => ({ GET: fetcher }) }) - fetcher.mockResolvedValue(Response.json({ data: [{ source: 'builtin' }], nextCursor: null })) - expect((await proxySandboxResourceRequest(request('/api/v2/blocks'), token)).status).toBe(200) - expect(recordInput).toHaveBeenCalledWith('mothership-chat:chat', false) -}) diff --git a/apps/sim/lib/mothership/tools/sandbox-resource-transport.ts b/apps/sim/lib/mothership/tools/sandbox-resource-transport.ts index 635dbf15591..e35d6ffaf9b 100644 --- a/apps/sim/lib/mothership/tools/sandbox-resource-transport.ts +++ b/apps/sim/lib/mothership/tools/sandbox-resource-transport.ts @@ -1,20 +1,18 @@ import { createLogger } from '@sim/logger' import { generateId } from '@sim/utils/id' import { NextRequest } from 'next/server' -import { - v2GetBlockContract, - v2GetToolContract, - v2ListBlocksContract, - v2ListConnectorTypesContract, - v2ListToolsContract, -} from '@/lib/api/contracts/v2/catalog' import { v2DownloadFileContract, v2ReadFileTextContract } from '@/lib/api/contracts/v2/files' import { markCopilotRequest } from '@/lib/api/server/routes/copilot-request' import { matchV2Route } from '@/lib/api/server/routes/in-process-transport' import { withWorkspaceInvocationScope } from '@/lib/core/application/workspace-invocation-scope' import { asOrchestrationError, statusForOrchestrationError } from '@/lib/core/orchestration/types' import { getInternalApiBaseUrl } from '@/lib/core/utils/urls' -import type { DurableSecretProvenance } from '@/lib/execution/durable-secret-provenance' +import { + type DurableSecretProvenance, + durableSecretProvenanceFromEnvelope, + EXACT_EMPTY_DURABLE_SECRET_PROVENANCE, + mergeDurableSecretProvenance, +} from '@/lib/execution/durable-secret-provenance' import { recordExistingSessionFileInput } from '@/lib/execution/remote-sandbox/session-file-provenance' import { createResourceEffectTransport } from '@/lib/mothership/agent-cli/resource-effects' import { resolveInvocationWorkspace } from '@/lib/mothership/application/workspace-target' @@ -24,6 +22,7 @@ import { recordSandboxResourceEffects, } from '@/lib/mothership/tools/sandbox-resources' import { chatSandboxSessionKey } from '@/lib/mothership/tools/sandbox-session-key' +import { observeTableRowDelivery } from '@/lib/table/application/row-delivery-observer' import { observeWorkspaceFileDelivery } from '@/lib/workspace-files/application/file-delivery-observer' const logger = createLogger('MothershipSandboxResourceTransport') @@ -101,6 +100,7 @@ async function proxyAuthorizedSandboxRequest( const forwarded = new Request(target, init) const effects: ResourceChange[] = [] let response: Response | undefined + let rowProvenance: DurableSecretProvenance | undefined let dispatched = false /** Invoke the same handler with private request identity; the declared use case still authorizes current domain access. */ const matched = matchV2Route(path) @@ -122,23 +122,6 @@ async function proxyAuthorizedSandboxRequest( if (!(result instanceof Response)) throw new Error('Invalid sandbox API response') return result } - // Only these producer-owned catalog responses contain no workspace data or execution output. - const publicCatalog = - method === 'GET' && - [v2ListToolsContract, v2GetToolContract, v2ListConnectorTypesContract].some( - (contract) => - contract.path.replace(/\[([^\]]+)\]/g, (_match, key) => - encodeURIComponent(matched.params[key] ?? '') - ) === path - ) - const blockCatalog = - method === 'GET' && - [v2ListBlocksContract, v2GetBlockContract].find( - (contract) => - contract.path.replace(/\[([^\]]+)\]/g, (_match, key) => - encodeURIComponent(matched.params[key] ?? '') - ) === path - ) const recordInput = (provenance: boolean | DurableSecretProvenance) => recordExistingSessionFileInput(chatSandboxSessionKey(scope.chatId), provenance) const fileRead = @@ -151,21 +134,31 @@ async function proxyAuthorizedSandboxRequest( ) const deliver = async () => { dispatched = true - if (!publicCatalog && !fileRead && !blockCatalog) await recordInput(false) - let observed = false - const result = await observeWorkspaceFileDelivery(async (provenance) => { - await recordInput(provenance?.status === 'exact' ? provenance : false) - observed = true - }, dispatch) + let fileObserved = false + const result = await observeTableRowDelivery( + async (provenance, _values, extras) => { + rowProvenance = mergeDurableSecretProvenance( + rowProvenance ?? EXACT_EMPTY_DURABLE_SECRET_PROVENANCE, + extras.unprovenancedErrorText + ? { status: 'unknown' } + : durableSecretProvenanceFromEnvelope(provenance) + ) + }, + () => + observeWorkspaceFileDelivery(async (provenance) => { + await recordInput(provenance?.status === 'exact' ? provenance : false) + fileObserved = true + }, dispatch) + ) try { - if (fileRead && !observed && result.ok && result.body) await recordInput(false) - if (blockCatalog && result.ok && result.body) { - const parsed = blockCatalog.response.schema.safeParse(await result.clone().json()) - const data = parsed.success ? parsed.data.data : undefined - const safe = Array.isArray(data) - ? data.every((block) => block.source === 'builtin') - : data?.source === 'builtin' - if (!safe) await recordInput(false) + if (fileRead && !fileObserved && result.ok && result.body) await recordInput(false) + else if (!fileObserved && !rowProvenance) { + /** Missing producer evidence is unrecorded, not proof that the machine received a secret. */ + logger.warn('Sandbox API response has no recorded secret provenance', { + method, + route: matched.pattern, + toolCallId: scope.toolCallId, + }) } } catch (error) { await result.body?.cancel().catch(() => {}) @@ -219,6 +212,24 @@ async function proxyAuthorizedSandboxRequest( } }) ) + if (rowProvenance) { + // Row mutations have already committed; admission must not interrupt completion or effects. + try { + await recordInput(rowProvenance) + } catch { + await response.body?.cancel().catch(() => {}) + logger.warn('Sandbox response provenance could not be recorded after API completion', { + toolCallId: scope.toolCallId, + status: response.status, + }) + response = Response.json( + { + error: `API request completed with HTTP ${response.status}, but its result could not be returned safely. Do not retry a mutation automatically; read the resource to check its current state.`, + }, + { status: 502 } + ) + } + } return new Response(request.method === 'HEAD' ? null : response.body, { status: response.status, statusText: response.statusText, From 045b07ccd2bd1d10916cb05e41504a1f8defd205 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 13:13:30 -0700 Subject: [PATCH 12/31] fix(mothership): stop duplicating chat resource tabs in workspace chats (#8537) * fix(mothership): stop duplicating chat resource tabs in workspace chats * fix(mothership): scope tool side-effect resources inside handleResourceSideEffects * fix(mothership): assert emitted resource events instead of mock calls --- .../contracts/mothership-resource-tools.ts | 1 - apps/sim/lib/mothership/agent-cli/index.ts | 3 +- .../agent-cli/resource-owner.test.ts | 7 ++- .../lib/mothership/request/tools/executor.ts | 5 +- .../request/tools/resources.test.ts | 55 ++++++++++--------- .../lib/mothership/request/tools/resources.ts | 4 +- .../registry/server-tool-adapter.test.ts | 4 +- .../tools/server/open-resource.test.ts | 1 - .../mothership/tools/server/open-resource.ts | 2 +- 9 files changed, 45 insertions(+), 37 deletions(-) diff --git a/apps/sim/lib/api/contracts/mothership-resource-tools.ts b/apps/sim/lib/api/contracts/mothership-resource-tools.ts index 08de9dba462..c7187963192 100644 --- a/apps/sim/lib/api/contracts/mothership-resource-tools.ts +++ b/apps/sim/lib/api/contracts/mothership-resource-tools.ts @@ -29,7 +29,6 @@ export const openResourceOutputSchema = z.object({ type: z.enum(['workflow', 'table', 'knowledgebase', 'file', 'dashboard', 'log']), id: z.string(), title: z.string(), - workspaceId: z.string(), viewId: z.string().optional(), executionId: z.string().optional(), }) diff --git a/apps/sim/lib/mothership/agent-cli/index.ts b/apps/sim/lib/mothership/agent-cli/index.ts index 7f8b904b0b3..324a36dd401 100644 --- a/apps/sim/lib/mothership/agent-cli/index.ts +++ b/apps/sim/lib/mothership/agent-cli/index.ts @@ -177,7 +177,8 @@ async function executeBoundAgentCliRequest( } else throw new Error('Service invocation must use the service bridge') if (resources.length) result = { ...result, resources: [...resources, ...(result.resources ?? [])] } - if (result.resources?.length) + // Only organization chats address a workspace; a workspace chat's resources leave it implicit. + if (context.chatOrganizationId && result.resources?.length) result = { ...result, resources: result.resources.map((effect) => { diff --git a/apps/sim/lib/mothership/agent-cli/resource-owner.test.ts b/apps/sim/lib/mothership/agent-cli/resource-owner.test.ts index 2f836fa2041..7b6965d0ad7 100644 --- a/apps/sim/lib/mothership/agent-cli/resource-owner.test.ts +++ b/apps/sim/lib/mothership/agent-cli/resource-owner.test.ts @@ -85,7 +85,7 @@ beforeEach(() => { }) describe('scoped CLI resource ownership', () => { - it('gives a workspace upload completion and explicit open the same resource identity', async () => { + it('gives a workspace upload completion, explicit open, and an open tab one identity', async () => { const upload = await executeAgentCliRequest( { invocation: { kind: 'cli', argv: ['files', 'upload', '@/tmp/upload-acceptance.txt'] } }, context @@ -102,9 +102,10 @@ describe('scoped CLI resource ownership', () => { id: fileId, title: file.name, path: 'files/upload-acceptance.txt', - workspaceId: first, }) - expect(getChatResourceKey(resource)).toBe(getChatResourceKey(opened.resources[0])) + const openTab = getChatResourceKey({ type: 'file', id: fileId }) + expect(getChatResourceKey(resource)).toBe(openTab) + expect(getChatResourceKey(opened.resources[0])).toBe(openTab) expect(mocks.transport).toHaveBeenCalledTimes(1) }) diff --git a/apps/sim/lib/mothership/request/tools/executor.ts b/apps/sim/lib/mothership/request/tools/executor.ts index fabe535218e..4262b930af0 100644 --- a/apps/sim/lib/mothership/request/tools/executor.ts +++ b/apps/sim/lib/mothership/request/tools/executor.ts @@ -1040,7 +1040,10 @@ async function executeToolAndReportInner( execContext.chatId, options?.onEvent, () => abortRequested(context, execContext, options), - toolCall.targetWorkspaceId ?? execContext.workspaceId, + { + organizationId: execContext.organizationId, + workspaceId: toolCall.targetWorkspaceId ?? execContext.workspaceId, + }, execContext.userId ) } diff --git a/apps/sim/lib/mothership/request/tools/resources.test.ts b/apps/sim/lib/mothership/request/tools/resources.test.ts index a7a700f16a2..659f9572895 100644 --- a/apps/sim/lib/mothership/request/tools/resources.test.ts +++ b/apps/sim/lib/mothership/request/tools/resources.test.ts @@ -24,6 +24,7 @@ vi.mock('@/lib/mothership/resources/persistence', () => ({ import { MothershipStreamV1EventType } from '@/lib/mothership/generated/mothership-stream-v1' import { handleResourceSideEffects } from '@/lib/mothership/request/tools/resources' +import type { StreamEvent } from '@/lib/mothership/request/types' import type { MothershipResource } from '@/lib/mothership/resources/types' describe('handleResourceSideEffects', () => { @@ -97,31 +98,35 @@ describe('handleResourceSideEffects', () => { }) } ) - it.each(['workspace-a'])('addresses extracted exports to admitted %s', async (workspaceId) => { - const resource = { type: 'file' as const, id: 'export', title: 'decisions.csv' } - mocks.extractResourcesFromToolResult.mockReturnValue([resource]) - const onEvent = vi.fn() - await handleResourceSideEffects( - 'run_function', - undefined, - { success: true, output: {} }, - { success: true, output: {} }, - 'org-chat', - onEvent, - () => false, - workspaceId - ) - expect(mocks.persistChatResources).toHaveBeenCalledWith('org-chat', [ - { ...resource, workspaceId }, - ]) - expect(onEvent).toHaveBeenCalledWith({ - type: 'resource', - payload: { - op: 'upsert', - resource: { ...resource, workspaceId }, - }, - }) - }) + it.each([ + { chat: 'organization', organizationId: 'org', expected: { workspaceId: 'workspace-a' } }, + { chat: 'workspace', organizationId: undefined, expected: {} }, + ])( + 'addresses extracted exports in a $chat chat like its open tabs', + async ({ organizationId, expected }) => { + const resource = { type: 'file' as const, id: 'export', title: 'decisions.csv' } + mocks.extractResourcesFromToolResult.mockReturnValue([resource]) + const events: StreamEvent[] = [] + await handleResourceSideEffects( + 'run_function', + undefined, + { success: true, output: {} }, + { success: true, output: {} }, + 'chat', + (event) => { + events.push(event) + }, + () => false, + { organizationId, workspaceId: 'workspace-a' } + ) + expect(events).toEqual([ + { + type: 'resource', + payload: { op: 'upsert', resource: { ...resource, ...expected } }, + }, + ]) + } + ) }) it('emits authorized Search results beside the persisted address, never inside it', async () => { diff --git a/apps/sim/lib/mothership/request/tools/resources.ts b/apps/sim/lib/mothership/request/tools/resources.ts index a7c980ecad4..c74853cfda4 100644 --- a/apps/sim/lib/mothership/request/tools/resources.ts +++ b/apps/sim/lib/mothership/request/tools/resources.ts @@ -36,9 +36,11 @@ export async function handleResourceSideEffects( chatId: string, onEvent: ((event: StreamEvent) => void | Promise) | undefined, isAborted: () => boolean, - workspaceId?: string, + owner?: { organizationId?: string; workspaceId?: string }, actorUserId?: string ): Promise { + // Only organization chats address a workspace; a workspace chat's resources leave it implicit. + const workspaceId = owner?.organizationId ? owner.workspaceId : undefined // Cheap early exit so we don't emit a span for tools that can never // produce resources (most of them). The span only shows up for tools // that might actually do resource work. diff --git a/apps/sim/lib/mothership/tools/registry/server-tool-adapter.test.ts b/apps/sim/lib/mothership/tools/registry/server-tool-adapter.test.ts index 162ff99678c..428853aa89f 100644 --- a/apps/sim/lib/mothership/tools/registry/server-tool-adapter.test.ts +++ b/apps/sim/lib/mothership/tools/registry/server-tool-adapter.test.ts @@ -120,9 +120,7 @@ describe('server tool adapter authority boundary', () => { }) it('forwards canonical open-resource effects without replacing target assertions or injecting a workflow', async () => { - const resources = [ - { type: 'workflow', id: 'flow', title: 'Canonical', workspaceId: 'workspace-1' }, - ] + const resources = [{ type: 'workflow', id: 'flow', title: 'Canonical' }] mocks.routeExecution.mockResolvedValue({ resources }) const params = { workspaceId: 'asserted-workspace', diff --git a/apps/sim/lib/mothership/tools/server/open-resource.test.ts b/apps/sim/lib/mothership/tools/server/open-resource.test.ts index baae77a31d9..67610aab989 100644 --- a/apps/sim/lib/mothership/tools/server/open-resource.test.ts +++ b/apps/sim/lib/mothership/tools/server/open-resource.test.ts @@ -87,7 +87,6 @@ describe('resource opening authorization boundary', () => { 'Canonical knowledge', 'Workflow run', ]) - expect(result.resources.every((resource) => resource.workspaceId === 'ws-a')).toBe(true) for (const read of Object.values(mocks)) expect(read).toHaveBeenCalledWith( expect.objectContaining({ diff --git a/apps/sim/lib/mothership/tools/server/open-resource.ts b/apps/sim/lib/mothership/tools/server/open-resource.ts index c84e8543c3b..ad622bc329b 100644 --- a/apps/sim/lib/mothership/tools/server/open-resource.ts +++ b/apps/sim/lib/mothership/tools/server/open-resource.ts @@ -56,7 +56,7 @@ export const openResourceServerTool: BaseServerTool Date: Thu, 1 Oct 2026 13:24:58 -0700 Subject: [PATCH 13/31] chore(search): add paced projection replacement and data retirement (#8534) * chore(search): add paced projection replacement and data retirement * chore(search): keep retirement usage in the CLI * fix(db): fence schema push and require direct retirement sessions * fix(db): preserve test connection URL parameters --- packages/db/drizzle.config.ts | 2 + packages/db/maintenance/index.ts | 18 + .../search-retirement-health.test.ts | 266 +++++ .../maintenance/search-retirement-health.ts | 239 +++++ .../search-retirement.integration.ts | 836 +++++++++++++++ packages/db/maintenance/search-retirement.ts | 972 ++++++++++++++++++ packages/db/package.json | 1 + .../0027_retire_search_embeddings.ts | 41 +- packages/db/script-migrations/index.ts | 2 +- .../search-embedding-retirement.md | 179 ---- packages/db/scripts/push.test.ts | 5 - packages/db/scripts/push.ts | 122 ++- packages/db/scripts/retire-indexed-search.ts | 262 +++++ packages/db/vitest.config.ts | 7 +- 14 files changed, 2702 insertions(+), 250 deletions(-) create mode 100644 packages/db/maintenance/index.ts create mode 100644 packages/db/maintenance/search-retirement-health.test.ts create mode 100644 packages/db/maintenance/search-retirement-health.ts create mode 100644 packages/db/maintenance/search-retirement.integration.ts create mode 100644 packages/db/maintenance/search-retirement.ts delete mode 100644 packages/db/script-migrations/search-embedding-retirement.md create mode 100644 packages/db/scripts/retire-indexed-search.ts diff --git a/packages/db/drizzle.config.ts b/packages/db/drizzle.config.ts index 42d5b878d2e..cffe391ded6 100644 --- a/packages/db/drizzle.config.ts +++ b/packages/db/drizzle.config.ts @@ -20,5 +20,7 @@ export default { '!script_migrations', '!search_embedding_cleanup_progress', '!search_embedding_cleanup_targets', + '!search_retirement_*', + '!embedding_search_retirement_*', ], } satisfies Config diff --git a/packages/db/maintenance/index.ts b/packages/db/maintenance/index.ts new file mode 100644 index 00000000000..624527cdbe7 --- /dev/null +++ b/packages/db/maintenance/index.ts @@ -0,0 +1,18 @@ +export { + abortSearchRetirement, + advanceSearchRetirement, + beginSearchRetirementPurge, + cutoverSearchRetirement, + finalizeSearchRetirement, + getSearchRetirementStatus, + initializeSearchRetirement, + SearchRetirementError, + type SearchRetirementPhase, + type SearchRetirementStatus, +} from '@sim/db/maintenance/search-retirement' +export { + RetireSearchHealthError, + readSearchRetirementHealth, + readSearchRetirementHealthLimits, + type SearchRetirementHealthLimits, +} from '@sim/db/maintenance/search-retirement-health' diff --git a/packages/db/maintenance/search-retirement-health.test.ts b/packages/db/maintenance/search-retirement-health.test.ts new file mode 100644 index 00000000000..56ea78fedbe --- /dev/null +++ b/packages/db/maintenance/search-retirement-health.test.ts @@ -0,0 +1,266 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + RetireSearchHealthError, + readSearchRetirementHealth, + readSearchRetirementHealthLimits, + type SearchRetirementHealthLimits, +} from '@sim/db/maintenance/search-retirement-health' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const NOW = Date.parse('2026-10-01T12:00:00.000Z') +const LIMITS: SearchRetirementHealthLimits = { + databaseId: 'test-database', + maxReplicaLagBytes: 1_000, + maxReplicaLagSeconds: 2, + maxWalBytesPerSecond: 2_000, + maxDatabaseP95Ms: 100, + maxCpuPercent: 50, + minFreeStorageBytes: 5_000, + maxSampleAgeMs: 10_000, +} +const SAMPLE = { + observedAt: '2026-10-01T12:00:00.000Z', + databaseId: 'test-database', + healthy: true, + replicaLagBytes: 0, + replicaLagSeconds: 0, + walBytesPerSecond: 100, + databaseP95Ms: 10, + cpuPercent: 10, + freeStorageBytes: 10_000, + maintenanceAllowed: true, + cutoverAllowed: false, +} + +describe('Search retirement external health gate', () => { + let directory: string + let path: string + + beforeEach(async () => { + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(NOW) + directory = await mkdtemp(join(tmpdir(), 'search-retirement-health-')) + path = join(directory, 'health.json') + await writeFile(path, JSON.stringify(SAMPLE)) + }) + + afterEach(async () => { + vi.useRealTimers() + await rm(directory, { recursive: true, force: true }) + }) + + it('reads a replacement sample before allowing another page', async () => { + expect(await readSearchRetirementHealth(path, LIMITS)).toBe(NOW) + await writeFile(path, JSON.stringify({ ...SAMPLE, maintenanceAllowed: false })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'maintenance_refused', + }) + }) + + it('requires separate cutover approval while allowing ordinary maintenance', async () => { + expect(await readSearchRetirementHealth(path, LIMITS)).toBe(NOW) + await expect(readSearchRetirementHealth(path, LIMITS, { cutover: true })).rejects.toMatchObject( + { + reason: 'cutover_refused', + } + ) + await writeFile(path, JSON.stringify({ ...SAMPLE, cutoverAllowed: true })) + expect(await readSearchRetirementHealth(path, LIMITS, { cutover: true })).toBe(NOW) + }) + + it('rejects an operator CPU ceiling above one hundred percent', async () => { + await writeFile(path, JSON.stringify({ ...LIMITS, maxCpuPercent: 100.001 })) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + }) + + it.each(['missing', 'directory'] as const)('rejects a %s health file', async (kind) => { + await expect( + readSearchRetirementHealth(kind === 'missing' ? join(directory, 'absent') : directory, LIMITS) + ).rejects.toBeInstanceOf(RetireSearchHealthError) + }) + + it('accepts at most eight KiB, including UTF-8 bytes and trailing whitespace', async () => { + const encoded = JSON.stringify(SAMPLE) + await writeFile(path, encoded.padEnd(8_192, ' ')) + expect(await readSearchRetirementHealth(path, LIMITS)).toBe(NOW) + await writeFile(path, encoded.padEnd(8_193, ' ')) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'health_file_too_large', + }) + await writeFile(path, JSON.stringify({ ...SAMPLE, padding: 'é'.repeat(4_096) })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'health_file_too_large', + }) + }) + + it.each(['', '{', 'null', '[]', 'true', '{"observedAt":123}'])( + 'rejects malformed or incomplete wire content: %s', + async (content) => { + await writeFile(path, content) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + } + ) + + it('rejects invalid UTF-8 instead of substituting a replacement character', async () => { + const encoded = Buffer.from(JSON.stringify(SAMPLE)) + const location = encoded.indexOf('test-database') + encoded[location] = 0xff + await writeFile(path, encoded) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it.each([ + { healthy: 'true' }, + { maintenanceAllowed: 1 }, + { cutoverAllowed: 'true' }, + { cutoverAllowed: undefined }, + { databaseId: '' }, + { unexpected: true }, + { observedAt: null }, + { replicaLagBytes: -1 }, + { replicaLagSeconds: null }, + { walBytesPerSecond: '100' }, + { databaseP95Ms: -1 }, + { cpuPercent: null }, + { freeStorageBytes: -1 }, + ])('rejects malformed sample field %j', async (fields) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, ...fields })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it('rejects a nonfinite JSON number', async () => { + await writeFile( + path, + JSON.stringify(SAMPLE).replace('"replicaLagBytes":0', '"replicaLagBytes":1e999') + ) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it.each([ + '2026-10-01T12:00:00+00:00', + '2026-10-01 12:00:00Z', + '2026-02-30T12:00:00.000Z', + '2026-10-01T24:00:00.000Z', + ])('rejects a noncanonical or impossible UTC timestamp: %s', async (observedAt) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, observedAt })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'invalid_sample', + }) + }) + + it.each([ + ['2026-10-01T11:59:49.999Z', 'stale_sample'], + ['2026-10-01T12:00:01.001Z', 'future_sample'], + ])('rejects a sample outside the time budget: %s', async (observedAt, reason) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, observedAt })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ reason }) + }) + + it('does not reuse a previously fresh sample once it expires', async () => { + await readSearchRetirementHealth(path, LIMITS) + vi.setSystemTime(NOW + LIMITS.maxSampleAgeMs + 1) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ + reason: 'stale_sample', + }) + }) + + it.each([ + [{ databaseId: 'other-database' }, 'identity_mismatch'], + [{ healthy: false }, 'unhealthy'], + [{ maintenanceAllowed: false }, 'maintenance_refused'], + [{ replicaLagBytes: 1_001 }, 'replica_lag_bytes'], + [{ replicaLagSeconds: 2.001 }, 'replica_lag_seconds'], + [{ walBytesPerSecond: 2_001 }, 'wal_rate'], + [{ databaseP95Ms: 101 }, 'database_latency'], + [{ cpuPercent: 51 }, 'cpu_usage'], + [{ freeStorageBytes: 4_999 }, 'storage_headroom'], + ] as const)('refuses unsafe sample %j', async (fields, reason) => { + await writeFile(path, JSON.stringify({ ...SAMPLE, ...fields })) + await expect(readSearchRetirementHealth(path, LIMITS)).rejects.toMatchObject({ reason }) + }) + + it('permits zero-lag limits and refuses any replication debt', async () => { + const limits = { ...LIMITS, maxReplicaLagBytes: 0, maxReplicaLagSeconds: 0 } + await readSearchRetirementHealth(path, limits) + await writeFile(path, JSON.stringify({ ...SAMPLE, replicaLagBytes: 1 })) + await expect(readSearchRetirementHealth(path, limits)).rejects.toMatchObject({ + reason: 'replica_lag_bytes', + }) + }) + + it.each([ + { databaseId: '' }, + { maxReplicaLagBytes: -1 }, + { maxReplicaLagSeconds: Number.NaN }, + { maxWalBytesPerSecond: 0 }, + { maxDatabaseP95Ms: Number.POSITIVE_INFINITY }, + { maxCpuPercent: 0 }, + { minFreeStorageBytes: 0 }, + { maxSampleAgeMs: 0 }, + { maxSampleAgeMs: 30_001 }, + ])('rejects limits that disable a guard: %j', async (fields) => { + await expect(readSearchRetirementHealth(path, { ...LIMITS, ...fields })).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + await writeFile(path, JSON.stringify({ ...LIMITS, ...fields })) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + }) + + it('loads the bounded operator policy and enforces it against the health sample', async () => { + const policyPath = join(directory, 'policy.json') + await writeFile(policyPath, JSON.stringify({ ...LIMITS, maxReplicaLagBytes: 0 })) + const limits = await readSearchRetirementHealthLimits(policyPath) + await writeFile(path, JSON.stringify({ ...SAMPLE, replicaLagBytes: 1 })) + await expect(readSearchRetirementHealth(path, limits)).rejects.toMatchObject({ + reason: 'replica_lag_bytes', + }) + }) + + it.each(['{}', 'null', '{', JSON.stringify({ ...LIMITS, unexpected: true })])( + 'rejects malformed operator policy content: %s', + async (content) => { + await writeFile(path, content) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'invalid_limits', + }) + } + ) + + it('enforces the same eight-KiB cap on operator policies', async () => { + await writeFile(path, JSON.stringify(LIMITS).padEnd(8_193, ' ')) + await expect(readSearchRetirementHealthLimits(path)).rejects.toMatchObject({ + reason: 'health_file_too_large', + }) + }) + + it('returns only typed generic reasons for payload and filesystem failures', async () => { + const privateMarker = 'private-operator-marker' + await writeFile(path, JSON.stringify({ ...SAMPLE, databaseId: privateMarker })) + for (const candidate of [path, join(directory, privateMarker)]) { + try { + await readSearchRetirementHealth(candidate, LIMITS) + expect.fail('The health gate should refuse this sample') + } catch (error) { + expect(error).toBeInstanceOf(RetireSearchHealthError) + if (!(error instanceof RetireSearchHealthError)) throw error + expect(error.message).not.toContain(privateMarker) + expect(error.message).not.toContain(directory) + expect(error.cause).toBeUndefined() + } + } + }) +}) diff --git a/packages/db/maintenance/search-retirement-health.ts b/packages/db/maintenance/search-retirement-health.ts new file mode 100644 index 00000000000..61d18109820 --- /dev/null +++ b/packages/db/maintenance/search-retirement-health.ts @@ -0,0 +1,239 @@ +import { constants } from 'node:fs' +import { open } from 'node:fs/promises' +import { isRecordLike } from '@sim/utils/object' + +const MAX_FILE_BYTES = 8_192 +const MAX_SAMPLE_AGE_MS = 30_000 +const MAX_FUTURE_SKEW_MS = 1_000 + +export interface SearchRetirementHealthLimits { + databaseId: string + maxReplicaLagBytes: number + maxReplicaLagSeconds: number + maxWalBytesPerSecond: number + maxDatabaseP95Ms: number + maxCpuPercent: number + minFreeStorageBytes: number + maxSampleAgeMs: number +} + +interface SearchRetirementHealthSample { + observedAt: string + databaseId: string + healthy: boolean + replicaLagBytes: number + replicaLagSeconds: number + walBytesPerSecond: number + databaseP95Ms: number + cpuPercent: number + freeStorageBytes: number + maintenanceAllowed: boolean + cutoverAllowed: boolean +} + +type HealthRefusalReason = + | 'health_file_unreadable' + | 'health_file_too_large' + | 'health_file_changed' + | 'invalid_sample' + | 'invalid_limits' + | 'identity_mismatch' + | 'stale_sample' + | 'future_sample' + | 'unhealthy' + | 'maintenance_refused' + | 'cutover_refused' + | 'replica_lag_bytes' + | 'replica_lag_seconds' + | 'wal_rate' + | 'database_latency' + | 'cpu_usage' + | 'storage_headroom' + +/** A safe refusal reason that never includes operator paths, database identity, or sample values. */ +export class RetireSearchHealthError extends Error { + constructor(readonly reason: HealthRefusalReason) { + super(`Search retirement health gate refused: ${reason}`) + this.name = 'RetireSearchHealthError' + } +} + +const SAMPLE_FIELDS = [ + 'observedAt', + 'databaseId', + 'healthy', + 'replicaLagBytes', + 'replicaLagSeconds', + 'walBytesPerSecond', + 'databaseP95Ms', + 'cpuPercent', + 'freeStorageBytes', + 'maintenanceAllowed', + 'cutoverAllowed', +] as const + +const LIMIT_FIELDS = [ + 'databaseId', + 'maxReplicaLagBytes', + 'maxReplicaLagSeconds', + 'maxWalBytesPerSecond', + 'maxDatabaseP95Ms', + 'maxCpuPercent', + 'minFreeStorageBytes', + 'maxSampleAgeMs', +] as const + +function isNonnegativeFinite(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) && value >= 0 +} + +function isPositiveFinite(value: unknown): value is number { + return isNonnegativeFinite(value) && value > 0 +} + +function hasExactFields(value: Record, fields: readonly string[]): boolean { + return Object.keys(value).length === fields.length && fields.every((field) => field in value) +} + +function assertLimits(value: unknown): asserts value is SearchRetirementHealthLimits { + if ( + !isRecordLike(value) || + !hasExactFields(value, LIMIT_FIELDS) || + typeof value.databaseId !== 'string' || + value.databaseId.trim().length === 0 || + !isNonnegativeFinite(value.maxReplicaLagBytes) || + !isNonnegativeFinite(value.maxReplicaLagSeconds) || + !isPositiveFinite(value.maxWalBytesPerSecond) || + !isPositiveFinite(value.maxDatabaseP95Ms) || + !isPositiveFinite(value.maxCpuPercent) || + value.maxCpuPercent > 100 || + !isPositiveFinite(value.minFreeStorageBytes) || + !isPositiveFinite(value.maxSampleAgeMs) || + value.maxSampleAgeMs > MAX_SAMPLE_AGE_MS + ) { + throw new RetireSearchHealthError('invalid_limits') + } +} + +function assertSample(value: unknown): asserts value is SearchRetirementHealthSample { + if ( + !isRecordLike(value) || + !hasExactFields(value, SAMPLE_FIELDS) || + typeof value.observedAt !== 'string' || + typeof value.databaseId !== 'string' || + value.databaseId.trim().length === 0 || + typeof value.healthy !== 'boolean' || + typeof value.maintenanceAllowed !== 'boolean' || + typeof value.cutoverAllowed !== 'boolean' || + !isNonnegativeFinite(value.replicaLagBytes) || + !isNonnegativeFinite(value.replicaLagSeconds) || + !isNonnegativeFinite(value.walBytesPerSecond) || + !isNonnegativeFinite(value.databaseP95Ms) || + !isNonnegativeFinite(value.cpuPercent) || + !isNonnegativeFinite(value.freeStorageBytes) + ) { + throw new RetireSearchHealthError('invalid_sample') + } +} + +async function readBoundedJson( + path: string, + invalidReason: 'invalid_sample' | 'invalid_limits' +): Promise { + try { + const file = await open(path, constants.O_RDONLY | constants.O_NONBLOCK) + try { + const before = await file.stat() + if (!before.isFile()) throw new RetireSearchHealthError('health_file_unreadable') + if (before.size > MAX_FILE_BYTES) throw new RetireSearchHealthError('health_file_too_large') + + const bytes = Buffer.alloc(MAX_FILE_BYTES) + let length = 0 + while (length < bytes.length) { + const read = await file.read(bytes, length, bytes.length - length, length) + if (read.bytesRead === 0) break + length += read.bytesRead + } + + const after = await file.stat() + if (after.size > MAX_FILE_BYTES) throw new RetireSearchHealthError('health_file_too_large') + if ( + before.size !== after.size || + before.mtimeMs !== after.mtimeMs || + before.ctimeMs !== after.ctimeMs || + length !== after.size + ) { + throw new RetireSearchHealthError('health_file_changed') + } + + try { + const text = new TextDecoder('utf-8', { fatal: true }).decode(bytes.subarray(0, length)) + const value: unknown = JSON.parse(text) + return value + } catch { + throw new RetireSearchHealthError(invalidReason) + } + } finally { + await file.close() + } + } catch (error) { + if (error instanceof RetireSearchHealthError) throw error + throw new RetireSearchHealthError('health_file_unreadable') + } +} + +/** Reads and validates the operator's bounded policy file without exposing its contents in errors. */ +export async function readSearchRetirementHealthLimits( + path: string +): Promise { + const limits = await readBoundedJson(path, 'invalid_limits') + assertLimits(limits) + return limits +} + +/** Reads a fresh external health sample before one bounded maintenance page; unknown health refuses work. */ +export async function readSearchRetirementHealth( + path: string, + limits: SearchRetirementHealthLimits, + options: { cutover?: boolean } = {} +): Promise { + assertLimits(limits) + const sample = await readBoundedJson(path, 'invalid_sample') + assertSample(sample) + if (!/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/.test(sample.observedAt)) { + throw new RetireSearchHealthError('invalid_sample') + } + const observedAt = Date.parse(sample.observedAt) + const canonicalTimestamp = sample.observedAt.includes('.') + ? sample.observedAt + : sample.observedAt.replace('Z', '.000Z') + if (!Number.isFinite(observedAt) || new Date(observedAt).toISOString() !== canonicalTimestamp) { + throw new RetireSearchHealthError('invalid_sample') + } + if (sample.databaseId !== limits.databaseId) + throw new RetireSearchHealthError('identity_mismatch') + const age = Date.now() - observedAt + if (age < -MAX_FUTURE_SKEW_MS) throw new RetireSearchHealthError('future_sample') + if (age > limits.maxSampleAgeMs) throw new RetireSearchHealthError('stale_sample') + if (!sample.healthy) throw new RetireSearchHealthError('unhealthy') + if (!sample.maintenanceAllowed) throw new RetireSearchHealthError('maintenance_refused') + if (options.cutover && !sample.cutoverAllowed) + throw new RetireSearchHealthError('cutover_refused') + if (sample.replicaLagBytes > limits.maxReplicaLagBytes) { + throw new RetireSearchHealthError('replica_lag_bytes') + } + if (sample.replicaLagSeconds > limits.maxReplicaLagSeconds) { + throw new RetireSearchHealthError('replica_lag_seconds') + } + if (sample.walBytesPerSecond > limits.maxWalBytesPerSecond) { + throw new RetireSearchHealthError('wal_rate') + } + if (sample.databaseP95Ms > limits.maxDatabaseP95Ms) { + throw new RetireSearchHealthError('database_latency') + } + if (sample.cpuPercent > limits.maxCpuPercent) throw new RetireSearchHealthError('cpu_usage') + if (sample.freeStorageBytes < limits.minFreeStorageBytes) { + throw new RetireSearchHealthError('storage_headroom') + } + return observedAt +} diff --git a/packages/db/maintenance/search-retirement.integration.ts b/packages/db/maintenance/search-retirement.integration.ts new file mode 100644 index 00000000000..0ab85b505ef --- /dev/null +++ b/packages/db/maintenance/search-retirement.integration.ts @@ -0,0 +1,836 @@ +import { spawn, spawnSync } from 'node:child_process' +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { runKnowledgeProjection } from '@sim/db/knowledge-projection' +import { + abortSearchRetirement, + advanceSearchRetirement, + beginSearchRetirementPurge, + cutoverSearchRetirement, + finalizeSearchRetirement, + getSearchRetirementStatus, + initializeSearchRetirement, +} from '@sim/db/maintenance/search-retirement' +import { readTestDatabaseUrl } from '@sim/db/testing/test-infrastructure' +import { generateId } from '@sim/utils/id' +import postgres, { type Sql } from 'postgres' +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +/** + * Real PostgreSQL proof: interrupted copy, late writes behind the cursor, deferred vectors, + * width preservation, generation races, marker invalidation, NOWAIT cutover, cached writers, + * destructive phase gates, and ordinary-KB isolation. The integration runner writes its JSON + * artifact to INTEGRATION_REPORT_PATH (CI uploads test-results/integration.json by default). + */ +describe('operator-driven Search retirement in PostgreSQL', () => { + const databaseUrl = readTestDatabaseUrl() + let admin: Sql + let sql: Sql + let writer: Sql + let database: string + const template = `sim_test_retirement_template_${generateId().replaceAll('-', '')}` + + beforeAll(async () => { + const source = postgres(databaseUrl, { max: 1, onnotice: () => undefined }) + let migrated: boolean + try { + const [row] = await source<{ migrated: boolean }[]>` + SELECT to_regclass('drizzle.__drizzle_migrations') IS NOT NULL AS migrated` + migrated = row.migrated + } finally { + await source.end() + } + const controlUrl = new URL(databaseUrl) + controlUrl.pathname = '/postgres' + const control = postgres(controlUrl.toString(), { max: 1, onnotice: () => undefined }) + try { + await control.unsafe(`CREATE DATABASE "${template}" TEMPLATE template0`) + } finally { + await control.end() + } + const url = new URL(databaseUrl) + url.pathname = `/${template}` + const setup = postgres(url.toString(), { max: 1, onnotice: () => undefined }) + try { + for (const extension of ['vector', 'btree_gin', 'pg_trgm']) { + await setup`CREATE EXTENSION IF NOT EXISTS ${setup(extension)}` + } + } finally { + await setup.end() + } + // A private template avoids cloning the shared fixture while other suites hold connections. + const provision = spawnSync( + 'bun', + ['--no-env-file', `./scripts/${migrated ? 'migrate' : 'push'}.ts`], + { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { + ...process.env, + DATABASE_URL: url.toString(), + MIGRATION_DATABASE_URL: url.toString(), + }, + encoding: 'utf8', + timeout: 90_000, + maxBuffer: 8 * 1_024 * 1_024, + } + ) + expect(provision.status, provision.stdout + provision.stderr).toBe(0) + }, 120_000) + + afterAll(async () => { + const url = new URL(databaseUrl) + url.pathname = '/postgres' + const control = postgres(url.toString(), { max: 1, onnotice: () => undefined }) + try { + await control.unsafe(`DROP DATABASE IF EXISTS "${template}"`) + } finally { + await control.end() + } + }) + + beforeEach(async () => { + database = `sim_test_retirement_${generateId().replaceAll('-', '')}` + const controlUrl = new URL(databaseUrl) + controlUrl.pathname = '/postgres' + admin = postgres(controlUrl.toString(), { max: 1, onnotice: () => undefined }) + await admin.unsafe(`CREATE DATABASE "${database}" TEMPLATE "${template.replaceAll('"', '""')}"`) + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const options = { max: 1, onnotice: () => undefined } + sql = postgres(fixtureUrl.toString(), { ...options, prepare: false }) + writer = postgres(fixtureUrl.toString(), { ...options, prepare: true }) + await sql`INSERT INTO "user" (id, name, email, email_verified, created_at, updated_at) + VALUES ('reader', 'Synthetic reader', 'reader@example.test', true, now(), now())` + await sql`INSERT INTO workspace (id, name, owner_id, billed_account_user_id) + VALUES ('workspace', 'Synthetic workspace', 'reader', 'reader')` + await sql`CREATE TABLE search_embedding_cleanup_progress ( + id integer PRIMARY KEY, knowledge_base_id text, phase text, after_id text)` + await sql`INSERT INTO search_embedding_cleanup_progress VALUES (1, 'search', 'embeddings', '')` + await sql`CREATE TABLE search_embedding_cleanup_targets (knowledge_base_id text PRIMARY KEY)` + await sql`INSERT INTO search_embedding_cleanup_targets VALUES ('search')` + const bases = [ + { id: 'prefix', width: 1536, model: 'text-embedding-3-small' }, + ...[384, 768, 1024, 1536, 3072].map((width) => ({ + id: `full-${width}`, + width, + model: 'full-width-fixture', + })), + { id: 'search', width: 1536, model: 'text-embedding-3-small' }, + ] + for (const base of bases) { + await sql`INSERT INTO knowledge_base (id, user_id, workspace_id, name, embedding_model, embedding_dimension, is_search_index) + VALUES (${base.id}, 'reader', 'workspace', ${base.id}, ${base.model}, ${base.width}, ${base.id === 'search'})` + await sql`INSERT INTO document (id, knowledge_base_id, filename, file_url, file_size, mime_type, + processing_queue_token, acl) + VALUES (${`${base.id}-doc`}, ${base.id}, 'fixture.txt', 'fixture', 10, 'text/plain', 'dispatch', ARRAY['u:reader@example.test'])` + const column = base.width === 1536 ? 'embedding' : `embedding_${base.width}` + await sql.unsafe( + `INSERT INTO embedding + (id, knowledge_base_id, document_id, chunk_index, chunk_hash, content, content_length, token_count, start_offset, end_offset, ${column}) + SELECT $1 || '-' || n, $1, $1 || '-doc', n, 'hash-' || n, 'Synthetic content', 17, 4, 0, 17, + array_fill(0.01::real, ARRAY[${base.width}])::vector(${base.width}) + FROM generate_series(1, 4) n`, + [base.id] + ) + } + // An older deferred writer can leave a canonical chunk without a projected vector. + await sql`DELETE FROM embedding_search WHERE id = 'prefix-4'` + }, 60_000) + + afterEach(async () => { + await writer?.end() + await sql?.end() + if (admin) { + await admin.unsafe(`DROP DATABASE IF EXISTS "${database}"`) + await admin.end() + } + }) + + async function nextPage(pageSize = 5) { + return advanceSearchRetirement(sql, { pageSize }) + } + + async function finishCopy() { + for (let page = 0; page < 100; page++) { + const state = await nextPage() + if (state.phase === 'ready') return state + } + throw new Error('Fixture retirement did not reach the cutover gate') + } + + it('resumes copy, includes late writes, replaces all widths, and only then purges Search chunks', async () => { + expect(await getSearchRetirementStatus(sql)).toBeNull() + await initializeSearchRetirement(sql) + for (let page = 0; page < 10; page++) { + if ((await nextPage()).copied > 0) break + } + expect((await getSearchRetirementStatus(sql))?.copied).toBeGreaterThan(0) + expect( + await sql`SELECT id FROM embedding_search_retirement_shadow WHERE id = 'full-1024-1'` + ).toHaveLength(1) + await writer`UPDATE embedding SET enabled = false WHERE id = 'full-1024-1'` + await writer`DELETE FROM embedding WHERE id = 'full-1024-3'` + await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, + content, content_length, token_count, start_offset, end_offset, embedding) + VALUES ('000-late', 'prefix', 'prefix-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, + array_fill(0.02::real, ARRAY[1536])::vector(1536))` + await initializeSearchRetirement(sql) + await finishCopy() + expect( + (await sql`SELECT count(*)::int AS n FROM embedding WHERE knowledge_base_id = 'search'`)[0].n + ).toBe(4) + await expect(beginSearchRetirementPurge(sql)).rejects.toThrow() + // Warm a prepared reader and trigger writer before replacing the relation. + const preparedRead = () => + writer`SELECT id, enabled FROM embedding_search WHERE id = 'prefix-1'` + await preparedRead() + await writer`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + await finishCopy() + await cutoverSearchRetirement(sql) + expect(await preparedRead()).toEqual([{ id: 'prefix-1', enabled: false }]) + await writer`UPDATE embedding SET enabled = true WHERE id = 'prefix-1'` + expect(await preparedRead()).toEqual([{ id: 'prefix-1', enabled: true }]) + expect((await sql`SELECT count(*)::int AS n FROM embedding_search`)[0].n).toBe(24) + expect( + await sql`SELECT id FROM embedding_search WHERE knowledge_base_id = 'search'` + ).toHaveLength(0) + expect(await sql`SELECT id FROM embedding_search WHERE id = 'prefix-4'`).toHaveLength(1) + expect(await sql`SELECT id FROM embedding_search WHERE id = 'full-1024-3'`).toHaveLength(0) + expect(await sql`SELECT enabled FROM embedding_search WHERE id = 'full-1024-1'`).toEqual([ + { enabled: false }, + ]) + expect( + ( + await sql`SELECT vector_dims(vector_512) AS width FROM embedding_search WHERE id = '000-late'` + )[0].width + ).toBe(512) + for (const width of [384, 768, 1024, 1536, 3072]) { + const column = width === 1536 ? 'vector' : `vector_${width}` + expect( + ( + await sql.unsafe( + `SELECT vector_dims(${column}) AS width FROM embedding_search WHERE id = $1`, + [`full-${width}-2`] + ) + )[0].width + ).toBe(width) + } + await expect(abortSearchRetirement(sql)).rejects.toThrow() + await beginSearchRetirementPurge(sql) + for (let page = 0; page < 100; page++) { + const state = await nextPage() + if (state.phase === 'finalize') break + } + expect((await getSearchRetirementStatus(sql))?.phase).toBe('finalize') + await finalizeSearchRetirement(sql) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('done') + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(24) + expect((await sql`SELECT count(*)::int AS n FROM knowledge_base`)[0].n).toBe(7) + expect( + ( + await sql`SELECT enabled, user_excluded, processing_queue_token FROM document WHERE id = 'search-doc'` + )[0] + ).toEqual({ enabled: false, user_excluded: true, processing_queue_token: null }) + expect( + ( + await sql`SELECT enabled, user_excluded, processing_queue_token, acl FROM document WHERE id = 'prefix-doc'` + )[0] + ).toEqual({ + enabled: true, + user_excluded: false, + processing_queue_token: 'dispatch', + acl: ['u:reader@example.test'], + }) + await writer`DELETE FROM embedding WHERE id = '000-late'` + expect(await sql`SELECT id FROM embedding_search WHERE id = '000-late'`).toHaveLength(0) + }) + + it('refuses cutover immediately while an ordinary reader holds the active projection', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + const [{ oid }] = await sql`SELECT 'embedding_search'::regclass::oid AS oid` + await writer`BEGIN` + try { + await writer`SELECT id FROM embedding_search LIMIT 1` + await expect(cutoverSearchRetirement(sql)).rejects.toMatchObject({ code: '55P03' }) + expect((await sql`SELECT 'embedding_search'::regclass::oid AS oid`)[0].oid).toBe(oid) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + } finally { + await writer`ROLLBACK` + } + await cutoverSearchRetirement(sql) + expect((await sql`SELECT 'embedding_search'::regclass::oid AS oid`)[0].oid).not.toBe(oid) + }) + + it('refuses incomplete copy and invalidates the job when a KB changes classification', async () => { + await initializeSearchRetirement(sql) + await expect(cutoverSearchRetirement(sql)).rejects.toThrow() + await writer`UPDATE knowledge_base SET is_search_index = false WHERE id = 'search'` + await expect(nextPage()).rejects.toThrow() + expect((await getSearchRetirementStatus(sql))?.invalidated).toBe(true) + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + await abortSearchRetirement(sql) + expect(await getSearchRetirementStatus(sql)).toBeNull() + await writer`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + false + ) + }) + + it('rolls back a failed page with its cursor and retries without losing or duplicating rows', async () => { + await initializeSearchRetirement(sql) + for (let n = 0; n < 20; n++) { + if ((await getSearchRetirementStatus(sql))?.phase === 'copy') break + await nextPage() + } + const before = await getSearchRetirementStatus(sql) + await writer`BEGIN` + try { + await writer`LOCK TABLE embedding IN ACCESS EXCLUSIVE MODE` + await expect(nextPage()).rejects.toMatchObject({ code: '55P03' }) + } finally { + await writer`ROLLBACK` + } + expect(await getSearchRetirementStatus(sql)).toEqual(before) + await finishCopy() + await cutoverSearchRetirement(sql) + expect((await sql`SELECT count(*)::int AS n FROM embedding_search`)[0].n).toBe(24) + }) + + it('refuses unknown inbound dependencies instead of silently redirecting only some readers', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await sql`CREATE VIEW external_projection_reader AS SELECT id FROM embedding_search` + await expect(cutoverSearchRetirement(sql)).rejects.toThrow() + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + await sql`DROP VIEW external_projection_reader` + await cutoverSearchRetirement(sql) + }) + + it('fences the old cursor-based command while replacement work exists', async () => { + await initializeSearchRetirement(sql) + await expect( + sql`UPDATE search_embedding_cleanup_progress SET after_id = 'late' WHERE id = 1` + ).rejects.toThrow() + expect( + (await sql`SELECT after_id FROM search_embedding_cleanup_progress WHERE id = 1`)[0].after_id + ).toBe('') + await abortSearchRetirement(sql) + await sql`UPDATE search_embedding_cleanup_progress SET after_id = 'late' WHERE id = 1` + }) + + it('refuses a repeatable-read snapshot taken before copy without any projection lock', async () => { + await writer`BEGIN ISOLATION LEVEL REPEATABLE READ` + try { + await writer`SELECT id FROM workspace LIMIT 1` + await initializeSearchRetirement(sql) + await finishCopy() + await expect(cutoverSearchRetirement(sql)).rejects.toThrow(/Transactions predate/) + expect( + ( + await sql`SELECT count(*)::int AS n FROM embedding_search WHERE knowledge_base_id = 'search'` + )[0].n + ).toBe(4) + } finally { + await writer`ROLLBACK` + } + await cutoverSearchRetirement(sql) + expect((await sql`SELECT count(*)::int AS n FROM embedding_search`)[0].n).toBe(24) + }) + + it('keeps deferred ordinary-vector repair working after cutover and metadata changes', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await cutoverSearchRetirement(sql) + await writer.begin(async (tx) => { + await tx`SET LOCAL sim.projection_mode = 'async'` + await tx`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + }) + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + true + ) + await runKnowledgeProjection(writer, { budgetMs: 5_000, pageSize: 2 }) + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + false + ) + await writer`UPDATE knowledge_base SET embedding_model = 'updated-full-width-fixture' WHERE id = 'full-1536'` + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await getSearchRetirementStatus(sql))?.phase).toBe('finalize') + await finalizeSearchRetirement(sql) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('done') + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(24) + }) + + it('retains a newer dirty generation written while its older image is being copied', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await writer`UPDATE embedding SET enabled = false WHERE id = 'prefix-1'` + const barrierUrl = new URL(databaseUrl) + barrierUrl.pathname = `/${database}` + const barrier = postgres(barrierUrl.toString(), { max: 1, onnotice: () => undefined }) + await sql`CREATE FUNCTION retirement_test_barrier() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + PERFORM set_config('lock_timeout', '1500ms', true); + PERFORM pg_advisory_xact_lock(791184); + RETURN NEW; + END $$` + await sql`CREATE TRIGGER retirement_test_barrier BEFORE INSERT OR UPDATE ON embedding_search_retirement_shadow + FOR EACH ROW EXECUTE FUNCTION retirement_test_barrier()` + await barrier`SELECT pg_advisory_lock(791184)` + const [{ pid }] = await sql`SELECT pg_backend_pid() AS pid` + const copy = Promise.allSettled([nextPage()]) + try { + await vi.waitFor( + async () => { + const [{ waiting }] = await barrier`SELECT EXISTS ( + SELECT 1 FROM pg_locks WHERE pid = ${pid} AND locktype = 'advisory' AND NOT granted + ) AS waiting` + expect(waiting).toBe(true) + }, + { interval: 5, timeout: 1_000 } + ) + await writer`UPDATE embedding SET enabled = true WHERE id = 'prefix-1'` + await barrier`SELECT pg_advisory_unlock(791184)` + const [result] = await copy + if (result.status === 'rejected') throw result.reason + expect( + await sql`SELECT embedding_id FROM search_retirement_changes WHERE embedding_id = 'prefix-1'` + ).toHaveLength(1) + expect( + (await sql`SELECT enabled FROM embedding_search_retirement_shadow WHERE id = 'prefix-1'`)[0] + .enabled + ).toBe(false) + } finally { + await barrier`SELECT pg_advisory_unlock_all()` + await copy + await barrier.end() + await sql`DROP TRIGGER retirement_test_barrier ON embedding_search_retirement_shadow` + await sql`DROP FUNCTION retirement_test_barrier()` + } + await finishCopy() + await cutoverSearchRetirement(sql) + expect((await sql`SELECT enabled FROM embedding_search WHERE id = 'prefix-1'`)[0].enabled).toBe( + true + ) + }) + + it('catches a late Search write behind the purge cursor before completing document retirement', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await cutoverSearchRetirement(sql) + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'documents') break + } + await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, + content, content_length, token_count, start_offset, end_offset, embedding) + VALUES ('000-late-search', 'search', 'search-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, + array_fill(0.02::real, ARRAY[1536])::vector(1536))` + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await getSearchRetirementStatus(sql))?.phase).toBe('finalize') + await finalizeSearchRetirement(sql) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('done') + expect(await sql`SELECT id FROM embedding WHERE knowledge_base_id = 'search'`).toHaveLength(0) + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(24) + expect( + await sql`SELECT tgname FROM pg_trigger WHERE tgrelid = 'embedding'::regclass + AND tgname = 'search_retirement_capture'` + ).toHaveLength(0) + }) + + it.each(['abort', 'begin-purge', 'finalize'] as const)( + 'rolls back %s if its implicit DROP lock conflicts with a reader', + async (command) => { + await initializeSearchRetirement(sql) + if (command !== 'abort') { + await finishCopy() + await cutoverSearchRetirement(sql) + } + if (command === 'finalize') { + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + } + const before = await getSearchRetirementStatus(sql) + const operation = + command === 'abort' + ? abortSearchRetirement + : command === 'begin-purge' + ? beginSearchRetirementPurge + : finalizeSearchRetirement + await writer`BEGIN` + try { + await writer`SELECT id FROM embedding LIMIT 1` + await expect(operation(sql)).rejects.toMatchObject({ code: '55P03' }) + expect(await getSearchRetirementStatus(sql)).toEqual(before) + expect(await writer`SELECT id FROM embedding_search WHERE id = 'prefix-1'`).toHaveLength(1) + } finally { + await writer`ROLLBACK` + } + await operation(sql) + expect((await getSearchRetirementStatus(sql))?.phase ?? null).toBe( + command === 'abort' ? null : command === 'begin-purge' ? 'purge' : 'done' + ) + } + ) + + it('returns from finalization to bounded purge when a late Search write arrives', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await cutoverSearchRetirement(sql) + await beginSearchRetirementPurge(sql) + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await nextPage()).phase).toBe('finalize') + expect( + await sql`SELECT tgname FROM pg_trigger WHERE tgrelid = 'embedding'::regclass + AND tgname = 'search_retirement_capture'` + ).toHaveLength(1) + await writer`UPDATE document SET enabled = true, user_excluded = false WHERE id = 'search-doc'` + await writer`INSERT INTO embedding (id, knowledge_base_id, document_id, chunk_index, chunk_hash, + content, content_length, token_count, start_offset, end_offset, embedding) + VALUES ('000-final-late-search', 'search', 'search-doc', 5, 'late', 'Late synthetic content', 22, 4, 0, 22, + array_fill(0.02::real, ARRAY[1536])::vector(1536))` + expect((await finalizeSearchRetirement(sql)).phase).toBe('purge') + for (let n = 0; n < 100; n++) { + if ((await nextPage()).phase === 'finalize') break + } + expect((await finalizeSearchRetirement(sql)).phase).toBe('done') + expect(await sql`SELECT id FROM embedding WHERE knowledge_base_id = 'search'`).toHaveLength(0) + expect( + (await sql`SELECT enabled, user_excluded FROM document WHERE id = 'search-doc'`)[0] + ).toEqual({ enabled: false, user_excluded: true }) + }) + + it('refuses a stored SQL function that would retain the old projection identity', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await sql`CREATE FUNCTION retirement_dependent_function() RETURNS bigint + LANGUAGE SQL RETURN (SELECT count(*) FROM embedding_search)` + await expect(cutoverSearchRetirement(sql)).rejects.toThrow(/dependencies/) + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + expect((await sql`SELECT retirement_dependent_function() AS n`)[0].n).toBe('27') + }) + + it('rejects a replacement index changed after preparation', async () => { + await initializeSearchRetirement(sql) + await finishCopy() + await sql`DROP INDEX embedding_search_retirement_shadow_512_hnsw_idx` + await sql`CREATE INDEX embedding_search_retirement_shadow_512_hnsw_idx + ON embedding_search_retirement_shadow (enabled)` + await expect(cutoverSearchRetirement(sql)).rejects.toThrow() + expect((await getSearchRetirementStatus(sql))?.phase).toBe('ready') + expect( + ( + await sql`SELECT count(*)::int AS n FROM embedding_search WHERE knowledge_base_id = 'search'` + )[0].n + ).toBe(4) + }) + + it('refuses a disabled canonical writer before creating maintenance objects', async () => { + await sql`ALTER TABLE embedding DISABLE TRIGGER embedding_search_sync` + await expect(initializeSearchRetirement(sql)).rejects.toThrow() + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + }) + + it('serializes with another maintenance operator without advancing its checkpoint', async () => { + await initializeSearchRetirement(sql) + const before = await getSearchRetirementStatus(sql) + await writer`SELECT pg_advisory_lock(hashtextextended('sim:search-retirement-maintenance', 0))` + try { + await expect(nextPage()).rejects.toThrow(/Another migration or retirement operation/) + } finally { + await writer`SELECT pg_advisory_unlock_all()` + } + expect(await getSearchRetirementStatus(sql)).toEqual(before) + }) + + it('refuses db:push before schema reconciliation once replacement retirement exists', async () => { + await initializeSearchRetirement(sql) + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const script = fileURLToPath(new URL('../scripts/push.ts', import.meta.url)) + const result = spawnSync('bun', ['--no-env-file', script, '--retirement-test-invalid-option'], { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { ...process.env, NODE_ENV: 'development', DATABASE_URL: fixtureUrl.toString() }, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + }) + expect(result.status).toBe(1) + expect(`${result.stdout}${result.stderr}`).toContain( + 'Schema push is disabled after Search retirement starts' + ) + expect(`${result.stdout}${result.stderr}`).not.toContain('Unrecognized options') + expect((await getSearchRetirementStatus(sql))?.phase).toBe('snapshot') + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + }) + + it('refuses db:push while retirement holds its maintenance fence before creating a receipt', async () => { + await writer`SELECT pg_advisory_lock(hashtextextended('sim:search-retirement-maintenance', 0))` + try { + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const result = spawnSync( + 'bun', + ['--no-env-file', './scripts/push.ts', '--retirement-test-invalid-option'], + { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { ...process.env, NODE_ENV: 'development', DATABASE_URL: fixtureUrl.toString() }, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + } + ) + expect(result.status).toBe(1) + expect(`${result.stdout}${result.stderr}`).toContain( + 'Another migration or retirement operation' + ) + expect(`${result.stdout}${result.stderr}`).not.toContain('Unrecognized options') + expect(await getSearchRetirementStatus(sql)).toBeNull() + } finally { + await writer`SELECT pg_advisory_unlock_all()` + } + }) + + it('fences retirement during db:push and stops database preparation when its fence connection closes', async () => { + await sql`ALTER TABLE workspace_files ADD COLUMN size bigint` + await writer`BEGIN` + await writer`LOCK TABLE workspace_files IN ACCESS EXCLUSIVE MODE` + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const child = spawn( + 'bun', + ['--no-env-file', './scripts/push.ts', '--force', '--retirement-test-invalid-option'], + { + cwd: fileURLToPath(new URL('..', import.meta.url)), + env: { ...process.env, NODE_ENV: 'development', DATABASE_URL: fixtureUrl.toString() }, + stdio: ['ignore', 'pipe', 'pipe'], + timeout: 10_000, + } + ) + let output = '' + child.stdout.on('data', (chunk: Buffer) => { + output += chunk.toString() + }) + child.stderr.on('data', (chunk: Buffer) => { + output += chunk.toString() + }) + const exited = new Promise((resolve, reject) => { + child.once('error', reject) + child.once('close', resolve) + }) + try { + await vi.waitFor( + async () => { + expect( + await sql`SELECT pid FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND query = 'LOCK TABLE public.workspace_files IN ACCESS EXCLUSIVE MODE'` + ).toHaveLength(1) + }, + { timeout: 5_000 } + ) + await expect(initializeSearchRetirement(sql)).rejects.toThrow( + /Another migration or retirement operation/ + ) + const [guard] = await sql`SELECT pid, backend_xmin FROM pg_stat_activity + WHERE datname = current_database() AND application_name = 'sim-db-push'` + expect(guard.backend_xmin).toBeNull() + await sql`SELECT pg_terminate_backend(${guard.pid})` + expect(await exited).toBe(1) + expect(output).toContain('Schema-push lock connection closed') + expect(output).not.toContain('Unrecognized options') + await writer`ROLLBACK` + await vi.waitFor(async () => { + expect( + await sql`SELECT pid FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND query = 'LOCK TABLE public.workspace_files IN ACCESS EXCLUSIVE MODE'` + ).toHaveLength(0) + }) + expect( + await sql`SELECT attname FROM pg_attribute + WHERE attrelid = 'workspace_files'::regclass AND attname = 'size' AND NOT attisdropped` + ).toHaveLength(1) + expect(await getSearchRetirementStatus(sql)).toBeNull() + } finally { + await writer`ROLLBACK` + await exited + } + }) + + it('the operator CLI refuses unverified endpoints and connection overrides before connecting', () => { + const script = fileURLToPath(new URL('../scripts/retire-indexed-search.ts', import.meta.url)) + const overrideUrl = new URL(databaseUrl) + overrideUrl.searchParams.set('statement_timeout', '0') + for (const url of [ + 'postgresql://reader@pool.example.invalid:5432/postgres', + 'postgresql://reader@fixture.pg.psdb.cloud:6432/postgres?sslmode=verify-full&sslrootcert=system', + 'postgresql://reader@fixture.pg.psdb.cloud.example.invalid:5432/postgres', + 'postgresql://reader@localhost:5432/production', + 'postgresql://reader@fixture.pg.psdb.cloud:5432/postgres?sslmode=disable', + 'postgresql://fixture.pg.psdb.cloud:5432/postgres?sslmode=verify-full&sslrootcert=system', + 'postgresql://reader@fixture.pg.psdb.cloud:5432/?sslmode=verify-full&sslrootcert=system', + overrideUrl.toString(), + ]) { + const result = spawnSync('bun', ['--no-env-file', script, 'identity'], { + env: { ...process.env, NODE_ENV: 'development', MIGRATION_DATABASE_URL: url }, + encoding: 'utf8', + timeout: 5_000, + maxBuffer: 64 * 1_024, + }) + expect(result.status).toBe(1) + expect(`${result.stdout}${result.stderr}`).not.toContain(url) + } + }) + + it('the CLI respects health gates, keeps status read-only, and stops after losing its session', async () => { + const fixtureUrl = new URL(databaseUrl) + fixtureUrl.pathname = `/${database}` + const script = fileURLToPath(new URL('../scripts/retire-indexed-search.ts', import.meta.url)) + const environment = { + ...process.env, + NODE_ENV: 'development', + MIGRATION_DATABASE_URL: fixtureUrl.toString(), + } + const invoke = (...args: string[]) => + spawnSync('bun', ['--no-env-file', script, ...args], { + env: environment, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + }) + expect(invoke('status').status).toBe(0) + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect(invoke('prepare', '--ack-release-drained').status).toBe(1) + expect(await getSearchRetirementStatus(sql)).toBeNull() + const identity = invoke('identity') + expect(identity.status).toBe(0) + const databaseId = identity.stdout.trim() + const directory = await mkdtemp(join(tmpdir(), 'retirement-cli-')) + try { + const policy = join(directory, 'policy.json') + const health = join(directory, 'health.json') + await writeFile( + policy, + JSON.stringify({ + databaseId, + maxReplicaLagBytes: 1, + maxReplicaLagSeconds: 1, + maxWalBytesPerSecond: 1_000, + maxDatabaseP95Ms: 100, + maxCpuPercent: 50, + minFreeStorageBytes: 100, + maxSampleAgeMs: 30_000, + }) + ) + const sample = { + databaseId, + observedAt: new Date().toISOString(), + healthy: false, + maintenanceAllowed: true, + cutoverAllowed: false, + replicaLagBytes: 0, + replicaLagSeconds: 0, + walBytesPerSecond: 0, + databaseP95Ms: 1, + cpuPercent: 1, + freeStorageBytes: 1_000, + } + await writeFile(health, JSON.stringify(sample)) + const refused = invoke( + 'prepare', + '--ack-release-drained', + '--health-file', + health, + '--health-policy', + policy + ) + expect(refused.status).toBe(1) + expect(`${refused.stdout}${refused.stderr}`).toContain('unhealthy') + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect(invoke('status').status).toBe(0) + await writeFile(health, JSON.stringify({ ...sample, healthy: true })) + expect( + invoke( + 'prepare', + '--ack-release-drained', + '--health-file', + health, + '--health-policy', + policy + ).status + ).toBe(0) + const runner = spawn( + 'bun', + [ + '--no-env-file', + script, + 'run', + '--pages', + '2', + '--health-file', + health, + '--health-policy', + policy, + ], + { + env: environment, + stdio: 'ignore', + timeout: 20_000, + } + ) + const exited = new Promise((resolve, reject) => { + runner.once('error', reject) + runner.once('close', resolve) + }) + try { + await vi.waitFor( + async () => { + expect( + (await sql`SELECT after_id FROM search_retirement_state WHERE id = 1`)[0].after_id + ).not.toBe('') + }, + { timeout: 5_000 } + ) + const before = await getSearchRetirementStatus(sql) + const [session] = await sql`SELECT pid FROM pg_stat_activity + WHERE datname = current_database() AND application_name = 'sim-search-data-retirement'` + await sql`SELECT pg_terminate_backend(${session.pid})` + const exitCode = await exited + expect(await getSearchRetirementStatus(sql)).toEqual(before) + expect(exitCode).toBe(1) + } finally { + runner.kill('SIGKILL') + await exited + } + await abortSearchRetirement(sql) + } finally { + await rm(directory, { recursive: true, force: true }) + } + const oldScript = fileURLToPath( + new URL('../script-migrations/0027_retire_search_embeddings.ts', import.meta.url) + ) + expect( + spawnSync('bun', ['--no-env-file', oldScript, '--maintenance'], { + env: environment, + encoding: 'utf8', + timeout: 10_000, + maxBuffer: 64 * 1_024, + }).status + ).toBe(1) + expect(await getSearchRetirementStatus(sql)).toBeNull() + expect((await sql`SELECT count(*)::int AS n FROM embedding`)[0].n).toBe(28) + }) +}) diff --git a/packages/db/maintenance/search-retirement.ts b/packages/db/maintenance/search-retirement.ts new file mode 100644 index 00000000000..6dac2422670 --- /dev/null +++ b/packages/db/maintenance/search-retirement.ts @@ -0,0 +1,972 @@ +import type { Sql, TransactionSql } from 'postgres' + +const STATE = 'public.search_retirement_state' +const SHADOW = 'public.embedding_search_retirement_shadow' +const BACKUP = 'public.embedding_search_retirement_backup' +const JOB_LOCK = 'sim:search-retirement-replacement' +const MAINTENANCE_LOCK = 'sim:search-retirement-maintenance' +const MIGRATION_LOCK = '4961002270' +const WIDTHS = [1536, 384, 512, 768, 1024, 3072] as const +const SOURCE_WIDTHS = [1536, 384, 768, 1024, 3072] as const +const column = (prefix: string, width: number) => (width === 1536 ? prefix : `${prefix}_${width}`) +const VECTOR_COLUMNS = WIDTHS.map((width) => column('vector', width)) +const BINARY_COLUMNS = SOURCE_WIDTHS.map((width) => column('binary', width)) +const PROJECTED_COLUMNS = [ + 'id', + 'knowledge_base_id', + 'document_id', + 'enabled', + ...BINARY_COLUMNS, + ...VECTOR_COLUMNS, +] +const SHARED_INDEXES = [ + { name: 'embedding_search_pkey', suffix: 'pkey', definition: 'UNIQUE (id)' }, + { name: 'embedding_search_kb_idx', suffix: 'kb_idx', definition: '(knowledge_base_id)' }, + { + name: 'embedding_search_document_lookup_idx', + suffix: 'document_lookup_idx', + definition: '(document_id, knowledge_base_id, id) WHERE enabled', + }, + ...WIDTHS.map((width) => ({ + name: + width === 1536 + ? 'embedding_search_cosine_hnsw_idx' + : `embedding_search_${width}_cosine_hnsw_idx`, + suffix: `${width}_hnsw_idx`, + definition: `USING hnsw (${column('vector', width)} halfvec_cosine_ops) WITH (m = 16, ef_construction = 64)`, + })), +] as const + +export type SearchRetirementPhase = + | 'snapshot' + | 'copy' + | 'catchup' + | 'validate-source' + | 'validate-shadow' + | 'ready' + | 'cutover' + | 'purge' + | 'documents' + | 'finalize' + | 'done' + +/** Deliberate operator messages never include database values, IDs, or statement parameters. */ +export class SearchRetirementError extends Error { + override name = 'SearchRetirementError' +} + +export interface SearchRetirementStatus { + phase: SearchRetirementPhase + invalidated: boolean + invalidationReason: string | null + sourceScanned: number + copied: number + reconciled: number + validatedSource: number + validatedShadow: number + purgedEmbeddings: number + retiredDocuments: number + pendingChanges: boolean + changeBacklogAtLimit: boolean + backupRetained: boolean +} + +interface StateRow { + version: number + phase: SearchRetirementPhase + resume_phase: 'validate-source' | 'ready' + after_id: string + invalidated: boolean + invalidation_reason: string | null + original_oid: number + replacement_oid: number + source_scanned: string + copied: string + reconciled: string + validated_source: string + validated_shadow: string + purged_embeddings: string + retired_documents: string + round_mutations: string + ready_at: Date | null +} + +interface PageResult { + after_id: string | null + scanned: number + changed: number +} + +const identifier = (value: string) => `"${value.replaceAll('"', '""')}"` +const columns = PROJECTED_COLUMNS.map(identifier).join(', ') +const assignments = PROJECTED_COLUMNS.filter((name) => name !== 'id') + .map((name) => `${identifier(name)} = EXCLUDED.${identifier(name)}`) + .join(', ') +const comparison = (left: string, right: string) => + `(${PROJECTED_COLUMNS.map((name) => `${left}.${identifier(name)}`).join(', ')}) IS DISTINCT FROM (${PROJECTED_COLUMNS.map((name) => `${right}.${identifier(name)}`).join(', ')})` + +const REPLACEMENT_SHAPE = `SELECT jsonb_build_object( + 'columns', (SELECT jsonb_agg(to_jsonb(a) ORDER BY a.attnum) FROM ( + SELECT a.attnum, a.attname, a.atttypid, a.atttypmod, a.attnotnull, a.attidentity, + a.attgenerated, a.attcollation, a.attacl, pg_get_expr(d.adbin, d.adrelid) AS default_value + FROM pg_attribute a LEFT JOIN pg_attrdef d ON d.adrelid = a.attrelid AND d.adnum = a.attnum + WHERE a.attrelid = c.oid AND a.attnum > 0 AND NOT a.attisdropped ORDER BY a.attnum LIMIT 32 + ) a), + 'constraints', (SELECT jsonb_agg(to_jsonb(k) ORDER BY k.conname) FROM ( + SELECT conname, contype, convalidated, condeferrable, condeferred, pg_get_constraintdef(oid) AS definition + FROM pg_constraint WHERE conrelid = c.oid ORDER BY conname LIMIT 32 + ) k), + 'table', jsonb_build_array(c.relkind, c.relpersistence, c.relrowsecurity, + c.relforcerowsecurity, c.relreplident, c.reloptions, c.reltablespace), + 'unexpected_dependencies', EXISTS (SELECT 1 FROM pg_constraint WHERE confrelid = c.oid) + OR EXISTS (SELECT 1 FROM pg_depend WHERE refobjid = c.oid + AND refclassid = 'pg_class'::regclass + AND classid IN ('pg_rewrite'::regclass, 'pg_proc'::regclass, 'pg_policy'::regclass)) + OR EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = c.oid OR inhparent = c.oid) + OR EXISTS (SELECT 1 FROM pg_policy WHERE polrelid = c.oid) + OR EXISTS (SELECT 1 FROM pg_trigger WHERE tgrelid = c.oid AND NOT tgisinternal) + OR EXISTS (SELECT 1 FROM pg_publication_tables WHERE schemaname = 'public' + AND tablename = 'embedding_search_retirement_shadow') +) FROM pg_class c WHERE c.oid = 'public.embedding_search_retirement_shadow'::regclass` + +/** Matches the current synchronous writer, including OpenAI's 512-dimensional prefix. */ +function projectedValues(source: string, model: string): string { + const shortened = `${model} IN ('text-embedding-3-small', 'text-embedding-3-large') AND ${source}.embedding_384 IS NULL` + return [ + `${source}.id`, + `${source}.knowledge_base_id`, + `${source}.document_id`, + `${source}.enabled`, + ...SOURCE_WIDTHS.map( + (width) => + `binary_quantize(${source}.${column('embedding', width)})::bit(${width}) AS ${identifier(column('binary', width))}` + ), + ...WIDTHS.map((width) => + width === 512 + ? `CASE WHEN ${shortened} THEN subvector(coalesce(${SOURCE_WIDTHS.map((size) => `${source}.${column('embedding', size)}`).join(', ')}), 1, 512)::halfvec(512) END AS vector_512` + : `CASE WHEN NOT (${shortened}) THEN ${source}.${column('embedding', width)}::halfvec(${width}) END AS ${column('vector', width)}` + ), + ].join(', ') +} + +async function relationExists(tx: Sql | TransactionSql, name: string): Promise { + const [row] = await tx<{ present: boolean }[]>`SELECT to_regclass(${name}) IS NOT NULL AS present` + return row.present +} + +async function stateOf(tx: Sql | TransactionSql): Promise { + const [row] = await tx`SELECT * FROM public.search_retirement_state WHERE id = 1` + if (!row || row.version !== 1) + throw new SearchRetirementError('Unrecognized retirement state; no changes were made') + return row +} + +async function statusOf(tx: Sql | TransactionSql): Promise { + const state = await stateOf(tx) + const [relations] = await tx<{ pending: boolean; backlog: boolean; backup: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM public.search_retirement_changes) AS pending, + EXISTS (SELECT 1 FROM public.search_retirement_changes OFFSET 10000 LIMIT 1) AS backlog, + to_regclass('public.embedding_search_retirement_backup') IS NOT NULL AS backup` + return { + phase: state.phase, + invalidated: state.invalidated, + invalidationReason: state.invalidation_reason, + sourceScanned: Number(state.source_scanned), + copied: Number(state.copied), + reconciled: Number(state.reconciled), + validatedSource: Number(state.validated_source), + validatedShadow: Number(state.validated_shadow), + purgedEmbeddings: Number(state.purged_embeddings), + retiredDocuments: Number(state.retired_documents), + pendingChanges: relations.pending, + changeBacklogAtLimit: relations.backlog, + backupRetained: relations.backup, + } +} + +/** Read-only: inspecting an uninitialized database never creates maintenance objects. */ +export async function getSearchRetirementStatus(sql: Sql): Promise { + if (!(await relationExists(sql, STATE))) return null + return statusOf(sql) +} + +async function operation( + sql: Sql, + work: (tx: TransactionSql) => Promise, + ddl = false +): Promise { + return sql.begin('isolation level read committed', async (tx) => { + const [version] = await tx<{ supported: boolean }[]>` + SELECT current_setting('server_version_num')::int >= 170000 AS supported` + if (!version.supported) + throw new SearchRetirementError( + 'Retirement requires PostgreSQL 17 or newer for a bounded transaction deadline' + ) + await tx.unsafe("SET LOCAL transaction_timeout = '3s'") + await tx.unsafe(`SET LOCAL statement_timeout = '${ddl ? '5s' : '2s'}'`) + await tx.unsafe("SET LOCAL lock_timeout = '100ms'") + await tx.unsafe('SET LOCAL search_path = pg_catalog, public') + const [locks] = await tx<{ job: boolean; maintenance: boolean; migration: boolean }[]>` + SELECT pg_try_advisory_xact_lock(hashtextextended(${JOB_LOCK}, 0)) AS job, + pg_try_advisory_xact_lock(hashtextextended(${MAINTENANCE_LOCK}, 0)) AS maintenance, + pg_try_advisory_xact_lock(${MIGRATION_LOCK}::bigint) AS migration` + if (!locks.job || !locks.maintenance || !locks.migration) { + throw new SearchRetirementError( + 'Another migration or retirement operation is active; retry later' + ) + } + if (await relationExists(tx, 'public.search_embedding_cleanup_progress')) { + await tx`SELECT id FROM public.search_embedding_cleanup_progress WHERE id = 1 FOR UPDATE NOWAIT` + } + return work(tx) + }) as Promise +} + +function requireValid(state: StateRow): void { + if (state.invalidated) { + if (['cutover', 'purge', 'documents', 'finalize', 'done'].includes(state.phase)) { + throw new SearchRetirementError( + 'A captured target changed after cutover; investigate its scope before resuming. Retirement state is preserved' + ) + } + throw new SearchRetirementError( + 'Knowledge-base indexing metadata changed; abort and prepare a new replacement' + ) + } +} + +async function verifyRelationIdentity(tx: TransactionSql, state: StateRow): Promise { + const swapped = ['cutover', 'purge', 'documents', 'finalize', 'done'].includes(state.phase) + const [row] = await tx<{ active: number; shadow: number | null; backup: number | null }[]>` + SELECT to_regclass('public.embedding_search')::oid AS active, + to_regclass('public.embedding_search_retirement_shadow')::oid AS shadow, + to_regclass('public.embedding_search_retirement_backup')::oid AS backup` + if ( + row.active !== (swapped ? state.replacement_oid : state.original_oid) || + (!swapped && row.shadow !== state.replacement_oid) || + (state.phase === 'cutover' && row.backup !== state.original_oid) + ) { + throw new SearchRetirementError( + 'Retirement relation identity changed; refusing to adopt or replace it' + ) + } +} + +/** Catalog checks are bounded and reject dependencies that a name swap would strand on the old OID. */ +async function inspectProjection(tx: TransactionSql): Promise { + const [relation] = await tx<{ safe: boolean }[]>` + SELECT c.relkind = 'r' AND c.relpersistence = 'p' AND NOT c.relrowsecurity + AND NOT c.relforcerowsecurity + AND NOT EXISTS (SELECT 1 FROM pg_policy WHERE polrelid = c.oid) + AND NOT EXISTS (SELECT 1 FROM pg_attribute WHERE attrelid = c.oid AND attacl IS NOT NULL) + AND NOT EXISTS (SELECT 1 FROM pg_constraint WHERE confrelid = c.oid) + AND NOT EXISTS (SELECT 1 FROM pg_depend WHERE refobjid = c.oid + AND refclassid = 'pg_class'::regclass + AND classid IN ('pg_rewrite'::regclass, 'pg_proc'::regclass, 'pg_policy'::regclass)) + AND NOT EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = c.oid OR inhparent = c.oid) + AND NOT EXISTS (SELECT 1 FROM pg_publication_tables + WHERE schemaname = 'public' AND tablename IN ('embedding', 'embedding_search', 'knowledge_base')) + AS safe + FROM pg_class c WHERE c.oid = 'public.embedding_search'::regclass` + if (!relation?.safe) + throw new SearchRetirementError( + 'Projection dependencies, publication, or row security require a separate migration' + ) + const [constraints] = await tx<{ safe: boolean }[]>` + SELECT NOT EXISTS (SELECT 1 FROM pg_constraint c WHERE c.conrelid = 'public.embedding_search'::regclass + AND NOT ( + (c.contype = 'p' AND c.conkey = ARRAY[1]::smallint[]) + OR (c.contype = 'f' AND c.conkey = ARRAY[1]::smallint[] + AND c.confrelid = 'public.embedding'::regclass AND c.confkey = ARRAY[1]::smallint[] + AND c.confdeltype = 'c' AND c.confupdtype = 'a') + OR (c.contype = 'c' AND c.conname = 'embedding_search_width_check') + OR c.contype = 'n' + )) AS safe` + if (!constraints.safe) + throw new SearchRetirementError( + 'Projection constraints differ from the reviewed compatible shape' + ) + const actual = await tx<{ name: string; type: string; required: boolean }[]>` + SELECT attname AS name, format_type(atttypid, atttypmod) AS type, attnotnull AS required + FROM pg_attribute WHERE attrelid = 'public.embedding_search'::regclass + AND attnum > 0 AND NOT attisdropped ORDER BY attnum LIMIT 32` + const expected = new Map([ + ['id', 'text:true'], + ['knowledge_base_id', 'text:true'], + ['document_id', 'text:true'], + ['enabled', 'boolean:true'], + ['connector_id', 'text:false'], + ['acl', 'text[]:false'], + ...SOURCE_WIDTHS.map((width) => [column('binary', width), `bit(${width}):false`] as const), + ...WIDTHS.map((width) => [column('vector', width), `halfvec(${width}):false`] as const), + ]) + if ( + actual.length !== expected.size || + actual.some((entry) => expected.get(entry.name) !== `${entry.type}:${entry.required}`) + ) { + throw new SearchRetirementError('Projection columns differ from the reviewed compatible shape') + } + const [triggers] = await tx<{ unknown: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM pg_trigger t JOIN pg_proc p ON p.oid = t.tgfoid + WHERE t.tgrelid = 'public.embedding_search'::regclass AND NOT t.tgisinternal + AND (t.tgname <> 'embedding_search_source_acl_set' OR p.proname <> 'set_projection_source_acl' + OR p.pronamespace <> 'public'::regnamespace OR t.tgenabled <> 'O')) AS unknown` + if (triggers.unknown) + throw new SearchRetirementError( + 'Projection has an unrecognized trigger; no replacement was prepared' + ) + const [writer] = await tx<{ safe: boolean; guard: string | null }[]>` + SELECT t.tgenabled = 'O' AND t.tgtype = 21 AND t.tgnargs = 0 + AND t.tgfoid = to_regprocedure('public.sync_embedding_search()') + AND ARRAY(SELECT a.attname::text FROM pg_attribute a + WHERE a.attrelid = t.tgrelid AND a.attnum = ANY(t.tgattr) ORDER BY a.attname) + = ARRAY['document_id', 'embedding', 'embedding_1024', 'embedding_3072', + 'embedding_384', 'embedding_768', 'enabled', 'knowledge_base_id']::text[] AS safe, + pg_get_expr(t.tgqual, t.tgrelid) AS guard + FROM pg_trigger t WHERE t.tgrelid = 'public.embedding'::regclass + AND t.tgname = 'embedding_search_sync' AND NOT t.tgisinternal` + const guard = writer?.guard?.replace(/[\s()]/g, '') + const setting = "current_setting'sim.projection_mode'::text,true" + if ( + !writer?.safe || + (guard && + ![ + `${setting}ISDISTINCTFROM'async'::text`, + `NOT${setting}ISNOTDISTINCTFROM'async'::text`, + `NOTNOT${setting}ISDISTINCTFROM'async'::text`, + ].includes(guard)) + ) { + throw new SearchRetirementError( + 'Canonical vector synchronization trigger differs from the reviewed writer' + ) + } +} + +async function clonePrivileges(tx: TransactionSql): Promise { + const [owner] = await tx<{ name: string }[]>` + SELECT pg_get_userbyid(relowner) AS name FROM pg_class WHERE oid = 'public.embedding_search'::regclass` + await tx.unsafe(`ALTER TABLE ${SHADOW} OWNER TO ${identifier(owner.name)}`) + const previousGrants = await tx<{ role: string }[]>` + SELECT DISTINCT CASE WHEN a.grantee = 0 THEN 'PUBLIC' ELSE pg_get_userbyid(a.grantee) END AS role + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search_retirement_shadow'::regclass AND a.grantee <> c.relowner LIMIT 129` + if (previousGrants.length > 128) + throw new SearchRetirementError( + 'Default table privileges exceed the reviewed maintenance bound' + ) + for (const grant of previousGrants) { + await tx.unsafe( + `REVOKE ALL PRIVILEGES ON TABLE ${SHADOW} FROM ${grant.role === 'PUBLIC' ? 'PUBLIC' : identifier(grant.role)}` + ) + } + const grants = await tx<{ role: string; privilege: string; grantable: boolean }[]>` + SELECT CASE WHEN a.grantee = 0 THEN 'PUBLIC' ELSE pg_get_userbyid(a.grantee) END AS role, + a.privilege_type AS privilege, a.is_grantable AS grantable + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search'::regclass AND a.grantee <> c.relowner LIMIT 129` + if (grants.length > 128) + throw new SearchRetirementError( + 'Projection privilege set exceeds the reviewed maintenance bound' + ) + for (const grant of grants) { + if ( + ![ + 'SELECT', + 'INSERT', + 'UPDATE', + 'DELETE', + 'TRUNCATE', + 'REFERENCES', + 'TRIGGER', + 'MAINTAIN', + ].includes(grant.privilege) + ) { + throw new SearchRetirementError('Projection has an unrecognized privilege') + } + await tx.unsafe( + `GRANT ${grant.privilege} ON TABLE ${SHADOW} TO ${grant.role === 'PUBLIC' ? 'PUBLIC' : identifier(grant.role)}${grant.grantable ? ' WITH GRANT OPTION' : ''}` + ) + } +} + +async function installCapture(tx: TransactionSql): Promise { + await tx.unsafe(`CREATE FUNCTION public.capture_search_retirement_change() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + IF TG_OP <> 'INSERT' THEN + INSERT INTO public.search_retirement_changes (embedding_id) VALUES (OLD.id) + ON CONFLICT (embedding_id) DO UPDATE SET generation = search_retirement_changes.generation + 1; + END IF; + IF TG_OP <> 'DELETE' AND (TG_OP <> 'UPDATE' OR OLD.id IS DISTINCT FROM NEW.id) THEN + INSERT INTO public.search_retirement_changes (embedding_id) VALUES (NEW.id) + ON CONFLICT (embedding_id) DO UPDATE SET generation = search_retirement_changes.generation + 1; + END IF; + RETURN NULL; + END $$`) + await tx.unsafe(`CREATE TRIGGER search_retirement_capture AFTER INSERT OR UPDATE OR DELETE + ON public.embedding FOR EACH ROW EXECUTE FUNCTION public.capture_search_retirement_change()`) + await tx.unsafe(`CREATE FUNCTION public.invalidate_search_retirement() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + UPDATE public.search_retirement_state SET invalidated = true, + invalidation_reason = 'knowledge-base-metadata-changed' WHERE id = 1; + RETURN NULL; + END $$`) + await tx.unsafe(`CREATE TRIGGER search_retirement_invalidate + AFTER UPDATE OF is_search_index, embedding_model, embedding_dimension ON public.knowledge_base + FOR EACH ROW WHEN (OLD.is_search_index IS DISTINCT FROM NEW.is_search_index + OR OLD.embedding_model IS DISTINCT FROM NEW.embedding_model + OR OLD.embedding_dimension IS DISTINCT FROM NEW.embedding_dimension) + EXECUTE FUNCTION public.invalidate_search_retirement()`) + if (await relationExists(tx, 'public.search_embedding_cleanup_progress')) { + await tx.unsafe( + 'LOCK TABLE public.search_embedding_cleanup_progress IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe(`CREATE FUNCTION public.guard_legacy_search_retirement() RETURNS trigger + LANGUAGE plpgsql AS $$ BEGIN + RAISE EXCEPTION 'Legacy cleanup is fenced by replacement retirement' USING ERRCODE = '55000'; + END $$`) + await tx.unsafe(`CREATE TRIGGER search_retirement_legacy_guard BEFORE INSERT OR UPDATE OR DELETE + ON public.search_embedding_cleanup_progress FOR EACH ROW + EXECUTE FUNCTION public.guard_legacy_search_retirement()`) + } +} + +/** Only empty-object DDL runs here. Copying, validation, cutover, and deletion require separate calls. */ +export async function initializeSearchRetirement(sql: Sql): Promise { + return operation( + sql, + async (tx) => { + if (await relationExists(tx, STATE)) { + const state = await stateOf(tx) + await verifyRelationIdentity(tx, state) + return statusOf(tx) + } + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe('LOCK TABLE public.embedding_search IN ACCESS SHARE MODE NOWAIT') + await inspectProjection(tx) + for (const relation of [ + SHADOW, + BACKUP, + 'public.search_retirement_targets', + 'public.search_retirement_changes', + ]) { + if (await relationExists(tx, relation)) + throw new SearchRetirementError( + 'Unowned retirement objects already exist; refusing to replace them' + ) + } + await tx.unsafe(`CREATE TABLE ${STATE} ( + id integer PRIMARY KEY CHECK (id = 1), version integer NOT NULL CHECK (version = 1), + phase text NOT NULL, resume_phase text NOT NULL DEFAULT 'validate-source', after_id text NOT NULL DEFAULT '', + invalidated boolean NOT NULL DEFAULT false, invalidation_reason text, + original_oid oid NOT NULL, replacement_oid oid NOT NULL, + source_scanned bigint NOT NULL DEFAULT 0, copied bigint NOT NULL DEFAULT 0, + reconciled bigint NOT NULL DEFAULT 0, validated_source bigint NOT NULL DEFAULT 0, + validated_shadow bigint NOT NULL DEFAULT 0, purged_embeddings bigint NOT NULL DEFAULT 0, + retired_documents bigint NOT NULL DEFAULT 0, round_mutations bigint NOT NULL DEFAULT 0 + , ready_at timestamptz, index_manifest jsonb NOT NULL DEFAULT '{}', relation_manifest jsonb NOT NULL DEFAULT '{}' + )`) + await tx.unsafe( + 'CREATE TABLE public.search_retirement_targets (knowledge_base_id text PRIMARY KEY)' + ) + await tx.unsafe( + 'CREATE TABLE public.search_retirement_changes (embedding_id text PRIMARY KEY, generation bigint NOT NULL DEFAULT 1)' + ) + await tx.unsafe( + `CREATE TABLE ${SHADOW} (LIKE public.embedding_search INCLUDING DEFAULTS INCLUDING GENERATED INCLUDING CONSTRAINTS INCLUDING STORAGE)` + ) + await tx.unsafe(`ALTER TABLE ${SHADOW} ADD CONSTRAINT embedding_search_retirement_shadow_pkey PRIMARY KEY (id), + ADD CONSTRAINT embedding_search_retirement_shadow_embedding_fk FOREIGN KEY (id) REFERENCES public.embedding(id) ON DELETE CASCADE`) + for (const index of SHARED_INDEXES.slice(1)) { + await tx.unsafe( + `CREATE INDEX ${identifier(`embedding_search_retirement_shadow_${index.suffix}`)} ON ${SHADOW} ${index.definition}` + ) + } + await clonePrivileges(tx) + await tx`INSERT INTO public.search_retirement_state (id, version, phase, original_oid, replacement_oid) + VALUES (1, 1, 'snapshot', 'public.embedding_search'::regclass, 'public.embedding_search_retirement_shadow'::regclass)` + await tx`UPDATE public.search_retirement_state SET index_manifest = ( + SELECT jsonb_object_agg(c.relname, pg_get_indexdef(i.indexrelid)) FROM pg_index i + JOIN pg_class c ON c.oid = i.indexrelid WHERE i.indrelid = 'public.embedding_search_retirement_shadow'::regclass + ) WHERE id = 1` + await tx.unsafe(`UPDATE ${STATE} SET relation_manifest = (${REPLACEMENT_SHAPE}) WHERE id = 1`) + await installCapture(tx) + return statusOf(tx) + }, + true + ) +} + +async function copyPage( + tx: TransactionSql, + state: StateRow, + pageSize: number +): Promise { + const [page] = await tx.unsafe( + `WITH page AS MATERIALIZED ( + SELECT id FROM public.embedding WHERE id > $1 ORDER BY id LIMIT $2 + ), expected AS MATERIALIZED ( + SELECT ${projectedValues('e', 'k.embedding_model')} FROM page p + JOIN public.embedding e ON e.id = p.id JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE NOT k.is_search_index + ), written AS ( + INSERT INTO ${SHADOW} AS s (${columns}) SELECT ${columns} FROM expected + ON CONFLICT (id) DO UPDATE SET ${assignments} WHERE ${comparison('s', 'EXCLUDED')} + RETURNING 1 + ) SELECT max(id) AS after_id, count(*)::int AS scanned, + (SELECT count(*)::int FROM written) AS changed FROM page`, + [state.after_id, pageSize] + ) + return page +} + +/** Read a generation before source state; a newer committed write keeps its queue entry for retry. */ +async function reconcilePage(tx: TransactionSql, pageSize: number): Promise { + const queued = await tx<{ embedding_id: string; generation: string }[]>` + SELECT embedding_id, generation::text FROM public.search_retirement_changes ORDER BY embedding_id LIMIT ${pageSize}` + if (queued.length === 0) return 0 + const ids = queued.map((row) => row.embedding_id) + await tx.unsafe( + `WITH expected AS MATERIALIZED ( + SELECT ${projectedValues('e', 'k.embedding_model')} FROM public.embedding e + JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE e.id = ANY($1::text[]) AND NOT k.is_search_index + ) INSERT INTO ${SHADOW} AS s (${columns}) SELECT ${columns} FROM expected + ON CONFLICT (id) DO UPDATE SET ${assignments} WHERE ${comparison('s', 'EXCLUDED')}`, + [ids] + ) + await tx.unsafe( + `DELETE FROM ${SHADOW} s WHERE s.id = ANY($1::text[]) AND NOT EXISTS ( + SELECT 1 FROM public.embedding e JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE e.id = s.id AND NOT k.is_search_index)`, + [ids] + ) + await tx`DELETE FROM public.search_retirement_changes q USING + unnest(${ids}::text[], ${queued.map((row) => row.generation)}::bigint[]) AS done(id, generation) + WHERE q.embedding_id = done.id AND q.generation = done.generation` + return queued.length +} + +async function validatePage( + tx: TransactionSql, + state: StateRow, + pageSize: number +): Promise { + const source = state.phase === 'validate-source' + const [page] = await tx.unsafe<(PageResult & { invalid: boolean })[]>( + `WITH page AS MATERIALIZED ( + SELECT id FROM ${source ? 'public.embedding' : SHADOW} WHERE id > $1 ORDER BY id LIMIT $2 + ), expected AS MATERIALIZED ( + SELECT ${projectedValues('e', 'k.embedding_model')} FROM page p + JOIN public.embedding e ON e.id = p.id JOIN public.knowledge_base k ON k.id = e.knowledge_base_id + WHERE NOT k.is_search_index + ) SELECT max(p.id) AS after_id, count(*)::int AS scanned, 0 AS changed, + coalesce(bool_or(NOT EXISTS (SELECT 1 FROM public.search_retirement_changes q WHERE q.embedding_id = p.id) + AND ${source ? `e.id IS NOT NULL AND (s.id IS NULL OR ${comparison('s', 'e')})` : `(e.id IS NULL OR ${comparison('s', 'e')})`}), false) AS invalid + FROM page p LEFT JOIN expected e ON e.id = p.id LEFT JOIN ${SHADOW} s ON s.id = p.id`, + [state.after_id, pageSize] + ) + if (page.invalid) + throw new SearchRetirementError( + 'Replacement validation found an uncaptured mismatch; cutover is not permitted' + ) + return page +} + +async function transition(tx: TransactionSql, phase: SearchRetirementPhase): Promise { + await tx`UPDATE public.search_retirement_state SET phase = ${phase}, after_id = '', round_mutations = 0 WHERE id = 1` +} + +/** Late writes retain their generation until the corresponding canonical row is safely retired. */ +async function purgeCapturedPage(tx: TransactionSql, pageSize: number): Promise { + const queued = await tx<{ embedding_id: string; generation: string }[]>` + SELECT embedding_id, generation::text FROM public.search_retirement_changes + ORDER BY embedding_id LIMIT ${Math.min(pageSize, 25)}` + if (queued.length === 0) return false + const ids = queued.map((row) => row.embedding_id) + const [page] = await tx<{ changed: number; invalid: boolean }[]>` + WITH targets AS MATERIALIZED ( + SELECT e.id, e.knowledge_base_id FROM public.embedding e + JOIN public.search_retirement_targets t ON t.knowledge_base_id = e.knowledge_base_id + WHERE e.id = ANY(${ids}::text[]) + ), bases AS MATERIALIZED ( + SELECT k.id, k.is_search_index FROM public.knowledge_base k + WHERE k.id IN (SELECT knowledge_base_id FROM targets) ORDER BY k.id FOR SHARE OF k + ), changed AS ( + DELETE FROM public.embedding e USING targets t, bases b + WHERE e.id = t.id AND e.knowledge_base_id = t.knowledge_base_id + AND b.id = t.knowledge_base_id AND b.is_search_index RETURNING e.id + ) SELECT (SELECT count(*)::int FROM changed) AS changed, + EXISTS (SELECT 1 FROM targets t LEFT JOIN bases b ON b.id = t.knowledge_base_id + WHERE b.id IS NULL OR NOT b.is_search_index) AS invalid` + if (page.invalid) + throw new SearchRetirementError( + 'A captured cleanup target is no longer Search-marked; no page was committed' + ) + await tx`DELETE FROM public.search_retirement_changes q USING + unnest(${ids}::text[], ${queued.map((row) => row.generation)}::bigint[]) AS done(id, generation) + WHERE q.embedding_id = done.id AND q.generation = done.generation` + await tx`UPDATE public.search_retirement_state SET purged_embeddings = purged_embeddings + ${page.changed} + WHERE id = 1` + return true +} + +async function removeCapture(tx: TransactionSql): Promise { + // Trigger removal can upgrade its relation lock; refuse a busy reader within one millisecond. + await tx.unsafe("SET LOCAL lock_timeout = '1ms'") + await tx.unsafe('DROP TRIGGER search_retirement_capture ON public.embedding') + await tx.unsafe('DROP TRIGGER search_retirement_invalidate ON public.knowledge_base') + await tx.unsafe( + 'DROP FUNCTION public.capture_search_retirement_change(), public.invalidate_search_retirement()' + ) +} + +async function purgePage(tx: TransactionSql, state: StateRow, pageSize: number): Promise { + if (await purgeCapturedPage(tx, pageSize)) { + if (state.phase === 'documents') await transition(tx, 'purge') + return + } + const documents = state.phase === 'documents' + const mutation = documents + ? `UPDATE public.document d SET user_excluded = true, enabled = false, processing_queue_token = NULL, + processing_queued_at = NULL, processing_deferred_until = NULL + FROM targets t WHERE d.id = t.id AND d.knowledge_base_id = t.knowledge_base_id + AND (NOT d.user_excluded OR d.enabled OR d.processing_queue_token IS NOT NULL + OR d.processing_queued_at IS NOT NULL OR d.processing_deferred_until IS NOT NULL) RETURNING d.id` + : `DELETE FROM public.embedding e USING targets t + WHERE e.id = t.id AND e.knowledge_base_id = t.knowledge_base_id RETURNING e.id` + const [page] = await tx.unsafe<(PageResult & { invalid: boolean })[]>( + `WITH page AS MATERIALIZED ( + SELECT id, knowledge_base_id FROM public.${documents ? 'document' : 'embedding'} + WHERE id > $1 ORDER BY id LIMIT $2 + ), targets AS MATERIALIZED ( + SELECT p.* FROM page p JOIN public.search_retirement_targets t ON t.knowledge_base_id = p.knowledge_base_id + ), bases AS MATERIALIZED ( + SELECT k.id, k.is_search_index FROM public.knowledge_base k + WHERE k.id IN (SELECT knowledge_base_id FROM targets) ORDER BY k.id FOR SHARE OF k + ), safe_targets AS MATERIALIZED ( + SELECT t.* FROM targets t JOIN bases b ON b.id = t.knowledge_base_id WHERE b.is_search_index + ), changed AS (${mutation.replace('FROM targets t', 'FROM safe_targets t').replace('USING targets t', 'USING safe_targets t')}) + SELECT max(id) AS after_id, count(*)::int AS scanned, + (SELECT count(*)::int FROM changed) AS changed, + EXISTS (SELECT 1 FROM targets t LEFT JOIN bases b ON b.id = t.knowledge_base_id + WHERE b.id IS NULL OR NOT b.is_search_index) AS invalid FROM page`, + [state.after_id, Math.min(pageSize, 25)] + ) + if (page.invalid) + throw new SearchRetirementError( + 'A captured cleanup target is no longer Search-marked; no page was committed' + ) + if (page.after_id !== null) { + await tx.unsafe( + `UPDATE ${STATE} SET after_id = $1, round_mutations = round_mutations + $2, + ${documents ? 'retired_documents' : 'purged_embeddings'} = ${documents ? 'retired_documents' : 'purged_embeddings'} + $2 WHERE id = 1`, + [page.after_id, page.changed] + ) + return + } + if (Number(state.round_mutations) > 0) { + await transition(tx, state.phase) + return + } + await transition(tx, documents ? 'finalize' : 'documents') +} + +/** Commits at most one bounded page. A timeout rolls back both its writes and cursor. */ +export async function advanceSearchRetirement( + sql: Sql, + options: { pageSize: number } +): Promise { + if (!Number.isInteger(options.pageSize) || options.pageSize < 1 || options.pageSize > 100) { + throw new SearchRetirementError('Page size must be an integer from 1 to 100') + } + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + const size = options.pageSize + const before = await statusOf(tx) + if ( + before.changeBacklogAtLimit && + !['ready', 'catchup', 'cutover', 'purge', 'documents', 'finalize', 'done'].includes( + state.phase + ) + ) { + const reconciled = await reconcilePage(tx, size) + await tx`UPDATE public.search_retirement_state SET reconciled = reconciled + ${reconciled} WHERE id = 1` + return statusOf(tx) + } + if (state.phase === 'snapshot') { + const [page] = await tx`WITH page AS MATERIALIZED ( + SELECT id, is_search_index FROM public.knowledge_base WHERE id > ${state.after_id} ORDER BY id LIMIT ${size} + ), inserted AS (INSERT INTO public.search_retirement_targets (knowledge_base_id) + SELECT id FROM page WHERE is_search_index ON CONFLICT DO NOTHING RETURNING 1) + SELECT max(id) AS after_id, count(*)::int AS scanned, (SELECT count(*)::int FROM inserted) AS changed FROM page` + if (page.after_id === null) await transition(tx, 'copy') + else + await tx`UPDATE public.search_retirement_state SET after_id = ${page.after_id} WHERE id = 1` + } else if (state.phase === 'copy') { + const page = await copyPage(tx, state, size) + if (page.after_id === null) await transition(tx, 'catchup') + else + await tx`UPDATE public.search_retirement_state SET after_id = ${page.after_id}, + source_scanned = source_scanned + ${page.scanned}, copied = copied + ${page.changed} WHERE id = 1` + } else if (state.phase === 'catchup' || state.phase === 'ready') { + const reconciled = await reconcilePage(tx, size) + if (reconciled > 0) { + await tx`UPDATE public.search_retirement_state SET reconciled = reconciled + ${reconciled}, ready_at = NULL WHERE id = 1` + if (state.phase === 'ready') await transition(tx, 'catchup') + } + if (reconciled === 0) { + if (state.resume_phase === 'ready' && !state.ready_at) { + await tx.unsafe(`ANALYZE ${SHADOW} (id, knowledge_base_id, document_id, enabled)`) + await tx`UPDATE public.search_retirement_state SET ready_at = clock_timestamp() WHERE id = 1` + } + if (state.phase === 'catchup') await transition(tx, state.resume_phase) + } + } else if (state.phase === 'validate-source' || state.phase === 'validate-shadow') { + const page = await validatePage(tx, state, size) + if (page.after_id === null) { + if (state.phase === 'validate-source') await transition(tx, 'validate-shadow') + else { + await tx`UPDATE public.search_retirement_state SET resume_phase = 'ready' WHERE id = 1` + await transition(tx, 'catchup') + } + } else { + const counter = state.phase === 'validate-source' ? 'validated_source' : 'validated_shadow' + await tx.unsafe( + `UPDATE ${STATE} SET after_id = $1, ${counter} = ${counter} + $2 WHERE id = 1`, + [page.after_id, page.scanned] + ) + } + } else if (state.phase === 'purge' || state.phase === 'documents') { + await purgePage(tx, state, size) + } + return statusOf(tx) + }) +} + +async function replaceVectorWriter(tx: TransactionSql): Promise { + await tx.unsafe(`CREATE OR REPLACE FUNCTION public.sync_embedding_search() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + IF NOT EXISTS (SELECT 1 FROM public.knowledge_base WHERE id = NEW.knowledge_base_id AND NOT is_search_index) THEN + DELETE FROM public.embedding_search WHERE id = NEW.id; + RETURN NEW; + END IF; + INSERT INTO public.embedding_search AS s (${columns}) + SELECT ${projectedValues('NEW', 'k.embedding_model')} FROM public.knowledge_base k WHERE k.id = NEW.knowledge_base_id + ON CONFLICT (id) DO UPDATE SET ${assignments}; + RETURN NEW; + END $$`) +} + +/** Ordinary writes no longer need capture once the replacement receives synchronous projections. */ +async function retainSearchCapture(tx: TransactionSql): Promise { + await tx.unsafe(`CREATE OR REPLACE FUNCTION public.capture_search_retirement_change() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + IF TG_OP <> 'DELETE' AND EXISTS (SELECT 1 FROM public.search_retirement_targets + WHERE knowledge_base_id = NEW.knowledge_base_id) THEN + INSERT INTO public.search_retirement_changes (embedding_id) VALUES (NEW.id) + ON CONFLICT (embedding_id) DO UPDATE SET generation = search_retirement_changes.generation + 1; + END IF; + RETURN NULL; + END $$`) + await tx.unsafe(`CREATE OR REPLACE FUNCTION public.invalidate_search_retirement() RETURNS trigger + LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN + IF OLD.is_search_index IS DISTINCT FROM NEW.is_search_index + AND EXISTS (SELECT 1 FROM public.search_retirement_targets WHERE knowledge_base_id = NEW.id) THEN + UPDATE public.search_retirement_state SET invalidated = true, + invalidation_reason = 'captured-target-marker-changed' WHERE id = 1; + END IF; + RETURN NULL; + END $$`) +} + +/** A zero-backlog metadata swap; busy relations cause an immediate rollback, never a wait queue. */ +export async function cutoverSearchRetirement(sql: Sql): Promise { + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + if (state.phase !== 'ready') + throw new SearchRetirementError( + 'Replacement must complete copy, catchup, and validation before cutover' + ) + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe(`LOCK TABLE public.embedding_search, ${SHADOW} IN ACCESS EXCLUSIVE MODE NOWAIT`) + requireValid(await stateOf(tx)) + await inspectProjection(tx) + if ((await statusOf(tx)).pendingChanges) + throw new SearchRetirementError( + 'Concurrent changes remain; run another bounded catchup page before cutover' + ) + const [snapshots] = await tx<{ observable: boolean; stale: boolean; prepared: boolean }[]>` + SELECT (SELECT rolsuper FROM pg_roles WHERE rolname = current_user) + OR pg_has_role(current_user, 'pg_read_all_stats', 'USAGE') AS observable, + EXISTS (SELECT 1 FROM pg_stat_activity a WHERE a.datid = (SELECT oid FROM pg_database WHERE datname = current_database()) + AND a.pid <> pg_backend_pid() AND a.xact_start <= (SELECT ready_at FROM public.search_retirement_state WHERE id = 1)) AS stale, + (SELECT ready_at IS NOT NULL FROM public.search_retirement_state WHERE id = 1) AS prepared` + if (!snapshots.observable) + throw new SearchRetirementError( + 'Cutover requires pg_read_all_stats to verify the transaction snapshot barrier' + ) + if (!snapshots.prepared || snapshots.stale) + throw new SearchRetirementError( + 'Transactions predate the replacement readiness barrier; let them finish and retry cutover' + ) + const [indexes] = await tx<{ valid: boolean }[]>` + SELECT coalesce(bool_and(i.indisvalid AND i.indisready), false) + AND jsonb_object_agg(c.relname, pg_get_indexdef(i.indexrelid)) = + (SELECT index_manifest FROM public.search_retirement_state WHERE id = 1) AS valid + FROM pg_index i JOIN pg_class c ON c.oid = i.indexrelid + WHERE i.indrelid = 'public.embedding_search_retirement_shadow'::regclass` + if (!indexes.valid) + throw new SearchRetirementError('Replacement index definitions changed after preparation') + const [shape] = await tx.unsafe<{ same: boolean }[]>(`SELECT (${REPLACEMENT_SHAPE}) = + (SELECT relation_manifest FROM ${STATE} WHERE id = 1) AS same`) + if (!shape.same) { + throw new SearchRetirementError( + 'Replacement columns, constraints, or dependencies changed after preparation' + ) + } + const [privileges] = await tx<{ same: boolean }[]>` + WITH source AS (SELECT a.grantee, a.privilege_type, a.is_grantable + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search'::regclass), + replacement AS (SELECT a.grantee, a.privilege_type, a.is_grantable + FROM pg_class c CROSS JOIN LATERAL aclexplode(coalesce(c.relacl, acldefault('r', c.relowner))) a + WHERE c.oid = 'public.embedding_search_retirement_shadow'::regclass) + SELECT a.relowner = b.relowner AND NOT EXISTS (SELECT * FROM source EXCEPT SELECT * FROM replacement) + AND NOT EXISTS (SELECT * FROM replacement EXCEPT SELECT * FROM source) AS same + FROM pg_class a, pg_class b WHERE a.oid = 'public.embedding_search'::regclass + AND b.oid = 'public.embedding_search_retirement_shadow'::regclass` + if (!privileges.same) + throw new SearchRetirementError( + 'Projection privileges changed; replacement cutover is not safe' + ) + const [aclTrigger] = await tx<{ present: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM pg_trigger WHERE tgrelid = 'public.embedding_search'::regclass + AND tgname = 'embedding_search_source_acl_set' AND NOT tgisinternal) AS present` + for (const index of SHARED_INDEXES) { + const [installed] = await tx<{ present: boolean; valid: boolean; original: boolean }[]>` + SELECT EXISTS (SELECT 1 FROM pg_class WHERE oid = to_regclass(${`public.${index.name}`})) AS present, + EXISTS (SELECT 1 FROM pg_index WHERE indexrelid = to_regclass(${`public.${index.name}`}) + AND indrelid = 'public.embedding_search'::regclass) AS original, + EXISTS (SELECT 1 FROM pg_index WHERE indexrelid = to_regclass(${`public.embedding_search_retirement_shadow_${index.suffix}`}) + AND indrelid = 'public.embedding_search_retirement_shadow'::regclass AND indisvalid AND indisready) AS valid` + if (!installed.valid) + throw new SearchRetirementError( + 'Replacement index is missing or invalid; cutover is not permitted' + ) + if (installed.present && !installed.original) + throw new SearchRetirementError('A shared index name belongs to an unrelated relation') + if (installed.present) + await tx.unsafe( + `ALTER INDEX public.${identifier(index.name)} RENAME TO ${identifier(`embedding_search_retirement_backup_${index.suffix}`)}` + ) + await tx.unsafe( + `ALTER INDEX public.${identifier(`embedding_search_retirement_shadow_${index.suffix}`)} RENAME TO ${identifier(index.name)}` + ) + } + await tx.unsafe( + `ALTER TABLE public.embedding_search RENAME TO embedding_search_retirement_backup` + ) + await tx.unsafe(`ALTER TABLE ${SHADOW} RENAME TO embedding_search`) + await tx.unsafe(`ALTER TABLE public.embedding_search + RENAME CONSTRAINT embedding_search_retirement_shadow_embedding_fk TO embedding_search_id_embedding_id_fk`) + if (aclTrigger.present) { + await tx.unsafe(`CREATE TRIGGER embedding_search_source_acl_set + BEFORE INSERT OR UPDATE OF document_id, enabled ON public.embedding_search + FOR EACH ROW WHEN (current_setting('sim.projection_mode', true) IS DISTINCT FROM 'async') + EXECUTE FUNCTION public.set_projection_source_acl()`) + } + await replaceVectorWriter(tx) + await retainSearchCapture(tx) + await transition(tx, 'cutover') + return statusOf(tx) + }) +} + +/** Explicitly ends the backup observation window before any canonical deletion can touch its HNSW graphs. */ +export async function beginSearchRetirementPurge(sql: Sql): Promise { + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + if (state.phase !== 'cutover') + throw new SearchRetirementError( + 'Purge requires a completed cutover and an explicit end to backup retention' + ) + await tx.unsafe('LOCK TABLE public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT') + await tx.unsafe(`LOCK TABLE ${BACKUP} IN ACCESS EXCLUSIVE MODE NOWAIT`) + // Removing the backup FK can upgrade its parent lock; keep that wait below a normal page's limit. + await tx.unsafe("SET LOCAL lock_timeout = '1ms'") + await tx.unsafe(`DROP TABLE ${BACKUP} RESTRICT`) + await transition(tx, 'purge') + return statusOf(tx) + }) +} + +/** Explicitly removes capture after operators clear the final primary and replica DDL window. */ +export async function finalizeSearchRetirement(sql: Sql): Promise { + return operation(sql, async (tx) => { + const state = await stateOf(tx) + requireValid(state) + await verifyRelationIdentity(tx, state) + if (state.phase !== 'finalize') { + throw new SearchRetirementError( + 'Finalization requires completed canonical and document retirement' + ) + } + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + requireValid(await stateOf(tx)) + if ((await statusOf(tx)).pendingChanges) { + await transition(tx, 'purge') + return statusOf(tx) + } + await removeCapture(tx) + await transition(tx, 'done') + return statusOf(tx) + }) +} + +/** Before cutover, abort removes only this job's empty-or-partial replacement and capture machinery. */ +export async function abortSearchRetirement(sql: Sql): Promise { + return operation(sql, async (tx) => { + if (!(await relationExists(tx, STATE))) return + const state = await stateOf(tx) + await verifyRelationIdentity(tx, state) + if (['cutover', 'purge', 'documents', 'finalize', 'done'].includes(state.phase)) { + throw new SearchRetirementError( + 'A cut-over replacement cannot be rolled back by renaming a stale backup' + ) + } + await tx.unsafe( + 'LOCK TABLE public.knowledge_base, public.embedding IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe(`LOCK TABLE ${SHADOW} IN ACCESS EXCLUSIVE MODE NOWAIT`) + await removeCapture(tx) + if (await relationExists(tx, 'public.search_embedding_cleanup_progress')) { + await tx.unsafe( + 'LOCK TABLE public.search_embedding_cleanup_progress IN SHARE ROW EXCLUSIVE MODE NOWAIT' + ) + await tx.unsafe( + 'DROP TRIGGER IF EXISTS search_retirement_legacy_guard ON public.search_embedding_cleanup_progress' + ) + } + await tx.unsafe('DROP FUNCTION IF EXISTS public.guard_legacy_search_retirement()') + await tx.unsafe( + `DROP TABLE ${SHADOW}, public.search_retirement_changes, public.search_retirement_targets, ${STATE} RESTRICT` + ) + }) +} diff --git a/packages/db/package.json b/packages/db/package.json index 381203f99a1..31cf5145650 100644 --- a/packages/db/package.json +++ b/packages/db/package.json @@ -36,6 +36,7 @@ "db:migrate": "bun --env-file=.env run ./scripts/migrate.ts", "db:reconcile-fork-kb-file-ownership": "bun --env-file=.env run ./scripts/reconcile-fork-kb-file-ownership.ts", "db:reconcile-workspace-storage": "bun --env-file=.env run ./scripts/reconcile-workspace-storage.ts", + "db:retire-indexed-search": "bun --no-env-file ./scripts/retire-indexed-search.ts", "db:studio": "bunx drizzle-kit studio --config=./drizzle.config.ts", "test": "vitest run", "type-check": "tsc --noEmit", diff --git a/packages/db/script-migrations/0027_retire_search_embeddings.ts b/packages/db/script-migrations/0027_retire_search_embeddings.ts index f58f5b14b01..e9022b633f1 100644 --- a/packages/db/script-migrations/0027_retire_search_embeddings.ts +++ b/packages/db/script-migrations/0027_retire_search_embeddings.ts @@ -1,11 +1,9 @@ -import { parseArgs } from 'node:util' -import { resolveMigrationDatabaseUrl } from '@sim/db/script-migrations/database-url' import type { ScriptMigration } from '@sim/db/script-migrations/types' import { retryOnLockTimeout } from '@sim/db/scripts/lock-timeout-retry' import { createLogger } from '@sim/logger' import { getPostgresCancellationReason } from '@sim/utils/errors' import { sleep } from '@sim/utils/helpers' -import postgres, { type Sql, type TransactionSql } from 'postgres' +import type { Sql, TransactionSql } from 'postgres' const logger = createLogger('RetireSearchEmbeddings') /** Most IDs one page reads in primary-key order; reading is cheap next to the mutation. */ @@ -409,37 +407,10 @@ async function validateTargetMarkers(tx: TransactionSql): Promise { } } -/** - * The operator entry: resumes the saved cursor and, with `--maintenance`, also rebuilds the indexes, - * vacuums, and journals the completed cleanup. `--pause-ratio` and `--max-rows` set the pacing. - */ +/** Historical implementation remains for migration replay tests; the unbounded CLI is retired. */ if (import.meta.main) { - const { values } = parseArgs({ - options: { - maintenance: { type: 'boolean', default: false }, - 'pause-ratio': { type: 'string' }, - 'max-rows': { type: 'string' }, - }, - }) - const pacing: RetirementPacing = { - pauseRatio: Number(values['pause-ratio'] ?? DEFAULT_RETIREMENT_PACING.pauseRatio), - maxRows: Number(values['max-rows'] ?? DEFAULT_RETIREMENT_PACING.maxRows), - } - const url = resolveMigrationDatabaseUrl() - if (!url) throw new Error('DATABASE_URL is required for Search retirement') - const sql = postgres(url, { max: 1, max_lifetime: null, onnotice: () => undefined }) - try { - if (values.maintenance) { - const { runScriptMigrations } = await import('@sim/db/script-migrations/index') - const { retireAllSearchEmbeddings } = await import( - '@sim/db/script-migrations/0029_retire_all_search_embeddings' - ) - await runScriptMigrations(sql, [retireAllSearchEmbeddings(pacing)]) - } else { - await retireSearchEmbeddings(sql, pacing) - logger.info('Search retirement pass finished; run with --maintenance off-peak to complete it') - } - } finally { - await sql.end() - } + logger.error( + 'Use packages/db/scripts/retire-indexed-search.ts. The legacy delete/reindex command is disabled.' + ) + process.exitCode = 1 } diff --git a/packages/db/script-migrations/index.ts b/packages/db/script-migrations/index.ts index 7700a8f80a3..084c707c333 100644 --- a/packages/db/script-migrations/index.ts +++ b/packages/db/script-migrations/index.ts @@ -60,7 +60,7 @@ export const scriptMigrations: readonly ScriptMigration[] = [ userTableSchemaForWriteMigration, /** * Search retirement (0027–0029) is an operator-run maintenance command, not a deploy step: - * see `search-embedding-retirement.md`. + * run `packages/db/scripts/retire-indexed-search.ts --help` for usage. */ ] diff --git a/packages/db/script-migrations/search-embedding-retirement.md b/packages/db/script-migrations/search-embedding-retirement.md deleted file mode 100644 index 7b8864c4499..00000000000 --- a/packages/db/script-migrations/search-embedding-retirement.md +++ /dev/null @@ -1,179 +0,0 @@ -# Retiring legacy Search indexes - -The retirement is an **operator-run maintenance command, not a deploy step**. Deploy migrations no -longer register it: a long cleanup inside the deploy migration generated heavy WAL and stalled -application writes, and it held the release until it finished. It is optional storage reclamation -once live Search is on, so it runs separately, paced, at a time the operator chooses. Self-hosted -operators can run the same command. - -`0029_retire_all_search_embeddings` snapshots every knowledge base whose persisted `is_search_index` -marker is true and supersedes the single-KB retirement and maintenance receipts (`0027`/`0028`), -including databases that already recorded either. No Search KB is a completed no-op. Once saved, the -snapshot stays fixed across runs even if another Search KB is created. Ordinary KBs and the selected -KBs' live source/credential configuration, document metadata, and backing files are preserved. - -## Before running - -The app and workers must already use live Search, and older indexing jobs must be drained. -Enterprise Search no longer has an indexed backend or an environment toggle to re-enable it. -Older releases could re-enable indexed Search, so their workers must be drained before retirement. Live source setup may still -create a Search KB for configuration; it does not index content. Document uploads, dispatch and -queued processing also honor the indexed-search gate. - -## Running it - -From the repository root, with the migration role's writer DSN on a **direct or session-pooled** -connection (the run holds a session advisory lock and session settings; PgBouncer transaction pooling -is unsupported, and reserving a postgres.js client does not pin a backend through it): - -```sh -# Retire documents and delete their chunks, resuming the saved cursor. Safe to stop and rerun. -MIGRATION_DATABASE_URL= bun run packages/db/script-migrations/0027_retire_search_embeddings.ts - -# Off-peak: finish any remaining retirement, rebuild the HNSW indexes, vacuum, and record completion. -MIGRATION_DATABASE_URL= bun run packages/db/script-migrations/0027_retire_search_embeddings.ts --maintenance -``` - -| Flag | Default | Effect | -| --- | --- | --- | -| `--pause-ratio N` | `2` | After each page, pause N × the page's duration (at most one minute), so the run is busy at most `1 / (1 + N)` of the time. Raise it to go gentler. | -| `--max-rows N` | `2000` | The most rows one page may update or delete (25–8,000). Lower it to make each page lighter. | -| `--maintenance` | off | After retirement, run the index rebuilds and vacuums and journal `0029` with its superseded names. | - -Run it as the migration role: maintenance needs `pg_maintain`, which the application roles lack. Run -it outside peak traffic, and run `--maintenance` in the quietest window you have: concurrent HNSW -rebuilds are long and write a lot of WAL (GitLab, for example, schedules automatic reindexing for -weekends). Keep one run at a time. - -**Pausing.** Ctrl-C is safe at any point. The in-flight page rolls back with its cursor, and an -interrupted concurrent rebuild's leftover index is removed on the next run. Rerun the same command to -resume; completed pages stay committed. - -**Watching.** Every ten pages the run logs the phase, cursor, rows mutated so far and current row -limit; it also logs each halving after a slow page, each phase change, and the start of the completion -recheck. In PostgreSQL, watch for `checkpoint starting: wal` in quick succession, slow checkpoint -sync times, `canceling wait for synchronous replication`, and replica lag. If they appear, stop the -run and resume later with a higher `--pause-ratio` or lower `--max-rows`. - -```sql -SELECT * FROM search_embedding_cleanup_progress; -SELECT name, applied_at FROM script_migrations -WHERE name IN ('0027_retire_search_embeddings', '0028_maintain_search_retirement', - '0029_retire_all_search_embeddings'); -``` - -A plain run does not journal anything; only a `--maintenance` run that finishes records `0029` and its -superseded names. On upgrading a legacy single-KB checkpoint, the snapshot and cursor reset commit -atomically. The scan starts at the beginning once so it includes other KBs behind the old cursor; -previous deletes remain committed. Maintenance checkpoints also reset once because the expanded -cleanup creates new dead entries. A completed legacy checkpoint does not require its former KB to -still exist or remain Search-marked; the new snapshot selects current Search KBs and preserves any KB -now marked ordinary. An unfinished legacy checkpoint still requires its target to remain -Search-marked. - -## How a run paces itself - -Each page mutates at most a row limit of target rows and reads at most four IDs per row of that -limit, never more than 25,000 IDs. Pages execute -sequentially, and each is followed by a pause of `--pause-ratio` times its duration, up to one -minute. Because a page is timed through its commit, a slow synchronous replica or a checkpoint stall -lengthens the following pause by the same factor. Retiring a document is a non-HOT update that writes every index on -`document`, and deleting a chunk cascades into its projections, so a page's cost follows the target -rows it mutates, not the IDs it reads. A page that reaches the row limit advances the cursor only to -its last mutated row; the rest of its scan is read again by the next page. Tying the scan window to -the limit keeps that re-reading proportional to the work, even after the limit shrinks. Documents -that are already retired never count against the limit. - -The row limit starts at 2,000 rows, or `--max-rows` if lower. A page is timed from the start of its transaction through its -commit, including the synchronous-replication wait and any lock-timeout retries. A page slower than -30 seconds halves the limit. A fast page, one under 7.5 seconds, doubles it up to `--max-rows`, which -also widens the scan window, so sparse stretches are not crawled in small windows. The limit never drops below 25 rows. Phase changes do not adjust it. - -Materialized SQL pages keep the IDs inside PostgreSQL; the migration process receives only a cursor -and a validation result. Each page uses a two-minute statement timeout and a one-second lock -timeout. If a page's mutating statement exceeds the statement timeout, the page rolls back with its -cursor and is retried with half the row limit after the usual pause. From then on, fast pages grow -the limit only up to that halved size, so a size that timed out is never tried again. A page that -still times out at 25 rows fails the migration. Any other statement timeout fails the migration at once, because a smaller page cannot -speed it up. The completion rechecks, which walk every captured KB once, run with a 30-minute -timeout. Brief lock timeouts retry the rolled-back page with bounded backoff for up to one minute. -Other errors, or exhausted lock retries, fail the migration without a completion receipt. - -The runner-owned `search_embedding_cleanup_targets` table stores the frozen KB set, populated in -bounded SQL pages within one repeatable-read transaction. The existing `search_embedding_cleanup_progress` -row stores the shared phase and ID cursor; its legacy `knowledge_base_id` remains an informational -anchor, not the full deletion scope. Page mutations and cursor advancement commit together. -The one-off migration journal records only completion. `db:push` excludes both bookkeeping tables -from schema diffing. - -The documents phase fences queued and in-flight processing by marking only target documents -excluded/disabled and clearing their dispatch stamps. The embeddings phase deletes only target -chunks, and refuses a page whose target document was not retired or has inconsistent ownership. -The existing foreign keys cascade to vector/keyword projections and chunk provenance. Both phases -walk the primary key once for the entire captured set in bounded pages; ordinary rows are never -updated. Each page locks and rechecks the Search markers for its target KBs before mutation, and -fails atomically if any target changed to an ordinary KB. This avoids a separate full-table scan -per KB or sorting a whole KB when no suitable composite cleanup index exists. Resumption continues the saved scan, including across pages containing only -unrelated rows. Before completion, the cleanup checks for unretired documents and remaining chunks -behind either cursor and restarts the affected phase if needed. A final bounded pass validates all -captured KB markers, including empty KBs and KBs whose rows were already scanned, holding shared -marker locks until the completion checkpoint commits. Resuming a completed cleanup before maintenance -also revalidates the captured set, with the same 30-minute timeout as the completion recheck. Keep target writers stopped and -do not change their Search markers during the pass. - -Inspect progress with: - -```sql -SELECT * FROM search_embedding_cleanup_progress; -SELECT name, applied_at FROM script_migrations -WHERE name IN ('0027_retire_search_embeddings', '0028_maintain_search_retirement', - '0029_retire_all_search_embeddings'); -``` - -After completion, check the selected KBs have no `embedding` rows, verify their live Search, and verify -ordinary KB retrieval. A zero-row absence check may still scan index entries; use an appropriate -timeout. The cleanup is destructive and not reversible by flipping the search flag. Re-enabling -indexed Search requires deliberately restoring document eligibility and fully resyncing its sources. - -## Storage maintenance - -With `--maintenance`, after deletion, `0029` invokes the existing maintenance implementation to run `REINDEX INDEX CONCURRENTLY` on each HNSW index of `embedding_search`, -then `VACUUM (ANALYZE, TRUNCATE FALSE)` on the vector and keyword projections, chunk provenance, -embeddings, and documents. These operations execute sequentially outside transactions. Rebuilds -keep ordinary reads and writes available and require temporary index space and WAL capacity. -They wait for older transactions and can dominate total runtime. PostgreSQL's `pg_stat_progress_create_index` and `pg_stat_progress_vacuum` expose progress. - -The existing progress row gains `reindexed_through` and `vacuumed_tables` checkpoints. Completed -indexes and tables are skipped on retry; interruption between an operation and its checkpoint may -repeat that one operation. Invalid `_ccnew`/`_ccold` siblings from an interrupted concurrent rebuild -are removed concurrently before retrying their original index. One advisory lock serializes the -maintenance worker. Do not run other index maintenance on these tables at the same time. - -Ordinary vacuum makes dead space reusable and refreshes planner statistics; it generally does not -shrink table files. This migration does not run `VACUUM FULL` or rewrite tables. See the -[PostgreSQL vacuum documentation](https://www.postgresql.org/docs/17/sql-vacuum.html), -[concurrent reindex recovery](https://www.postgresql.org/docs/17/sql-reindex.html#SQL-REINDEX-CONCURRENTLY), -and [pgvector maintenance guidance](https://github.com/pgvector/pgvector#vacuuming). - -Document counts are historical until a later document cleanup. Do not raw-delete documents or -bucket objects: their application hard-delete path also enqueues identity-bound storage cleanup -and applies accounting. Its ordinary scoped mode excludes retired documents, so a follow-up must -explicitly support these rows while preserving those side effects. Do not delete source accounts, -integration policies or permission grants used by live Search. - -## Why it runs this way - -Long data changes belong outside deploy migrations, in batches, throttled on database health, and -resumable from a cursor: - -- [strong_migrations: Backfilling data](https://github.com/ankane/strong_migrations#backfilling-data) -- [GitLab batched background migrations](https://docs.gitlab.com/development/database/batched_background_migrations/) - and [automatic reindexing](https://docs.gitlab.com/omnibus/settings/database/) -- [Shopify maintenance_tasks](https://github.com/Shopify/maintenance_tasks) -- [gh-ost throttling](https://github.com/github/gh-ost/blob/master/doc/throttle.md) and - [pt-online-schema-change](https://docs.percona.com/percona-toolkit/pt-online-schema-change.html) -- [Stripe: online migrations at scale](https://stripe.com/blog/online-migrations) -- PostgreSQL 17: [WAL configuration](https://www.postgresql.org/docs/17/wal-configuration.html), - [synchronous replication](https://www.postgresql.org/docs/17/warm-standby.html#SYNCHRONOUS-REPLICATION), - [replication statistics](https://www.postgresql.org/docs/17/monitoring-stats.html) -- [PlanetScale: the only scalable delete](https://planetscale.com/blog/the-only-scalable-delete) diff --git a/packages/db/scripts/push.test.ts b/packages/db/scripts/push.test.ts index acca74d8f55..c1e95771e60 100644 --- a/packages/db/scripts/push.test.ts +++ b/packages/db/scripts/push.test.ts @@ -33,11 +33,6 @@ afterEach(() => { }) describe('db:push policy and process boundaries', () => { - it('does not implicitly approve data loss', async () => { - expect(await runPush([])).toBe(0) - expect(spawn.mock.calls[0][0]).not.toContain('--force') - }) - it('rejects interactive renames without a terminal before any database commands', async () => { expect(await runPush(['--interactive-renames', '--force'])).toBe(1) expect(spawn).not.toHaveBeenCalled() diff --git a/packages/db/scripts/push.ts b/packages/db/scripts/push.ts index 2d10c1c4e0e..5c2725097f2 100644 --- a/packages/db/scripts/push.ts +++ b/packages/db/scripts/push.ts @@ -1,4 +1,7 @@ +import { fileURLToPath } from 'node:url' import { createLogger } from '@sim/logger' +import { getPostgresErrorCode } from '@sim/utils/errors' +import postgres, { type Sql } from 'postgres' const logger = createLogger('DatabasePush') const RECONCILIATION_COMMANDS = [ @@ -12,6 +15,61 @@ const RECONCILIATION_COMMANDS = [ ['bun', '--env-file=.env', 'run', './script-migrations/0026_user_table_schema_for_write.ts'], ] +/** Keep preparation, Drizzle and reconcilers fenced without a session lock or a retained snapshot. */ +async function withRetirementFence(run: (signal: AbortSignal) => Promise): Promise { + const rawUrl = process.env.DATABASE_URL + if (!rawUrl) { + logger.error('DATABASE_URL is required for schema push') + return 1 + } + const abort = new AbortController() + let sql: Sql | undefined + try { + const url = new URL(rawUrl) + for (const parameter of ['application_name', 'statement_timeout', 'lock_timeout', 'options']) { + url.searchParams.delete(parameter) + } + sql = postgres(url.toString(), { + max: 1, + prepare: false, + connect_timeout: 5, + max_lifetime: null, + connection: { application_name: 'sim-db-push', statement_timeout: 2_000, lock_timeout: 100 }, + onnotice: () => undefined, + onclose: () => abort.abort(), + }) + return (await sql.begin('isolation level read committed read only', async (tx) => { + // Simple protocol closes SELECT portals so concurrent index builds do not wait on their snapshots. + const [locks] = await tx<{ maintenance: boolean; migration: boolean }[]>` + SELECT pg_try_advisory_xact_lock(hashtextextended('sim:search-retirement-maintenance', 0)) AS maintenance, + pg_try_advisory_xact_lock(4961002270::bigint) AS migration`.simple() + if (!locks.maintenance || !locks.migration) { + logger.error('Another migration or retirement operation is active; retry schema push later') + return 1 + } + const [state] = await tx<{ present: boolean }[]>` + SELECT to_regclass('public.search_retirement_state') IS NOT NULL AS present`.simple() + if (state.present) { + logger.error( + 'Schema push is disabled after Search retirement starts, including completed retirement. Use reviewed versioned migrations; push reconcilers would restore retired projections' + ) + return 1 + } + return run(abort.signal) + })) as number + } catch (error) { + logger.error( + abort.signal.aborted + ? 'Schema-push lock connection closed; stopped all commands' + : 'Schema push stopped', + { code: getPostgresErrorCode(error) } + ) + return 1 + } finally { + await sql?.end({ timeout: 1 }).catch(() => undefined) + } +} + /** * Push treats additions and removals as distinct objects by default. The pinned * Drizzle patch reads this policy only in the push subprocess; generation keeps @@ -31,39 +89,45 @@ export async function runPush(args: string[]): Promise { return 1 } - if (!help && args.includes('--force')) { - const preparation = Bun.spawn(['bun', '--env-file=.env', 'run', './scripts/prepare-push.ts'], { - stdin: 'inherit', - stdout: 'inherit', - stderr: 'inherit', - }) - const preparationExit = await preparation.exited - if (preparationExit !== 0) return preparationExit - } + async function runCommands(signal?: AbortSignal): Promise { + async function runCommand(command: string[]): Promise { + signal?.throwIfAborted() + const child = Bun.spawn(command, { + env: { ...process.env, SIM_DB_PUSH_RENAME_MODE: interactiveRenames ? 'prompt' : 'create' }, + stdin: 'inherit', + stdout: 'inherit', + stderr: 'inherit', + signal, + killSignal: 'SIGKILL', + }) + const code = await child.exited + signal?.throwIfAborted() + return code + } - const pushArgs = args.filter((arg) => arg !== '--interactive-renames') - const child = Bun.spawn( - ['bunx', '--no-install', 'drizzle-kit', 'push', '--config=./drizzle.config.ts', ...pushArgs], - { - env: { ...process.env, SIM_DB_PUSH_RENAME_MODE: interactiveRenames ? 'prompt' : 'create' }, - stdin: 'inherit', - stdout: 'inherit', - stderr: 'inherit', + if (!help && args.includes('--force')) { + const code = await runCommand(['bun', '--env-file=.env', 'run', './scripts/prepare-push.ts']) + if (code !== 0) return code } - ) - const exitCode = await child.exited - if (exitCode !== 0 || help) return exitCode + const pushArgs = args.filter((arg) => arg !== '--interactive-renames') + // Launch the installed CLI directly so losing the fence also terminates its database connection. + const cli = fileURLToPath(new URL('bin.cjs', import.meta.resolve('drizzle-kit'))) + const exitCode = await runCommand([ + 'bun', + cli, + 'push', + '--config=./drizzle.config.ts', + ...pushArgs, + ]) + if (exitCode !== 0 || help) return exitCode - for (const command of RECONCILIATION_COMMANDS) { - const reconciliation = Bun.spawn(command, { - stdin: 'inherit', - stdout: 'inherit', - stderr: 'inherit', - }) - const reconciliationExit = await reconciliation.exited - if (reconciliationExit !== 0) return reconciliationExit + for (const command of RECONCILIATION_COMMANDS) { + const code = await runCommand(command) + if (code !== 0) return code + } + return 0 } - return 0 + return help ? runCommands() : withRetirementFence(runCommands) } if (import.meta.main) { diff --git a/packages/db/scripts/retire-indexed-search.ts b/packages/db/scripts/retire-indexed-search.ts new file mode 100644 index 00000000000..0692e1390a9 --- /dev/null +++ b/packages/db/scripts/retire-indexed-search.ts @@ -0,0 +1,262 @@ +import { createHash } from 'node:crypto' +import { parseArgs } from 'node:util' +import { + abortSearchRetirement, + advanceSearchRetirement, + beginSearchRetirementPurge, + cutoverSearchRetirement, + finalizeSearchRetirement, + getSearchRetirementStatus, + initializeSearchRetirement, + RetireSearchHealthError, + readSearchRetirementHealth, + readSearchRetirementHealthLimits, + SearchRetirementError, +} from '@sim/db/maintenance' +import { createLogger, LogLevel } from '@sim/logger' +import { getPostgresErrorCode } from '@sim/utils/errors' +import { sleep } from '@sim/utils/helpers' +import postgres from 'postgres' + +class RetirementCommandError extends Error {} + +const logger = createLogger('SearchDataRetirement', { enabled: true, logLevel: LogLevel.INFO }) +const HELP = `Operator-only indexed Search retirement. No work runs on deployment. + +bun --no-env-file packages/db/scripts/retire-indexed-search.ts [options] + +Commands: + identity Print the non-secret connection fingerprint for the health policy/collector. + status Read progress; never initialize or resume work. + prepare Create empty indexed replacement and capture writes (requires --ack-release-drained). + run Advance bounded pages; stops at each manual gate (default: one page). + cutover Attempt NOWAIT swap after validation (requires --ack-release-drained). + begin-purge Retire the backup and allow chunk deletion (requires --ack-retire-backup). + finalize Remove capture after deletion and a separate DDL health clearance. + abort Remove capture and discard the replacement before cutover. + +Every write except abort requires --health-file PATH --health-policy PATH. +Policy databaseId must equal the identity command's fingerprint. +Run options: --page-size 25 (1–100), --pages 1 (1–120), --seconds 60 (1–600). +A fixed pause of at least 5 seconds follows each page. Timeouts stop; no automatic retry. +Connection: MIGRATION_DATABASE_URL, PostgreSQL 17+, verified PlanetScale primary on port 5432. +Other endpoints are refused; loopback databases with a test name segment are for local fixtures only. + +Deploy the code-removal release and drain old workers first. Then: + prepare -> run until ready -> cutover -> verify behavior -> begin-purge + -> run until finalize -> finalize. Repeat run if late writes return the phase to purge. +Start with default one-page runs and watch telemetry before requesting longer runs. +Before begin-purge, verify ordinary-KB retrieval, ACL denials, connector ingestion and live Search. +Reads use the old projection until cutover. The retained backup is not an instant rollback. +Cutover requires pg_read_all_stats and no old snapshots; never automatically retry DDL gates. +Abort, backup removal and finalization cap implicit DROP lock waits at 1 ms. +After retirement starts, use versioned migrations; db:push would restore old projection machinery. + +Health files are UTF-8 JSON, at most 8 KiB, atomically refreshed by trusted telemetry. +Sample fields: observedAt (oldest metric timestamp, UTC ISO), databaseId, healthy, + maintenanceAllowed, cutoverAllowed, replicaLagBytes, replicaLagSeconds, + walBytesPerSecond, databaseP95Ms, cpuPercent, freeStorageBytes. +Policy fields: databaseId, maxReplicaLagBytes, maxReplicaLagSeconds, maxWalBytesPerSecond, + maxDatabaseP95Ms, maxCpuPercent, minFreeStorageBytes, maxSampleAgeMs (at most 30000). +Choose limits from actual capacity and latency requirements; missing/stale telemetry stops work. +Use worst replica lag and CPU, and minimum free storage; healthy includes application error/latency checks. +Set cutoverAllowed only after primary and replica snapshots are clear for DDL. +Pacing limits load but cannot eliminate latency or replica-conflict risk.` + +function integerOption(value: string | undefined, fallback: number, ceiling: number): number { + const parsed = value === undefined ? fallback : Number(value) + if (!Number.isInteger(parsed) || parsed < 1 || parsed > ceiling) { + throw new RetirementCommandError(`Expected an integer between 1 and ${ceiling}`) + } + return parsed +} + +async function main(): Promise { + const { values, positionals } = parseArgs({ + args: process.argv.slice(2), + allowPositionals: true, + options: { + help: { type: 'boolean', default: false }, + 'health-file': { type: 'string' }, + 'health-policy': { type: 'string' }, + 'page-size': { type: 'string' }, + pages: { type: 'string' }, + seconds: { type: 'string' }, + 'ack-release-drained': { type: 'boolean', default: false }, + 'ack-retire-backup': { type: 'boolean', default: false }, + }, + }) + if (values.help || positionals.length === 0) { + process.stdout.write(`${HELP}\n`) + return + } + const [command] = positionals + if ( + positionals.length !== 1 || + ![ + 'identity', + 'status', + 'prepare', + 'run', + 'cutover', + 'begin-purge', + 'finalize', + 'abort', + ].includes(command) + ) { + throw new RetirementCommandError('Unknown command; use --help') + } + const rawUrl = process.env.MIGRATION_DATABASE_URL + if (!rawUrl) + throw new RetirementCommandError( + 'MIGRATION_DATABASE_URL is required; there is no application DSN fallback' + ) + const url = new URL(rawUrl) + if (!['postgres:', 'postgresql:'].includes(url.protocol)) + throw new RetirementCommandError('Expected a PostgreSQL URL') + if (!url.username || !url.pathname.slice(1)) { + throw new RetirementCommandError('The connection URL must include its role and database name') + } + const localFixture = + ['localhost', '127.0.0.1', '[::1]'].includes(url.hostname) && + /(^|_)test(_|$)/.test(decodeURIComponent(url.pathname.slice(1))) + const hosted = + /^[a-z0-9-]+\.(?:pg|horizon)\.psdb\.cloud$/.test(url.hostname) && + (url.port || '5432') === '5432' && + !decodeURIComponent(url.username).includes('|') + if (!hosted && !localFixture) { + throw new RetirementCommandError( + 'Use a direct PlanetScale primary endpoint on port 5432; unverified endpoints and poolers are unsupported' + ) + } + if ( + [...url.searchParams.keys()].some( + (key) => !['sslmode', 'sslrootcert', 'sslnegotiation'].includes(key) + ) + ) { + throw new RetirementCommandError( + 'Connection overrides are unsupported; only TLS URL parameters are allowed' + ) + } + if ( + hosted && + (url.searchParams.get('sslmode') !== 'verify-full' || + url.searchParams.get('sslrootcert') !== 'system') + ) { + throw new RetirementCommandError( + 'PlanetScale requires sslmode=verify-full and sslrootcert=system' + ) + } + const databaseId = createHash('sha256') + .update( + JSON.stringify([url.hostname.toLowerCase(), url.port || '5432', url.pathname, url.username]) + ) + .digest('hex') + if (command === 'identity') { + process.stdout.write(`${databaseId}\n`) + return + } + const pageSize = integerOption(values['page-size'], 25, 100) + const pages = integerOption(values.pages, 1, 120) + const budgetMs = integerOption(values.seconds, 60, 600) * 1_000 + if (['prepare', 'cutover'].includes(command) && !values['ack-release-drained']) { + throw new RetirementCommandError( + 'Confirm the code-removal release is fully deployed and old workers/maintenance jobs drained with --ack-release-drained' + ) + } + if (command === 'begin-purge' && !values['ack-retire-backup']) { + throw new RetirementCommandError( + 'Confirm ordinary KB retrieval and live Search after cutover with --ack-retire-backup' + ) + } + let checkHealth = async () => {} + if (!['status', 'abort'].includes(command)) { + const healthFile = values['health-file'] + const policyFile = values['health-policy'] + if (!healthFile || !policyFile) + throw new RetirementCommandError('--health-file and --health-policy are required') + const limits = await readSearchRetirementHealthLimits(policyFile) + if (limits.databaseId !== databaseId) { + throw new RetirementCommandError('Health policy does not match this database connection') + } + checkHealth = async () => { + await readSearchRetirementHealth(healthFile, limits, { + cutover: ['cutover', 'begin-purge', 'finalize'].includes(command), + }) + } + await checkHealth() + } + const sql = postgres(rawUrl, { + port: Number(url.port || '5432'), + ssl: hosted ? 'verify-full' : false, + max: 1, + prepare: false, + connect_timeout: 5, + max_lifetime: null, + connection: { + application_name: 'sim-search-data-retirement', + statement_timeout: 2_000, + lock_timeout: 100, + idle_in_transaction_session_timeout: 5_000, + }, + onnotice: () => undefined, + onclose: () => { + void sql.end({ timeout: 0 }).catch(() => undefined) + }, + }) + try { + if (command === 'status') { + logger.info('Search retirement status', { progress: await getSearchRetirementStatus(sql) }) + return + } + const [{ writable }] = await sql<{ writable: boolean }[]>` + SELECT NOT pg_is_in_recovery() AND current_setting('transaction_read_only') = 'off' AS writable` + if (!writable) + throw new RetirementCommandError('Maintenance requires a writable primary connection') + const [{ acquired }] = await sql<{ acquired: boolean }[]>` + SELECT pg_try_advisory_lock(hashtextextended('sim:search-retirement-maintenance', 0)) AS acquired` + if (!acquired) + throw new RetirementCommandError('Another retirement or index-maintenance command is running') + await checkHealth() + if (command === 'prepare') await initializeSearchRetirement(sql) + else if (command === 'cutover') await cutoverSearchRetirement(sql) + else if (command === 'begin-purge') await beginSearchRetirementPurge(sql) + else if (command === 'finalize') await finalizeSearchRetirement(sql) + else if (command === 'abort') await abortSearchRetirement(sql) + else { + const started = performance.now() + for (let page = 0; page < pages && performance.now() - started < budgetMs; page++) { + await checkHealth() + const before = performance.now() + const progress = await advanceSearchRetirement(sql, { pageSize }) + logger.info('Search retirement page committed', { page: page + 1, progress }) + if (['ready', 'cutover', 'finalize', 'done'].includes(progress.phase)) break + await sleep(Math.max(5_000, (performance.now() - before) * 9)) + } + } + logger.info('Search retirement command finished', { + progress: await getSearchRetirementStatus(sql), + }) + } finally { + await sql.end({ timeout: 5 }) + } +} + +try { + await main() +} catch (error) { + // PostgreSQL errors may contain row values or credentials; never log their query, detail or stack. + const code = getPostgresErrorCode(error) + logger.error('Search retirement stopped; committed pages remain resumable', { + reason: + error instanceof RetireSearchHealthError || + error instanceof RetirementCommandError || + error instanceof SearchRetirementError + ? error.message + : code + ? 'Database refused the operation' + : 'Preflight or command failed; use --help to check options', + ...(code ? { code } : {}), + }) + process.exitCode = 1 +} diff --git a/packages/db/vitest.config.ts b/packages/db/vitest.config.ts index 8ab9cee962d..f8159f6b59c 100644 --- a/packages/db/vitest.config.ts +++ b/packages/db/vitest.config.ts @@ -14,7 +14,12 @@ export default defineConfig(({ mode }) => { test: { include: integration ? ['**/*.integration.ts'] - : ['scripts/**/*.test.ts', 'script-migrations/**/*.test.ts', '*.test.ts'], + : [ + 'scripts/**/*.test.ts', + 'script-migrations/**/*.test.ts', + 'maintenance/**/*.test.ts', + '*.test.ts', + ], ...(integration && { setupFiles: ['./vitest.integration.setup.ts'] }), }, }) From 90e3b918ebaade434f2a375a78d89289c79a3610 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 14:10:00 -0700 Subject: [PATCH 14/31] fix(library): ignore sentence punctuation after bare internal URLs in check:library-content (#8540) * fix(library): ignore sentence punctuation after bare internal URLs in check:library-content * fix(library): only strip trailing punctuation from bare URLs, not explicit link targets --- scripts/check-library-content.test.ts | 20 ++++++++++++++++++++ scripts/check-library-content.ts | 7 ++++++- 2 files changed, 26 insertions(+), 1 deletion(-) diff --git a/scripts/check-library-content.test.ts b/scripts/check-library-content.test.ts index 65db6936e0d..ce8d01c9649 100644 --- a/scripts/check-library-content.test.ts +++ b/scripts/check-library-content.test.ts @@ -77,6 +77,7 @@ describe('check-library-content', () => { [ '## Overview', 'See [the guide](/library/kept-guide) and https://www.sim.ai/library/kept-guide#intro.', + 'Read https://www.sim.ai/library/kept-guide. Then https://www.sim.ai/library/moved-post, too.', '![diagram](/library/clean/diagram.png) and [tags](/library/tags) are not posts.', 'An apex https://sim.ai/library/missing link belongs to check:site-urls.', '| a | b |', @@ -177,6 +178,8 @@ describe('check-library-content', () => { '[b](https://www.sim.ai/library/old-guide)', 'c', '[d](/blog/kept-guide)', + 'Also see https://www.sim.ai/library/gone.', + '[e](/library/kept-guide.) and f', ].join('\n') ) expect(await findingsFor('library', 'post')).toEqual([ @@ -200,6 +203,23 @@ describe('check-library-content', () => { rule: 'internal-link', message: '/blog/kept-guide does not exist (no apps/sim/content/blog/kept-guide/index.mdx).', }, + { + line: 14, + rule: 'internal-link', + message: '/library/gone does not exist (no apps/sim/content/library/gone/index.mdx).', + }, + { + line: 15, + rule: 'internal-link', + message: + '/library/kept-guide. does not exist (no apps/sim/content/library/kept-guide./index.mdx).', + }, + { + line: 15, + rule: 'internal-link', + message: + '/library/kept-guide. does not exist (no apps/sim/content/library/kept-guide./index.mdx).', + }, ]) }) diff --git a/scripts/check-library-content.ts b/scripts/check-library-content.ts index 796ebb36cdd..e59d8432d76 100644 --- a/scripts/check-library-content.ts +++ b/scripts/check-library-content.ts @@ -76,6 +76,10 @@ export interface PostRef { */ const INTERNAL_LINK = /(?:https?:\/\/www\.sim\.ai|(?<=\]\(\s*|href=\{?["'`]))\/(library|blog|customers)\/([^\s)"'`#?/<>\]]+)(\/[^\s)"'`#?<>\]]*)?/g +/** Sentence punctuation that ends a bare URL in prose (`…see https://www.sim.ai/library/x.`). */ +const TRAILING_PUNCTUATION = /[.,;:!]+$/ +/** Text just before a Markdown link target or `href` value, whose URL ends at its delimiter. */ +const LINK_TARGET_OPENER = /(?:\]\(\s*|href=\{?["'`])$/ const MARKDOWN_LINK = /\[[^\]\n]*\]\([^)\n]*\)/ const FAQ_HEADING = /^#{1,6}\s+FAQs?\s*:?\s*$/i const CODE_FENCE = /^\s*(```|~~~)/ @@ -330,7 +334,8 @@ export async function checkPost( } for (const match of text.matchAll(INTERNAL_LINK)) { const linkSection = match[1] as Section - const target = match[2] + const isBareUrl = !LINK_TARGET_OPENER.test(text.slice(0, match.index)) + const target = isBareUrl ? match[2].replace(TRAILING_PUNCTUATION, '') : match[2] const rest = match[3] ?? '' // A deeper path is a public asset (`/library//cover.jpg`) or a static sub-route. if (rest !== '' && rest !== '/') continue From efca99ce4db41914c91f8b74195d79e07910adc8 Mon Sep 17 00:00:00 2001 From: Waleed Date: Thu, 1 Oct 2026 14:48:18 -0700 Subject: [PATCH 15/31] fix(ci): isolate HTTP end-to-end suites and stop the SCIM readiness flake (#8544) * fix(ci): isolate HTTP end-to-end suites and stop the SCIM readiness flake The SCIM step cold-compiles the app under next dev before its first request; readiness took 42-150s on 8 vCPU runners against a fixed 120s deadline, so the slow tail failed. It also ran on both provisioning legs of the integration matrix, which are the PR critical path (~13 min). - Move SCIM and the four search HTTP suites into their own job with its own database, off the integration legs and run once against migrate. - Readiness waits up to 300s, fails fast if the server exits, prints the server log tail, and uploads the server log with the reports. - Factor Bun/Node/cache/install setup into a setup-workspace action. * fix(ci): drop SCIM step timeout and update the SCIM CI guide --- .github/actions/setup-workspace/action.yml | 61 ++++++ .github/workflows/test-build.yml | 238 +++++++-------------- apps/sim/ee/scim/TESTING.md | 19 +- 3 files changed, 154 insertions(+), 164 deletions(-) create mode 100644 .github/actions/setup-workspace/action.yml diff --git a/.github/actions/setup-workspace/action.yml b/.github/actions/setup-workspace/action.yml new file mode 100644 index 00000000000..b6e2e48fdf2 --- /dev/null +++ b/.github/actions/setup-workspace/action.yml @@ -0,0 +1,61 @@ +name: Setup Workspace +description: Install the pinned Bun and Node toolchain, mount the dependency (and optionally Turbo) caches, and install workspace dependencies. + +inputs: + provider: + description: The CI_PROVIDER repo variable, forwarded to cache-mount. + required: false + default: '' + turbo-cache-key: + description: Suffix for a Turbo cache mounted at ./.turbo. Empty skips the mount. Jobs that write Turbo entries need distinct suffixes, or last-writer-wins commits evict each other's entries. + required: false + default: '' + +# Cache keys are scoped by event name, and fork PRs get their own namespace on +# top: untrusted fork runs must never share a cache with push runs (whose caches +# feed production image builds) or with trusted internal-PR runs. +# +# node_modules also keys on the lockfile hash: a sticky disk is a mutable volume, +# and `bun install --frozen-lockfile` adds what the lockfile needs without +# pruning what it dropped, so branches on different lockfiles were contaminating +# each other (a stale @next/swc 16.2.6 outlived the 16.2.11 bump). The bun and +# Turbo caches are content/hash-addressed, so they stay shared — that is what +# keeps a fresh node_modules disk cheap to fill. +runs: + using: composite + steps: + - name: Setup Bun + uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 + with: + bun-version: 1.4.2 + + - name: Setup Node + uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6 + with: + node-version: 24 + + - name: Mount Bun cache + uses: ./.github/actions/cache-mount + with: + provider: ${{ inputs.provider }} + key: ${{ github.repository }}-bun-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} + path: ~/.bun/install/cache + + - name: Mount node_modules + uses: ./.github/actions/cache-mount + with: + provider: ${{ inputs.provider }} + key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }} + path: ./node_modules + + - name: Mount Turbo cache + if: inputs.turbo-cache-key != '' + uses: ./.github/actions/cache-mount + with: + provider: ${{ inputs.provider }} + key: ${{ github.repository }}-${{ inputs.turbo-cache-key }}-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} + path: ./.turbo + + - name: Install dependencies + shell: bash + run: bun install --frozen-lockfile --ignore-scripts diff --git a/.github/workflows/test-build.yml b/.github/workflows/test-build.yml index e62e49a6676..70a44a2b896 100644 --- a/.github/workflows/test-build.yml +++ b/.github/workflows/test-build.yml @@ -8,7 +8,7 @@ permissions: contents: read jobs: - oauth-postgres: + postgres-integration: # Runs the real-infrastructure test layer: every `*.integration.ts` in packages/db and # apps/sim, discovered by glob (`vitest run --mode integration`), against the database each # provisioning path produces. A new integration suite needs no workflow change. @@ -67,25 +67,10 @@ jobs: - name: Checkout code uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - - name: Setup Bun - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 - with: - bun-version: 1.4.2 - - - name: Setup Node - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6 - with: - node-version: 24 - - - name: Mount Bun cache - uses: ./.github/actions/cache-mount + - name: Setup workspace + uses: ./.github/actions/setup-workspace with: provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-bun-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} - path: ~/.bun/install/cache - - - name: Install dependencies - run: bun install --frozen-lockfile --ignore-scripts - name: Provision a fresh database through the supported command working-directory: packages/db @@ -129,78 +114,87 @@ jobs: if-no-files-found: warn retention-days: 14 + # Acceptance suites that cross a real HTTP boundary. Its own job, off the + # integration legs' critical path and on its own database: the SCIM app boots + # hosted, which starts background usage replay against DATABASE_URL, so it must + # never share a database with suites asserting on billing rows. The suites + # exercise HTTP behavior rather than a provisioning path, so they run once, + # against the production (migrate) path. + http-e2e: + name: End-to-end over real HTTP + runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || 'ubuntu-latest' }} + timeout-minutes: 20 + services: + postgres: + image: pgvector/pgvector:pg17 + env: + POSTGRES_USER: postgres + POSTGRES_PASSWORD: postgres + POSTGRES_DB: sim_test + ports: + - 5432:5432 + options: >- + --health-cmd "pg_isready -U postgres -d sim_test" + --health-interval 5s + --health-timeout 5s + --health-retries 10 + env: + DATABASE_URL: postgresql://postgres:postgres@127.0.0.1:5432/sim_test + BETTER_AUTH_SECRET: http-e2e-ci-secret-at-least-32-characters + ENCRYPTION_KEY: '0000000000000000000000000000000000000000000000000000000000000000' + + steps: + - name: Checkout code + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 + + - name: Setup workspace + uses: ./.github/actions/setup-workspace + with: + provider: ${{ vars.CI_PROVIDER }} + + # Migrations create their own extensions, as on a fresh self-hosted install. + - name: Provision the database through migrations + working-directory: packages/db + run: bun run db:migrate + - name: Verify Google document reads over real HTTP - if: matrix.provision == 'push' working-directory: apps/sim env: NEXT_PUBLIC_APP_URL: http://127.0.0.1:3040 NEXT_PUBLIC_FORCE_HOSTED: 'false' - SEARCH_GOOGLE_CONTENT_REPORT_PATH: ${{ runner.temp }}/search-google-content.json + SEARCH_GOOGLE_CONTENT_REPORT_PATH: ${{ runner.temp }}/e2e/search-google-content.json run: bun scripts/test-search-google-content-e2e.ts - - name: Upload Google content acceptance report - if: failure() && matrix.provision == 'push' - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: search-google-content - path: ${{ runner.temp }}/search-google-content.json - if-no-files-found: ignore - retention-days: 7 - - name: Verify Lucid MCP search and complete diagram reads over real HTTP - if: matrix.provision == 'push' working-directory: apps/sim env: NEXT_PUBLIC_APP_URL: http://127.0.0.1:3040 NEXT_PUBLIC_FORCE_HOSTED: 'false' - SEARCH_LUCID_REPORT_PATH: ${{ runner.temp }}/search-lucid.json + SEARCH_LUCID_REPORT_PATH: ${{ runner.temp }}/e2e/search-lucid.json run: bun scripts/test-search-lucid-e2e.ts - - name: Upload Lucid acceptance report - if: failure() && matrix.provision == 'push' - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: search-lucid - path: ${{ runner.temp }}/search-lucid.json - if-no-files-found: ignore - retention-days: 7 - - name: Verify Zoom search over real HTTP - if: matrix.provision == 'push' working-directory: apps/sim env: NEXT_PUBLIC_APP_URL: http://127.0.0.1:3040 NEXT_PUBLIC_FORCE_HOSTED: 'false' - SEARCH_ZOOM_REPORT_PATH: ${{ runner.temp }}/search-zoom.json + SEARCH_ZOOM_REPORT_PATH: ${{ runner.temp }}/e2e/search-zoom.json run: bun scripts/test-search-zoom-e2e.ts - - name: Upload Zoom acceptance report - if: failure() && matrix.provision == 'push' - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: search-zoom - path: ${{ runner.temp }}/search-zoom.json - if-no-files-found: ignore - retention-days: 7 - - name: Verify Google Meet search over real HTTP - if: matrix.provision == 'push' working-directory: apps/sim env: NEXT_PUBLIC_APP_URL: http://127.0.0.1:3040 NEXT_PUBLIC_FORCE_HOSTED: 'false' - SEARCH_GOOGLE_MEET_REPORT_PATH: ${{ runner.temp }}/search-google-meet.json + SEARCH_GOOGLE_MEET_REPORT_PATH: ${{ runner.temp }}/e2e/search-google-meet.json run: bun scripts/test-search-google-meet-e2e.ts - - name: Upload Google Meet acceptance report - if: failure() && matrix.provision == 'push' - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: search-google-meet - path: ${{ runner.temp }}/search-google-meet.json - if-no-files-found: ignore - retention-days: 7 - + # The first request cold-compiles the app under Turbopack, which took 42-150s + # on this runner class: a fixed 120s readiness deadline failed on the slow + # tail. The deadline only has to catch a hung boot; an exited server fails + # immediately, and either way the server log tail lands in the job log. No + # step timeout: the job's bound covers a hang without cutting a slow but + # healthy suite short of writing its report. - name: Verify SCIM and administration over real HTTP working-directory: apps/sim env: @@ -222,42 +216,44 @@ jobs: DISABLE_TELEMETRY: 'true' NEXT_TELEMETRY_DISABLED: '1' NEXT_PUBLIC_CHAT_DISABLED: 'true' + READY_TIMEOUT_SECONDS: 300 run: | - server_log="$RUNNER_TEMP/scim-next.log" + report_dir="$RUNNER_TEMP/e2e" + server_log="$report_dir/scim-next.log" + mkdir -p "$report_dir" node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3017 > "$server_log" 2>&1 & server_pid=$! finish() { kill "$server_pid" 2>/dev/null || true wait "$server_pid" 2>/dev/null || true - awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$RUNNER_TEMP/scim-http-status.log" + awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/scim-http-status.log" } trap finish EXIT - deadline=$((SECONDS + 120)) - until curl --fail --silent --max-time 3 http://127.0.0.1:3017/api/health > /dev/null; do - if ! kill -0 "$server_pid" 2>/dev/null; then - echo 'Local SCIM app exited during startup.' - exit 1 - fi - if [ "$SECONDS" -ge "$deadline" ]; then - echo 'Local SCIM app did not become ready within 120 seconds.' - exit 1 - fi + fail_startup() { + echo "::error::$1" + tail -n 200 "$server_log" + exit 1 + } + started=$SECONDS + until curl --fail --silent --max-time 10 http://127.0.0.1:3017/api/health > /dev/null; do + kill -0 "$server_pid" 2>/dev/null || fail_startup 'Local SCIM app exited during startup.' + [ $((SECONDS - started)) -lt "$READY_TIMEOUT_SECONDS" ] || + fail_startup "Local SCIM app did not become ready within $READY_TIMEOUT_SECONDS seconds." sleep 2 done + echo "Local SCIM app ready after $((SECONDS - started))s" SCIM_E2E_BASE_URL="$NEXT_PUBLIC_APP_URL" \ SCIM_E2E_DATABASE_URL="$DATABASE_URL" \ SCIM_E2E_AUTH_SECRET="$BETTER_AUTH_SECRET" \ - SCIM_E2E_REPORT_PATH="$RUNNER_TEMP/scim-e2e-report.json" \ + SCIM_E2E_REPORT_PATH="$report_dir/scim-e2e-report.json" \ bun run test:scim:e2e - - name: Upload SCIM failure report and HTTP status log + - name: Upload end-to-end reports and server logs if: failure() uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 with: - name: scim-failure-${{ matrix.provision }} - path: | - ${{ runner.temp }}/scim-e2e-report.json - ${{ runner.temp }}/scim-http-status.log + name: http-e2e-reports + path: ${{ runner.temp }}/e2e/ if-no-files-found: ignore retention-days: 7 @@ -280,50 +276,11 @@ jobs: with: fetch-depth: 2 - - name: Setup Bun - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 - with: - bun-version: 1.4.2 - - - name: Setup Node - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6 - with: - node-version: 24 - - # Cache keys are scoped by event name, and fork PRs get their own - # namespace on top: untrusted fork runs must never share a cache with - # push runs (whose caches feed production image builds) or with trusted - # internal-PR runs. - # - # node_modules also keys on the lockfile hash: a sticky disk is a mutable - # volume, and `bun install --frozen-lockfile` adds what the lockfile needs - # without pruning what it dropped, so branches on different lockfiles were - # contaminating each other (a stale @next/swc 16.2.6 outlived the 16.2.11 - # bump). The bun and Turbo caches are content/hash-addressed, so they stay - # shared — that is what keeps a fresh node_modules disk cheap to fill. - - name: Mount Bun cache - uses: ./.github/actions/cache-mount - with: - provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-bun-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} - path: ~/.bun/install/cache - - - name: Mount node_modules - uses: ./.github/actions/cache-mount + - name: Setup workspace + uses: ./.github/actions/setup-workspace with: provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }} - path: ./node_modules - - - name: Mount Turbo cache - uses: ./.github/actions/cache-mount - with: - provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-turbo-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} - path: ./.turbo - - - name: Install dependencies - run: bun install --frozen-lockfile --ignore-scripts + turbo-cache-key: turbo-cache # Surfaces known CVEs in the dependency tree. Non-blocking until the # existing advisory backlog is triaged, then flip to a required gate by @@ -375,9 +332,6 @@ jobs: # # Depth stays at 1 — without a merge-base the migration audit diffs the two # tips, which under `--diff-filter=AM` is exactly the migrations new here. - # Resolved once for both diff-based audits, and never with `|| true`: a - # swallowed fetch leaves the base absent, which neither audit can tell apart - # from a branch that changed nothing. # # On push the base is `github.event.before`, the tip the branch had before # this push — not `HEAD~1`, which names only the last commit and would let a @@ -481,36 +435,11 @@ jobs: - name: Checkout code uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - - name: Setup Bun - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 - with: - bun-version: 1.4.2 - - - name: Setup Node - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6 - with: - node-version: 24 - - - name: Mount Bun cache - uses: ./.github/actions/cache-mount - with: - provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-bun-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} - path: ~/.bun/install/cache - - - name: Mount node_modules - uses: ./.github/actions/cache-mount + - name: Setup workspace + uses: ./.github/actions/setup-workspace with: provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }} - path: ./node_modules - - - name: Mount Turbo cache - uses: ./.github/actions/cache-mount - with: - provider: ${{ vars.CI_PROVIDER }} - key: ${{ github.repository }}-turbo-cache-build-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }} - path: ./.turbo + turbo-cache-key: turbo-cache-build # No `.next/cache` mount: the Turbopack persistent build cache is off. A # controlled A/B on one branch (PR #6078) with a byte-identical module graph @@ -534,9 +463,6 @@ jobs: echo "::warning::Runner has ${TOTAL_GB} GB. A cold-cache build peaks ~51 GB, so this run may be OOM-killed (reported only as 'the runner has received a shutdown signal'). Warm/partial builds should still fit." fi - - name: Install dependencies - run: bun install --frozen-lockfile --ignore-scripts - - name: Build application env: NODE_OPTIONS: '--no-warnings --max-old-space-size=8192' diff --git a/apps/sim/ee/scim/TESTING.md b/apps/sim/ee/scim/TESTING.md index bf56e6fc546..bb73c2a942f 100644 --- a/apps/sim/ee/scim/TESTING.md +++ b/apps/sim/ee/scim/TESTING.md @@ -76,13 +76,15 @@ timeout. ## Continuous integration and PostgreSQL regressions -The `PostgreSQL integration` job in `.github/workflows/test-build.yml` runs -against both supported database provisioning paths, `db:push` and `db:migrate`. -After the integration layer (every `*.integration.ts`), it starts a local Next.js app with -hosted Enterprise configuration and runs the HTTP suite above. Startup is -bounded to 120 seconds; the server is stopped when the step exits. A failure -uploads the credential-free scenario report and an allowlist of HTTP status log -lines. Raw application logs are not uploaded. +The `End-to-end over real HTTP` job in `.github/workflows/test-build.yml` +provisions its own database through `db:migrate`, starts a local Next.js app +with hosted Enterprise configuration, and runs the HTTP suite above. It has its +own database because the hosted app runs background usage replay against it. +Readiness is bounded to 300 seconds, since the first request cold-compiles the +app; the server is stopped when the step exits. A failure prints the server log +tail and uploads the credential-free scenario report, an allowlist of HTTP +status log lines, and the local app's server log. The CI app uses only fixture +secrets. The focused PostgreSQL suite is part of the integration layer and reads the same `TEST_DATABASE_URL` as every other integration suite: @@ -100,7 +102,8 @@ transaction. Hosted billing flags are configured for the test; subscription and entitlement reads use real PostgreSQL with the transaction tripwire enabled. The suite checks an active Enterprise subscription, an ended one, and a real billing query failure that must propagate instead of releasing directory locks. -The same CI job runs `lib/auth/sso/application/admit-sso-user.integration.ts`, +The `PostgreSQL integration` job, which runs every `*.integration.ts` against +both `db:push` and `db:migrate`, also runs `lib/auth/sso/application/admit-sso-user.integration.ts`, which verifies that SCIM's `disableJit` setting blocks fresh SSO membership, preserves existing membership, and permits JIT when disabled. These checks run the admission operation and Enterprise entitlement reads through PostgreSQL. From ec1a997dab824e1b5c371db99fa9101b7ff8885d Mon Sep 17 00:00:00 2001 From: Theodore Li Date: Thu, 1 Oct 2026 14:53:30 -0700 Subject: [PATCH 16/31] fix(dashboards): keep DST fall-back range endpoints and resolve threshold axis from first series (#8541) * fix(dashboards): keep DST fall-back range endpoints and resolve threshold axis from first series * fix(dashboards): reject invalid first-series axis refs and show exclusive end for sub-minute ranges * fix(dashboards): show inclusive range end in exact caption and cover id/x-axis threshold lookups * fix(charts): resolve threshold axis by index before id, matching ECharts --- apps/sim/lib/charts/annotations.test.ts | 94 +++++++++++++++++++++++++ apps/sim/lib/charts/annotations.ts | 38 ++++++++-- apps/sim/lib/dashboards/time.test.ts | 18 +++++ apps/sim/lib/dashboards/time.ts | 19 +++++ 4 files changed, 163 insertions(+), 6 deletions(-) diff --git a/apps/sim/lib/charts/annotations.test.ts b/apps/sim/lib/charts/annotations.test.ts index 60aafc426bd..280936cb110 100644 --- a/apps/sim/lib/charts/annotations.test.ts +++ b/apps/sim/lib/charts/annotations.test.ts @@ -116,4 +116,98 @@ describe('chart annotations', () => { ) ).toThrow('Use highlights and thresholds instead of markArea or markLine on the series') }) + + it('measures thresholds on the axes the first series is plotted on', () => { + const option = applyChartAnnotations( + { + xAxis: { type: 'time' }, + yAxis: [{ type: 'category' }, { id: 'latency', type: 'value' }], + series: [{ type: 'line', yAxisIndex: 1 }], + }, + { thresholds: [{ value: 5 }] }, + palette + ) + expect((option.series as Series[])[0].markLine).toMatchObject({ data: [{ yAxis: 5 }] }) + expect(() => + applyChartAnnotations( + { + xAxis: { type: 'time' }, + yAxis: [{ type: 'value' }, { id: 'stage', type: 'category' }], + series: [{ type: 'line', yAxisId: 'stage' }], + }, + { thresholds: [{ value: 5 }] }, + palette + ) + ).toThrow('Thresholds require a value axis') + }) + + it('resolves a value axis selected by id, including a value x-axis on horizontal bars', () => { + const byId = applyChartAnnotations( + { + xAxis: { type: 'time' }, + yAxis: [{ type: 'category' }, { id: 'latency', type: 'value' }], + series: [{ type: 'line', yAxisId: 'latency' }], + }, + { thresholds: [{ value: 5 }] }, + palette + ) + expect((byId.series as Series[])[0].markLine).toMatchObject({ data: [{ yAxis: 5 }] }) + const horizontal = applyChartAnnotations( + { + xAxis: [{ type: 'category' }, { id: 'count', type: 'value' }], + yAxis: [{ type: 'value' }, { id: 'stage', type: 'category', inverse: true }], + series: [{ type: 'bar', xAxisId: 'count', yAxisIndex: 1 }], + }, + { thresholds: [{ value: 5, label: 'Limit' }] }, + palette + ) + const [first, labels] = horizontal.series as Series[] + expect(first.markLine).toMatchObject({ data: [{ xAxis: 5 }] }) + expect(labels.markLine).toMatchObject({ data: [{ xAxis: 5, label: { position: 'start' } }] }) + }) + + it('prefers the axis index over the axis id, as ECharts does', () => { + const option = applyChartAnnotations( + { + xAxis: { type: 'time' }, + yAxis: [ + { id: 'stage', type: 'category' }, + { id: 7, type: 'value' }, + ], + series: [{ type: 'line', yAxisIndex: 1, yAxisId: 'stage' }], + }, + { thresholds: [{ value: 5 }] }, + palette + ) + expect((option.series as Series[])[0].markLine).toMatchObject({ data: [{ yAxis: 5 }] }) + expect(() => + applyChartAnnotations( + { ...option, series: [{ type: 'line', yAxisId: '7' }] }, + { thresholds: [{ value: 5 }] }, + palette + ) + ).not.toThrow() + }) + + it('rejects a first series that references an axis the chart does not define', () => { + for (const reference of [{ yAxisIndex: 1 }, { yAxisIndex: -1 }, { yAxisIndex: 0.5 }]) + expect(() => + applyChartAnnotations( + { + xAxis: { type: 'time' }, + yAxis: { type: 'value' }, + series: [{ type: 'line', ...reference }], + }, + { thresholds: [{ value: 5 }] }, + palette + ) + ).toThrow('The first series references a missing yAxis') + expect(() => + applyChartAnnotations( + { xAxis: { type: 'time' }, yAxis: [], series: [{ type: 'line' }] }, + { thresholds: [{ value: 5 }] }, + palette + ) + ).toThrow('The first series references a missing yAxis') + }) }) diff --git a/apps/sim/lib/charts/annotations.ts b/apps/sim/lib/charts/annotations.ts index 79c41a1e2f6..93da7c9758c 100644 --- a/apps/sim/lib/charts/annotations.ts +++ b/apps/sim/lib/charts/annotations.ts @@ -27,17 +27,43 @@ export interface ChartAnnotations { const BAND_OPACITY = 0.08 -function firstAxis(axis: unknown): Record { - return toRecord(Array.isArray(axis) ? axis[0] : axis) +function firstSeries(option: Record): Record { + return toRecord(Array.isArray(option.series) ? option.series[0] : option.series) } -/** The axis a threshold is measured on; ECharts defaults an unspecified yAxis to a value axis. */ +/** + * The axis the first series is plotted on. Like ECharts, `*AxisIndex` wins over `*AxisId`, and ids + * match across string and number. + */ +function seriesAxis( + option: Record, + key: 'xAxis' | 'yAxis' +): Record { + const axes = Array.isArray(option[key]) ? (option[key] as unknown[]) : [option[key]] + const series = firstSeries(option) + const id = series[`${key}Id`] + if (series[`${key}Index`] === undefined && id !== undefined) { + const axis = axes.find((candidate) => String(toRecord(candidate).id) === String(id)) + if (axis === undefined) throw new Error(`The first series references a missing ${key} "${id}"`) + return toRecord(axis) + } + const index = series[`${key}Index`] ?? 0 + if (option[key] === undefined && index === 0) return {} + if (!Number.isInteger(index) || axes[index as number] === undefined) + throw new Error(`The first series references a missing ${key} at index ${String(index)}`) + return toRecord(axes[index as number]) +} + +/** + * The axis a threshold is measured on, among the axes the first series is plotted on; ECharts + * defaults an unspecified yAxis to a value axis. + */ export function valueAxisKey(option: Record): 'xAxis' | 'yAxis' { if (option.xAxis === undefined && option.yAxis === undefined) throw new Error('Thresholds require a value axis') - const y = firstAxis(option.yAxis) + const y = seriesAxis(option, 'yAxis') if (y.type === undefined || y.type === 'value' || y.type === 'log') return 'yAxis' - const x = firstAxis(option.xAxis) + const x = seriesAxis(option, 'xAxis') if (x.type === 'value' || x.type === 'log') return 'xAxis' throw new Error('Thresholds require a value axis') } @@ -47,7 +73,7 @@ export function valueAxisKey(option: Record): 'xAxis' | 'yAxis' * y axis is inverted (as horizontal bar charts usually are). */ function verticalTop(option: Record): 'start' | 'end' { - return firstAxis(option.yAxis).inverse === true ? 'start' : 'end' + return seriesAxis(option, 'yAxis').inverse === true ? 'start' : 'end' } /** A chart annotations can draw on: a first series with no hand-written marks to collide with. */ diff --git a/apps/sim/lib/dashboards/time.test.ts b/apps/sim/lib/dashboards/time.test.ts index 04e91090222..50f658a83eb 100644 --- a/apps/sim/lib/dashboards/time.test.ts +++ b/apps/sim/lib/dashboards/time.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import { dashboardAxisFormatter, dashboardRangeFromCalendar, + dashboardRangeText, dashboardTimeLabel, dashboardZoomRange, parseDashboardCustomRange, @@ -47,4 +48,21 @@ describe('dashboard time interactions', () => { expect(dashboardTimeLabel(stamp, 'America/Los_Angeles')).toContain('PDT') expect(dashboardTimeLabel('2026-12-20T02:30:00Z', 'America/Los_Angeles')).toContain('PST') }) + + it('keeps both ends of a range that repeats the same wall-clock minute across a DST fall-back', () => { + const text = dashboardRangeText( + { from: '2026-11-01T05:30:00.000Z', to: '2026-11-01T06:31:00.000Z' }, + 'America/New_York' + ) + expect(text).toBe('Nov 1, 01:30:00 EDT – Nov 1, 01:30:59 EST') + }) + + it('shows distinct endpoints for a sub-minute zoom range', () => { + expect( + dashboardRangeText( + { from: '2026-09-20T14:30:00.000Z', to: '2026-09-20T14:30:01.000Z' }, + 'UTC' + ) + ).toBe('Sep 20, 14:30:00.000 UTC – Sep 20, 14:30:00.999 UTC') + }) }) diff --git a/apps/sim/lib/dashboards/time.ts b/apps/sim/lib/dashboards/time.ts index d3a8dfd9a83..16012862cb9 100644 --- a/apps/sim/lib/dashboards/time.ts +++ b/apps/sim/lib/dashboards/time.ts @@ -89,6 +89,25 @@ export function dashboardRangeText(range: DashboardTimeRange, timeZone: string): const to = new Date(Date.parse(range.to) - 1) const fromLocal = zonedWallClock(from, timeZone) const toLocal = zonedWallClock(to, timeZone) + if (fromLocal === toLocal) { + // Same wall-clock minute (e.g. a DST fall-back repeat): formatRange would collapse both ends, + // so show each inclusive end exactly, down to milliseconds if seconds still collide. + const exact = (fractionalSecondDigits?: 3) => + new Intl.DateTimeFormat('en-US', { + timeZone, + month: 'short', + day: 'numeric', + hour: '2-digit', + minute: '2-digit', + second: '2-digit', + fractionalSecondDigits, + hourCycle: 'h23', + timeZoneName: 'short', + }) + const seconds = exact() + const format = seconds.format(from) === seconds.format(to) ? exact(3) : seconds + return `${format.format(from)} – ${format.format(to)}` + } const sameDay = fromLocal.slice(0, 10) === toLocal.slice(0, 10) return new Intl.DateTimeFormat('en-US', { timeZone, From 428a945ea2174920e5133cb040560bf0e1c01060 Mon Sep 17 00:00:00 2001 From: Vikhyath Mondreti Date: Thu, 1 Oct 2026 15:01:06 -0700 Subject: [PATCH 17/31] fix(search): correct onboarding and live citations (#8542) * fix(search): correct onboarding and live citations * fix(search): clear onboarding readiness after credential disconnect --- .../components/get-started/get-started.tsx | 12 +- .../knowledge/search-integrations.ts | 1 + .../search-mcp-setup.integration.ts | 218 ++++++++++++++++++ .../application/search-integrations.ts | 107 ++++++++- .../sim/lib/knowledge/search/citation.test.ts | 53 ----- apps/sim/lib/knowledge/search/citation.ts | 28 --- .../lib/knowledge/search/source-url.test.ts | 26 +++ .../mothership/generated/tool-catalog-v1.ts | 1 - .../mothership/generated/tool-schemas-v1.ts | 4 - .../server/knowledge/workspace-search.test.ts | 74 ++++++ .../server/knowledge/workspace-search.ts | 21 +- packages/db/schema.ts | 18 +- 12 files changed, 444 insertions(+), 119 deletions(-) delete mode 100644 apps/sim/lib/knowledge/search/citation.test.ts create mode 100644 apps/sim/lib/knowledge/search/source-url.test.ts diff --git a/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx b/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx index f9e42b544c8..42f8f800cac 100644 --- a/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx +++ b/apps/sim/app/o/[organizationId]/home/components/get-started/get-started.tsx @@ -71,7 +71,7 @@ function StepMark({ complete }: { complete: boolean }) { * The organization home's onboarding list under the composer. Same chrome as * the workspace home's suggested actions: a hover-revealed disclosure header * over hairline-separated rows. Each step leads to the page that completes it, - * and reads as done from the organization's real state: a connected account and an OAuth app authorized to use Search. + * and reads as done from the organization's real state: a configured integration and an OAuth app authorized to use Search. */ export function GetStarted() { const { organization, viewer, connectedAccountsAvailable } = useOrganizationContext() @@ -142,7 +142,15 @@ export function GetStarted() { 'connect-sim-search': routes.settingsSection('search-mcp'), } const completed: Record = { - 'connect-integration': Boolean(hasSearchConnection), + 'connect-integration': Boolean( + hasSearchConnection || + integrations?.some( + (integration) => + integration.approved && + integration.available !== false && + integration.configuredServiceSource + ) + ), 'connect-sim-search': hasSearchAuthorization, } const steps = STEPS.filter((step) => diff --git a/apps/sim/lib/api/contracts/knowledge/search-integrations.ts b/apps/sim/lib/api/contracts/knowledge/search-integrations.ts index b0e199baef0..ba92d243653 100644 --- a/apps/sim/lib/api/contracts/knowledge/search-integrations.ts +++ b/apps/sim/lib/api/contracts/knowledge/search-integrations.ts @@ -13,6 +13,7 @@ export type SearchIntegrationApproval = z.output diff --git a/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts b/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts index 41edfd940a4..fc9bc7f2756 100644 --- a/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts +++ b/apps/sim/lib/knowledge/__integration__/search-mcp-setup.integration.ts @@ -3,6 +3,8 @@ import { credential, credentialGroup, credentialGroupEnrollment, + knowledgeBase, + knowledgeConnector, mcpServers, member, organization, @@ -30,11 +32,16 @@ import { import { getCredentialGroup } from '@/lib/credential-groups/service' import { SLACK_MANAGED_USER_SCOPES } from '@/lib/credential-groups/slack-managed-user-scopes' import { createOrganizationAccountsGroup } from '@/lib/credential-groups/workspace-accounts' +import { deleteConnectionCredential } from '@/lib/credentials/deletion' import { acquireAdvisoryXactLock, tryAcquireAdvisoryXactLock } from '@/lib/db/advisory-locks' import { approveSearchIntegration, listSearchIntegrations, } from '@/lib/knowledge/application/search-integrations' +import { + GITHUB_INSTALLATION_PROVIDER_ID, + type GitHubInstallationBinding, +} from '@/lib/oauth/github-installation-types' import { defaultLiveSearchPolicy } from '@/lib/sim-search/live/policy-schema' import { SLACK_RTS_USER_SCOPES } from '@/lib/sim-search/live/scopes' @@ -247,6 +254,217 @@ describe('atomic organization live Search MCP setup', () => { } ) + async function seedServiceSource(provider: 'google_drive' | 'github' | 'gitlab') { + const knowledgeBaseId = generateId() + const connectorId = generateId() + const credentialId = generateId() + const installation = { + type: 'github_app_installation', + version: 1, + appId: '1', + appClientId: 'fixture-github-app', + installationId: '21', + accountId: '11', + accountType: 'Organization', + accountLogin: 'fixture-owner', + repositorySelection: 'selected', + } satisfies GitHubInstallationBinding + const encryptedInstallation = + provider === 'github' ? await encryptSecret(JSON.stringify(installation)) : undefined + await db.insert(knowledgeBase).values({ + id: knowledgeBaseId, + userId: ids.owner, + organizationId: ids.organization, + isSearchIndex: true, + name: 'Service source fixture', + }) + await db.insert(credential).values({ + id: credentialId, + organizationId: ids.organization, + type: 'service_account', + providerId: provider === 'github' ? GITHUB_INSTALLATION_PROVIDER_ID : 'google-drive', + ...(encryptedInstallation + ? { + encryptedServiceAccountKey: encryptedInstallation.encrypted, + providerSubjectId: installation.installationId, + providerTenantId: installation.accountId, + authorizationAppId: installation.appClientId, + } + : {}), + displayName: 'Service source fixture', + createdBy: ids.owner, + }) + await db.insert(knowledgeConnector).values({ + id: connectorId, + knowledgeBaseId, + connectorType: provider, + credentialId, + encryptedApiKey: provider === 'gitlab' ? 'synthetic-encrypted-key' : null, + sourceConfig: + provider === 'github' + ? { repository: 'fixture-owner/repository', githubRepositoryId: '101' } + : {}, + accessMode: provider === 'github' ? 'members' : 'admin', + status: 'active', + }) + await db.insert(organizationSearchIntegration).values({ + organizationId: ids.organization, + connectorType: provider, + approved: true, + }) + await db + .update(organization) + .set({ + metadata: { + liveSearchPolicies: { + [provider]: { + ...defaultLiveSearchPolicy(provider), + accessMode: 'service_account', + ...(provider === 'google_drive' ? { sourceId: connectorId } : {}), + }, + }, + }, + }) + .where(eq(organization.id, ids.organization)) + return { knowledgeBaseId, connectorId, credentialId } + } + + async function integrationStatus(provider: string) { + const data = await listSearchIntegrations.execute({ + principal: createSessionPrincipal({ userId: ids.member, sessionId: generateId() }), + input: { organizationId: ids.organization }, + }) + return listSearchIntegrationsContract.response.schema + .parse({ success: true, data }) + .data.find((entry) => entry.connectorType === provider) + } + + it.each(['google_drive', 'github', 'gitlab'] as const)( + 'reports a configured %s service source without requiring a member account', + async (provider) => { + const source = await seedServiceSource(provider) + expect((await snapshot()).groups).toEqual([]) + expect(await integrationStatus(provider)).toMatchObject({ configuredServiceSource: true }) + await db + .update(knowledgeConnector) + .set({ status: 'disabled' }) + .where(eq(knowledgeConnector.id, source.connectorId)) + expect(await integrationStatus(provider)).toMatchObject({ configuredServiceSource: false }) + await db + .update(knowledgeConnector) + .set({ status: 'active', archivedAt: new Date() }) + .where(eq(knowledgeConnector.id, source.connectorId)) + expect(await integrationStatus(provider)).toMatchObject({ configuredServiceSource: false }) + await db + .update(knowledgeConnector) + .set({ archivedAt: null }) + .where(eq(knowledgeConnector.id, source.connectorId)) + await db + .update(knowledgeBase) + .set({ deletedAt: new Date() }) + .where(eq(knowledgeBase.id, source.knowledgeBaseId)) + expect(await integrationStatus(provider)).toMatchObject({ configuredServiceSource: false }) + await db + .update(knowledgeBase) + .set({ deletedAt: null }) + .where(eq(knowledgeBase.id, source.knowledgeBaseId)) + await db + .update(organizationSearchIntegration) + .set({ approved: false }) + .where(eq(organizationSearchIntegration.organizationId, ids.organization)) + expect(await integrationStatus(provider)).toMatchObject({ configuredServiceSource: false }) + } + ) + + it('stops reporting a service source as configured after its credential is deleted', async () => { + const source = await seedServiceSource('google_drive') + expect(await integrationStatus('google_drive')).toMatchObject({ configuredServiceSource: true }) + await deleteConnectionCredential({ + credentialId: source.credentialId, + organizationId: ids.organization, + reason: 'user_delete', + }) + expect(await integrationStatus('google_drive')).toMatchObject({ + configuredServiceSource: false, + }) + }) + + it('requires the selected service source to belong to this organization and provider', async () => { + const source = await seedServiceSource('google_drive') + const otherOrganizationId = generateId() + await db.insert(organization).values({ + id: otherOrganizationId, + name: 'Other service fixture', + slug: otherOrganizationId, + }) + try { + await db + .update(knowledgeBase) + .set({ organizationId: otherOrganizationId }) + .where(eq(knowledgeBase.id, source.knowledgeBaseId)) + expect(await integrationStatus('google_drive')).toMatchObject({ + configuredServiceSource: false, + }) + await db + .update(knowledgeBase) + .set({ organizationId: ids.organization }) + .where(eq(knowledgeBase.id, source.knowledgeBaseId)) + await db + .update(knowledgeConnector) + .set({ connectorType: 'confluence' }) + .where(eq(knowledgeConnector.id, source.connectorId)) + expect(await integrationStatus('google_drive')).toMatchObject({ + configuredServiceSource: false, + }) + await db + .update(knowledgeConnector) + .set({ connectorType: 'google_drive' }) + .where(eq(knowledgeConnector.id, source.connectorId)) + await db + .update(organization) + .set({ + metadata: { + liveSearchPolicies: { + google_drive: { + ...defaultLiveSearchPolicy(), + accessMode: 'service_account', + sourceId: generateId(), + }, + }, + }, + }) + .where(eq(organization.id, ids.organization)) + expect(await integrationStatus('google_drive')).toMatchObject({ + configuredServiceSource: false, + }) + } finally { + await db.delete(organization).where(eq(organization.id, otherOrganizationId)) + } + }) + + it('does not count a GitHub member source without its active installation credential', async () => { + const source = await seedServiceSource('github') + await db + .update(credential) + .set({ revokedAt: new Date() }) + .where(eq(credential.id, source.credentialId)) + expect(await integrationStatus('github')).toMatchObject({ configuredServiceSource: false }) + await db + .update(credential) + .set({ revokedAt: null }) + .where(eq(credential.id, source.credentialId)) + await db + .update(knowledgeConnector) + .set({ memberSyncStatus: 'disabled' }) + .where(eq(knowledgeConnector.id, source.connectorId)) + expect(await integrationStatus('github')).toMatchObject({ configuredServiceSource: false }) + await db + .update(knowledgeConnector) + .set({ memberSyncStatus: 'idle', sourceConfig: {} }) + .where(eq(knowledgeConnector.id, source.connectorId)) + expect(await integrationStatus('github')).toMatchObject({ configuredServiceSource: false }) + }) + it('keeps disabled Zoom approvals visible and removable without permitting reapproval', async () => { const connectorType = 'zoom' await db.insert(organizationSearchIntegration).values({ diff --git a/apps/sim/lib/knowledge/application/search-integrations.ts b/apps/sim/lib/knowledge/application/search-integrations.ts index 2e5b97ca5db..9ed99a094bd 100644 --- a/apps/sim/lib/knowledge/application/search-integrations.ts +++ b/apps/sim/lib/knowledge/application/search-integrations.ts @@ -1,17 +1,25 @@ import { AuditAction, AuditResourceType } from '@sim/audit' import { requirePrincipalSubjectUserId } from '@sim/auth/principal' import { db } from '@sim/db' -import { organization, organizationSearchIntegration } from '@sim/db/schema' -import { eq, sql } from 'drizzle-orm' +import { + credential, + knowledgeBase, + knowledgeConnector, + organization, + organizationSearchIntegration, +} from '@sim/db/schema' +import { and, eq, exists, inArray, isNotNull, isNull, ne, or, sql } from 'drizzle-orm' import { OrchestrationError } from '@/lib/core/orchestration/types' import { CredentialGroupProviderConfigurationError } from '@/lib/credential-groups/provider-adapter' import { isScopedCredentialGroupsAvailable } from '@/lib/credential-groups/scoped-availability' import { addOrganizationAccountProvider } from '@/lib/credential-groups/service' import { SLACK_SEARCH_USER_SCOPES } from '@/lib/credential-groups/slack-managed-user-scopes' +import { resolveKnowledgeAccessAvailability } from '@/lib/knowledge/access/availability' import { defineAuthorizedKnowledgeUseCase } from '@/lib/knowledge/application/authorized-knowledge-use-case' import { resolveKnowledgeOwnerContext } from '@/lib/knowledge/application/contexts' import { knowledgeOperations } from '@/lib/knowledge/application/operations' import { listOrganizationSearchApprovals } from '@/lib/knowledge/search/integration-policy' +import { GITHUB_INSTALLATION_PROVIDER_ID } from '@/lib/oauth/github-installation-types' import { refuseCapability } from '@/lib/permission-groups/capabilities' import { isOrganizationCapabilityWithheld } from '@/lib/permission-groups/capability-assertions' import { NativeSearchError } from '@/lib/sim-search/live/http' @@ -26,6 +34,7 @@ import { normalizeLiveSearchPolicy, } from '@/lib/sim-search/live/policy-schema' import { livePolicyFor, loadLiveSearchPolicies } from '@/lib/sim-search/live/policy-store' +import { supportsLiveSearchMode } from '@/lib/sim-search/live/provider-catalog' import { isSearchProviderEnabled } from '@/lib/sim-search/live/provider-rollout' import { loadLiveServiceSource } from '@/lib/sim-search/live/service-sources' import { @@ -33,6 +42,7 @@ import { liveSearchMcpConnector, liveSearchMemberAccountProvider, } from '@/lib/sim-search/live/source-catalog' +import { getConnectorMeta } from '@/connectors/registry' interface SearchIntegrationInput { organizationId: string @@ -54,7 +64,7 @@ export const listSearchIntegrations = defineAuthorizedKnowledgeUseCase({ const policies = await loadLiveSearchPolicies({ organizationId: context.organizationId }) const scope = { kind: 'organization', organizationId: context.organizationId } as const const zoomEnabled = await isSearchProviderEnabled('zoom', scope) - return LIVE_SEARCH_SOURCE_TYPES.map(([connectorType]) => ({ + const integrations = LIVE_SEARCH_SOURCE_TYPES.map(([connectorType]) => ({ connectorType, approved: approvals.get(connectorType) ?? false, ...(policies @@ -64,6 +74,97 @@ export const listSearchIntegrations = defineAuthorizedKnowledgeUseCase({ } : {}), })) + const servicePolicies = integrations.filter( + (integration) => + integration.approved && + integration.available !== false && + integration.policy?.accessMode === 'service_account' && + supportsLiveSearchMode(integration.connectorType, 'service_account') + ) + const availability = servicePolicies.length + ? await resolveKnowledgeAccessAvailability(context) + : null + const candidates = servicePolicies.filter( + (integration) => + availability?.sourceMirrored && + (availability.memberScoped || + (integration.connectorType !== 'github' && + !getConnectorMeta(integration.connectorType)?.requiresMemberIdentity)) + ) + const configured = candidates.length + ? await db + .select({ provider: sql`requested.provider` }) + .from( + sql`(VALUES ${sql.join( + candidates.map( + (integration) => + sql`(${integration.connectorType}::text, ${integration.policy?.sourceId ?? null}::text)` + ), + sql`, ` + )}) AS requested(provider, source_id)` + ) + .where( + exists( + db + .select({ id: knowledgeConnector.id }) + .from(knowledgeConnector) + .innerJoin(knowledgeBase, eq(knowledgeBase.id, knowledgeConnector.knowledgeBaseId)) + .where( + and( + eq(knowledgeBase.organizationId, context.organizationId), + eq(knowledgeBase.isSearchIndex, true), + isNull(knowledgeBase.deletedAt), + eq(knowledgeConnector.connectorType, sql`requested.provider`), + inArray(knowledgeConnector.status, ['active', 'pending', 'syncing', 'error']), + isNull(knowledgeConnector.archivedAt), + isNull(knowledgeConnector.deletedAt), + or( + and( + sql`requested.provider = 'github'`, + eq(knowledgeConnector.accessMode, 'members'), + ne(knowledgeConnector.memberSyncStatus, 'disabled'), + sql`jsonb_typeof(${knowledgeConnector.sourceConfig}::jsonb->'githubRepositoryId') = 'string'`, + exists( + db + .select({ id: credential.id }) + .from(credential) + .where( + and( + eq(credential.id, knowledgeConnector.credentialId), + eq(credential.organizationId, context.organizationId), + eq(credential.type, 'service_account'), + eq(credential.providerId, GITHUB_INSTALLATION_PROVIDER_ID), + isNull(credential.revokedAt) + ) + ) + ) + ), + and( + sql`requested.provider <> 'github'`, + eq(knowledgeConnector.accessMode, 'admin'), + or( + and( + sql`requested.provider = 'gitlab'`, + isNotNull(knowledgeConnector.encryptedApiKey) + ), + and( + sql`requested.provider <> 'gitlab'`, + eq(knowledgeConnector.id, sql`requested.source_id`), + isNotNull(knowledgeConnector.credentialId) + ) + ) + ) + ) + ) + ) + ) + ) + : [] + const configuredProviders = new Set(configured.map((row) => row.provider)) + return integrations.map((integration) => ({ + ...integration, + configuredServiceSource: configuredProviders.has(integration.connectorType), + })) }, }) diff --git a/apps/sim/lib/knowledge/search/citation.test.ts b/apps/sim/lib/knowledge/search/citation.test.ts deleted file mode 100644 index d9332b9c65b..00000000000 --- a/apps/sim/lib/knowledge/search/citation.test.ts +++ /dev/null @@ -1,53 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { createKnowledgeDocumentCitation } from '@/lib/knowledge/search/citation' -import { isKnowledgeSourceUrl } from '@/lib/knowledge/search/source-url' - -const input = { - scope: { kind: 'workspace', workspaceId: 'workspace/a' } as const, - knowledgeBaseId: 'kb/b', - documentId: 'doc/c', - baseUrl: 'https://www.sim.ai', - sourceUrl: 'https://docs.google.com/document/d/abc/edit?tab=t.0#heading', -} - -describe('knowledge citations', () => { - it.each([ - null, - '', - 'javascript:alert(1)', - 'data:text/html,secret', - 'https://user:secret@source.test/doc', - 'https://source.test/with\nnewline', - 'https://source.test\\@evil.test/doc', - '/relative/path', - 'https:source.test/doc', - '//source.test/doc', - ])('uses a scoped Sim link when the source URL is %s', (sourceUrl) => { - expect(createKnowledgeDocumentCitation({ ...input, sourceUrl }).citationUrl).toBe( - 'https://www.sim.ai/workspace/workspace%2Fa/knowledge/kb%2Fb/doc%2Fc' - ) - }) - - it('links organization documents under their own organization', () => { - expect( - createKnowledgeDocumentCitation({ - ...input, - scope: { kind: 'organization', organizationId: 'org/a' }, - sourceUrl: null, - }).citationUrl - ).toBe('https://www.sim.ai/o/org%2Fa/knowledge/kb%2Fb/doc%2Fc') - }) - - it.each(['file:///tmp/app', 'javascript:alert(1)', 'https://secret@sim.ai'])( - 'rejects unsafe application URL %s', - (baseUrl) => { - expect(() => createKnowledgeDocumentCitation({ ...input, baseUrl })).toThrow( - 'Invalid citation base URL' - ) - } - ) - - it('rejects whitespace in exact provider references', () => { - expect(isKnowledgeSourceUrl(` ${input.sourceUrl}`)).toBe(false) - }) -}) diff --git a/apps/sim/lib/knowledge/search/citation.ts b/apps/sim/lib/knowledge/search/citation.ts index 02880594948..c6eb6489991 100644 --- a/apps/sim/lib/knowledge/search/citation.ts +++ b/apps/sim/lib/knowledge/search/citation.ts @@ -1,34 +1,6 @@ import { sha256Hex } from '@sim/security/hash' -import type { ResourceScope } from '@/lib/core/resource-scope' -import { isKnowledgeSourceUrl } from '@/lib/knowledge/search/source-url' /** Stable opaque citation IDs keep the model from rewriting long live references. */ export function liveCitationId(documentId: string): string { return `live:${sha256Hex(documentId).slice(0, 32)}` } - -interface KnowledgeDocumentCitationInput { - scope: ResourceScope - knowledgeBaseId: string - documentId: string - sourceUrl: string | null - baseUrl: string -} - -/** Uses the original source when safe, otherwise the authorized Sim document page. */ -export function createKnowledgeDocumentCitation(input: KnowledgeDocumentCitationInput) { - if (!isKnowledgeSourceUrl(input.baseUrl)) throw new Error('Invalid citation base URL') - const ownerPath = - input.scope.kind === 'organization' - ? `/o/${encodeURIComponent(input.scope.organizationId)}` - : `/workspace/${encodeURIComponent(input.scope.workspaceId)}` - const documentPath = `${ownerPath}/knowledge/${encodeURIComponent(input.knowledgeBaseId)}/${encodeURIComponent(input.documentId)}` - const sourceUrl = input.sourceUrl?.trim() - return { - citationId: `document:${input.documentId}`, - citationUrl: - sourceUrl && isKnowledgeSourceUrl(sourceUrl) - ? sourceUrl - : new URL(documentPath, input.baseUrl).href, - } -} diff --git a/apps/sim/lib/knowledge/search/source-url.test.ts b/apps/sim/lib/knowledge/search/source-url.test.ts new file mode 100644 index 00000000000..c8560bdfa05 --- /dev/null +++ b/apps/sim/lib/knowledge/search/source-url.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest' +import { isKnowledgeSourceUrl } from '@/lib/knowledge/search/source-url' + +describe('knowledge source URLs', () => { + it.each([ + '', + 'file:///tmp/app', + 'javascript:alert(1)', + 'data:text/html,secret', + 'https://user:secret@source.test/doc', + 'https://source.test/with\nnewline', + 'https://source.test\\@evil.test/doc', + '/relative/path', + 'https:source.test/doc', + '//source.test/doc', + ' https://source.test/doc', + ])('rejects unsafe provider link %s', (sourceUrl) => { + expect(isKnowledgeSourceUrl(sourceUrl)).toBe(false) + }) + + it('accepts a provider link with its query and fragment intact', () => { + expect( + isKnowledgeSourceUrl('https://docs.google.com/document/d/abc/edit?tab=t.0#heading') + ).toBe(true) + }) +}) diff --git a/apps/sim/lib/mothership/generated/tool-catalog-v1.ts b/apps/sim/lib/mothership/generated/tool-catalog-v1.ts index eb42cd17d8d..bf94e2d5695 100644 --- a/apps/sim/lib/mothership/generated/tool-catalog-v1.ts +++ b/apps/sim/lib/mothership/generated/tool-catalog-v1.ts @@ -7893,7 +7893,6 @@ export const SearchSources: ToolCatalogEntry = { type: 'string', maxLength: 200, }, - mine: { description: 'Only used for action: list. Omit for other actions.', type: 'boolean' }, connectorId: { description: 'Required for action: get. Only used for action: get. Omit for other actions.', type: 'string', diff --git a/apps/sim/lib/mothership/generated/tool-schemas-v1.ts b/apps/sim/lib/mothership/generated/tool-schemas-v1.ts index ac3bacd5b5b..07020b2a423 100644 --- a/apps/sim/lib/mothership/generated/tool-schemas-v1.ts +++ b/apps/sim/lib/mothership/generated/tool-schemas-v1.ts @@ -8064,10 +8064,6 @@ export const TOOL_RUNTIME_SCHEMAS: Record = { type: 'string', maxLength: 200, }, - mine: { - description: 'Only used for action: list. Omit for other actions.', - type: 'boolean', - }, connectorId: { description: 'Required for action: get. Only used for action: get. Omit for other actions.', diff --git a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts index 1b78b765008..a926b851cdc 100644 --- a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts +++ b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.test.ts @@ -27,6 +27,8 @@ vi.mock('@/lib/sim-search/live/application', () => ({ })) import { knowledgeOperations } from '@/lib/knowledge/application/operations' +import { collectRetrievalCitationEvidence } from '@/lib/mothership/chat/citation-evidence' +import { compactRetrievalCitations } from '@/lib/mothership/chat/retrieval-citations' import { readDocumentServerTool, searchWorkspaceServerTool, @@ -81,6 +83,78 @@ describe('Assistant retrieval tools', () => { next: null, }) }) + for (const tool of [searchWorkspaceServerTool, readDocumentServerTool]) { + describe(`${tool.name} live citations`, () => { + it.each([ + { + name: 'the source link before its container', + sourceUrl: 'https://source.test/document', + sourceContainerUrl: 'https://source.test/container', + citationUrl: 'https://source.test/document', + }, + { + name: 'the container when the source has no link', + sourceUrl: null, + sourceContainerUrl: 'https://source.test/container', + citationUrl: 'https://source.test/container', + }, + { + name: 'no citation link when neither destination is available', + sourceUrl: null, + sourceContainerUrl: undefined, + citationUrl: null, + }, + ])('preserves $name through saved citation evidence', async (links) => { + const document = { + knowledgeBaseId: '', + documentId: 'live:opaque-reference', + documentName: 'Live document', + sourceUrl: links.sourceUrl, + sourceContainerUrl: links.sourceContainerUrl, + connectorType: 'slack', + content: 'Retrieved evidence', + } + const searching = tool.name === 'search_workspace' + if (searching) { + mocks.search.mockResolvedValueOnce({ + retrieval: { status: 'complete', timedOutLegs: [] }, + results: [document], + }) + } else { + mocks.read.mockResolvedValueOnce({ + ...document, + chunks: [{ chunkIndex: 0, content: document.content }], + hasMore: false, + next: null, + }) + } + const result = await tool.execute( + searching ? { query: 'evidence' } : { documentId: document.documentId }, + { ...context, assistantSearch: undefined } + ) + const expected = { documentId: document.documentId, citationUrl: links.citationUrl } + expect(result).toMatchObject({ + success: true, + data: searching ? { results: [expected] } : expected, + }) + const evidence = collectRetrievalCitationEvidence([ + { + toolCall: { + name: tool.name, + status: 'success', + result: { + success: true, + output: compactRetrievalCitations(tool.name, result), + }, + }, + }, + ]) + expect([...evidence.values()].map((source) => source.url)).toEqual( + links.citationUrl ? [links.citationUrl] : [] + ) + }) + }) + } it.each([{ startDate: '2026-09-01T00:00:00Z' }, { sortBy: 'newest' }, { sortBy: 'oldest' }])( 'returns actionable validation for empty Notion native queries with %j', async (bound) => { diff --git a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts index ce9957cfec1..098f5e0191c 100644 --- a/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts +++ b/apps/sim/lib/mothership/tools/server/knowledge/workspace-search.ts @@ -5,10 +5,9 @@ import { searchWorkspaceInputSchema, } from '@/lib/api/contracts/mothership-assistant-tools' import { getValidationErrorMessage } from '@/lib/api/server/validation' -import { getBaseUrl } from '@/lib/core/utils/urls' import { EmbeddingConfigurationError } from '@/lib/embeddings/configuration-error' import { SearchDeadlineError } from '@/lib/knowledge/search/budget' -import { createKnowledgeDocumentCitation, liveCitationId } from '@/lib/knowledge/search/citation' +import { liveCitationId } from '@/lib/knowledge/search/citation' import { withSearchDiagnostics } from '@/lib/knowledge/search/diagnostics' import { intersectWorkspaceSearchFilters } from '@/lib/knowledge/search/filters' import { @@ -25,7 +24,7 @@ import { projectResolvedSecretModelContent } from '@/executor/utils/resolved-sec const logger = createLogger('WorkspaceSearchTool') const CITATION_INSTRUCTION = - 'Cite the evidence you use as {"id":""}. Use only IDs returned by these tools.' + + 'Cite the evidence you use as {"id":""}. Use only IDs returned by these tools with a non-null citationUrl.' + ' When referring to a Slack conversation, link the returned sourceContainerName to its sourceContainerUrl when available.' export const searchWorkspaceServerTool: BaseServerTool = { @@ -93,14 +92,8 @@ export const searchWorkspaceServerTool: BaseServerTool = { results: data.results.map((item) => ({ ...item, siteName: connectorDisplayName(item.connectorType ?? ''), - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: '', - documentId: item.documentId, - sourceUrl: item.sourceUrl, - baseUrl: getBaseUrl(), - }), citationId: liveCitationId(item.documentId), + citationUrl: item.sourceUrl ?? item.sourceContainerUrl ?? null, })), }, } @@ -166,14 +159,8 @@ export const readDocumentServerTool: BaseServerTool = { message: CITATION_INSTRUCTION, data: { ...data, - ...createKnowledgeDocumentCitation({ - scope, - knowledgeBaseId: '', - documentId: data.documentId, - sourceUrl: data.sourceUrl, - baseUrl: getBaseUrl(), - }), citationId: liveCitationId(data.documentId), + citationUrl: data.sourceUrl ?? data.sourceContainerUrl ?? null, }, } } catch (error) { diff --git a/packages/db/schema.ts b/packages/db/schema.ts index 0a47dffac6e..523c8f7e970 100644 --- a/packages/db/schema.ts +++ b/packages/db/schema.ts @@ -3820,23 +3820,19 @@ export const embeddingSearch = pgTable( knowledgeBaseId: text('knowledge_base_id').notNull(), documentId: text('document_id').notNull(), enabled: boolean('enabled').notNull(), - /** - * The connector whose documents this chunk belongs to, copied from the document so a vector - * index can cover one source. A member reads a source whole or barely at all, so searching - * each readable source in its own index finds their nearest chunks; one index over every - * source spends its scan budget on chunks the graph reached but the member cannot read. - * NULL for uploads. - */ - // contract-pending(after the indexed-search retirement release and source/ACL projection writers have drained): drop connector_id — regular KB retrieval checks the parent document. + /** contract-pending(after #8528 is fully deployed and source/ACL projection writers have drained): drop connector_id — regular KB retrieval checks the parent document. */ connectorId: text('connector_id'), - /** The document's ACL, mirrored by trigger, so a walk can test readability on the row it visits. */ - // contract-pending(after the indexed-search retirement release and source/ACL projection writers have drained): drop acl — regular KB retrieval retains document-level access checks. + /** contract-pending(after #8528 is fully deployed and source/ACL projection writers have drained): drop acl — regular KB retrieval retains document-level access checks. */ acl: text('acl').array(), - // contract-pending(after vector writers stop computing binary projections and embedding_search_width_check is replaced): drop binary and all binary_* columns — their ANN indexes were dropped in 0372 and no reader uses them. + /** contract-pending(after vector writers stop computing binary projections and embedding_search_width_check is replaced): drop binary and all binary_* columns — their ANN indexes were dropped in 0372 and no reader uses them. */ binary: bit('binary', { dimensions: 1536 }), + /** @deprecated Remove with the binary projection contract above. */ binary384: bit('binary_384', { dimensions: 384 }), + /** @deprecated Remove with the binary projection contract above. */ binary768: bit('binary_768', { dimensions: 768 }), + /** @deprecated Remove with the binary projection contract above. */ binary1024: bit('binary_1024', { dimensions: 1024 }), + /** @deprecated Remove with the binary projection contract above. */ binary3072: bit('binary_3072', { dimensions: 3072 }), vector: halfvec('vector', { dimensions: 1536 }), vector384: halfvec('vector_384', { dimensions: 384 }), From 39067d0c7b0c6f89c6ce6892869323c8151c4f70 Mon Sep 17 00:00:00 2001 From: Theodore Li Date: Thu, 1 Oct 2026 15:04:30 -0700 Subject: [PATCH 18/31] feat(files): render diff fences and theme mermaid diagrams (#8543) * improvement(files): theme mermaid diagrams with app tokens * feat(files): render diff fences with source-checked state * improvement(files): frame diffs as cards with tinted change rows * improvement(files): render prose diffs as source excerpt cards and compare two documents * improvement(files): put the open link beside the excerpt title * improvement(files): render inline bold and code in prose diffs * improvement(files): render edits as line diffs with softer change colors * improvement(files): draw mermaid flowcharts like workflow blocks and edges * fix(files): keep mermaid labels inside their nodes and fork edges cleanly * improvement(files): drop diff source checks and split view, size tables to content * improvement(files): keep the diff parser out of the editor bundle * fix(files): parse multi-file and copy git diffs, cap word diffs, readable pie labels * fix(files): title renames by their new path and expand collapsed lines one run at a time --- .../file-viewer/mermaid-diagram.tsx | 19 +- .../components/file-viewer/mermaid-theme.ts | 88 ++++++ .../rich-markdown-editor/code-block.tsx | 21 +- .../rich-markdown-editor.css | 21 +- .../components/dashboards/dashboard-panel.tsx | 4 +- apps/sim/components/diff/diff-embed.tsx | 284 ++++++++++++++++++ apps/sim/components/diff/diff-view.tsx | 203 +++++++++++++ apps/sim/lib/diff/embed-language.ts | 2 + apps/sim/lib/diff/unified.test.ts | 96 ++++++ apps/sim/lib/diff/unified.ts | 143 +++++++++ ...check-tool-registry-boundary.baseline.json | 16 +- 11 files changed, 877 insertions(+), 20 deletions(-) create mode 100644 apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-theme.ts create mode 100644 apps/sim/components/diff/diff-embed.tsx create mode 100644 apps/sim/components/diff/diff-view.tsx create mode 100644 apps/sim/lib/diff/embed-language.ts create mode 100644 apps/sim/lib/diff/unified.test.ts create mode 100644 apps/sim/lib/diff/unified.ts diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-diagram.tsx b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-diagram.tsx index 10019822f72..9e8c3c634c4 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-diagram.tsx +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-diagram.tsx @@ -4,6 +4,10 @@ import { memo, useEffect, useState } from 'react' import { toError } from '@sim/utils/errors' import { generateShortId } from '@sim/utils/id' import { useTheme } from 'next-themes' +import { + mermaidWorkflowCss, + readMermaidThemeVariables, +} from '@/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-theme' import { PreviewLoadingFrame } from './preview-shared' import { ZoomablePreview } from './zoomable-preview' @@ -124,10 +128,23 @@ export const MermaidDiagram = memo(function MermaidDiagram({ const { default: mermaid } = await import('mermaid') if (cancelled) return + const themeVariables = readMermaidThemeVariables( + document.documentElement, + mermaidTheme === 'dark' + ) mermaid.initialize({ startOnLoad: false, securityLevel: 'strict', - theme: mermaidTheme, + theme: 'base', + themeCSS: mermaidWorkflowCss(String(themeVariables.fontFamily)), + flowchart: { + curve: 'basis', + padding: 12, + nodeSpacing: 36, + rankSpacing: 56, + wrappingWidth: 180, + }, + themeVariables, }) mermaid.setParseErrorHandler?.(() => undefined) diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-theme.ts b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-theme.ts new file mode 100644 index 00000000000..e5f43e0802b --- /dev/null +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/mermaid-theme.ts @@ -0,0 +1,88 @@ +/** + * Mermaid `base` theme variables drawn from the app's tokens, so diagrams match the editor and + * charts instead of Mermaid's stock palette. Mermaid derives shades from these with its own colour + * library, so every token read here must resolve to a plain hex or rgb value. + */ +export function readMermaidThemeVariables( + element: Element, + darkMode: boolean +): Record { + const styles = getComputedStyle(element) + const token = (name: string) => { + const value = styles.getPropertyValue(name).trim() + if (!value) throw new Error(`Missing diagram theme token ${name}`) + return value + } + const background = token('--bg') + const node = token('--surface-2') + const muted = token('--surface-4') + const subtle = token('--surface-5') + const border = token('--border') + const text = token('--text-primary') + const body = token('--text-body') + const line = token('--text-icon') + return { + darkMode, + fontFamily: getComputedStyle(document.body).fontFamily, + fontSize: '14px', + background, + primaryColor: node, + primaryTextColor: text, + primaryBorderColor: border, + secondaryColor: muted, + secondaryTextColor: body, + secondaryBorderColor: border, + tertiaryColor: subtle, + tertiaryTextColor: body, + tertiaryBorderColor: border, + mainBkg: node, + nodeBorder: border, + textColor: body, + lineColor: token('--workflow-edge'), + clusterBkg: token('--surface-3'), + clusterBorder: border, + edgeLabelBackground: background, + noteBkgColor: muted, + noteTextColor: body, + noteBorderColor: border, + actorBkg: node, + actorBorder: border, + actorTextColor: text, + actorLineColor: line, + signalColor: line, + signalTextColor: body, + labelBoxBkgColor: subtle, + labelBoxBorderColor: border, + labelTextColor: body, + loopTextColor: body, + sequenceNumberColor: background, + pie1: body, + pie2: line, + pie3: token('--text-subtle'), + pie4: token('--text-muted'), + pieStrokeColor: background, + pieTitleTextColor: text, + pieSectionTextColor: background, + pieLegendTextColor: body, + } +} + +/** + * Flowcharts drawn like workflow canvas blocks and edges: rounded cards with a 1.5px outline, + * 1.5px edges without arrowheads, and subgraphs as rounded subflow containers. Labels are pinned + * to the font Mermaid measured them with; otherwise they inherit the surrounding document's font + * and line height and overflow their boxes. Edge rules are scoped to flowchart classes so sequence + * and other diagrams keep their arrows. + */ +export function mermaidWorkflowCss(fontFamily: string): string { + return ` + .label, .nodeLabel, .edgeLabel, .label p, .nodeLabel p, .edgeLabel p { + font-family: ${fontFamily}; font-size: 14px; line-height: 1.5; letter-spacing: normal; + } + .edgeLabel, .edgeLabel p { font-size: 12px; } + .node rect, .node polygon, .node circle, .node path { stroke-width: 1.5px; } + .node rect { rx: 10px; ry: 10px; } + .cluster rect { rx: 14px; ry: 14px; stroke-width: 1.5px; } + .flowchart-link { stroke-width: 1.5px; marker-end: none !important; } +` +} diff --git a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx index 227331565a1..80b86ad0d98 100644 --- a/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx +++ b/apps/sim/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/code-block.tsx @@ -12,6 +12,7 @@ import { Check, ChevronDown, Code, Duplicate, Eye, Wrap } from '@sim/emcn/icons' import type { ReactNodeViewProps } from '@tiptap/react' import { NodeViewContent, NodeViewWrapper, ReactNodeViewRenderer } from '@tiptap/react' import { DASHBOARD_EMBED_LANGUAGE } from '@/lib/dashboards/embed-language' +import { DIFF_EMBED_LANGUAGE } from '@/lib/diff/embed-language' import { MarkdownStreamingContext } from '@/app/workspace/[workspaceId]/files/components/file-viewer/rich-markdown-editor/markdown-streaming-context' import { looksLikeMermaid, MermaidDiagram } from '../mermaid-diagram' import { MarkdownCodeBlock } from './code-block-schema' @@ -22,6 +23,9 @@ import { useEditorEditable } from './use-editor-editable' const DashboardEmbed = lazy(() => import('@/components/dashboards/dashboard-embed').then((m) => ({ default: m.DashboardEmbed })) ) +const DiffEmbed = lazy(() => + import('@/components/diff/diff-embed').then((m) => ({ default: m.DiffEmbed })) +) const PLAIN = 'plain' const MERMAID = 'mermaid' @@ -74,7 +78,8 @@ function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeView const text = node.textContent const isMermaid = explicitLanguage === MERMAID || (!explicitLanguage && looksLikeMermaid(text)) const isDashboard = explicitLanguage === DASHBOARD_EMBED_LANGUAGE - const isRendered = isMermaid || isDashboard + const isDiff = explicitLanguage === DIFF_EMBED_LANGUAGE + const isRendered = isMermaid || isDashboard || isDiff // Editable Mermaid shows source while the caret is focused inside the block and re-renders the // diagram on blur (the Linear/GitHub model). The Show source / Show diagram control drives this by @@ -148,7 +153,13 @@ function CodeBlockView({ node, updateAttributes, editor, getPos }: ReactNodeView