From fba0f1393b209f9e9d519d8f779943acb5a79f70 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Fri, 14 Aug 2026 12:04:46 -0700 Subject: [PATCH 1/5] feat(connectors): add 9 knowledge base connectors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Box, Zoho Desk, PagerDuty, Trello, Microsoft Excel, Google Slides, Google Vault, Mintlify, and SFTP. Selected by intersecting the published connector catalogs of Glean, Onyx, Dust, Vectara, Writer, Guru, Elastic, Microsoft 365 Copilot, Notion AI, Unstructured, and Airbyte against services that already ship a Sim block, so OAuth providers, credentials, and icons are reused. Box was the largest gap, appearing in 7-8 of ~10 catalogs. Every connector was validated against live provider documentation twice, the second pass treating the first pass's conclusions as unproven. Notable correctness work that came out of that: Listing truncation. The sync engine hard-deletes documents past a cap that is not flagged with `listingCapped`, and five connectors had a path there — an empty Mintlify discovery, Zoho Desk's exact-multiple default caps, Trello's archived lists and 1000-card ceiling, a Google Vault cursor bailout, and a PagerDuty stalled page. The engine also gained a backstop: an empty or collapsed listing blocks deletion reconciliation until the same observation repeats on a consecutive sync, reconstructed from existing sync-log counters so no migration is needed. API alignment. `desk.zoho.ca` does not resolve (Canada is `desk.zohocloud.ca`, and Singapore and UAE were missing); `modifiedTime` is absent from Zoho's ticket list projection, so every ticket re-embedded on every sync; Trello's `dateLastActivity` is documented to miss some edits; PagerDuty's 10,000-record ceiling bounds `offset + limit`, not offset; Excel indexed dates as raw serial numbers while Google Sheets renders them; Google Vault truncated at roughly 249 matters. Security. SFTP followed symlinks in `getDocument` and composed unchecked server-supplied filenames into paths; it now also supports optional host-key fingerprint verification, which runs during key exchange before any password is sent. Trello interpolated user-supplied board ids into URL paths. Google Vault is narrowed to `ediscovery.readonly`. `getDataverseBaseUrl` accepted any host while attaching a bearer token, and is pinned to Microsoft's Dataverse domains — pre-existing shipped code, fixed here. Also adds `ConnectorAuthConfig.optional` so a public source can be configured without inventing an API key, and teaches the scope check that a granted read-write scope satisfies a required `.readonly` sibling. Microsoft Dataverse was built and then removed: its OAuth cannot complete consent. Dataverse requires a per-environment resource URI, the provider declares a static `https://dynamics.microsoft.com/user_impersonation` that is not an Entra Application ID URI, and the environment URL is only collected after the credential exists. That predates this change and also affects the 12 shipped Dataverse tools. --- .../docs/en/knowledgebase/connectors.mdx | 20 +- apps/sim/app/api/tools/sftp/utils.ts | 105 ++- .../add-connector-modal.tsx | 12 +- apps/sim/connectors/box/box.ts | 632 ++++++++++++++ apps/sim/connectors/box/index.ts | 1 + apps/sim/connectors/box/meta.ts | 42 + .../connectors/google-slides/google-slides.ts | 495 +++++++++++ apps/sim/connectors/google-slides/index.ts | 1 + apps/sim/connectors/google-slides/meta.ts | 68 ++ .../connectors/google-vault/google-vault.ts | 722 ++++++++++++++++ apps/sim/connectors/google-vault/index.ts | 1 + apps/sim/connectors/google-vault/meta.ts | 73 ++ apps/sim/connectors/microsoft-excel/index.ts | 1 + apps/sim/connectors/microsoft-excel/meta.ts | 67 ++ .../microsoft-excel/microsoft-excel.ts | 669 +++++++++++++++ apps/sim/connectors/mintlify/index.ts | 1 + apps/sim/connectors/mintlify/meta.ts | 63 ++ apps/sim/connectors/mintlify/mintlify.ts | 620 ++++++++++++++ apps/sim/connectors/pagerduty/index.ts | 1 + apps/sim/connectors/pagerduty/meta.ts | 95 +++ apps/sim/connectors/pagerduty/pagerduty.ts | 711 ++++++++++++++++ apps/sim/connectors/registry.server.ts | 18 + apps/sim/connectors/registry.ts | 18 + apps/sim/connectors/sftp/index.ts | 1 + apps/sim/connectors/sftp/meta.ts | 107 +++ apps/sim/connectors/sftp/sftp.ts | 709 ++++++++++++++++ apps/sim/connectors/trello/index.ts | 1 + apps/sim/connectors/trello/meta.ts | 71 ++ apps/sim/connectors/trello/trello.ts | 797 ++++++++++++++++++ apps/sim/connectors/types.ts | 13 +- apps/sim/connectors/zoho-desk/index.ts | 1 + apps/sim/connectors/zoho-desk/meta.ts | 180 ++++ apps/sim/connectors/zoho-desk/zoho-desk.ts | 746 ++++++++++++++++ .../tools/server/knowledge/knowledge-base.ts | 8 - .../lib/knowledge/application/connectors.ts | 12 +- .../knowledge/connectors/sync-engine.test.ts | 92 +- .../lib/knowledge/connectors/sync-engine.ts | 189 ++++- .../lib/knowledge/orchestration/connectors.ts | 4 +- apps/sim/lib/oauth/oauth.ts | 12 + apps/sim/lib/oauth/utils.ts | 19 +- apps/sim/tools/microsoft_dataverse/utils.ts | 49 +- 41 files changed, 7416 insertions(+), 31 deletions(-) create mode 100644 apps/sim/connectors/box/box.ts create mode 100644 apps/sim/connectors/box/index.ts create mode 100644 apps/sim/connectors/box/meta.ts create mode 100644 apps/sim/connectors/google-slides/google-slides.ts create mode 100644 apps/sim/connectors/google-slides/index.ts create mode 100644 apps/sim/connectors/google-slides/meta.ts create mode 100644 apps/sim/connectors/google-vault/google-vault.ts create mode 100644 apps/sim/connectors/google-vault/index.ts create mode 100644 apps/sim/connectors/google-vault/meta.ts create mode 100644 apps/sim/connectors/microsoft-excel/index.ts create mode 100644 apps/sim/connectors/microsoft-excel/meta.ts create mode 100644 apps/sim/connectors/microsoft-excel/microsoft-excel.ts create mode 100644 apps/sim/connectors/mintlify/index.ts create mode 100644 apps/sim/connectors/mintlify/meta.ts create mode 100644 apps/sim/connectors/mintlify/mintlify.ts create mode 100644 apps/sim/connectors/pagerduty/index.ts create mode 100644 apps/sim/connectors/pagerduty/meta.ts create mode 100644 apps/sim/connectors/pagerduty/pagerduty.ts create mode 100644 apps/sim/connectors/sftp/index.ts create mode 100644 apps/sim/connectors/sftp/meta.ts create mode 100644 apps/sim/connectors/sftp/sftp.ts create mode 100644 apps/sim/connectors/trello/index.ts create mode 100644 apps/sim/connectors/trello/meta.ts create mode 100644 apps/sim/connectors/trello/trello.ts create mode 100644 apps/sim/connectors/zoho-desk/index.ts create mode 100644 apps/sim/connectors/zoho-desk/meta.ts create mode 100644 apps/sim/connectors/zoho-desk/zoho-desk.ts diff --git a/apps/docs/content/docs/en/knowledgebase/connectors.mdx b/apps/docs/content/docs/en/knowledgebase/connectors.mdx index 4b69acdc3d1..76e0e23a8ef 100644 --- a/apps/docs/content/docs/en/knowledgebase/connectors.mdx +++ b/apps/docs/content/docs/en/knowledgebase/connectors.mdx @@ -14,23 +14,24 @@ Connectors continuously sync documents from external services into your knowledg Connect Source picker showing a searchable list of available connectors including Airtable, Asana, Confluence, Discord, Dropbox, Evernote, Fireflies, GitHub, and Gmail -Sim ships with 49 built-in connectors: +Sim ships with 61 built-in connectors: | Category | Connectors | |----------|-----------| -| **Productivity** | Notion, Confluence, Asana, Linear, Jira, Jira Service Management, Monday, Google Calendar, Google Sheets, Google Forms, Typeform | -| **Cloud Storage** | Google Drive, Dropbox, OneDrive, SharePoint, Amazon S3 | -| **Documents** | Google Docs, WordPress, Webflow, DocuSign | +| **Productivity** | Notion, Confluence, Asana, Linear, Jira, Jira Service Management, Monday, Trello, ClickUp, Google Calendar, Google Sheets, Google Forms, Microsoft Excel, Typeform | +| **Cloud Storage** | Google Drive, Dropbox, OneDrive, SharePoint, Box, Amazon S3, SFTP | +| **Documents** | Google Docs, Google Slides, Mintlify, WordPress, Webflow, DocuSign | | **Development** | GitHub, GitLab, Azure DevOps, Sentry | -| **Communication** | Slack, Discord, Microsoft Teams, Reddit, YouTube | +| **Communication** | Slack, Discord, Microsoft Teams, Reddit, X, YouTube | | **Email** | Gmail, Outlook | | **CRM** | HubSpot, Salesforce | -| **Support** | Intercom, ServiceNow, Zendesk | -| **Incident Management** | incident.io, Rootly | +| **Support** | Intercom, ServiceNow, Zendesk, Zoho Desk | +| **Incident Management** | incident.io, Rootly, PagerDuty | | **Data** | Airtable | | **Note-taking** | Evernote, Obsidian | -| **Meetings** | Zoom, Gong, Grain, Granola, Fathom, Fireflies | +| **Meetings** | Zoom, Google Meet, Gong, Grain, Granola, Fathom, Fireflies | | **Recruiting** | Greenhouse, Ashby | +| **Compliance** | Google Vault | ## Adding a Connector @@ -55,6 +56,9 @@ Other connectors use **API keys** or **personal access tokens** instead. The set | **YouTube** | YouTube Data API key from the Google Cloud Console | | **Amazon S3** | Secret Access Key (the Access Key ID, region, and bucket are entered as config fields) | | **Sentry** | Auth token with `project:read` and `event:read` scopes | +| **PagerDuty** | REST API key from Integrations → API Access Keys | +| **SFTP** | Password or unencrypted private key (host, port, username, and root path are entered as config fields) | +| **Mintlify** | API key — optional for public documentation sites, which sync from `llms.txt` | If you rotate an API key in the external service, update it in Sim as well — OAuth tokens refresh automatically, but API keys do not. diff --git a/apps/sim/app/api/tools/sftp/utils.ts b/apps/sim/app/api/tools/sftp/utils.ts index ea81b52793c..17ad9c57623 100644 --- a/apps/sim/app/api/tools/sftp/utils.ts +++ b/apps/sim/app/api/tools/sftp/utils.ts @@ -1,8 +1,13 @@ +import { createHash } from 'node:crypto' +import { createLogger } from '@sim/logger' +import { safeCompare } from '@sim/security/compare' import { toError } from '@sim/utils/errors' import { type Attributes, Client, type ConnectConfig, type SFTPWrapper } from 'ssh2' import { validateDatabaseHost } from '@/lib/core/security/input-validation.server' import { readNodeStreamToBufferWithLimit } from '@/lib/core/utils/stream-limits' +const logger = createLogger('SftpUtils') + const S_IFMT = 0o170000 const S_IFDIR = 0o040000 const S_IFREG = 0o100000 @@ -15,9 +20,44 @@ export interface SftpConnectionConfig { password?: string | null privateKey?: string | null passphrase?: string | null + /** + * Idle socket timeout in ms, forwarded to ssh2's `sock.setTimeout`. Left + * unset the socket has no idle timeout at all (ssh2 defaults it to `0`). + */ timeout?: number keepaliveInterval?: number readyTimeout?: number + /** + * Expected SHA-256 host key fingerprint in the format `ssh-keyscan` and + * OpenSSH print (`SHA256:`). The `SHA256:` prefix and any base64 + * padding are optional. When set, a server presenting a different host key is + * rejected before authentication runs. When omitted, the host is not + * verified — ssh2's default behavior. + */ + hostFingerprint?: string | null +} + +/** + * Normalizes a user-supplied SHA-256 fingerprint for comparison: trims, drops + * an optional `SHA256:` prefix, and strips base64 `=` padding, which OpenSSH + * omits but copy/paste sources sometimes include. + */ +function normalizeSha256Fingerprint(value: string): string { + return value + .trim() + .replace(/^sha256:/i, '') + .replace(/=+$/, '') + .trim() +} + +/** + * Computes the OpenSSH SHA-256 fingerprint of a host key. ssh2 hands the + * verifier the raw SSH wire-format public key blob — the same bytes OpenSSH + * base64-encodes into `known_hosts` — so hashing it directly reproduces the + * unpadded base64 digest that `ssh-keyscan | ssh-keygen -lf -` prints. + */ +function computeHostKeyFingerprint(hostKey: Buffer): string { + return createHash('sha256').update(hostKey).digest('base64').replace(/=+$/, '') } /** @@ -93,6 +133,11 @@ function formatSftpError(err: Error, config: { host: string; port: number }): Er /** * Creates an SSH connection for SFTP using the provided configuration. * Uses ssh2 library defaults which align with OpenSSH standards. + * + * When `hostFingerprint` is supplied the server's host key is pinned to it and + * a mismatch aborts the handshake before any credential is sent. Without it + * ssh2 accepts whatever host key answers, which is the pre-existing behavior + * kept for backward compatibility. */ export async function createSftpConnection(config: SftpConnectionConfig): Promise { const host = config.host @@ -132,6 +177,50 @@ export async function createSftpConnection(config: SftpConnectionConfig): Promis if (config.keepaliveInterval !== undefined) { connectConfig.keepaliveInterval = config.keepaliveInterval } + if (config.timeout !== undefined) { + connectConfig.timeout = config.timeout + } + + const suppliedFingerprint = config.hostFingerprint?.trim() + const expectedFingerprint = suppliedFingerprint + ? normalizeSha256Fingerprint(suppliedFingerprint) + : undefined + + /** + * Fail closed rather than silently skipping verification. A value that is + * non-blank but normalizes away (`SHA256:`, `=`) would otherwise leave no + * `hostVerifier` installed, trusting whatever host answers — the opposite + * of what supplying a fingerprint asks for. + */ + if (suppliedFingerprint && !expectedFingerprint) { + throw new Error( + 'Host key fingerprint is not a valid SHA-256 fingerprint. Expected the base64 form printed by `ssh-keyscan | ssh-keygen -lf -`.' + ) + } + + /** + * Set when the pinned fingerprint does not match. ssh2 reports the + * rejection through a generic `'error'` event, so the precise cause is + * carried out of the verifier rather than re-derived from that message. + */ + let hostKeyRejection: Error | undefined + + if (expectedFingerprint) { + connectConfig.hostVerifier = (hostKey: Buffer): boolean => { + const actualFingerprint = computeHostKeyFingerprint(hostKey) + if (safeCompare(actualFingerprint, expectedFingerprint)) { + return true + } + hostKeyRejection = new Error( + `Host key verification failed for ${host}:${port}. ` + + `Expected SHA256:${expectedFingerprint} but the server presented SHA256:${actualFingerprint}. ` + + `Either the server's host key changed, or the connection was intercepted. ` + + `Re-run "ssh-keyscan -t rsa,ecdsa,ed25519 ${host}" to confirm the current key before updating the fingerprint.` + ) + logger.warn('SFTP host key fingerprint mismatch', { host, port }) + return false + } + } if (hasPrivateKey) { connectConfig.privateKey = config.privateKey! @@ -147,7 +236,21 @@ export async function createSftpConnection(config: SftpConnectionConfig): Promis }) client.on('error', (err) => { - reject(formatSftpError(err, { host, port })) + reject(hostKeyRejection ?? formatSftpError(err, { host, port })) + }) + + /** + * ssh2 only re-emits the socket's `'timeout'` event; it never destroys the + * socket, so without this the connection would sit open forever after the + * idle timeout elapsed. + */ + client.on('timeout', () => { + client.destroy() + reject( + new Error( + `Connection to ${host}:${port} timed out after ${config.timeout}ms of inactivity.` + ) + ) }) try { diff --git a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx index 705746d348f..a547065ff13 100644 --- a/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx +++ b/apps/sim/app/workspace/[workspaceId]/knowledge/[id]/components/add-connector-modal/add-connector-modal.tsx @@ -82,6 +82,9 @@ export function AddConnectorModal({ const connectorConfig = selectedType ? CONNECTOR_META_REGISTRY[selectedType] : null const isApiKeyMode = connectorConfig?.auth.mode === 'apiKey' + /** True when the connector declares its key optional (public sources need none). */ + const isApiKeyOptional = + connectorConfig?.auth.mode === 'apiKey' && connectorConfig.auth.optional === true const connectorProviderId = useMemo( () => connectorConfig && connectorConfig.auth.mode === 'oauth' @@ -160,7 +163,7 @@ export function AddConnectorModal({ const canSubmit = useMemo(() => { if (!connectorConfig) return false if (isApiKeyMode) { - if (!apiKeyValue.trim()) return false + if (!isApiKeyOptional && !apiKeyValue.trim()) return false } else { if (!effectiveCredentialId) return false } @@ -174,6 +177,7 @@ export function AddConnectorModal({ }, [ connectorConfig, isApiKeyMode, + isApiKeyOptional, apiKeyValue, effectiveCredentialId, isFieldVisible, @@ -207,7 +211,11 @@ export function AddConnectorModal({ { knowledgeBaseId, connectorType: selectedType, - ...(isApiKeyMode ? { apiKey: apiKeyValue } : { credentialId: effectiveCredentialId! }), + ...(isApiKeyMode + ? apiKeyValue.trim() + ? { apiKey: apiKeyValue } + : {} + : { credentialId: effectiveCredentialId! }), sourceConfig: finalSourceConfig, syncIntervalMinutes: syncInterval, }, diff --git a/apps/sim/connectors/box/box.ts b/apps/sim/connectors/box/box.ts new file mode 100644 index 00000000000..2d82a9a6f2b --- /dev/null +++ b/apps/sim/connectors/box/box.ts @@ -0,0 +1,632 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import { sleep } from '@sim/utils/helpers' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { boxConnectorMeta } from '@/connectors/box/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { + CONNECTOR_MAX_FILE_BYTES, + ConnectorFileTooLargeError, + htmlToPlainText, + isSkippedDocument, + markSkipped, + parseTagDate, + readBodyWithLimit, + sizeLimitSkipReason, + stubOrSkipBySize, + takeIndexableWithinCap, +} from '@/connectors/utils' + +const logger = createLogger('BoxConnector') + +const BOX_API_BASE = 'https://api.box.com/2.0' +const BOX_ROOT_FOLDER_ID = '0' + +/** Maximum items per `GET /folders/:id/items` page (Box hard limit is 1000). */ +const FOLDER_ITEMS_PAGE_SIZE = 1000 + +/** + * Fields requested on every listed item. Supplying `fields` drops Box's standard + * field set, so every field the stub and tag mapping need must be named here. + */ +const ITEM_FIELDS = 'type,id,name,etag,size,created_at,modified_at,extension,path_collection' + +const FILE_FIELDS = `${ITEM_FIELDS},item_status,trashed_at` + +/** + * Folder pages drained per `listDocuments` call. Box has no recursive listing, so + * a naive one-folder-per-call walk would spend the sync engine's page budget + * (`MAX_PAGES`) on folder depth rather than on documents. + */ +const FOLDER_PAGES_PER_CALL = 10 + +/** Soft ceiling on stubs accumulated in one call, so a wide tree still yields early. */ +const MAX_FILES_PER_CALL = 2000 + +const MAX_FILE_SIZE = CONNECTOR_MAX_FILE_BYTES + +/** + * Extensions read straight from `GET /files/:id/content` as UTF-8. These need no + * Box-side conversion, so they skip the representation round trip entirely. + */ +const PLAIN_TEXT_EXTENSIONS = new Set([ + 'as', + 'as3', + 'asm', + 'bat', + 'c', + 'cc', + 'cmake', + 'cpp', + 'cs', + 'css', + 'csv', + 'cxx', + 'diff', + 'erb', + 'groovy', + 'h', + 'haml', + 'hh', + 'htm', + 'html', + 'java', + 'js', + 'json', + 'less', + 'log', + 'm', + 'make', + 'markdown', + 'md', + 'ml', + 'mm', + 'php', + 'pl', + 'plist', + 'properties', + 'py', + 'rb', + 'rst', + 'sass', + 'scala', + 'scm', + 'script', + 'sh', + 'sml', + 'sql', + 'tsv', + 'txt', + 'vim', + 'vtt', + 'xhtml', + 'xml', + 'xsd', + 'xsl', + 'yaml', + 'yml', +]) + +/** + * Formats Box lists with `Text? = Yes` in the representation "Supported File + * Types" table whose raw bytes are not usable as UTF-8 (binary, proprietary, or + * markup-heavy). Content for these is pulled from the `extracted_text` + * representation rather than from `GET /files/:id/content`. + */ +const REPRESENTATION_EXTENSIONS = new Set([ + 'boxcanvas', + 'boxnote', + 'doc', + 'docx', + 'fdx', + 'gdoc', + 'gsheet', + 'gslide', + 'gslides', + 'msg', + 'odp', + 'ods', + 'odt', + 'otp', + 'pdf', + 'ppt', + 'pptx', + 'rtf', + 'vi', + 'webdoc', + 'wpd', + 'xbd', + 'xdw', + 'xls', + 'xlsm', + 'xlsx', +]) + +/** Extensions rendered as HTML that must be stripped before indexing. */ +const HTML_EXTENSIONS = new Set(['htm', 'html', 'xhtml']) + +/** Bounded wait for Box to finish generating an on-demand text representation. */ +const REPRESENTATION_POLL_ATTEMPTS = 5 +const REPRESENTATION_POLL_DELAY_MS = 2000 + +/** Hosts Box serves representation and content downloads from. */ +const BOX_DOWNLOAD_HOST_SUFFIXES = ['.box.com', '.boxcloud.com'] + +interface BoxPathEntry { + id?: string + name?: string +} + +interface BoxItem { + type?: string + id: string + name?: string + etag?: string | null + size?: number + created_at?: string + modified_at?: string + extension?: string + item_status?: string + trashed_at?: string | null + path_collection?: { entries?: BoxPathEntry[] } +} + +interface BoxFolderItemsResponse { + entries?: BoxItem[] + next_marker?: string | null +} + +interface BoxRepresentationEntry { + representation?: string + content?: { url_template?: string } + info?: { url?: string } + status?: { state?: string } +} + +interface BoxFileWithRepresentations extends BoxItem { + representations?: { entries?: BoxRepresentationEntry[] } +} + +/** + * Traversal position across pages of a single sync run. Box has no recursive + * listing endpoint, so the connector walks the folder tree breadth-first and + * carries the pending-folder queue plus the current folder's marker in the cursor. + */ +interface BoxTraversalState { + /** Folder IDs discovered but not yet listed. */ + queue: string[] + /** Folder currently being listed. */ + folderId: string + /** Box marker for the next page of `folderId`, if any. */ + marker?: string +} + +function encodeCursor(state: BoxTraversalState): string { + return Buffer.from(JSON.stringify(state), 'utf8').toString('base64url') +} + +function decodeCursor(cursor: string): BoxTraversalState | null { + try { + const parsed = JSON.parse(Buffer.from(cursor, 'base64url').toString('utf8')) as unknown + if (typeof parsed !== 'object' || parsed === null) return null + const candidate = parsed as Partial + if (typeof candidate.folderId !== 'string' || !Array.isArray(candidate.queue)) return null + return { + folderId: candidate.folderId, + queue: candidate.queue.filter((id): id is string => typeof id === 'string'), + marker: typeof candidate.marker === 'string' ? candidate.marker : undefined, + } + } catch { + return null + } +} + +function normalizeFolderId(value: unknown): string { + const raw = typeof value === 'string' ? value.trim() : '' + return raw || BOX_ROOT_FOLDER_ID +} + +function getExtension(item: BoxItem): string { + if (item.extension) return item.extension.toLowerCase() + const name = item.name ?? '' + const dotIndex = name.lastIndexOf('.') + return dotIndex === -1 ? '' : name.slice(dotIndex + 1).toLowerCase() +} + +function isSupportedFile(item: BoxItem): boolean { + if (item.type !== 'file') return false + const extension = getExtension(item) + return PLAIN_TEXT_EXTENSIONS.has(extension) || REPRESENTATION_EXTENSIONS.has(extension) +} + +/** + * Builds a human-readable path from Box's `path_collection` ancestry, dropping the + * synthetic root entry (id `0`, "All Files") so paths read as `/Reports/q3.pdf`. + */ +function buildPath(item: BoxItem): string { + const ancestors = (item.path_collection?.entries ?? []) + .filter((entry) => entry.id !== BOX_ROOT_FOLDER_ID && typeof entry.name === 'string') + .map((entry) => entry.name as string) + return `/${[...ancestors, item.name ?? ''].join('/')}` +} + +/** + * Metadata-only stub. `etag` is Box's per-version entity tag and changes on every + * new file version, so it is the change indicator; `modified_at` is the fallback + * for the rare item where Box omits an etag. + */ +function fileToStub(item: BoxItem): ExternalDocument { + return { + externalId: item.id, + title: item.name || 'Untitled', + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: `https://app.box.com/file/${item.id}`, + contentHash: `box:${item.id}:${item.etag ?? item.modified_at ?? ''}`, + metadata: { + path: buildPath(item), + extension: getExtension(item), + lastModified: item.modified_at, + fileSize: item.size, + }, + } +} + +function assertBoxDownloadHost(url: string): void { + const host = new URL(url).hostname.toLowerCase() + const allowed = BOX_DOWNLOAD_HOST_SUFFIXES.some( + (suffix) => host === suffix.slice(1) || host.endsWith(suffix) + ) + if (!allowed) { + throw new Error(`Refusing to download Box content from unexpected host: ${host}`) + } +} + +/** + * Streams a Box download against the connector size cap, raising + * {@link ConnectorFileTooLargeError} when the body exceeds it so an oversized file + * surfaces as a skipped row instead of being buffered whole. + */ +async function downloadWithinLimit(url: string, accessToken: string): Promise { + assertBoxDownloadHost(url) + + const response = await fetchWithRetry(url, { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}` }, + }) + + if (response.status === 202) { + throw new Error('Box content is not yet ready for download') + } + if (!response.ok) { + throw new Error(`Failed to download Box content: ${response.status}`) + } + + const buffer = await readBodyWithLimit(response, MAX_FILE_SIZE) + if (!buffer) { + throw new ConnectorFileTooLargeError(MAX_FILE_SIZE) + } + return buffer +} + +async function fetchPlainTextContent( + accessToken: string, + fileId: string, + extension: string +): Promise { + const buffer = await downloadWithinLimit(`${BOX_API_BASE}/files/${fileId}/content`, accessToken) + const text = buffer.toString('utf8') + return HTML_EXTENSIONS.has(extension) ? htmlToPlainText(text) : text +} + +/** + * Resolves the `extracted_text` representation for a file. + * + * Box generates text representations on demand: the first request reports state + * `none`, and calling the representation's `info.url` starts generation. The state + * is then polled a bounded number of times. When generation has not finished in + * time the caller returns `null` so the document is retried on the next sync rather + * than being stored with empty content. + */ +async function fetchExtractedText( + accessToken: string, + entry: BoxRepresentationEntry +): Promise { + const infoUrl = entry.info?.url + let urlTemplate = entry.content?.url_template + let state = entry.status?.state + if (!urlTemplate && !infoUrl) return null + + for (let attempt = 0; attempt <= REPRESENTATION_POLL_ATTEMPTS; attempt++) { + if ((state === 'success' || state === 'viewable') && urlTemplate) { + const buffer = await downloadWithinLimit( + urlTemplate.replace('{+asset_path}', ''), + accessToken + ) + return buffer.toString('utf8') + } + if (state === 'error' || !infoUrl) return null + if (attempt === REPRESENTATION_POLL_ATTEMPTS) break + + assertBoxDownloadHost(infoUrl) + const response = await fetchWithRetry(infoUrl, { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}` }, + }) + if (!response.ok && response.status !== 202) return null + + if (response.status === 202) { + state = 'pending' + } else { + const info = (await response.json()) as BoxRepresentationEntry + state = info.status?.state ?? 'pending' + urlTemplate = info.content?.url_template ?? urlTemplate + } + + if (state !== 'success' && state !== 'viewable') { + await sleep(REPRESENTATION_POLL_DELAY_MS) + } + } + + return null +} + +/** + * Lists one page of a folder. A folder the credential can no longer read is + * reported rather than thrown, so one inaccessible subtree does not abort the + * whole listing — the caller flags the listing as capped instead. + */ +async function listFolderPage( + accessToken: string, + folderId: string, + marker: string | undefined +): Promise { + const params = new URLSearchParams({ + fields: ITEM_FIELDS, + limit: String(FOLDER_ITEMS_PAGE_SIZE), + usemarker: 'true', + }) + if (marker) params.set('marker', marker) + + const response = await fetchWithRetry( + `${BOX_API_BASE}/folders/${encodeURIComponent(folderId)}/items?${params.toString()}`, + { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}` }, + } + ) + + if (response.status === 403 || response.status === 404) { + logger.warn('Skipping inaccessible Box folder', { folderId, status: response.status }) + return null + } + + if (!response.ok) { + const errorText = await response.text() + logger.error('Failed to list Box folder items', { + folderId, + status: response.status, + error: errorText, + }) + throw new Error(`Failed to list Box folder items: ${response.status}`) + } + + return (await response.json()) as BoxFolderItemsResponse +} + +export const boxConnector: ConnectorConfig = { + ...boxConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + cursor?: string, + syncContext?: Record + ): Promise => { + const rootFolderId = normalizeFolderId(sourceConfig.folderId) + const state: BoxTraversalState = cursor + ? (decodeCursor(cursor) ?? { queue: [], folderId: rootFolderId }) + : { queue: [], folderId: rootFolderId } + + const queue = [...state.queue] + const files: BoxItem[] = [] + let position: { folderId: string; marker?: string } | null = { + folderId: state.folderId, + marker: state.marker, + } + + for (let fetched = 0; fetched < FOLDER_PAGES_PER_CALL && position; fetched++) { + const page = await listFolderPage(accessToken, position.folderId, position.marker) + + if (page) { + for (const item of page.entries ?? []) { + if (item.type === 'folder') { + queue.push(item.id) + } else if (isSupportedFile(item)) { + files.push(item) + } + } + } else if (syncContext) { + /** + * A folder was skipped, so documents that still exist in Box are absent from + * this listing. Without this flag the engine would reconcile them as deleted. + */ + syncContext.listingCapped = true + } + + const nextMarker = page?.next_marker || undefined + if (nextMarker) { + position = { folderId: position.folderId, marker: nextMarker } + } else { + const nextFolderId = queue.shift() + position = nextFolderId ? { folderId: nextFolderId } : null + } + + if (files.length >= MAX_FILES_PER_CALL) break + } + + const nextState: BoxTraversalState | null = position + ? { queue, folderId: position.folderId, marker: position.marker } + : null + + const maxFiles = sourceConfig.maxFiles ? Number(sourceConfig.maxFiles) : 0 + const previouslyFetched = (syncContext?.totalDocsFetched as number) ?? 0 + + const stubs = files.map((item) => stubOrSkipBySize(fileToStub(item), item.size, MAX_FILE_SIZE)) + + const { documents, indexableCount, capReached } = takeIndexableWithinCap( + stubs, + isSkippedDocument, + maxFiles, + previouslyFetched + ) + + if (syncContext) syncContext.totalDocsFetched = previouslyFetched + indexableCount + + /** + * The cap truncates the listing when it stopped traversal with folders still + * pending, or when it dropped items from this very page. Reaching the cap on + * the final item of the final page hides nothing, so deletion reconciliation + * stays enabled in that case. + */ + const droppedInPage = documents.length < stubs.length + const hitLimit = capReached && (nextState !== null || droppedInPage) + if (hitLimit && syncContext) syncContext.listingCapped = true + + return { + documents, + nextCursor: hitLimit || !nextState ? undefined : encodeCursor(nextState), + hasMore: !hitLimit && nextState !== null, + } + }, + + getDocument: async ( + accessToken: string, + _sourceConfig: Record, + externalId: string + ): Promise => { + try { + const response = await fetchWithRetry( + `${BOX_API_BASE}/files/${encodeURIComponent(externalId)}?fields=${FILE_FIELDS},representations`, + { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + 'x-rep-hints': '[extracted_text]', + }, + } + ) + + if (response.status === 404 || response.status === 403) return null + if (!response.ok) { + throw new Error(`Failed to get Box file metadata: ${response.status}`) + } + + const file = (await response.json()) as BoxFileWithRepresentations + if (file.trashed_at || (file.item_status && file.item_status !== 'active')) return null + if (!isSupportedFile({ ...file, type: file.type ?? 'file' })) return null + + const stub = fileToStub(file) + if (file.size && file.size > MAX_FILE_SIZE) { + return markSkipped(stub, sizeLimitSkipReason(MAX_FILE_SIZE)) + } + + const extension = getExtension(file) + + let content: string | null + try { + if (PLAIN_TEXT_EXTENSIONS.has(extension)) { + content = await fetchPlainTextContent(accessToken, file.id, extension) + } else { + const entry = (file.representations?.entries ?? []).find( + (candidate) => candidate.representation === 'extracted_text' + ) + content = entry ? await fetchExtractedText(accessToken, entry) : null + } + } catch (error) { + if (error instanceof ConnectorFileTooLargeError) { + return markSkipped(stub, sizeLimitSkipReason(error.limitBytes)) + } + throw error + } + + if (!content?.trim()) return null + + return { ...stub, content, contentDeferred: false } + } catch (error) { + logger.warn(`Failed to fetch Box document ${externalId}`, { + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const maxFiles = sourceConfig.maxFiles as string | undefined + if (maxFiles && (Number.isNaN(Number(maxFiles)) || Number(maxFiles) <= 0)) { + return { valid: false, error: 'Max files must be a positive number' } + } + + const folderId = normalizeFolderId(sourceConfig.folderId) + if (!/^\d+$/.test(folderId)) { + return { + valid: false, + error: 'Folder ID must be a numeric Box folder ID (use 0 for the root)', + } + } + + try { + const response = await fetchWithRetry( + `${BOX_API_BASE}/folders/${folderId}?fields=id,name`, + { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}` }, + }, + VALIDATE_RETRY_OPTIONS + ) + + if (response.status === 404) { + return { valid: false, error: 'Folder not found. Check the folder ID and try again.' } + } + if (response.status === 401 || response.status === 403) { + return { + valid: false, + error: 'Box denied access to this folder. Reconnect or pick another folder.', + } + } + if (!response.ok) { + return { valid: false, error: `Failed to access Box: ${response.status}` } + } + + return { valid: true } + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Failed to validate configuration') } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + if (typeof metadata.path === 'string' && metadata.path) { + result.path = metadata.path + } + + if (typeof metadata.extension === 'string' && metadata.extension) { + result.extension = metadata.extension + } + + const lastModified = parseTagDate(metadata.lastModified) + if (lastModified) result.lastModified = lastModified + + if (metadata.fileSize != null) { + const num = Number(metadata.fileSize) + if (!Number.isNaN(num)) result.fileSize = num + } + + return result + }, +} diff --git a/apps/sim/connectors/box/index.ts b/apps/sim/connectors/box/index.ts new file mode 100644 index 00000000000..03077f48037 --- /dev/null +++ b/apps/sim/connectors/box/index.ts @@ -0,0 +1 @@ +export { boxConnector } from '@/connectors/box/box' diff --git a/apps/sim/connectors/box/meta.ts b/apps/sim/connectors/box/meta.ts new file mode 100644 index 00000000000..a8c2643ef30 --- /dev/null +++ b/apps/sim/connectors/box/meta.ts @@ -0,0 +1,42 @@ +import { BoxCompanyIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const boxConnectorMeta: ConnectorMeta = { + id: 'box', + name: 'Box', + description: 'Sync text-extractable files from Box', + version: '1.0.0', + icon: BoxCompanyIcon, + + auth: { + mode: 'oauth', + provider: 'box', + requiredScopes: ['root_readwrite'], + }, + + configFields: [ + { + id: 'folderId', + title: 'Folder ID', + type: 'short-input', + placeholder: 'e.g. 123456789 (default: entire account)', + required: false, + description: + 'Numeric Box folder ID to sync recursively. Leave empty (or use 0) to sync all files.', + }, + { + id: 'maxFiles', + title: 'Max Files', + type: 'short-input', + required: false, + placeholder: 'e.g. 500 (default: unlimited)', + }, + ], + + tagDefinitions: [ + { id: 'path', displayName: 'File Path', fieldType: 'text' }, + { id: 'extension', displayName: 'Extension', fieldType: 'text' }, + { id: 'lastModified', displayName: 'Last Modified', fieldType: 'date' }, + { id: 'fileSize', displayName: 'File Size (bytes)', fieldType: 'number' }, + ], +} diff --git a/apps/sim/connectors/google-slides/google-slides.ts b/apps/sim/connectors/google-slides/google-slides.ts new file mode 100644 index 00000000000..fad9dc5cbae --- /dev/null +++ b/apps/sim/connectors/google-slides/google-slides.ts @@ -0,0 +1,495 @@ +import { createLogger } from '@sim/logger' +import { toError } from '@sim/utils/errors' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { googleSlidesConnectorMeta } from '@/connectors/google-slides/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { + buildDriveParentsClause, + joinTagArray, + parseMultiValue, + parseTagDate, +} from '@/connectors/utils' + +const logger = createLogger('GoogleSlidesConnector') + +const PRESENTATION_MIME_TYPE = 'application/vnd.google-apps.presentation' + +/** Reason recorded for a presentation whose slides contain no extractable text. */ +const NO_TEXT = 'No extractable text' + +/** Guards against a pathological (or cyclic) group nesting depth. */ +const MAX_GROUP_DEPTH = 12 + +/** + * Represents a Google Drive file entry returned by the Drive API. + */ +interface DriveFile { + id: string + name: string + mimeType: string + modifiedTime?: string + createdTime?: string + webViewLink?: string + owners?: { displayName?: string; emailAddress?: string }[] +} + +/** + * A single formatting-consistent run of text within a Slides text element. + */ +interface SlidesTextElement { + textRun?: { content?: string } +} + +/** + * The Slides API `TextContent` object carried by shapes and table cells. + */ +interface SlidesTextContent { + textElements?: SlidesTextElement[] +} + +interface SlidesTableCell { + text?: SlidesTextContent +} + +interface SlidesTableRow { + tableCells?: SlidesTableCell[] +} + +/** + * A page element on a slide. Exactly one of the visual properties is set; + * only `shape`, `table`, `wordArt`, and `elementGroup` can carry extractable + * text — `image`, `video`, `line`, `sheetsChart`, and `speakerSpotlight` do not. + */ +interface SlidesPageElement { + objectId?: string + shape?: { text?: SlidesTextContent } + table?: { tableRows?: SlidesTableRow[] } + wordArt?: { renderedText?: string } + elementGroup?: { children?: SlidesPageElement[] } +} + +/** + * A Slides API `Page`. Slides carry `slideProperties.notesPage`, whose + * `notesProperties.speakerNotesObjectId` names the element holding the notes. + */ +interface SlidesPage { + objectId?: string + pageElements?: SlidesPageElement[] + notesProperties?: { speakerNotesObjectId?: string } + slideProperties?: { notesPage?: SlidesPage } +} + +interface SlidesPresentation { + slides?: SlidesPage[] +} + +/** + * Flattens a Slides `TextContent` into plain text. Slides encodes a soft line + * break as a vertical tab, which is normalized to a newline. + */ +function extractTextContent(text: SlidesTextContent | undefined): string { + const elements = text?.textElements + if (!elements) return '' + + return elements + .map((element) => element.textRun?.content ?? '') + .join('') + .replace(/\v/g, '\n') + .replace(/\n+$/, '') +} + +/** + * Walks page elements in document order, appending each element's text to + * `parts`. Groups are recursed into so nested shapes are not dropped. + */ +function collectElementText( + elements: SlidesPageElement[] | undefined, + parts: string[], + depth: number +): void { + if (!elements || depth > MAX_GROUP_DEPTH) return + + for (const element of elements) { + const shapeText = extractTextContent(element.shape?.text) + if (shapeText.trim()) parts.push(shapeText) + + const wordArtText = element.wordArt?.renderedText + if (wordArtText?.trim()) parts.push(wordArtText.trim()) + + const rows = element.table?.tableRows + if (rows) { + for (const row of rows) { + const cells = (row.tableCells ?? []) + .map((cell) => extractTextContent(cell.text).replace(/\n/g, ' ').trim()) + .filter(Boolean) + if (cells.length > 0) parts.push(cells.join(' | ')) + } + } + + if (element.elementGroup?.children) { + collectElementText(element.elementGroup.children, parts, depth + 1) + } + } +} + +/** + * Extracts the speaker notes for a slide. The notes page mirrors the slide's + * body placeholders, so only the element named by `speakerNotesObjectId` is + * read — anything else would duplicate the slide's own text. + */ +function extractSpeakerNotes(slide: SlidesPage): string { + const notesPage = slide.slideProperties?.notesPage + const notesObjectId = notesPage?.notesProperties?.speakerNotesObjectId + if (!notesPage || !notesObjectId) return '' + + const notesElement = notesPage.pageElements?.find((element) => element.objectId === notesObjectId) + return extractTextContent(notesElement?.shape?.text) +} + +/** + * Renders a presentation as plain text, preserving slide order. Each slide is + * introduced by a Markdown heading so retrieved chunks keep their position. + */ +function extractTextFromPresentation( + presentation: SlidesPresentation, + includeSpeakerNotes: boolean +): string { + const slides = presentation.slides + if (!slides) return '' + + const sections: string[] = [] + + for (let index = 0; index < slides.length; index++) { + const parts: string[] = [] + collectElementText(slides[index].pageElements, parts, 0) + + if (includeSpeakerNotes) { + const notes = extractSpeakerNotes(slides[index]) + if (notes.trim()) parts.push(`Speaker notes: ${notes}`) + } + + if (parts.length > 0) { + sections.push(`## Slide ${index + 1}\n${parts.join('\n')}`) + } + } + + return sections.join('\n\n').trim() +} + +/** + * Fetches a presentation via the Slides API and extracts its text. Only the + * `slides` field is requested — masters, layouts, and the notes master carry + * template boilerplate that would pollute the index. + */ +async function fetchPresentationContent( + accessToken: string, + presentationId: string, + includeSpeakerNotes: boolean +): Promise { + const url = `https://slides.googleapis.com/v1/presentations/${encodeURIComponent(presentationId)}?fields=slides` + + const response = await fetchWithRetry(url, { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }) + + if (!response.ok) { + throw new Error( + `Failed to fetch Google Slides presentation ${presentationId}: ${response.status}` + ) + } + + const presentation = (await response.json()) as SlidesPresentation + return extractTextFromPresentation(presentation, includeSpeakerNotes) +} + +/** + * Resolves the speaker-notes preference. Notes are included unless the user + * explicitly opted out, so an unset legacy config keeps the richer content. + */ +function shouldIncludeSpeakerNotes(sourceConfig: Record): boolean { + return sourceConfig.includeSpeakerNotes !== 'no' +} + +/** + * Creates a lightweight stub from a Drive file entry. Content is deferred + * and only fetched via getDocument for new or changed documents. + */ +function fileToStub(file: DriveFile): ExternalDocument { + return { + externalId: file.id, + title: file.name || 'Untitled', + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: file.webViewLink || `https://docs.google.com/presentation/d/${file.id}/edit`, + contentHash: `gslides:${file.id}:${file.modifiedTime ?? ''}`, + metadata: { + modifiedTime: file.modifiedTime, + createdTime: file.createdTime, + owners: file.owners?.map((o) => o.displayName || o.emailAddress).filter(Boolean), + }, + } +} + +/** + * Builds the Drive API query string for listing Google Slides presentations. + * When `lastSyncAt` is supplied the listing is narrowed to presentations + * touched since the previous sync. + */ +function buildQuery(sourceConfig: Record, lastSyncAt?: Date): string { + const parts: string[] = ['trashed = false', `mimeType = '${PRESENTATION_MIME_TYPE}'`] + + const parentsClause = buildDriveParentsClause(parseMultiValue(sourceConfig.folderId)) + if (parentsClause) parts.push(parentsClause) + + if (lastSyncAt && !Number.isNaN(lastSyncAt.getTime())) { + parts.push(`modifiedTime > '${lastSyncAt.toISOString()}'`) + } + + return parts.join(' and ') +} + +export const googleSlidesConnector: ConnectorConfig = { + ...googleSlidesConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + cursor?: string, + syncContext?: Record, + lastSyncAt?: Date + ): Promise => { + const query = buildQuery(sourceConfig, lastSyncAt) + const pageSize = 100 + + const queryParams = new URLSearchParams({ + q: query, + pageSize: String(pageSize), + fields: + 'nextPageToken,incompleteSearch,files(id,name,mimeType,modifiedTime,createdTime,webViewLink,owners)', + supportsAllDrives: 'true', + includeItemsFromAllDrives: 'true', + }) + + if (cursor) { + queryParams.set('pageToken', cursor) + } + + const url = `https://www.googleapis.com/drive/v3/files?${queryParams.toString()}` + + logger.info('Listing Google Slides presentations', { query, cursor: cursor ?? 'initial' }) + + const response = await fetchWithRetry(url, { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }) + + if (!response.ok) { + const errorText = await response.text() + logger.error('Failed to list Google Slides presentations', { + status: response.status, + error: errorText, + }) + throw new Error(`Failed to list Google Slides presentations: ${response.status}`) + } + + const data = await response.json() + const files = (data.files || []) as DriveFile[] + + /** + * Drive sets `incompleteSearch` when it could not search every corpus (it + * arises with the `allDrives` scope enabled by `includeItemsFromAllDrives`). + * A partial listing drops still-existing presentations, so reconciliation + * must be suppressed to avoid hard-deleting valid documents. + */ + const incompleteSearch = data.incompleteSearch === true + + const maxDocs = sourceConfig.maxDocs ? Number(sourceConfig.maxDocs) : 0 + const previouslyFetched = (syncContext?.totalDocsFetched as number) ?? 0 + + let documents = files.map(fileToStub) + let slicedSome = false + if (maxDocs > 0) { + const remaining = maxDocs - previouslyFetched + if (documents.length > remaining) { + slicedSome = true + documents = documents.slice(0, remaining) + } + } + + const totalFetched = previouslyFetched + documents.length + if (syncContext) syncContext.totalDocsFetched = totalFetched + const hitLimit = maxDocs > 0 && totalFetched >= maxDocs + + const nextPageToken = data.nextPageToken as string | undefined + + /** + * Mark the listing as incomplete so the sync engine skips deletion + * reconciliation when this page does not represent the full source set: + * - `slicedSome`: the page held more presentations than `maxDocs` allowed. + * - `hitLimit` with a next page: the cap was reached while pages remain. + * - `incompleteSearch`: Drive could not search every corpus, so the page is + * partial and may omit still-existing presentations. + * Reconciliation against any of these would hard-delete valid documents. + */ + if (syncContext && (slicedSome || (hitLimit && Boolean(nextPageToken)) || incompleteSearch)) { + syncContext.listingCapped = true + } + + return { + documents, + nextCursor: hitLimit ? undefined : nextPageToken, + hasMore: hitLimit ? false : Boolean(nextPageToken), + } + }, + + getDocument: async ( + accessToken: string, + sourceConfig: Record, + externalId: string + ): Promise => { + const fields = 'id,name,mimeType,modifiedTime,createdTime,webViewLink,owners,trashed' + const url = `https://www.googleapis.com/drive/v3/files/${encodeURIComponent(externalId)}?fields=${encodeURIComponent(fields)}&supportsAllDrives=true` + + const response = await fetchWithRetry(url, { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }) + + if (!response.ok) { + if (response.status === 404) return null + throw new Error(`Failed to get Google Slides metadata: ${response.status}`) + } + + const file = (await response.json()) as DriveFile & { trashed?: boolean } + + if (file.trashed) return null + if (file.mimeType !== PRESENTATION_MIME_TYPE) return null + + const content = await fetchPresentationContent( + accessToken, + file.id, + shouldIncludeSpeakerNotes(sourceConfig) + ) + + /** + * An image-only deck carries no extractable text. Surfacing it as a skipped + * row keeps it visible in the knowledge base UI — returning `null` would + * make the engine drop the document with no reason recorded, so the + * presentation would simply be missing and re-fetched on every sync. + */ + if (!content.trim()) { + return { ...fileToStub(file), content: '', contentDeferred: false, skippedReason: NO_TEXT } + } + + return { ...fileToStub(file), content, contentDeferred: false } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const folderIds = parseMultiValue(sourceConfig.folderId) + const maxDocs = sourceConfig.maxDocs as string | undefined + + if (maxDocs && (Number.isNaN(Number(maxDocs)) || Number(maxDocs) <= 0)) { + return { valid: false, error: 'Max presentations must be a positive number' } + } + + const includeSpeakerNotes = sourceConfig.includeSpeakerNotes + if ( + includeSpeakerNotes != null && + includeSpeakerNotes !== '' && + includeSpeakerNotes !== 'yes' && + includeSpeakerNotes !== 'no' + ) { + return { valid: false, error: 'Speaker notes must be either "yes" or "no"' } + } + + try { + if (folderIds.length > 0) { + for (const folderId of folderIds) { + const url = `https://www.googleapis.com/drive/v3/files/${encodeURIComponent(folderId)}?fields=id,name,mimeType&supportsAllDrives=true` + const response = await fetchWithRetry( + url, + { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }, + VALIDATE_RETRY_OPTIONS + ) + + if (!response.ok) { + if (response.status === 404) { + return { + valid: false, + error: `Folder "${folderId}" not found. Check the folder ID and permissions.`, + } + } + return { + valid: false, + error: `Failed to access folder "${folderId}": ${response.status}`, + } + } + + const folder = await response.json() + if (folder.mimeType !== 'application/vnd.google-apps.folder') { + return { valid: false, error: `"${folderId}" is not a folder` } + } + } + } else { + const probeParams = new URLSearchParams({ + pageSize: '1', + q: `trashed = false and mimeType = '${PRESENTATION_MIME_TYPE}'`, + fields: 'files(id)', + supportsAllDrives: 'true', + includeItemsFromAllDrives: 'true', + }) + const response = await fetchWithRetry( + `https://www.googleapis.com/drive/v3/files?${probeParams.toString()}`, + { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }, + VALIDATE_RETRY_OPTIONS + ) + + if (!response.ok) { + return { valid: false, error: `Failed to access Google Slides: ${response.status}` } + } + } + + return { valid: true } + } catch (error) { + return { valid: false, error: toError(error).message || 'Failed to validate configuration' } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + const owners = joinTagArray(metadata.owners) + if (owners) result.owners = owners + + const lastModified = parseTagDate(metadata.modifiedTime) + if (lastModified) result.lastModified = lastModified + + return result + }, +} diff --git a/apps/sim/connectors/google-slides/index.ts b/apps/sim/connectors/google-slides/index.ts new file mode 100644 index 00000000000..56d00401c45 --- /dev/null +++ b/apps/sim/connectors/google-slides/index.ts @@ -0,0 +1 @@ +export { googleSlidesConnector } from '@/connectors/google-slides/google-slides' diff --git a/apps/sim/connectors/google-slides/meta.ts b/apps/sim/connectors/google-slides/meta.ts new file mode 100644 index 00000000000..a4cc9a17aef --- /dev/null +++ b/apps/sim/connectors/google-slides/meta.ts @@ -0,0 +1,68 @@ +import { GoogleSlidesIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const googleSlidesConnectorMeta: ConnectorMeta = { + id: 'google_slides', + name: 'Google Slides', + description: 'Sync Google Slides presentations', + version: '1.0.0', + icon: GoogleSlidesIcon, + + /** + * The Slides API has no dedicated Sim OAuth service. `presentations.get` + * accepts `https://www.googleapis.com/auth/drive`, which the `google-drive` + * provider already grants — the same provider every `google_slides` tool uses. + */ + auth: { + mode: 'oauth', + provider: 'google-drive', + requiredScopes: ['https://www.googleapis.com/auth/drive'], + }, + + configFields: [ + { + id: 'folderSelector', + title: 'Folders', + type: 'selector', + selectorKey: 'google.drive', + mimeType: 'application/vnd.google-apps.folder', + canonicalParamId: 'folderId', + mode: 'basic', + multi: true, + placeholder: 'Select one or more folders (optional)', + required: false, + }, + { + id: 'folderId', + title: 'Folder IDs', + type: 'short-input', + canonicalParamId: 'folderId', + mode: 'advanced', + multi: true, + placeholder: 'e.g. 1aBcDeFg…, 2cDeFgHi… (comma-separated for multiple)', + required: false, + }, + { + id: 'includeSpeakerNotes', + title: 'Speaker Notes', + type: 'dropdown', + required: false, + options: [ + { label: 'Include speaker notes', id: 'yes' }, + { label: 'Slide text only', id: 'no' }, + ], + }, + { + id: 'maxDocs', + title: 'Max Presentations', + type: 'short-input', + required: false, + placeholder: 'e.g. 500 (default: unlimited)', + }, + ], + + tagDefinitions: [ + { id: 'owners', displayName: 'Owner', fieldType: 'text' }, + { id: 'lastModified', displayName: 'Last Modified', fieldType: 'date' }, + ], +} diff --git a/apps/sim/connectors/google-vault/google-vault.ts b/apps/sim/connectors/google-vault/google-vault.ts new file mode 100644 index 00000000000..7c94e6205fd --- /dev/null +++ b/apps/sim/connectors/google-vault/google-vault.ts @@ -0,0 +1,722 @@ +import { createLogger } from '@sim/logger' +import { toError } from '@sim/utils/errors' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { googleVaultConnectorMeta } from '@/connectors/google-vault/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { computeContentHash, parseTagDate, takeIndexableWithinCap } from '@/connectors/utils' + +const logger = createLogger('GoogleVaultConnector') + +const VAULT_API_BASE = 'https://vault.googleapis.com/v1' + +/** Vault caps both `matters.list` and `matters.holds.list` page sizes at 100. */ +const PAGE_SIZE = 100 + +/** + * Matters whose child resources are enumerated in a single `listDocuments` call. + * + * The sync engine allows a bounded number of `listDocuments` calls per run, so + * spending one call per matter per child kind would truncate the listing after a + * few hundred matters. Draining a batch of matters per call keeps the call count + * proportional to `matters / BATCH` instead of `matters × kinds`. + */ +const CHILD_MATTER_BATCH = 8 + +/** Concurrent child listings issued within one batch. */ +const CHILD_CONCURRENCY = 4 + +/** + * Upper bound on child pages drained for a single matter/kind pair. Vault returns at + * most 100 children per page, so this covers 5,000 holds or saved queries in one + * matter; exceeding it marks the listing capped rather than looping unbounded. + */ +const MAX_CHILD_PAGES = 50 + +/** + * Google Vault matter, as returned by `matters.list`/`matters.get` with `view=FULL`. + * @see https://developers.google.com/workspace/vault/reference/rest/v1/matters + */ +interface VaultMatter { + matterId?: string + name?: string + description?: string + state?: string + matterRegion?: string + matterPermissions?: { accountId?: string; role?: string }[] +} + +/** Held account entry on a hold (`matters.holds` with `view=FULL_HOLD`). */ +interface VaultHeldAccount { + accountId?: string + email?: string + firstName?: string + lastName?: string + holdTime?: string +} + +/** Service-specific query options attached to a hold. */ +interface VaultCorpusQuery { + driveQuery?: { includeSharedDriveFiles?: boolean } + mailQuery?: { terms?: string; startTime?: string; endTime?: string } + groupsQuery?: { terms?: string; startTime?: string; endTime?: string } + hangoutsChatQuery?: { includeRooms?: boolean } + voiceQuery?: { coveredData?: string[] } +} + +/** + * Google Vault hold. + * @see https://developers.google.com/workspace/vault/reference/rest/v1/matters.holds + */ +interface VaultHold { + holdId?: string + name?: string + corpus?: string + updateTime?: string + orgUnit?: { orgUnitId?: string; holdTime?: string } + accounts?: VaultHeldAccount[] + query?: VaultCorpusQuery +} + +/** + * Search parameters shared by saved queries and exports. + * @see https://developers.google.com/workspace/vault/reference/rest/v1/Query + */ +interface VaultQuery { + corpus?: string + dataScope?: string + method?: string + searchMethod?: string + terms?: string + startTime?: string + endTime?: string + timeZone?: string + accountInfo?: { emails?: string[] } + orgUnitInfo?: { orgUnitId?: string } + sharedDriveInfo?: { sharedDriveIds?: string[] } + teamDriveInfo?: { teamDriveIds?: string[] } + hangoutsChatInfo?: { roomId?: string[] } + sitesUrlInfo?: { urls?: string[] } + driveDocumentInfo?: { documentIds?: { ids?: string[] } } +} + +/** + * Google Vault saved query. + * @see https://developers.google.com/workspace/vault/reference/rest/v1/matters.savedQueries + */ +interface VaultSavedQuery { + savedQueryId?: string + displayName?: string + matterId?: string + createTime?: string + query?: VaultQuery +} + +/** Child resource families enumerated under each matter. */ +type VaultChildKind = 'holds' | 'savedQueries' + +/** + * Opaque pagination state for a Vault sync. + * + * A single sync interleaves two levels of pagination: a page of matters, then the + * child resources of every matter on that page. The cursor carries the matter IDs of + * the current page so a resumed call never has to re-derive them. + * + * Child pagination is fully drained inside one call per batch of matters, so the + * cursor only has to remember how far through the matter page it has walked. + */ +interface VaultCursor { + phase: 'matters' | 'children' + /** Page token used to fetch the matters page processed in the `matters` phase. */ + mattersPageToken?: string + /** Page token for the matters page that follows the one currently being walked. */ + nextMattersPageToken?: string + matterIds?: string[] + /** Index into `matterIds` of the first matter in the next batch to process. */ + matterIndex?: number +} + +function encodeCursor(cursor: VaultCursor): string { + return Buffer.from(JSON.stringify(cursor), 'utf8').toString('base64url') +} + +function decodeCursor(cursor?: string): VaultCursor { + if (!cursor) return { phase: 'matters' } + try { + return JSON.parse(Buffer.from(cursor, 'base64url').toString('utf8')) as VaultCursor + } catch { + return { phase: 'matters' } + } +} + +function readString(value: unknown): string | undefined { + const text = typeof value === 'string' ? value.trim() : '' + return text.length > 0 ? text : undefined +} + +function readBoolean(value: unknown, fallback: boolean): boolean { + if (typeof value === 'boolean') return value + const text = readString(value)?.toLowerCase() + if (text === 'true') return true + if (text === 'false') return false + return fallback +} + +/** Parses a positive integer cap; `0` means unlimited. */ +function readMaxDocuments(value: unknown): number { + const text = readString(value) + if (!text) return 0 + const parsed = Number(text) + return Number.isFinite(parsed) && parsed > 0 ? Math.floor(parsed) : 0 +} + +function enabledChildKinds(sourceConfig: Record): VaultChildKind[] { + const kinds: VaultChildKind[] = [] + if (readBoolean(sourceConfig.includeHolds, true)) kinds.push('holds') + if (readBoolean(sourceConfig.includeSavedQueries, true)) kinds.push('savedQueries') + return kinds +} + +async function vaultGet(accessToken: string, url: string, label: string): Promise { + return fetchWithRetry(url, { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }).catch((error) => { + throw new Error(`Failed to reach Google Vault (${label}): ${toError(error).message}`) + }) +} + +function appendLine(lines: string[], label: string, value?: string): void { + if (value?.trim()) lines.push(`${label}: ${value.trim()}`) +} + +/** Renders a matter as indexable plain text. */ +function renderMatter(matter: VaultMatter): string { + const lines: string[] = [] + appendLine(lines, 'Matter', matter.name) + appendLine(lines, 'Matter ID', matter.matterId) + appendLine(lines, 'State', matter.state) + appendLine(lines, 'Region', matter.matterRegion) + appendLine(lines, 'Description', matter.description) + + const permissions = (matter.matterPermissions ?? []) + .map((permission) => + [permission.accountId, permission.role].filter((part) => Boolean(part)).join(' — ') + ) + .filter((entry) => entry.length > 0) + if (permissions.length > 0) { + lines.push('Permissions:') + for (const permission of permissions) lines.push(` - ${permission}`) + } + + return lines.join('\n') +} + +/** Renders a hold, including its scope and search refinements, as plain text. */ +function renderHold(hold: VaultHold, matterId: string): string { + const lines: string[] = [] + appendLine(lines, 'Hold', hold.name) + appendLine(lines, 'Hold ID', hold.holdId) + appendLine(lines, 'Matter ID', matterId) + appendLine(lines, 'Corpus', hold.corpus) + appendLine(lines, 'Last updated', hold.updateTime) + appendLine(lines, 'Organizational unit', hold.orgUnit?.orgUnitId) + + const accounts = (hold.accounts ?? []) + .map((account) => { + const name = [account.firstName, account.lastName].filter(Boolean).join(' ') + return [account.email ?? account.accountId, name].filter(Boolean).join(' — ') + }) + .filter((entry) => entry.length > 0) + if (accounts.length > 0) { + lines.push('Held accounts:') + for (const account of accounts) lines.push(` - ${account}`) + } + + const mailLike = hold.query?.mailQuery ?? hold.query?.groupsQuery + appendLine(lines, 'Search terms', mailLike?.terms) + appendLine(lines, 'Start time', mailLike?.startTime) + appendLine(lines, 'End time', mailLike?.endTime) + if (hold.query?.driveQuery?.includeSharedDriveFiles !== undefined) { + appendLine( + lines, + 'Includes shared drive files', + String(hold.query.driveQuery.includeSharedDriveFiles) + ) + } + if (hold.query?.hangoutsChatQuery?.includeRooms !== undefined) { + appendLine(lines, 'Includes chat spaces', String(hold.query.hangoutsChatQuery.includeRooms)) + } + const coveredData = hold.query?.voiceQuery?.coveredData + if (coveredData && coveredData.length > 0) { + appendLine(lines, 'Covered Voice data', coveredData.join(', ')) + } + + return lines.join('\n') +} + +/** Renders a saved query and its search parameters as plain text. */ +function renderSavedQuery(savedQuery: VaultSavedQuery, matterId: string): string { + const lines: string[] = [] + appendLine(lines, 'Saved query', savedQuery.displayName) + appendLine(lines, 'Saved query ID', savedQuery.savedQueryId) + appendLine(lines, 'Matter ID', matterId) + appendLine(lines, 'Created', savedQuery.createTime) + + const query = savedQuery.query + appendLine(lines, 'Corpus', query?.corpus) + appendLine(lines, 'Data scope', query?.dataScope) + appendLine(lines, 'Search method', query?.method ?? query?.searchMethod) + appendLine(lines, 'Search terms', query?.terms) + appendLine(lines, 'Start time', query?.startTime) + appendLine(lines, 'End time', query?.endTime) + appendLine(lines, 'Time zone', query?.timeZone) + appendLine(lines, 'Accounts', query?.accountInfo?.emails?.join(', ')) + appendLine(lines, 'Organizational unit', query?.orgUnitInfo?.orgUnitId) + appendLine( + lines, + 'Shared drives', + (query?.sharedDriveInfo?.sharedDriveIds ?? query?.teamDriveInfo?.teamDriveIds)?.join(', ') + ) + appendLine(lines, 'Chat spaces', query?.hangoutsChatInfo?.roomId?.join(', ')) + appendLine(lines, 'Site URLs', query?.sitesUrlInfo?.urls?.join(', ')) + appendLine(lines, 'Drive documents', query?.driveDocumentInfo?.documentIds?.ids?.join(', ')) + + return lines.join('\n') +} + +/** + * Builds a matter document. + * + * Vault matters expose no modification timestamp, so change detection hashes the + * matter's own metadata fields. Every field in the hash comes from the `view=FULL` + * payload, which `listDocuments` and `getDocument` both request — so the hash is + * identical on both paths. + */ +async function matterToDocument(matter: VaultMatter): Promise { + const matterId = matter.matterId ?? '' + const canonical = JSON.stringify({ + name: matter.name ?? '', + description: matter.description ?? '', + state: matter.state ?? '', + matterRegion: matter.matterRegion ?? '', + permissions: (matter.matterPermissions ?? []) + .map((permission) => `${permission.accountId ?? ''}:${permission.role ?? ''}`) + .sort(), + }) + + return { + externalId: `matter:${matterId}`, + title: matter.name || `Matter ${matterId}`, + content: renderMatter(matter), + mimeType: 'text/plain', + contentHash: `gvault:matter:${matterId}:${await computeContentHash(canonical)}`, + metadata: { + resourceType: 'matter', + matterId, + state: matter.state, + }, + } +} + +/** + * Builds a hold document. Holds carry `updateTime`, which Vault bumps on every + * modification, so it is a sufficient change indicator. + */ +function holdToDocument(hold: VaultHold, matterId: string): ExternalDocument { + const holdId = hold.holdId ?? '' + return { + externalId: `hold:${matterId}:${holdId}`, + title: hold.name || `Hold ${holdId}`, + content: renderHold(hold, matterId), + mimeType: 'text/plain', + contentHash: `gvault:hold:${matterId}:${holdId}:${hold.updateTime ?? ''}`, + metadata: { + resourceType: 'hold', + matterId, + corpus: hold.corpus, + lastModified: hold.updateTime, + }, + } +} + +/** + * Builds a saved query document. Saved queries are immutable once created (the API + * exposes only create, get, list, and delete), so `createTime` identifies the version. + * The `v2` token in the hash tracks the rendering itself: because the source can never + * change, a rendering change would otherwise never re-index existing documents. + */ +function savedQueryToDocument(savedQuery: VaultSavedQuery, matterId: string): ExternalDocument { + const savedQueryId = savedQuery.savedQueryId ?? '' + return { + externalId: `savedQuery:${matterId}:${savedQueryId}`, + title: savedQuery.displayName || `Saved query ${savedQueryId}`, + content: renderSavedQuery(savedQuery, matterId), + mimeType: 'text/plain', + contentHash: `gvault:savedquery:v2:${matterId}:${savedQueryId}:${savedQuery.createTime ?? ''}`, + metadata: { + resourceType: 'savedQuery', + matterId, + corpus: savedQuery.query?.corpus, + lastModified: savedQuery.createTime, + }, + } +} + +/** Fetches one page of matters (or the single configured matter). */ +async function fetchMattersPage( + accessToken: string, + sourceConfig: Record, + pageToken?: string +): Promise<{ matters: VaultMatter[]; nextPageToken?: string }> { + const singleMatterId = readString(sourceConfig.matterId) + + if (singleMatterId) { + const response = await vaultGet( + accessToken, + `${VAULT_API_BASE}/matters/${encodeURIComponent(singleMatterId)}?view=FULL`, + 'matters.get' + ) + if (!response.ok) { + throw new Error(`Failed to fetch Vault matter ${singleMatterId}: ${response.status}`) + } + return { matters: [(await response.json()) as VaultMatter] } + } + + const params = new URLSearchParams({ view: 'FULL', pageSize: String(PAGE_SIZE) }) + const state = readString(sourceConfig.matterState) + if (state && state !== 'ALL') params.set('state', state) + if (pageToken) params.set('pageToken', pageToken) + + const response = await vaultGet( + accessToken, + `${VAULT_API_BASE}/matters?${params.toString()}`, + 'matters.list' + ) + if (!response.ok) { + throw new Error(`Failed to list Vault matters: ${response.status}`) + } + + const data = (await response.json()) as { matters?: VaultMatter[]; nextPageToken?: string } + return { matters: data.matters ?? [], nextPageToken: data.nextPageToken } +} + +/** Fetches one page of a matter's child resources of the given kind. */ +async function fetchChildPage( + accessToken: string, + matterId: string, + kind: VaultChildKind, + pageToken?: string +): Promise<{ documents: ExternalDocument[]; nextPageToken?: string }> { + const params = new URLSearchParams({ pageSize: String(PAGE_SIZE) }) + if (kind === 'holds') params.set('view', 'FULL_HOLD') + if (pageToken) params.set('pageToken', pageToken) + + const response = await vaultGet( + accessToken, + `${VAULT_API_BASE}/matters/${encodeURIComponent(matterId)}/${kind}?${params.toString()}`, + `matters.${kind}.list` + ) + if (!response.ok) { + throw new Error(`Failed to list ${kind} for matter ${matterId}: ${response.status}`) + } + + if (kind === 'holds') { + const data = (await response.json()) as { holds?: VaultHold[]; nextPageToken?: string } + return { + documents: (data.holds ?? []).map((hold) => holdToDocument(hold, matterId)), + nextPageToken: data.nextPageToken, + } + } + + const data = (await response.json()) as { + savedQueries?: VaultSavedQuery[] + nextPageToken?: string + } + return { + documents: (data.savedQueries ?? []).map((savedQuery) => + savedQueryToDocument(savedQuery, matterId) + ), + nextPageToken: data.nextPageToken, + } +} + +/** + * Drains every page of one matter's child listing of the given kind. + * + * `capped` is true when the listing was cut short — by the page bound or by a request + * failure (a matter the caller cannot read, or a transient error). The caller turns + * that into `syncContext.listingCapped` so deletion reconciliation is skipped for the + * run rather than hard-deleting documents that still exist at the source. + */ +async function fetchAllChildren( + accessToken: string, + matterId: string, + kind: VaultChildKind +): Promise<{ documents: ExternalDocument[]; capped: boolean }> { + const documents: ExternalDocument[] = [] + let pageToken: string | undefined + + try { + for (let page = 0; page < MAX_CHILD_PAGES; page++) { + const result = await fetchChildPage(accessToken, matterId, kind, pageToken) + documents.push(...result.documents) + if (!result.nextPageToken) return { documents, capped: false } + pageToken = result.nextPageToken + } + } catch (error) { + logger.warn(`Failed to list ${kind} for Vault matter ${matterId}`, { + error: toError(error).message, + }) + return { documents, capped: true } + } + + logger.warn(`Stopped listing ${kind} for Vault matter ${matterId} at the page bound`, { + maxChildPages: MAX_CHILD_PAGES, + }) + return { documents, capped: true } +} + +/** + * Lists every enabled child resource for a batch of matters, bounded concurrency. + * + * One `listDocuments` call covers a whole batch, which keeps the number of calls a + * sync needs proportional to the matter count rather than to `matters × kinds × + * child pages`. + */ +async function fetchChildrenForMatters( + accessToken: string, + matterIds: string[], + kinds: VaultChildKind[] +): Promise<{ documents: ExternalDocument[]; capped: boolean }> { + const tasks: { matterId: string; kind: VaultChildKind }[] = [] + for (const matterId of matterIds) { + for (const kind of kinds) tasks.push({ matterId, kind }) + } + + const documents: ExternalDocument[] = [] + let capped = false + + for (let index = 0; index < tasks.length; index += CHILD_CONCURRENCY) { + const results = await Promise.all( + tasks + .slice(index, index + CHILD_CONCURRENCY) + .map((task) => fetchAllChildren(accessToken, task.matterId, task.kind)) + ) + for (const result of results) { + documents.push(...result.documents) + if (result.capped) capped = true + } + } + + return { documents, capped } +} + +export const googleVaultConnector: ConnectorConfig = { + ...googleVaultConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + cursor?: string, + syncContext?: Record + ): Promise => { + const kinds = enabledChildKinds(sourceConfig) + const state = decodeCursor(cursor) + + let pageDocuments: ExternalDocument[] = [] + let nextState: VaultCursor | undefined + + if (state.phase === 'matters') { + const { matters, nextPageToken } = await fetchMattersPage( + accessToken, + sourceConfig, + state.mattersPageToken + ) + pageDocuments = await Promise.all(matters.map(matterToDocument)) + + const matterIds = matters + .map((matter) => matter.matterId) + .filter((matterId): matterId is string => Boolean(matterId)) + + if (kinds.length > 0 && matterIds.length > 0) { + nextState = { + phase: 'children', + matterIds, + matterIndex: 0, + nextMattersPageToken: nextPageToken, + } + } else if (nextPageToken) { + nextState = { phase: 'matters', mattersPageToken: nextPageToken } + } + } else { + const matterIds = state.matterIds ?? [] + const batchStart = state.matterIndex ?? 0 + const batch = matterIds.slice(batchStart, batchStart + CHILD_MATTER_BATCH) + + if (batch.length > 0 && kinds.length > 0) { + const children = await fetchChildrenForMatters(accessToken, batch, kinds) + pageDocuments = children.documents + if (children.capped && syncContext) syncContext.listingCapped = true + } + + const nextMatterIndex = batchStart + batch.length + if (nextMatterIndex < matterIds.length && kinds.length > 0) { + nextState = { ...state, matterIndex: nextMatterIndex } + } else if (state.nextMattersPageToken) { + nextState = { phase: 'matters', mattersPageToken: state.nextMattersPageToken } + } + } + + const maxDocuments = readMaxDocuments(sourceConfig.maxDocuments) + const alreadyFetched = (syncContext?.totalDocsFetched as number | undefined) ?? 0 + const { documents, indexableCount, capReached } = takeIndexableWithinCap( + pageDocuments, + () => false, + maxDocuments, + alreadyFetched + ) + if (syncContext) syncContext.totalDocsFetched = alreadyFetched + indexableCount + + const truncated = documents.length < pageDocuments.length + if (syncContext && (truncated || (capReached && nextState !== undefined))) { + syncContext.listingCapped = true + } + + return { + documents, + nextCursor: capReached || !nextState ? undefined : encodeCursor(nextState), + hasMore: !capReached && nextState !== undefined, + } + }, + + getDocument: async ( + accessToken: string, + _sourceConfig: Record, + externalId: string + ): Promise => { + const [kind, first, second] = externalId.split(':') + + try { + if (kind === 'matter') { + const response = await vaultGet( + accessToken, + `${VAULT_API_BASE}/matters/${encodeURIComponent(first)}?view=FULL`, + 'matters.get' + ) + if (response.status === 404 || response.status === 403) return null + if (!response.ok) throw new Error(`Failed to fetch Vault matter: ${response.status}`) + return await matterToDocument((await response.json()) as VaultMatter) + } + + if (kind === 'hold') { + const response = await vaultGet( + accessToken, + `${VAULT_API_BASE}/matters/${encodeURIComponent(first)}/holds/${encodeURIComponent(second)}?view=FULL_HOLD`, + 'matters.holds.get' + ) + if (response.status === 404 || response.status === 403) return null + if (!response.ok) throw new Error(`Failed to fetch Vault hold: ${response.status}`) + return holdToDocument((await response.json()) as VaultHold, first) + } + + if (kind === 'savedQuery') { + const response = await vaultGet( + accessToken, + `${VAULT_API_BASE}/matters/${encodeURIComponent(first)}/savedQueries/${encodeURIComponent(second)}`, + 'matters.savedQueries.get' + ) + if (response.status === 404 || response.status === 403) return null + if (!response.ok) throw new Error(`Failed to fetch Vault saved query: ${response.status}`) + return savedQueryToDocument((await response.json()) as VaultSavedQuery, first) + } + + logger.warn('Unrecognized Google Vault external ID', { externalId }) + return null + } catch (error) { + logger.warn(`Failed to fetch Google Vault document ${externalId}`, { + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const maxDocumentsInput = readString(sourceConfig.maxDocuments) + if (maxDocumentsInput && readMaxDocuments(maxDocumentsInput) === 0) { + return { valid: false, error: 'Max documents must be a positive number' } + } + + const matterState = readString(sourceConfig.matterState) + if (matterState && !['ALL', 'OPEN', 'CLOSED'].includes(matterState)) { + return { valid: false, error: `Unsupported matter state "${matterState}"` } + } + + const matterId = readString(sourceConfig.matterId) + + try { + const url = matterId + ? `${VAULT_API_BASE}/matters/${encodeURIComponent(matterId)}?view=BASIC` + : `${VAULT_API_BASE}/matters?pageSize=1&view=BASIC` + + const response = await fetchWithRetry( + url, + { + method: 'GET', + headers: { + Authorization: `Bearer ${accessToken}`, + Accept: 'application/json', + }, + }, + VALIDATE_RETRY_OPTIONS + ) + + if (!response.ok) { + if (matterId && response.status === 404) { + return { + valid: false, + error: `Matter "${matterId}" not found. Check the matter ID and your Vault permissions.`, + } + } + if (response.status === 401 || response.status === 403) { + return { + valid: false, + error: + 'Google Vault access denied. The account needs Vault privileges and the eDiscovery scope.', + } + } + return { valid: false, error: `Failed to access Google Vault: ${response.status}` } + } + + return { valid: true } + } catch (error) { + return { valid: false, error: toError(error).message || 'Failed to validate configuration' } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + const resourceType = readString(metadata.resourceType) + if (resourceType) result.resourceType = resourceType + + const matterId = readString(metadata.matterId) + if (matterId) result.matterId = matterId + + const state = readString(metadata.state) + if (state) result.state = state + + const corpus = readString(metadata.corpus) + if (corpus) result.corpus = corpus + + const lastModified = parseTagDate(metadata.lastModified) + if (lastModified) result.lastModified = lastModified + + return result + }, +} diff --git a/apps/sim/connectors/google-vault/index.ts b/apps/sim/connectors/google-vault/index.ts new file mode 100644 index 00000000000..11eafbbded2 --- /dev/null +++ b/apps/sim/connectors/google-vault/index.ts @@ -0,0 +1 @@ +export { googleVaultConnector } from '@/connectors/google-vault/google-vault' diff --git a/apps/sim/connectors/google-vault/meta.ts b/apps/sim/connectors/google-vault/meta.ts new file mode 100644 index 00000000000..26dae0e504f --- /dev/null +++ b/apps/sim/connectors/google-vault/meta.ts @@ -0,0 +1,73 @@ +import { GoogleVaultIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const googleVaultConnectorMeta: ConnectorMeta = { + id: 'google_vault', + name: 'Google Vault', + description: 'Sync Google Vault matters, holds, and saved queries into your knowledge base', + version: '1.0.0', + icon: GoogleVaultIcon, + + auth: { + mode: 'oauth', + provider: 'google-vault', + requiredScopes: ['https://www.googleapis.com/auth/ediscovery.readonly'], + }, + + configFields: [ + { + id: 'matterId', + title: 'Matter ID', + type: 'short-input', + placeholder: 'e.g. 12345678901234567890 (leave blank to sync every matter)', + required: false, + description: 'Restrict the sync to a single matter. Leave blank to sync all matters.', + }, + { + id: 'matterState', + title: 'Matter State', + type: 'dropdown', + required: false, + options: [ + { label: 'All states', id: 'ALL' }, + { label: 'Open only', id: 'OPEN' }, + { label: 'Closed only', id: 'CLOSED' }, + ], + }, + { + id: 'includeHolds', + title: 'Include Holds', + type: 'dropdown', + required: false, + options: [ + { label: 'Yes', id: 'true' }, + { label: 'No', id: 'false' }, + ], + }, + { + id: 'includeSavedQueries', + title: 'Include Saved Queries', + type: 'dropdown', + required: false, + options: [ + { label: 'Yes', id: 'true' }, + { label: 'No', id: 'false' }, + ], + }, + { + id: 'maxDocuments', + title: 'Max Documents', + type: 'short-input', + required: false, + placeholder: 'e.g. 500 (default: unlimited)', + }, + ], + + tagDefinitions: [ + { id: 'resourceType', displayName: 'Resource Type', fieldType: 'text' }, + { id: 'matterId', displayName: 'Matter ID', fieldType: 'text' }, + { id: 'state', displayName: 'Matter State', fieldType: 'text' }, + { id: 'corpus', displayName: 'Corpus', fieldType: 'text' }, + { id: 'lastModified', displayName: 'Last Modified', fieldType: 'date' }, + ], +} diff --git a/apps/sim/connectors/microsoft-excel/index.ts b/apps/sim/connectors/microsoft-excel/index.ts new file mode 100644 index 00000000000..57fecb895d1 --- /dev/null +++ b/apps/sim/connectors/microsoft-excel/index.ts @@ -0,0 +1 @@ +export { microsoftExcelConnector } from '@/connectors/microsoft-excel/microsoft-excel' diff --git a/apps/sim/connectors/microsoft-excel/meta.ts b/apps/sim/connectors/microsoft-excel/meta.ts new file mode 100644 index 00000000000..05ac3517d45 --- /dev/null +++ b/apps/sim/connectors/microsoft-excel/meta.ts @@ -0,0 +1,67 @@ +import { MicrosoftExcelIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const microsoftExcelConnectorMeta: ConnectorMeta = { + id: 'microsoft_excel', + name: 'Microsoft Excel', + description: 'Sync workbook sheet data from Microsoft Excel', + version: '1.0.0', + icon: MicrosoftExcelIcon, + + auth: { + mode: 'oauth', + provider: 'microsoft-excel', + requiredScopes: ['Files.ReadWrite'], + }, + + configFields: [ + { + id: 'driveId', + title: 'Drive ID (SharePoint)', + type: 'short-input', + required: false, + placeholder: 'Leave empty for your own OneDrive', + description: + 'The SharePoint document library (drive) ID holding the workbook. Leave empty to use your own OneDrive for Business. Workbooks stored in consumer OneDrive are not supported by the Excel API.', + }, + { + id: 'spreadsheetSelector', + title: 'Workbook', + type: 'selector', + selectorKey: 'microsoft.excel', + canonicalParamId: 'spreadsheetId', + mode: 'basic', + dependsOn: ['driveId'], + placeholder: 'Select a workbook', + required: true, + }, + { + id: 'spreadsheetId', + title: 'Workbook ID', + type: 'short-input', + canonicalParamId: 'spreadsheetId', + mode: 'advanced', + dependsOn: ['driveId'], + placeholder: 'e.g. 01ABC123DEF456', + required: true, + description: 'The Microsoft Graph drive item ID of the .xlsx workbook', + }, + { + id: 'sheetFilter', + title: 'Sheets to Sync', + type: 'dropdown', + required: false, + options: [ + { label: 'All sheets', id: 'all' }, + { label: 'First sheet only', id: 'first' }, + ], + }, + ], + + tagDefinitions: [ + { id: 'sheetTitle', displayName: 'Sheet Name', fieldType: 'text' }, + { id: 'rowCount', displayName: 'Row Count', fieldType: 'number' }, + { id: 'columnCount', displayName: 'Column Count', fieldType: 'number' }, + { id: 'lastModified', displayName: 'Last Modified', fieldType: 'date' }, + ], +} diff --git a/apps/sim/connectors/microsoft-excel/microsoft-excel.ts b/apps/sim/connectors/microsoft-excel/microsoft-excel.ts new file mode 100644 index 00000000000..2eb4761aa0a --- /dev/null +++ b/apps/sim/connectors/microsoft-excel/microsoft-excel.ts @@ -0,0 +1,669 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import type { RetryOptions } from '@/lib/knowledge/documents/utils' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { microsoftExcelConnectorMeta } from '@/connectors/microsoft-excel/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { markSkipped, parseTagDate, readBodyWithLimit } from '@/connectors/utils' +import type { ExcelCellValue } from '@/tools/microsoft_excel/types' +import { + escapeODataString, + getItemBasePath, + parseGraphErrorMessage, + trimTrailingEmptyRowsAndColumns, +} from '@/tools/microsoft_excel/utils' + +const logger = createLogger('MicrosoftExcelConnector') + +/** + * Separator between the workbook drive-item ID and the worksheet ID inside an + * `externalId`. Graph worksheet IDs are brace-wrapped GUIDs (`{FC03…A0}`), so they + * never contain this token. + */ +const SHEET_SEPARATOR = '__sheet__' + +/** + * Version token folded into `contentHash`. The hash is metadata-based (workbook + * `lastModifiedDateTime`), so a change to *how* cell content is rendered would + * otherwise leave every already-indexed worksheet on the old rendering until the + * workbook itself is edited. Bump this whenever the indexed text changes shape. + * `v2` = displayed `text` values instead of raw serial-number `values`. + */ +const CONTENT_FORMAT_VERSION = 'v2' + +/** + * Hard ceiling on the number of worksheets listed from a single workbook. Excel + * allows far more sheets than any knowledge base should absorb in one sync, so the + * listing stops here and flags `listingCapped` to keep deletion reconciliation from + * purging the sheets past the cap. + */ +const MAX_WORKSHEETS = 500 + +/** Maximum rows read from a single worksheet's used range. */ +const MAX_ROWS = 5000 + +/** Maximum columns read from a single worksheet's used range. */ +const MAX_COLUMNS = 200 + +/** + * Maximum cells read from a single worksheet. A workbook can declare a used range + * of millions of cells, and Graph serializes every one of them into the JSON body, + * so the row/column caps alone are not enough — the row cap is tightened further + * until the rectangle fits this budget. + */ +const MAX_CELLS = 200_000 + +/** + * Byte ceiling on a single range response. The caps above bound the *requested* + * rectangle, but individual cells carry arbitrary user text, so the body is read + * through a streaming limiter and abandoned rather than buffered if it overruns. + */ +const MAX_RANGE_RESPONSE_BYTES = 16 * 1024 * 1024 + +/** + * Byte ceiling on the used-range *metadata* response. `$select=address` keeps that + * body to a few hundred bytes, but Graph documents that unsupported query + * parameters can "fail silently" (https://learn.microsoft.com/en-us/graph/query-parameters + * — "Error handling for query parameters"), and the `usedRange` reference page — unlike + * `worksheets` — has no "Optional query parameters" section promising `$select` support. + * If the projection is ever dropped, Graph serializes the whole grid (values, text, + * formulas, numberFormat, valueTypes) into this response, so it is read through the + * streaming limiter too: the dimensions-only design degrades to a skipped worksheet + * rather than an unbounded buffer. + */ +const MAX_USED_RANGE_RESPONSE_BYTES = 1024 * 1024 + +interface Worksheet { + id: string + name: string + position: number + visibility?: string +} + +interface WorksheetListResponse { + value?: Worksheet[] +} + +interface WorkbookItem { + id: string + name?: string + webUrl?: string + lastModifiedDateTime?: string +} + +interface UsedRangeMetadata { + address?: string +} + +interface RangeValues { + address?: string + text?: string[][] + values?: ExcelCellValue[][] +} + +/** A1-style rectangle, all bounds 1-based and inclusive. */ +interface CellRect { + startRow: number + startColumn: number + endRow: number + endColumn: number +} + +/** Converts an A1 column label (`A`, `Z`, `AA`) to its 1-based index. */ +export function columnLabelToIndex(label: string): number { + let index = 0 + for (const char of label.toUpperCase()) { + index = index * 26 + (char.charCodeAt(0) - 64) + } + return index +} + +/** Converts a 1-based column index to its A1 label (`1` → `A`, `27` → `AA`). */ +export function columnIndexToLabel(index: number): string { + let remaining = index + let label = '' + while (remaining > 0) { + const rest = (remaining - 1) % 26 + label = String.fromCharCode(65 + rest) + label + remaining = Math.floor((remaining - 1) / 26) + } + return label +} + +/** + * Parses a Graph range address (`Sheet1!B2:F400`, `'My Sheet'!A1`) into its + * rectangle. The sheet-name prefix is split on the LAST `!` because Excel permits + * `!` inside a quoted sheet name. Returns `null` when the address is not an + * absolute A1 rectangle we can bound. + */ +export function parseRangeAddress(address: string): CellRect | null { + const bangIndex = address.lastIndexOf('!') + const local = bangIndex === -1 ? address : address.slice(bangIndex + 1) + const cellPattern = /^\$?([A-Za-z]+)\$?(\d+)$/ + + const [startCell, endCell] = local.split(':') + const start = startCell?.match(cellPattern) + if (!start) return null + + const end = endCell ? endCell.match(cellPattern) : start + if (!end) return null + + return { + startRow: Number(start[2]), + startColumn: columnLabelToIndex(start[1]), + endRow: Number(end[2]), + endColumn: columnLabelToIndex(end[1]), + } +} + +/** + * Shrinks a used-range rectangle to the connector's row, column, and cell caps. + * The rectangle is always anchored at the used range's top-left cell so the header + * row survives; the column cap is applied first, then the row cap is tightened + * further until the remaining rectangle fits {@link MAX_CELLS}. + */ +export function capRect(rect: CellRect): { rect: CellRect; capped: boolean } { + const cappedEndColumn = Math.min(rect.endColumn, rect.startColumn + MAX_COLUMNS - 1) + const columns = cappedEndColumn - rect.startColumn + 1 + const rowBudget = Math.max(1, Math.min(MAX_ROWS, Math.floor(MAX_CELLS / columns))) + const cappedEndRow = Math.min(rect.endRow, rect.startRow + rowBudget - 1) + + return { + rect: { ...rect, endColumn: cappedEndColumn, endRow: cappedEndRow }, + capped: cappedEndColumn < rect.endColumn || cappedEndRow < rect.endRow, + } +} + +/** Renders a rectangle as a sheet-relative A1 address (`B2:F400`). */ +function formatRect(rect: CellRect): string { + const start = `${columnIndexToLabel(rect.startColumn)}${rect.startRow}` + const end = `${columnIndexToLabel(rect.endColumn)}${rect.endRow}` + return start === end ? start : `${start}:${end}` +} + +/** Normalizes a Graph cell value to the plain string used in indexed content. */ +function cellToString(value: ExcelCellValue): string { + if (value === null || value === undefined) return '' + return String(value) +} + +/** + * Formats worksheet rows into an LLM-friendly text representation, labelling each + * row by index and each cell by its header name. Mirrors the Google Sheets + * connector so both spreadsheet sources chunk identically. + */ +export function formatSheetContent(headers: string[], rows: ExcelCellValue[][]): string { + if (headers.length === 0) return '' + + const lines: string[] = [] + for (let i = 0; i < rows.length; i++) { + const row = rows[i] ?? [] + lines.push(`Row ${i + 1}:`) + for (let j = 0; j < headers.length; j++) { + lines.push(` ${headers[j]}: ${cellToString(row[j])}`) + } + lines.push('') + } + + return lines.join('\n').trim() +} + +/** Builds the Graph URL for a worksheet, addressing it by name (Graph accepts id or name). */ +function worksheetUrl(basePath: string, sheetName: string): string { + return `${basePath}/workbook/worksheets('${encodeURIComponent(escapeODataString(sheetName))}')` +} + +/** Throws a Graph-formatted error for a failed response. */ +async function graphError(response: Response, context: string): Promise { + const body = await response.text().catch(() => '') + const detail = parseGraphErrorMessage(response.status, response.statusText, body) + throw new Error(`${context}: ${detail}`) +} + +/** Fetches the workbook drive item (name, webUrl, lastModifiedDateTime). */ +async function fetchWorkbookItem( + accessToken: string, + basePath: string, + retryOptions?: RetryOptions +): Promise { + const response = await fetchWithRetry( + `${basePath}?$select=id,name,webUrl,lastModifiedDateTime`, + { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}`, Accept: 'application/json' }, + }, + retryOptions + ) + + if (response.status === 404 || response.status === 410) return null + if (!response.ok) await graphError(response, 'Failed to fetch workbook') + + return (await response.json()) as WorkbookItem +} + +/** Lists the workbook's worksheets in tab order. */ +async function fetchWorksheets(accessToken: string, basePath: string): Promise { + const response = await fetchWithRetry( + `${basePath}/workbook/worksheets?$select=id,name,position,visibility&$orderby=position`, + { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}`, Accept: 'application/json' }, + } + ) + + if (!response.ok) await graphError(response, 'Failed to list worksheets') + + const data = (await response.json()) as WorksheetListResponse + return data.value ?? [] +} + +/** + * Fetches the worksheet's used-range dimensions WITHOUT its values. + * `$select` keeps the response to a few bytes, so the connector can decide how much + * of a potentially enormous sheet to read before requesting any cell data. + */ +async function fetchUsedRangeMetadata( + accessToken: string, + basePath: string, + sheetName: string +): Promise { + const url = `${worksheetUrl(basePath, sheetName)}/usedRange(valuesOnly=true)?$select=address` + const response = await fetchWithRetry(url, { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}`, Accept: 'application/json' }, + }) + + if (response.status === 404) return null + if (!response.ok) await graphError(response, `Failed to read used range for "${sheetName}"`) + + const buffer = await readBodyWithLimit(response, MAX_USED_RANGE_RESPONSE_BYTES) + if (!buffer) throw new RangeTooLargeError(MAX_USED_RANGE_RESPONSE_BYTES) + + return JSON.parse(buffer.toString('utf8')) as UsedRangeMetadata +} + +/** + * Raised when a range response exceeds its byte ceiling. The body is abandoned + * mid-stream rather than buffered, so the worksheet surfaces as a skipped document + * instead of pulling an unbounded payload into memory. + */ +class RangeTooLargeError extends Error { + constructor(maxBytes: number) { + super(`Worksheet range response exceeds ${maxBytes} bytes`) + this.name = 'RangeTooLargeError' + } +} + +/** + * Fetches the capped rectangle of cell values from a worksheet. + * + * Both `text` and `values` are projected. `values` carries the *raw* cell values, so a + * date or currency cell comes back as its underlying serial number (`42019`, not + * `1/15/2015`) — useless for retrieval. `text` carries the displayed strings and, + * per the Range reference, "doesn't depend on the cell width. The # sign substitution + * that happens in Excel UI doesn't affect the text value returned by the API" + * (https://learn.microsoft.com/en-us/graph/api/resources/range), so it never degrades + * to `#######`. This matches the Google Sheets connector's `FORMATTED_VALUE`. + * `values` is kept only as a fallback for the rows `text` does not cover. + */ +async function fetchRangeValues( + accessToken: string, + basePath: string, + sheetName: string, + address: string +): Promise { + const url = `${worksheetUrl(basePath, sheetName)}/range(address='${encodeURIComponent(address)}')?$select=address,text,values` + const response = await fetchWithRetry(url, { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}`, Accept: 'application/json' }, + }) + + if (!response.ok) await graphError(response, `Failed to read range for "${sheetName}"`) + + const buffer = await readBodyWithLimit(response, MAX_RANGE_RESPONSE_BYTES) + if (!buffer) throw new RangeTooLargeError(MAX_RANGE_RESPONSE_BYTES) + + return JSON.parse(buffer.toString('utf8')) as RangeValues +} + +interface WorkbookSnapshot { + workbook: WorkbookItem | null + worksheets: Worksheet[] +} + +/** + * Loads the workbook drive item and its worksheet list, memoizing the in-flight promise + * on `syncContext` for the duration of the sync run. The sync engine hydrates deferred + * documents concurrently, so without this every worksheet in the workbook would repeat + * both calls — 2N Graph requests against an API that throttles aggressively. A rejected + * promise is evicted so a transient failure does not poison the rest of the run. + */ +async function loadWorkbookSnapshot( + accessToken: string, + basePath: string, + spreadsheetId: string, + syncContext?: Record +): Promise { + const cacheKey = `workbookSnapshot:${spreadsheetId}` + const cached = syncContext?.[cacheKey] as Promise | undefined + if (cached) return cached + + const pending = (async (): Promise => { + const workbook = await fetchWorkbookItem(accessToken, basePath) + if (!workbook) return { workbook: null, worksheets: [] } + return { workbook, worksheets: await fetchWorksheets(accessToken, basePath) } + })() + + if (syncContext) { + syncContext[cacheKey] = pending + pending.catch(() => { + if (syncContext[cacheKey] === pending) delete syncContext[cacheKey] + }) + } + + return pending +} + +/** Composes the stable external ID for one worksheet inside a workbook. */ +function buildExternalId(spreadsheetId: string, worksheetId: string): string { + return `${spreadsheetId}${SHEET_SEPARATOR}${worksheetId}` +} + +/** Splits an external ID back into its workbook and worksheet IDs. */ +export function parseExternalId( + externalId: string +): { spreadsheetId: string; worksheetId: string } | null { + const index = externalId.indexOf(SHEET_SEPARATOR) + if (index <= 0) return null + + const spreadsheetId = externalId.slice(0, index) + const worksheetId = externalId.slice(index + SHEET_SEPARATOR.length) + if (!spreadsheetId || !worksheetId) return null + + return { spreadsheetId, worksheetId } +} + +/** + * Builds the deferred listing stub for one worksheet. Used by both `listDocuments` + * and `getDocument` so the metadata-based `contentHash` is byte-identical on both + * paths — the sync engine compares them directly to decide what to re-index. + */ +function sheetToStub( + spreadsheetId: string, + workbook: WorkbookItem, + sheet: Worksheet +): ExternalDocument { + const workbookTitle = workbook.name ?? 'Workbook' + return { + externalId: buildExternalId(spreadsheetId, sheet.id), + title: `${workbookTitle} - ${sheet.name}`, + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: workbook.webUrl, + contentHash: `microsoft_excel:${CONTENT_FORMAT_VERSION}:${spreadsheetId}:${sheet.id}:${workbook.lastModifiedDateTime ?? ''}`, + metadata: { + spreadsheetId, + workbookName: workbookTitle, + sheetTitle: sheet.name, + worksheetId: sheet.id, + position: sheet.position, + visibility: sheet.visibility, + lastModifiedDateTime: workbook.lastModifiedDateTime, + }, + } +} + +/** Resolves the workbook's Graph base path from the connector's source config. */ +function resolveBasePath(sourceConfig: Record): { + spreadsheetId: string + basePath: string +} { + const spreadsheetId = + typeof sourceConfig.spreadsheetId === 'string' ? sourceConfig.spreadsheetId.trim() : '' + if (!spreadsheetId) { + throw new Error('Workbook ID is required') + } + + const driveId = typeof sourceConfig.driveId === 'string' ? sourceConfig.driveId.trim() : '' + return { spreadsheetId, basePath: getItemBasePath(spreadsheetId, driveId || undefined) } +} + +export const microsoftExcelConnector: ConnectorConfig = { + ...microsoftExcelConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + _cursor?: string, + syncContext?: Record + ): Promise => { + const { spreadsheetId, basePath } = resolveBasePath(sourceConfig) + + const { workbook, worksheets } = await loadWorkbookSnapshot( + accessToken, + basePath, + spreadsheetId, + syncContext + ) + + /** + * A 404/410 means the drive item is gone for good, so the listing is genuinely + * empty and reconciliation should purge the workbook's sheets. `listingCapped` + * is deliberately NOT set here — a permissions failure surfaces as 401/403 and + * throws from `fetchWorkbookItem` instead. + */ + if (!workbook) { + logger.info('Workbook not found; listing no documents', { spreadsheetId }) + return { documents: [], hasMore: false } + } + + const sheetFilter = typeof sourceConfig.sheetFilter === 'string' ? sourceConfig.sheetFilter : '' + const scoped = sheetFilter === 'first' ? worksheets.slice(0, 1) : worksheets + + const selected = scoped.slice(0, MAX_WORKSHEETS) + if (selected.length < scoped.length && syncContext) { + logger.warn('Worksheet listing truncated by the connector cap', { + spreadsheetId, + total: scoped.length, + cap: MAX_WORKSHEETS, + }) + syncContext.listingCapped = true + } + + logger.info('Listing Microsoft Excel worksheets', { + spreadsheetId, + workbookName: workbook.name, + sheetCount: selected.length, + }) + + return { + documents: selected.map((sheet) => sheetToStub(spreadsheetId, workbook, sheet)), + hasMore: false, + } + }, + + getDocument: async ( + accessToken: string, + sourceConfig: Record, + externalId: string, + syncContext?: Record + ): Promise => { + const parsed = parseExternalId(externalId) + if (!parsed) { + logger.warn('Invalid external ID format', { externalId }) + return null + } + + const driveId = typeof sourceConfig.driveId === 'string' ? sourceConfig.driveId.trim() : '' + const basePath = getItemBasePath(parsed.spreadsheetId, driveId || undefined) + + const { workbook, worksheets } = await loadWorkbookSnapshot( + accessToken, + basePath, + parsed.spreadsheetId, + syncContext + ) + if (!workbook) { + logger.info('Workbook not found', { spreadsheetId: parsed.spreadsheetId }) + return null + } + + const sheet = worksheets.find((candidate) => candidate.id === parsed.worksheetId) + if (!sheet) { + logger.info('Worksheet no longer exists in the workbook', { externalId }) + return null + } + + const stub = sheetToStub(parsed.spreadsheetId, workbook, sheet) + + try { + const usedRange = await fetchUsedRangeMetadata(accessToken, basePath, sheet.name) + const address = usedRange?.address + if (!address) return null + + const rect = parseRangeAddress(address) + if (!rect) { + logger.warn('Unparseable used-range address', { externalId, address }) + return null + } + + const { rect: capped, capped: wasCapped } = capRect(rect) + if (wasCapped) { + logger.warn('Worksheet content truncated by the connector cell caps', { + externalId, + usedRangeAddress: address, + indexedRangeAddress: formatRect(capped), + }) + } + const range = await fetchRangeValues(accessToken, basePath, sheet.name, formatRect(capped)) + const values = trimTrailingEmptyRowsAndColumns(range.text ?? range.values ?? []) + + if (values.length < 2) return null + + const headers = values[0].map((header, index) => { + const label = cellToString(header).trim() + return label || `Column ${index + 1}` + }) + + const body = formatSheetContent(headers, values.slice(1)) + if (!body.trim()) return null + + const content = wasCapped + ? `${body}\n\n[Truncated: only ${formatRect(capped)} of ${address} was indexed]` + : body + + return { + ...stub, + content, + contentDeferred: false, + metadata: { + ...stub.metadata, + rowCount: values.length - 1, + columnCount: headers.length, + usedRangeAddress: address, + indexedRangeAddress: formatRect(capped), + truncated: wasCapped, + }, + } + } catch (error) { + if (error instanceof RangeTooLargeError) { + logger.info('Skipping oversized worksheet range', { externalId }) + return markSkipped(stub, 'Worksheet exceeds the connector size limit and was not indexed') + } + logger.warn('Failed to extract content from worksheet', { + externalId, + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const sheetFilter = sourceConfig.sheetFilter + if ( + sheetFilter !== undefined && + sheetFilter !== '' && + sheetFilter !== 'all' && + sheetFilter !== 'first' + ) { + return { valid: false, error: 'Sheets to Sync must be either "all" or "first"' } + } + + let basePath: string + try { + basePath = resolveBasePath(sourceConfig).basePath + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Workbook ID is required') } + } + + try { + const workbook = await fetchWorkbookItem(accessToken, basePath, VALIDATE_RETRY_OPTIONS) + if (!workbook) { + return { + valid: false, + error: 'Workbook not found. Check the workbook ID and that your account can access it.', + } + } + + const response = await fetchWithRetry( + `${basePath}/workbook/worksheets?$select=id&$top=1`, + { + method: 'GET', + headers: { Authorization: `Bearer ${accessToken}`, Accept: 'application/json' }, + }, + VALIDATE_RETRY_OPTIONS + ) + + if (!response.ok) { + if (response.status === 403) { + return { + valid: false, + error: 'Access denied. Ensure the workbook is shared with your Microsoft account.', + } + } + if (response.status === 404) { + return { + valid: false, + error: 'This file is not an Excel workbook, or it no longer exists.', + } + } + const body = await response.text().catch(() => '') + return { + valid: false, + error: parseGraphErrorMessage(response.status, response.statusText, body), + } + } + + return { valid: true } + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Failed to validate configuration') } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + if (typeof metadata.sheetTitle === 'string') { + result.sheetTitle = metadata.sheetTitle + } + + if (typeof metadata.rowCount === 'number') { + result.rowCount = metadata.rowCount + } + + if (typeof metadata.columnCount === 'number') { + result.columnCount = metadata.columnCount + } + + const lastModified = parseTagDate(metadata.lastModifiedDateTime) + if (lastModified) { + result.lastModified = lastModified + } + + return result + }, +} diff --git a/apps/sim/connectors/mintlify/index.ts b/apps/sim/connectors/mintlify/index.ts new file mode 100644 index 00000000000..80880199d18 --- /dev/null +++ b/apps/sim/connectors/mintlify/index.ts @@ -0,0 +1 @@ +export { mintlifyConnector } from '@/connectors/mintlify/mintlify' diff --git a/apps/sim/connectors/mintlify/meta.ts b/apps/sim/connectors/mintlify/meta.ts new file mode 100644 index 00000000000..7800875a90d --- /dev/null +++ b/apps/sim/connectors/mintlify/meta.ts @@ -0,0 +1,63 @@ +import { MintlifyIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +/** Default ceiling on indexed pages per sync when `maxPages` is not configured. */ +export const DEFAULT_MAX_PAGES = 500 + +/** Hard ceiling on the configurable `maxPages` value. */ +export const MAX_PAGES_LIMIT = 5000 + +export const mintlifyConnectorMeta: ConnectorMeta = { + id: 'mintlify', + name: 'Mintlify', + description: 'Sync pages from a hosted Mintlify documentation site', + version: '1.0.0', + icon: MintlifyIcon, + + /** + * The key is sent as a bearer token to the documentation host itself, not to + * `api.mintlify.com` — Mintlify's own API keys authenticate the dashboard and + * assistant APIs, which have no page-listing endpoint. It is therefore only + * useful for a site fronted by a proxy that accepts a bearer token, so it is + * declared optional and public sites can be connected with no key at all. + */ + auth: { + mode: 'apiKey', + label: 'Access Token', + placeholder: 'Only needed if your docs site requires a bearer token', + optional: true, + }, + + configFields: [ + { + id: 'siteUrl', + title: 'Documentation Site URL', + type: 'short-input', + placeholder: 'https://docs.yourcompany.com', + required: true, + description: + 'Base URL of your published Mintlify site. Pages are discovered from the llms.txt file Mintlify hosts there.', + }, + { + id: 'pathPrefix', + title: 'Path Prefix', + type: 'short-input', + placeholder: 'e.g. /guides', + required: false, + description: 'Only sync pages whose path starts with this prefix (leave empty for all pages)', + }, + { + id: 'maxPages', + title: 'Max Pages', + type: 'short-input', + placeholder: `e.g. 200 (default: ${DEFAULT_MAX_PAGES}, max: ${MAX_PAGES_LIMIT})`, + required: false, + description: 'Maximum number of documentation pages to index', + }, + ], + + tagDefinitions: [ + { id: 'section', displayName: 'Section', fieldType: 'text' }, + { id: 'description', displayName: 'Description', fieldType: 'text' }, + ], +} diff --git a/apps/sim/connectors/mintlify/mintlify.ts b/apps/sim/connectors/mintlify/mintlify.ts new file mode 100644 index 00000000000..1bb49e4ed07 --- /dev/null +++ b/apps/sim/connectors/mintlify/mintlify.ts @@ -0,0 +1,620 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import { validateExternalUrl } from '@/lib/core/security/input-validation' +import { + type SecureFetchRetryOptions, + secureFetchWithRetry, +} from '@/lib/knowledge/documents/secure-fetch.server' +import { VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { + DEFAULT_MAX_PAGES, + MAX_PAGES_LIMIT, + mintlifyConnectorMeta, +} from '@/connectors/mintlify/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { computeContentHash, htmlToPlainText } from '@/connectors/utils' + +const logger = createLogger('MintlifyConnector') + +/** Documents returned per `listDocuments` call. */ +const DOCS_PER_PAGE = 50 + +/** Byte cap for the page-index file (`llms.txt` / `sitemap.xml`) of a user-supplied host. */ +const INDEX_MAX_BYTES = 5 * 1024 * 1024 + +/** Byte cap for a single documentation page body. */ +const PAGE_MAX_BYTES = 1024 * 1024 + +/** Child sitemaps followed from a ``, bounding a hostile or huge index. */ +const MAX_CHILD_SITEMAPS = 20 + +/** A page discovered from the site's index file. */ +interface MintlifyPageLink { + /** Site-absolute path without the `.md` extension, e.g. `/docs/quickstart`. */ + path: string + title: string + description?: string + /** Nearest markdown heading above the link in `llms.txt`. */ + section?: string +} + +interface MintlifySite { + /** Scheme + host of the configured site, e.g. `https://docs.example.com`. */ + origin: string + /** Configured base URL with any trailing slash removed, e.g. `https://docs.example.com/docs`. */ + baseUrl: string + /** Hostname with a leading `www.` removed, used for same-site link filtering. */ + hostKey: string + /** Path portion of `baseUrl` without a trailing slash, e.g. `/docs`, or `''` at the origin root. */ + basePath: string +} + +/** + * Normalizes the configured documentation site URL and runs an early structural + * SSRF check via the shared `validateExternalUrl` policy. + * + * The authoritative SSRF boundary is enforced at request time: every site request + * goes through {@link secureFetchWithRetry}, which resolves DNS, re-checks the + * resolved IP, and pins the connection to it — closing the DNS-rebinding gap a + * synchronous string check cannot. + */ +function resolveSite(rawUrl: string | undefined): MintlifySite { + let url = (rawUrl || '').trim().replace(/\/+$/, '') + if (!url) { + throw new Error('Documentation site URL is required') + } + if (!url.startsWith('https://') && !url.startsWith('http://')) { + url = `https://${url}` + } + + const validation = validateExternalUrl(url, 'siteUrl') + if (!validation.isValid) { + throw new Error(validation.error || 'Invalid documentation site URL') + } + + const parsed = new URL(url) + return { + origin: parsed.origin, + baseUrl: url, + hostKey: parsed.hostname.replace(/^www\./, ''), + basePath: parsed.pathname.replace(/\/+$/, ''), + } +} + +/** Resolves the configured page cap, clamped to {@link MAX_PAGES_LIMIT}. */ +function resolveMaxPages(value: unknown): number { + const parsed = Number(value) + if (!Number.isFinite(parsed) || parsed <= 0) return DEFAULT_MAX_PAGES + return Math.min(Math.floor(parsed), MAX_PAGES_LIMIT) +} + +/** Normalizes an optional path prefix filter to a leading-slash form, or `''`. */ +function resolvePathPrefix(value: unknown): string { + const prefix = typeof value === 'string' ? value.trim() : '' + if (!prefix) return '' + return prefix.startsWith('/') ? prefix : `/${prefix}` +} + +/** Bearer headers, omitted when no key is configured (public sites need none). */ +function siteHeaders(accessToken: string, accept: string): Record { + const headers: Record = { Accept: accept } + if (accessToken?.trim()) { + headers.Authorization = `Bearer ${accessToken.trim()}` + } + return headers +} + +/** + * Fetches a text resource from the user-supplied documentation host. + * + * `stripAuthOnRedirect` keeps the configured key from being forwarded to a + * redirect target on another origin. + */ +async function fetchSiteText( + url: string, + accessToken: string, + accept: string, + maxBytes: number, + retryOptions?: SecureFetchRetryOptions +): Promise<{ body: string; contentType: string } | null> { + const response = await secureFetchWithRetry( + url, + { + method: 'GET', + headers: siteHeaders(accessToken, accept), + stripAuthOnRedirect: true, + }, + { ...retryOptions, maxResponseBytes: maxBytes } + ) + + if (!response.ok) { + if (response.status === 404) return null + throw new Error(`Mintlify site returned status ${response.status} for ${url}`) + } + + return { + body: await response.text(), + contentType: response.headers.get('content-type') ?? '', + } +} + +/** + * Converts a discovered link into a site-absolute path, or `null` when it points + * off-site, is not an http(s) URL, or is not a documentation page. A leading + * `www.` is ignored on both sides because Mintlify sites commonly redirect + * between the apex and `www` host while emitting the canonical one in `llms.txt`. + */ +function toPagePath(rawHref: string, site: MintlifySite): string | null { + let parsed: URL + try { + parsed = new URL(rawHref, `${site.origin}/`) + } catch { + return null + } + + if (parsed.protocol !== 'https:' && parsed.protocol !== 'http:') return null + if (parsed.hostname.replace(/^www\./, '') !== site.hostKey) return null + + let path = parsed.pathname.replace(/\.md$/i, '').replace(/\/+$/, '') + if (!path) path = '/' + if (/\.(xml|json|ya?ml|txt|png|jpe?g|svg|gif|webp|pdf|zip|css|js)$/i.test(path)) return null + return path +} + +/** Derives a human-readable title from a page path, e.g. `/docs/api-keys` → `Api keys`. */ +function titleFromPath(path: string): string { + const slug = path.split('/').filter(Boolean).pop() || 'Index' + const words = slug.replace(/[-_]+/g, ' ').trim() + return words.charAt(0).toUpperCase() + words.slice(1) +} + +const LLMS_LINK_PATTERN = /^\s*[-*]?\s*\[([^\]]+)\]\(([^)\s]+)\)\s*(.*)$/ +const MARKDOWN_HEADING_PATTERN = /^(#{1,6})\s+(.+?)\s*$/ + +/** + * Strips the separator between a link and its description. `llms.txt` is a loose + * convention, not a spec: Mintlify, Trigger.dev, and Resend emit + * `](url.md): description` while Anthropic emits `](url.md) - description`, so + * both a colon and a dash lead-in are removed. + */ +const LLMS_DESCRIPTION_SEPARATOR = /^[\s:\-–—]+/ + +/** + * Parses the markdown link list Mintlify publishes at `/llms.txt`. Each entry is + * a `- [Title](https://site/path.md): description` line, grouped under markdown + * headings that name the navigation section. + */ +function parseLlmsTxt(body: string, site: MintlifySite): MintlifyPageLink[] { + const pages: MintlifyPageLink[] = [] + const seen = new Set() + let section: string | undefined + + for (const line of body.split('\n')) { + const heading = MARKDOWN_HEADING_PATTERN.exec(line) + if (heading) { + section = heading[2] + continue + } + + const match = LLMS_LINK_PATTERN.exec(line) + if (!match) continue + + const path = toPagePath(match[2], site) + if (!path || seen.has(path)) continue + seen.add(path) + + const description = match[3]?.replace(LLMS_DESCRIPTION_SEPARATOR, '').trim() + pages.push({ + path, + title: match[1].trim() || titleFromPath(path), + description: description || undefined, + section, + }) + } + + return pages +} + +const SITEMAP_LOC_PATTERN = /\s*([^<\s]+)\s*<\/loc>/gi +const SITEMAP_INDEX_PATTERN = /]/i + +/** Extracts every `` value from a sitemap document. */ +function sitemapLocations(body: string): string[] { + return [...body.matchAll(SITEMAP_LOC_PATTERN)].map((match) => match[1]) +} + +/** Parses `` page entries from a sitemap, used when the site has no `llms.txt`. */ +function parseSitemap(locations: string[], site: MintlifySite): MintlifyPageLink[] { + const pages: MintlifyPageLink[] = [] + const seen = new Set() + + for (const location of locations) { + const path = toPagePath(location, site) + if (!path || seen.has(path)) continue + seen.add(path) + pages.push({ path, title: titleFromPath(path) }) + } + + return pages +} + +/** + * Resolves a same-site child-sitemap URL, or `null` when it points off-host. + * `toPagePath` cannot be reused here because it rejects `.xml` by design. + */ +function sameSiteSitemapUrl(rawHref: string, site: MintlifySite): string | null { + try { + const parsed = new URL(rawHref, `${site.origin}/`) + if (parsed.protocol !== 'https:' && parsed.protocol !== 'http:') return null + if (parsed.hostname.replace(/^www\./, '') !== site.hostKey) return null + return parsed.toString() + } catch { + return null + } +} + +/** + * Reads the site's sitemap, following a `` into its child sitemaps. + * Without this a site that splits its sitemap would parse to zero pages — every + * `` would be a `.xml` URL that {@link toPagePath} discards. + */ +async function discoverFromSitemap( + site: MintlifySite, + accessToken: string, + retryOptions?: SecureFetchRetryOptions +): Promise { + const root = await fetchSiteText( + `${site.baseUrl}/sitemap.xml`, + accessToken, + 'application/xml', + INDEX_MAX_BYTES, + retryOptions + ) + if (!root) return [] + + if (!SITEMAP_INDEX_PATTERN.test(root.body)) { + return parseSitemap(sitemapLocations(root.body), site) + } + + const allChildUrls = sitemapLocations(root.body) + .map((location) => sameSiteSitemapUrl(location, site)) + .filter((url): url is string => Boolean(url)) + + if (allChildUrls.length > MAX_CHILD_SITEMAPS) { + throw new Error( + `Sitemap index at ${site.baseUrl}/sitemap.xml lists ${allChildUrls.length} child sitemaps (limit ${MAX_CHILD_SITEMAPS})` + ) + } + + logger.info('Following Mintlify sitemap index', { children: allChildUrls.length }) + + const locations: string[] = [] + for (const childUrl of allChildUrls) { + const child = await fetchSiteText( + childUrl, + accessToken, + 'application/xml', + INDEX_MAX_BYTES, + retryOptions + ) + /** + * A missing child sitemap is fatal rather than skipped. `fetchSiteText` + * maps 404 to `null`, so continuing here would hand the sync engine a + * listing short by one child's worth of pages — a partial listing the + * engine cannot distinguish from genuine deletions, which reconciles every + * page of that child out of the knowledge base. + */ + if (!child) { + throw new Error(`Child sitemap ${childUrl} listed in the sitemap index is unavailable`) + } + locations.push(...sitemapLocations(child.body)) + } + + return parseSitemap(locations, site) +} + +/** + * Restricts discovered pages to the configured base path. + * + * Applied to every discovery source, not just the origin-level index: a sub-path + * site's `sitemap.xml` is equally free to enumerate the whole host, and a listing + * that reaches outside the configured scope indexes pages the user did not ask + * for. A no-op when the site is configured at the host root. + */ +function withinBasePath(pages: MintlifyPageLink[], site: MintlifySite): MintlifyPageLink[] { + if (!site.basePath) return pages + return pages.filter( + (page) => page.path === site.basePath || page.path.startsWith(`${site.basePath}/`) + ) +} + +/** + * Discovers every page of the site. + * + * Mintlify's REST API has no page-enumeration endpoint — its discovery API only + * supports query-driven search and path-addressed page reads — so the index files + * Mintlify auto-publishes are the enumeration path. `/llms.txt` is preferred + * because it carries titles, descriptions, and section grouping; `/sitemap.xml` + * is the fallback for sites that disabled it. + */ +async function discoverPages( + site: MintlifySite, + accessToken: string, + retryOptions?: SecureFetchRetryOptions +): Promise { + /** + * The origin-level index is only consulted for a site published at the host + * root. On a sub-path site (`https://example.com/docs`) it describes the whole + * host — the marketing site — and its handful of incidental `/docs` links is a + * far worse listing than the sitemap's. Trusting it there produced a listing of + * 13 pages for `trigger.dev/docs` against the 306 its own index publishes, + * which the sync engine would reconcile as ~293 deletions. + */ + const indexUrls = [ + ...new Set([ + `${site.baseUrl}/llms.txt`, + `${site.baseUrl}/.well-known/llms.txt`, + ...(site.basePath ? [] : [`${site.origin}/llms.txt`]), + ]), + ] + + for (const indexUrl of indexUrls) { + const result = await fetchSiteText( + indexUrl, + accessToken, + 'text/plain', + INDEX_MAX_BYTES, + retryOptions + ) + if (!result) continue + const pages = withinBasePath(parseLlmsTxt(result.body, site), site) + if (pages.length > 0) return pages + } + + return withinBasePath(await discoverFromSitemap(site, accessToken, retryOptions), site) +} + +/** Elements whose *text content* is markup/data, never prose. */ +const NON_CONTENT_ELEMENT_PATTERN = + /<(script|style|noscript|template|svg|head)\b[^>]*>[\s\S]*?<\/\1>/gi + +/** `
` / `
` body, the prose region of a rendered docs page. */ +const MAIN_REGION_PATTERN = /<(main|article)\b[^>]*>([\s\S]*)<\/\1>/i + +const HTML_COMMENT_PATTERN = //g + +/** + * Extracts prose from a full HTML document. + * + * The shared {@link htmlToPlainText} only removes tags, so a script's *contents* + * survive as text. Every other connector feeds it fragment HTML from an API + * field, where that is fine; this connector is the only one that hands it a whole + * server-rendered page. On a Next.js-rendered docs site that is catastrophic — + * `docs.sim.ai/introduction` yields 310KB of "text" of which 294KB is the RSC + * flight payload and site-wide navigation JSON, which both swamps the real page + * content in retrieval and makes every page's extraction near-identical. + * + * So non-content elements are dropped whole, and the `
`/`
` region + * is preferred over the full document to shed chrome (nav, sidebar, footer). + */ +function htmlPageToPlainText(html: string): string { + const stripped = html.replace(HTML_COMMENT_PATTERN, ' ').replace(NON_CONTENT_ELEMENT_PATTERN, ' ') + + const main = MAIN_REGION_PATTERN.exec(stripped) + const region = main ? main[2] : stripped + const text = htmlToPlainText(region) + + /** + * A site that renders its content entirely on the client leaves an empty + * `
`; fall back to the whole document rather than reporting the page as + * empty (which `getDocument` would turn into a dropped document). + */ + return text || htmlToPlainText(stripped) +} + +/** + * Builds the listing stub for a page. + * + * A Mintlify page exposes no version, ETag, or trustworthy `Last-Modified` + * (the hosted `.md` route reports the current time), so no metadata-derived + * change indicator exists. The stub therefore carries a path-only hash that can + * never equal a stored hash, which makes the sync engine re-hydrate every page; + * `getDocument` then returns a content-derived hash and the engine skips the + * write when it matches the stored one. Same trade-off as the Obsidian connector. + */ +function pageToStub(page: MintlifyPageLink, site: MintlifySite): ExternalDocument { + return { + externalId: page.path, + title: page.title, + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: `${site.origin}${page.path}`, + contentHash: `mintlify:${page.path}`, + metadata: { + path: page.path, + section: page.section, + description: page.description, + }, + } +} + +export const mintlifyConnector: ConnectorConfig = { + ...mintlifyConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + cursor?: string, + syncContext?: Record + ): Promise => { + const site = resolveSite(sourceConfig.siteUrl as string) + const pathPrefix = resolvePathPrefix(sourceConfig.pathPrefix) + const maxPages = resolveMaxPages(sourceConfig.maxPages) + + let pages = syncContext?.pages as MintlifyPageLink[] | undefined + if (!pages) { + const discovered = await discoverPages(site, accessToken) + /** + * An empty discovery means the site's index files were unreachable or + * unparseable, not that the docs are empty — `validateConfig` refuses a + * site with no index at setup time. Returning an empty listing would let + * the sync engine reconcile every stored page away, so this fails the sync + * instead: `shouldReconcileDeletions` never runs on a thrown sync. + */ + if (discovered.length === 0) { + throw new Error( + `No pages found at ${site.baseUrl} — the site's llms.txt and sitemap.xml are unavailable` + ) + } + + const filtered = pathPrefix + ? discovered.filter((page) => page.path.startsWith(pathPrefix)) + : discovered + + if (filtered.length > maxPages && syncContext) { + /** + * The listing is truncated while more pages exist, so deletion + * reconciliation must be suppressed — otherwise every page past the cap + * would be hard-deleted from the knowledge base. + */ + syncContext.listingCapped = true + logger.info('Mintlify page listing capped', { + discovered: filtered.length, + maxPages, + }) + } + + pages = filtered.slice(0, maxPages) + if (syncContext) { + syncContext.pages = pages + } + } + + const offset = cursor ? Number(cursor) : 0 + const pageSlice = pages.slice(offset, offset + DOCS_PER_PAGE) + const nextOffset = offset + pageSlice.length + const hasMore = nextOffset < pages.length + + return { + documents: pageSlice.map((page) => pageToStub(page, site)), + nextCursor: hasMore ? String(nextOffset) : undefined, + hasMore, + } + }, + + getDocument: async ( + accessToken: string, + sourceConfig: Record, + externalId: string, + syncContext?: Record + ): Promise => { + const site = resolveSite(sourceConfig.siteUrl as string) + const path = toPagePath(externalId, site) + if (!path) { + logger.warn('Skipping Mintlify page outside the configured site', { externalId }) + return null + } + + const pages = syncContext?.pages as MintlifyPageLink[] | undefined + const listed = pages?.find((page) => page.path === path) + + try { + /** + * Mintlify serves every page as raw Markdown at `{page}.md`. A site that + * does not (a non-Mintlify host that merely publishes an `llms.txt`, or a + * page removed from the Markdown route) answers 404 there, so the rendered + * HTML page is the fallback and gets stripped to text. + */ + const result = + (await fetchSiteText( + `${site.origin}${path === '/' ? '/index' : path}.md`, + accessToken, + 'text/markdown', + PAGE_MAX_BYTES + )) ?? + (await fetchSiteText(`${site.origin}${path}`, accessToken, 'text/html', PAGE_MAX_BYTES)) + if (!result) return null + + /** + * The `.md` route serves Markdown, which is already plain text. An HTML + * body — from the fallback above, or from a rewrite that ignores the + * extension — is stripped so raw markup is never indexed. + */ + const isHtml = + result.contentType.includes('html') || /^\s*<(!doctype\s+html|html\b)/i.test(result.body) + const content = isHtml ? htmlPageToPlainText(result.body) : result.body.trim() + if (!content) return null + + const stub = pageToStub(listed ?? { path, title: titleFromPath(path) }, site) + return { + ...stub, + content, + contentDeferred: false, + contentHash: `mintlify:${path}:${await computeContentHash(content)}`, + } + } catch (error) { + logger.warn('Failed to fetch Mintlify page', { + path, + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + let site: MintlifySite + try { + site = resolveSite(sourceConfig.siteUrl as string) + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Invalid documentation site URL') } + } + + const rawMaxPages = sourceConfig.maxPages + if (rawMaxPages !== undefined && rawMaxPages !== null && rawMaxPages !== '') { + const parsed = Number(rawMaxPages) + if (!Number.isFinite(parsed) || parsed <= 0) { + return { valid: false, error: 'Max Pages must be a positive number' } + } + } + + try { + const pages = await discoverPages(site, accessToken, VALIDATE_RETRY_OPTIONS) + if (pages.length === 0) { + return { + valid: false, + error: `No pages found at ${site.baseUrl}. The site must publish an llms.txt or sitemap.xml index.`, + } + } + + const pathPrefix = resolvePathPrefix(sourceConfig.pathPrefix) + if (pathPrefix && !pages.some((page) => page.path.startsWith(pathPrefix))) { + return { valid: false, error: `No pages match the path prefix "${pathPrefix}"` } + } + + return { valid: true } + } catch (error) { + return { + valid: false, + error: getErrorMessage(error, 'Failed to reach the Mintlify documentation site'), + } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + if (typeof metadata.section === 'string' && metadata.section.trim()) { + result.section = metadata.section.trim() + } + + if (typeof metadata.description === 'string' && metadata.description.trim()) { + result.description = metadata.description.trim() + } + + return result + }, +} diff --git a/apps/sim/connectors/pagerduty/index.ts b/apps/sim/connectors/pagerduty/index.ts new file mode 100644 index 00000000000..3de263ab54e --- /dev/null +++ b/apps/sim/connectors/pagerduty/index.ts @@ -0,0 +1 @@ +export { pagerdutyConnector } from '@/connectors/pagerduty/pagerduty' diff --git a/apps/sim/connectors/pagerduty/meta.ts b/apps/sim/connectors/pagerduty/meta.ts new file mode 100644 index 00000000000..78312e5d37e --- /dev/null +++ b/apps/sim/connectors/pagerduty/meta.ts @@ -0,0 +1,95 @@ +import { PagerDutyIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const pagerdutyConnectorMeta: ConnectorMeta = { + id: 'pagerduty', + name: 'PagerDuty', + description: 'Sync incidents, notes, and response timelines from PagerDuty', + version: '1.0.0', + icon: PagerDutyIcon, + + auth: { + mode: 'apiKey', + label: 'REST API Key', + placeholder: 'Enter your PagerDuty REST API key', + }, + + /** + * Deliberately absent. PagerDuty's `since`/`until` filter incident *creation* + * time, and the REST API exposes no modified-since filter, so an incremental + * listing would never surface a status change, a new note, or a resolution on + * an incident created before the window — an incident synced while triggered + * would stay triggered forever. Every sync therefore lists the full history, + * gated by `contentHash` so unchanged incidents are never re-hydrated. + */ + + configFields: [ + { + id: 'statuses', + title: 'Status', + type: 'dropdown', + required: false, + options: [ + { label: 'All (default)', id: '' }, + { label: 'Triggered', id: 'triggered' }, + { label: 'Acknowledged', id: 'acknowledged' }, + { label: 'Resolved', id: 'resolved' }, + ], + description: 'Only sync incidents in this status. Leave on All to sync every status.', + }, + { + id: 'urgency', + title: 'Urgency', + type: 'dropdown', + required: false, + mode: 'advanced', + options: [ + { label: 'All (default)', id: '' }, + { label: 'High', id: 'high' }, + { label: 'Low', id: 'low' }, + ], + description: 'Only sync incidents at this urgency. Requires the urgencies ability.', + }, + { + id: 'serviceIds', + title: 'Filter by Services', + type: 'short-input', + required: false, + mode: 'advanced', + multi: true, + placeholder: 'Service IDs (comma-separated, default: all)', + description: 'Only sync incidents on these PagerDuty service IDs (e.g. PIJ90N7).', + }, + { + id: 'teamIds', + title: 'Filter by Teams', + type: 'short-input', + required: false, + mode: 'advanced', + multi: true, + placeholder: 'Team IDs (comma-separated, default: all)', + description: + 'Only sync incidents owned by these PagerDuty team IDs. Requires the teams ability.', + }, + { + id: 'maxIncidents', + title: 'Max Incidents', + type: 'short-input', + required: false, + placeholder: 'e.g. 200 (default: unlimited)', + description: 'Cap the number of incidents synced. Leave empty to sync all incidents.', + }, + ], + + tagDefinitions: [ + { id: 'status', displayName: 'Status', fieldType: 'text' }, + { id: 'urgency', displayName: 'Urgency', fieldType: 'text' }, + { id: 'priority', displayName: 'Priority', fieldType: 'text' }, + { id: 'service', displayName: 'Service', fieldType: 'text' }, + { id: 'teams', displayName: 'Teams', fieldType: 'text' }, + { id: 'incidentType', displayName: 'Incident Type', fieldType: 'text' }, + { id: 'incidentDate', displayName: 'Incident Date', fieldType: 'date' }, + { id: 'resolvedDate', displayName: 'Resolved Date', fieldType: 'date' }, + { id: 'incidentNumber', displayName: 'Incident Number', fieldType: 'number' }, + ], +} diff --git a/apps/sim/connectors/pagerduty/pagerduty.ts b/apps/sim/connectors/pagerduty/pagerduty.ts new file mode 100644 index 00000000000..dab4a6e7034 --- /dev/null +++ b/apps/sim/connectors/pagerduty/pagerduty.ts @@ -0,0 +1,711 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { pagerdutyConnectorMeta } from '@/connectors/pagerduty/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { htmlToPlainText, joinTagArray, parseMultiValue, parseTagDate } from '@/connectors/utils' + +const logger = createLogger('PagerDutyConnector') + +const PAGERDUTY_API_BASE = 'https://api.pagerduty.com' +/** PagerDuty caps `limit` at 100 on every classic-pagination listing. */ +const PAGE_SIZE = 100 +/** Cap on log entries appended to a document so a noisy incident stays bounded. */ +const MAX_LOG_ENTRIES = 200 +/** + * Hard ceiling on classic (offset) pagination in PagerDuty's REST API v2: a + * request whose `offset + limit` exceeds this value is answered with + * `400 Invalid Request` rather than served. The listing therefore stops while the + * *next* request would still fit, and the caller marks the sync as capped instead + * of letting deletion reconciliation purge the unseen tail. + */ +const MAX_LISTING_WINDOW = 10000 + +const VALID_STATUSES = new Set(['triggered', 'acknowledged', 'resolved']) +const VALID_URGENCIES = new Set(['high', 'low']) + +/** PagerDuty's common reference envelope, present on service/team/priority/agent fields. */ +interface PagerDutyReference { + id?: string + type?: string + summary?: string + html_url?: string +} + +/** Additional incident body, returned only when `include[]=body` is requested. */ +interface PagerDutyIncidentBody { + type?: string + /** + * Documented as an object in PagerDuty's OpenAPI schema, but the Incident + * Creation API only accepts string bodies, so both forms occur in practice. + */ + details?: unknown +} + +interface PagerDutyIncident { + id?: string + incident_number?: number + title?: string + status?: string + urgency?: string + incident_key?: string + created_at?: string + updated_at?: string + last_status_change_at?: string + resolved_at?: string + html_url?: string + service?: PagerDutyReference + escalation_policy?: PagerDutyReference + teams?: PagerDutyReference[] + priority?: PagerDutyReference + assignments?: Array<{ assignee?: PagerDutyReference }> + incident_type?: { name?: string } + resolve_reason?: { type?: string; incident?: PagerDutyReference } + body?: PagerDutyIncidentBody +} + +interface PagerDutyIncidentsListResponse { + incidents?: PagerDutyIncident[] + limit?: number + offset?: number + more?: boolean +} + +interface PagerDutyIncidentShowResponse { + incident?: PagerDutyIncident +} + +interface PagerDutyNote { + id?: string + content?: string + created_at?: string + updated_at?: string + user?: PagerDutyReference +} + +interface PagerDutyNotesResponse { + notes?: PagerDutyNote[] +} + +interface PagerDutyLogEntry { + id?: string + type?: string + summary?: string + created_at?: string + agent?: PagerDutyReference + note?: string +} + +interface PagerDutyLogEntriesResponse { + log_entries?: PagerDutyLogEntry[] + more?: boolean +} + +/** + * Metadata persisted on every incident document. Produced by one function so the + * deferred list stub and the hydrated document carry identical tag values. + */ +interface IncidentMetadata { + status?: string + urgency?: string + priority?: string + service?: string + teams?: string[] + incidentType?: string + incidentDate?: string + resolvedDate?: string + incidentNumber?: number +} + +/** + * Builds PagerDuty's REST headers. The REST API authenticates with a + * `Token token=` scheme rather than Bearer, and pins the v2 schema through + * the versioned Accept header. + */ +function buildHeaders(accessToken: string): Record { + return { + Authorization: `Token token=${accessToken}`, + Accept: 'application/vnd.pagerduty+json;version=2', + 'Content-Type': 'application/json', + } +} + +/** + * Metadata-based content hash keyed on PagerDuty's own change indicator. + * + * `updated_at` is bumped whenever the incident is modified, so the hash is stable + * between the list stub and `getDocument`. Notes and log entries are child + * resources: PagerDuty does not guarantee they bump the parent's `updated_at`, so + * a full resync is the way to pick up note-only changes. + */ +function buildContentHash(incident: PagerDutyIncident): string { + return `pagerduty:${incident.id}:${incident.updated_at ?? ''}` +} + +function buildTitle(incident: PagerDutyIncident): string { + const title = incident.title?.trim() + const number = incident.incident_number + if (title && number != null) return `#${number}: ${title}` + return title || (number != null ? `Incident #${number}` : `Incident ${incident.id ?? ''}`.trim()) +} + +/** Extracts the human-readable summaries from a reference array. */ +function referenceLabels(references: PagerDutyReference[] | undefined): string[] | undefined { + if (!Array.isArray(references)) return undefined + const labels: string[] = [] + for (const reference of references) { + const label = reference.summary?.trim() + if (label) labels.push(label) + } + return labels.length > 0 ? labels : undefined +} + +function buildMetadata(incident: PagerDutyIncident): IncidentMetadata { + return { + status: incident.status ?? undefined, + urgency: incident.urgency ?? undefined, + priority: incident.priority?.summary ?? undefined, + service: incident.service?.summary ?? undefined, + teams: referenceLabels(incident.teams), + incidentType: incident.incident_type?.name ?? undefined, + incidentDate: incident.created_at ?? undefined, + resolvedDate: incident.resolved_at ?? undefined, + incidentNumber: + typeof incident.incident_number === 'number' ? incident.incident_number : undefined, + } +} + +/** + * Renders the incident body details, which arrive either as a plain/HTML string + * (Incident Creation API) or as a structured object (Events API payloads). + */ +function renderBodyDetails(details: unknown): string | undefined { + if (typeof details === 'string') { + const text = htmlToPlainText(details) + return text.trim() || undefined + } + if (details && typeof details === 'object') { + const lines: string[] = [] + for (const [key, value] of Object.entries(details as Record)) { + if (value == null || typeof value === 'object') continue + lines.push(`${key}: ${String(value)}`) + } + return lines.length > 0 ? lines.join('\n') : undefined + } + return undefined +} + +function incidentToStub(incident: PagerDutyIncident): ExternalDocument | null { + if (!incident.id) return null + return { + externalId: incident.id, + title: buildTitle(incident), + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: incident.html_url || undefined, + contentHash: buildContentHash(incident), + metadata: { ...buildMetadata(incident) }, + } +} + +/** + * Fetches every note on an incident. `GET /incidents/{id}/notes` takes no + * pagination parameters and returns the full set in one response. + */ +async function fetchNotes(accessToken: string, incidentId: string): Promise { + try { + const response = await fetchWithRetry( + `${PAGERDUTY_API_BASE}/incidents/${encodeURIComponent(incidentId)}/notes`, + { method: 'GET', headers: buildHeaders(accessToken) } + ) + + if (!response.ok) { + logger.warn('Failed to fetch PagerDuty incident notes', { + incidentId, + status: response.status, + }) + return [] + } + + const data = (await response.json()) as PagerDutyNotesResponse + return data.notes ?? [] + } catch (error) { + logger.warn('Error fetching PagerDuty incident notes', { + incidentId, + error: toError(error).message, + }) + return [] + } +} + +/** + * Fetches the incident's response timeline via classic offset pagination. + * + * `is_overview=true` narrows the log to the significant lifecycle changes + * (trigger, acknowledge, escalate, resolve, annotate) instead of every + * notification delivery, which keeps the indexed timeline readable. + * + * `since`/`until` are bounded to the incident's own lifetime. PagerDuty documents + * a one-month default range on `GET /incidents` but says nothing either way about + * `GET /incidents/{id}/log_entries`, and an inherited default would silently + * return an empty timeline for every incident older than the window. Every log + * entry necessarily falls inside `[created_at, now]`, so pinning that range can + * only remove ambiguity — and if PagerDuty rejects the range (an undocumented + * maximum span), the walk restarts unbounded rather than losing the timeline. + */ +async function fetchLogEntries( + accessToken: string, + incident: PagerDutyIncident +): Promise { + const incidentId = incident.id as string + const entries: PagerDutyLogEntry[] = [] + let offset = 0 + let bounded = Boolean(incident.created_at) + let truncated = false + + try { + while (entries.length < MAX_LOG_ENTRIES) { + const url = new URL( + `${PAGERDUTY_API_BASE}/incidents/${encodeURIComponent(incidentId)}/log_entries` + ) + url.searchParams.set('limit', String(PAGE_SIZE)) + url.searchParams.set('offset', String(offset)) + url.searchParams.set('is_overview', 'true') + if (bounded && incident.created_at) { + url.searchParams.set('since', incident.created_at) + url.searchParams.set('until', new Date().toISOString()) + } + + const response = await fetchWithRetry(url.toString(), { + method: 'GET', + headers: buildHeaders(accessToken), + }) + + if (!response.ok) { + if (bounded && offset === 0 && response.status === 400) { + logger.warn('PagerDuty rejected the log entry date range; retrying unbounded', { + incidentId, + }) + bounded = false + continue + } + logger.warn('Failed to fetch PagerDuty incident log entries', { + incidentId, + status: response.status, + }) + break + } + + const data = (await response.json()) as PagerDutyLogEntriesResponse + const page = data.log_entries ?? [] + entries.push(...page) + + if (!data.more || page.length === 0) break + if (entries.length >= MAX_LOG_ENTRIES) { + truncated = true + break + } + offset += page.length + } + } catch (error) { + logger.warn('Error fetching PagerDuty incident log entries', { + incidentId, + error: toError(error).message, + }) + } + + if (truncated || entries.length > MAX_LOG_ENTRIES) { + logger.warn('Truncated PagerDuty incident timeline at the per-document cap', { + incidentId, + cap: MAX_LOG_ENTRIES, + }) + } + + return entries.slice(0, MAX_LOG_ENTRIES) +} + +/** + * Formats an incident, its notes, and its timeline into a single plain-text + * document. Sections without data are omitted so open incidents do not carry + * empty resolution headers. + */ +function formatIncidentContent( + incident: PagerDutyIncident, + notes: PagerDutyNote[], + logEntries: PagerDutyLogEntry[] +): string { + const parts: string[] = [] + + parts.push(`Incident: ${buildTitle(incident)}`) + if (incident.status) parts.push(`Status: ${incident.status}`) + if (incident.urgency) parts.push(`Urgency: ${incident.urgency}`) + if (incident.priority?.summary) parts.push(`Priority: ${incident.priority.summary}`) + if (incident.service?.summary) parts.push(`Service: ${incident.service.summary}`) + if (incident.incident_type?.name) parts.push(`Type: ${incident.incident_type.name}`) + + const teams = referenceLabels(incident.teams) + if (teams) parts.push(`Teams: ${teams.join(', ')}`) + + const assignees = referenceLabels( + incident.assignments?.map((assignment) => assignment.assignee ?? {}) + ) + if (assignees) parts.push(`Assigned to: ${assignees.join(', ')}`) + + if (incident.escalation_policy?.summary) { + parts.push(`Escalation Policy: ${incident.escalation_policy.summary}`) + } + if (incident.created_at) parts.push(`Triggered: ${incident.created_at}`) + if (incident.resolved_at) parts.push(`Resolved: ${incident.resolved_at}`) + if (incident.resolve_reason?.type) parts.push(`Resolve Reason: ${incident.resolve_reason.type}`) + if (incident.incident_key) parts.push(`Incident Key: ${incident.incident_key}`) + + const details = renderBodyDetails(incident.body?.details) + if (details) { + parts.push('') + parts.push('--- Details ---') + parts.push(details) + } + + const noteLines = notes + .map((note) => { + const content = note.content?.trim() + if (!content) return undefined + const author = note.user?.summary?.trim() + const prefix = [note.created_at ? `[${note.created_at}]` : '', author ?? ''] + .filter(Boolean) + .join(' ') + return prefix ? `${prefix}: ${content}` : content + }) + .filter((line): line is string => Boolean(line)) + if (noteLines.length > 0) { + parts.push('') + parts.push('--- Notes ---') + parts.push(...noteLines) + } + + const timelineLines = logEntries + .map((entry) => { + const summary = entry.summary?.trim() + const note = entry.note?.trim() + const text = [summary, note].filter(Boolean).join(' — ') + if (!text) return undefined + const agent = entry.agent?.summary?.trim() + const prefix = [entry.created_at ? `[${entry.created_at}]` : '', agent ?? ''] + .filter(Boolean) + .join(' ') + return prefix ? `${prefix}: ${text}` : text + }) + .filter((line): line is string => Boolean(line)) + if (timelineLines.length > 0) { + parts.push('') + parts.push('--- Timeline ---') + parts.push(...timelineLines) + } + + return parts.join('\n').trim() +} + +/** + * Reads the optional `maxIncidents` cap, returning 0 (unlimited) when unset or + * not a positive number. + */ +function parseMaxIncidents(sourceConfig: Record): number { + const raw = sourceConfig.maxIncidents + if (raw == null || raw === '') return 0 + const value = Number(raw) + return Number.isFinite(value) && value > 0 ? Math.floor(value) : 0 +} + +function readStringConfig(sourceConfig: Record, key: string): string { + const value = sourceConfig[key] + return typeof value === 'string' ? value.trim() : '' +} + +/** + * Fetches a single incident, asking for the extra sections the document renders. + * + * PagerDuty's OpenAPI schema documents `Incident.body` as "only returned if the + * `include[]=body` query parameter is provided", yet omits `body` from the same + * spec's `include[]` enum. The value is therefore requested on the strength of + * the field docs, and a rejected request is retried once without it rather than + * failing the document. Returns `null` when the incident no longer exists. + */ +async function fetchIncident( + accessToken: string, + incidentId: string +): Promise { + const base = `${PAGERDUTY_API_BASE}/incidents/${encodeURIComponent(incidentId)}` + const includes = ['teams', 'priorities', 'services'] + + for (const withBody of [true, false]) { + const url = new URL(base) + for (const include of includes) url.searchParams.append('include[]', include) + if (withBody) url.searchParams.append('include[]', 'body') + + const response = await fetchWithRetry(url.toString(), { + method: 'GET', + headers: buildHeaders(accessToken), + }) + + if (response.ok) { + const data = (await response.json()) as PagerDutyIncidentShowResponse + return data.incident ?? null + } + + if (response.status === 404 || response.status === 410) return null + if (withBody && response.status === 400) { + logger.warn('PagerDuty rejected the body include; retrying without it', { incidentId }) + continue + } + + throw new Error(`Failed to fetch PagerDuty incident: ${response.status}`) + } + + return null +} + +export const pagerdutyConnector: ConnectorConfig = { + ...pagerdutyConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + cursor?: string, + syncContext?: Record + ): Promise => { + const maxIncidents = parseMaxIncidents(sourceConfig) + const status = readStringConfig(sourceConfig, 'statuses') + const urgency = readStringConfig(sourceConfig, 'urgency') + const serviceIds = parseMultiValue(sourceConfig.serviceIds) + const teamIds = parseMultiValue(sourceConfig.teamIds) + + const parsedCursor = cursor ? Number(cursor) : 0 + const offset = Number.isFinite(parsedCursor) && parsedCursor > 0 ? Math.floor(parsedCursor) : 0 + + const url = new URL(`${PAGERDUTY_API_BASE}/incidents`) + url.searchParams.set('limit', String(PAGE_SIZE)) + url.searchParams.set('offset', String(offset)) + url.searchParams.append('sort_by', 'created_at:asc') + if (status) url.searchParams.append('statuses[]', status) + if (urgency) url.searchParams.append('urgencies[]', urgency) + for (const serviceId of serviceIds) url.searchParams.append('service_ids[]', serviceId) + for (const teamId of teamIds) url.searchParams.append('team_ids[]', teamId) + + /** + * Without an explicit range PagerDuty defaults `since`/`until` to the last + * month, which would silently truncate the listing and make deletion + * reconciliation purge every older document. `date_range=all` disables that + * default and returns the complete history. + */ + url.searchParams.set('date_range', 'all') + + logger.info('Listing PagerDuty incidents', { offset, maxIncidents }) + + const response = await fetchWithRetry(url.toString(), { + method: 'GET', + headers: buildHeaders(accessToken), + }) + + if (!response.ok) { + const errorText = await response.text().catch(() => '') + logger.error('Failed to list PagerDuty incidents', { + status: response.status, + error: errorText.slice(0, 500), + }) + throw new Error(`Failed to list PagerDuty incidents: ${response.status}`) + } + + const data = (await response.json()) as PagerDutyIncidentsListResponse + const incidents = data.incidents ?? [] + + const allDocuments: ExternalDocument[] = [] + let skipped = 0 + for (const incident of incidents) { + const stub = incidentToStub(incident) + if (stub) { + allDocuments.push(stub) + } else { + skipped += 1 + } + } + + /** + * An incident dropped for a missing ID is still present in PagerDuty, so the + * listing is incomplete and must not drive deletion reconciliation. + */ + if (skipped > 0 && syncContext) { + logger.warn('Skipped PagerDuty incidents without an ID', { skipped }) + syncContext.listingCapped = true + } + + const prevFetched = (syncContext?.totalDocsFetched as number) ?? 0 + let documents = allDocuments + if (maxIncidents > 0) { + const remaining = Math.max(0, maxIncidents - prevFetched) + if (allDocuments.length > remaining) { + documents = allDocuments.slice(0, remaining) + } + } + + const totalFetched = prevFetched + documents.length + if (syncContext) syncContext.totalDocsFetched = totalFetched + const hitLimit = maxIncidents > 0 && totalFetched >= maxIncidents + const sourceHasMore = Boolean(data.more) + /** + * The cap only truncates the listing when it actually withheld something: a + * `maxIncidents` that happens to equal the source's exact incident count + * still yields a complete listing, and marking it capped would block deletion + * reconciliation forever. + */ + if (hitLimit && (sourceHasMore || documents.length < allDocuments.length) && syncContext) { + syncContext.listingCapped = true + } + + const nextOffset = offset + incidents.length + /** + * Measured against the request that *would* come next: PagerDuty rejects + * `offset + limit > 10000`, so a short page (which leaves `nextOffset` off a + * clean page boundary) must still stop before the sum overruns. + */ + const hitOffsetCeiling = nextOffset + PAGE_SIZE > MAX_LISTING_WINDOW + /** + * PagerDuty reported more results but served none, so the walk cannot + * advance — `offset` only moves by what the page returned. Treat it as a + * truncated listing rather than source exhaustion. + */ + const stalledPage = sourceHasMore && incidents.length === 0 + const hasMore = !hitLimit && !hitOffsetCeiling && sourceHasMore && incidents.length > 0 + + if (!hitLimit && hitOffsetCeiling && sourceHasMore && syncContext) { + logger.warn('Stopping PagerDuty listing at the pagination window ceiling', { + offset: nextOffset, + }) + syncContext.listingCapped = true + } + + if (!hitLimit && stalledPage && syncContext) { + logger.warn('PagerDuty reported more incidents but returned an empty page', { offset }) + syncContext.listingCapped = true + } + + return { + documents, + nextCursor: hasMore ? String(nextOffset) : undefined, + hasMore, + } + }, + + getDocument: async ( + accessToken: string, + _sourceConfig: Record, + externalId: string + ): Promise => { + try { + if (!externalId) return null + + const incident = await fetchIncident(accessToken, externalId) + if (!incident?.id) return null + + const [notes, logEntries] = await Promise.all([ + fetchNotes(accessToken, incident.id), + fetchLogEntries(accessToken, incident), + ]) + + const content = formatIncidentContent(incident, notes, logEntries) + if (!content.trim()) { + logger.info('Skipping PagerDuty incident with no indexable content', { externalId }) + return null + } + + return { + externalId: incident.id, + title: buildTitle(incident), + content, + contentDeferred: false, + mimeType: 'text/plain', + sourceUrl: incident.html_url || undefined, + contentHash: buildContentHash(incident), + metadata: { ...buildMetadata(incident) }, + } + } catch (error) { + logger.warn('Failed to get PagerDuty incident', { + externalId, + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const maxIncidents = sourceConfig.maxIncidents as string | undefined + if (maxIncidents && (Number.isNaN(Number(maxIncidents)) || Number(maxIncidents) < 0)) { + return { valid: false, error: 'Max incidents must be a non-negative number' } + } + + const status = readStringConfig(sourceConfig, 'statuses') + if (status && !VALID_STATUSES.has(status)) { + return { + valid: false, + error: 'Status must be one of triggered, acknowledged, or resolved', + } + } + + const urgency = readStringConfig(sourceConfig, 'urgency') + if (urgency && !VALID_URGENCIES.has(urgency)) { + return { valid: false, error: 'Urgency must be either high or low' } + } + + try { + const response = await fetchWithRetry( + `${PAGERDUTY_API_BASE}/incidents?limit=1&date_range=all`, + { + method: 'GET', + headers: buildHeaders(accessToken), + }, + VALIDATE_RETRY_OPTIONS + ) + + if (!response.ok) { + const errorText = await response.text().catch(() => '') + return { + valid: false, + error: `PagerDuty access failed: ${response.status}${errorText ? ` — ${errorText.slice(0, 200)}` : ''}`, + } + } + + return { valid: true } + } catch (error) { + const message = getErrorMessage(error, 'Failed to validate configuration') + return { valid: false, error: message } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + for (const key of ['status', 'urgency', 'priority', 'service', 'incidentType'] as const) { + const value = metadata[key] + if (typeof value === 'string' && value.trim()) result[key] = value + } + + const teams = joinTagArray(metadata.teams) + if (teams) result.teams = teams + + const incidentDate = parseTagDate(metadata.incidentDate) + if (incidentDate) result.incidentDate = incidentDate + + const resolvedDate = parseTagDate(metadata.resolvedDate) + if (resolvedDate) result.resolvedDate = resolvedDate + + if (metadata.incidentNumber != null) { + const incidentNumber = Number(metadata.incidentNumber) + if (!Number.isNaN(incidentNumber)) result.incidentNumber = incidentNumber + } + + return result + }, +} diff --git a/apps/sim/connectors/registry.server.ts b/apps/sim/connectors/registry.server.ts index ba870e2af41..229d9b5331f 100644 --- a/apps/sim/connectors/registry.server.ts +++ b/apps/sim/connectors/registry.server.ts @@ -2,6 +2,7 @@ import { airtableConnector } from '@/connectors/airtable' import { asanaConnector } from '@/connectors/asana' import { ashbyConnector } from '@/connectors/ashby' import { azureDevopsConnector } from '@/connectors/azure-devops' +import { boxConnector } from '@/connectors/box' import { clickupConnector } from '@/connectors/clickup' import { confluenceConnector } from '@/connectors/confluence' import { discordConnector } from '@/connectors/discord' @@ -20,6 +21,8 @@ import { googleDriveConnector } from '@/connectors/google-drive' import { googleFormsConnector } from '@/connectors/google-forms' import { googleMeetConnector } from '@/connectors/google-meet' import { googleSheetsConnector } from '@/connectors/google-sheets' +import { googleSlidesConnector } from '@/connectors/google-slides' +import { googleVaultConnector } from '@/connectors/google-vault' import { grainConnector } from '@/connectors/grain' import { granolaConnector } from '@/connectors/granola' import { greenhouseConnector } from '@/connectors/greenhouse' @@ -29,20 +32,25 @@ import { intercomConnector } from '@/connectors/intercom' import { jiraConnector } from '@/connectors/jira' import { jsmConnector } from '@/connectors/jsm' import { linearConnector } from '@/connectors/linear' +import { microsoftExcelConnector } from '@/connectors/microsoft-excel' import { microsoftTeamsConnector } from '@/connectors/microsoft-teams' +import { mintlifyConnector } from '@/connectors/mintlify' import { mondayConnector } from '@/connectors/monday' import { notionConnector } from '@/connectors/notion' import { obsidianConnector } from '@/connectors/obsidian' import { onedriveConnector } from '@/connectors/onedrive' import { outlookConnector } from '@/connectors/outlook' +import { pagerdutyConnector } from '@/connectors/pagerduty' import { redditConnector } from '@/connectors/reddit' import { rootlyConnector } from '@/connectors/rootly' import { s3Connector } from '@/connectors/s3' import { salesforceConnector } from '@/connectors/salesforce' import { sentryConnector } from '@/connectors/sentry' import { servicenowConnector } from '@/connectors/servicenow' +import { sftpConnector } from '@/connectors/sftp' import { sharepointConnector } from '@/connectors/sharepoint' import { slackConnector } from '@/connectors/slack' +import { trelloConnector } from '@/connectors/trello' import { typeformConnector } from '@/connectors/typeform' import type { ConnectorRegistry } from '@/connectors/types' import { webflowConnector } from '@/connectors/webflow' @@ -50,6 +58,7 @@ import { wordpressConnector } from '@/connectors/wordpress' import { xConnector } from '@/connectors/x' import { youtubeConnector } from '@/connectors/youtube' import { zendeskConnector } from '@/connectors/zendesk' +import { zohoDeskConnector } from '@/connectors/zoho-desk' import { zoomConnector } from '@/connectors/zoom' /** @@ -64,6 +73,7 @@ export const CONNECTOR_REGISTRY: ConnectorRegistry = { asana: asanaConnector, ashby: ashbyConnector, azure_devops: azureDevopsConnector, + box: boxConnector, clickup: clickupConnector, confluence: confluenceConnector, discord: discordConnector, @@ -82,6 +92,8 @@ export const CONNECTOR_REGISTRY: ConnectorRegistry = { google_forms: googleFormsConnector, google_meet: googleMeetConnector, google_sheets: googleSheetsConnector, + google_slides: googleSlidesConnector, + google_vault: googleVaultConnector, grain: grainConnector, granola: granolaConnector, greenhouse: greenhouseConnector, @@ -91,25 +103,31 @@ export const CONNECTOR_REGISTRY: ConnectorRegistry = { jira: jiraConnector, jsm: jsmConnector, linear: linearConnector, + microsoft_excel: microsoftExcelConnector, microsoft_teams: microsoftTeamsConnector, + mintlify: mintlifyConnector, monday: mondayConnector, notion: notionConnector, obsidian: obsidianConnector, onedrive: onedriveConnector, outlook: outlookConnector, + pagerduty: pagerdutyConnector, reddit: redditConnector, rootly: rootlyConnector, s3: s3Connector, salesforce: salesforceConnector, sentry: sentryConnector, servicenow: servicenowConnector, + sftp: sftpConnector, sharepoint: sharepointConnector, slack: slackConnector, + trello: trelloConnector, typeform: typeformConnector, webflow: webflowConnector, wordpress: wordpressConnector, x: xConnector, youtube: youtubeConnector, zendesk: zendeskConnector, + zoho_desk: zohoDeskConnector, zoom: zoomConnector, } diff --git a/apps/sim/connectors/registry.ts b/apps/sim/connectors/registry.ts index b1fd50e9736..8d46de4c98a 100644 --- a/apps/sim/connectors/registry.ts +++ b/apps/sim/connectors/registry.ts @@ -2,6 +2,7 @@ import { airtableConnectorMeta } from '@/connectors/airtable/meta' import { asanaConnectorMeta } from '@/connectors/asana/meta' import { ashbyConnectorMeta } from '@/connectors/ashby/meta' import { azureDevopsConnectorMeta } from '@/connectors/azure-devops/meta' +import { boxConnectorMeta } from '@/connectors/box/meta' import { clickupConnectorMeta } from '@/connectors/clickup/meta' import { confluenceConnectorMeta } from '@/connectors/confluence/meta' import { discordConnectorMeta } from '@/connectors/discord/meta' @@ -20,6 +21,8 @@ import { googleDriveConnectorMeta } from '@/connectors/google-drive/meta' import { googleFormsConnectorMeta } from '@/connectors/google-forms/meta' import { googleMeetConnectorMeta } from '@/connectors/google-meet/meta' import { googleSheetsConnectorMeta } from '@/connectors/google-sheets/meta' +import { googleSlidesConnectorMeta } from '@/connectors/google-slides/meta' +import { googleVaultConnectorMeta } from '@/connectors/google-vault/meta' import { grainConnectorMeta } from '@/connectors/grain/meta' import { granolaConnectorMeta } from '@/connectors/granola/meta' import { greenhouseConnectorMeta } from '@/connectors/greenhouse/meta' @@ -29,20 +32,25 @@ import { intercomConnectorMeta } from '@/connectors/intercom/meta' import { jiraConnectorMeta } from '@/connectors/jira/meta' import { jsmConnectorMeta } from '@/connectors/jsm/meta' import { linearConnectorMeta } from '@/connectors/linear/meta' +import { microsoftExcelConnectorMeta } from '@/connectors/microsoft-excel/meta' import { microsoftTeamsConnectorMeta } from '@/connectors/microsoft-teams/meta' +import { mintlifyConnectorMeta } from '@/connectors/mintlify/meta' import { mondayConnectorMeta } from '@/connectors/monday/meta' import { notionConnectorMeta } from '@/connectors/notion/meta' import { obsidianConnectorMeta } from '@/connectors/obsidian/meta' import { onedriveConnectorMeta } from '@/connectors/onedrive/meta' import { outlookConnectorMeta } from '@/connectors/outlook/meta' +import { pagerdutyConnectorMeta } from '@/connectors/pagerduty/meta' import { redditConnectorMeta } from '@/connectors/reddit/meta' import { rootlyConnectorMeta } from '@/connectors/rootly/meta' import { s3ConnectorMeta } from '@/connectors/s3/meta' import { salesforceConnectorMeta } from '@/connectors/salesforce/meta' import { sentryConnectorMeta } from '@/connectors/sentry/meta' import { servicenowConnectorMeta } from '@/connectors/servicenow/meta' +import { sftpConnectorMeta } from '@/connectors/sftp/meta' import { sharepointConnectorMeta } from '@/connectors/sharepoint/meta' import { slackConnectorMeta } from '@/connectors/slack/meta' +import { trelloConnectorMeta } from '@/connectors/trello/meta' import { typeformConnectorMeta } from '@/connectors/typeform/meta' import type { ConnectorMeta, ConnectorMetaRegistry } from '@/connectors/types' import { webflowConnectorMeta } from '@/connectors/webflow/meta' @@ -50,6 +58,7 @@ import { wordpressConnectorMeta } from '@/connectors/wordpress/meta' import { xConnectorMeta } from '@/connectors/x/meta' import { youtubeConnectorMeta } from '@/connectors/youtube/meta' import { zendeskConnectorMeta } from '@/connectors/zendesk/meta' +import { zohoDeskConnectorMeta } from '@/connectors/zoho-desk/meta' import { zoomConnectorMeta } from '@/connectors/zoom/meta' /** @@ -64,6 +73,7 @@ export const CONNECTOR_META_REGISTRY: ConnectorMetaRegistry = { asana: asanaConnectorMeta, ashby: ashbyConnectorMeta, azure_devops: azureDevopsConnectorMeta, + box: boxConnectorMeta, clickup: clickupConnectorMeta, confluence: confluenceConnectorMeta, discord: discordConnectorMeta, @@ -82,6 +92,8 @@ export const CONNECTOR_META_REGISTRY: ConnectorMetaRegistry = { google_forms: googleFormsConnectorMeta, google_meet: googleMeetConnectorMeta, google_sheets: googleSheetsConnectorMeta, + google_slides: googleSlidesConnectorMeta, + google_vault: googleVaultConnectorMeta, grain: grainConnectorMeta, granola: granolaConnectorMeta, greenhouse: greenhouseConnectorMeta, @@ -91,26 +103,32 @@ export const CONNECTOR_META_REGISTRY: ConnectorMetaRegistry = { jira: jiraConnectorMeta, jsm: jsmConnectorMeta, linear: linearConnectorMeta, + microsoft_excel: microsoftExcelConnectorMeta, microsoft_teams: microsoftTeamsConnectorMeta, + mintlify: mintlifyConnectorMeta, monday: mondayConnectorMeta, notion: notionConnectorMeta, obsidian: obsidianConnectorMeta, onedrive: onedriveConnectorMeta, outlook: outlookConnectorMeta, + pagerduty: pagerdutyConnectorMeta, reddit: redditConnectorMeta, rootly: rootlyConnectorMeta, s3: s3ConnectorMeta, salesforce: salesforceConnectorMeta, sentry: sentryConnectorMeta, servicenow: servicenowConnectorMeta, + sftp: sftpConnectorMeta, sharepoint: sharepointConnectorMeta, slack: slackConnectorMeta, + trello: trelloConnectorMeta, typeform: typeformConnectorMeta, webflow: webflowConnectorMeta, wordpress: wordpressConnectorMeta, x: xConnectorMeta, youtube: youtubeConnectorMeta, zendesk: zendeskConnectorMeta, + zoho_desk: zohoDeskConnectorMeta, zoom: zoomConnectorMeta, } diff --git a/apps/sim/connectors/sftp/index.ts b/apps/sim/connectors/sftp/index.ts new file mode 100644 index 00000000000..aaf2a5c21fe --- /dev/null +++ b/apps/sim/connectors/sftp/index.ts @@ -0,0 +1 @@ +export { sftpConnector } from '@/connectors/sftp/sftp' diff --git a/apps/sim/connectors/sftp/meta.ts b/apps/sim/connectors/sftp/meta.ts new file mode 100644 index 00000000000..29ac8ada690 --- /dev/null +++ b/apps/sim/connectors/sftp/meta.ts @@ -0,0 +1,107 @@ +import { SftpIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const sftpConnectorMeta: ConnectorMeta = { + id: 'sftp', + name: 'SFTP', + description: + 'Sync text-based files from a remote SFTP (SSH File Transfer Protocol) directory tree into your knowledge base', + version: '1.0.0', + icon: SftpIcon, + + auth: { + mode: 'apiKey', + label: 'Password or Private Key', + placeholder: 'Password, or paste an unencrypted OpenSSH private key', + }, + + supportsIncrementalSync: true, + + configFields: [ + { + id: 'host', + title: 'Host', + type: 'short-input', + placeholder: 'e.g. sftp.example.com', + required: true, + description: + 'Hostname of the SFTP server. Private, loopback, and link-local addresses are rejected.', + }, + { + id: 'port', + title: 'Port', + type: 'short-input', + placeholder: '22', + required: false, + description: 'SSH port. Defaults to 22.', + }, + { + id: 'username', + title: 'Username', + type: 'short-input', + placeholder: 'e.g. sftp-user', + required: true, + }, + { + id: 'authMethod', + title: 'Authentication Method', + type: 'dropdown', + required: false, + options: [ + { label: 'Password', id: 'password' }, + { label: 'Private Key', id: 'privateKey' }, + ], + description: + 'How the secret above is interpreted. Private keys must be unencrypted (no passphrase).', + }, + { + id: 'hostFingerprint', + title: 'Host Key Fingerprint', + type: 'short-input', + placeholder: 'e.g. SHA256:abc123... (optional)', + required: false, + description: + 'Expected SHA-256 host key fingerprint. Get it with "ssh-keyscan -t rsa,ecdsa,ed25519 | ssh-keygen -lf -" and paste the SHA256:... value. If it does not match, the connection is refused before any credential is sent. Leave empty to skip host verification (the server is then trusted on sight).', + }, + { + id: 'rootPath', + title: 'Root Path', + type: 'short-input', + placeholder: 'e.g. /home/sftp-user/docs', + required: true, + description: 'Absolute remote directory to sync. Only files under this path are indexed.', + }, + { + id: 'extensions', + title: 'File Extensions', + type: 'short-input', + placeholder: 'e.g. txt, md, csv (optional)', + required: false, + description: + 'Comma-separated list of file extensions to sync. Leave blank to use the built-in text formats.', + }, + { + id: 'maxDepth', + title: 'Max Directory Depth', + type: 'short-input', + placeholder: 'e.g. 5 (default: 5, max: 10)', + required: false, + description: 'How many directory levels below the root path to walk.', + }, + { + id: 'maxFiles', + title: 'Max Files', + type: 'short-input', + placeholder: 'e.g. 2000 (default: 2000, max: 10000)', + required: false, + description: 'Stop syncing after this many files.', + }, + ], + + tagDefinitions: [ + { id: 'directory', displayName: 'Folder', fieldType: 'text' }, + { id: 'extension', displayName: 'Extension', fieldType: 'text' }, + { id: 'fileSize', displayName: 'Size (bytes)', fieldType: 'number' }, + { id: 'lastModified', displayName: 'Last Modified', fieldType: 'date' }, + ], +} diff --git a/apps/sim/connectors/sftp/sftp.ts b/apps/sim/connectors/sftp/sftp.ts new file mode 100644 index 00000000000..97d2297450a --- /dev/null +++ b/apps/sim/connectors/sftp/sftp.ts @@ -0,0 +1,709 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import type { Attributes, Client, SFTPWrapper } from 'ssh2' +import { isPayloadSizeLimitError } from '@/lib/core/utils/stream-limits' +import { + createSftpConnection, + getFileType, + getSftp, + isPathSafe, + readSftpFileCapped, + sanitizePath, +} from '@/app/api/tools/sftp/utils' +import { sftpConnectorMeta } from '@/connectors/sftp/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { + CONNECTOR_MAX_FILE_BYTES, + htmlToPlainText, + markSkipped, + parseTagDate, + sizeLimitSkipReason, + stubOrSkipBySize, +} from '@/connectors/utils' + +const logger = createLogger('SftpConnector') + +/** Maximum bytes read from a single remote file. Larger files are surfaced as skipped. */ +const MAX_FILE_SIZE = CONNECTOR_MAX_FILE_BYTES + +/** Directory levels below the root path walked when `maxDepth` is not configured. */ +const DEFAULT_MAX_DEPTH = 5 + +/** Hard ceiling on directory recursion, regardless of the configured `maxDepth`. */ +const MAX_ALLOWED_DEPTH = 10 + +/** Files listed per sync when `maxFiles` is not configured. */ +const DEFAULT_MAX_FILES = 2000 + +/** Hard ceiling on files listed per sync, regardless of the configured `maxFiles`. */ +const MAX_ALLOWED_FILES = 10_000 + +/** Hard ceiling on `readdir` calls per sync, bounding wide (rather than deep) trees. */ +const MAX_DIRECTORIES = 1000 + +/** + * Hard ceiling on entries emitted by a single walk, counting oversized files. + * Oversized files deliberately do not consume the `maxFiles` budget, so without + * this second ceiling a tree of nothing but oversized files would grow the + * listing until the walk ran out of directories. + */ +const MAX_LISTED_ENTRIES = MAX_ALLOWED_FILES + +/** Seconds to wait for the SSH handshake before giving up. */ +const READY_TIMEOUT_MS = 20_000 + +/** + * Keepalive cadence. ssh2 tears the connection down after three unanswered + * keepalives, which is what turns a server that accepts the TCP connection and + * then goes silent into an error rather than an indefinite wait. + */ +const KEEPALIVE_INTERVAL_MS = 10_000 + +/** + * Wall-clock ceiling on a directory walk. `readyTimeout` bounds only the SSH + * handshake; every SFTP request after it is unbounded, so a server that answers + * the handshake and then trickles (or never answers) `readdir` would otherwise + * hold the sync task open until its own 30-minute deadline. + */ +const LISTING_TIMEOUT_MS = 10 * 60_000 + +/** Wall-clock ceiling on fetching a single document, for the same reason. */ +const DOCUMENT_TIMEOUT_MS = 2 * 60_000 + +/** + * Slack subtracted from the incremental cutoff. `mtime` comes from the remote + * server's clock: if it runs behind ours, a file written just after a sync gets + * an `mtime` below the cutoff and would be skipped by every later incremental + * sync, silently and permanently. Re-listing a few minutes of overlap is cheap + * because unchanged documents are hash-gated by the sync engine. + */ +const INCREMENTAL_CLOCK_SKEW_SECONDS = 300 + +/** Bytes inspected when sniffing a downloaded file for binary content. */ +const BINARY_SNIFF_BYTES = 8192 + +/** + * File extensions considered safely text-extractable. Anything else (or a file + * with no extension) is skipped, since its bytes cannot be reliably decoded to + * plain text. Users override this list via the `extensions` config field. + */ +const DEFAULT_EXTENSIONS = new Set([ + 'txt', + 'md', + 'markdown', + 'csv', + 'tsv', + 'json', + 'jsonl', + 'ndjson', + 'html', + 'htm', + 'xml', + 'yaml', + 'yml', + 'log', + 'rtf', +]) + +/** Extensions whose content is rendered markup and must be flattened before indexing. */ +const HTML_EXTENSIONS = new Set(['html', 'htm']) + +/** + * Minimal shape of an `SFTPWrapper.readdir` entry. Declared structurally rather + * than importing ssh2's `FileEntryWithStats` so the connector depends only on + * the fields it reads. + */ +interface SftpDirEntry { + filename: string + attrs: Attributes +} + +/** A remote file selected for syncing during the directory walk. */ +interface SftpFileEntry { + /** Absolute remote path, used as the document's externalId. */ + path: string + /** Absolute remote path of the containing directory. */ + directory: string + size: number + /** Modification time in epoch seconds, as reported by the server. */ + mtime: number +} + +/** Connection and scope parameters resolved from sourceConfig + the stored secret. */ +interface SftpContext { + host: string + port: number + username: string + password?: string + privateKey?: string + /** Optional pinned SHA-256 host key fingerprint; empty means no verification. */ + hostFingerprint?: string + rootPath: string + allowedExtensions: Set + maxDepth: number + maxFiles: number +} + +/** + * Parses the comma-separated `extensions` override into a normalized set + * (lowercased, no leading dot). Falls back to the built-in text formats. + */ +function resolveExtensions(raw: unknown): Set { + if (typeof raw !== 'string') return DEFAULT_EXTENSIONS + const exts = raw + .split(',') + .map((e) => e.trim().toLowerCase().replace(/^\./, '')) + .filter(Boolean) + return exts.length > 0 ? new Set(exts) : DEFAULT_EXTENSIONS +} + +/** + * Clamps a numeric config value into `[1, max]`, falling back to `fallback` + * when the value is absent or not a positive number. + */ +function resolveBoundedNumber(raw: unknown, fallback: number, max: number): number { + const parsed = typeof raw === 'number' ? raw : Number((raw as string) ?? '') + if (!Number.isFinite(parsed) || parsed <= 0) return fallback + return Math.min(Math.floor(parsed), max) +} + +/** + * Unpadded base64 of a SHA-256 digest — what OpenSSH prints after the `SHA256:` + * prefix (32 digest bytes encode to 43 base64 characters). + */ +const SHA256_FINGERPRINT_PATTERN = /^[A-Za-z0-9+/]{43}$/ + +/** + * Normalizes and validates the pinned host key fingerprint. Validation matters + * because host verification is opt-in: a value that normalizes to nothing (a + * bare `SHA256:`), or an MD5 fingerprint, would otherwise be dropped and the + * connection would silently fall back to trusting whatever host answers. + */ +function resolveHostFingerprint(raw: unknown): string | undefined { + if (typeof raw !== 'string') return undefined + const trimmed = raw.trim() + if (!trimmed) return undefined + + const normalized = trimmed + .replace(/^sha256:/i, '') + .replace(/=+$/, '') + .trim() + if (!SHA256_FINGERPRINT_PATTERN.test(normalized)) { + throw new Error( + 'Host key fingerprint must be a SHA-256 fingerprint, e.g. "SHA256:<43 base64 characters>". ' + + 'Get it with "ssh-keyscan -t rsa,ecdsa,ed25519 | ssh-keygen -lf -".' + ) + } + return normalized +} + +/** Extracts the lowercased extension of a path segment, or '' when there is none. */ +function getExtension(filePath: string): string { + const name = filePath.split('/').pop() ?? '' + const dotIndex = name.lastIndexOf('.') + if (dotIndex <= 0 || dotIndex === name.length - 1) return '' + return name.slice(dotIndex + 1).toLowerCase() +} + +/** + * Normalizes a remote path to an absolute, separator-collapsed form without a + * trailing slash (the root `/` is preserved). + */ +function normalizeRemotePath(raw: string): string { + const sanitized = sanitizePath(raw) + const absolute = sanitized.startsWith('/') ? sanitized : `/${sanitized}` + const trimmed = absolute.replace(/\/+$/, '') + return trimmed === '' ? '/' : trimmed +} + +/** Joins a directory and a child name into an absolute remote path. */ +function joinRemotePath(directory: string, name: string): string { + return directory === '/' ? `/${name}` : `${directory}/${name}` +} + +/** True when `candidate` is the root path itself or lives beneath it. */ +function isWithinRoot(candidate: string, rootPath: string): boolean { + if (rootPath === '/') return true + return candidate === rootPath || candidate.startsWith(`${rootPath}/`) +} + +/** + * Resolves connection parameters from the connector's sourceConfig and the + * decrypted secret (delivered as `accessToken`). The secret is interpreted as a + * password or an OpenSSH private key depending on `authMethod`. + */ +function resolveContext(accessToken: string, sourceConfig: Record): SftpContext { + const host = ((sourceConfig.host as string) ?? '').trim() + const username = ((sourceConfig.username as string) ?? '').trim() + const rawRootPath = ((sourceConfig.rootPath as string) ?? '').trim() + const secret = (accessToken ?? '').trim() + const authMethod = ((sourceConfig.authMethod as string) ?? 'password').trim() + + if (!host) throw new Error('Missing SFTP host') + if (!username) throw new Error('Missing SFTP username') + if (!rawRootPath) throw new Error('Missing root path') + if (!secret) throw new Error('Missing SFTP password or private key') + if (!isPathSafe(rawRootPath)) { + throw new Error('Root path must not contain path traversal sequences') + } + + const port = resolveBoundedNumber(sourceConfig.port, 22, 65535) + const hostFingerprint = resolveHostFingerprint(sourceConfig.hostFingerprint) + + return { + host, + port, + username, + hostFingerprint, + password: authMethod === 'privateKey' ? undefined : secret, + privateKey: authMethod === 'privateKey' ? secret : undefined, + rootPath: normalizeRemotePath(rawRootPath), + allowedExtensions: resolveExtensions(sourceConfig.extensions), + maxDepth: resolveBoundedNumber(sourceConfig.maxDepth, DEFAULT_MAX_DEPTH, MAX_ALLOWED_DEPTH), + maxFiles: resolveBoundedNumber(sourceConfig.maxFiles, DEFAULT_MAX_FILES, MAX_ALLOWED_FILES), + } +} + +/** + * Opens an SSH/SFTP session, runs `fn` under a wall-clock deadline, and always + * tears the connection down — including on the error and timeout paths — so a + * failed sync never leaks a socket. + * + * Host validation (DNS resolution plus private/loopback/reserved-IP rejection, + * with the connection pinned to the resolved address) happens inside + * {@link createSftpConnection}, which is the SSH counterpart to the HTTP + * `secureFetchWithRetry` boundary used by the other file-storage connectors. + * When the source is configured with a host key fingerprint, that same helper + * also pins the server's host key before any credential is sent. + */ +async function withSftpSession( + ctx: SftpContext, + timeoutMs: number, + fn: (sftp: SFTPWrapper) => Promise +): Promise { + let client: Client | undefined + let timer: NodeJS.Timeout | undefined + try { + client = await createSftpConnection({ + host: ctx.host, + port: ctx.port, + username: ctx.username, + password: ctx.password, + privateKey: ctx.privateKey, + hostFingerprint: ctx.hostFingerprint, + readyTimeout: READY_TIMEOUT_MS, + keepaliveInterval: KEEPALIVE_INTERVAL_MS, + }) + const sftp = await getSftp(client) + const connection = client + const deadline = new Promise((_, reject) => { + timer = setTimeout(() => { + /** + * `destroy`, not `end`: a graceful close half-closes the socket and then + * waits for a FIN the unresponsive server that caused this timeout is + * unlikely to send, leaving the descriptor open. + */ + connection.destroy() + reject(new Error(`SFTP session exceeded ${Math.round(timeoutMs / 1000)}s`)) + }, timeoutMs) + }) + /** + * `race` keeps a rejection handler attached to `fn`, so the requests the + * timeout cancels settle without surfacing as unhandled rejections. + */ + return await Promise.race([fn(sftp), deadline]) + } finally { + if (timer) clearTimeout(timer) + client?.end() + } +} + +/** Promise wrapper around `SFTPWrapper.readdir`. */ +function readRemoteDirectory(sftp: SFTPWrapper, directory: string): Promise { + return new Promise((resolve, reject) => { + sftp.readdir(directory, (err, list) => { + if (err) reject(err) + else resolve(list) + }) + }) +} + +/** True for the SFTP status the server returns when a path no longer exists. */ +function isNotFoundError(error: unknown): boolean { + return /no such file|not found|ENOENT/i.test(getErrorMessage(error, '')) +} + +/** + * Promise wrapper around `SFTPWrapper.stat`/`lstat`, resolving null when the + * path is gone. + * + * `follow: false` issues `SSH_FXP_LSTAT`, which describes the link itself + * instead of its target. Document reads use it so a symlink planted (or swapped + * in) under the root cannot be resolved into a file outside it; the root-path + * check in `validateConfig` follows links deliberately, since a symlinked root + * directory is a legitimate configuration. + */ +function statRemotePath( + sftp: SFTPWrapper, + remotePath: string, + { follow }: { follow: boolean } +): Promise { + return new Promise((resolve, reject) => { + const stat = follow ? sftp.stat.bind(sftp) : sftp.lstat.bind(sftp) + stat(remotePath, (err, stats) => { + if (err) { + if (isNotFoundError(err)) resolve(null) + else reject(err) + } else { + resolve(stats) + } + }) + }) +} + +/** Outcome of a bounded directory walk. */ +interface WalkResult { + files: SftpFileEntry[] + /** + * True when the walk stopped short of the full tree — a cap was hit or a + * directory could not be read — meaning still-present files are missing from + * the listing and deletion reconciliation must be suppressed. + */ + truncated: boolean +} + +/** + * Walks the remote tree breadth-first from the root path, collecting files whose + * extension is indexable. Bounded on three axes — recursion depth, number of + * `readdir` calls, and number of indexable files — so a hostile or merely huge + * remote tree can never drive an unbounded traversal. + * + * Symlinks are never followed: they are the mechanism by which a remote tree can + * escape the configured root or cycle forever. `readdir` reports link entries + * with `lstat` semantics, so a symlink is classified as `symlink` here and falls + * through both the directory and the file branch. + * + * Oversized files still ride along as skipped stubs (they surface as failed rows + * in the knowledge base) and do not consume the file budget, so they are bounded + * separately by {@link MAX_LISTED_ENTRIES}. + */ +async function walkTree( + sftp: SFTPWrapper, + ctx: SftpContext, + lastSyncAt?: Date +): Promise { + const cutoffSeconds = lastSyncAt + ? Math.floor(lastSyncAt.getTime() / 1000) - INCREMENTAL_CLOCK_SKEW_SECONDS + : undefined + const files: SftpFileEntry[] = [] + const queue: Array<{ path: string; depth: number }> = [{ path: ctx.rootPath, depth: 0 }] + + let indexableCount = 0 + let directoriesRead = 0 + let truncated = false + + while (queue.length > 0) { + if (indexableCount >= ctx.maxFiles) { + truncated = true + break + } + if (directoriesRead >= MAX_DIRECTORIES) { + truncated = true + break + } + if (files.length >= MAX_LISTED_ENTRIES) { + truncated = true + break + } + + const current = queue.shift() + if (!current) break + + let entries: SftpDirEntry[] + try { + entries = await readRemoteDirectory(sftp, current.path) + directoriesRead += 1 + } catch (error) { + /** + * A directory that cannot be read may still hold live documents, so the + * listing is incomplete and must not trigger deletion reconciliation. + */ + logger.warn('Failed to read SFTP directory', { + directory: current.path, + error: toError(error).message, + }) + truncated = true + continue + } + + for (const entry of entries) { + if (entry.filename === '.' || entry.filename === '..') continue + + /** + * Directory entries come from the remote server, which is not trusted to + * return real POSIX names: a filename carrying separators or NUL bytes + * would compose a path pointing outside the configured root. + */ + if (/[/\\\0]/.test(entry.filename)) { + logger.warn('Skipping SFTP entry with an illegal filename', { directory: current.path }) + continue + } + + const childPath = joinRemotePath(current.path, entry.filename) + const type = getFileType(entry.attrs) + + if (type === 'directory') { + /** + * `maxDepth` is a configured scope filter, not a cap: files below it are + * never indexed in the first place, so their absence is not evidence of + * a partial listing. Flagging it would leave `listingCapped` set on + * every sync of any tree deeper than the limit, permanently suppressing + * deletion reconciliation. + */ + if (current.depth + 1 > ctx.maxDepth) continue + /** + * The pending queue is capped as well as the number of reads: a single + * directory holding millions of subdirectories would otherwise grow the + * queue without bound long before the read ceiling stopped the walk. + */ + if (queue.length >= MAX_DIRECTORIES) { + truncated = true + continue + } + queue.push({ path: childPath, depth: current.depth + 1 }) + continue + } + + if (type !== 'file') continue + if (!ctx.allowedExtensions.has(getExtension(entry.filename))) continue + + const size = entry.attrs.size ?? 0 + if (size <= 0) continue + + const mtime = entry.attrs.mtime ?? 0 + if (cutoffSeconds !== undefined && mtime < cutoffSeconds) continue + + if (files.length >= MAX_LISTED_ENTRIES) { + truncated = true + break + } + + const oversized = size > MAX_FILE_SIZE + if (!oversized) { + if (indexableCount >= ctx.maxFiles) { + truncated = true + break + } + indexableCount += 1 + } + + files.push({ path: childPath, directory: current.path, size, mtime }) + } + } + + return { files, truncated } +} + +/** + * Builds a metadata stub for a remote file. The hash is derived purely from + * listing metadata (path, mtime, size) so change detection never requires + * downloading content, and it is produced here for both `listDocuments` and + * `getDocument` so the two can never disagree. + */ +function fileToStub(ctx: SftpContext, entry: SftpFileEntry): ExternalDocument { + const title = entry.path.split('/').pop() || entry.path + + return { + externalId: entry.path, + title, + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: `sftp://${ctx.host}:${ctx.port}${entry.path}`, + contentHash: `sftp:${entry.path}:${entry.mtime}:${entry.size}`, + metadata: { + path: entry.path, + directory: entry.directory, + extension: getExtension(entry.path), + fileSize: entry.size, + lastModified: new Date(entry.mtime * 1000).toISOString(), + }, + } +} + +/** + * Heuristic binary check: a NUL byte in the leading bytes of a file never occurs + * in the UTF-8 text formats this connector indexes. + */ +function looksBinary(buffer: Buffer): boolean { + const end = Math.min(buffer.length, BINARY_SNIFF_BYTES) + for (let i = 0; i < end; i++) { + if (buffer[i] === 0) return true + } + return false +} + +export const sftpConnector: ConnectorConfig = { + ...sftpConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + _cursor?: string, + syncContext?: Record, + lastSyncAt?: Date + ): Promise => { + const ctx = resolveContext(accessToken, sourceConfig) + + logger.info('Listing SFTP files', { + host: ctx.host, + rootPath: ctx.rootPath, + incremental: Boolean(lastSyncAt), + }) + + const { files, truncated } = await withSftpSession(ctx, LISTING_TIMEOUT_MS, (sftp) => + walkTree(sftp, ctx, lastSyncAt) + ) + + const documents = files.map((entry) => + stubOrSkipBySize(fileToStub(ctx, entry), entry.size, MAX_FILE_SIZE) + ) + + /** + * A truncated walk means still-present files are absent from this listing. + * Without this flag the sync engine would hard-delete every document past + * the cap. + */ + if (truncated && syncContext) syncContext.listingCapped = true + + return { documents, hasMore: false } + }, + + getDocument: async ( + accessToken: string, + sourceConfig: Record, + externalId: string + ): Promise => { + const ctx = resolveContext(accessToken, sourceConfig) + + if (!isPathSafe(externalId)) { + logger.warn('Rejecting SFTP path with traversal sequences', { externalId }) + return null + } + const remotePath = normalizeRemotePath(externalId) + if (!isWithinRoot(remotePath, ctx.rootPath)) { + logger.warn('Rejecting SFTP path outside the configured root', { remotePath }) + return null + } + + return await withSftpSession(ctx, DOCUMENT_TIMEOUT_MS, async (sftp) => { + const stats = await statRemotePath(sftp, remotePath, { follow: false }) + if (!stats) return null + /** + * `lstat` above means a symlink reports as `symlink`, not as whatever it + * points at, so a link swapped in under the root between listing and + * fetch is rejected here rather than read through. + */ + if (getFileType(stats) !== 'file') return null + + const size = stats.size ?? 0 + const entry: SftpFileEntry = { + path: remotePath, + directory: remotePath.slice(0, remotePath.lastIndexOf('/')) || '/', + size, + mtime: stats.mtime ?? 0, + } + const stub = fileToStub(ctx, entry) + + if (size > MAX_FILE_SIZE) { + logger.warn('Skipping oversized SFTP file', { remotePath, size }) + return markSkipped(stub, sizeLimitSkipReason(MAX_FILE_SIZE)) + } + + let buffer: Buffer + try { + buffer = await readSftpFileCapped(sftp, remotePath, MAX_FILE_SIZE, 'SFTP connector sync') + } catch (error) { + /** + * The reported `stat` size is attacker-controlled, so a server can + * understate it and then stream unbounded data. `readSftpFileCapped` + * destroys the stream at the cap and throws, which lands here. + */ + if (isPayloadSizeLimitError(error)) { + logger.warn('SFTP file exceeded the size cap while streaming', { remotePath }) + return markSkipped(stub, sizeLimitSkipReason(MAX_FILE_SIZE)) + } + /** + * The file was removed between the listing and this read. That is an + * absence, not a failure, so it resolves null; every other error is + * rethrown so the sync records a failed document instead of silently + * dropping one. + */ + if (isNotFoundError(error)) { + logger.warn('SFTP file disappeared before it could be read', { remotePath }) + return null + } + throw error + } + + if (looksBinary(buffer)) { + logger.warn('Skipping binary SFTP file', { remotePath }) + return markSkipped(stub, 'File appears to be binary and was not indexed') + } + + const raw = buffer.toString('utf-8') + const content = HTML_EXTENSIONS.has(getExtension(remotePath)) ? htmlToPlainText(raw) : raw + if (!content.trim()) return null + + return { ...stub, content, contentDeferred: false } + }) + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + let ctx: SftpContext + try { + ctx = resolveContext(accessToken, sourceConfig) + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Invalid configuration') } + } + + try { + const valid = await withSftpSession(ctx, DOCUMENT_TIMEOUT_MS, async (sftp) => { + const stats = await statRemotePath(sftp, ctx.rootPath, { follow: true }) + if (!stats) return false + return getFileType(stats) === 'directory' + }) + if (!valid) { + return { valid: false, error: `Root path is not an accessible directory: ${ctx.rootPath}` } + } + return { valid: true } + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Failed to connect to the SFTP server') } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + if (typeof metadata.directory === 'string' && metadata.directory.length > 0) { + result.directory = metadata.directory + } + + if (typeof metadata.extension === 'string' && metadata.extension.length > 0) { + result.extension = metadata.extension + } + + if (metadata.fileSize != null) { + const num = Number(metadata.fileSize) + if (!Number.isNaN(num)) result.fileSize = num + } + + const lastModified = parseTagDate(metadata.lastModified) + if (lastModified) result.lastModified = lastModified + + return result + }, +} diff --git a/apps/sim/connectors/trello/index.ts b/apps/sim/connectors/trello/index.ts new file mode 100644 index 00000000000..aec23722680 --- /dev/null +++ b/apps/sim/connectors/trello/index.ts @@ -0,0 +1 @@ +export { trelloConnector } from '@/connectors/trello/trello' diff --git a/apps/sim/connectors/trello/meta.ts b/apps/sim/connectors/trello/meta.ts new file mode 100644 index 00000000000..24b003b8755 --- /dev/null +++ b/apps/sim/connectors/trello/meta.ts @@ -0,0 +1,71 @@ +import { TrelloIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +export const trelloConnectorMeta: ConnectorMeta = { + id: 'trello', + name: 'Trello', + description: 'Sync board cards, descriptions, checklists, and comments from Trello', + version: '1.1.0', + icon: TrelloIcon, + + auth: { + mode: 'oauth', + provider: 'trello', + requiredScopes: ['read'], + }, + + configFields: [ + { + id: 'boardSelector', + title: 'Boards', + type: 'selector', + selectorKey: 'trello.boards', + canonicalParamId: 'boardIds', + mode: 'basic', + multi: true, + required: false, + placeholder: 'Select boards (empty = all open boards)', + description: + 'Boards to sync. Leave empty to sync cards from every open board you can access.', + }, + { + id: 'boardIds', + title: 'Board IDs', + type: 'short-input', + canonicalParamId: 'boardIds', + mode: 'advanced', + multi: true, + required: false, + placeholder: 'e.g. 5f2b1c8e9a1d2b0011223344 (empty = all open boards)', + description: + 'Comma-separated board IDs (24-character hex). Leave empty to sync cards from every open board you can access.', + }, + { + id: 'cardFilter', + title: 'Cards', + type: 'dropdown', + required: false, + options: [ + { label: 'Open cards only', id: 'open' }, + { label: 'All cards (including archived)', id: 'all' }, + ], + description: 'Which cards to sync. Defaults to open cards only.', + }, + { + id: 'maxCards', + title: 'Max Cards', + type: 'short-input', + required: false, + placeholder: 'e.g. 1000 (default: unlimited)', + }, + ], + + tagDefinitions: [ + { id: 'boardName', displayName: 'Board', fieldType: 'text' }, + { id: 'listName', displayName: 'List', fieldType: 'text' }, + { id: 'labels', displayName: 'Labels', fieldType: 'text' }, + { id: 'closed', displayName: 'Archived', fieldType: 'boolean' }, + { id: 'due', displayName: 'Due Date', fieldType: 'date' }, + { id: 'lastActivity', displayName: 'Last Activity', fieldType: 'date' }, + ], +} diff --git a/apps/sim/connectors/trello/trello.ts b/apps/sim/connectors/trello/trello.ts new file mode 100644 index 00000000000..721e78215df --- /dev/null +++ b/apps/sim/connectors/trello/trello.ts @@ -0,0 +1,797 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import { env } from '@/lib/core/config/env' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import { trelloConnectorMeta } from '@/connectors/trello/meta' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { joinTagArray, parseMultiValue, parseTagDate } from '@/connectors/utils' + +const logger = createLogger('TrelloConnector') + +/** + * Trello REST API base. Every request authenticates with the Sim application + * `key` plus the user's OAuth `token` as query parameters — Trello does not + * accept a bearer header. + * @see https://developer.atlassian.com/cloud/trello/guides/rest-api/authorization/ + */ +const TRELLO_API_BASE_URL = 'https://api.trello.com/1' + +/** + * Card fields requested when listing. Kept in sync with the single-card fetch so + * the stub and the hydrated document describe the same card. `badges` carries the + * comment/checklist/attachment counters that feed the change-detection hash. + * @see https://developer.atlassian.com/cloud/trello/guides/rest-api/object-definitions/ + */ +const CARD_FIELDS = + 'id,name,desc,url,shortUrl,closed,due,dueComplete,dateLastActivity,idList,idBoard,labels,badges' + +/** + * Maximum cards requested per Trello request. The API caps long collections at + * 1000 results and documents `before`/`since` as the way to page past that cap. + * Neither the ordering of `GET /lists/{id}/cards` nor which 1000 results `limit` + * keeps is documented, so a list that needs a second request is reported as + * capped even though the extra pages are still collected. + * @see https://developer.atlassian.com/cloud/trello/guides/rest-api/api-introduction/ + */ +const CARD_PAGE_LIMIT = 1000 + +/** + * Soft per-call document target. Trello has no board-wide card cursor, so the + * listing walks board → list → card page. Emitting one list per call would burn + * one sync-engine page per list and hit its `MAX_PAGES` ceiling on workspaces + * with many small lists, which permanently truncates the listing. + */ +const CARD_TARGET_PER_CALL = 500 + +/** Maximum comment actions requested for, and rendered into, a card's content. */ +const COMMENT_LIMIT = 50 + +/** + * Upper bound on Trello requests issued by a single `listDocuments` call. The + * traversal is board → list → card page, so a workspace made of many small or + * empty lists would otherwise issue thousands of sequential requests inside one + * call and exhaust the sync task's time budget before returning a single page. + */ +const MAX_REQUESTS_PER_CALL = 40 + +/** Concurrency used when resolving names for explicitly configured boards. */ +const BOARD_LOOKUP_CONCURRENCY = 4 + +interface TrelloLabel { + id?: string + name?: string | null + color?: string | null +} + +interface TrelloBadges { + comments?: number | null + checkItems?: number | null + checkItemsChecked?: number | null + attachments?: number | null + description?: boolean | null +} + +interface TrelloAttachment { + id?: string + name?: string | null + url?: string | null +} + +interface TrelloMember { + id?: string + fullName?: string | null + username?: string | null +} + +interface TrelloCard { + id: string + name?: string | null + desc?: string | null + url?: string | null + shortUrl?: string | null + closed?: boolean | null + due?: string | null + dueComplete?: boolean | null + dateLastActivity?: string | null + idList?: string | null + idBoard?: string | null + labels?: TrelloLabel[] | null + badges?: TrelloBadges | null + board?: { id?: string | null; name?: string | null } | null + list?: { id?: string | null; name?: string | null } | null + actions?: TrelloAction[] | null + attachments?: TrelloAttachment[] | null + members?: TrelloMember[] | null +} + +interface TrelloAction { + id?: string + type?: string | null + date?: string | null + data?: { text?: string | null } | null + memberCreator?: { fullName?: string | null; username?: string | null } | null +} + +interface TrelloChecklistItem { + id?: string + name?: string | null + state?: string | null +} + +interface TrelloChecklist { + id?: string + name?: string | null + checkItems?: TrelloChecklistItem[] | null +} + +interface TrelloBoardRef { + id: string + name?: string | null +} + +interface TrelloListRef { + id: string + name?: string | null +} + +/** + * Pagination state encoded into `nextCursor`: which board of the resolved set is + * being read, which of that board's lists is being read, and — when that list + * holds more cards than one request can return — the id of the oldest card + * already emitted, replayed as the `before` bound for the next request. + */ +interface CursorState { + boardIndex: number + listIndex: number + beforeId?: string +} + +function encodeCursor(state: CursorState): string { + return Buffer.from(JSON.stringify(state), 'utf8').toString('base64url') +} + +function decodeCursor(cursor?: string): CursorState { + if (!cursor) return { boardIndex: 0, listIndex: 0 } + try { + const parsed = JSON.parse(Buffer.from(cursor, 'base64url').toString('utf8')) as + | Partial + | undefined + return { + boardIndex: Number(parsed?.boardIndex) || 0, + listIndex: Number(parsed?.listIndex) || 0, + beforeId: typeof parsed?.beforeId === 'string' ? parsed.beforeId : undefined, + } + } catch { + return { boardIndex: 0, listIndex: 0 } + } +} + +/** + * Raised when a Trello request fails, carrying the HTTP status so callers can + * distinguish a deleted card (404) from a transport or permission failure. + */ +class TrelloApiError extends Error { + constructor( + message: string, + readonly status: number + ) { + super(message) + this.name = 'TrelloApiError' + } +} + +/** + * Reads Sim's Trello application key. Trello binds every OAuth token to the key + * that issued it, so requests are unauthenticated without both halves. + */ +function requireApiKey(): string { + const apiKey = env.TRELLO_API_KEY + if (!apiKey) { + throw new Error('TRELLO_API_KEY environment variable is not set') + } + return apiKey +} + +/** + * Performs an authenticated GET against the Trello REST API and parses the JSON + * body. `params` values are appended verbatim; `key` and `token` are always added. + */ +async function trelloGet( + accessToken: string, + path: string, + params: Record = {}, + retryOptions?: Parameters[2] +): Promise { + const url = new URL(`${TRELLO_API_BASE_URL}${path}`) + for (const [name, value] of Object.entries(params)) { + url.searchParams.set(name, value) + } + url.searchParams.set('key', requireApiKey()) + url.searchParams.set('token', accessToken) + + const response = await fetchWithRetry( + url.toString(), + { method: 'GET', headers: { Accept: 'application/json' } }, + retryOptions + ) + + if (!response.ok) { + const errorText = await response.text().catch(() => '') + throw new TrelloApiError( + `Trello API error: ${response.status}${errorText ? ` — ${errorText.slice(0, 200)}` : ''}`, + response.status + ) + } + + return (await response.json()) as T +} + +/** + * Normalizes the configured card scope. Trello's cards nested resource accepts + * `all`, `closed`, `none`, `open`, and `visible`; this connector exposes only the + * two meaningful read scopes and defaults to open cards. + */ +function resolveCardFilter(sourceConfig: Record): 'open' | 'all' { + return sourceConfig.cardFilter === 'all' ? 'all' : 'open' +} + +/** + * Resolves the boards to sync. Explicitly configured board IDs are looked up so + * their names are available for tagging; otherwise every open board the member + * belongs to is enumerated. + * @see https://developer.atlassian.com/cloud/trello/rest/api-group-members/#api-members-id-boards-get + */ +async function resolveBoards( + accessToken: string, + sourceConfig: Record +): Promise { + const configured = parseMultiValue(sourceConfig.boardIds) + if (configured.length > 0) { + const resolved: TrelloBoardRef[] = [] + for (let index = 0; index < configured.length; index += BOARD_LOOKUP_CONCURRENCY) { + const batch = configured.slice(index, index + BOARD_LOOKUP_CONCURRENCY) + const settled = await Promise.all( + batch.map(async (id): Promise => { + try { + const board = await trelloGet( + accessToken, + `/boards/${encodeURIComponent(id)}`, + { fields: 'id,name' } + ) + return { id, name: board?.name ?? null } + } catch (error) { + logger.warn('Failed to resolve Trello board name', { + boardId: id, + error: toError(error).message, + }) + return { id } + } + }) + ) + resolved.push(...settled) + } + return resolved + } + + const boards = await trelloGet(accessToken, '/members/me/boards', { + filter: 'open', + fields: 'id,name', + }) + return Array.isArray(boards) ? boards.filter((board) => Boolean(board?.id)) : [] +} + +/** + * Fetches a board's lists. The list scope tracks the configured card scope so + * "All cards (including archived)" also reaches cards sitting in archived lists, + * which an `open`-only list filter would hide entirely. + * @see https://developer.atlassian.com/cloud/trello/rest/api-group-boards/#api-boards-id-lists-get + */ +async function listBoardLists( + accessToken: string, + boardId: string, + cardFilter: 'open' | 'all' +): Promise { + const lists = await trelloGet( + accessToken, + `/boards/${encodeURIComponent(boardId)}/lists`, + { filter: cardFilter === 'all' ? 'all' : 'open', fields: 'id,name' } + ) + return Array.isArray(lists) ? lists.filter((list) => Boolean(list?.id)) : [] +} + +function labelNames(card: TrelloCard): string[] { + if (!Array.isArray(card.labels)) return [] + return card.labels + .map((label) => label?.name?.trim() || label?.color?.trim() || '') + .filter((name) => name.length > 0) +} + +/** + * Change-detection hash. + * + * `dateLastActivity` alone is not a sufficient change signal: Trello documents + * neither which events bump it nor any guarantee that every content-bearing + * event does, and comment/checklist edits are rendered into the document body. + * The `badges` counters — comment count, checklist item counts, attachment count + * and the description flag — come back with the listing at no extra request cost + * and change whenever those bodies change, so they are folded in as a + * belt-and-braces signal. Must be produced identically by the stub and the + * hydrated document. + */ +function buildContentHash(card: TrelloCard): string { + const badges = card.badges ?? {} + const counters = [ + badges.comments ?? 0, + badges.checkItems ?? 0, + badges.checkItemsChecked ?? 0, + badges.attachments ?? 0, + badges.description === true ? 1 : 0, + ].join('.') + return `trello:${card.id}:${card.dateLastActivity ?? ''}:${counters}` +} + +/** + * Metadata carried on every document and fed to `mapTags`. Board and list names + * are passed in because the listing resolves them from its traversal while the + * single-card fetch resolves them from the expanded `board`/`list` objects. + */ +function cardMetadata( + card: TrelloCard, + boardName: string, + listName: string +): Record { + return { + boardId: card.idBoard ?? '', + boardName, + listId: card.idList ?? '', + listName, + labels: labelNames(card), + closed: card.closed === true, + due: card.due ?? undefined, + lastActivity: card.dateLastActivity ?? undefined, + } +} + +/** + * Builds the lightweight listing stub. Content is deferred because checklists and + * comments each require their own per-card request. + */ +function cardToStub(card: TrelloCard, boardName: string, listName: string): ExternalDocument { + return { + externalId: card.id, + title: card.name?.trim() || 'Untitled Card', + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: card.url ?? card.shortUrl ?? undefined, + contentHash: buildContentHash(card), + metadata: cardMetadata(card, boardName, listName), + } +} + +/** + * Renders a card, its description, checklists, attachments, members, and comments + * as plain text. Trello returns description and comment text as Markdown, never + * HTML, so no HTML stripping is applied. + */ +function buildCardContent( + card: TrelloCard, + boardName: string, + listName: string, + checklists: TrelloChecklist[], + comments: TrelloAction[] +): string { + const parts: string[] = [] + + if (boardName) parts.push(`Board: ${boardName}`) + if (listName) parts.push(`List: ${listName}`) + parts.push(`Card: ${card.name?.trim() || 'Untitled Card'}`) + + const labels = labelNames(card) + if (labels.length > 0) parts.push(`Labels: ${labels.join(', ')}`) + + const members = (Array.isArray(card.members) ? card.members : []) + .map((member) => member?.fullName?.trim() || member?.username?.trim() || '') + .filter((name) => name.length > 0) + if (members.length > 0) parts.push(`Members: ${members.join(', ')}`) + + if (card.due) parts.push(`Due: ${card.due}${card.dueComplete === true ? ' (complete)' : ''}`) + if (card.closed === true) parts.push('Archived: Yes') + + const description = card.desc?.trim() + if (description) { + parts.push('') + parts.push('--- Description ---') + parts.push(description) + } + + const populatedChecklists = checklists.filter( + (checklist) => (checklist.checkItems?.length ?? 0) > 0 + ) + if (populatedChecklists.length > 0) { + parts.push('') + parts.push('--- Checklists ---') + for (const checklist of populatedChecklists) { + parts.push(`${checklist.name?.trim() || 'Checklist'}:`) + for (const item of checklist.checkItems ?? []) { + const name = item.name?.trim() + if (!name) continue + parts.push(`- [${item.state === 'complete' ? 'x' : ' '}] ${name}`) + } + } + } + + const attachments = (Array.isArray(card.attachments) ? card.attachments : []).filter( + (attachment) => attachment?.name?.trim() || attachment?.url?.trim() + ) + if (attachments.length > 0) { + parts.push('') + parts.push('--- Attachments ---') + for (const attachment of attachments) { + const name = attachment.name?.trim() + const url = attachment.url?.trim() + parts.push(name && url ? `- ${name}: ${url}` : `- ${name || url}`) + } + } + + const populatedComments = comments.filter((comment) => comment.data?.text?.trim()) + if (populatedComments.length > 0) { + parts.push('') + parts.push('--- Comments ---') + for (const comment of populatedComments) { + const author = + comment.memberCreator?.fullName?.trim() || + comment.memberCreator?.username?.trim() || + 'Unknown' + parts.push(`Comment by ${author}: ${comment.data?.text?.trim()}`) + } + } + + return parts.join('\n') +} + +/** + * Reads a board's lists once per sync run. The listing walks a board across + * several pages, so re-fetching its lists on every page would multiply requests + * and risk an inconsistent traversal mid-sync. + */ +async function getCachedLists( + accessToken: string, + boardId: string, + cardFilter: 'open' | 'all', + syncContext: Record | undefined +): Promise<{ lists: TrelloListRef[]; fetched: boolean }> { + const cacheKey = `lists:${boardId}` + const cached = syncContext?.[cacheKey] as TrelloListRef[] | undefined + if (cached) return { lists: cached, fetched: false } + + const lists = await listBoardLists(accessToken, boardId, cardFilter) + if (syncContext) syncContext[cacheKey] = lists + return { lists, fetched: true } +} + +export const trelloConnector: ConnectorConfig = { + ...trelloConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + cursor?: string, + syncContext?: Record + ): Promise => { + const maxCards = sourceConfig.maxCards ? Number(sourceConfig.maxCards) : 0 + const cardFilter = resolveCardFilter(sourceConfig) + const state = decodeCursor(cursor) + + const boards = + (syncContext?.boards as TrelloBoardRef[] | undefined) ?? + (await resolveBoards(accessToken, sourceConfig)) + if (syncContext) syncContext.boards = boards + + const markCapped = () => { + if (syncContext) syncContext.listingCapped = true + } + + const documents: ExternalDocument[] = [] + const previouslyFetched = (syncContext?.totalDocsFetched as number) ?? 0 + let boardIndex = state.boardIndex + let listIndex = state.listIndex + let beforeId = state.beforeId + let hitLimit = false + let requestsUsed = 0 + + while (boardIndex < boards.length) { + if (documents.length >= CARD_TARGET_PER_CALL || requestsUsed >= MAX_REQUESTS_PER_CALL) break + + const board = boards[boardIndex] + const boardName = board.name?.trim() ?? '' + + /** + * A list-level failure would drop still-existing cards from the listing, so + * the board is skipped and the listing is marked capped rather than letting + * deletion reconciliation purge those documents. + */ + let lists: TrelloListRef[] + try { + const result = await getCachedLists(accessToken, board.id, cardFilter, syncContext) + lists = result.lists + if (result.fetched) requestsUsed += 1 + } catch (error) { + requestsUsed += 1 + logger.warn('Failed to list Trello lists for board', { + boardId: board.id, + error: toError(error).message, + }) + markCapped() + boardIndex += 1 + listIndex = 0 + beforeId = undefined + continue + } + + if (listIndex >= lists.length) { + boardIndex += 1 + listIndex = 0 + beforeId = undefined + continue + } + + const list = lists[listIndex] + const listName = list.name?.trim() ?? '' + + let cards: TrelloCard[] + requestsUsed += 1 + try { + cards = await trelloGet( + accessToken, + `/lists/${encodeURIComponent(list.id)}/cards`, + { + filter: cardFilter, + fields: CARD_FIELDS, + limit: String(CARD_PAGE_LIMIT), + ...(beforeId ? { before: beforeId } : {}), + } + ) + } catch (error) { + logger.warn('Failed to list Trello cards', { + boardId: board.id, + listId: list.id, + error: toError(error).message, + }) + markCapped() + listIndex += 1 + beforeId = undefined + continue + } + + const rawCards = (Array.isArray(cards) ? cards : []).filter((card) => Boolean(card?.id)) + const pageFull = rawCards.length >= CARD_PAGE_LIMIT + + /** + * Trello's `before` bound is a creation date derived from the id, so it is + * only second-granular and a card created in the same second as the bound + * can come back again. Card ids are Mongo ObjectIds whose leading four + * bytes are the big-endian creation timestamp, so lowercase hex ordering + * is creation ordering down to the second and total (arbitrary but stable) + * within a second — enough to drop anything not strictly older than the + * bound as already emitted. + */ + const bound = beforeId + const newCards = bound ? rawCards.filter((card) => card.id < bound) : rawCards + + let oldestId: string | undefined + for (const card of newCards) { + if (!oldestId || card.id < oldestId) oldestId = card.id + } + + let stubs = newCards.map((card) => cardToStub(card, boardName, listName)) + + let slicedByCap = false + if (maxCards > 0) { + const remaining = Math.max(0, maxCards - previouslyFetched - documents.length) + if (stubs.length > remaining) { + stubs = stubs.slice(0, remaining) + slicedByCap = true + } + } + + documents.push(...stubs) + + if (pageFull && oldestId && (!beforeId || oldestId < beforeId)) { + /** + * Paging past Trello's 1000-result ceiling with `before` is only complete + * if the truncated response is the newest 1000 of the collection. Trello + * documents neither the ordering of `GET /lists/{id}/cards` nor which + * 1000 results `limit` keeps, so a card newer than the bound but absent + * from the previous response is unreachable and would look deleted. The + * listing is therefore reported as capped for any list that needs a + * second request — the cards are still collected, but deletion + * reconciliation is withheld until a deliberate full resync. + */ + markCapped() + beforeId = oldestId + } else { + if (pageFull) { + /** + * The page came back full yet yielded no strictly-older card, so + * `before` cannot advance and the rest of this list is unreachable. + */ + logger.warn('Trello list pagination stalled; remaining cards unreachable', { + boardId: board.id, + listId: list.id, + }) + markCapped() + } + listIndex += 1 + beforeId = undefined + } + + if (maxCards > 0 && previouslyFetched + documents.length >= maxCards) { + hitLimit = true + /** + * The cap only truncates the listing when source cards were actually + * left behind. Reaching the cap exactly at source exhaustion is a + * complete listing and must stay eligible for deletion reconciliation. + */ + const moreRemains = + slicedByCap || + beforeId !== undefined || + listIndex < lists.length || + boardIndex + 1 < boards.length + if (moreRemains) markCapped() + break + } + } + + const totalFetched = previouslyFetched + documents.length + if (syncContext) syncContext.totalDocsFetched = totalFetched + + const exhausted = hitLimit || boardIndex >= boards.length + const nextCursor = exhausted ? undefined : encodeCursor({ boardIndex, listIndex, beforeId }) + + logger.info('Listing Trello cards', { + boardIndex, + boardTotal: boards.length, + listIndex, + cardCount: documents.length, + totalFetched, + }) + + return { documents, nextCursor, hasMore: !exhausted } + }, + + getDocument: async ( + accessToken: string, + _sourceConfig: Record, + externalId: string + ): Promise => { + if (!externalId) return null + + const cardId = encodeURIComponent(externalId) + + try { + const card = await trelloGet(accessToken, `/cards/${cardId}`, { + fields: CARD_FIELDS, + board: 'true', + board_fields: 'id,name', + list: 'true', + actions: 'commentCard', + actions_limit: String(COMMENT_LIMIT), + attachments: 'true', + attachment_fields: 'name,url', + members: 'true', + member_fields: 'fullName,username', + }) + + if (!card?.id) return null + + const boardName = card.board?.name?.trim() ?? '' + const listName = card.list?.name?.trim() ?? '' + + /** + * Checklists come from their own endpoint so `checkItems` are explicitly + * requested. A failure here degrades content rather than dropping the card. + */ + let checklists: TrelloChecklist[] = [] + try { + const fetched = await trelloGet( + accessToken, + `/cards/${cardId}/checklists`, + { checkItems: 'all', checkItem_fields: 'name,state', fields: 'name' } + ) + if (Array.isArray(fetched)) checklists = fetched + } catch (error) { + logger.warn('Failed to fetch Trello checklists', { + externalId, + error: toError(error).message, + }) + } + + const comments = (Array.isArray(card.actions) ? card.actions : []) + .filter((action) => action?.type === 'commentCard') + .slice(0, COMMENT_LIMIT) + + return { + externalId: card.id, + title: card.name?.trim() || 'Untitled Card', + content: buildCardContent(card, boardName, listName, checklists, comments), + contentDeferred: false, + mimeType: 'text/plain', + sourceUrl: card.url ?? card.shortUrl ?? undefined, + contentHash: buildContentHash(card), + metadata: cardMetadata(card, boardName, listName), + } + } catch (error) { + if (error instanceof TrelloApiError && error.status === 404) return null + logger.warn('Failed to get Trello card', { + externalId, + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const maxCards = sourceConfig.maxCards as string | undefined + if (maxCards && (Number.isNaN(Number(maxCards)) || Number(maxCards) < 0)) { + return { valid: false, error: 'Max cards must be a non-negative number' } + } + + const cardFilter = sourceConfig.cardFilter + if (cardFilter != null && cardFilter !== '' && cardFilter !== 'open' && cardFilter !== 'all') { + return { valid: false, error: 'Cards must be either "open" or "all"' } + } + + if (!env.TRELLO_API_KEY) { + return { + valid: false, + error: 'Trello is not configured on this deployment (missing Trello application key)', + } + } + + try { + await trelloGet(accessToken, '/members/me', { fields: 'id' }, VALIDATE_RETRY_OPTIONS) + + for (const boardId of parseMultiValue(sourceConfig.boardIds)) { + await trelloGet( + accessToken, + `/boards/${encodeURIComponent(boardId)}`, + { fields: 'id' }, + VALIDATE_RETRY_OPTIONS + ) + } + + return { valid: true } + } catch (error) { + return { valid: false, error: getErrorMessage(error, 'Failed to validate configuration') } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + if (typeof metadata.boardName === 'string' && metadata.boardName.trim()) { + result.boardName = metadata.boardName + } + + if (typeof metadata.listName === 'string' && metadata.listName.trim()) { + result.listName = metadata.listName + } + + const labels = joinTagArray(metadata.labels) + if (labels) result.labels = labels + + if (typeof metadata.closed === 'boolean') result.closed = metadata.closed + + const due = parseTagDate(metadata.due) + if (due) result.due = due + + const lastActivity = parseTagDate(metadata.lastActivity) + if (lastActivity) result.lastActivity = lastActivity + + return result + }, +} diff --git a/apps/sim/connectors/types.ts b/apps/sim/connectors/types.ts index 71ad9ad6926..cc96e68a7af 100644 --- a/apps/sim/connectors/types.ts +++ b/apps/sim/connectors/types.ts @@ -8,7 +8,18 @@ import type { SelectorKey } from '@/hooks/selectors/types' */ export type ConnectorAuthConfig = | { mode: 'oauth'; provider: OAuthService; requiredScopes?: string[] } - | { mode: 'apiKey'; label?: string; placeholder?: string } + | { + mode: 'apiKey' + label?: string + placeholder?: string + /** + * When true, the key may be left blank — the source is reachable without + * authentication (e.g. a public documentation site). A blank key is + * stored as `null` rather than an encrypted empty string, and the + * connector receives an empty access token. + */ + optional?: boolean + } /** * A single document fetched from an external source. diff --git a/apps/sim/connectors/zoho-desk/index.ts b/apps/sim/connectors/zoho-desk/index.ts new file mode 100644 index 00000000000..e36f95a4369 --- /dev/null +++ b/apps/sim/connectors/zoho-desk/index.ts @@ -0,0 +1 @@ +export { zohoDeskConnector } from '@/connectors/zoho-desk/zoho-desk' diff --git a/apps/sim/connectors/zoho-desk/meta.ts b/apps/sim/connectors/zoho-desk/meta.ts new file mode 100644 index 00000000000..b016c59a827 --- /dev/null +++ b/apps/sim/connectors/zoho-desk/meta.ts @@ -0,0 +1,180 @@ +import { ZohoDeskIcon } from '@/components/icons' +import type { ConnectorMeta } from '@/connectors/types' + +/** Default ceiling on tickets pulled per sync when the user leaves Max Tickets blank. */ +export const DEFAULT_MAX_TICKETS = 500 + +/** Default ceiling on Help Center articles pulled per sync. */ +export const DEFAULT_MAX_ARTICLES = 500 + +/** + * Zoho Desk REST hosts per data center. Zoho scopes every portal to the data + * center it was created in, and a token minted in one is rejected by the others, + * so the host is part of the connector configuration rather than a constant. + * + * Kept as a closed map (never a user-supplied host) so the OAuth token can only + * ever be sent to a Zoho-owned origin. + */ +export const ZOHO_DESK_DATA_CENTER_BASES = { + us: 'https://desk.zoho.com', + eu: 'https://desk.zoho.eu', + in: 'https://desk.zoho.in', + au: 'https://desk.zoho.com.au', + jp: 'https://desk.zoho.jp', + uk: 'https://desk.zoho.uk', + // Canada is the one region that is not a `zoho.` host: Zoho serves it from + // `zohocloud.ca` (accounts.zohocloud.ca / www.zohoapis.ca). `desk.zoho.ca` does + // not resolve at all. + ca: 'https://desk.zohocloud.ca', + sa: 'https://desk.zoho.sa', + cn: 'https://desk.zoho.com.cn', + sg: 'https://desk.zoho.sg', + ae: 'https://desk.zoho.ae', +} as const + +export type ZohoDeskDataCenter = keyof typeof ZOHO_DESK_DATA_CENTER_BASES + +/** Data center assumed when the user has not chosen one. */ +export const DEFAULT_ZOHO_DESK_DATA_CENTER: ZohoDeskDataCenter = 'us' + +export const zohoDeskConnectorMeta: ConnectorMeta = { + id: 'zoho_desk', + name: 'Zoho Desk', + description: 'Sync Help Center articles and support tickets from Zoho Desk', + version: '1.0.0', + icon: ZohoDeskIcon, + + auth: { + mode: 'oauth', + provider: 'zoho-desk', + requiredScopes: [ + 'Desk.basic.READ', + 'Desk.tickets.READ', + 'Desk.articles.READ', + 'Desk.organization.READ', + ], + }, + + configFields: [ + { + id: 'orgSelector', + title: 'Organization', + type: 'selector', + selectorKey: 'zoho_desk.organizations', + canonicalParamId: 'orgId', + mode: 'basic', + placeholder: 'Select an organization', + required: true, + description: 'Zoho Desk portal to sync from', + }, + { + id: 'orgId', + title: 'Organization ID', + type: 'short-input', + canonicalParamId: 'orgId', + mode: 'advanced', + placeholder: 'e.g. 706989253', + required: true, + description: 'Zoho Desk organization ID', + }, + { + id: 'dataCenter', + title: 'Data Center', + type: 'dropdown', + required: true, + description: 'Zoho data center your portal was created in', + options: [ + { label: 'United States (desk.zoho.com)', id: 'us' }, + { label: 'Europe (desk.zoho.eu)', id: 'eu' }, + { label: 'India (desk.zoho.in)', id: 'in' }, + { label: 'Australia (desk.zoho.com.au)', id: 'au' }, + { label: 'Japan (desk.zoho.jp)', id: 'jp' }, + { label: 'United Kingdom (desk.zoho.uk)', id: 'uk' }, + { label: 'Canada (desk.zohocloud.ca)', id: 'ca' }, + { label: 'Saudi Arabia (desk.zoho.sa)', id: 'sa' }, + { label: 'China (desk.zoho.com.cn)', id: 'cn' }, + { label: 'Singapore (desk.zoho.sg)', id: 'sg' }, + { label: 'United Arab Emirates (desk.zoho.ae)', id: 'ae' }, + ], + }, + { + id: 'contentType', + title: 'Content Type', + type: 'dropdown', + required: true, + description: 'What content to sync from Zoho Desk', + options: [ + { label: 'Articles & Tickets', id: 'both' }, + { label: 'Help Center Articles Only', id: 'articles' }, + { label: 'Support Tickets Only', id: 'tickets' }, + ], + }, + { + id: 'ticketStatus', + title: 'Ticket Status Filter', + type: 'short-input', + required: false, + placeholder: 'e.g. Open,On Hold (default: all statuses)', + description: + 'Comma-separated ticket statuses. Free text because a Zoho Desk portal can define its own statuses.', + }, + { + id: 'departmentIds', + title: 'Department IDs', + type: 'short-input', + required: false, + multi: true, + placeholder: 'e.g. 1892000000006907 (default: all departments)', + description: 'Restrict the ticket sync to specific departments', + }, + { + id: 'articleStatus', + title: 'Article Status Filter', + type: 'dropdown', + required: false, + description: 'Publishing status of the Help Center articles to sync', + options: [ + { label: 'All Statuses', id: 'all' }, + { label: 'Published', id: 'Published' }, + { label: 'Draft', id: 'Draft' }, + { label: 'Review', id: 'Review' }, + { label: 'Expired', id: 'Expired' }, + { label: 'Unpublished', id: 'Unpublished' }, + ], + }, + { + id: 'articleCategoryId', + title: 'Article Category ID', + type: 'short-input', + required: false, + placeholder: 'e.g. 4000000013240 (default: all categories)', + description: 'Restrict the article sync to a single knowledge base category', + }, + { + id: 'maxTickets', + title: 'Max Tickets', + type: 'short-input', + required: false, + placeholder: `e.g. 200 (default: ${DEFAULT_MAX_TICKETS})`, + description: 'Maximum number of tickets to sync', + }, + { + id: 'maxArticles', + title: 'Max Articles', + type: 'short-input', + required: false, + placeholder: `e.g. 200 (default: ${DEFAULT_MAX_ARTICLES})`, + description: 'Maximum number of Help Center articles to sync', + }, + ], + + tagDefinitions: [ + { id: 'contentType', displayName: 'Content Type', fieldType: 'text' }, + { id: 'status', displayName: 'Status', fieldType: 'text' }, + { id: 'priority', displayName: 'Priority', fieldType: 'text' }, + { id: 'category', displayName: 'Category', fieldType: 'text' }, + { id: 'tags', displayName: 'Tags', fieldType: 'text' }, + { id: 'updatedAt', displayName: 'Last Updated', fieldType: 'date' }, + { id: 'commentCount', displayName: 'Comment Count', fieldType: 'number' }, + ], +} diff --git a/apps/sim/connectors/zoho-desk/zoho-desk.ts b/apps/sim/connectors/zoho-desk/zoho-desk.ts new file mode 100644 index 00000000000..dbb7fc4891c --- /dev/null +++ b/apps/sim/connectors/zoho-desk/zoho-desk.ts @@ -0,0 +1,746 @@ +import { createLogger } from '@sim/logger' +import { getErrorMessage, toError } from '@sim/utils/errors' +import { fetchWithRetry, VALIDATE_RETRY_OPTIONS } from '@/lib/knowledge/documents/utils' +import type { ConnectorConfig, ExternalDocument, ExternalDocumentList } from '@/connectors/types' +import { htmlToPlainText, joinTagArray, parseMultiValue, parseTagDate } from '@/connectors/utils' +import { + DEFAULT_MAX_ARTICLES, + DEFAULT_MAX_TICKETS, + DEFAULT_ZOHO_DESK_DATA_CENTER, + ZOHO_DESK_DATA_CENTER_BASES, + type ZohoDeskDataCenter, + zohoDeskConnectorMeta, +} from '@/connectors/zoho-desk/meta' + +const logger = createLogger('ZohoDeskConnector') + +/** Zoho caps the article list at 50 per request and the ticket list at 100. */ +const ARTICLES_PAGE_SIZE = 50 +const TICKETS_PAGE_SIZE = 100 + +/** Zoho caps `/conversations` at 200 entries per request. */ +const CONVERSATIONS_PAGE_SIZE = 200 + +/** + * Highest `from` index any Desk list API accepts. Zoho allows paginating over at + * most 5000 records and answers `from >= 5000` with HTTP 422 + * `UNPROCESSABLE_ENTITY: The value passed for field 'from' exceeds the range of + * '0-4999'` — an error, not an empty page. Walking into it would abort the whole + * listing, so the drain stops at the ceiling and reports the listing as capped. + */ +const MAX_LIST_OFFSET = 4999 + +/** Upper bound on conversation entries folded into one ticket document. */ +const MAX_CONVERSATION_ENTRIES = 400 + +/** + * Upper bound on threads hydrated to their full body for one ticket. Beyond this + * the truncated `summary` from `/conversations` is used, so a thousand-message + * ticket cannot turn a single `getDocument` call into a thousand HTTP requests. + */ +const MAX_HYDRATED_THREADS = 50 + +interface ZohoDeskArticleSummary { + id: string + title?: string + status?: string + locale?: string + summary?: string + modifiedTime?: string + createdTime?: string + permalink?: string + portalUrl?: string + webUrl?: string + categoryId?: string + category?: { id?: string; name?: string } + tags?: string[] + isTrashed?: boolean +} + +interface ZohoDeskArticleDetail extends ZohoDeskArticleSummary { + answer?: string +} + +interface ZohoDeskTicket { + id: string + ticketNumber?: string + subject?: string + description?: string + status?: string + statusType?: string + priority?: string + category?: string + subCategory?: string + classification?: string + channel?: string + departmentId?: string + createdTime?: string + /** Present on `GET /tickets/{id}` only — the list projection omits it. */ + modifiedTime?: string + closedTime?: string | null + customerResponseTime?: string | null + threadCount?: string + commentCount?: string + resolution?: string | null + webUrl?: string + isTrashed?: boolean +} + +interface ZohoDeskConversationEntry { + id: string + type?: string + summary?: string + content?: string + contentType?: string + createdTime?: string + commentedTime?: string + visibility?: string + isPublic?: boolean + direction?: string + channel?: string + author?: { name?: string; email?: string; type?: string } + commenter?: { name?: string; email?: string; type?: string } +} + +interface ZohoDeskThreadDetail { + id: string + content?: string + plainText?: string + contentType?: string + summary?: string +} + +interface ZohoDeskOrganization { + id: string + companyName?: string +} + +/** + * Resolves the Zoho Desk REST base (`{deskHost}/api/v1`) for the configured data + * center. The host comes from a closed map, never from user input, so the OAuth + * token can only reach a Zoho-owned origin. + * + * @throws {Error} when the configured data center is not recognized. + */ +function resolveApiBase(sourceConfig: Record): string { + const raw = typeof sourceConfig.dataCenter === 'string' ? sourceConfig.dataCenter.trim() : '' + const key = (raw || DEFAULT_ZOHO_DESK_DATA_CENTER) as ZohoDeskDataCenter + const host = ZOHO_DESK_DATA_CENTER_BASES[key] + if (!host) { + throw new Error(`Unsupported Zoho Desk data center: ${raw}`) + } + return `${host}/api/v1` +} + +/** + * Reads the required organization ID from the connector config. + * + * @throws {Error} when it is missing. + */ +function requireOrgId(sourceConfig: Record): string { + const orgId = typeof sourceConfig.orgId === 'string' ? sourceConfig.orgId.trim() : '' + if (!orgId) { + throw new Error('Organization ID is required') + } + return orgId +} + +/** + * Reads an optional positive integer cap, falling back to `fallback` when unset + * and throwing when it is present but not a positive number. + */ +function resolveMax(value: unknown, fallback: number, label: string): number { + if (value === undefined || value === null || value === '') return fallback + const parsed = Number(value) + if (!Number.isFinite(parsed) || parsed < 1) { + throw new Error(`${label} must be a positive number`) + } + return Math.floor(parsed) +} + +/** + * Performs an authenticated GET against the Zoho Desk API. + * + * Zoho answers an empty collection with `204 No Content` (which has no JSON body), + * so that case is normalized to an empty object rather than left to throw. + */ +async function deskGet( + url: string, + accessToken: string, + orgId: string | undefined, + retryOptions?: Parameters[2] +): Promise> { + const headers: Record = { + Authorization: `Zoho-oauthtoken ${accessToken}`, + Accept: 'application/json', + } + if (orgId) headers.orgId = orgId + + const response = await fetchWithRetry(url, { method: 'GET', headers }, retryOptions) + + if (response.status === 204) return {} + + const body = (await response.json().catch(() => ({}))) as Record + + if (!response.ok) { + const message = + typeof body.message === 'string' && body.message.trim() + ? body.message + : typeof body.errorCode === 'string' && body.errorCode.trim() + ? body.errorCode + : `Zoho Desk API HTTP error: ${response.status}` + throw new Error(message) + } + + return body +} + +/** Reads the `data` array Zoho wraps every list response in. */ +function readDataArray(body: Record): T[] { + return Array.isArray(body.data) ? (body.data as T[]) : [] +} + +/** + * Renders a Zoho content value as plain text. Zoho spells the HTML discriminator + * both as `html` (comments) and `text/html` (threads), so both are matched; a + * value without an HTML content type is passed through untouched so genuinely + * plain bodies are not mangled by tag stripping. + */ +function toPlainText(content: string | undefined, contentType: string | undefined): string { + if (!content) return '' + const normalized = contentType?.trim().toLowerCase() ?? '' + const isHtml = normalized === 'html' || normalized.startsWith('text/html') + return isHtml ? htmlToPlainText(content) : content +} + +/** + * Drains a Zoho `from`/`limit` paginated list into a single array, stopping at + * `max`. + * + * The final page asks only for the records still needed, so the walk never pulls + * more than `max`. When the cap is reached on an exactly-full page the source may + * or may not hold more, and Zoho's list responses carry no "has more" marker, so + * one extra single-record probe settles it. + * + * @param probeForMore when false the cap is treated as non-truncating. Only for + * callers that ignore `truncated` (per-ticket conversation folding), so they do + * not pay for the probe. + * @returns the collected items and whether the source still had more to give + * when the cap stopped the walk — the caller must surface that as + * `syncContext.listingCapped` so the sync engine does not hard-delete every + * document past the cap. + */ +async function drainPaginated( + fetchPage: (from: number, limit: number) => Promise, + pageSize: number, + max: number, + probeForMore = true +): Promise<{ items: T[]; truncated: boolean }> { + const items: T[] = [] + let from = 0 + + while (items.length < max) { + // Zoho refuses `from >= 5000` outright, so the walk stops one page short of + // the ceiling and declares itself capped rather than throwing a 422. + if (from > MAX_LIST_OFFSET) return { items, truncated: true } + const limit = Math.min(pageSize, max - items.length) + const page = await fetchPage(from, limit) + if (page.length === 0) return { items, truncated: false } + + items.push(...page) + + // A short page means the source is exhausted, which is exactly the case the + // sync engine must be allowed to reconcile deletions against. + if (page.length < limit) return { items, truncated: false } + from += page.length + } + + if (!probeForMore) return { items, truncated: false } + if (from > MAX_LIST_OFFSET) return { items, truncated: true } + + const probe = await fetchPage(from, 1) + return { items, truncated: probe.length > 0 } +} + +/** Lists Help Center articles, newest-first-stable on `createdTime`. */ +async function fetchArticles( + apiBase: string, + accessToken: string, + orgId: string, + options: { status?: string; categoryId?: string; max: number } +): Promise<{ items: ZohoDeskArticleSummary[]; truncated: boolean }> { + return drainPaginated( + async (from, limit) => { + const query = new URLSearchParams({ + from: String(from), + limit: String(limit), + // Stable ordering matters across pages: sorting on modifiedTime would + // reshuffle rows mid-walk as the portal is edited, silently skipping some. + sortBy: 'createdTime', + }) + if (options.status && options.status !== 'all') query.set('status', options.status) + if (options.categoryId) query.set('categoryId', options.categoryId) + const body = await deskGet(`${apiBase}/articles?${query.toString()}`, accessToken, orgId) + return readDataArray(body) + }, + ARTICLES_PAGE_SIZE, + options.max + ) +} + +/** Lists tickets, optionally filtered by status and department. */ +async function fetchTickets( + apiBase: string, + accessToken: string, + orgId: string, + options: { status?: string; departmentIds: string[]; max: number } +): Promise<{ items: ZohoDeskTicket[]; truncated: boolean }> { + return drainPaginated( + async (from, limit) => { + const query = new URLSearchParams({ + from: String(from), + limit: String(limit), + sortBy: 'createdTime', + }) + if (options.status) query.set('status', options.status) + if (options.departmentIds.length > 0) { + query.set('departmentIds', options.departmentIds.join(',')) + } + const body = await deskGet(`${apiBase}/tickets?${query.toString()}`, accessToken, orgId) + return readDataArray(body) + }, + TICKETS_PAGE_SIZE, + options.max + ) +} + +/** Lists the threads and comments recorded on a ticket, oldest page first. */ +async function fetchConversations( + apiBase: string, + accessToken: string, + orgId: string, + ticketId: string +): Promise { + const { items } = await drainPaginated( + async (from, limit) => { + const query = new URLSearchParams({ from: String(from), limit: String(limit) }) + const body = await deskGet( + `${apiBase}/tickets/${encodeURIComponent(ticketId)}/conversations?${query.toString()}`, + accessToken, + orgId + ) + return readDataArray(body) + }, + CONVERSATIONS_PAGE_SIZE, + MAX_CONVERSATION_ENTRIES, + false + ) + return items +} + +/** + * Fetches a thread's full body. `/conversations` returns only a truncated + * `summary` for threads, so the message text has to be read per thread. + */ +async function fetchThread( + apiBase: string, + accessToken: string, + orgId: string, + ticketId: string, + threadId: string +): Promise { + const body = await deskGet( + `${apiBase}/tickets/${encodeURIComponent(ticketId)}/threads/${encodeURIComponent(threadId)}?include=plainText`, + accessToken, + orgId + ) + // double-cast-allowed: deskGet returns an untyped JSON record; `id` is checked above + return typeof body.id === 'string' ? (body as unknown as ZohoDeskThreadDetail) : null +} + +/** Formats one conversation entry as a labelled block of plain text. */ +function formatConversationEntry( + entry: ZohoDeskConversationEntry, + hydrated: ZohoDeskThreadDetail | null +): string { + const isComment = entry.type === 'comment' + const person = isComment ? entry.commenter : entry.author + const timestamp = entry.commentedTime || entry.createdTime || '' + const visibility = isComment + ? entry.isPublic === false + ? 'Internal' + : 'Public' + : entry.visibility || 'public' + + const body = hydrated + ? hydrated.plainText?.trim() || + toPlainText(hydrated.content, hydrated.contentType) || + hydrated.summary || + '' + : toPlainText(entry.content, entry.contentType) || entry.summary || '' + + const label = isComment ? 'Comment' : 'Thread' + const author = person?.name || person?.email || 'Unknown' + return `\n[${timestamp}] ${label} (${visibility}) — ${author}:\n${body}` +} + +/** Folds a ticket and its conversation into one plain-text document body. */ +function formatTicketContent(ticket: ZohoDeskTicket, conversation: string[]): string { + const parts: string[] = [] + + if (ticket.subject) parts.push(`Subject: ${ticket.subject}`) + if (ticket.status) parts.push(`Status: ${ticket.status}`) + if (ticket.priority) parts.push(`Priority: ${ticket.priority}`) + if (ticket.classification) parts.push(`Classification: ${ticket.classification}`) + if (ticket.category) parts.push(`Category: ${ticket.category}`) + if (ticket.channel) parts.push(`Channel: ${ticket.channel}`) + if (ticket.createdTime) parts.push(`Created: ${ticket.createdTime}`) + if (ticket.modifiedTime) parts.push(`Updated: ${ticket.modifiedTime}`) + + if (ticket.description) { + parts.push('') + parts.push('--- Description ---') + // Zoho ships no content-type discriminator for `description`, and it can be + // either HTML or plain text, so strip only when markup is actually present. + parts.push( + /<[a-z][\s\S]*>/i.test(ticket.description) + ? htmlToPlainText(ticket.description) + : ticket.description + ) + } + + // The agent's resolution note is the single most reusable answer on a ticket, + // and `GET /tickets/{id}` is the only place it is returned. + if (ticket.resolution) { + parts.push('') + parts.push('--- Resolution ---') + parts.push( + /<[a-z][\s\S]*>/i.test(ticket.resolution) + ? htmlToPlainText(ticket.resolution) + : ticket.resolution + ) + } + + if (conversation.length > 0) { + parts.push('') + parts.push('--- Conversation ---') + parts.push(...conversation) + } + + return parts.join('\n') +} + +/** + * Metadata-only change indicator for a ticket. + * + * Deliberately excludes `modifiedTime`: Zoho returns it on `GET /tickets/{id}` + * but NOT on the `GET /tickets` list projection, so mixing it in would make the + * listing stub and the hydrated document hash differently and re-index every + * ticket on every sync forever. Every field below appears on both responses. + * + * `createdTime` alone would never move, so the fields that do track activity ride + * along: the conversation counters, the workflow state, and the last customer + * response. + */ +function ticketContentHash(ticket: ZohoDeskTicket): string { + return [ + 'zoho_desk:ticket', + ticket.id, + ticket.createdTime ?? '', + ticket.threadCount ?? '', + ticket.commentCount ?? '', + ticket.status ?? '', + ticket.priority ?? '', + ticket.closedTime ?? '', + ticket.customerResponseTime ?? '', + ].join(':') +} + +/** + * Builds the deferred stub for a ticket. Content is resolved lazily in + * `getDocument` because the conversation needs one call per ticket (plus one per + * thread), which would exhaust the listing pass's time budget. + */ +function ticketToStub(ticket: ZohoDeskTicket): ExternalDocument { + const number = ticket.ticketNumber ? `#${ticket.ticketNumber}` : ticket.id + return { + externalId: `ticket-${ticket.id}`, + title: `Ticket ${number}: ${ticket.subject || 'Untitled'}`, + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: ticket.webUrl, + contentHash: ticketContentHash(ticket), + metadata: { + type: 'ticket', + ticketId: ticket.id, + ticketNumber: ticket.ticketNumber, + status: ticket.status, + priority: ticket.priority, + category: ticket.category, + departmentId: ticket.departmentId, + commentCount: Number(ticket.commentCount ?? 0), + updatedAt: ticket.modifiedTime || ticket.createdTime, + createdAt: ticket.createdTime, + }, + } +} + +/** Metadata-only change indicator for an article. */ +function articleContentHash(article: ZohoDeskArticleSummary): string { + return `zoho_desk:article:${article.id}:${article.modifiedTime || article.createdTime || ''}` +} + +/** + * Builds the deferred stub for an article. The article list projection carries + * only a truncated `summary`; the full `answer` body comes from the per-article + * endpoint, so content is deferred. + */ +function articleToStub(article: ZohoDeskArticleSummary): ExternalDocument { + return { + externalId: `article-${article.id}`, + title: article.title || 'Untitled', + content: '', + contentDeferred: true, + mimeType: 'text/plain', + sourceUrl: article.portalUrl || article.webUrl, + contentHash: articleContentHash(article), + metadata: { + type: 'article', + articleId: article.id, + status: article.status, + locale: article.locale, + category: article.category?.name, + categoryId: article.categoryId || article.category?.id, + tags: article.tags, + updatedAt: article.modifiedTime || article.createdTime, + createdAt: article.createdTime, + }, + } +} + +export const zohoDeskConnector: ConnectorConfig = { + ...zohoDeskConnectorMeta, + + listDocuments: async ( + accessToken: string, + sourceConfig: Record, + _cursor?: string, + syncContext?: Record + ): Promise => { + const apiBase = resolveApiBase(sourceConfig) + const orgId = requireOrgId(sourceConfig) + const contentType = (sourceConfig.contentType as string) || 'both' + + const documents: ExternalDocument[] = [] + let truncated = false + + if (contentType === 'articles' || contentType === 'both') { + const max = resolveMax(sourceConfig.maxArticles, DEFAULT_MAX_ARTICLES, 'Max articles') + const articleStatus = + typeof sourceConfig.articleStatus === 'string' + ? sourceConfig.articleStatus.trim() + : undefined + const categoryId = + typeof sourceConfig.articleCategoryId === 'string' + ? sourceConfig.articleCategoryId.trim() + : undefined + + const articles = await fetchArticles(apiBase, accessToken, orgId, { + status: articleStatus, + categoryId: categoryId || undefined, + max, + }) + logger.info(`Fetched ${articles.items.length} articles from Zoho Desk`, { orgId }) + truncated = truncated || articles.truncated + + for (const article of articles.items) { + if (article.isTrashed) continue + documents.push(articleToStub(article)) + } + } + + if (contentType === 'tickets' || contentType === 'both') { + const max = resolveMax(sourceConfig.maxTickets, DEFAULT_MAX_TICKETS, 'Max tickets') + const ticketStatus = + typeof sourceConfig.ticketStatus === 'string' ? sourceConfig.ticketStatus.trim() : undefined + + const tickets = await fetchTickets(apiBase, accessToken, orgId, { + status: ticketStatus || undefined, + departmentIds: parseMultiValue(sourceConfig.departmentIds), + max, + }) + logger.info(`Fetched ${tickets.items.length} tickets from Zoho Desk`, { orgId }) + truncated = truncated || tickets.truncated + + for (const ticket of tickets.items) { + if (ticket.isTrashed) continue + documents.push(ticketToStub(ticket)) + } + } + + // A capped listing is not the full source set. Without this flag the sync + // engine treats every document past the cap as deleted and removes it. + if (truncated && syncContext) { + syncContext.listingCapped = true + } + + return { documents, hasMore: false } + }, + + getDocument: async ( + accessToken: string, + sourceConfig: Record, + externalId: string + ): Promise => { + try { + const apiBase = resolveApiBase(sourceConfig) + const orgId = requireOrgId(sourceConfig) + + if (externalId.startsWith('article-')) { + const articleId = externalId.slice('article-'.length) + const body = await deskGet( + `${apiBase}/articles/${encodeURIComponent(articleId)}`, + accessToken, + orgId + ) + if (typeof body.id !== 'string') return null + // double-cast-allowed: deskGet returns an untyped JSON record; `id` is checked above + const article = body as unknown as ZohoDeskArticleDetail + if (article.isTrashed) return null + + const content = htmlToPlainText(article.answer || '') + if (!content.trim()) return null + + return { ...articleToStub(article), content, contentDeferred: false } + } + + if (externalId.startsWith('ticket-')) { + const ticketId = externalId.slice('ticket-'.length) + const body = await deskGet( + `${apiBase}/tickets/${encodeURIComponent(ticketId)}`, + accessToken, + orgId + ) + if (typeof body.id !== 'string') return null + // double-cast-allowed: deskGet returns an untyped JSON record; `id` is checked above + const ticket = body as unknown as ZohoDeskTicket + if (ticket.isTrashed) return null + + const entries = await fetchConversations(apiBase, accessToken, orgId, ticketId) + if (entries.length >= MAX_CONVERSATION_ENTRIES) { + logger.warn('Zoho Desk ticket conversation truncated at the entry cap', { + ticketId, + cap: MAX_CONVERSATION_ENTRIES, + }) + } + const blocks: string[] = [] + let hydratedThreads = 0 + + for (const entry of entries) { + let hydrated: ZohoDeskThreadDetail | null = null + if (entry.type === 'thread' && hydratedThreads < MAX_HYDRATED_THREADS) { + try { + hydrated = await fetchThread(apiBase, accessToken, orgId, ticketId, entry.id) + hydratedThreads += 1 + } catch (error) { + logger.warn('Failed to fetch Zoho Desk thread body; using summary', { + ticketId, + threadId: entry.id, + error: getErrorMessage(error), + }) + } + } + blocks.push(formatConversationEntry(entry, hydrated)) + } + + const content = formatTicketContent(ticket, blocks) + if (!content.trim()) return null + + return { ...ticketToStub(ticket), content, contentDeferred: false } + } + + return null + } catch (error) { + logger.warn('Failed to get Zoho Desk document', { + externalId, + error: toError(error).message, + }) + return null + } + }, + + validateConfig: async ( + accessToken: string, + sourceConfig: Record + ): Promise<{ valid: boolean; error?: string }> => { + const orgId = typeof sourceConfig.orgId === 'string' ? sourceConfig.orgId.trim() : '' + if (!orgId) { + return { valid: false, error: 'Organization ID is required' } + } + + const dataCenter = + typeof sourceConfig.dataCenter === 'string' ? sourceConfig.dataCenter.trim() : '' + if (!dataCenter || !(dataCenter in ZOHO_DESK_DATA_CENTER_BASES)) { + return { valid: false, error: 'A supported Zoho Desk data center is required' } + } + + const contentType = + typeof sourceConfig.contentType === 'string' ? sourceConfig.contentType.trim() : '' + if (!contentType) { + return { valid: false, error: 'Content type is required' } + } + + try { + resolveMax(sourceConfig.maxTickets, DEFAULT_MAX_TICKETS, 'Max tickets') + resolveMax(sourceConfig.maxArticles, DEFAULT_MAX_ARTICLES, 'Max articles') + } catch (error) { + return { valid: false, error: toError(error).message } + } + + try { + const apiBase = resolveApiBase(sourceConfig) + // `/organizations` is the one Desk endpoint that is not org-scoped, so it + // verifies the token and the data center, and confirms the configured + // organization is actually one this credential can reach. + const body = await deskGet( + `${apiBase}/organizations`, + accessToken, + undefined, + VALIDATE_RETRY_OPTIONS + ) + const organizations = readDataArray(body) + if (!organizations.some((organization) => String(organization.id) === orgId)) { + return { + valid: false, + error: `Organization ${orgId} is not accessible with this credential in the selected data center`, + } + } + return { valid: true } + } catch (error) { + return { valid: false, error: toError(error).message || 'Failed to validate configuration' } + } + }, + + mapTags: (metadata: Record): Record => { + const result: Record = {} + + if (typeof metadata.type === 'string') result.contentType = metadata.type + if (typeof metadata.status === 'string') result.status = metadata.status + if (typeof metadata.priority === 'string') result.priority = metadata.priority + if (typeof metadata.category === 'string') result.category = metadata.category + + const tags = joinTagArray(metadata.tags) + if (tags) result.tags = tags + + const updatedAt = parseTagDate(metadata.updatedAt) + if (updatedAt) result.updatedAt = updatedAt + + if (metadata.commentCount != null) { + const commentCount = Number(metadata.commentCount) + if (!Number.isNaN(commentCount)) result.commentCount = commentCount + } + + return result + }, +} diff --git a/apps/sim/lib/copilot/tools/server/knowledge/knowledge-base.ts b/apps/sim/lib/copilot/tools/server/knowledge/knowledge-base.ts index 2e97eaf547d..90115a49ce0 100644 --- a/apps/sim/lib/copilot/tools/server/knowledge/knowledge-base.ts +++ b/apps/sim/lib/copilot/tools/server/knowledge/knowledge-base.ts @@ -863,14 +863,6 @@ export const knowledgeBaseServerTool: BaseServerTool = { ...(args.sourceConfig ?? {}) } if (args.disabledTagIds?.length) { sourceConfig.disabledTagIds = args.disabledTagIds diff --git a/apps/sim/lib/knowledge/application/connectors.ts b/apps/sim/lib/knowledge/application/connectors.ts index c7e501d9dc6..536c905e8f4 100644 --- a/apps/sim/lib/knowledge/application/connectors.ts +++ b/apps/sim/lib/knowledge/application/connectors.ts @@ -173,12 +173,16 @@ async function validateConnectorSourceConfig(input: { let accessToken: string | null = null if (connectorConfig.auth.mode === 'apiKey') { if (!input.connector.encryptedApiKey) { - return { - message: 'API key not found. Please reconfigure the connector.', - errorCode: 'validation', + if (!connectorConfig.auth.optional) { + return { + message: 'API key not found. Please reconfigure the connector.', + errorCode: 'validation', + } } + accessToken = '' + } else { + accessToken = (await decryptApiKey(input.connector.encryptedApiKey)).decrypted } - accessToken = (await decryptApiKey(input.connector.encryptedApiKey)).decrypted } else { if (!input.connector.credentialId) { return { diff --git a/apps/sim/lib/knowledge/connectors/sync-engine.test.ts b/apps/sim/lib/knowledge/connectors/sync-engine.test.ts index b4d3a162480..2036c417664 100644 --- a/apps/sim/lib/knowledge/connectors/sync-engine.test.ts +++ b/apps/sim/lib/knowledge/connectors/sync-engine.test.ts @@ -4,6 +4,11 @@ import { authOAuthUtilsMock } from '@sim/testing' import { generateShortId } from '@sim/utils/id' import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + classifySuspectListing, + evaluateListingSafety, + type PreviousListingObservation, +} from '@/lib/knowledge/connectors/sync-engine' vi.mock('drizzle-orm', () => ({ and: vi.fn(), @@ -23,7 +28,7 @@ vi.mock('@/background/knowledge-connector-sync', () => ({ knowledgeConnectorSync: { trigger: vi.fn() }, })) -const mockMapTags = vi.fn() +const { mockMapTags } = vi.hoisted(() => ({ mockMapTags: vi.fn() })) vi.mock('@/connectors/registry.server', () => ({ CONNECTOR_REGISTRY: { @@ -537,3 +542,88 @@ describe('chunkOpsByByteBudget', () => { expect(chunks).toHaveLength(1) }) }) + +describe('classifySuspectListing', () => { + it('trusts a healthy listing', () => { + expect(classifySuspectListing(100, 100)).toBeNull() + expect(classifySuspectListing(90, 100)).toBeNull() + }) + + it('flags an empty listing against a real corpus', () => { + expect(classifySuspectListing(0, 3)).toBe('empty') + expect(classifySuspectListing(0, 10_000)).toBe('empty') + }) + + it('ignores an empty listing on a trivially small corpus', () => { + expect(classifySuspectListing(0, 0)).toBeNull() + expect(classifySuspectListing(0, 2)).toBeNull() + }) + + it('flags a near-total collapse on a large corpus', () => { + expect(classifySuspectListing(3, 10_000)).toBe('collapsed') + expect(classifySuspectListing(49, 500)).toBe('collapsed') + }) + + it('allows an ordinary bulk deletion through', () => { + expect(classifySuspectListing(1000, 10_000)).toBeNull() + expect(classifySuspectListing(1, 8)).toBeNull() + expect(classifySuspectListing(4, 49)).toBeNull() + }) +}) + +describe('evaluateListingSafety', () => { + const previous = ( + listedCount: number, + ownedCount: number, + trustworthy = true + ): PreviousListingObservation => ({ listedCount, ownedCount, trustworthy }) + + it('leaves a healthy listing untouched', () => { + expect(evaluateListingSafety(100, 100, null, undefined)).toEqual({ + reason: null, + blocked: false, + corroborated: false, + }) + }) + + it('blocks the first suspect empty listing', () => { + expect(evaluateListingSafety(0, 500, previous(500, 500), undefined)).toEqual({ + reason: 'empty', + blocked: true, + corroborated: false, + }) + }) + + it('blocks when there is no previous completed sync to corroborate', () => { + expect(evaluateListingSafety(0, 500, null, undefined).blocked).toBe(true) + }) + + it('reconciles once a consecutive sync sees the same empty listing', () => { + expect(evaluateListingSafety(0, 500, previous(0, 500), undefined)).toEqual({ + reason: 'empty', + blocked: false, + corroborated: true, + }) + }) + + it('refuses to be corroborated by a possibly-incremental previous run', () => { + expect(evaluateListingSafety(0, 500, previous(0, 500, false), undefined).blocked).toBe(true) + }) + + it('blocks then allows a proportional collapse across two syncs', () => { + expect(evaluateListingSafety(3, 10_000, previous(10_000, 10_000), undefined).blocked).toBe(true) + expect(evaluateListingSafety(3, 10_000, previous(2, 10_000), undefined)).toEqual({ + reason: 'collapsed', + blocked: false, + corroborated: true, + }) + }) + + it('lets an explicit fullSync override the guard', () => { + expect(evaluateListingSafety(0, 500, null, true)).toEqual({ + reason: 'empty', + blocked: false, + corroborated: false, + }) + }) +}) diff --git a/apps/sim/lib/knowledge/connectors/sync-engine.ts b/apps/sim/lib/knowledge/connectors/sync-engine.ts index d3197f88362..ccd804a50df 100644 --- a/apps/sim/lib/knowledge/connectors/sync-engine.ts +++ b/apps/sim/lib/knowledge/connectors/sync-engine.ts @@ -10,7 +10,7 @@ import { createLogger } from '@sim/logger' import { getErrorMessage, toError } from '@sim/utils/errors' import { generateId } from '@sim/utils/id' import { randomInt } from '@sim/utils/random' -import { and, eq, gt, inArray, isNotNull, isNull, lt, ne, or, sql } from 'drizzle-orm' +import { and, desc, eq, gt, inArray, isNotNull, isNull, lt, ne, or, sql } from 'drizzle-orm' import { decryptApiKey } from '@/lib/api-key/crypto' import { assertBillingAttributionSnapshot, @@ -246,6 +246,148 @@ export function shouldReconcileDeletions( return !syncContext?.listingCapped || Boolean(fullSync) } +/** + * Minimum number of documents a connector must still own before an empty + * listing is treated as suspect. Below it, an empty listing is far more likely + * to be a genuinely emptied source than a broken one, the blast radius of + * reconciling is a handful of documents, and any ratio-based judgement is + * statistically meaningless. + */ +const SUSPECT_LISTING_MIN_OWNED_DOCS = 3 +/** + * Minimum owned-document count before the proportional (collapse) guard + * applies. A source can legitimately shrink hard when it is small — going from + * 8 documents to 1 is ordinary editing — so the collapse guard only engages on + * corpora large enough that a near-total disappearance in a single sync is + * implausible without an upstream fault. + */ +const SUSPECT_COLLAPSE_MIN_OWNED_DOCS = 50 +/** + * A listing covering less than this fraction of the documents the connector + * still owns is treated as suspect. Deliberately far below any plausible + * bulk edit (10% means 10,000 documents collapsing to under 1,000) so normal + * housekeeping never trips it, while the partial-outage shapes seen in the + * wild — an auth wall or an interstitial served for most of a source — do. + */ +const SUSPECT_COLLAPSE_MAX_RATIO = 0.1 + +/** Why a listing is considered untrustworthy evidence of deletion. */ +export type SuspectListingReason = 'empty' | 'collapsed' + +/** + * A prior sync's listing, reconstructed from its sync-log counters. + * + * `trustworthy` is false when that run could have been an incremental listing: + * an incremental run that observed no changes is indistinguishable from a full + * run that observed nothing, and treating the former as corroboration would let + * a single bad listing confirm itself. + */ +export interface PreviousListingObservation { + listedCount: number + ownedCount: number + trustworthy: boolean +} + +/** + * Classifies a listing as untrustworthy evidence that documents were deleted. + * + * A connector that returns nothing (or almost nothing) while the knowledge base + * still holds a real corpus for it is far more likely to be broken than to be + * reporting a genuinely emptied source: observed causes include an HTTP 200 + * interstitial served instead of an index, and a source moved behind auth. + * Neither surfaces as an error, so the sync looks clean and the listing looks + * authoritative. + */ +export function classifySuspectListing( + listedCount: number, + ownedCount: number +): SuspectListingReason | null { + if (ownedCount < SUSPECT_LISTING_MIN_OWNED_DOCS) return null + if (listedCount === 0) return 'empty' + if ( + ownedCount >= SUSPECT_COLLAPSE_MIN_OWNED_DOCS && + listedCount < ownedCount * SUSPECT_COLLAPSE_MAX_RATIO + ) { + return 'collapsed' + } + return null +} + +/** + * Decides whether a suspect listing may still reconcile deletions. + * + * A suspect listing is only acted on once the *same* observation repeats on a + * consecutive sync, so a single transient upstream fault can never remove + * documents — not even reversibly, since a soft delete hides them from search + * immediately. A genuinely emptied source keeps reconciling: its second sync + * corroborates the first, tombstones everything, and the third sync completes + * the existing two-strike purge. + * + * A forced `fullSync` overrides the guard, matching its existing meaning + * elsewhere here — an explicit human request to reconcile against this listing + * right now. + */ +export function evaluateListingSafety( + listedCount: number, + ownedCount: number, + previous: PreviousListingObservation | null, + fullSync: boolean | undefined +): { reason: SuspectListingReason | null; blocked: boolean; corroborated: boolean } { + const reason = classifySuspectListing(listedCount, ownedCount) + if (!reason) return { reason: null, blocked: false, corroborated: false } + if (fullSync) return { reason, blocked: false, corroborated: false } + + const corroborated = Boolean( + previous?.trustworthy && classifySuspectListing(previous.listedCount, previous.ownedCount) + ) + return { reason, blocked: !corroborated, corroborated } +} + +/** + * Reconstructs the previous completed sync's listing from its log counters. + * + * No schema change is needed: every document the previous run listed landed in + * exactly one of added/updated/unchanged/failed, and `lastSyncDocCount` records + * how many documents the connector owned when that run finished. Documents the + * user excluded also land in `docsUnchanged`, which can only inflate the + * reconstructed listing — erring toward "the previous listing looked healthy", + * i.e. toward blocking deletions. + */ +async function loadPreviousListingObservation( + connectorId: string, + currentSyncLogId: string, + previousOwnedCount: number, + trustworthy: boolean +): Promise { + const rows = await db + .select({ + docsAdded: knowledgeConnectorSyncLog.docsAdded, + docsUpdated: knowledgeConnectorSyncLog.docsUpdated, + docsUnchanged: knowledgeConnectorSyncLog.docsUnchanged, + docsFailed: knowledgeConnectorSyncLog.docsFailed, + }) + .from(knowledgeConnectorSyncLog) + .where( + and( + eq(knowledgeConnectorSyncLog.connectorId, connectorId), + eq(knowledgeConnectorSyncLog.status, 'completed'), + ne(knowledgeConnectorSyncLog.id, currentSyncLogId) + ) + ) + .orderBy(desc(knowledgeConnectorSyncLog.startedAt)) + .limit(1) + + const previous = rows[0] + if (!previous) return null + + return { + listedCount: + previous.docsAdded + previous.docsUpdated + previous.docsUnchanged + previous.docsFailed, + ownedCount: previousOwnedCount, + trustworthy, + } +} + /** * Decides whether a sync should use the connector's incremental listing. * @@ -391,6 +533,9 @@ async function resolveAccessToken( ): Promise { if (connectorConfig.auth.mode === 'apiKey') { if (!connector.encryptedApiKey) { + if (connectorConfig.auth.optional) { + return '' + } throw new Error('API key connector is missing encrypted API key') } const { decrypted } = await decryptApiKey(connector.encryptedApiKey) @@ -1004,11 +1149,51 @@ export async function executeSync( options?.fullSync ) - const reconcileDeletionsAllowed = shouldReconcileDeletions( + let reconcileDeletionsAllowed = shouldReconcileDeletions( isIncremental, syncContext, options?.fullSync ) + + /** + * Backstop shared by every connector: a listing that reports (almost) + * nothing while this connector still owns a real corpus is treated as a + * fault, not as evidence of deletion, until a consecutive sync sees the + * same thing. Only evaluated when reconciliation would otherwise run, so + * healthy syncs pay nothing and no existing gate is loosened. + */ + const ownedDocCount = existingDocs.length + tombstonedDocs.length + if (reconcileDeletionsAllowed && classifySuspectListing(seenExternalIds.size, ownedDocCount)) { + const previousObservation = await loadPreviousListingObservation( + connectorId, + syncLogId, + connector.lastSyncDocCount ?? ownedDocCount, + !connectorConfig.supportsIncrementalSync || connector.syncMode === 'full' + ) + const listingSafety = evaluateListingSafety( + seenExternalIds.size, + ownedDocCount, + previousObservation, + options?.fullSync + ) + logger.warn('Suspect connector listing detected', { + connectorId, + connectorType: connector.connectorType, + reason: listingSafety.reason, + listedDocs: seenExternalIds.size, + ownedDocs: ownedDocCount, + liveDocs: existingDocs.length, + tombstonedDocs: tombstonedDocs.length, + previousListedDocs: previousObservation?.listedCount ?? null, + previousObservationTrusted: previousObservation?.trustworthy ?? false, + deletionReconciliation: listingSafety.blocked ? 'skipped' : 'proceeding', + syncRunId: syncContext.syncRunId, + }) + if (listingSafety.blocked) { + reconcileDeletionsAllowed = false + } + } + const gatedSoftDeleteIds = reconcileDeletionsAllowed ? softDeleteIds : [] const gatedHardDeleteIds = reconcileDeletionsAllowed ? hardDeleteIds : [] diff --git a/apps/sim/lib/knowledge/orchestration/connectors.ts b/apps/sim/lib/knowledge/orchestration/connectors.ts index 16ca5d23d8c..750c2a41bcc 100644 --- a/apps/sim/lib/knowledge/orchestration/connectors.ts +++ b/apps/sim/lib/knowledge/orchestration/connectors.ts @@ -155,10 +155,10 @@ export async function performCreateKnowledgeConnector( let accessToken: string if (connectorConfig.auth.mode === 'apiKey') { - if (!apiKey) { + if (!apiKey && !connectorConfig.auth.optional) { return fail('API key is required', 'validation') } - accessToken = apiKey + accessToken = apiKey ?? '' } else { if (!credentialId) { return fail('Credential is required', 'validation') diff --git a/apps/sim/lib/oauth/oauth.ts b/apps/sim/lib/oauth/oauth.ts index a98e9361e65..0ba237f5fa2 100644 --- a/apps/sim/lib/oauth/oauth.ts +++ b/apps/sim/lib/oauth/oauth.ts @@ -259,6 +259,11 @@ export const OAUTH_PROVIDERS: Record = { 'https://www.googleapis.com/auth/userinfo.email', 'https://www.googleapis.com/auth/userinfo.profile', 'https://www.googleapis.com/auth/ediscovery', + // Least-privilege scope for read-only consumers. The knowledge base + // connector only lists matters, holds, and saved queries, all of which + // accept ediscovery.readonly; the block's export tools still need the + // read-write scope above. + 'https://www.googleapis.com/auth/ediscovery.readonly', 'https://www.googleapis.com/auth/devstorage.read_only', ], serviceAccountProviderId: 'google-service-account', @@ -1184,6 +1189,13 @@ export const OAUTH_PROVIDERS: Record = { 'Desk.tickets.READ', 'Desk.tickets.UPDATE', 'Desk.contacts.READ', + // READ only: the knowledge base connector syncs Help Center articles + // via GET /articles and GET /articles/{id}; nothing authors one. + 'Desk.articles.READ', + // GET /organizations documents `Desk.organization.READ , Desk.basic.READ`. + // Sibling endpoints spell the same construction "requires X and Y" + // (dependencyMappings, roles), so the comma is AND, not OR. + 'Desk.organization.READ', // READ only: the agent picker for `assigneeId` lists agents, and no // tool creates, edits or deletes one. 'Desk.agents.READ', diff --git a/apps/sim/lib/oauth/utils.ts b/apps/sim/lib/oauth/utils.ts index eef2b12d759..ecd0846ccdb 100644 --- a/apps/sim/lib/oauth/utils.ts +++ b/apps/sim/lib/oauth/utils.ts @@ -36,6 +36,8 @@ export const SCOPE_DESCRIPTIONS: Record = { 'https://www.googleapis.com/auth/adwords': 'Manage Google Ads campaigns and reporting', 'https://www.googleapis.com/auth/bigquery': 'View and manage data in Google BigQuery', 'https://www.googleapis.com/auth/ediscovery': 'Access Google Vault for eDiscovery', + 'https://www.googleapis.com/auth/ediscovery.readonly': + 'View Google Vault matters, holds, and saved queries', 'https://www.googleapis.com/auth/devstorage.read_only': 'Read files from Google Cloud Storage', 'https://www.googleapis.com/auth/admin.directory.group': 'Manage Google Workspace groups', 'https://www.googleapis.com/auth/admin.directory.group.member': @@ -686,12 +688,27 @@ export function getMissingRequiredScopes( for (const s of requiredScopes) { if (IGNORED_SCOPES.has(s)) continue - if (!granted.has(s)) missing.push(s) + if (!granted.has(s) && !isScopeSatisfiedBy(s, granted)) missing.push(s) } return missing } +/** + * Whether a granted scope already covers `required` despite not matching it verbatim. + * + * A read-write scope subsumes its `.readonly` sibling — a credential holding + * `.../auth/ediscovery` is accepted by every method that documents + * `.../auth/ediscovery.readonly`. Without this, narrowing a consumer to the + * least-privileged scope would report every already-connected credential as + * missing it and prompt a re-consent that grants nothing new. + */ +function isScopeSatisfiedBy(required: string, granted: ReadonlySet): boolean { + const readonlySuffix = '.readonly' + if (!required.endsWith(readonlySuffix)) return false + return granted.has(required.slice(0, -readonlySuffix.length)) +} + /** * Build a mapping of providerId -> { baseProvider, serviceKey } from OAUTH_PROVIDERS * This is computed once at module load time diff --git a/apps/sim/tools/microsoft_dataverse/utils.ts b/apps/sim/tools/microsoft_dataverse/utils.ts index 9e6e7862dca..5ff377bcee3 100644 --- a/apps/sim/tools/microsoft_dataverse/utils.ts +++ b/apps/sim/tools/microsoft_dataverse/utils.ts @@ -1,8 +1,55 @@ +/** + * Registrable domains Microsoft serves Dataverse environments from — commercial and + * regional clouds (`*.crm[N].dynamics.com`), China (21Vianet), US Government and DoD, + * and the legacy German cloud. Matching on the registrable domain rather than each + * regional `crmN` prefix keeps new Microsoft regions working without a code change. + */ +const DATAVERSE_HOST_SUFFIXES = [ + '.dynamics.com', + '.dynamics.cn', + '.dynamics.de', + '.microsoftdynamics.us', + '.appsplatform.us', +] as const + /** * Normalizes a Dataverse environment URL into a base URL suitable for building Web API request * paths: trims incidental whitespace (common when pasted from a browser address bar) and strips * a trailing slash so callers can safely append `/api/data/v9.2/...`. + * + * The value is user-supplied and every request built from it carries the caller's OAuth bearer + * token, so the host is pinned to Microsoft's Dataverse domains — an arbitrary origin here would + * otherwise receive that token. + * + * @throws {Error} when the value is empty, not a parseable absolute URL, not HTTPS, carries + * credentials, or is not a Dataverse host. */ export function getDataverseBaseUrl(environmentUrl: string): string { - return environmentUrl.trim().replace(/\/$/, '') + const value = typeof environmentUrl === 'string' ? environmentUrl.trim() : '' + if (!value) { + throw new Error('Environment URL is required') + } + + let parsed: URL + try { + parsed = new URL(value) + } catch { + throw new Error('Environment URL must be an absolute URL, e.g. https://myorg.crm.dynamics.com') + } + + if (parsed.protocol !== 'https:') { + throw new Error('Environment URL must use https') + } + if (parsed.username || parsed.password) { + throw new Error('Environment URL must not contain credentials') + } + + const hostname = parsed.hostname.toLowerCase() + if (!DATAVERSE_HOST_SUFFIXES.some((suffix) => hostname.endsWith(suffix))) { + throw new Error( + 'Environment URL must be a Dataverse environment, e.g. https://myorg.crm.dynamics.com' + ) + } + + return parsed.origin } From 903c94e913e9e3cff7d956f673eb2bfd1feeb5a9 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Fri, 14 Aug 2026 12:59:11 -0700 Subject: [PATCH 2/5] fix(dataverse): strip the bearer token when a request redirects MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The host allowlist added alongside the connector work only constrains the initial destination. `secureFetchWithPinnedIP` follows redirects and keeps the `Authorization` header unless a tool opts out, so a redirect away from an allowed Dataverse origin would forward the caller's OAuth token to whatever host answers. Dataverse redirects in normal operation — file downloads hand back a signed storage URL, and environment hosts move between regional origins — so this is reachable without a compromised environment URL. Sets `stripAuthOnRedirect` on all 18 Dataverse tools, matching the existing GitHub job-logs and Windchill precedent. --- apps/sim/tools/microsoft_dataverse/associate.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/create_multiple.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/create_record.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/delete_record.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/disassociate.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/download_file.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/execute_action.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/execute_function.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/fetchxml_query.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/get_entity_metadata.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/get_record.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/list_records.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/search.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/update_multiple.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/update_record.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/upload_file.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/upsert_record.ts | 6 ++++++ apps/sim/tools/microsoft_dataverse/whoami.ts | 6 ++++++ 18 files changed, 108 insertions(+) diff --git a/apps/sim/tools/microsoft_dataverse/associate.ts b/apps/sim/tools/microsoft_dataverse/associate.ts index 32978ef631b..e901f4061ac 100644 --- a/apps/sim/tools/microsoft_dataverse/associate.ts +++ b/apps/sim/tools/microsoft_dataverse/associate.ts @@ -80,6 +80,12 @@ export const dataverseAssociateTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}(${params.recordId.trim()})/${params.navigationProperty.trim()}/$ref` }, method: (params) => (params.navigationType === 'single' ? 'PUT' : 'POST'), + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/create_multiple.ts b/apps/sim/tools/microsoft_dataverse/create_multiple.ts index 12db7cdc4d6..3fe1101a373 100644 --- a/apps/sim/tools/microsoft_dataverse/create_multiple.ts +++ b/apps/sim/tools/microsoft_dataverse/create_multiple.ts @@ -62,6 +62,12 @@ export const dataverseCreateMultipleTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}/Microsoft.Dynamics.CRM.CreateMultiple` }, method: 'POST', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/create_record.ts b/apps/sim/tools/microsoft_dataverse/create_record.ts index 35240b41f28..33ed6d0280a 100644 --- a/apps/sim/tools/microsoft_dataverse/create_record.ts +++ b/apps/sim/tools/microsoft_dataverse/create_record.ts @@ -55,6 +55,12 @@ export const dataverseCreateRecordTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}` }, method: 'POST', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/delete_record.ts b/apps/sim/tools/microsoft_dataverse/delete_record.ts index f9bac250fe4..c0fe5aafaab 100644 --- a/apps/sim/tools/microsoft_dataverse/delete_record.ts +++ b/apps/sim/tools/microsoft_dataverse/delete_record.ts @@ -53,6 +53,12 @@ export const dataverseDeleteRecordTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}(${params.recordId.trim()})` }, method: 'DELETE', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/disassociate.ts b/apps/sim/tools/microsoft_dataverse/disassociate.ts index d74f5a349c3..82e284a245c 100644 --- a/apps/sim/tools/microsoft_dataverse/disassociate.ts +++ b/apps/sim/tools/microsoft_dataverse/disassociate.ts @@ -74,6 +74,12 @@ export const dataverseDisassociateTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${entitySetName}(${recordId})/${navigationProperty}/$ref` }, method: 'DELETE', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/download_file.ts b/apps/sim/tools/microsoft_dataverse/download_file.ts index 155374fef27..d6f322bf723 100644 --- a/apps/sim/tools/microsoft_dataverse/download_file.ts +++ b/apps/sim/tools/microsoft_dataverse/download_file.ts @@ -60,6 +60,12 @@ export const dataverseDownloadFileTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}(${params.recordId.trim()})/${params.fileColumn.trim()}/$value` }, method: 'GET', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/execute_action.ts b/apps/sim/tools/microsoft_dataverse/execute_action.ts index 1a427bb151f..4cad4931cc3 100644 --- a/apps/sim/tools/microsoft_dataverse/execute_action.ts +++ b/apps/sim/tools/microsoft_dataverse/execute_action.ts @@ -78,6 +78,12 @@ export const dataverseExecuteActionTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${actionName}` }, method: 'POST', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/execute_function.ts b/apps/sim/tools/microsoft_dataverse/execute_function.ts index 4457af4113b..40a1392ce71 100644 --- a/apps/sim/tools/microsoft_dataverse/execute_function.ts +++ b/apps/sim/tools/microsoft_dataverse/execute_function.ts @@ -83,6 +83,12 @@ export const dataverseExecuteFunctionTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${functionName}${paramStr}${querySuffix}` }, method: 'GET', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/fetchxml_query.ts b/apps/sim/tools/microsoft_dataverse/fetchxml_query.ts index 0e6fd37fed4..fdad61edb5f 100644 --- a/apps/sim/tools/microsoft_dataverse/fetchxml_query.ts +++ b/apps/sim/tools/microsoft_dataverse/fetchxml_query.ts @@ -57,6 +57,12 @@ export const dataverseFetchXmlQueryTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}?fetchXml=${encodedFetchXml}` }, method: 'GET', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/get_entity_metadata.ts b/apps/sim/tools/microsoft_dataverse/get_entity_metadata.ts index 2dcb4d87eba..82ec14e0f3f 100644 --- a/apps/sim/tools/microsoft_dataverse/get_entity_metadata.ts +++ b/apps/sim/tools/microsoft_dataverse/get_entity_metadata.ts @@ -76,6 +76,12 @@ export const dataverseGetEntityMetadataTool: ToolConfig< return `${baseUrl}/api/data/v9.2/EntityDefinitions(LogicalName='${entityLogicalName}')${query}` }, method: 'GET', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/get_record.ts b/apps/sim/tools/microsoft_dataverse/get_record.ts index 344405ba28f..039179d77d1 100644 --- a/apps/sim/tools/microsoft_dataverse/get_record.ts +++ b/apps/sim/tools/microsoft_dataverse/get_record.ts @@ -71,6 +71,12 @@ export const dataverseGetRecordTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}(${params.recordId.trim()})${query}` }, method: 'GET', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', diff --git a/apps/sim/tools/microsoft_dataverse/list_records.ts b/apps/sim/tools/microsoft_dataverse/list_records.ts index b79bea05084..fc5ff1f6f40 100644 --- a/apps/sim/tools/microsoft_dataverse/list_records.ts +++ b/apps/sim/tools/microsoft_dataverse/list_records.ts @@ -93,6 +93,12 @@ export const dataverseListRecordsTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}${query}` }, method: 'GET', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => { // Dataverse ignores $top entirely when Prefer: odata.maxpagesize is also sent, so the // page-size preference is only applied when the caller hasn't requested an explicit $top. diff --git a/apps/sim/tools/microsoft_dataverse/search.ts b/apps/sim/tools/microsoft_dataverse/search.ts index d313074925f..e2d0b28c3d5 100644 --- a/apps/sim/tools/microsoft_dataverse/search.ts +++ b/apps/sim/tools/microsoft_dataverse/search.ts @@ -98,6 +98,12 @@ export const dataverseSearchTool: ToolConfig ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/update_multiple.ts b/apps/sim/tools/microsoft_dataverse/update_multiple.ts index d0702c1b1c9..84f411d80b5 100644 --- a/apps/sim/tools/microsoft_dataverse/update_multiple.ts +++ b/apps/sim/tools/microsoft_dataverse/update_multiple.ts @@ -62,6 +62,12 @@ export const dataverseUpdateMultipleTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}/Microsoft.Dynamics.CRM.UpdateMultiple` }, method: 'POST', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/update_record.ts b/apps/sim/tools/microsoft_dataverse/update_record.ts index 8bf3b1ff793..872f5e9461e 100644 --- a/apps/sim/tools/microsoft_dataverse/update_record.ts +++ b/apps/sim/tools/microsoft_dataverse/update_record.ts @@ -60,6 +60,12 @@ export const dataverseUpdateRecordTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}(${params.recordId.trim()})` }, method: 'PATCH', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/upload_file.ts b/apps/sim/tools/microsoft_dataverse/upload_file.ts index a8025fe0916..3bc72fbe887 100644 --- a/apps/sim/tools/microsoft_dataverse/upload_file.ts +++ b/apps/sim/tools/microsoft_dataverse/upload_file.ts @@ -70,6 +70,12 @@ export const dataverseUploadFileTool: ToolConfig< request: { url: '/api/tools/microsoft-dataverse/upload-file', method: 'POST', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: () => ({ 'Content-Type': 'application/json', }), diff --git a/apps/sim/tools/microsoft_dataverse/upsert_record.ts b/apps/sim/tools/microsoft_dataverse/upsert_record.ts index 04f82356f24..9642cecc72f 100644 --- a/apps/sim/tools/microsoft_dataverse/upsert_record.ts +++ b/apps/sim/tools/microsoft_dataverse/upsert_record.ts @@ -61,6 +61,12 @@ export const dataverseUpsertRecordTool: ToolConfig< return `${baseUrl}/api/data/v9.2/${params.entitySetName.trim()}(${params.recordId.trim()})` }, method: 'PATCH', + /** + * Dataverse endpoints redirect (file downloads issue a signed storage URL, + * and environment hosts redirect between regional origins), so drop the + * bearer token rather than forward it to whatever origin answers. + */ + stripAuthOnRedirect: true, headers: (params) => ({ Authorization: `Bearer ${params.accessToken}`, 'Content-Type': 'application/json', diff --git a/apps/sim/tools/microsoft_dataverse/whoami.ts b/apps/sim/tools/microsoft_dataverse/whoami.ts index d78174ecd42..95e2969b89e 100644 --- a/apps/sim/tools/microsoft_dataverse/whoami.ts +++ b/apps/sim/tools/microsoft_dataverse/whoami.ts @@ -39,6 +39,12 @@ export const dataverseWhoAmITool: ToolConfig ({ Authorization: `Bearer ${params.accessToken}`, 'OData-MaxVersion': '4.0', From 7426d3dc199eeac4872b0d94a455090de0e704b2 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Fri, 14 Aug 2026 13:02:02 -0700 Subject: [PATCH 3/5] fix(connectors): address review findings on listing and hashing - microsoft-excel: `fetchWorksheets` read only the first Graph page and never followed `@odata.nextLink`. A workbook with more sheets than fit in one page dropped the remainder from the listing without setting `listingCapped`, so the sync engine reconciled those documents away as deleted. The walk now pages, bounded by MAX_WORKSHEETS, and only follows a nextLink that stays on the Graph origin, since the link is server-supplied and carries the token. - google-slides: the listing `contentHash` covered only the file id and modified time, so toggling the speaker-notes option left every stored hash matching and no presentation was ever re-hydrated with the new scope. The setting is now part of the hash, in the single shared stub builder so the list and hydrate paths stay identical. - mintlify: `pathPrefix` filtered with a bare `startsWith`, so a prefix of `/guides` also matched a sibling like `/guides-archive`. It now shares the `/`-boundary rule `withinBasePath` already used, extracted as `isUnderPath`. --- .../connectors/google-slides/google-slides.ts | 28 ++++++++----- .../microsoft-excel/microsoft-excel.ts | 39 +++++++++++++++---- apps/sim/connectors/mintlify/mintlify.ts | 19 ++++++--- 3 files changed, 63 insertions(+), 23 deletions(-) diff --git a/apps/sim/connectors/google-slides/google-slides.ts b/apps/sim/connectors/google-slides/google-slides.ts index fad9dc5cbae..3f776360cd5 100644 --- a/apps/sim/connectors/google-slides/google-slides.ts +++ b/apps/sim/connectors/google-slides/google-slides.ts @@ -218,7 +218,7 @@ function shouldIncludeSpeakerNotes(sourceConfig: Record): boole * Creates a lightweight stub from a Drive file entry. Content is deferred * and only fetched via getDocument for new or changed documents. */ -function fileToStub(file: DriveFile): ExternalDocument { +function fileToStub(file: DriveFile, includeSpeakerNotes: boolean): ExternalDocument { return { externalId: file.id, title: file.name || 'Untitled', @@ -226,7 +226,12 @@ function fileToStub(file: DriveFile): ExternalDocument { contentDeferred: true, mimeType: 'text/plain', sourceUrl: file.webViewLink || `https://docs.google.com/presentation/d/${file.id}/edit`, - contentHash: `gslides:${file.id}:${file.modifiedTime ?? ''}`, + /** + * The speaker-notes setting selects what the rendered content contains, so it + * belongs in the hash. Without it, toggling the option leaves every stored + * hash matching and no presentation is ever re-hydrated with the new scope. + */ + contentHash: `gslides:${file.id}:${file.modifiedTime ?? ''}:${includeSpeakerNotes ? 'n1' : 'n0'}`, metadata: { modifiedTime: file.modifiedTime, createdTime: file.createdTime, @@ -314,7 +319,8 @@ export const googleSlidesConnector: ConnectorConfig = { const maxDocs = sourceConfig.maxDocs ? Number(sourceConfig.maxDocs) : 0 const previouslyFetched = (syncContext?.totalDocsFetched as number) ?? 0 - let documents = files.map(fileToStub) + const includeSpeakerNotes = shouldIncludeSpeakerNotes(sourceConfig) + let documents = files.map((file) => fileToStub(file, includeSpeakerNotes)) let slicedSome = false if (maxDocs > 0) { const remaining = maxDocs - previouslyFetched @@ -376,11 +382,8 @@ export const googleSlidesConnector: ConnectorConfig = { if (file.trashed) return null if (file.mimeType !== PRESENTATION_MIME_TYPE) return null - const content = await fetchPresentationContent( - accessToken, - file.id, - shouldIncludeSpeakerNotes(sourceConfig) - ) + const includeSpeakerNotes = shouldIncludeSpeakerNotes(sourceConfig) + const content = await fetchPresentationContent(accessToken, file.id, includeSpeakerNotes) /** * An image-only deck carries no extractable text. Surfacing it as a skipped @@ -389,10 +392,15 @@ export const googleSlidesConnector: ConnectorConfig = { * presentation would simply be missing and re-fetched on every sync. */ if (!content.trim()) { - return { ...fileToStub(file), content: '', contentDeferred: false, skippedReason: NO_TEXT } + return { + ...fileToStub(file, includeSpeakerNotes), + content: '', + contentDeferred: false, + skippedReason: NO_TEXT, + } } - return { ...fileToStub(file), content, contentDeferred: false } + return { ...fileToStub(file, includeSpeakerNotes), content, contentDeferred: false } }, validateConfig: async ( diff --git a/apps/sim/connectors/microsoft-excel/microsoft-excel.ts b/apps/sim/connectors/microsoft-excel/microsoft-excel.ts index 2eb4761aa0a..fc93d0537f4 100644 --- a/apps/sim/connectors/microsoft-excel/microsoft-excel.ts +++ b/apps/sim/connectors/microsoft-excel/microsoft-excel.ts @@ -39,6 +39,13 @@ const CONTENT_FORMAT_VERSION = 'v2' */ const MAX_WORKSHEETS = 500 +/** + * Origin every Graph response must stay on. `@odata.nextLink` is server-supplied + * and carries the bearer token when followed, so a link pointing anywhere else is + * dropped rather than requested. + */ +const GRAPH_API_BASE = 'https://graph.microsoft.com/' + /** Maximum rows read from a single worksheet's used range. */ const MAX_ROWS = 5000 @@ -82,6 +89,7 @@ interface Worksheet { interface WorksheetListResponse { value?: Worksheet[] + '@odata.nextLink'?: string } interface WorkbookItem { @@ -243,18 +251,33 @@ async function fetchWorkbookItem( /** Lists the workbook's worksheets in tab order. */ async function fetchWorksheets(accessToken: string, basePath: string): Promise { - const response = await fetchWithRetry( - `${basePath}/workbook/worksheets?$select=id,name,position,visibility&$orderby=position`, - { + const worksheets: Worksheet[] = [] + let url: string | undefined = + `${basePath}/workbook/worksheets?$select=id,name,position,visibility&$orderby=position` + + /** + * Graph paginates collection responses, so a workbook with more sheets than fit + * in one page must follow `@odata.nextLink`. Reading only the first page would + * drop the remainder from the listing without setting `listingCapped`, and the + * sync engine would then reconcile those documents away as deleted. The walk is + * bounded by `MAX_WORKSHEETS`, whose truncation the caller does flag. + */ + while (url && worksheets.length <= MAX_WORKSHEETS) { + const response = await fetchWithRetry(url, { method: 'GET', headers: { Authorization: `Bearer ${accessToken}`, Accept: 'application/json' }, - } - ) + }) + + if (!response.ok) await graphError(response, 'Failed to list worksheets') - if (!response.ok) await graphError(response, 'Failed to list worksheets') + const data = (await response.json()) as WorksheetListResponse + worksheets.push(...(data.value ?? [])) + + const next = data['@odata.nextLink'] + url = next && next.startsWith(GRAPH_API_BASE) ? next : undefined + } - const data = (await response.json()) as WorksheetListResponse - return data.value ?? [] + return worksheets } /** diff --git a/apps/sim/connectors/mintlify/mintlify.ts b/apps/sim/connectors/mintlify/mintlify.ts index 1bb49e4ed07..4cc7e02fd1c 100644 --- a/apps/sim/connectors/mintlify/mintlify.ts +++ b/apps/sim/connectors/mintlify/mintlify.ts @@ -323,9 +323,18 @@ async function discoverFromSitemap( */ function withinBasePath(pages: MintlifyPageLink[], site: MintlifySite): MintlifyPageLink[] { if (!site.basePath) return pages - return pages.filter( - (page) => page.path === site.basePath || page.path.startsWith(`${site.basePath}/`) - ) + return pages.filter((page) => isUnderPath(page.path, site.basePath)) +} + +/** + * Whether `path` is `prefix` itself or sits beneath it. + * + * A bare `startsWith` would also match a sibling whose name merely begins with + * the prefix — `/guides` would capture `/guides-archive` — so the boundary `/` + * is required. + */ +function isUnderPath(path: string, prefix: string): boolean { + return path === prefix || path.startsWith(`${prefix}/`) } /** @@ -469,7 +478,7 @@ export const mintlifyConnector: ConnectorConfig = { } const filtered = pathPrefix - ? discovered.filter((page) => page.path.startsWith(pathPrefix)) + ? discovered.filter((page) => isUnderPath(page.path, pathPrefix)) : discovered if (filtered.length > maxPages && syncContext) { @@ -591,7 +600,7 @@ export const mintlifyConnector: ConnectorConfig = { } const pathPrefix = resolvePathPrefix(sourceConfig.pathPrefix) - if (pathPrefix && !pages.some((page) => page.path.startsWith(pathPrefix))) { + if (pathPrefix && !pages.some((page) => isUnderPath(page.path, pathPrefix))) { return { valid: false, error: `No pages match the path prefix "${pathPrefix}"` } } From 74cda711f4c8c0af5ed528bd19db1383f8b764f4 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Fri, 14 Aug 2026 13:29:48 -0700 Subject: [PATCH 4/5] fix(connectors): list newest first in zoho desk, accept a trailing slash prefix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - zoho-desk: `sortBy: 'createdTime'` is ascending — Zoho denotes descending with a `-` prefix — so the default 500-record caps kept the oldest tickets and articles and recent ones were never listed. Because the cap sets listingCapped, that stale tail could not reconcile away either. Now sorts `-createdTime`. Still ordering on createdTime rather than modifiedTime, so rows do not reshuffle mid-walk. - mintlify: `resolvePathPrefix` kept a trailing slash while `isUnderPath` accepts an exact match or `prefix + '/'`, so `/guides/` matched neither `/guides` nor `/guides/intro` and the source synced nothing. A regression from the previous round, which replaced a bare `startsWith`. The prefix is now normalized before comparison. --- apps/sim/connectors/mintlify/mintlify.ts | 12 ++++++++++-- apps/sim/connectors/zoho-desk/zoho-desk.ts | 22 ++++++++++++++++++---- 2 files changed, 28 insertions(+), 6 deletions(-) diff --git a/apps/sim/connectors/mintlify/mintlify.ts b/apps/sim/connectors/mintlify/mintlify.ts index 4cc7e02fd1c..f4b9592abf9 100644 --- a/apps/sim/connectors/mintlify/mintlify.ts +++ b/apps/sim/connectors/mintlify/mintlify.ts @@ -88,9 +88,17 @@ function resolveMaxPages(value: unknown): number { return Math.min(Math.floor(parsed), MAX_PAGES_LIMIT) } -/** Normalizes an optional path prefix filter to a leading-slash form, or `''`. */ +/** + * Normalizes an optional path prefix filter to a leading-slash, no-trailing-slash + * form, or `''`. + * + * The trailing slash has to go: {@link isUnderPath} accepts an exact match or + * `prefix + '/'`, so a raw `/guides/` would match neither `/guides` nor + * `/guides/intro` and the source would silently sync nothing. + */ function resolvePathPrefix(value: unknown): string { - const prefix = typeof value === 'string' ? value.trim() : '' + const trimmed = typeof value === 'string' ? value.trim() : '' + const prefix = trimmed.replace(/\/+$/, '') if (!prefix) return '' return prefix.startsWith('/') ? prefix : `/${prefix}` } diff --git a/apps/sim/connectors/zoho-desk/zoho-desk.ts b/apps/sim/connectors/zoho-desk/zoho-desk.ts index dbb7fc4891c..badbd6a2c39 100644 --- a/apps/sim/connectors/zoho-desk/zoho-desk.ts +++ b/apps/sim/connectors/zoho-desk/zoho-desk.ts @@ -274,9 +274,15 @@ async function fetchArticles( const query = new URLSearchParams({ from: String(from), limit: String(limit), - // Stable ordering matters across pages: sorting on modifiedTime would - // reshuffle rows mid-walk as the portal is edited, silently skipping some. - sortBy: 'createdTime', + /** + * Newest first. Zoho treats a bare field as ascending and a `-` prefix as + * descending, so `createdTime` would fill the cap with the oldest records + * and leave recent tickets and articles permanently unreachable — the cap + * sets `listingCapped`, which stops that stale tail from ever reconciling + * away. Sorting on createdTime rather than modifiedTime keeps the order + * stable across pages; edits would otherwise reshuffle rows mid-walk. + */ + sortBy: '-createdTime', }) if (options.status && options.status !== 'all') query.set('status', options.status) if (options.categoryId) query.set('categoryId', options.categoryId) @@ -300,7 +306,15 @@ async function fetchTickets( const query = new URLSearchParams({ from: String(from), limit: String(limit), - sortBy: 'createdTime', + /** + * Newest first. Zoho treats a bare field as ascending and a `-` prefix as + * descending, so `createdTime` would fill the cap with the oldest records + * and leave recent tickets and articles permanently unreachable — the cap + * sets `listingCapped`, which stops that stale tail from ever reconciling + * away. Sorting on createdTime rather than modifiedTime keeps the order + * stable across pages; edits would otherwise reshuffle rows mid-walk. + */ + sortBy: '-createdTime', }) if (options.status) query.set('status', options.status) if (options.departmentIds.length > 0) { From eb1e55654da78fa0cd773ed787b8811fa0fbc987 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Fri, 14 Aug 2026 13:36:28 -0700 Subject: [PATCH 5/5] fix(dataverse): strip the bearer token on the upload route's own redirect MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `upload_file` posts to an internal route rather than calling Dataverse directly, so the tool-level `stripAuthOnRedirect` added in 903c94e9 only covers the same-origin hop into that route. The route's own outbound PATCH carries the caller's OAuth token and left redirect stripping at its default, so a redirect to a signed storage host — which is exactly how Dataverse serves file operations — would have handed that host a reusable credential. The other 17 tools build the Dataverse URL directly, so the tool-level flag already covers them. --- .../app/api/tools/microsoft-dataverse/upload-file/route.ts | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/apps/sim/app/api/tools/microsoft-dataverse/upload-file/route.ts b/apps/sim/app/api/tools/microsoft-dataverse/upload-file/route.ts index 56438618aa6..eb78c96325f 100644 --- a/apps/sim/app/api/tools/microsoft-dataverse/upload-file/route.ts +++ b/apps/sim/app/api/tools/microsoft-dataverse/upload-file/route.ts @@ -124,6 +124,13 @@ export const POST = withRouteHandler(async (request: NextRequest) => { 'x-ms-file-name': validatedData.fileName, }, body: fileBuffer, + /** + * The tool's own `stripAuthOnRedirect` only covers the hop to this + * route. Dataverse redirects file operations to signed storage hosts, + * so this outbound call has to drop the bearer token itself or the + * redirect target receives a reusable OAuth credential. + */ + stripAuthOnRedirect: true, }, 'environmentUrl' )