mirror of
https://github.com/simstudioai/sim.git
synced 2026-09-24 15:45:35 +08:00
feat(knowledge): include sourceUrl in KB search results (#4533)
* feat(knowledge): include sourceUrl in KB search results * improvement(kb-search): mark sourceUrl nullable and cover non-null happy path in tests
This commit is contained in:
@@ -60,6 +60,7 @@ Search for similar content in a knowledge base using vector similarity
|
||||
| `results` | array | Array of search results from the knowledge base |
|
||||
| ↳ `documentId` | string | Document ID |
|
||||
| ↳ `documentName` | string | Document name |
|
||||
| ↳ `sourceUrl` | string | URL to the original source document \(e.g., Confluence page, Google Doc, Notion page\). Null for documents without an external source. |
|
||||
| ↳ `content` | string | Content of the result |
|
||||
| ↳ `chunkIndex` | number | Index of the chunk within the document |
|
||||
| ↳ `similarity` | number | Similarity score of the result |
|
||||
|
||||
@@ -5030,6 +5030,7 @@
|
||||
{
|
||||
"documentId": "doc_abc123",
|
||||
"documentName": "Getting Started.pdf",
|
||||
"sourceUrl": "https://example.atlassian.net/wiki/spaces/DOCS/pages/12345",
|
||||
"content": "To reset your password, go to Settings > Security.",
|
||||
"chunkIndex": 3,
|
||||
"similarity": 0.95,
|
||||
@@ -6264,6 +6265,11 @@
|
||||
"type": "string",
|
||||
"description": "Filename of the source document."
|
||||
},
|
||||
"sourceUrl": {
|
||||
"type": "string",
|
||||
"nullable": true,
|
||||
"description": "URL to the original source document for connector-synced documents (e.g., a Confluence page, Google Doc, or Notion page). Null for documents without an external source."
|
||||
},
|
||||
"content": {
|
||||
"type": "string",
|
||||
"description": "The matched chunk content."
|
||||
|
||||
@@ -24,7 +24,7 @@ const {
|
||||
mockHandleTagAndVectorSearch,
|
||||
mockGetQueryStrategy,
|
||||
mockGenerateSearchEmbedding,
|
||||
mockGetDocumentNamesByIds,
|
||||
mockGetDocumentMetadataByIds,
|
||||
} = vi.hoisted(() => ({
|
||||
mockDbChain: {
|
||||
select: vi.fn().mockReturnThis(),
|
||||
@@ -43,7 +43,7 @@ const {
|
||||
mockHandleTagAndVectorSearch: vi.fn(),
|
||||
mockGetQueryStrategy: vi.fn(),
|
||||
mockGenerateSearchEmbedding: vi.fn(),
|
||||
mockGetDocumentNamesByIds: vi.fn(),
|
||||
mockGetDocumentMetadataByIds: vi.fn(),
|
||||
}))
|
||||
|
||||
const mockCheckKnowledgeBaseAccess = knowledgeApiUtilsMockFns.mockCheckKnowledgeBaseAccess
|
||||
@@ -101,7 +101,7 @@ vi.mock('./utils', () => ({
|
||||
handleTagAndVectorSearch: mockHandleTagAndVectorSearch,
|
||||
getQueryStrategy: mockGetQueryStrategy,
|
||||
generateSearchEmbedding: mockGenerateSearchEmbedding,
|
||||
getDocumentNamesByIds: mockGetDocumentNamesByIds,
|
||||
getDocumentMetadataByIds: mockGetDocumentMetadataByIds,
|
||||
APIError: class APIError extends Error {
|
||||
public status: number
|
||||
constructor(message: string, status: number) {
|
||||
@@ -159,9 +159,9 @@ describe('Knowledge Search API Route', () => {
|
||||
singleQueryOptimized: true,
|
||||
})
|
||||
mockGenerateSearchEmbedding.mockClear().mockResolvedValue([0.1, 0.2, 0.3, 0.4, 0.5])
|
||||
mockGetDocumentNamesByIds.mockClear().mockResolvedValue({
|
||||
doc1: 'Document 1',
|
||||
doc2: 'Document 2',
|
||||
mockGetDocumentMetadataByIds.mockClear().mockResolvedValue({
|
||||
doc1: { filename: 'Document 1', sourceUrl: null },
|
||||
doc2: { filename: 'Document 2', sourceUrl: null },
|
||||
})
|
||||
mockGetDocumentTagDefinitions.mockClear()
|
||||
hybridAuthMockFns.mockCheckSessionOrInternalAuth.mockClear().mockResolvedValue({
|
||||
@@ -998,8 +998,11 @@ describe('Knowledge Search API Route', () => {
|
||||
})
|
||||
|
||||
mockGenerateSearchEmbedding.mockResolvedValue([0.1, 0.2, 0.3])
|
||||
mockGetDocumentNamesByIds.mockResolvedValue({
|
||||
'doc-active': 'Active Document.pdf',
|
||||
mockGetDocumentMetadataByIds.mockResolvedValue({
|
||||
'doc-active': {
|
||||
filename: 'Active Document.pdf',
|
||||
sourceUrl: 'https://example.atlassian.net/wiki/spaces/DOCS/pages/12345',
|
||||
},
|
||||
})
|
||||
|
||||
const mockTagDefs = {
|
||||
@@ -1023,6 +1026,9 @@ describe('Knowledge Search API Route', () => {
|
||||
expect(data.data.results).toHaveLength(1)
|
||||
expect(data.data.results[0].documentId).toBe('doc-active')
|
||||
expect(data.data.results[0].documentName).toBe('Active Document.pdf')
|
||||
expect(data.data.results[0].sourceUrl).toBe(
|
||||
'https://example.atlassian.net/wiki/spaces/DOCS/pages/12345'
|
||||
)
|
||||
})
|
||||
|
||||
it('should exclude results from deleted documents in tag search', async () => {
|
||||
@@ -1067,8 +1073,8 @@ describe('Knowledge Search API Route', () => {
|
||||
singleQueryOptimized: true,
|
||||
})
|
||||
|
||||
mockGetDocumentNamesByIds.mockResolvedValue({
|
||||
'doc-active-tagged': 'Active Tagged Document.pdf',
|
||||
mockGetDocumentMetadataByIds.mockResolvedValue({
|
||||
'doc-active-tagged': { filename: 'Active Tagged Document.pdf', sourceUrl: null },
|
||||
})
|
||||
|
||||
const mockTagDefs = {
|
||||
@@ -1140,8 +1146,8 @@ describe('Knowledge Search API Route', () => {
|
||||
})
|
||||
|
||||
mockGenerateSearchEmbedding.mockResolvedValue([0.1, 0.2, 0.3])
|
||||
mockGetDocumentNamesByIds.mockResolvedValue({
|
||||
'doc-active-combined': 'Active Combined Search.pdf',
|
||||
mockGetDocumentMetadataByIds.mockResolvedValue({
|
||||
'doc-active-combined': { filename: 'Active Combined Search.pdf', sourceUrl: null },
|
||||
})
|
||||
|
||||
const mockTagDefs = {
|
||||
|
||||
@@ -16,7 +16,7 @@ import type { StructuredFilter } from '@/lib/knowledge/types'
|
||||
import { estimateTokenCount } from '@/lib/tokenization/estimators'
|
||||
import {
|
||||
generateSearchEmbedding,
|
||||
getDocumentNamesByIds,
|
||||
getDocumentMetadataByIds,
|
||||
getQueryStrategy,
|
||||
handleTagAndVectorSearch,
|
||||
handleTagOnlySearch,
|
||||
@@ -413,7 +413,7 @@ export const POST = withRouteHandler(async (request: NextRequest) => {
|
||||
})
|
||||
|
||||
const documentIds = results.map((result) => result.documentId)
|
||||
const documentNameMap = await getDocumentNamesByIds(documentIds)
|
||||
const documentMetadataMap = await getDocumentMetadataByIds(documentIds)
|
||||
|
||||
try {
|
||||
PlatformEvents.knowledgeBaseSearched({
|
||||
@@ -449,9 +449,11 @@ export const POST = withRouteHandler(async (request: NextRequest) => {
|
||||
})
|
||||
|
||||
const rerankerScore = rerankedScores.get(result.id)
|
||||
const docMeta = documentMetadataMap[result.documentId]
|
||||
return {
|
||||
documentId: result.documentId,
|
||||
documentName: documentNameMap[result.documentId] || undefined,
|
||||
documentName: docMeta?.filename || undefined,
|
||||
sourceUrl: docMeta?.sourceUrl ?? null,
|
||||
content: result.content,
|
||||
chunkIndex: result.chunkIndex,
|
||||
metadata: tags,
|
||||
|
||||
@@ -396,11 +396,11 @@ describe('Knowledge Search Utils', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('getDocumentNamesByIds', () => {
|
||||
describe('getDocumentMetadataByIds', () => {
|
||||
it('should handle empty input gracefully', async () => {
|
||||
const { getDocumentNamesByIds } = await import('./utils')
|
||||
const { getDocumentMetadataByIds } = await import('./utils')
|
||||
|
||||
const result = await getDocumentNamesByIds([])
|
||||
const result = await getDocumentMetadataByIds([])
|
||||
|
||||
expect(result).toEqual({})
|
||||
})
|
||||
|
||||
@@ -3,9 +3,22 @@ import { document, embedding } from '@sim/db/schema'
|
||||
import { and, eq, inArray, isNull, sql } from 'drizzle-orm'
|
||||
import type { StructuredFilter } from '@/lib/knowledge/types'
|
||||
|
||||
export async function getDocumentNamesByIds(
|
||||
export interface DocumentMetadata {
|
||||
filename: string
|
||||
sourceUrl: string | null
|
||||
}
|
||||
|
||||
/**
|
||||
* Batch-fetch display metadata for documents referenced by search results.
|
||||
* Excludes documents that are user-excluded, archived, or soft-deleted —
|
||||
* mirrors the visibility filters applied inside the search SQL itself, so
|
||||
* the lookup will never surface metadata for a row a caller could not have
|
||||
* legitimately matched. Returns a map keyed by document id; missing ids
|
||||
* indicate the document is no longer visible and should be skipped.
|
||||
*/
|
||||
export async function getDocumentMetadataByIds(
|
||||
documentIds: string[]
|
||||
): Promise<Record<string, string>> {
|
||||
): Promise<Record<string, DocumentMetadata>> {
|
||||
if (documentIds.length === 0) {
|
||||
return {}
|
||||
}
|
||||
@@ -15,6 +28,7 @@ export async function getDocumentNamesByIds(
|
||||
.select({
|
||||
id: document.id,
|
||||
filename: document.filename,
|
||||
sourceUrl: document.sourceUrl,
|
||||
})
|
||||
.from(document)
|
||||
.where(
|
||||
@@ -26,12 +40,12 @@ export async function getDocumentNamesByIds(
|
||||
)
|
||||
)
|
||||
|
||||
const documentNameMap: Record<string, string> = {}
|
||||
const map: Record<string, DocumentMetadata> = {}
|
||||
documents.forEach((doc) => {
|
||||
documentNameMap[doc.id] = doc.filename
|
||||
map[doc.id] = { filename: doc.filename, sourceUrl: doc.sourceUrl ?? null }
|
||||
})
|
||||
|
||||
return documentNameMap
|
||||
return map
|
||||
}
|
||||
|
||||
export interface SearchResult {
|
||||
|
||||
@@ -15,7 +15,7 @@ const {
|
||||
mockHandleTagAndVectorSearch,
|
||||
mockGetQueryStrategy,
|
||||
mockGenerateSearchEmbedding,
|
||||
mockGetDocumentNamesByIds,
|
||||
mockGetDocumentMetadataByIds,
|
||||
mockAuthenticateRequest,
|
||||
mockValidateWorkspaceAccess,
|
||||
} = vi.hoisted(() => ({
|
||||
@@ -24,7 +24,7 @@ const {
|
||||
mockHandleTagAndVectorSearch: vi.fn(),
|
||||
mockGetQueryStrategy: vi.fn(),
|
||||
mockGenerateSearchEmbedding: vi.fn(),
|
||||
mockGetDocumentNamesByIds: vi.fn(),
|
||||
mockGetDocumentMetadataByIds: vi.fn(),
|
||||
mockAuthenticateRequest: vi.fn(),
|
||||
mockValidateWorkspaceAccess: vi.fn(),
|
||||
}))
|
||||
@@ -35,7 +35,7 @@ vi.mock('@/app/api/knowledge/search/utils', () => ({
|
||||
handleTagAndVectorSearch: mockHandleTagAndVectorSearch,
|
||||
getQueryStrategy: mockGetQueryStrategy,
|
||||
generateSearchEmbedding: mockGenerateSearchEmbedding,
|
||||
getDocumentNamesByIds: mockGetDocumentNamesByIds,
|
||||
getDocumentMetadataByIds: mockGetDocumentMetadataByIds,
|
||||
}))
|
||||
|
||||
vi.mock('@/app/api/knowledge/utils', () => knowledgeApiUtilsMock)
|
||||
@@ -81,7 +81,7 @@ describe('v1 knowledge search route — per-KB embedding model', () => {
|
||||
mockGetQueryStrategy.mockReturnValue({ distanceThreshold: 0.5 })
|
||||
mockGenerateSearchEmbedding.mockResolvedValue([0.1, 0.2, 0.3])
|
||||
mockHandleVectorOnlySearch.mockResolvedValue([])
|
||||
mockGetDocumentNamesByIds.mockResolvedValue({})
|
||||
mockGetDocumentMetadataByIds.mockResolvedValue({})
|
||||
})
|
||||
|
||||
it('passes the KB embedding model into generateSearchEmbedding', async () => {
|
||||
@@ -127,6 +127,42 @@ describe('v1 knowledge search route — per-KB embedding model', () => {
|
||||
expect(mockGenerateSearchEmbedding).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('surfaces sourceUrl from document metadata in search results', async () => {
|
||||
mockCheckKnowledgeBaseAccess.mockResolvedValueOnce({
|
||||
hasAccess: true,
|
||||
knowledgeBase: baseKb('kb-confluence', 'text-embedding-3-small'),
|
||||
})
|
||||
mockHandleVectorOnlySearch.mockResolvedValue([
|
||||
{
|
||||
documentId: 'doc-confluence',
|
||||
knowledgeBaseId: 'kb-confluence',
|
||||
content: 'page content',
|
||||
chunkIndex: 0,
|
||||
distance: 0.1,
|
||||
},
|
||||
])
|
||||
mockGetDocumentMetadataByIds.mockResolvedValue({
|
||||
'doc-confluence': {
|
||||
filename: 'Runbook.md',
|
||||
sourceUrl: 'https://example.atlassian.net/wiki/spaces/DOCS/pages/12345',
|
||||
},
|
||||
})
|
||||
|
||||
const req = createMockRequest('POST', {
|
||||
workspaceId: 'ws-1',
|
||||
knowledgeBaseIds: 'kb-confluence',
|
||||
query: 'runbook',
|
||||
})
|
||||
const res = await POST(req)
|
||||
const body = await res.json()
|
||||
|
||||
expect(res.status).toBe(200)
|
||||
expect(body.data.results[0].sourceUrl).toBe(
|
||||
'https://example.atlassian.net/wiki/spaces/DOCS/pages/12345'
|
||||
)
|
||||
expect(body.data.results[0].documentName).toBe('Runbook.md')
|
||||
})
|
||||
|
||||
it('allows tag-only search across mixed embedding models', async () => {
|
||||
mockHandleTagOnlySearch.mockResolvedValue([])
|
||||
mockCheckKnowledgeBaseAccess.mockResolvedValueOnce({
|
||||
|
||||
@@ -8,7 +8,7 @@ import { buildUndefinedTagsError, validateTagValue } from '@/lib/knowledge/tags/
|
||||
import type { StructuredFilter } from '@/lib/knowledge/types'
|
||||
import {
|
||||
generateSearchEmbedding,
|
||||
getDocumentNamesByIds,
|
||||
getDocumentMetadataByIds,
|
||||
getQueryStrategy,
|
||||
handleTagAndVectorSearch,
|
||||
handleTagOnlySearch,
|
||||
@@ -205,7 +205,7 @@ export const POST = withRouteHandler(async (request: NextRequest) => {
|
||||
})
|
||||
|
||||
const documentIds = results.map((r) => r.documentId)
|
||||
const documentNameMap = await getDocumentNamesByIds(documentIds)
|
||||
const documentMetadataMap = await getDocumentMetadataByIds(documentIds)
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
@@ -222,9 +222,11 @@ export const POST = withRouteHandler(async (request: NextRequest) => {
|
||||
}
|
||||
})
|
||||
|
||||
const docMeta = documentMetadataMap[result.documentId]
|
||||
return {
|
||||
documentId: result.documentId,
|
||||
documentName: documentNameMap[result.documentId] || undefined,
|
||||
documentName: docMeta?.filename || undefined,
|
||||
sourceUrl: docMeta?.sourceUrl ?? null,
|
||||
content: result.content,
|
||||
chunkIndex: result.chunkIndex,
|
||||
metadata: tags,
|
||||
|
||||
@@ -177,6 +177,12 @@ export const knowledgeSearchTool: ToolConfig<any, KnowledgeSearchResponse> = {
|
||||
properties: {
|
||||
documentId: { type: 'string', description: 'Document ID' },
|
||||
documentName: { type: 'string', description: 'Document name' },
|
||||
sourceUrl: {
|
||||
type: 'string',
|
||||
nullable: true,
|
||||
description:
|
||||
'URL to the original source document (e.g., Confluence page, Google Doc, Notion page). Null for documents without an external source.',
|
||||
},
|
||||
content: { type: 'string', description: 'Content of the result' },
|
||||
chunkIndex: { type: 'number', description: 'Index of the chunk within the document' },
|
||||
similarity: { type: 'number', description: 'Similarity score of the result' },
|
||||
|
||||
@@ -36,6 +36,7 @@ export function inferDocumentFileInfo(documentName: string): {
|
||||
export interface KnowledgeSearchResult {
|
||||
documentId: string
|
||||
documentName: string
|
||||
sourceUrl: string | null
|
||||
content: string
|
||||
chunkIndex: number
|
||||
metadata: Record<string, any>
|
||||
|
||||
@@ -46,6 +46,7 @@ export interface OutputProperty {
|
||||
type: OutputType
|
||||
description?: string
|
||||
optional?: boolean
|
||||
nullable?: boolean
|
||||
properties?: Record<string, OutputProperty>
|
||||
items?: {
|
||||
type: OutputType
|
||||
|
||||
Reference in New Issue
Block a user