Skip to content

Commit b445152

Browse files
committed
fix(knowledge): walk a saturated slice and re-read connector state at hydration
- fall back to a graph walk when the sliced sources hold more readable documents than one exact ranking may enumerate, since that enumeration has no order and would otherwise rank an arbitrary subset - read content under the full predicate, which re-reads each connector's own lifecycle and approval, so a source deleted, archived or unapproved mid-search stops answering at the gate that returns content
1 parent 6104e1e commit b445152

3 files changed

Lines changed: 62 additions & 29 deletions

File tree

‎apps/sim/lib/knowledge/search/diagnostics.ts‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -105,6 +105,8 @@ export interface SearchDiagnosticMetadata {
105105
/** Documents in a bounded permitted set. */
106106
permittedDocumentCount?: number
107107
vectorSourcesSliced?: number
108+
/** The sliced sources held more readable documents than one exact ranking may enumerate. */
109+
vectorSlicedSaturated?: boolean
108110
vectorSourcesWalked?: number
109111
/**
110112
* Which index ranked an unbounded keyword leg: `tin` ranks by BM25 and checks access on the top

‎apps/sim/lib/knowledge/search/queries.test.ts‎

Lines changed: 26 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1311,6 +1311,32 @@ describe('permitted-document planner', () => {
13111311
expect(JSON.stringify(exact[0])).toContain('sliced-src')
13121312
})
13131313

1314+
it('walks the sliced sources when more documents are readable than one ranking may enumerate', async () => {
1315+
const eligibility = { workspace: [], admin: ['sliced-src'], members: [] }
1316+
/** The slice enumerates in no order, so a saturated one would rank an arbitrary subset. */
1317+
sourceExactRows = [{ id: 'arbitrary-hit', distance: 0.4, saturated: true }]
1318+
traversedRows = [{ id: 'walked-hit', distance: 0.2 }]
1319+
rerankRows = [hit('walked-hit', 'sliced-src')]
1320+
queueTableRows(schemaMock.embedding, rerankRows)
1321+
await handleVectorOnlySearch({
1322+
...params,
1323+
permitted: { kind: 'unbounded' },
1324+
accessPlan: {
1325+
connectors: eligibility,
1326+
observers: { confirmed: [], observed: [] },
1327+
memberSources: [],
1328+
},
1329+
})
1330+
const walks = statements().filter((query) => query.sql.includes('AS visible'))
1331+
expect(walks).toHaveLength(1)
1332+
expect(JSON.stringify(walks[0])).toContain('sliced-src')
1333+
const reranked = JSON.stringify(
1334+
statements().find((query) => query.sql.includes('scored_search_candidates'))
1335+
)
1336+
expect(reranked).toContain('walked-hit')
1337+
expect(reranked).not.toContain('arbitrary-hit')
1338+
})
1339+
13141340
it('ranks every source exactly when the caller is a member of none', async () => {
13151341
sourceExactRows = [{ id: 'sliced-hit', distance: 0.05 }]
13161342
rerankRows = [hit('sliced-hit', 'sliced-src')]

‎apps/sim/lib/knowledge/search/queries.ts‎

Lines changed: 34 additions & 29 deletions
Original file line numberDiff line numberDiff line change
@@ -19,7 +19,6 @@ import {
1919
knowledgeAclOverlapCondition,
2020
knowledgeCandidateAccessConditionForConnectors,
2121
knowledgeMetadataCandidateAccessCondition,
22-
liveSourceAccessCondition,
2322
type SearchAccessPlan,
2423
textArrayLiteral,
2524
} from '@/lib/knowledge/access/predicate'
@@ -659,10 +658,10 @@ async function selectAuthorizedSearchResults(input: {
659658
/**
660659
* Loads the content of candidates that survived ranking, under the read predicate.
661660
*
662-
* Where the search resolved its connectors, the predicate takes them from that resolution instead
663-
* of proving each one again per row: the same documents, without the lookup this page already paid
664-
* for once. Mirrored permissions decide a reader here exactly as they did during ranking — a
665-
* document's own source is asked again when its content is read directly, not on this path.
661+
* The resolved connector state bounds ranking, never this. A page of ranked identifiers is small,
662+
* so its content is read under the full predicate, which re-reads each connector's own lifecycle
663+
* and approval: a source deleted, archived or unapproved while the search was running stops
664+
* answering here, at the gate that returns content.
666665
*/
667666
function hydrateSearchCandidates(
668667
ids: string[],
@@ -671,16 +670,9 @@ function hydrateSearchCandidates(
671670
filters: WorkspaceSearchFilters | undefined,
672671
conditions: (SQL | undefined)[],
673672
leg: RetrievalLeg,
674-
budget?: SearchBudget,
675-
plan?: SearchAccessPlan
673+
budget?: SearchBudget
676674
) {
677-
const accessCondition = plan
678-
? knowledgeCandidateAccessConditionForConnectors(
679-
access,
680-
plan,
681-
liveSourceAccessCondition(access)
682-
)
683-
: knowledgeAccessCondition(access)
675+
const accessCondition = knowledgeAccessCondition(access)
684676
return runSearchQuery(budget, `${leg}.sql`, (executor) =>
685677
executor
686678
.select(getSearchResultFields(distance))
@@ -768,8 +760,7 @@ export async function handleTagOnlySearch(params: SearchParams): Promise<SearchR
768760
params.filters,
769761
conditions,
770762
'tags',
771-
params.budget,
772-
params.accessPlan
763+
params.budget
773764
),
774765
})
775766
}
@@ -1101,8 +1092,10 @@ async function selectSourceVectorCandidates(input: {
11011092
)
11021093
const readable = and(...input.documentConditions, input.documentTagCondition)
11031094
type RankedChunks = Promise<Array<{ id: string; distance: number }>>
1104-
const walks: Array<() => RankedChunks> = sources.walked.map(
1105-
(connectorId) => () =>
1095+
/** Walks one source's own index, or the sliced sources together when their slice saturated. */
1096+
const walk =
1097+
(scope: SQL): (() => RankedChunks) =>
1098+
() =>
11061099
withVectorScanSettings(
11071100
(executor) =>
11081101
executor.execute<{ id: string; distance: number }>(sql`
@@ -1113,33 +1106,47 @@ async function selectSourceVectorCandidates(input: {
11131106
WHERE ${and(eq(document.id, embeddingSearch.documentId), readable)}
11141107
LIMIT 1
11151108
) AS visible
1116-
WHERE ${and(base, eq(embeddingSearch.connectorId, connectorId))}
1109+
WHERE ${and(base, scope)}
11171110
ORDER BY ${input.candidateDistance} LIMIT ${input.candidateLimit}`),
11181111
input.budget,
11191112
'vector.source_walk'
11201113
)
1114+
const walks: Array<() => RankedChunks> = sources.walked.map((connectorId) =>
1115+
walk(eq(embeddingSearch.connectorId, connectorId))
11211116
)
1117+
const slicedScope = sql`(${embeddingSearch.connectorId} IS NULL
1118+
OR ${embeddingSearch.connectorId} = ANY(${textArrayLiteral([...sources.sliced])}))`
11221119
/** One statement for every sliced source: their readable documents, then exact ranking of those. */
11231120
const slice: Array<() => RankedChunks> =
11241121
sources.sliced.length === 0
11251122
? []
11261123
: [
1127-
() =>
1128-
runSearchQuery(input.budget, 'vector.source_exact', (executor) =>
1129-
executor.execute<{ id: string; distance: number }>(sql`
1124+
async () => {
1125+
const rows = await runSearchQuery(input.budget, 'vector.source_exact', (executor) =>
1126+
executor.execute<{ id: string; distance: number; saturated: boolean }>(sql`
11301127
WITH readable_documents AS MATERIALIZED (
11311128
SELECT ${document.id} AS id FROM ${document}
11321129
WHERE (${document.connectorId} IS NULL
11331130
OR ${document.connectorId} = ANY(${textArrayLiteral([...sources.sliced])}))
11341131
AND ${readable}
1135-
LIMIT ${SOURCE_EXACT_DOCUMENT_LIMIT}
1132+
LIMIT ${SOURCE_EXACT_DOCUMENT_LIMIT + 1}
11361133
)
1137-
SELECT ${embeddingSearch.id} AS id, (${input.candidateDistance}) + 0 AS distance
1134+
SELECT ${embeddingSearch.id} AS id, (${input.candidateDistance}) + 0 AS distance,
1135+
(SELECT count(*) FROM readable_documents) > ${SOURCE_EXACT_DOCUMENT_LIMIT} AS saturated
11381136
FROM ${embeddingSearch}
11391137
JOIN readable_documents ON readable_documents.id = ${embeddingSearch.documentId}
11401138
WHERE ${base}
11411139
ORDER BY distance LIMIT ${input.candidateLimit}`)
1142-
),
1140+
)
1141+
/**
1142+
* The slice enumerates readable documents in no particular order, so a set past its
1143+
* bound would rank an arbitrary subset and could miss the nearest chunks entirely.
1144+
* Walk those sources instead: approximate, but drawn from the whole of them.
1145+
*/
1146+
if (!rows.some((row) => row.saturated)) return rows
1147+
annotateSearchDiagnostics({ vectorSlicedSaturated: true })
1148+
return walk(slicedScope)()
1149+
},
11431150
]
11441151
const scored = await mapWithConcurrency([...walks, ...slice], SOURCE_RANKING_CONCURRENCY, (run) =>
11451152
run()
@@ -1383,8 +1390,7 @@ async function selectVectorResults(params: SearchParams): Promise<SearchResult[]
13831390
params.filters,
13841391
conditions,
13851392
'vector',
1386-
params.budget,
1387-
params.accessPlan
1393+
params.budget
13881394
),
13891395
})
13901396
}
@@ -1643,8 +1649,7 @@ export async function executeKeywordSearch(params: KeywordSearchParams): Promise
16431649
params.filters,
16441650
conditions,
16451651
'keyword',
1646-
params.budget,
1647-
params.accessPlan
1652+
params.budget
16481653
),
16491654
})
16501655
}

0 commit comments

Comments
 (0)