Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/eval/baseline-v1.5.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,8 +25,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi
| nDCG@10 | 0.9437 |
| Evidence precision@5 | 0.2133 |

Timing is informational only and is **not** frozen: indexing 1941 ms, query
p50 14.39 ms, p95 27.71 ms on the
Timing is informational only and is **not** frozen: indexing 2515 ms, query
p50 13.99 ms, p95 23.62 ms on the
machine that produced this file. Timing and index size depend on hardware and on the
corpus, so they must never be the reason two runs differ.

Expand Down
20 changes: 20 additions & 0 deletions src/main/ipc/knowledgeHandlers.ts
Original file line number Diff line number Diff line change
Expand Up @@ -150,6 +150,26 @@ export function registerKnowledgeHandlers(knowledgeService: KnowledgeService) {
})
)

// 字面搜索(#96):BM25,与语义搜索分开返回,UI 不融合两者。
ipcMain.handle(
'knowledge:search-text',
validate(KnowledgeSchemas.searchText, async (params) => {
Logger.debug('KnowledgeHandlers', 'search-text:', params)

try {
const results = await knowledgeService.searchText(
params.notebookId,
params.query,
params.options ?? {}
)
return { success: true, results }
} catch (error) {
Logger.error('KnowledgeHandlers', 'Error searching text:', error)
return { success: false, error: (error as Error).message, results: [] }
}
})
)

// 获取文档列表
ipcMain.handle(
'knowledge:get-documents',
Expand Down
12 changes: 12 additions & 0 deletions src/main/ipc/validation.ts
Original file line number Diff line number Diff line change
Expand Up @@ -202,6 +202,18 @@ export const KnowledgeSchemas = {
includeContent: z.boolean().optional()
})
.optional()
}),

// 字面检索(#96):BM25 over chunks_fts,与语义检索分开呈现,不融合。
searchText: z.object({
notebookId: z.string().min(1, '笔记本 ID 不能为空'),
query: z.string().min(1, '搜索查询不能为空').max(1000, '搜索查询不能超过1000个字符'),
options: z
.object({
limit: z.number().int().min(1).max(100).optional(),
documentIds: z.array(z.string().min(1)).optional()
})
.optional()
})
}

Expand Down
42 changes: 41 additions & 1 deletion src/main/services/KnowledgeService.ts
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@ import {
type IdentifiedBlockDraft
} from './blocks/documentBlocks'
import { resolveChunkProvenance, type ChunkProvenance } from './chunkProvenance'
import { DenseRetriever } from './retrieval'
import { DenseRetriever, hydrateEvidence } from './retrieval'
import {
advanceRun,
completeRun,
Expand All @@ -60,6 +60,12 @@ import type {
Retriever
} from './retrieval'
import { WebFetchService } from './WebFetchService'
import {
deleteDocumentChunksFts,
ensureChunksFts,
indexChunksFts,
searchChunksFts
} from './fts'
import { vectorStoreManager } from '../vectorstore'
import Logger from '../../shared/utils/logger'

Expand Down Expand Up @@ -467,13 +473,25 @@ export class KnowledgeService {
// 4. 到此为止没有破坏任何东西。现在才替换旧的派生索引。
advanceRun(runId, 'finalizing', 85)
onProgress?.('saving_chunks', 85)
ensureChunksFts()
await this.clearDerivedIndex(documentId)

this.persistDocumentBlocks(blocks)

// 保存分块与 chunk↔block 映射(同一事务,不会出现没有映射的 chunk)
const { chunkIds } = this.saveChunks(documentId, notebookId, chunkResults, now)

// 字面检索索引(#96)。与 chunk 写入在同一个 pipeline 里,所以两者不会各自
// 漂移;就算漂移,`ensureChunksFts()` 的补齐也会在下次索引时自愈。
indexChunksFts(
chunkResults.map((chunk, index) => ({
chunkId: chunkIds[index],
notebookId,
documentId,
content: chunk.content
}))
)

// 5. 保存嵌入元数据并添加到向量存储
onProgress?.('saving_embeddings', 90)
const vectorStore = await vectorStoreManager.getStore(
Expand Down Expand Up @@ -561,6 +579,10 @@ export class KnowledgeService {
const doc = db.select().from(documents).where(eq(documents.id, documentId)).get()
if (!doc) return

// 字面索引与 chunk 一起清掉;表不存在时先建出来,免得删除本身成了第一个错误。
ensureChunksFts()
deleteDocumentChunksFts(documentId)

const oldChunks = db
.select({ id: chunks.id })
.from(chunks)
Expand Down Expand Up @@ -905,6 +927,24 @@ export class KnowledgeService {
return toSearchResults(evidence, { includeContent })
}

/**
* 字面检索(#96):BM25 over `chunks_fts`,可限定来源(#94 的 scope)。
*
* 与向量检索共用 `SearchResult` 形状,所以引用/定位链路不需要第二套。两个信号在
* UI 里分开呈现,**不**做融合 —— 融合是 chat 检索(#77)的事。
*/
async searchText(
notebookId: string,
query: string,
options: { limit?: number; documentIds?: string[] } = {}
): Promise<SearchResult[]> {
ensureChunksFts()
const hits = searchChunksFts(notebookId, query, options)
if (hits.length === 0) return []

return toSearchResults(hydrateEvidence(getDatabase(), hits))
}

/**
* 获取 notebook 的所有文档
*/
Expand Down
95 changes: 95 additions & 0 deletions src/main/services/fts.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
import { getSqlite } from '../db'
import {
backfillChunksFtsSql,
buildFtsMatchQuery,
createChunksFtsSql,
deleteChunksFtsSql,
insertChunksFtsSql,
searchChunksFtsSql
} from './ftsSql'

/**
* Full-text index over chunks (#96) — the database operations.
*
* The SQL lives in `ftsSql.ts` (pure, testable without Electron); this file runs it
* against the real connection.
*/

export * from './ftsSql'

/**
* 建表 + 补齐缺失的行。
*
* 每次调用都补齐而不是只建一次表:在「chunk 已写入、FTS 行还没写」之间崩溃,或一份
* 旧库在引入 FTS 之前就已经索引过,都会留下没有 FTS 行的 chunk。补齐让这两种情况
* 自愈,代价只是一条 INSERT..SELECT。
*/
export function ensureChunksFts(): void {
const sqlite = getSqlite()
if (!sqlite) return
sqlite.exec(createChunksFtsSql())
sqlite.exec(backfillChunksFtsSql())
}

export interface ChunksFtsRow {
chunkId: string
notebookId: string
documentId: string
content: string
}

/** 写入刚索引好的 chunk。 */
export function indexChunksFts(rows: readonly ChunksFtsRow[]): void {
if (rows.length === 0) return
const sqlite = getSqlite()
if (!sqlite) return

const statement = sqlite.prepare(insertChunksFtsSql())
const insert = sqlite.transaction((items: readonly ChunksFtsRow[]) => {
for (const item of items) {
statement.run(item.content, item.chunkId, item.notebookId, item.documentId)
}
})
insert(rows)
}

/** 删除一份文档的全部 FTS 行(重新索引、删除文档时)。 */
export function deleteDocumentChunksFts(documentId: string): void {
const sqlite = getSqlite()
if (!sqlite) return
sqlite.prepare(deleteChunksFtsSql('document_id', 1)).run(documentId)
}

export interface ChunksFtsHit {
chunkId: string
/** BM25 转成正数,越大越相关。 */
score: number
}

/**
* 字面检索:BM25 排序,可限定来源(#94 的 scope)。
*
* 返回的顺序就是 BM25 顺序;命中已被删除的 chunk 由调用方补齐证据时丢弃。
*/
export function searchChunksFts(
notebookId: string,
query: string,
options: { limit?: number; documentIds?: string[] } = {}
): ChunksFtsHit[] {
const match = buildFtsMatchQuery(query)
if (!match) return []

const sqlite = getSqlite()
if (!sqlite) return []

const documentIds = options.documentIds ?? []
const limit = options.limit ?? 20
const statement = sqlite.prepare(searchChunksFtsSql({ documentCount: documentIds.length }))
const rows = (
documentIds.length > 0
? statement.all(match, notebookId, ...documentIds, limit)
: statement.all(match, notebookId, limit)
) as Array<{ chunk_id: string; score: number }>

return rows.map((row) => ({ chunkId: row.chunk_id, score: -row.score }))
}
89 changes: 89 additions & 0 deletions src/main/services/ftsSql.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,89 @@
/**
* Full-text index SQL (#96) — pure builders, no database handle.
*
* Kept separate from `fts.ts` for the same reason `vectorTableSql.ts` is separate
* from `SQLiteVectorStore.ts`: the statements and the query sanitiser must be
* testable against the real FTS5 engine without loading Electron (the db module
* imports `electron`, which plain Node cannot).
*
* Two consumers share this index and must **not** share ranking semantics:
*
* - the global search surface presents literal matches as their own signal;
* - chat retrieval (#77) fuses BM25 with dense into one ranking.
*/

const CHUNKS_FTS_TABLE = 'chunks_fts'

export function createChunksFtsSql(): string {
// `content` is indexed; the ids are stored but not tokenized, so a result can be
// joined back to `chunks` without ever parsing the chunk text out of a snippet.
return (
`CREATE VIRTUAL TABLE IF NOT EXISTS ${CHUNKS_FTS_TABLE} USING fts5(` +
'content, chunk_id UNINDEXED, notebook_id UNINDEXED, document_id UNINDEXED)'
)
}

export function insertChunksFtsSql(): string {
return `INSERT INTO ${CHUNKS_FTS_TABLE} (content, chunk_id, notebook_id, document_id) VALUES (?, ?, ?, ?)`
}

/** Backfill rows for chunks indexed before the FTS table existed. */
export function backfillChunksFtsSql(): string {
return (
`INSERT INTO ${CHUNKS_FTS_TABLE} (content, chunk_id, notebook_id, document_id) ` +
'SELECT content, id, notebook_id, document_id FROM chunks ' +
`WHERE id NOT IN (SELECT chunk_id FROM ${CHUNKS_FTS_TABLE})`
)
}

export function deleteChunksFtsSql(column: 'chunk_id' | 'document_id', count: number): string {
if (count <= 0) throw new Error('deleteChunksFtsSql needs at least one id')
const placeholders = Array.from({ length: count }, () => '?').join(', ')
return `DELETE FROM ${CHUNKS_FTS_TABLE} WHERE ${column} IN (${placeholders})`
}

/**
* Turn free text from a search box into an FTS5 MATCH expression.
*
* Every whitespace-separated term is quoted, so the user's text can never be
* parsed as FTS operators (`-`, `*`, `:`, `NEAR`, unbalanced quotes) — a query
* that is a syntax error is a query that returns nothing, which reads as "no
* results" rather than "your search is malformed". Terms are ANDed, which is
* FTS5's default for space-separated tokens.
*
* Returns null when there is nothing to search for.
*/
export function buildFtsMatchQuery(raw: string): string | null {
const terms = raw
.trim()
.split(/\s+/)
.filter((term) => term.length > 0)
if (terms.length === 0) return null
return terms.map((term) => `"${term.replace(/"/g, '""')}"`).join(' ')
}

export interface FtsSearchSqlOptions {
/** Filter to these documents, e.g. the chat scope from #94. Empty = no filter. */
documentCount?: number
}

/**
* BM25-ranked search within one notebook.
*
* `bm25()` returns a negative number where more negative is a better match, so the
* caller orders ascending and negates it for display.
*/
export function searchChunksFtsSql(options: FtsSearchSqlOptions = {}): string {
const documentCount = options.documentCount ?? 0
const documentFilter =
documentCount > 0
? ` AND document_id IN (${Array.from({ length: documentCount }, () => '?').join(', ')})`
: ''

return (
`SELECT chunk_id, bm25(${CHUNKS_FTS_TABLE}) AS score ` +
`FROM ${CHUNKS_FTS_TABLE} ` +
`WHERE ${CHUNKS_FTS_TABLE} MATCH ? AND notebook_id = ?${documentFilter} ` +
'ORDER BY score ASC LIMIT ?'
)
}
22 changes: 22 additions & 0 deletions src/main/smokeTest.ts
Original file line number Diff line number Diff line change
Expand Up @@ -713,6 +713,28 @@ async function runChecks(): Promise<string[]> {
)
pass('DenseRetriever returns retrieved evidence with a page/block locator')

// --- the literal signal has its own index (#96) ----------------------------
// FTS is written with the chunks and deleted with them, so a re-index must leave
// exactly one row: zero means the search silently went blind, two means a stale
// row survived.
const ftsDocId = await knowledge.addDocument(reindexNotebook, {
title: 'FTS smoke',
type: 'text',
content: 'the quick brown zephyrine jumps over the lazy dog'
})
const literal = await knowledge.searchText(reindexNotebook, 'zephyrine')
assert(
literal.length === 1 && literal[0].documentId === ftsDocId,
'full-text search did not find a literal term in the document it was indexed from'
)
await knowledge.reindexDocument(ftsDocId)
const literalAfterReindex = await knowledge.searchText(reindexNotebook, 'zephyrine')
assert(
literalAfterReindex.length === 1 && literalAfterReindex[0].documentId === ftsDocId,
'a re-index left the full-text index broken or duplicated'
)
pass('full-text search finds literal terms and survives a re-index')

// --- a failed re-index leaves the previous index usable (#95) ---------------
// The failure this guards: `reindexDocument()` used to clear the derived index
// *before* embedding, so an embedding failure at 70% left the document with no
Expand Down
6 changes: 6 additions & 0 deletions src/preload/index.d.ts
Original file line number Diff line number Diff line change
Expand Up @@ -274,6 +274,12 @@ declare global {
query: string,
options?: SearchOptions
) => Promise<{ success: boolean; results: KnowledgeSearchResult[]; error?: string }>
/** 字面搜索(#96):BM25,与语义搜索分开返回。 */
searchText: (
notebookId: string,
query: string,
options?: { limit?: number; documentIds?: string[] }
) => Promise<{ success: boolean; results: KnowledgeSearchResult[]; error?: string }>

// 文档管理
getDocuments: (notebookId: string) => Promise<KnowledgeDocument[]>
Expand Down
6 changes: 6 additions & 0 deletions src/preload/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -176,6 +176,12 @@ const api = {
// 搜索
search: (notebookId: string, query: string, options?: any) =>
ipcRenderer.invoke('knowledge:search', { notebookId, query, options }),
// 字面搜索(#96)
searchText: (
notebookId: string,
query: string,
options?: { limit?: number; documentIds?: string[] }
) => ipcRenderer.invoke('knowledge:search-text', { notebookId, query, options }),

// 文档管理
getDocuments: (notebookId: string) =>
Expand Down
4 changes: 4 additions & 0 deletions src/renderer/src/components/notebook/NotebookLayout.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ import ResizableLayout from '../layouts/ResizableLayout'
import SourcePanel from './SourcePanel'
import ProcessPanel from './ProcessPanel'
import NotePanel from './NotePanel'
import SearchPalette from './chat/SearchPalette'
import { useNotebookStore } from '../../store/notebookStore'
import { useChatStore } from '../../store/chatStore'
import { useUIStore } from '../../store/uiStore'
Expand Down Expand Up @@ -113,6 +114,9 @@ export default function NotebookLayout(): ReactElement {
onClose={handleDiscardClose}
onConfirm={handleDiscardConfirm}
/>

{/* Global search (#96): Ctrl/Cmd+K from anywhere in the notebook. */}
<SearchPalette />
</div>
)
}
Loading
Loading