2026-06-07 01:10:31 +08:00
use crate ::app ::AppState ;
use crate ::context ::RequestContext ;
use crate ::error ::WebError ;
use crate ::routes ::evidence ::{ citation_markdown_for_locator , citation_url_for_locator };
use crate ::routes ::local_folder_source ;
2026-06-09 22:12:09 +08:00
use axum ::body ::Body ;
2026-06-07 01:10:31 +08:00
use axum ::extract ::{ Extension , Json , Query , State };
2026-06-09 22:12:09 +08:00
use axum ::http ::{ header , StatusCode };
use axum ::response ::Response ;
2026-06-07 01:10:31 +08:00
use core_protocol ::evidence ::{
EvidenceBBox , EvidenceLocator , EvidenceOpenAction , EvidenceResourceKind ,
};
2026-06-09 22:12:09 +08:00
use futures_util ::TryStreamExt ;
2026-06-07 01:10:31 +08:00
use serde ::{ Deserialize , Serialize };
use serde_json ::{ json , Value };
use std ::collections ::hash_map ::DefaultHasher ;
2026-06-09 09:20:56 +08:00
use std ::collections ::{ BTreeMap , BTreeSet };
2026-06-07 01:10:31 +08:00
use std ::env ;
use std ::fs ;
use std ::hash ::{ Hash , Hasher };
use std ::path ::{ Path , PathBuf };
use std ::time ::{ Duration , SystemTime , UNIX_EPOCH };
const REGISTRY_SCHEMA : & str = "mnote.knowledge_rag.source_registry.v1" ;
const REFERENCE_SCHEMA : & str = "mnote.knowledge_rag.reference.v1" ;
const DEFAULT_LIGHTRAG_ENDPOINT : & str = "http://127.0.0.1:9621" ;
const DEFAULT_LIGHTRAG_INPUT_DIR : & str = "/mnt/Data1T/Mnote_data/lightrag/inputs" ;
const DEFAULT_LIGHTRAG_WORKING_DIR : & str = "/mnt/Data1T/Mnote_data/lightrag/rag_storage" ;
const MAX_INGEST_SOURCES_PER_REQUEST : usize = 200 ;
2026-06-08 20:35:49 +08:00
const LIGHTRAG_PROVIDER_MIN_QUERY_CHARS : usize = 2 ;
2026-06-13 22:20:01 +08:00
const LARGE_DOCUMENT_SKIP_KG_MIN_BYTES : u64 = 10 * 1024 * 1024 ;
const DOCUMENT_STRUCTURE_INDEX_SCHEMA : & str = "mnote.knowledge_rag.document_structure_index.v1" ;
const DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PER_QUERY : usize = 8 ;
const DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PERSISTED : usize = 256 ;
const DOCUMENT_STRUCTURE_INDEX_MAX_SECTIONS_PER_DOC : usize = 2_000 ;
2026-06-07 10:35:21 +08:00
const SOURCE_SCOPE_MODE_POST_FILTER : & str = "post_filter_mapped_references" ;
2026-06-07 01:10:31 +08:00
const KNOWLEDGE_RAG_SOURCE_EXTENSIONS : & [ & str ] = & [
"md" , "markdown" , "txt" , "pdf" , "doc" , "docx" , "ppt" , "pptx" , "xls" , "xlsx" , "csv" , "png" ,
"jpg" , "jpeg" , "webp" , "gif" , "bmp" , "tif" , "tiff" ,
];
const LIGHTRAG_SCAN_SOURCE_EXTENSIONS : & [ & str ] = & [
"md" , "markdown" , "mdx" , "txt" , "pdf" , "docx" , "pptx" , "xlsx" , "rtf" , "odt" , "tex" , "epub" ,
2026-06-08 20:35:49 +08:00
"html" , "htm" , "png" , "jpg" , "jpeg" , "webp" , "gif" , "bmp" , "tif" , "tiff" ,
2026-06-07 01:10:31 +08:00
];
2026-06-09 09:20:56 +08:00
#[derive(Debug, Clone, Default)]
struct CitationTextBundle {
raw_quote : String ,
display_quote : String ,
locator_evidence_text : String ,
search_query : String ,
normalized_fingerprint : String ,
locator_text_source : & 'static str ,
display_cleaned : bool ,
}
2026-06-07 01:10:31 +08:00
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagStatusQuery {
pub workspace_id : Option < String > ,
pub root_uri : Option < String > ,
}
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagIngestRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
pub sources : Vec < KnowledgeRagSourceInput > ,
pub force : Option < bool > ,
}
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagSourceInput {
pub source_path : Option < String > ,
pub path : Option < String > ,
pub parser_hint : Option < String > ,
}
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagQueryRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
pub question : Option < String > ,
pub query : Option < String > ,
pub mode : Option < String > ,
pub top_k : Option < u32 > ,
pub chunk_top_k : Option < u32 > ,
pub include_chunk_content : Option < bool > ,
pub source_paths : Option < Vec < String >> ,
2026-06-13 22:20:01 +08:00
pub include_document_structure_index : Option < bool > ,
2026-06-07 01:10:31 +08:00
}
2026-06-08 20:35:49 +08:00
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagSearchRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
pub query : String ,
pub mode : Option < String > ,
pub top_k : Option < u32 > ,
pub chunk_top_k : Option < u32 > ,
pub include_chunk_content : Option < bool > ,
pub source_paths : Option < Vec < String >> ,
}
2026-06-07 01:10:31 +08:00
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagOpenReferenceRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
pub reference : Option < Value > ,
pub reference_id : Option < String > ,
pub file_path : Option < String > ,
pub chunk_id : Option < String > ,
}
2026-06-13 22:20:01 +08:00
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagSectionContextRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
pub source_path : Option < String > ,
pub source_id : Option < String > ,
pub light_rag_doc_id : Option < String > ,
pub file_path : Option < String > ,
pub section_id : Option < String > ,
pub start_block_ordinal : Option < u64 > ,
pub end_block_ordinal : Option < u64 > ,
pub start_paragraph_ordinal : Option < u32 > ,
pub end_paragraph_ordinal : Option < u32 > ,
pub context_before : Option < u64 > ,
pub context_after : Option < u64 > ,
pub max_blocks : Option < usize > ,
pub max_chars : Option < usize > ,
}
2026-06-07 01:10:31 +08:00
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagDeleteSourceRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
pub source_path : Option < String > ,
pub light_rag_doc_id : Option < String > ,
}
#[derive(Debug, Deserialize)]
#[serde(rename_all = "camelCase" )]
pub struct KnowledgeRagPruneRegistryRequest {
pub workspace_id : Option < String > ,
pub root_uri : String ,
}
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
#[serde(rename_all = "camelCase" )]
struct KnowledgeRagSourceRegistry {
schema : String ,
workspace_id : String ,
root_uri : String ,
updated_at_ms : u128 ,
2026-06-08 20:35:49 +08:00
#[serde(default)]
indexed_roots : Vec < KnowledgeRagIndexedRoot > ,
2026-06-07 01:10:31 +08:00
entries : Vec < KnowledgeRagSourceRegistryEntry > ,
}
2026-06-08 20:35:49 +08:00
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
#[serde(rename_all = "camelCase" )]
struct KnowledgeRagIndexedRoot {
root_relative_path : String ,
recursive : bool ,
#[serde(default)]
exclude_patterns : Vec < String > ,
run_on_change : Option < bool > ,
updated_at_ms : u128 ,
}
2026-06-07 01:10:31 +08:00
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(rename_all = "camelCase" )]
struct KnowledgeRagSourceRegistryEntry {
source_id : String ,
workspace_id : String ,
root_uri : String ,
source_path : String ,
source_root_relative_path : String ,
source_hash : String ,
light_rag_doc_id : Option < String > ,
#[serde(default, skip_serializing_if = "Option::is_none" )]
light_rag_status : Option < String > ,
light_rag_file_path : String ,
symlink_path : String ,
parser_hint : Option < String > ,
indexed_at_ms : Option < u128 > ,
deleted_at_ms : Option < u128 > ,
stale : bool ,
updated_at_ms : u128 ,
}
#[derive(Debug, Clone, Default)]
pub ( crate ) struct KnowledgeRagSourceStatuses {
pub ( crate ) indexed_paths : BTreeSet < String > ,
pub ( crate ) indexing_paths : BTreeSet < String > ,
pub ( crate ) failed_paths : BTreeSet < String > ,
}
pub ( crate ) fn knowledge_rag_source_statuses (
root_path : & Path ,
workspace_id : & str ,
root_uri : & str ,
) -> Result < KnowledgeRagSourceStatuses , WebError > {
let registry = read_registry ( root_path , workspace_id , root_uri ) ? ;
let mut statuses = KnowledgeRagSourceStatuses ::default ();
for entry in registry . entries {
let path = entry . source_root_relative_path . trim ();
if path . is_empty () {
continue ;
}
let provider_status = entry . light_rag_status . as_deref (). unwrap_or_default ();
if provider_status == "delete_completed" {
continue ;
}
if provider_status == "failed" {
statuses . failed_paths . insert ( path . to_string ());
2026-06-07 10:35:21 +08:00
} else if provider_status == "delete_retry_required" {
statuses . failed_paths . insert ( path . to_string ());
2026-06-07 01:10:31 +08:00
} else if provider_status == "delete_submitted"
|| ( entry . deleted_at_ms . is_some () && entry . light_rag_doc_id . is_some ())
{
statuses . indexing_paths . insert ( path . to_string ());
} else if entry . stale {
statuses . failed_paths . insert ( path . to_string ());
} else if entry . deleted_at_ms . is_some () && entry . light_rag_doc_id . is_none () {
continue ;
} else if entry . indexed_at_ms . is_some () && entry . light_rag_doc_id . is_some () {
statuses . indexed_paths . insert ( path . to_string ());
} else {
statuses . indexing_paths . insert ( path . to_string ());
}
}
Ok ( statuses )
}
pub async fn retired_local_ocr_endpoint (
Extension ( context ) : Extension < RequestContext > ,
) -> ( StatusCode , Json < Value > ) {
(
StatusCode ::GONE ,
Json ( json! ({
"ok" : false ,
"code" : "mnote_local_ocr_retired" ,
"message" : "本地 OCR sidecar 已退役;图片、PDF、Office 与索引统一交给 LightRAG 资料库处理。" ,
"replacement" : {
"provider" : "lightrag" ,
"status" : "/api/knowledge-rag/status" ,
"ingest" : "/api/knowledge-rag/ingest" ,
"deleteSource" : "/api/knowledge-rag/delete-source"
},
"requestId" : context . trace . request_id ,
"traceId" : context . trace . trace_id ,
})),
)
}
pub async fn status (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Query ( query ) : Query < KnowledgeRagStatusQuery > ,
) -> Result < Json < Value > , WebError > {
2026-06-08 20:35:49 +08:00
let ( registry , registry_diagnostics ) =
2026-06-07 01:10:31 +08:00
if let Some ( root_uri ) = query . root_uri . as_deref (). filter ( | value | ! value . is_empty ()) {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state , & context , root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( query . workspace_id . as_deref (), root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
2026-06-08 20:35:49 +08:00
let diagnostics = knowledge_rag_source_content_diagnostics ( & registry );
( Some ( registry ), diagnostics )
2026-06-07 01:10:31 +08:00
} else {
2026-06-08 20:35:49 +08:00
( None , Value ::Array ( Vec ::new ()))
2026-06-07 01:10:31 +08:00
};
let documents =
match lightrag_json ( reqwest ::Method ::GET , "/documents" , None , true , & context ). await {
Ok ( value ) => json! ({
"ok" : true ,
"rawStatusGroups" : lightrag_document_status_group_counts ( & value ),
"documents" : lightrag_document_summaries ( & value ),
}),
Err ( error ) => json! ({
"ok" : false ,
"code" : error . code (),
"message" : error . message (),
"rawStatusGroups" : {},
"documents" : [],
}),
};
2026-06-08 20:35:49 +08:00
let pipeline = match lightrag_json (
reqwest ::Method ::GET ,
"/documents/pipeline_status" ,
None ,
true ,
& context ,
)
. await
{
Ok ( value ) => lightrag_pipeline_status_summary ( & value ),
Err ( error ) => json! ({
"ok" : false ,
"code" : error . code (),
"message" : error . message (),
}),
};
2026-06-07 01:10:31 +08:00
let endpoint = lightrag_endpoint ();
2026-06-09 09:20:56 +08:00
let health_raw = lightrag_json ( reqwest ::Method ::GET , "/health" , None , false , & context ). await ;
let rerank_status = health_raw
. as_ref ()
. ok ()
. map ( lightrag_rerank_status_summary )
. unwrap_or_else ( || {
json! ({
"enabled" : false ,
"available" : false ,
"binding" : Value ::Null ,
"model" : Value ::Null ,
"status" : "unknown" ,
})
});
let health = match health_raw {
2026-06-07 01:10:31 +08:00
Ok ( value ) => json! ({
"ok" : true ,
"health" : value ,
}),
Err ( error ) => json! ({
"ok" : false ,
"code" : error . code (),
"message" : error . message (),
}),
};
Ok ( Json ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.provider_status.v1" ,
"provider" : "lightrag" ,
"endpoint" : endpoint ,
"dashboardUrl" : lightrag_dashboard_url (),
"inputDir" : lightrag_input_dir (). display (). to_string (),
"health" : health ,
2026-06-09 09:20:56 +08:00
"rerank" : rerank_status ,
2026-06-07 01:10:31 +08:00
"documents" : documents ,
2026-06-08 20:35:49 +08:00
"pipeline" : pipeline ,
2026-06-07 01:10:31 +08:00
"registry" : registry ,
2026-06-08 20:35:49 +08:00
"registryDiagnostics" : registry_diagnostics ,
2026-06-07 01:10:31 +08:00
})))
}
2026-06-09 22:12:09 +08:00
pub async fn pipeline_events (
Extension ( context ) : Extension < RequestContext > ,
) -> Result < Response , WebError > {
let endpoint = lightrag_endpoint ();
let url = format! (
" {} /documents/pipeline_status/events" ,
endpoint . trim_end_matches ( '/' )
);
let client = reqwest ::Client ::builder ()
. connect_timeout ( Duration ::from_secs ( 10 ))
. build ()
. map_err ( | error | {
WebError ::internal ( format! ( "LightRAG SSE client 初始化失败: {error} " ))
. with_context ( & context )
}) ? ;
let mut request = client . get ( & url ). header ( "accept" , "text/event-stream" );
if let Some ( api_key ) = lightrag_api_key () {
request = request . header ( "X-API-Key" , api_key );
}
let upstream = request . send (). await . map_err ( | error | {
WebError ::bad_gateway_code (
"knowledge_rag_lightrag_events_unreachable" ,
format! ( "无法连接 LightRAG pipeline 事件流: {error} " ),
)
. with_context ( & context )
}) ? ;
let upstream_status = upstream . status ();
if ! upstream_status . is_success () {
let text = upstream . text (). await . unwrap_or_default ();
return Err ( WebError ::bad_gateway_code (
"knowledge_rag_lightrag_events_error" ,
format! ( "LightRAG pipeline 事件流返回 HTTP {upstream_status} : {text} " ),
)
. with_context ( & context ));
}
let stream = upstream . bytes_stream (). map_err ( | error | {
std ::io ::Error ::new (
std ::io ::ErrorKind ::Other ,
format! ( "LightRAG pipeline 事件流读取失败: {error} " ),
)
});
Response ::builder ()
. status ( StatusCode ::OK )
. header ( header ::CONTENT_TYPE , "text/event-stream; charset=utf-8" )
. header ( header ::CACHE_CONTROL , "no-cache, no-transform" )
. header ( "x-accel-buffering" , "no" )
. body ( Body ::from_stream ( stream ))
. map_err ( | error | {
WebError ::internal ( format! ( "LightRAG pipeline 事件响应构造失败: {error} " ))
. with_context ( & context )
})
}
2026-06-07 01:10:31 +08:00
pub ( crate ) async fn sync_registry_for_root (
state : & AppState ,
context : & RequestContext ,
root_uri : & str ,
workspace_id : Option <& str > ,
) -> Result < (), WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
state , context , root_uri ,
)
. map_err ( | error | error . with_context ( context )) ? ;
let workspace_id = effective_workspace_id ( workspace_id , root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , context ). await
}
pub async fn ingest (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagIngestRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let input_dir = lightrag_input_dir ();
fs ::create_dir_all ( & input_dir ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_input_dir_unavailable" ,
format! ( "无法创建 LightRAG input 目录: {error} " ),
)
. with_context ( & context )
}) ? ;
let mut configured = Vec ::new ();
let mut expanded_count = 0 usize ;
let mut seen_sources = BTreeSet ::< String > ::new ();
let force = body . force . unwrap_or ( false );
2026-06-09 09:32:36 +08:00
let mut requested_sources = body . sources ;
if requested_sources . is_empty () {
requested_sources = registry
. indexed_roots
. iter ()
. map ( | root | KnowledgeRagSourceInput {
source_path : Some ( root . root_relative_path . clone ()),
path : None ,
parser_hint : None ,
})
. collect ();
}
if requested_sources . is_empty () {
return Err ( WebError ::bad_request_code (
"knowledge_rag_source_required" ,
"资料库索引缺少 sourcePath,且没有可重建的已保存索引范围" ,
)
. with_context ( & context ));
}
for source in requested_sources {
2026-06-07 01:10:31 +08:00
let source_path = source . source_path . or ( source . path ). ok_or_else ( || {
WebError ::bad_request_code ( "knowledge_rag_source_required" , "资料库索引缺少 sourcePath" )
. with_context ( & context )
}) ? ;
2026-06-08 20:35:49 +08:00
upsert_indexed_root_for_request ( & mut registry , & root_path , & source_path , & context ) ? ;
let requested_parser_hint = normalize_parser_hint ( source . parser_hint . as_deref (), & context ) ? ;
2026-06-07 01:10:31 +08:00
let resolved_sources = resolve_knowledge_rag_sources ( & root_path , & source_path , & context ) ? ;
for resolved in resolved_sources {
expanded_count += 1 ;
if expanded_count > MAX_INGEST_SOURCES_PER_REQUEST {
return Err ( WebError ::bad_request_code (
"knowledge_rag_source_limit_exceeded" ,
format! (
"单次资料库索引最多支持 {MAX_INGEST_SOURCES_PER_REQUEST} 个文件,请缩小目录范围"
),
)
. with_context ( & context ));
}
let canonical = resolved . canonical_path ;
let canonical_key = canonical . display (). to_string ();
if ! seen_sources . insert ( canonical_key . clone ()) {
configured . push ( json! ({
"sourcePath" : canonical_key ,
"requestedSourcePath" : resolved . requested_path ,
"sourceKind" : resolved . source_kind ,
"skipped" : true ,
"skipReason" : "duplicate_request" ,
}));
continue ;
}
let relative = root_relative_path ( & root_path , & canonical ) ? ;
let source_hash = source_hash ( & canonical ) ? ;
let file_name = canonical
. file_name ()
. and_then ( | value | value . to_str ())
. unwrap_or ( "source" );
2026-06-08 20:35:49 +08:00
let parser_hint = requested_parser_hint
. clone ()
. or_else ( || default_lightrag_parser_hint_for_source ( & canonical , file_name ));
2026-06-07 01:10:31 +08:00
let direct_scan_source = lightrag_scan_supported_file ( & canonical );
2026-06-13 22:20:01 +08:00
let reingest_doc_ids = if force {
registry
. entries
. iter ()
. filter ( | entry | {
entry . source_path == canonical_key
&& entry . deleted_at_ms . is_none ()
&& ! entry . stale
})
. filter_map ( | entry | entry . light_rag_doc_id . clone ())
. collect ::< Vec < _ >> ()
} else {
Vec ::new ()
};
let provider_delete = if reingest_doc_ids . is_empty () {
Value ::Null
} else {
delete_lightrag_documents_for_reingest ( reingest_doc_ids , & context ). await ?
};
2026-06-07 01:10:31 +08:00
if ! force {
if let Some ( existing ) = registry . entries . iter (). find ( | entry | {
entry . source_path == canonical_key
&& entry . source_hash == source_hash
&& entry . deleted_at_ms . is_none ()
&& ! entry . stale
&& ( entry . indexed_at_ms . is_some ()
|| matches! (
entry . light_rag_status . as_deref (),
Some ( "processing" | "pending" | "submitted" | "parsed" )
))
}) {
configured . push ( json! ({
"sourceId" : existing . source_id ,
"sourcePath" : existing . source_path ,
"sourceRootRelativePath" : existing . source_root_relative_path ,
"requestedSourcePath" : resolved . requested_path ,
"sourceKind" : resolved . source_kind ,
"lightRagFilePath" : existing . light_rag_file_path ,
"lightRagDocId" : existing . light_rag_doc_id ,
"lightRagStatus" : existing . light_rag_status ,
"skipped" : true ,
"skipReason" : "already_registered" ,
}));
continue ;
}
}
let staged_source = stage_lightrag_source (
& canonical ,
file_name ,
parser_hint . as_deref (),
& input_dir ,
& context ,
) ? ;
let light_rag_file_path = staged_source . light_rag_file_path . clone ();
let symlink_path = staged_source . staged_path . clone ();
let now = now_ms ();
let source_id = format! (
"lightrag-source- {} " ,
short_hash ( & canonical . display (). to_string ())
);
upsert_registry_entry (
& mut registry ,
KnowledgeRagSourceRegistryEntry {
source_id : source_id . clone (),
workspace_id : workspace_id . clone (),
root_uri : body . root_uri . clone (),
source_path : canonical_key . clone (),
source_root_relative_path : relative . clone (),
source_hash : source_hash . clone (),
light_rag_doc_id : None ,
light_rag_status : Some ( "submitted" . into ()),
light_rag_file_path : light_rag_file_path . clone (),
symlink_path : symlink_path . display (). to_string (),
parser_hint : parser_hint . clone (),
indexed_at_ms : None ,
deleted_at_ms : None ,
stale : false ,
updated_at_ms : now ,
},
);
configured . push ( json! ({
"sourceId" : source_id ,
"sourcePath" : canonical_key ,
"sourceRootRelativePath" : relative ,
"requestedSourcePath" : resolved . requested_path ,
"sourceKind" : resolved . source_kind ,
"lightRagFilePath" : light_rag_file_path ,
"symlinkPath" : symlink_path . display (). to_string (),
2026-06-13 22:20:01 +08:00
"parserHint" : parser_hint ,
"providerDelete" : provider_delete ,
2026-06-07 01:10:31 +08:00
"scanMode" : if direct_scan_source { "direct" } else { "markdown_wrapper" },
"lightRagStatus" : "submitted" ,
}));
}
}
write_registry ( & root_path , & mut registry ) ? ;
let scan = lightrag_json (
reqwest ::Method ::POST ,
"/documents/scan" ,
None ,
true ,
& context ,
)
. await ? ;
let scan_status = scan . get ( "status" ). and_then ( Value ::as_str ). unwrap_or ( "" );
let retry_required = scan_status == "scanning_skipped_pipeline_busy" ;
Ok ( Json ( json! ({
"ok" : ! retry_required ,
"schema" : "mnote.knowledge_rag.ingest_result.v1" ,
"provider" : "lightrag" ,
"force" : body . force . unwrap_or ( false ),
"configuredSources" : configured ,
"scan" : scan ,
"retryRequired" : retry_required ,
"registry" : registry ,
})))
}
pub async fn query_rag (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagQueryRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let query = body . question . or ( body . query ). ok_or_else ( || {
WebError ::bad_request_code ( "knowledge_rag_query_required" , "资料库问答缺少 query" )
. with_context ( & context )
}) ? ;
2026-06-08 20:35:49 +08:00
validate_lightrag_provider_query_length ( & query , "knowledge_rag_query_too_short" , & context ) ? ;
2026-06-13 22:20:01 +08:00
let source_scope = normalize_source_scope ( body . source_paths . as_deref ());
let requested_mode = normalize_lightrag_query_mode ( body . mode . as_deref ());
let mode_decision =
resolve_lightrag_query_mode_for_scope ( & registry , & source_scope , & requested_mode );
let mode = mode_decision . mode . clone ();
2026-06-07 01:10:31 +08:00
let raw = lightrag_json (
reqwest ::Method ::POST ,
"/query/data" ,
Some ( json! ({
"query" : query . clone (),
2026-06-09 09:20:56 +08:00
"mode" : mode . clone (),
2026-06-07 01:10:31 +08:00
"top_k" : body . top_k ,
"chunk_top_k" : body . chunk_top_k ,
"include_references" : true ,
"include_chunk_content" : body . include_chunk_content . unwrap_or ( true ),
})),
true ,
& context ,
)
. await ? ;
2026-06-08 20:35:49 +08:00
let mut references =
mapped_references ( & raw , & registry , & body . root_uri , & root_path , Some ( & query ));
2026-06-07 01:10:31 +08:00
filter_mapped_references_by_source_scope ( & mut references , & source_scope );
2026-06-13 22:20:01 +08:00
let include_document_structure_index = body
. include_document_structure_index
. unwrap_or_else ( || mode_decision . reason == "source_scope_skip_kg_document" );
let document_structure_index = if include_document_structure_index {
document_structure_index_payload (
& root_path ,
& registry ,
& source_scope ,
& references ,
Some ( & query ),
DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PER_QUERY ,
) ?
} else {
Value ::Null
};
2026-06-07 01:10:31 +08:00
Ok ( Json ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.query_result.v1" ,
"provider" : "lightrag" ,
2026-06-13 22:20:01 +08:00
"requestedRetrievalMode" : requested_mode ,
2026-06-09 09:20:56 +08:00
"retrievalMode" : mode ,
2026-06-13 22:20:01 +08:00
"effectiveRetrievalMode" : mode_decision . mode ,
"retrievalModeReason" : mode_decision . reason ,
"documentStructureIndexIncluded" : include_document_structure_index ,
"documentStructureIndexReason" : if include_document_structure_index { "requested_or_skip_kg_source_scope" } else { "not_requested" },
2026-06-07 01:10:31 +08:00
"sourceScope" : source_scope ,
2026-06-07 10:35:21 +08:00
"sourceScopeMode" : SOURCE_SCOPE_MODE_POST_FILTER ,
"rawScopeFiltered" : false ,
2026-06-07 01:10:31 +08:00
"raw" : raw ,
2026-06-09 09:20:56 +08:00
"citations" : knowledge_rag_citations ( & references ),
2026-06-07 01:10:31 +08:00
"references" : references ,
2026-06-13 22:20:01 +08:00
"documentStructureIndex" : document_structure_index ,
2026-06-07 01:10:31 +08:00
})))
}
2026-06-08 20:35:49 +08:00
pub async fn search (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagSearchRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let query = body . query . trim (). to_string ();
if query . is_empty () {
return Err ( WebError ::bad_request_code (
"knowledge_rag_search_query_required" ,
"资料库检索缺少 query" ,
)
. with_context ( & context ));
}
validate_lightrag_provider_query_length (
& query ,
"knowledge_rag_search_query_too_short" ,
& context ,
) ? ;
2026-06-13 22:20:01 +08:00
let source_scope = normalize_source_scope ( body . source_paths . as_deref ());
let requested_search_mode = normalize_knowledge_rag_search_mode ( body . mode . as_deref ());
let mode_decision = resolve_knowledge_rag_search_mode_for_scope (
& registry ,
& source_scope ,
& requested_search_mode ,
);
let search_mode = mode_decision . mode . clone ();
2026-06-09 09:20:56 +08:00
let raw = if search_mode == "exact" {
lightrag_json (
reqwest ::Method ::POST ,
"/query/search" ,
Some ( json! ({
"query" : query . clone (),
"limit" : body . top_k . or ( body . chunk_top_k ). unwrap_or ( 50 ),
"max_per_chunk" : 12 ,
"include_chunk_content" : body . include_chunk_content . unwrap_or ( true ),
2026-06-09 18:40:48 +08:00
"include_sidecar" : true ,
2026-06-09 09:20:56 +08:00
})),
true ,
& context ,
)
. await ?
} else {
lightrag_json (
reqwest ::Method ::POST ,
"/query/data" ,
Some ( json! ({
"query" : query . clone (),
"mode" : search_mode . clone (),
"top_k" : body . top_k ,
"chunk_top_k" : body . chunk_top_k ,
"include_references" : true ,
"include_chunk_content" : body . include_chunk_content . unwrap_or ( true ),
})),
true ,
& context ,
)
. await ?
};
2026-06-08 20:35:49 +08:00
let mut references =
mapped_references ( & raw , & registry , & body . root_uri , & root_path , Some ( & query ));
filter_mapped_references_by_source_scope ( & mut references , & source_scope );
2026-06-09 09:20:56 +08:00
if search_mode == "exact" {
filter_mapped_references_by_search_query ( & mut references , & query );
rank_mapped_references_for_query ( & mut references , & query );
}
2026-06-08 20:35:49 +08:00
dedupe_mapped_references_by_locator ( & mut references );
let results = references
. iter ()
. enumerate ()
. map ( | ( index , reference ) | {
knowledge_rag_search_result ( reference , index , & body . root_uri , & query )
})
. collect ::< Vec < _ >> ();
Ok ( Json ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.search_results.v1" ,
"provider" : "lightrag" ,
"query" : query ,
2026-06-13 22:20:01 +08:00
"requestedRetrievalMode" : requested_search_mode ,
2026-06-09 09:20:56 +08:00
"retrievalMode" : search_mode ,
2026-06-13 22:20:01 +08:00
"effectiveRetrievalMode" : mode_decision . mode ,
"retrievalModeReason" : mode_decision . reason ,
2026-06-09 09:20:56 +08:00
"rawRetrievalMode" : raw
. get ( "metadata" )
. and_then ( | metadata | metadata . get ( "query_mode" ))
. cloned ()
. unwrap_or ( Value ::Null ),
2026-06-08 20:35:49 +08:00
"sourceScope" : source_scope ,
"sourceScopeMode" : SOURCE_SCOPE_MODE_POST_FILTER ,
"rawScopeFiltered" : false ,
"results" : results ,
2026-06-09 09:20:56 +08:00
"citations" : knowledge_rag_citations ( & references ),
2026-06-08 20:35:49 +08:00
"references" : references ,
"registry" : {
"schema" : registry . schema ,
"workspaceId" : registry . workspace_id ,
"rootUri" : registry . root_uri ,
"updatedAtMs" : registry . updated_at_ms ,
"indexedRoots" : registry . indexed_roots ,
"sourceCount" : registry . entries . len (),
},
})))
}
fn knowledge_rag_search_result (
reference : & Value ,
index : usize ,
root_uri : & str ,
query : & str ,
) -> Value {
let source_path = reference
. get ( "sourceRootRelativePath" )
. and_then ( Value ::as_str )
. or_else ( || reference . get ( "filePath" ). and_then ( Value ::as_str ))
. unwrap_or_default ();
let title = Path ::new ( source_path )
. file_name ()
. and_then ( | value | value . to_str ())
. unwrap_or ( source_path );
let chunk_id = reference
. get ( "chunkId" )
. and_then ( Value ::as_str )
. or_else ( || {
reference
. get ( "chunkId" )
. and_then ( | value | value . get ( "id" ))
. and_then ( Value ::as_str )
})
. unwrap_or_default ();
let source_id = reference
. get ( "sourceId" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let id = if ! source_id . is_empty () || ! chunk_id . is_empty () {
format! ( "knowledge-rag: {source_id} : {chunk_id} " )
} else {
format! ( "knowledge-rag:reference: {index} " )
};
2026-06-09 09:20:56 +08:00
let quote = reference_display_quote ( reference );
2026-06-08 20:35:49 +08:00
let locator_precision = locator_precision_for_reference ( reference );
json! ({
"id" : id ,
"documentId" : id ,
2026-06-09 09:20:56 +08:00
"citationId" : reference . get ( "citationId" ). cloned (). unwrap_or ( Value ::Null ),
"citationLabel" : reference . get ( "citationLabel" ). cloned (). unwrap_or ( Value ::Null ),
2026-06-08 20:35:49 +08:00
"title" : title ,
"path" : source_path ,
"resourceType" : resource_type_for_path ( source_path ),
"sourceKind" : "local_folder" ,
"rootUri" : root_uri ,
"snippet" : quote . chars (). take ( 220 ). collect ::< String > (),
"quote" : quote ,
2026-06-09 09:20:56 +08:00
"rawQuote" : reference . get ( "rawQuote" ). cloned (). unwrap_or ( Value ::Null ),
"displayQuote" : reference . get ( "displayQuote" ). cloned (). unwrap_or_else ( || json! ( quote )),
"locatorEvidenceText" : reference . get ( "locatorEvidenceText" ). cloned (). unwrap_or ( Value ::Null ),
"normalizedFingerprint" : reference . get ( "normalizedFingerprint" ). cloned (). unwrap_or ( Value ::Null ),
"citationDiagnostics" : reference . get ( "citationDiagnostics" ). cloned (). unwrap_or ( Value ::Null ),
2026-06-08 20:35:49 +08:00
"matchSource" : reference . get ( "matchSource" ). and_then ( Value ::as_str ). unwrap_or ( "lightrag_reference" ),
"occurrenceIndex" : reference . get ( "occurrenceIndex" ). cloned (). unwrap_or ( Value ::Null ),
"provider" : "lightrag" ,
"query" : query ,
"hasOcr" : true ,
"locator" : reference . get ( "locator" ). cloned (). unwrap_or ( Value ::Null ),
"locatorDegraded" : reference . get ( "locatorDegraded" ). cloned (). unwrap_or_else ( || json! ( true )),
"locatorPrecision" : locator_precision ,
"citationUrl" : reference . get ( "citationUrl" ). cloned (). unwrap_or ( Value ::Null ),
"citationMarkdown" : reference . get ( "citationMarkdown" ). cloned (). unwrap_or ( Value ::Null ),
"publicPath" : reference . get ( "citationUrl" ). cloned (). unwrap_or ( Value ::Null ),
"openAction" : reference . get ( "openAction" ). cloned (). unwrap_or ( Value ::Null ),
"source" : {
"locator" : reference . get ( "locator" ). cloned (). unwrap_or ( Value ::Null ),
"reference" : reference ,
}
})
}
2026-06-09 09:20:56 +08:00
fn knowledge_rag_citations ( references : & [ Value ]) -> Vec < Value > {
references
. iter ()
. enumerate ()
. filter_map ( | ( index , reference ) | knowledge_rag_citation ( reference , index ))
. collect ()
}
fn knowledge_rag_citation ( reference : & Value , index : usize ) -> Option < Value > {
let citation_url = reference . get ( "citationUrl" ). cloned (). unwrap_or ( Value ::Null );
if citation_url . is_null () && reference . get ( "locator" ). is_none_or ( Value ::is_null ) {
return None ;
}
let citation_id = reference
. get ( "citationId" )
. and_then ( Value ::as_str )
. map ( ToOwned ::to_owned )
. unwrap_or_else ( || {
short_citation_id (
reference
. get ( "sourceId" )
. and_then ( Value ::as_str )
. unwrap_or_default (),
reference
. get ( "chunkId" )
. and_then ( Value ::as_str )
. unwrap_or_default (),
reference
. get ( "locator" )
. and_then ( | locator | locator . get ( "blockId" ))
. and_then ( Value ::as_str )
. unwrap_or_default (),
index as u64 ,
reference
. get ( "normalizedFingerprint" )
. and_then ( Value ::as_str )
. unwrap_or_default (),
)
});
Some ( json! ({
"schema" : "mnote.knowledge_rag.citation.v1" ,
"provider" : reference . get ( "provider" ). cloned (). unwrap_or_else ( || json! ( "lightrag" )),
"citationId" : citation_id . clone (),
"citationLabel" : format ! ( "[{}]" , citation_id ),
"sourceId" : reference . get ( "sourceId" ). cloned (). unwrap_or ( Value ::Null ),
"sourcePath" : reference . get ( "sourcePath" ). cloned (). unwrap_or ( Value ::Null ),
"sourceRootRelativePath" : reference . get ( "sourceRootRelativePath" ). cloned (). unwrap_or ( Value ::Null ),
"lightRagFilePath" : reference . get ( "filePath" ). cloned (). unwrap_or ( Value ::Null ),
"lightRagChunkId" : reference . get ( "chunkId" ). cloned (). unwrap_or ( Value ::Null ),
"blockId" : reference
. get ( "locator" )
. and_then ( | locator | locator . get ( "blockId" ))
. cloned ()
. unwrap_or ( Value ::Null ),
"headingPath" : reference . get ( "headingPath" ). cloned (). unwrap_or_else ( || json! ([])),
"rawQuote" : reference . get ( "rawQuote" ). cloned (). unwrap_or ( Value ::Null ),
"displayQuote" : reference . get ( "displayQuote" ). cloned (). unwrap_or_else ( || json! ( reference_display_quote ( reference ))),
"locatorEvidenceText" : reference . get ( "locatorEvidenceText" ). cloned (). unwrap_or_else ( || json! ( reference_locator_evidence_text ( reference ))),
"searchQuery" : reference . get ( "searchQuery" ). cloned (). unwrap_or ( Value ::Null ),
"quoteSource" : reference . get ( "quoteSource" ). cloned (). unwrap_or ( Value ::Null ),
"locatorPrecision" : reference . get ( "locatorPrecision" ). cloned (). unwrap_or_else ( || json! ( locator_precision_for_reference ( reference ))),
"locatorDegraded" : reference . get ( "locatorDegraded" ). cloned (). unwrap_or_else ( || json! ( true )),
"locator" : reference . get ( "locator" ). cloned (). unwrap_or ( Value ::Null ),
"citationUrl" : citation_url ,
"citationMarkdown" : reference . get ( "citationMarkdown" ). cloned (). unwrap_or ( Value ::Null ),
"relevanceScore" : reference . get ( "relevanceScore" ). cloned (). unwrap_or ( Value ::Null ),
"diagnostics" : reference . get ( "citationDiagnostics" ). cloned (). unwrap_or ( Value ::Null ),
}))
}
fn heading_path_for_sidecar_block ( block : Option <& Value > ) -> Vec < String > {
let Some ( block ) = block else {
return Vec ::new ();
};
let mut headings = block
. get ( "parent_headings" )
. or_else ( || block . get ( "parentHeadings" ))
. and_then ( Value ::as_array )
. map ( | items | {
items
. iter ()
. filter_map ( Value ::as_str )
. map ( clean_lightrag_text_for_display )
. filter ( | value | ! value . is_empty ())
. collect ::< Vec < _ >> ()
})
. unwrap_or_default ();
if let Some ( heading ) = block
. get ( "heading" )
. and_then ( Value ::as_str )
. map ( clean_lightrag_text_for_display )
. filter ( | value | ! value . is_empty ())
{
if headings . last (). is_none_or ( | last | last != & heading ) {
headings . push ( heading );
}
}
headings
}
2026-06-08 20:35:49 +08:00
fn locator_precision_for_reference ( reference : & Value ) -> & 'static str {
let locator = reference . get ( "locator" ). unwrap_or ( & Value ::Null );
locator_precision_for_locator_value ( locator )
}
fn locator_precision_for_locator_value ( locator : & Value ) -> & 'static str {
if locator . is_null () {
return "file" ;
}
let has_bbox = locator . get ( "bbox" ). is_some_and ( | value | ! value . is_null ());
let has_page = locator
. get ( "page" )
. and_then ( Value ::as_u64 )
. is_some_and ( | page | page > 0 );
if has_bbox && has_page {
return "bbox" ;
}
if has_page {
return "page" ;
}
let has_block = locator
. get ( "blockId" )
. and_then ( Value ::as_str )
. is_some_and ( | value | ! value . trim (). is_empty ());
if has_block {
return "paragraph" ;
}
"file"
}
fn locator_precision_for_locator ( locator : & EvidenceLocator ) -> & 'static str {
if locator . page . is_some () && locator . bbox . is_some () {
return "bbox" ;
}
if locator . page . is_some () {
return "page" ;
}
if locator
. block_id
. as_deref ()
. is_some_and ( | value | ! value . trim (). is_empty ())
{
return "paragraph" ;
}
"file"
}
fn non_whitespace_char_count ( query : & str ) -> usize {
query . chars (). filter ( | ch | ! ch . is_whitespace ()). count ()
}
fn validate_lightrag_provider_query_length (
query : & str ,
code : & 'static str ,
context : & RequestContext ,
) -> Result < (), WebError > {
if non_whitespace_char_count ( query ) >= LIGHTRAG_PROVIDER_MIN_QUERY_CHARS {
return Ok (());
}
Err ( WebError ::bad_request_code ( code , "请输入至少 2 个字再搜索" ). with_context ( context ))
}
2026-06-07 01:10:31 +08:00
fn normalize_source_scope ( source_paths : Option <& [ String ] > ) -> Vec < String > {
source_paths
. unwrap_or ( & [])
. iter ()
. map ( | value | value . trim (). trim_matches ( '/' ). replace ( '\\' , "/" ))
. filter ( | value | ! value . is_empty () && value != "." )
. collect ::< std ::collections ::BTreeSet < _ >> ()
. into_iter ()
. collect ()
}
fn filter_mapped_references_by_source_scope ( references : & mut Vec < Value > , source_scope : & [ String ]) {
if source_scope . is_empty () {
return ;
}
references . retain ( | reference | {
let source_path = reference
. get ( "sourceRootRelativePath" )
. and_then ( Value ::as_str )
. unwrap_or_default ()
. trim_matches ( '/' )
. replace ( '\\' , "/" );
source_scope
. iter ()
. any ( | scope | source_path == * scope || source_path . starts_with ( & format! ( " {scope} /" )))
});
}
2026-06-08 20:35:49 +08:00
fn filter_mapped_references_by_search_query ( references : & mut Vec < Value > , query : & str ) {
let query_normalized = normalize_text_for_match ( query ). to_ascii_lowercase ();
if query_normalized . is_empty () {
return ;
}
references
. retain ( | reference | mapped_reference_matches_search_query ( reference , & query_normalized ));
}
fn mapped_reference_matches_search_query ( reference : & Value , query_normalized : & str ) -> bool {
2026-06-09 09:20:56 +08:00
let quote = reference_display_quote ( reference );
2026-06-08 20:35:49 +08:00
let source_path = reference
. get ( "sourceRootRelativePath" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let combined =
normalize_text_for_match ( & format! ( " {source_path} \n {quote} " )). to_ascii_lowercase ();
if combined . contains ( query_normalized ) {
return true ;
}
let terms = query_match_terms ( query_normalized );
2026-06-09 09:20:56 +08:00
if query_normalized . chars (). any ( is_cjk_char ) {
if query_is_single_cjk_lookup ( query_normalized ) {
return false ;
}
return terms
. iter ()
. filter ( | term | significant_cjk_query_term ( term ))
. any ( | term | block_matches_query_term ( & combined , term ));
}
2026-06-08 20:35:49 +08:00
! terms . is_empty ()
&& terms
. iter ()
. all ( | term | block_matches_query_term ( & combined , term ))
}
fn dedupe_mapped_references_by_locator ( references : & mut Vec < Value > ) {
let mut seen = BTreeSet ::< String > ::new ();
references . retain ( | reference | {
let source = reference
. get ( "sourceRootRelativePath" )
. and_then ( Value ::as_str )
. or_else ( || reference . get ( "filePath" ). and_then ( Value ::as_str ))
. unwrap_or_default ();
let block_id = reference
. get ( "locator" )
. and_then ( | locator | locator . get ( "blockId" ))
. and_then ( Value ::as_str )
. unwrap_or_default ();
let chunk_id = reference
. get ( "chunkId" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
2026-06-09 09:20:56 +08:00
let fingerprint = reference
. get ( "normalizedFingerprint" )
. and_then ( Value ::as_str )
. filter ( | value | ! value . trim (). is_empty ())
. map ( ToOwned ::to_owned )
. unwrap_or_else ( || normalized_quote_fingerprint ( reference_display_quote ( reference )));
2026-06-08 20:35:49 +08:00
let key = if ! block_id . is_empty () {
format! ( " {source} \n block: {block_id} " )
} else if ! chunk_id . is_empty () {
2026-06-09 09:20:56 +08:00
format! ( " {source} \n chunk: {chunk_id} \n fingerprint: {fingerprint} " )
2026-06-08 20:35:49 +08:00
} else {
2026-06-09 09:20:56 +08:00
format! ( " {source} \n quote: {fingerprint} " )
2026-06-08 20:35:49 +08:00
};
seen . insert ( key )
});
}
2026-06-07 01:10:31 +08:00
fn rank_mapped_references_for_query ( references : & mut Vec < Value > , query : & str ) {
let mut indexed = references
. drain ( .. )
. enumerate ()
. map ( | ( index , reference ) | {
let score = reference_query_score ( & reference , query );
( index , score , reference )
})
. collect ::< Vec < _ >> ();
indexed . sort_by ( | left , right | right . 1. cmp ( & left . 1 ). then_with ( || left . 0. cmp ( & right . 0 )));
references . extend ( indexed . into_iter (). map ( | ( _ , _ , reference ) | reference ));
}
fn reference_query_score ( reference : & Value , query : & str ) -> i64 {
2026-06-08 20:35:49 +08:00
let query_normalized = normalize_text_for_match ( query ). to_ascii_lowercase ();
if query_normalized . is_empty () {
2026-06-07 01:10:31 +08:00
return 0 ;
}
2026-06-08 20:35:49 +08:00
let tokens = query_match_terms ( & query_normalized );
2026-06-07 01:10:31 +08:00
let source_path = reference
. get ( "sourceRootRelativePath" )
. and_then ( Value ::as_str )
. unwrap_or_default ()
. to_ascii_lowercase ();
let file_path = reference
. get ( "filePath" )
. and_then ( Value ::as_str )
. unwrap_or_default ()
. to_ascii_lowercase ();
2026-06-09 09:20:56 +08:00
let quote = reference_display_quote ( reference ). to_ascii_lowercase ();
2026-06-07 01:10:31 +08:00
let citation = reference
. get ( "citationMarkdown" )
. and_then ( Value ::as_str )
. unwrap_or_default ()
. to_ascii_lowercase ();
let mut score = 0 ;
if quote . contains ( & query_normalized ) {
score += 100 ;
}
if source_path . contains ( & query_normalized ) || file_path . contains ( & query_normalized ) {
score += 60 ;
}
for token in tokens {
2026-06-08 20:35:49 +08:00
if source_path . contains ( & token ) {
2026-06-07 01:10:31 +08:00
score += 20 ;
}
2026-06-08 20:35:49 +08:00
if file_path . contains ( & token ) {
2026-06-07 01:10:31 +08:00
score += 12 ;
}
2026-06-08 20:35:49 +08:00
if quote . contains ( & token ) {
2026-06-07 01:10:31 +08:00
score += 8 ;
}
2026-06-08 20:35:49 +08:00
if citation . contains ( & token ) {
2026-06-07 01:10:31 +08:00
score += 4 ;
}
}
score
}
fn normalize_lightrag_query_mode ( mode : Option <& str > ) -> String {
2026-06-09 09:20:56 +08:00
match mode
. map ( | value | value . trim (). to_ascii_lowercase ())
. filter ( | value | ! value . is_empty ())
. as_deref ()
{
Some ( "local" ) => "local" . into (),
Some ( "global" ) => "global" . into (),
Some ( "hybrid" ) => "hybrid" . into (),
Some ( "naive" ) => "naive" . into (),
Some ( "mix" ) => "mix" . into (),
Some ( "bypass" ) => "bypass" . into (),
Some ( "exact" | "keyword" | "full_text" | "full-text" ) => "naive" . into (),
Some ( _ ) => "mix" . into (),
2026-06-07 01:10:31 +08:00
None => "mix" . into (),
}
}
2026-06-13 22:20:01 +08:00
#[derive(Debug, Clone, PartialEq, Eq)]
struct LightRagQueryModeDecision {
mode : String ,
reason : String ,
}
fn resolve_lightrag_query_mode_for_scope (
registry : & KnowledgeRagSourceRegistry ,
source_scope : & [ String ],
requested_mode : & str ,
) -> LightRagQueryModeDecision {
if lightrag_query_scope_has_skip_kg_document ( registry , source_scope )
&& ! matches! ( requested_mode , "naive" | "bypass" )
{
return LightRagQueryModeDecision {
mode : "naive" . into (),
reason : "source_scope_skip_kg_document" . into (),
};
}
LightRagQueryModeDecision {
mode : requested_mode . to_string (),
reason : "requested_mode" . into (),
}
}
fn lightrag_query_scope_has_skip_kg_document (
registry : & KnowledgeRagSourceRegistry ,
source_scope : & [ String ],
) -> bool {
if source_scope . is_empty () {
return false ;
}
registry . entries . iter (). any ( | entry | {
! entry . stale
&& entry . deleted_at_ms . is_none ()
&& entry_matches_source_scope ( entry , source_scope )
&& registry_entry_has_skip_kg_parser_hint ( entry )
})
}
fn entry_matches_source_scope (
entry : & KnowledgeRagSourceRegistryEntry ,
source_scope : & [ String ],
) -> bool {
let source_path = entry
. source_root_relative_path
. trim_matches ( '/' )
. replace ( '\\' , "/" );
source_scope
. iter ()
. any ( | scope | source_path == * scope || source_path . starts_with ( & format! ( " {scope} /" )))
}
fn registry_entry_has_skip_kg_parser_hint ( entry : & KnowledgeRagSourceRegistryEntry ) -> bool {
entry
. parser_hint
. as_deref ()
. is_some_and ( parser_hint_contains_skip_kg )
|| parser_hint_segment_from_file_name ( & entry . light_rag_file_path )
. as_deref ()
. is_some_and ( parser_hint_contains_skip_kg )
}
fn parser_hint_contains_skip_kg ( hint : & str ) -> bool {
normalize_supported_parser_hint ( hint )
. as_deref ()
. is_some_and ( | normalized | {
normalized
. split_once ( '-' )
. map ( | ( _ , options ) | options )
. or_else ( || normalized . strip_prefix ( '-' ))
. is_some_and ( | options | options . contains ( '!' ))
})
}
fn parser_hint_segment_from_file_name ( file_name : & str ) -> Option < String > {
let mut search_start = 0 usize ;
while let Some ( offset ) = file_name . get ( search_start .. ) ? . find ( ".[" ) {
let start = search_start + offset ;
let hint_start = start + 2 ;
let Some ( end_offset ) = file_name . get ( hint_start .. ) ? . find ( ']' ) else {
break ;
};
let hint_end = hint_start + end_offset ;
let hint = file_name . get ( hint_start .. hint_end ) ? ;
if normalize_supported_parser_hint ( hint ). is_some () {
return Some ( hint . to_string ());
}
search_start = hint_end + 1 ;
}
None
}
2026-06-09 09:20:56 +08:00
fn normalize_knowledge_rag_search_mode ( mode : Option <& str > ) -> String {
match mode
. map ( | value | value . trim (). to_ascii_lowercase ())
. filter ( | value | ! value . is_empty ())
. as_deref ()
{
Some ( "exact" | "keyword" | "keywords" | "full_text" | "full-text" ) => "exact" . into (),
Some ( "local" ) => "local" . into (),
Some ( "global" ) => "global" . into (),
Some ( "hybrid" ) => "hybrid" . into (),
Some ( "naive" | "vector" ) => "naive" . into (),
Some ( "mix" | "mixed" ) => "mix" . into (),
2026-06-09 09:32:36 +08:00
Some ( _ ) => "mix" . into (),
None => "exact" . into (),
2026-06-09 09:20:56 +08:00
}
}
2026-06-13 22:20:01 +08:00
fn resolve_knowledge_rag_search_mode_for_scope (
registry : & KnowledgeRagSourceRegistry ,
source_scope : & [ String ],
requested_mode : & str ,
) -> LightRagQueryModeDecision {
if lightrag_query_scope_has_skip_kg_document ( registry , source_scope )
&& ! matches! ( requested_mode , "exact" | "naive" | "bypass" )
{
return LightRagQueryModeDecision {
mode : "naive" . into (),
reason : "source_scope_skip_kg_document" . into (),
};
}
LightRagQueryModeDecision {
mode : requested_mode . to_string (),
reason : "requested_mode" . into (),
}
}
2026-06-07 01:10:31 +08:00
pub async fn open_reference (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagOpenReferenceRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let file_path = body
. file_path
. or_else ( || {
body . reference
. as_ref ()
. and_then ( | value | value . get ( "file_path" ))
. and_then ( Value ::as_str )
. map ( ToOwned ::to_owned )
})
. ok_or_else ( || {
WebError ::bad_request_code (
"knowledge_rag_reference_file_path_required" ,
"资料库引用打开缺少 filePath" ,
)
. with_context ( & context )
}) ? ;
let reference = map_reference_plan (
& json! ({
"reference_id" : body . reference_id ,
"file_path" : file_path ,
"chunk_id" : body . chunk_id ,
}),
& registry ,
& body . root_uri ,
& root_path ,
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 01:10:31 +08:00
);
Ok ( Json ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.open_reference_result.v1" ,
"reference" : reference ,
"registry" : registry ,
})))
}
2026-06-13 22:20:01 +08:00
pub async fn section_context (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagSectionContextRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let payload = sidecar_section_context_payload ( & root_path , & registry , & body )
. map_err ( | error | error . with_context ( & context )) ? ;
Ok ( Json ( payload ))
}
2026-06-07 01:10:31 +08:00
pub async fn delete_source (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagDeleteSourceRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let source_path = body . source_path . as_deref (). map ( str ::trim );
let doc_id = body . light_rag_doc_id . as_deref (). map ( str ::trim );
let now = now_ms ();
let mut matched_doc_ids = Vec ::new ();
for entry in & mut registry . entries {
let source_matches = source_path
. filter ( | value | ! value . is_empty ())
. is_some_and ( | value | {
value == entry . source_path || value == entry . source_root_relative_path
});
let doc_matches = doc_id
. filter ( | value | ! value . is_empty ())
. is_some_and ( | value | entry . light_rag_doc_id . as_deref () == Some ( value ));
if source_matches || doc_matches {
if let Some ( doc_id ) = entry . light_rag_doc_id . clone () {
matched_doc_ids . push ( doc_id );
}
entry . stale = true ;
entry . deleted_at_ms = Some ( now );
entry . light_rag_status = Some (
if entry . light_rag_doc_id . is_some () {
"delete_submitted"
} else {
"delete_completed"
}
. into (),
);
entry . updated_at_ms = now ;
}
}
matched_doc_ids . sort ();
matched_doc_ids . dedup ();
let delete_result = if matched_doc_ids . is_empty () {
json! ({
"status" : "no_lightrag_doc" ,
"message" : "没有可删除的 LightRAG doc id; registry 已按 source 标记 stale" ,
})
} else {
lightrag_json (
reqwest ::Method ::DELETE ,
"/documents/delete_document" ,
Some ( json! ({
"doc_ids" : matched_doc_ids ,
"delete_file" : false ,
"delete_llm_cache" : false ,
})),
true ,
& context ,
)
. await ?
};
write_registry ( & root_path , & mut registry ) ? ;
Ok ( Json ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.delete_source_result.v1" ,
"provider" : "lightrag" ,
"deleteResult" : delete_result ,
"registry" : registry ,
})))
}
pub async fn prune_registry (
State ( state ) : State < AppState > ,
Extension ( context ) : Extension < RequestContext > ,
Json ( body ) : Json < KnowledgeRagPruneRegistryRequest > ,
) -> Result < Json < Value > , WebError > {
let root_path = local_folder_source ::ensure_local_workspace_read_access_with_state (
& state ,
& context ,
& body . root_uri ,
)
. map_err ( | error | error . with_context ( & context )) ? ;
let workspace_id = effective_workspace_id ( body . workspace_id . as_deref (), & body . root_uri );
let mut registry = read_registry ( & root_path , & workspace_id , & body . root_uri ) ? ;
sync_registry_with_documents ( & root_path , & mut registry , & context ). await ? ;
let before = registry . entries . len ();
registry
. entries
. retain ( | entry | ! knowledge_rag_registry_entry_prunable ( entry ));
let removed = before . saturating_sub ( registry . entries . len ());
write_registry ( & root_path , & mut registry ) ? ;
Ok ( Json ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.prune_registry_result.v1" ,
"provider" : "lightrag" ,
"removed" : removed ,
"registry" : registry ,
})))
}
fn knowledge_rag_registry_entry_prunable ( entry : & KnowledgeRagSourceRegistryEntry ) -> bool {
2026-06-07 10:35:21 +08:00
if matches! (
entry . light_rag_status . as_deref (),
Some ( "delete_submitted" | "delete_retry_required" )
) {
return false ;
}
if entry . light_rag_doc_id . is_some () && ( entry . deleted_at_ms . is_some () || entry . stale ) {
2026-06-07 01:10:31 +08:00
return false ;
}
entry . deleted_at_ms . is_some ()
|| entry . stale
|| matches! (
entry . light_rag_status . as_deref (),
Some ( "delete_completed" | "failed" )
)
}
2026-06-07 10:35:21 +08:00
fn knowledge_rag_provider_delete_confirmed ( entry : & KnowledgeRagSourceRegistryEntry ) -> bool {
entry . light_rag_doc_id . is_some ()
&& ( entry . deleted_at_ms . is_some ()
|| entry . stale
|| matches! (
entry . light_rag_status . as_deref (),
Some ( "delete_submitted" | "delete_retry_required" )
))
}
fn mark_registry_entry_delete_completed ( entry : & mut KnowledgeRagSourceRegistryEntry , now : u128 ) {
entry . light_rag_doc_id = None ;
entry . indexed_at_ms = None ;
entry . light_rag_status = Some ( "delete_completed" . into ());
entry . updated_at_ms = now ;
}
2026-06-07 01:10:31 +08:00
async fn sync_registry_with_documents (
root_path : & Path ,
registry : & mut KnowledgeRagSourceRegistry ,
context : & RequestContext ,
) -> Result < (), WebError > {
if registry . entries . is_empty () {
return Ok (());
}
let docs = lightrag_json ( reqwest ::Method ::GET , "/documents" , None , true , context ). await ? ;
let by_file_path = lightrag_documents_by_file_path ( & docs );
let now = now_ms ();
let mut changed = false ;
2026-06-07 10:35:21 +08:00
let mut retry_doc_ids = Vec ::new ();
2026-06-07 01:10:31 +08:00
for entry in & mut registry . entries {
if let Some ( doc ) = document_for_registry_entry ( & by_file_path , entry ) {
if let Some ( id ) = doc . get ( "id" ). and_then ( Value ::as_str ) {
entry . light_rag_doc_id = Some ( id . to_string ());
}
2026-06-07 10:35:21 +08:00
let delete_pending = matches! (
entry . light_rag_status . as_deref (),
Some ( "delete_submitted" | "delete_retry_required" )
) || entry . deleted_at_ms . is_some ();
if delete_pending {
if let Some ( doc_id ) = entry . light_rag_doc_id . clone () {
retry_doc_ids . push ( doc_id );
}
if entry . light_rag_status . is_none () {
entry . light_rag_status = Some ( "delete_submitted" . into ());
}
} else if let Some ( status ) = doc . get ( "status" ). and_then ( Value ::as_str ) {
entry . light_rag_status = Some ( status . to_string ());
2026-06-07 01:10:31 +08:00
}
2026-06-07 10:35:21 +08:00
if ! entry . stale
&& entry . deleted_at_ms . is_none ()
2026-06-07 01:10:31 +08:00
&& doc . get ( "status" ). and_then ( Value ::as_str ) == Some ( "processed" )
{
entry . indexed_at_ms . get_or_insert ( now );
entry . stale = ! Path ::new ( & entry . source_path ). exists ();
}
entry . updated_at_ms = now ;
changed = true ;
2026-06-07 10:35:21 +08:00
} else if knowledge_rag_provider_delete_confirmed ( entry ) {
mark_registry_entry_delete_completed ( entry , now );
2026-06-07 01:10:31 +08:00
changed = true ;
} else if entry . deleted_at_ms . is_some ()
&& entry . light_rag_doc_id . is_none ()
&& entry . light_rag_status . as_deref () != Some ( "delete_completed" )
{
entry . indexed_at_ms = None ;
entry . light_rag_status = Some ( "delete_completed" . into ());
entry . updated_at_ms = now ;
changed = true ;
} else if ! Path ::new ( & entry . source_path ). exists () {
entry . stale = true ;
entry . updated_at_ms = now ;
changed = true ;
}
}
2026-06-07 10:35:21 +08:00
let mut stale_doc_ids = sync_registry_source_state ( registry , now ) ? ;
stale_doc_ids . extend ( retry_doc_ids );
stale_doc_ids . sort ();
stale_doc_ids . dedup ();
2026-06-07 01:10:31 +08:00
if ! stale_doc_ids . is_empty () {
2026-06-07 10:35:21 +08:00
let delete_result = lightrag_json (
2026-06-07 01:10:31 +08:00
reqwest ::Method ::DELETE ,
"/documents/delete_document" ,
Some ( json! ({
"doc_ids" : stale_doc_ids ,
"delete_file" : false ,
"delete_llm_cache" : false ,
})),
true ,
context ,
)
. await ;
2026-06-07 10:35:21 +08:00
for entry in & mut registry . entries {
if entry
. light_rag_doc_id
. as_deref ()
. is_some_and ( | doc_id | stale_doc_ids . iter (). any ( | item | item == doc_id ))
{
entry . light_rag_status = Some (
if delete_result . is_err () {
"delete_retry_required"
} else {
"delete_submitted"
}
. into (),
);
entry . updated_at_ms = now ;
}
}
2026-06-07 01:10:31 +08:00
changed = true ;
}
if changed {
write_registry ( root_path , registry ) ? ;
2026-06-13 22:20:01 +08:00
refresh_document_structure_index ( root_path , registry ) ? ;
2026-06-07 01:10:31 +08:00
}
Ok (())
}
2026-06-13 22:20:01 +08:00
async fn delete_lightrag_documents_for_reingest (
doc_ids : Vec < String > ,
context : & RequestContext ,
) -> Result < Value , WebError > {
let mut doc_ids = doc_ids
. into_iter ()
. map ( | value | value . trim (). to_string ())
. filter ( | value | ! value . is_empty ())
. collect ::< Vec < _ >> ();
doc_ids . sort ();
doc_ids . dedup ();
if doc_ids . is_empty () {
return Ok ( Value ::Null );
}
let delete_result = lightrag_json (
reqwest ::Method ::DELETE ,
"/documents/delete_document" ,
Some ( json! ({
"doc_ids" : doc_ids ,
"delete_file" : false ,
"delete_llm_cache" : false ,
})),
true ,
context ,
)
. await ? ;
if delete_result . get ( "status" ). and_then ( Value ::as_str ) == Some ( "busy" ) {
return Err ( WebError ::bad_request_code (
"knowledge_rag_force_reingest_provider_busy" ,
"LightRAG 当前仍在处理旧任务,不能安全强制重建;请等待 pipeline idle 后重试" ,
)
. with_context ( context ));
}
wait_lightrag_pipeline_idle_for_reingest ( context ). await ? ;
Ok ( delete_result )
}
async fn wait_lightrag_pipeline_idle_for_reingest (
context : & RequestContext ,
) -> Result < (), WebError > {
for _ in 0 .. 180 {
let status = lightrag_json (
reqwest ::Method ::GET ,
"/documents/pipeline_status" ,
None ,
true ,
context ,
)
. await ? ;
let busy = status . get ( "busy" ). and_then ( Value ::as_bool ). unwrap_or ( false );
let scanning = status
. get ( "scanning" )
. and_then ( Value ::as_bool )
. unwrap_or ( false );
let destructive_busy = status
. get ( "destructive_busy" )
. and_then ( Value ::as_bool )
. unwrap_or ( false );
if ! busy && ! scanning && ! destructive_busy {
return Ok (());
}
tokio ::time ::sleep ( Duration ::from_secs ( 1 )). await ;
}
Err ( WebError ::bad_gateway_code (
"knowledge_rag_force_reingest_delete_timeout" ,
"LightRAG 删除旧 doc 后 pipeline 未在 180 秒内恢复 idle,已停止本次重建以避免复用旧索引" ,
)
. with_context ( context ))
}
2026-06-07 01:10:31 +08:00
fn lightrag_document_status_group_counts ( docs : & Value ) -> BTreeMap < String , usize > {
let mut counts = BTreeMap ::new ();
for ( status , items ) in docs
. get ( "statuses" )
. and_then ( Value ::as_object )
. into_iter ()
. flat_map ( | map | map . iter ())
{
counts . insert ( status . clone (), items . as_array (). map_or ( 0 , Vec ::len ));
}
counts
}
fn lightrag_document_summaries ( docs : & Value ) -> Vec < Value > {
let mut items = Vec ::new ();
for ( status_group , doc ) in docs
. get ( "statuses" )
. and_then ( Value ::as_object )
. into_iter ()
. flat_map ( | map | map . iter ())
. flat_map ( | ( status , items ) | {
items
. as_array ()
. into_iter ()
. flat_map ( move | docs | docs . iter (). map ( move | doc | ( status , doc )))
})
{
items . push ( json! ({
"id" : doc . get ( "id" ). and_then ( Value ::as_str ). unwrap_or_default (),
"filePath" : doc . get ( "file_path" ). and_then ( Value ::as_str ). unwrap_or_default (),
"status" : doc . get ( "status" ). and_then ( Value ::as_str ). unwrap_or ( status_group ),
"statusGroup" : status_group ,
"summary" : doc . get ( "summary" ). and_then ( Value ::as_str ). unwrap_or_default (),
"chunksCount" : doc . get ( "chunks_count" ). or_else ( || doc . get ( "chunks" )). cloned (). unwrap_or ( Value ::Null ),
"createdAt" : doc . get ( "created_at" ). or_else ( || doc . get ( "created" )). cloned (). unwrap_or ( Value ::Null ),
"updatedAt" : doc . get ( "updated_at" ). or_else ( || doc . get ( "updated" )). cloned (). unwrap_or ( Value ::Null ),
}));
}
items
}
2026-06-08 20:35:49 +08:00
fn lightrag_pipeline_status_summary ( value : & Value ) -> Value {
let latest_message = value
. get ( "latest_message" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
2026-06-09 22:12:09 +08:00
let history_messages = lightrag_pipeline_history_messages ( value );
let progress = parse_lightrag_chunk_progress ( latest_message ). or_else ( || {
history_messages
. iter ()
. rev ()
. find_map ( | message | parse_lightrag_chunk_progress ( message ))
});
2026-06-08 20:35:49 +08:00
json! ({
"ok" : true ,
"busy" : value . get ( "busy" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
2026-06-09 22:12:09 +08:00
"destructiveBusy" : value . get ( "destructive_busy" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
2026-06-08 20:35:49 +08:00
"scanning" : value . get ( "scanning" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
2026-06-09 22:12:09 +08:00
"scanningExclusive" : value . get ( "scanning_exclusive" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
"requestPending" : value . get ( "request_pending" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
"pendingEnqueues" : value . get ( "pending_enqueues" ). and_then ( Value ::as_u64 ). unwrap_or ( 0 ),
"pendingRequests" : value . get ( "pending_requests" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
"docs" : value . get ( "docs" ). and_then ( Value ::as_u64 ). unwrap_or ( 0 ),
"batches" : value . get ( "batchs" ). and_then ( Value ::as_u64 ). unwrap_or ( 0 ),
"currentBatch" : value . get ( "cur_batch" ). and_then ( Value ::as_u64 ). unwrap_or ( 0 ),
2026-06-08 20:35:49 +08:00
"jobName" : value . get ( "job_name" ). and_then ( Value ::as_str ). unwrap_or_default (),
2026-06-09 22:12:09 +08:00
"jobStart" : value . get ( "job_start" ). cloned (). unwrap_or ( Value ::Null ),
2026-06-08 20:35:49 +08:00
"latestMessage" : latest_message ,
2026-06-09 22:12:09 +08:00
"historyMessages" : history_messages ,
2026-06-08 20:35:49 +08:00
"cancellationRequested" : value . get ( "cancellation_requested" ). and_then ( Value ::as_bool ). unwrap_or ( false ),
"cancellationReason" : value . get ( "cancellation_reason" ). cloned (). unwrap_or ( Value ::Null ),
"progress" : progress . map ( | progress | json! ({
"current" : progress . current ,
"total" : progress . total ,
"docId" : progress . doc_id ,
})). unwrap_or ( Value ::Null ),
})
}
2026-06-09 22:12:09 +08:00
fn lightrag_pipeline_history_messages ( value : & Value ) -> Vec < String > {
value
. get ( "history_messages" )
. and_then ( Value ::as_array )
. map ( | messages | {
messages
. iter ()
. filter_map ( Value ::as_str )
. rev ()
. take ( 80 )
. map ( ToOwned ::to_owned )
. collect ::< Vec < _ >> ()
. into_iter ()
. rev ()
. collect ()
})
. unwrap_or_default ()
}
2026-06-09 09:20:56 +08:00
fn lightrag_rerank_status_summary ( value : & Value ) -> Value {
let configuration = value . get ( "configuration" ). unwrap_or ( & Value ::Null );
let queue = value . get ( "rerank_queue_status" ). unwrap_or ( & Value ::Null );
let enabled = configuration
. get ( "enable_rerank" )
. and_then ( Value ::as_bool )
. unwrap_or ( false );
let available = queue
. get ( "available" )
. and_then ( Value ::as_bool )
. unwrap_or ( false );
let binding = configuration
. get ( "rerank_binding" )
. cloned ()
. unwrap_or ( Value ::Null );
let model = configuration
. get ( "rerank_model" )
. cloned ()
. unwrap_or ( Value ::Null );
json! ({
"enabled" : enabled ,
"available" : available ,
"providerRerankEnabled" : enabled ,
"providerRerankAvailable" : available ,
"rerankModel" : model . clone (),
"binding" : binding ,
"model" : model ,
"minScore" : configuration . get ( "min_rerank_score" ). cloned (). unwrap_or ( Value ::Null ),
"status" : if enabled && available { "available" } else if enabled { "unavailable" } else { "disabled" },
"implementation" : "provider_status_only" ,
})
}
2026-06-08 20:35:49 +08:00
#[derive(Debug, Clone, PartialEq, Eq)]
struct LightRagChunkProgress {
current : u64 ,
total : u64 ,
doc_id : String ,
}
fn parse_lightrag_chunk_progress ( message : & str ) -> Option < LightRagChunkProgress > {
let rest = message . trim (). strip_prefix ( "Chunk " ) ? ;
let ( current , rest ) = rest . split_once ( " of " ) ? ;
let current = current . trim (). parse ::< u64 > (). ok () ? ;
let ( total , _ ) = rest . split_once ( ' ' ) ? ;
let total = total . trim (). parse ::< u64 > (). ok () ? ;
2026-06-09 22:12:09 +08:00
let doc_token = message . split_whitespace (). find ( | token | {
token . starts_with ( "doc-" ) && ( token . contains ( "-chunk-" ) || token . contains ( "-mm-" ))
}) ? ;
2026-06-08 20:35:49 +08:00
let doc_id = doc_token
. rsplit_once ( "-chunk-" )
2026-06-09 22:12:09 +08:00
. or_else ( || doc_token . rsplit_once ( "-mm-" ))
2026-06-08 20:35:49 +08:00
. map ( | ( doc_id , _ ) | doc_id )
. unwrap_or ( doc_token )
2026-06-09 22:12:09 +08:00
. trim_end_matches ( | ch : char | ! ch . is_ascii_alphanumeric ())
2026-06-08 20:35:49 +08:00
. to_string ();
if total == 0 || current == 0 || doc_id . is_empty () {
return None ;
}
Some ( LightRagChunkProgress {
current : current . min ( total ),
total ,
doc_id ,
})
}
2026-06-07 01:10:31 +08:00
fn lightrag_documents_by_file_path ( docs : & Value ) -> BTreeMap < String , Value > {
let mut by_file_path = BTreeMap ::< String , Value > ::new ();
for doc in docs
. get ( "statuses" )
. and_then ( Value ::as_object )
. into_iter ()
. flat_map ( | map | map . values ())
. filter_map ( Value ::as_array )
. flat_map ( | items | items . iter ())
{
if let Some ( file_path ) = doc . get ( "file_path" ). and_then ( Value ::as_str ) {
by_file_path . insert ( file_path . to_string (), doc . clone ());
}
}
by_file_path
}
fn sync_registry_source_state (
registry : & mut KnowledgeRagSourceRegistry ,
now : u128 ,
) -> Result < Vec < String > , WebError > {
let mut stale_doc_ids = Vec ::new ();
for entry in & mut registry . entries {
if entry . deleted_at_ms . is_some () {
continue ;
}
let Some ( doc_id ) = entry . light_rag_doc_id . clone () else {
continue ;
};
let source_path = Path ::new ( & entry . source_path );
if ! source_path . exists () {
entry . stale = true ;
entry . deleted_at_ms = Some ( now );
entry . indexed_at_ms = None ;
2026-06-07 10:35:21 +08:00
entry . light_rag_status = Some ( "delete_submitted" . into ());
2026-06-07 01:10:31 +08:00
entry . updated_at_ms = now ;
stale_doc_ids . push ( doc_id );
continue ;
}
let current_hash = source_hash ( source_path ) ? ;
if current_hash != entry . source_hash {
entry . stale = true ;
entry . source_hash = current_hash ;
entry . indexed_at_ms = None ;
2026-06-07 10:35:21 +08:00
entry . light_rag_status = Some ( "delete_submitted" . into ());
2026-06-07 01:10:31 +08:00
entry . updated_at_ms = now ;
stale_doc_ids . push ( doc_id );
}
}
stale_doc_ids . sort ();
stale_doc_ids . dedup ();
Ok ( stale_doc_ids )
}
async fn lightrag_json (
method : reqwest ::Method ,
path : & str ,
body : Option < Value > ,
use_api_key : bool ,
context : & RequestContext ,
) -> Result < Value , WebError > {
let endpoint = lightrag_endpoint ();
let url = format! ( " {}{} " , endpoint . trim_end_matches ( '/' ), path );
let client = reqwest ::Client ::builder ()
. timeout ( Duration ::from_secs ( 180 ))
. build ()
. map_err ( | error | {
WebError ::internal ( format! ( "LightRAG HTTP client 初始化失败: {error} " ))
. with_context ( context )
}) ? ;
2026-06-09 18:40:48 +08:00
let api_key = use_api_key . then ( lightrag_api_key ). flatten ();
for attempt in 0 .. 2 {
let mut request = client . request ( method . clone (), & url );
if let Some ( api_key ) = api_key . as_deref () {
2026-06-07 01:10:31 +08:00
request = request . header ( "X-API-Key" , api_key );
}
2026-06-09 18:40:48 +08:00
if let Some ( body ) = body . as_ref () {
request = request . json ( body );
}
let response = request . send (). await . map_err ( | error | {
WebError ::bad_gateway_code (
"knowledge_rag_lightrag_unreachable" ,
format! ( "无法访问 LightRAG provider: {error} " ),
)
. with_context ( context )
}) ? ;
let status = response . status ();
let text = response . text (). await . unwrap_or_default ();
if ! status . is_success () {
if attempt == 0 && should_retry_lightrag_provider_error ( status , & text ) {
tokio ::time ::sleep ( Duration ::from_millis ( 700 )). await ;
continue ;
}
return Err ( WebError ::bad_gateway_code (
"knowledge_rag_lightrag_error" ,
format! ( "LightRAG provider 返回 HTTP {status} : {text} " ),
)
. with_context ( context ));
}
return serde_json ::from_str ( & text ). map_err ( | error | {
WebError ::bad_gateway_code (
"knowledge_rag_lightrag_json_invalid" ,
format! ( "LightRAG provider 返回非 JSON 响应: {error} " ),
)
. with_context ( context )
});
2026-06-07 01:10:31 +08:00
}
2026-06-09 18:40:48 +08:00
unreachable! ( "lightrag retry loop returns on every branch" )
}
fn should_retry_lightrag_provider_error ( status : StatusCode , body : & str ) -> bool {
status . is_server_error ()
&& ( body . contains ( "RetryError" )
|| body . contains ( "InvalidResponseError" )
|| body . contains ( "Received empty content" ))
2026-06-07 01:10:31 +08:00
}
fn mapped_references (
raw : & Value ,
registry : & KnowledgeRagSourceRegistry ,
root_uri : & str ,
root_path : & Path ,
2026-06-08 20:35:49 +08:00
query : Option <& str > ,
2026-06-07 01:10:31 +08:00
) -> Vec < Value > {
reference_array ( raw )
. into_iter ()
2026-06-08 20:35:49 +08:00
. flat_map ( | reference | {
2026-06-07 01:10:31 +08:00
let enriched = enrich_reference_with_chunks ( raw , & reference );
2026-06-08 20:35:49 +08:00
expand_reference_by_chunks ( & enriched )
. into_iter ()
. map ( | candidate | {
map_reference_plan ( & candidate , registry , root_uri , root_path , query )
})
. collect ::< Vec < _ >> ()
2026-06-07 01:10:31 +08:00
})
. filter ( | reference | {
! reference
. get ( "stale" )
. and_then ( Value ::as_bool )
. unwrap_or ( false )
&& ! reference
. get ( "deleted" )
. and_then ( Value ::as_bool )
. unwrap_or ( false )
2026-06-07 10:35:21 +08:00
&& ! reference
. get ( "unmapped" )
. and_then ( Value ::as_bool )
. unwrap_or ( false )
2026-06-07 01:10:31 +08:00
})
. collect ()
}
2026-06-08 20:35:49 +08:00
fn expand_reference_by_chunks ( reference : & Value ) -> Vec < Value > {
let Some ( chunks ) = reference . get ( "chunks" ). and_then ( Value ::as_array ) else {
return vec! [ reference . clone ()];
};
if chunks . is_empty () {
return vec! [ reference . clone ()];
}
chunks
. iter ()
. enumerate ()
. map ( | ( index , chunk ) | {
let mut candidate = reference . clone ();
if let Some ( map ) = candidate . as_object_mut () {
map . insert ( "chunks" . into (), Value ::Array ( vec! [ chunk . clone ()]));
if let Some ( chunk_id ) = chunk . get ( "chunk_id" ). cloned () {
map . insert ( "chunk_id" . into (), chunk_id );
}
map . insert ( "chunkIndex" . into (), json! ( index ));
}
candidate
})
. collect ()
}
2026-06-07 01:10:31 +08:00
fn reference_array ( raw : & Value ) -> Vec < Value > {
if let Some ( references ) = raw
. get ( "references" )
. or_else ( || raw . get ( "data" ). and_then ( | value | value . get ( "references" )))
. and_then ( Value ::as_array )
{
return references . clone ();
}
let mut by_key = BTreeMap ::< String , Value > ::new ();
for chunk in chunk_array ( raw ) {
let file_path = chunk . get ( "file_path" ). and_then ( Value ::as_str ). unwrap_or ( "" );
if file_path . is_empty () {
continue ;
}
let reference_id = chunk
. get ( "reference_id" )
. and_then ( Value ::as_str )
. unwrap_or ( file_path );
let key = format! ( " {file_path} \n {reference_id} " );
by_key . entry ( key ). or_insert_with ( || {
json! ({
"file_path" : file_path ,
"reference_id" : reference_id ,
"chunks" : [ chunk . clone ()],
})
});
}
by_key . into_values (). collect ()
}
fn chunk_array ( raw : & Value ) -> Vec <& Value > {
raw . get ( "chunks" )
. or_else ( || raw . get ( "data" ). and_then ( | value | value . get ( "chunks" )))
. and_then ( Value ::as_array )
. map ( | items | items . iter (). collect ())
. unwrap_or_default ()
}
fn enrich_reference_with_chunks ( raw : & Value , reference : & Value ) -> Value {
let reference_id = reference . get ( "reference_id" ). and_then ( Value ::as_str );
let file_path = reference . get ( "file_path" ). and_then ( Value ::as_str );
let chunks = chunk_array ( raw )
. into_iter ()
. filter ( | chunk | {
let chunk_reference_id = chunk . get ( "reference_id" ). and_then ( Value ::as_str );
let chunk_file_path = chunk . get ( "file_path" ). and_then ( Value ::as_str );
reference_id
. zip ( chunk_reference_id )
. is_some_and ( | ( left , right ) | left == right )
|| file_path
. zip ( chunk_file_path )
. is_some_and ( | ( left , right ) | left == right )
})
. cloned ()
. collect ::< Vec < _ >> ();
if chunks . is_empty () {
return reference . clone ();
}
let mut enriched = reference . clone ();
if let Some ( map ) = enriched . as_object_mut () {
map . insert ( "chunks" . into (), Value ::Array ( chunks ));
}
enriched
}
fn map_reference_plan (
reference : & Value ,
registry : & KnowledgeRagSourceRegistry ,
root_uri : & str ,
root_path : & Path ,
2026-06-08 20:35:49 +08:00
query : Option <& str > ,
2026-06-07 01:10:31 +08:00
) -> Value {
let file_path = reference
. get ( "file_path" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let doc_id = reference . get ( "doc_id" ). and_then ( Value ::as_str );
let entry = registry . entries . iter (). find ( | entry | {
lightrag_file_path_matches ( entry , file_path )
|| doc_id . is_some_and ( | doc_id | entry . light_rag_doc_id . as_deref () == Some ( doc_id ))
});
let locator_degraded = entry . is_none_or ( | entry | entry . stale || entry . deleted_at_ms . is_some ());
let source_path = entry . map ( | entry | entry . source_path . clone ());
let source_root_relative_path = entry . map ( | entry | entry . source_root_relative_path . clone ());
let primary_chunk = reference
. get ( "chunks" )
. and_then ( Value ::as_array )
. and_then ( | chunks | chunks . first ());
let chunk_id = reference
. get ( "chunk_id" )
. or_else ( || primary_chunk . and_then ( | chunk | chunk . get ( "chunk_id" )))
. cloned ()
. unwrap_or ( Value ::Null );
2026-06-09 18:40:48 +08:00
let source_chunk_id = reference
. get ( "source_chunk_id" )
. or_else ( || primary_chunk . and_then ( | chunk | chunk . get ( "source_chunk_id" )))
. cloned ()
. or_else ( || {
chunk_id
. as_str ()
. and_then ( | value | value . split_once ( "#match-" ). map ( | ( base , _ ) | json! ( base )))
})
. unwrap_or ( Value ::Null );
2026-06-08 20:35:49 +08:00
let occurrence_index = reference
. get ( "occurrence_index" )
. or_else ( || primary_chunk . and_then ( | chunk | chunk . get ( "occurrence_index" )))
. cloned ()
. unwrap_or ( Value ::Null );
2026-06-09 18:40:48 +08:00
let chunk_sidecar =
lightrag_reference_sidecar ( reference , primary_chunk , & source_chunk_id , & chunk_id );
2026-06-08 20:35:49 +08:00
let ( quote , quote_source ) = primary_chunk
2026-06-07 01:10:31 +08:00
. and_then ( | chunk | chunk . get ( "content" ))
. and_then ( Value ::as_str )
2026-06-08 20:35:49 +08:00
. map ( | value | ( value . to_owned (), "chunk" ))
. or_else ( || {
2026-06-09 18:40:48 +08:00
lightrag_chunk_content_for_reference_ids ( & source_chunk_id , & chunk_id )
2026-06-08 20:35:49 +08:00
. map ( | value | ( value , "kv_store" ))
})
. map ( | ( value , source ) | {
let quote = query
. and_then ( | query | query_centered_quote ( & value , query , 500 ))
. unwrap_or_else ( || value . chars (). take ( 500 ). collect ::< String > ());
( quote , source )
})
. map_or (( None , "missing" ), | ( value , source ) | ( Some ( value ), source ));
2026-06-09 09:20:56 +08:00
let locator_sidecar_block = entry . and_then ( | entry | {
2026-06-09 18:40:48 +08:00
find_lightrag_sidecar_block ( entry , query , & occurrence_index , chunk_sidecar . as_ref ())
2026-06-09 09:20:56 +08:00
});
let locator_sidecar_text = locator_sidecar_block
. as_ref ()
. and_then ( | block | block . get ( "content" ))
. and_then ( Value ::as_str )
. map ( ToOwned ::to_owned );
let heading_path = heading_path_for_sidecar_block ( locator_sidecar_block . as_ref ());
let text_bundle = citation_text_bundle (
quote . as_deref (),
query ,
locator_sidecar_text . as_deref (),
quote_source ,
);
2026-06-08 20:35:49 +08:00
let content_diagnostics = quote_content_diagnostics ( quote . as_deref (), quote_source );
2026-06-07 01:10:31 +08:00
let locator = entry . and_then ( | entry | {
2026-06-09 18:40:48 +08:00
lightrag_locator_for_reference (
root_path ,
root_uri ,
entry ,
& chunk_id ,
query ,
Some ( & text_bundle . locator_evidence_text ),
& occurrence_index ,
chunk_sidecar . as_ref (),
)
2026-06-07 01:10:31 +08:00
});
2026-06-08 20:35:49 +08:00
let locator_precision = locator
. as_ref ()
. map ( locator_precision_for_locator )
. unwrap_or ( "file" );
2026-06-13 22:20:01 +08:00
let locator_degraded = locator_degraded || matches! ( locator_precision , "file" );
2026-06-07 01:10:31 +08:00
let fallback_citation_url =
entry . and_then ( | entry | fallback_resource_citation_url ( root_path , root_uri , entry ));
let citation_url = locator
. as_ref ()
. map ( citation_url_for_locator )
. or ( fallback_citation_url );
let citation_markdown = locator
. as_ref ()
. map ( citation_markdown_for_locator )
. or_else ( || {
citation_url . as_ref (). map ( | url | {
format! (
"[来源定位降级: {} ]( {} )" ,
markdown_link_label_escape ( file_path ),
url . replace ( ')' , "%29" )
)
})
});
2026-06-09 09:20:56 +08:00
let citation_id = short_citation_id (
entry
. map ( | entry | entry . source_id . as_str ())
. unwrap_or_default (),
chunk_id . as_str (). unwrap_or_default (),
locator
. as_ref ()
. and_then ( | locator | locator . block_id . as_deref ())
. unwrap_or_default (),
occurrence_index . as_u64 (). unwrap_or ( 0 ),
& text_bundle . normalized_fingerprint ,
);
2026-06-07 01:10:31 +08:00
json! ({
"schema" : REFERENCE_SCHEMA ,
"provider" : "lightrag" ,
2026-06-09 09:20:56 +08:00
"citationId" : citation_id . clone (),
"citationLabel" : format ! ( "[{}]" , citation_id ),
2026-06-08 20:35:49 +08:00
"matchSource" : if occurrence_index . is_null () { "lightrag_reference" } else { "lightrag_search" },
2026-06-07 01:10:31 +08:00
"reference" : reference ,
"filePath" : file_path ,
"chunkId" : chunk_id ,
2026-06-09 18:40:48 +08:00
"sourceChunkId" : source_chunk_id ,
2026-06-08 20:35:49 +08:00
"occurrenceIndex" : occurrence_index ,
2026-06-09 09:20:56 +08:00
"rawQuote" : text_bundle . raw_quote . clone (),
"displayQuote" : text_bundle . display_quote . clone (),
"locatorEvidenceText" : text_bundle . locator_evidence_text . clone (),
"searchQuery" : text_bundle . search_query . clone (),
"normalizedFingerprint" : text_bundle . normalized_fingerprint . clone (),
"headingPath" : heading_path ,
"quote" : text_bundle . display_quote . clone (),
2026-06-08 20:35:49 +08:00
"quoteSource" : quote_source ,
"contentDiagnostics" : content_diagnostics ,
2026-06-09 09:20:56 +08:00
"citationDiagnostics" : {
"quoteSource" : quote_source ,
"rawReferenceMapped" : entry . is_some (),
"sidecarBlockMapped" : locator_sidecar_text . is_some (),
"displayCleaned" : text_bundle . display_cleaned ,
"locatorTextSource" : text_bundle . locator_text_source ,
},
2026-06-07 01:10:31 +08:00
"locator" : locator ,
2026-06-08 20:35:49 +08:00
"locatorPrecision" : locator_precision ,
2026-06-07 01:10:31 +08:00
"citationUrl" : citation_url ,
"sourceId" : entry . map ( | entry | entry . source_id . clone ()),
"sourcePath" : source_path ,
"sourceRootRelativePath" : source_root_relative_path ,
2026-06-07 10:35:21 +08:00
"unmapped" : entry . is_none (),
2026-06-07 01:10:31 +08:00
"stale" : entry . is_some_and ( | entry | entry . stale ),
"deleted" : entry . is_some_and ( | entry | entry . deleted_at_ms . is_some ()),
"locatorDegraded" : locator_degraded ,
"openAction" : {
"kind" : "mnote.local_resource.open" ,
"params" : {
"rootUri" : root_uri ,
"path" : entry . map ( | entry | entry . source_root_relative_path . clone ()),
"provider" : "lightrag" ,
"filePath" : file_path ,
"chunkId" : chunk_id ,
2026-06-08 20:35:49 +08:00
"occurrenceIndex" : occurrence_index ,
2026-06-09 09:20:56 +08:00
"searchQuery" : query . unwrap_or_default (),
"evidenceText" : text_bundle . locator_evidence_text . clone (),
"displayQuote" : text_bundle . display_quote . clone (),
2026-06-07 01:10:31 +08:00
},
},
"citationMarkdown" : citation_markdown . unwrap_or_else ( || format! ( "[来源定位降级: {} ](#)" , file_path )),
})
}
fn fallback_resource_citation_url (
root_path : & Path ,
root_uri : & str ,
entry : & KnowledgeRagSourceRegistryEntry ,
) -> Option < String > {
if entry . stale || entry . deleted_at_ms . is_some () {
return None ;
}
let owner_document_id =
fallback_owner_document_id ( root_path , & entry . source_root_relative_path ) ? ;
let mut url = format! ( "/documents/ {owner_document_id} " );
append_query_param ( & mut url , "sourceKind" , "local_folder" );
append_query_param ( & mut url , "rootUri" , root_uri );
append_query_param (
& mut url ,
"resourceTab" ,
& format! (
"resource:file: {} : {} " ,
root_uri , entry . source_root_relative_path
),
);
append_query_param ( & mut url , "resourcePath" , & entry . source_root_relative_path );
Some ( url )
}
fn fallback_owner_document_id ( root_path : & Path , relative_path : & str ) -> Option < String > {
if is_markdown_like_path ( relative_path ) {
return Some ( local_markdown_document_id ( relative_path ));
}
let source_path = Path ::new ( relative_path );
if let Some ( parent ) = source_path
. parent ()
. filter ( | parent | ! parent . as_os_str (). is_empty ())
{
let parent_path = root_path . join ( parent );
if let Ok ( entries ) = fs ::read_dir ( parent_path ) {
let mut candidates = entries
. filter_map ( Result ::ok )
. filter_map ( | entry | {
let path = entry . path ();
if ! path . is_file () || ! is_markdown_like_path ( & path . to_string_lossy ()) {
return None ;
}
let file_name = path . file_name () ? . to_string_lossy ();
if file_name . starts_with ( '.' ) {
return None ;
}
let rel = parent . join ( file_name . as_ref ());
Some ( rel . to_string_lossy (). replace ( '\\' , "/" ))
})
. collect ::< Vec < _ >> ();
candidates . sort ();
if let Some ( candidate ) = candidates . first () {
return Some ( local_markdown_document_id ( candidate ));
}
}
}
for root_candidate in [ "README.md" , "index.md" ] {
if root_path . join ( root_candidate ). is_file () {
return Some ( local_markdown_document_id ( root_candidate ));
}
}
None
}
fn is_markdown_like_path ( value : & str ) -> bool {
Path ::new ( value )
. extension ()
. and_then ( | extension | extension . to_str ())
. is_some_and ( | extension | {
extension . eq_ignore_ascii_case ( "md" ) || extension . eq_ignore_ascii_case ( "markdown" )
})
}
fn local_markdown_document_id ( relative_path : & str ) -> String {
format! (
"local-md: {} " ,
encode_local_document_id_segment ( relative_path )
)
}
fn append_query_param ( url : & mut String , key : & str , value : & str ) {
let value = value . trim ();
if value . is_empty () {
return ;
}
let separator = if url . contains ( '?' ) { '&' } else { '?' };
url . push ( separator );
url . push_str ( & encode_query_component ( key ));
url . push ( '=' );
url . push_str ( & encode_query_component ( value ));
}
fn encode_query_component ( value : & str ) -> String {
let mut encoded = String ::with_capacity ( value . len ());
for byte in value . as_bytes () {
if byte . is_ascii_alphanumeric () || matches! ( byte , b '-' | b '_' | b '.' | b '~' ) {
encoded . push ( * byte as char );
} else {
encoded . push_str ( & format! ( "% {byte:02X} " ));
}
}
encoded
}
fn encode_local_document_id_segment ( relative_path : & str ) -> String {
relative_path
. replace ( '\\' , "/" )
. bytes ()
. flat_map ( | byte | match byte {
b 'A' ..= b 'Z' | b 'a' ..= b 'z' | b '0' ..= b '9' | b '-' | b '_' | b '.' => {
vec! [ byte as char ]
}
b '/' => "~2F" . chars (). collect ::< Vec < _ >> (),
_ => format! ( "~ {byte:02X} " ). chars (). collect ::< Vec < _ >> (),
})
. collect ()
}
fn markdown_link_label_escape ( value : & str ) -> String {
value . replace ( '[' , " \\ [" ). replace ( ']' , " \\ ]" )
}
2026-06-08 20:35:49 +08:00
fn quote_content_diagnostics ( quote : Option <& str > , quote_source : & str ) -> Value {
let quote = quote . unwrap_or_default ();
let non_empty_lines = quote
. lines ()
. map ( str ::trim )
. filter ( | line | ! line . is_empty ())
. collect ::< Vec < _ >> ();
let non_heading_lines = non_empty_lines
. iter ()
. copied ()
. filter ( | line | ! line . starts_with ( '#' ))
. collect ::< Vec < _ >> ();
let quote_only_image_placeholder = ! non_heading_lines . is_empty ()
&& non_heading_lines
. iter ()
. all ( | line | is_markdown_image_placeholder_line ( line ));
let ocr_text_exposed = ! quote . trim (). is_empty () && ! quote_only_image_placeholder ;
json! ({
"quoteEmpty" : quote . trim (). is_empty (),
"quoteOnlyImagePlaceholder" : quote_only_image_placeholder ,
"ocrTextExposed" : ocr_text_exposed ,
"quoteSource" : quote_source ,
})
}
2026-06-09 09:20:56 +08:00
fn citation_text_bundle (
raw_quote : Option <& str > ,
search_query : Option <& str > ,
locator_source_text : Option <& str > ,
quote_source : & 'static str ,
) -> CitationTextBundle {
let raw_quote = raw_quote . unwrap_or_default (). trim (). to_string ();
let search_query = search_query . unwrap_or_default (). trim (). to_string ();
let locator_raw = locator_source_text
. filter ( | value | ! value . trim (). is_empty ())
. unwrap_or ( & raw_quote );
let locator_window = if ! search_query . is_empty () {
query_centered_quote ( locator_raw , & search_query , 700 )
. or_else ( || query_centered_quote ( & raw_quote , & search_query , 700 ))
. unwrap_or_else ( || locator_raw . chars (). take ( 700 ). collect ::< String > ())
} else {
locator_raw . chars (). take ( 700 ). collect ::< String > ()
};
2026-06-09 18:40:48 +08:00
let display_source = locator_source_text
. filter ( | value | ! value . trim (). is_empty ())
. unwrap_or ( & raw_quote );
let display_quote = clean_lightrag_text_for_display ( display_source );
2026-06-09 09:20:56 +08:00
let locator_evidence_text = query_centered_clean_window (
& clean_lightrag_text_for_locator ( & locator_window ),
& search_query ,
700 ,
);
CitationTextBundle {
raw_quote : raw_quote . clone (),
display_cleaned : display_quote != normalize_text_for_match ( & raw_quote ),
normalized_fingerprint : normalized_quote_fingerprint ( & display_quote ),
display_quote ,
locator_evidence_text ,
search_query ,
locator_text_source : if locator_source_text . is_some () {
"sidecar_block"
} else {
quote_source
},
}
2026-06-08 20:35:49 +08:00
}
2026-06-09 09:20:56 +08:00
fn clean_lightrag_text_for_display ( value : & str ) -> String {
let without_tags = strip_lightrag_markup ( value );
normalize_latex_plain_text ( & without_tags )
}
fn clean_lightrag_text_for_locator ( value : & str ) -> String {
clean_lightrag_text_for_display ( value )
}
fn strip_lightrag_markup ( value : & str ) -> String {
let mut output = String ::new ();
let mut chars = value . chars (). peekable ();
while let Some ( ch ) = chars . next () {
if ch != '<' {
output . push ( ch );
continue ;
}
let mut tag = String ::new ();
let mut closed = false ;
for next in chars . by_ref () {
if next == '>' {
closed = true ;
break ;
}
tag . push ( next );
if tag . chars (). count () > 240 {
break ;
}
}
output . push ( ' ' );
if ! closed {
break ;
}
2026-06-08 20:35:49 +08:00
}
2026-06-09 09:20:56 +08:00
output
}
fn normalize_latex_plain_text ( value : & str ) -> String {
let mut output = String ::new ();
let mut chars = value . chars (). peekable ();
while let Some ( ch ) = chars . next () {
if ch == '\\' {
let mut command = String ::new ();
while let Some ( next ) = chars . peek (). copied () {
if next . is_ascii_alphabetic () {
command . push ( next );
chars . next ();
} else {
break ;
}
}
match command . as_str () {
"gt" => output . push ( '>' ),
"lt" => output . push ( '<' ),
"sim" => output . push ( '∼ ' ),
"circ" => output . push ( '∘' ),
"alpha" => output . push ( 'α ' ),
"beta" => output . push ( 'β' ),
"gamma" => output . push ( 'γ ' ),
"delta" => output . push ( 'δ' ),
"Delta" => output . push ( '△' ),
"left" | "right" | "mathrm" | "text" | "operatorname" => output . push ( ' ' ),
"" => output . push ( ' ' ),
_ => output . push ( ' ' ),
}
continue ;
}
match ch {
'{' | '}' | '_' | '^' => output . push ( ' ' ),
_ => output . push ( ch ),
}
}
normalize_text_for_match ( & output )
}
fn normalized_quote_fingerprint ( value : & str ) -> String {
normalize_text_for_match ( value )
. chars ()
. filter ( | ch | ch . is_alphanumeric () || is_cjk_char ( * ch ))
. take ( 160 )
. collect ::< String > ()
2026-06-08 20:35:49 +08:00
. to_ascii_lowercase ()
2026-06-09 09:20:56 +08:00
}
fn query_centered_clean_window ( value : & str , query : & str , max_chars : usize ) -> String {
let text = normalize_text_for_match ( value );
let query = normalize_text_for_match ( query );
if text . is_empty () || query . is_empty () {
return text . chars (). take ( max_chars ). collect ();
}
let lower = text . to_ascii_lowercase ();
let query_lower = query . to_ascii_lowercase ();
let start = lower . find ( & query_lower ). unwrap_or ( 0 );
text . get ( start .. )
. unwrap_or ( & text )
. chars ()
. take ( max_chars )
. collect ()
}
fn reference_display_quote ( reference : & Value ) -> & str {
reference
. get ( "displayQuote" )
. and_then ( Value ::as_str )
. or_else ( || reference . get ( "quote" ). and_then ( Value ::as_str ))
. unwrap_or_default ()
}
fn reference_locator_evidence_text ( reference : & Value ) -> & str {
reference
. get ( "locatorEvidenceText" )
. and_then ( Value ::as_str )
. or_else ( || reference . get ( "quote" ). and_then ( Value ::as_str ))
. unwrap_or_default ()
2026-06-08 20:35:49 +08:00
}
fn query_centered_quote ( content : & str , query : & str , max_chars : usize ) -> Option < String > {
let query_normalized = normalize_text_for_match ( query ). to_ascii_lowercase ();
if query_normalized . is_empty () {
return None ;
}
let lines = content . lines (). collect ::< Vec < _ >> ();
for ( index , line ) in lines . iter (). enumerate () {
let line_normalized = normalize_text_for_match ( line ). to_ascii_lowercase ();
if ! line_normalized . contains ( & query_normalized ) {
continue ;
}
let mut quote = String ::new ();
if index > 0 {
let previous = lines [ index - 1 ]. trim ();
if previous . starts_with ( '#' ) {
quote . push_str ( previous );
quote . push ( '\n' );
}
}
for line in lines . iter (). skip ( index ) {
let line = line . trim ();
if line . is_empty () {
if ! quote . is_empty () {
quote . push ( '\n' );
}
continue ;
}
let next_len = quote . chars (). count () + line . chars (). count () + 1 ;
if next_len > max_chars && ! quote . trim (). is_empty () {
break ;
}
if ! quote . is_empty () && ! quote . ends_with ( '\n' ) {
quote . push ( '\n' );
}
quote . push_str ( line );
if quote . chars (). count () >= max_chars {
break ;
}
}
let quote = quote . trim ();
if ! quote . is_empty () {
return Some ( quote . chars (). take ( max_chars ). collect ());
}
}
let content_lower = content . to_ascii_lowercase ();
let byte_index = content_lower . find ( & query_normalized ) ? ;
Some ( char_window_around_byte ( content , byte_index , max_chars ))
}
fn char_window_around_byte ( content : & str , byte_index : usize , max_chars : usize ) -> String {
let target_char_index = content [ .. byte_index ]. chars (). count ();
let before = max_chars / 3 ;
let start = target_char_index . saturating_sub ( before );
content
. chars ()
. skip ( start )
. take ( max_chars )
. collect ::< String > ()
. trim ()
. to_string ()
}
fn is_markdown_image_placeholder_line ( line : & str ) -> bool {
line . starts_with ( " && line . ends_with ( ')' )
}
2026-06-07 01:10:31 +08:00
fn lightrag_locator_for_reference (
root_path : & Path ,
root_uri : & str ,
entry : & KnowledgeRagSourceRegistryEntry ,
chunk_id : & Value ,
2026-06-08 20:35:49 +08:00
query : Option <& str > ,
2026-06-09 09:20:56 +08:00
locator_evidence_text : Option <& str > ,
2026-06-09 18:40:48 +08:00
occurrence_index : & Value ,
chunk_sidecar : Option <& Value > ,
2026-06-07 01:10:31 +08:00
) -> Option < EvidenceLocator > {
2026-06-09 18:40:48 +08:00
let block = find_lightrag_sidecar_block ( entry , query , occurrence_index , chunk_sidecar ) ? ;
2026-06-09 09:20:56 +08:00
let positions = block
2026-06-08 20:35:49 +08:00
. get ( "positions" )
. and_then ( Value ::as_array )
2026-06-09 09:20:56 +08:00
. map ( | positions | parse_lightrag_locator_positions ( positions . as_slice ()))
. unwrap_or_default ();
2026-06-07 01:10:31 +08:00
let resource_path = entry . source_root_relative_path . clone ();
2026-06-09 09:20:56 +08:00
let block_content = block
. get ( "content" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let evidence_text = locator_evidence_text
. filter ( | value | ! value . trim (). is_empty ())
. map ( ToOwned ::to_owned )
. or_else ( || {
query
. and_then ( | query | query_centered_quote ( block_content , query , 700 ))
. map ( | value | clean_lightrag_text_for_locator ( & value ))
})
2026-06-09 18:40:48 +08:00
. unwrap_or_else ( || clean_lightrag_text_for_locator ( block_content ));
2026-06-09 09:20:56 +08:00
let mut open_params = json! ({
"rootUri" : root_uri ,
"resourcePath" : resource_path ,
"provider" : "lightrag" ,
"chunkId" : chunk_id ,
"searchQuery" : query . unwrap_or_default (),
"query" : evidence_text ,
"evidenceText" : evidence_text ,
});
if let Some ( map ) = open_params . as_object_mut () {
if let Some ( paragraph_ordinal ) = positions . paragraph_ordinal {
map . insert ( "paragraphOrdinal" . into (), json! ( paragraph_ordinal ));
}
if let Some ( para_id_start ) = positions . para_id_start . as_deref () {
map . insert ( "paraIdStart" . into (), json! ( para_id_start ));
}
if let Some ( para_id_end ) = positions . para_id_end . as_deref () {
map . insert ( "paraIdEnd" . into (), json! ( para_id_end ));
}
if let Some ( text_fingerprint ) = positions . text_fingerprint . as_deref () {
map . insert ( "textFingerprint" . into (), json! ( text_fingerprint ));
}
}
2026-06-07 01:10:31 +08:00
let mut locator = EvidenceLocator ::new (
root_uri ,
2026-06-08 20:35:49 +08:00
fallback_owner_document_id ( root_path , & resource_path ). unwrap_or_default (),
2026-06-07 01:10:31 +08:00
& resource_path ,
2026-06-08 20:35:49 +08:00
evidence_resource_kind_for_path ( & resource_path ),
2026-06-07 01:10:31 +08:00
EvidenceOpenAction {
action_type : "mnote.open_resource_locator" . into (),
url : "/" . into (),
2026-06-09 09:20:56 +08:00
params : open_params ,
2026-06-07 01:10:31 +08:00
},
);
locator . resource_path = Some ( entry . source_root_relative_path . clone ());
2026-06-09 09:20:56 +08:00
if let Some ( position ) = positions . bbox {
2026-06-08 20:35:49 +08:00
locator . page = Some ( position . page );
locator . bbox = Some ( position . bbox );
}
2026-06-07 01:10:31 +08:00
locator . block_id = block
. get ( "blockid" )
. and_then ( Value ::as_str )
. map ( ToOwned ::to_owned )
. or_else ( || chunk_id . as_str (). map ( ToOwned ::to_owned ));
locator . source_map_path =
sidecar_blocks_path ( entry ). and_then ( | path | root_relative_path ( root_path , & path ). ok ());
Some ( locator )
}
2026-06-08 20:35:49 +08:00
fn evidence_resource_kind_for_path ( path : & str ) -> EvidenceResourceKind {
match Path ::new ( path )
. extension ()
. and_then ( | value | value . to_str ())
. map ( | value | value . to_ascii_lowercase ())
. as_deref ()
{
Some ( "pdf" ) => EvidenceResourceKind ::Pdf ,
Some ( "png" | "jpg" | "jpeg" | "gif" | "webp" | "bmp" | "svg" ) => {
EvidenceResourceKind ::Image
}
Some ( "md" | "markdown" | "txt" ) => EvidenceResourceKind ::Markdown ,
Some ( "doc" | "docx" | "xls" | "xlsx" | "ppt" | "pptx" | "csv" ) => {
EvidenceResourceKind ::Office
}
_ => EvidenceResourceKind ::RawFile ,
}
}
2026-06-07 01:10:31 +08:00
#[derive(Debug, Clone)]
2026-06-09 09:20:56 +08:00
struct LightRagBboxPosition {
2026-06-07 01:10:31 +08:00
page : u32 ,
bbox : EvidenceBBox ,
}
2026-06-09 09:20:56 +08:00
#[derive(Debug, Clone, Default)]
struct LightRagLocatorPositions {
bbox : Option < LightRagBboxPosition > ,
paragraph_ordinal : Option < u32 > ,
para_id_start : Option < String > ,
para_id_end : Option < String > ,
text_fingerprint : Option < String > ,
}
fn parse_lightrag_locator_positions ( positions : & [ Value ]) -> LightRagLocatorPositions {
let mut parsed = LightRagLocatorPositions ::default ();
for position in positions {
match position . get ( "type" ). and_then ( Value ::as_str ) {
Some ( "bbox" ) if parsed . bbox . is_none () => {
parsed . bbox = parse_bbox_position ( position );
}
Some ( "paraid" ) => {
if parsed . paragraph_ordinal . is_none () {
parsed . paragraph_ordinal = position . get ( "anchor" ). and_then ( value_to_u32 );
}
if parsed . para_id_start . is_none () || parsed . para_id_end . is_none () {
if let Some ( range ) = position . get ( "range" ). and_then ( Value ::as_array ) {
parsed . para_id_start = range
. first ()
. and_then ( Value ::as_str )
. map ( | value | value . trim (). to_string ())
. filter ( | value | ! value . is_empty ());
parsed . para_id_end = range
. get ( 1 )
. and_then ( Value ::as_str )
. map ( | value | value . trim (). to_string ())
. filter ( | value | ! value . is_empty ())
. or_else ( || parsed . para_id_start . clone ());
}
}
}
Some ( "text_fingerprint" ) if parsed . text_fingerprint . is_none () => {
parsed . text_fingerprint = position
. get ( "anchor" )
. and_then ( Value ::as_str )
. map ( | value | value . trim (). to_string ())
. filter ( | value | ! value . is_empty ());
}
_ => {}
}
}
parsed
}
fn parse_bbox_position ( value : & Value ) -> Option < LightRagBboxPosition > {
let page = value . get ( "anchor" ). and_then ( value_to_u32 ) ? ;
2026-06-07 01:10:31 +08:00
let range = value . get ( "range" ) ? . as_array () ? ;
if range . len () != 4 {
return None ;
}
2026-06-09 09:20:56 +08:00
Some ( LightRagBboxPosition {
2026-06-07 01:10:31 +08:00
page ,
bbox : EvidenceBBox {
x0 : range . first () ? . as_f64 () ? ,
y0 : range . get ( 1 ) ? . as_f64 () ? ,
x1 : range . get ( 2 ) ? . as_f64 () ? ,
y1 : range . get ( 3 ) ? . as_f64 () ? ,
},
})
}
2026-06-09 09:20:56 +08:00
fn value_to_u32 ( value : & Value ) -> Option < u32 > {
value
. as_u64 ()
. and_then ( | value | u32 ::try_from ( value ). ok ())
. or_else ( || {
value
. as_str ()
. and_then ( | value | value . trim (). parse ::< u32 > (). ok ())
})
}
2026-06-07 01:10:31 +08:00
fn find_lightrag_sidecar_block (
entry : & KnowledgeRagSourceRegistryEntry ,
2026-06-08 20:35:49 +08:00
query : Option <& str > ,
2026-06-09 18:40:48 +08:00
occurrence_index : & Value ,
chunk_sidecar : Option <& Value > ,
2026-06-07 01:10:31 +08:00
) -> Option < Value > {
2026-06-09 18:40:48 +08:00
let ref_ids = lightrag_sidecar_ref_ids ( chunk_sidecar ? );
if ref_ids . is_empty () {
2026-06-07 01:10:31 +08:00
return None ;
}
2026-06-09 18:40:48 +08:00
let blocks = lightrag_sidecar_blocks_by_id ( entry , & ref_ids ) ? ;
let query = query
. map ( str ::trim )
. filter ( | value | ! value . is_empty ())
. map ( ToOwned ::to_owned );
let Some ( query ) = query else {
return ref_ids . into_iter (). find_map ( | id | blocks . get ( & id ). cloned ());
};
let target_occurrence = occurrence_index . as_u64 (). unwrap_or ( 0 );
let mut seen_occurrences = 0 u64 ;
for id in ref_ids {
let Some ( block ) = blocks . get ( & id ) else {
2026-06-07 01:10:31 +08:00
continue ;
2026-06-09 18:40:48 +08:00
};
let count = block
2026-06-07 01:10:31 +08:00
. get ( "content" )
. and_then ( Value ::as_str )
2026-06-09 18:40:48 +08:00
. map ( | content | exact_match_count ( content , & query ))
. unwrap_or ( 0 );
if count == 0 {
2026-06-07 01:10:31 +08:00
continue ;
}
2026-06-09 18:40:48 +08:00
if seen_occurrences + count > target_occurrence {
return Some ( block . clone ());
}
seen_occurrences += count ;
}
2026-06-13 22:20:01 +08:00
let query_terms = locator_query_match_terms ( & query );
if query_terms . is_empty () {
return None ;
}
let ref_ids = lightrag_sidecar_ref_ids ( chunk_sidecar ? );
if query_terms . len () <= 6 {
for id in & ref_ids {
let Some ( block ) = blocks . get ( id ) else {
continue ;
};
let block_normalized = block
. get ( "content" )
. and_then ( Value ::as_str )
. map ( | content | normalize_text_for_match ( content ). to_lowercase ())
. unwrap_or_default ();
if ! block_normalized . is_empty ()
&& query_terms
. iter ()
. all ( | term | block_matches_query_term ( & block_normalized , term ))
{
return Some ( block . clone ());
}
}
}
for id in & ref_ids {
let Some ( block ) = blocks . get ( id ) else {
continue ;
};
let block_normalized = block
. get ( "content" )
. and_then ( Value ::as_str )
. map ( | content | normalize_text_for_match ( content ). to_lowercase ())
. unwrap_or_default ();
if ! block_normalized . is_empty ()
&& query_terms
. iter ()
. any ( | term | block_matches_query_term ( & block_normalized , term ))
{
return Some ( block . clone ());
}
}
2026-06-09 18:40:48 +08:00
None
}
fn lightrag_sidecar_ref_ids ( sidecar : & Value ) -> Vec < String > {
let mut ids = Vec ::new ();
if let Some ( id ) = sidecar . get ( "id" ). and_then ( Value ::as_str ) {
ids . push ( id . trim (). to_string ());
}
if let Some ( refs ) = sidecar . get ( "refs" ). and_then ( Value ::as_array ) {
for item in refs {
if let Some ( id ) = item . get ( "id" ). and_then ( Value ::as_str ) {
ids . push ( id . trim (). to_string ());
2026-06-08 20:35:49 +08:00
}
}
2026-06-09 18:40:48 +08:00
}
ids . retain ( | id | ! id . is_empty ());
ids . dedup ();
ids
}
fn lightrag_sidecar_blocks_by_id (
entry : & KnowledgeRagSourceRegistryEntry ,
ref_ids : & [ String ],
) -> Option < BTreeMap < String , Value >> {
let wanted = ref_ids . iter (). cloned (). collect ::< BTreeSet < _ >> ();
let path = sidecar_blocks_path ( entry ) ? ;
let content = fs ::read_to_string ( path ). ok () ? ;
let mut blocks = BTreeMap ::new ();
for line in content . lines () {
let block = serde_json ::from_str ::< Value > ( line ). ok () ? ;
let Some ( block_id ) = block . get ( "blockid" ). and_then ( Value ::as_str ) else {
continue ;
};
if wanted . contains ( block_id ) && block . get ( "positions" ). and_then ( Value ::as_array ). is_some () {
blocks . insert ( block_id . to_string (), block );
}
if blocks . len () == wanted . len () {
break ;
2026-06-07 01:10:31 +08:00
}
}
2026-06-09 18:40:48 +08:00
Some ( blocks )
2026-06-08 20:35:49 +08:00
}
2026-06-09 18:40:48 +08:00
fn exact_match_count ( content : & str , query : & str ) -> u64 {
if query . is_empty () {
return 0 ;
}
let content_lower = content . to_lowercase ();
let query_lower = query . to_lowercase ();
let mut count = 0 u64 ;
let mut start = 0 usize ;
while let Some ( index ) = content_lower [ start .. ]. find ( & query_lower ) {
count += 1 ;
start += index + query_lower . len (). max ( 1 );
}
count
2026-06-08 20:35:49 +08:00
}
2026-06-09 18:40:48 +08:00
fn lightrag_reference_sidecar (
reference : & Value ,
primary_chunk : Option <& Value > ,
source_chunk_id : & Value ,
chunk_id : & Value ,
) -> Option < Value > {
reference
. get ( "sidecar" )
. cloned ()
. or_else ( || {
primary_chunk
. and_then ( | chunk | chunk . get ( "sidecar" ))
. cloned ()
})
. or_else ( || lightrag_chunk_sidecar_for_reference_ids ( source_chunk_id , chunk_id ))
2026-06-08 20:35:49 +08:00
}
fn block_matches_query_term ( block_normalized : & str , term : & str ) -> bool {
if block_normalized . contains ( term ) {
return true ;
}
let term_chars = term . chars (). collect ::< Vec < _ >> ();
if term_chars . len () < 2 || ! term_chars . iter (). all ( | ch | is_cjk_char ( * ch )) {
return false ;
}
let mut index = 0 usize ;
let mut gap = 0 usize ;
for ch in block_normalized . chars () {
if ch == term_chars [ index ] {
index += 1 ;
gap = 0 ;
if index == term_chars . len () {
return true ;
}
continue ;
}
if index > 0 {
gap += 1 ;
if gap > 4 {
index = 0 ;
gap = 0 ;
}
}
}
false
}
fn query_match_terms ( query_normalized : & str ) -> Vec < String > {
let mut terms = query_normalized
. split ( | ch : char | ch . is_whitespace () || ch . is_ascii_punctuation ())
. map ( str ::trim )
. filter ( | term | ! term . is_empty ())
. map ( ToOwned ::to_owned )
. collect ::< Vec < _ >> ();
for term in terms . clone () {
let chars = term . chars (). collect ::< Vec < _ >> ();
if chars . len () < 4 || ! chars . iter (). any ( | ch | is_cjk_char ( * ch )) {
continue ;
}
for window in chars . windows ( 2 ) {
terms . push ( window . iter (). collect ::< String > ());
}
for window in chars . windows ( 3 ) {
terms . push ( window . iter (). collect ::< String > ());
}
}
terms . sort ();
terms . dedup ();
terms
}
2026-06-13 22:20:01 +08:00
fn locator_query_match_terms ( query : & str ) -> Vec < String > {
let query_normalized = normalize_text_for_match ( query ). to_lowercase ();
let mut terms = query_match_terms ( & query_normalized )
. into_iter ()
. filter ( | term | {
let chars = term . chars (). collect ::< Vec < _ >> ();
if chars . iter (). any ( | ch | is_cjk_char ( * ch )) {
return chars . len () <= 4
&& chars . iter (). all ( | ch | is_cjk_char ( * ch ))
&& significant_cjk_query_term ( term );
}
chars . len () >= 3
})
. collect ::< Vec < _ >> ();
terms . sort ();
terms . dedup ();
terms
}
2026-06-09 09:20:56 +08:00
fn query_is_single_cjk_lookup ( query_normalized : & str ) -> bool {
let chars = query_normalized . chars (). collect ::< Vec < _ >> ();
! chars . is_empty ()
&& chars . len () <= 8
&& chars
. iter ()
. all ( | ch | is_cjk_char ( * ch ) || ch . is_ascii_alphanumeric ())
}
fn significant_cjk_query_term ( term : & str ) -> bool {
let chars = term . chars (). collect ::< Vec < _ >> ();
if chars . len () < 2 || ! chars . iter (). all ( | ch | is_cjk_char ( * ch )) {
return false ;
}
! matches! (
term ,
"请用"
| "使用"
| "资料"
| "资料库"
| "知识"
| "知识库"
| "检索"
| "搜索"
| "回答"
| "说明"
| "解释"
| "总结"
| "用途"
| "应用"
| "来源"
| "出处"
| "证据"
| "定位"
| "相关"
| "什么"
| "哪些"
| "怎么"
| "如何"
| "中的"
| "有什么"
)
}
2026-06-08 20:35:49 +08:00
fn is_cjk_char ( ch : char ) -> bool {
( '\u{4e00}' ..= '\u{9fff}' ). contains ( & ch )
|| ( '\u{3400}' ..= '\u{4dbf}' ). contains ( & ch )
|| ( '\u{f900}' ..= '\u{faff}' ). contains ( & ch )
2026-06-07 01:10:31 +08:00
}
fn normalize_text_for_match ( value : & str ) -> String {
value
. split_whitespace ()
. collect ::< Vec < _ >> ()
. join ( " " )
. chars ()
. collect ::< String > ()
}
fn sidecar_blocks_path ( entry : & KnowledgeRagSourceRegistryEntry ) -> Option < PathBuf > {
for file_path in lightrag_sidecar_file_path_candidates ( entry ) {
let stem = Path ::new ( & file_path ). file_stem () ? . to_string_lossy ();
let path = lightrag_input_dir ()
. join ( "__parsed__" )
. join ( format! ( " {file_path} .parsed" ))
. join ( format! ( " {stem} .blocks.jsonl" ));
if path . exists () {
return Some ( path );
}
}
None
}
fn lightrag_sidecar_file_path_candidates ( entry : & KnowledgeRagSourceRegistryEntry ) -> Vec < String > {
let mut candidates = vec! [ entry . light_rag_file_path . clone ()];
if let Some ( stripped ) = strip_one_supported_parser_hint ( & entry . light_rag_file_path ) {
candidates . push ( stripped );
}
2026-06-08 20:35:49 +08:00
if let Some ( asset_path ) = image_wrapper_asset_file_path ( & entry . light_rag_file_path ) {
candidates . push ( asset_path );
}
2026-06-07 01:10:31 +08:00
candidates . sort ();
candidates . dedup ();
candidates
}
2026-06-08 20:35:49 +08:00
fn image_wrapper_asset_file_path ( file_path : & str ) -> Option < String > {
let asset_path = file_path . strip_suffix ( ".md" ) ? ;
if resource_type_for_path ( asset_path ) == "image" {
return Some ( asset_path . to_string ());
}
None
}
fn resource_type_for_path ( path : & str ) -> & 'static str {
match Path ::new ( path )
. extension ()
. and_then ( | value | value . to_str ())
. map ( | value | value . to_ascii_lowercase ())
. as_deref ()
{
Some ( "png" | "jpg" | "jpeg" | "webp" | "gif" | "bmp" | "tif" | "tiff" ) => "image" ,
Some ( "pdf" ) => "pdf" ,
Some ( "doc" | "docx" ) => "docx" ,
Some ( "ppt" | "pptx" ) => "pptx" ,
Some ( "xls" | "xlsx" ) => "xlsx" ,
_ => "resource" ,
}
}
fn knowledge_rag_source_content_diagnostics ( registry : & KnowledgeRagSourceRegistry ) -> Value {
Value ::Array (
registry
. entries
. iter ()
. map ( | entry | {
let sidecar_path = sidecar_blocks_path ( entry );
let sidecar_stats = sidecar_path
. as_ref ()
. and_then ( | path | sidecar_text_stats ( path ). ok ());
json! ({
"sourceId" : entry . source_id ,
"sourceRootRelativePath" : entry . source_root_relative_path ,
"lightRagDocId" : entry . light_rag_doc_id ,
"lightRagFilePath" : entry . light_rag_file_path ,
"lightRagStatus" : entry . light_rag_status ,
"directImageScan" : resource_type_for_path ( & entry . light_rag_file_path ) == "image" ,
"sidecarExists" : sidecar_path . is_some (),
"sidecarPath" : sidecar_path . map ( | path | path . display (). to_string ()),
"ocrTextExposed" : sidecar_stats
. as_ref ()
. map ( | stats | stats . meaningful_blocks > 0 )
. unwrap_or ( false ),
"sidecarBlocks" : sidecar_stats . as_ref (). map ( | stats | stats . blocks ). unwrap_or ( 0 ),
"sidecarMeaningfulBlocks" : sidecar_stats
. as_ref ()
. map ( | stats | stats . meaningful_blocks )
. unwrap_or ( 0 ),
})
})
. collect (),
)
}
2026-06-13 22:20:01 +08:00
#[derive(Debug, Clone)]
struct SidecarContextBlock {
block_ordinal : u64 ,
block_id : String ,
text : String ,
heading_path : Vec < String > ,
positions : LightRagLocatorPositions ,
}
fn sidecar_section_context_payload (
root_path : & Path ,
registry : & KnowledgeRagSourceRegistry ,
request : & KnowledgeRagSectionContextRequest ,
) -> Result < Value , WebError > {
let entry = resolve_section_context_entry ( registry , request ) ? ;
let sidecar_path = sidecar_blocks_path ( entry ). ok_or_else ( || {
WebError ::bad_request_code (
"knowledge_rag_section_sidecar_missing" ,
"该资料没有可读取的 LightRAG native sidecar blocks" ,
)
}) ? ;
let blocks = read_sidecar_context_blocks ( & sidecar_path ) ? ;
let ( selected_start , selected_end ) =
section_context_selected_range ( & blocks , request , entry , root_path ) ? ;
let context_before = request . context_before . unwrap_or ( 1 ). min ( 20 );
let context_after = request . context_after . unwrap_or ( 1 ). min ( 20 );
let range_start = selected_start . saturating_sub ( context_before );
let range_end = selected_end . saturating_add ( context_after );
let max_blocks = request . max_blocks . unwrap_or ( 24 ). clamp ( 1 , 80 );
let max_chars = request . max_chars . unwrap_or ( 12_000 ). clamp ( 500 , 40_000 );
let mut remaining_chars = max_chars ;
let mut returned_blocks = Vec ::new ();
let mut text_parts = Vec ::new ();
let mut truncated = false ;
for block in blocks
. iter ()
. filter ( | block | block . block_ordinal >= range_start && block . block_ordinal <= range_end )
{
if returned_blocks . len () >= max_blocks {
truncated = true ;
break ;
}
if remaining_chars == 0 {
truncated = true ;
break ;
}
let ( display_text , text_truncated ) =
take_chars_with_truncation ( & block . text , remaining_chars );
remaining_chars = remaining_chars . saturating_sub ( display_text . chars (). count ());
if text_truncated {
truncated = true ;
}
text_parts . push ( display_text . clone ());
returned_blocks . push ( json! ({
"blockOrdinal" : block . block_ordinal ,
"blockId" : block . block_id . clone (),
"paragraphOrdinal" : block . positions . paragraph_ordinal ,
"headingPath" : block . heading_path . clone (),
"text" : display_text ,
"textTruncated" : text_truncated ,
"locator" : sidecar_context_locator ( & block . positions ),
}));
if text_truncated {
break ;
}
}
let chunks = section_context_chunks ( & returned_blocks , 2_400 );
Ok ( json! ({
"ok" : true ,
"schema" : "mnote.knowledge_rag.section_context.v1" ,
"provider" : "lightrag" ,
"sourceId" : entry . source_id ,
"sourceRootRelativePath" : entry . source_root_relative_path ,
"lightRagDocId" : entry . light_rag_doc_id ,
"lightRagFilePath" : entry . light_rag_file_path ,
"sidecarPath" : root_relative_path ( root_path , & sidecar_path ). unwrap_or_else ( | _ | sidecar_path . display (). to_string ()),
"section" : {
"sectionId" : request . section_id ,
"startBlockOrdinal" : request . start_block_ordinal ,
"endBlockOrdinal" : request . end_block_ordinal ,
"startParagraphOrdinal" : request . start_paragraph_ordinal ,
"endParagraphOrdinal" : request . end_paragraph_ordinal ,
"selectedStartBlockOrdinal" : selected_start ,
"selectedEndBlockOrdinal" : selected_end ,
},
"limits" : {
"contextBefore" : context_before ,
"contextAfter" : context_after ,
"maxBlocks" : max_blocks ,
"maxChars" : max_chars ,
"returnedBlocks" : returned_blocks . len (),
"returnedChars" : max_chars . saturating_sub ( remaining_chars ),
"truncated" : truncated ,
},
"blocks" : returned_blocks ,
"chunks" : chunks ,
"text" : text_parts . join ( " \n\n " ),
}))
}
fn resolve_section_context_entry < 'a > (
registry : & 'a KnowledgeRagSourceRegistry ,
request : & KnowledgeRagSectionContextRequest ,
) -> Result <& 'a KnowledgeRagSourceRegistryEntry , WebError > {
let source_id = request
. source_id
. as_deref ()
. map ( str ::trim )
. filter ( | value | ! value . is_empty ())
. or_else ( || {
request
. section_id
. as_deref ()
. and_then ( | value | value . split_once ( '#' ). map ( | ( source_id , _ ) | source_id . trim ()))
. filter ( | value | ! value . is_empty ())
});
let source_path = request
. source_path
. as_deref ()
. map ( | value | value . trim (). trim_matches ( '/' ). replace ( '\\' , "/" ))
. filter ( | value | ! value . is_empty ());
let light_rag_doc_id = request
. light_rag_doc_id
. as_deref ()
. map ( str ::trim )
. filter ( | value | ! value . is_empty ());
let file_path = request
. file_path
. as_deref ()
. map ( str ::trim )
. filter ( | value | ! value . is_empty ());
if source_id . is_none ()
&& source_path . is_none ()
&& light_rag_doc_id . is_none ()
&& file_path . is_none ()
{
return Err ( WebError ::bad_request_code (
"knowledge_rag_section_source_required" ,
"按章节读取资料上下文需要 sourcePath、sourceId、lightRagDocId 或 filePath" ,
));
}
registry
. entries
. iter ()
. find ( | entry | {
! entry . stale
&& entry . deleted_at_ms . is_none ()
&& entry
. light_rag_status
. as_deref ()
. is_none_or ( | status | status == "processed" )
&& source_id . is_none_or ( | value | entry . source_id == value )
&& source_path
. as_deref ()
. is_none_or ( | value | entry . source_root_relative_path == value )
&& light_rag_doc_id
. is_none_or ( | value | entry . light_rag_doc_id . as_deref () == Some ( value ))
&& file_path . is_none_or ( | value | lightrag_file_path_matches ( entry , value ))
})
. ok_or_else ( || {
WebError ::bad_request_code (
"knowledge_rag_section_source_not_found" ,
"没有找到匹配且已处理完成的资料来源" ,
)
})
}
fn read_sidecar_context_blocks ( path : & Path ) -> Result < Vec < SidecarContextBlock > , WebError > {
let content = fs ::read_to_string ( path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_sidecar_read_failed" ,
format! ( "无法读取 LightRAG sidecar {} : {error} " , path . display ()),
)
}) ? ;
let mut blocks = Vec ::new ();
let mut ordinal = 0 u64 ;
for line in content . lines () {
let Ok ( block ) = serde_json ::from_str ::< Value > ( line ) else {
continue ;
};
if block . get ( "type" ). and_then ( Value ::as_str ) != Some ( "content" ) {
continue ;
}
let raw_text = block
. get ( "content" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let positions = block
. get ( "positions" )
. and_then ( Value ::as_array )
. map ( | positions | parse_lightrag_locator_positions ( positions . as_slice ()))
. unwrap_or_default ();
blocks . push ( SidecarContextBlock {
block_ordinal : ordinal ,
block_id : block
. get ( "blockid" )
. and_then ( Value ::as_str )
. unwrap_or_default ()
. to_string (),
text : clean_lightrag_text_for_display ( raw_text ),
heading_path : heading_path_for_sidecar_block ( Some ( & block )),
positions ,
});
ordinal += 1 ;
}
Ok ( blocks )
}
fn section_context_selected_range (
blocks : & [ SidecarContextBlock ],
request : & KnowledgeRagSectionContextRequest ,
entry : & KnowledgeRagSourceRegistryEntry ,
root_path : & Path ,
) -> Result < ( u64 , u64 ), WebError > {
if blocks . is_empty () {
return Err ( WebError ::bad_request_code (
"knowledge_rag_section_sidecar_empty" ,
"该资料的 LightRAG sidecar 没有 content blocks" ,
));
}
if let Some ( start ) = request . start_block_ordinal {
let end = request . end_block_ordinal . unwrap_or ( start );
if start > end {
return Err ( WebError ::bad_request_code (
"knowledge_rag_section_block_range_invalid" ,
"startBlockOrdinal 不能大于 endBlockOrdinal" ,
));
}
return Ok (( start , end ));
}
if let Some ( start ) = request . start_paragraph_ordinal {
let end = request . end_paragraph_ordinal . unwrap_or ( start );
if start > end {
return Err ( WebError ::bad_request_code (
"knowledge_rag_section_paragraph_range_invalid" ,
"startParagraphOrdinal 不能大于 endParagraphOrdinal" ,
));
}
let matching = blocks
. iter ()
. filter ( | block | {
block
. positions
. paragraph_ordinal
. is_some_and ( | ordinal | ordinal >= start && ordinal <= end )
})
. map ( | block | block . block_ordinal )
. collect ::< Vec < _ >> ();
let first = matching . first (). copied (). ok_or_else ( || {
WebError ::bad_request_code (
"knowledge_rag_section_paragraph_range_not_found" ,
"没有在 sidecar blocks 中找到对应 paragraph ordinal 范围" ,
)
}) ? ;
let last = matching . last (). copied (). unwrap_or ( first );
return Ok (( first , last ));
}
if let Some ( section_id ) = request
. section_id
. as_deref ()
. map ( str ::trim )
. filter ( | value | ! value . is_empty ())
{
if let Some (( start , end )) =
section_context_range_for_section_id ( root_path , entry , section_id )
{
return Ok (( start , end ));
}
}
Err ( WebError ::bad_request_code (
"knowledge_rag_section_range_required" ,
"按章节读取资料上下文需要 sectionId、startBlockOrdinal 或 startParagraphOrdinal" ,
))
}
fn section_context_range_for_section_id (
root_path : & Path ,
entry : & KnowledgeRagSourceRegistryEntry ,
section_id : & str ,
) -> Option < ( u64 , u64 ) > {
let structure = document_structure_for_entry ( root_path , entry , & BTreeSet ::new (), None ) ? ;
let sections = structure . get ( "sections" ). and_then ( Value ::as_array ) ? ;
sections . iter (). find_map ( | section | {
let matches_id = section
. get ( "sectionId" )
. and_then ( Value ::as_str )
. is_some_and ( | value | value == section_id );
if ! matches_id {
return None ;
}
let start = section . get ( "startBlockOrdinal" ). and_then ( Value ::as_u64 ) ? ;
let end = section
. get ( "endBlockOrdinal" )
. and_then ( Value ::as_u64 )
. unwrap_or ( start );
Some (( start , end ))
})
}
fn sidecar_context_locator ( positions : & LightRagLocatorPositions ) -> Value {
let bbox = positions . bbox . as_ref (). map ( | position | {
json! ({
"page" : position . page ,
"bbox" : {
"x0" : position . bbox . x0 ,
"y0" : position . bbox . y0 ,
"x1" : position . bbox . x1 ,
"y1" : position . bbox . y1 ,
}
})
});
json! ({
"paragraphOrdinal" : positions . paragraph_ordinal ,
"paraIdStart" : positions . para_id_start . clone (),
"paraIdEnd" : positions . para_id_end . clone (),
"textFingerprint" : positions . text_fingerprint . clone (),
"bbox" : bbox ,
})
}
fn section_context_chunks ( blocks : & [ Value ], max_chunk_chars : usize ) -> Vec < Value > {
let mut chunks = Vec ::new ();
let mut current_text = String ::new ();
let mut current_block_ids = Vec ::new ();
let mut start_block_ordinal : Option < u64 > = None ;
let mut end_block_ordinal : Option < u64 > = None ;
let flush_current = | chunks : & mut Vec < Value > ,
current_text : & mut String ,
current_block_ids : & mut Vec < String > ,
start_block_ordinal : & mut Option < u64 > ,
end_block_ordinal : & mut Option < u64 >| {
if current_text . trim (). is_empty () {
current_block_ids . clear ();
* start_block_ordinal = None ;
* end_block_ordinal = None ;
return ;
}
let chunk_ordinal = chunks . len ();
let text = current_text . trim (). to_string ();
let block_ids = current_block_ids . clone ();
let start = * start_block_ordinal ;
let end = * end_block_ordinal ;
chunks . push ( json! ({
"chunkOrdinal" : chunk_ordinal ,
"startBlockOrdinal" : start ,
"endBlockOrdinal" : end ,
"blockIds" : block_ids ,
"text" : text ,
}));
current_text . clear ();
current_block_ids . clear ();
* start_block_ordinal = None ;
* end_block_ordinal = None ;
};
for block in blocks {
let text = block
. get ( "text" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
if text . trim (). is_empty () {
continue ;
}
let block_ordinal = block . get ( "blockOrdinal" ). and_then ( Value ::as_u64 );
let block_id = block
. get ( "blockId" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let projected_chars = current_text . chars (). count () + text . chars (). count () + 2 ;
if ! current_text . is_empty () && projected_chars > max_chunk_chars {
flush_current (
& mut chunks ,
& mut current_text ,
& mut current_block_ids ,
& mut start_block_ordinal ,
& mut end_block_ordinal ,
);
}
if start_block_ordinal . is_none () {
start_block_ordinal = block_ordinal ;
}
end_block_ordinal = block_ordinal . or ( end_block_ordinal );
if ! block_id . is_empty () {
current_block_ids . push ( block_id . to_string ());
}
if ! current_text . is_empty () {
current_text . push_str ( " \n\n " );
}
current_text . push_str ( text );
}
flush_current (
& mut chunks ,
& mut current_text ,
& mut current_block_ids ,
& mut start_block_ordinal ,
& mut end_block_ordinal ,
);
chunks
}
fn take_chars_with_truncation ( value : & str , max_chars : usize ) -> ( String , bool ) {
let mut output = String ::new ();
let mut truncated = false ;
for ( index , ch ) in value . chars (). enumerate () {
if index >= max_chars {
truncated = true ;
break ;
}
output . push ( ch );
}
( output , truncated )
}
#[derive(Debug, Clone, Default, Serialize)]
#[serde(rename_all = "camelCase" )]
struct DocumentStructureSection {
section_id : String ,
title : String ,
level : u64 ,
heading_path : Vec < String > ,
start_block_ordinal : u64 ,
end_block_ordinal : u64 ,
block_count : u64 ,
start_paragraph_ordinal : Option < u32 > ,
end_paragraph_ordinal : Option < u32 > ,
start_block_id : String ,
end_block_id : String ,
query_match_count : u64 ,
matched_reference_count : u64 ,
sample : String ,
}
fn document_structure_index_payload (
root_path : & Path ,
registry : & KnowledgeRagSourceRegistry ,
source_scope : & [ String ],
references : & [ Value ],
query : Option <& str > ,
max_documents : usize ,
) -> Result < Value , WebError > {
let reference_source_ids = references
. iter ()
. filter_map ( | reference | reference . get ( "sourceId" ). and_then ( Value ::as_str ))
. map ( ToOwned ::to_owned )
. collect ::< BTreeSet < _ >> ();
let mut entries = registry
. entries
. iter ()
. filter ( | entry | {
! entry . stale
&& entry . deleted_at_ms . is_none ()
&& entry
. light_rag_status
. as_deref ()
. is_none_or ( | status | status == "processed" )
})
. filter ( | entry | {
if source_scope . is_empty () {
reference_source_ids . is_empty () || reference_source_ids . contains ( & entry . source_id )
} else {
source_scope . iter (). any ( | scope | {
entry . source_root_relative_path == * scope
|| entry
. source_root_relative_path
. starts_with ( & format! ( " {scope} /" ))
})
}
})
. collect ::< Vec < _ >> ();
entries . sort_by ( | left , right | {
left . source_root_relative_path
. cmp ( & right . source_root_relative_path )
});
entries . truncate ( max_documents );
let reference_block_ids = references
. iter ()
. filter_map ( | reference | {
reference
. get ( "locator" )
. and_then ( | locator | locator . get ( "blockId" ))
. and_then ( Value ::as_str )
})
. map ( ToOwned ::to_owned )
. collect ::< BTreeSet < _ >> ();
let documents = entries
. into_iter ()
. map ( | entry | {
document_structure_for_entry ( root_path , entry , & reference_block_ids , query )
. unwrap_or_else ( || {
json! ({
"sourceId" : entry . source_id ,
"sourceRootRelativePath" : entry . source_root_relative_path ,
"lightRagDocId" : entry . light_rag_doc_id ,
"lightRagFilePath" : entry . light_rag_file_path ,
"sidecarAvailable" : false ,
"sections" : [],
})
})
})
. collect ::< Vec < _ >> ();
Ok ( json! ({
"schema" : DOCUMENT_STRUCTURE_INDEX_SCHEMA ,
"generatedAtMs" : now_ms (),
"mode" : "sidecar_heading_sections" ,
"sourceScope" : source_scope ,
"referenceCount" : references . len (),
"documents" : documents ,
}))
}
fn document_structure_for_entry (
root_path : & Path ,
entry : & KnowledgeRagSourceRegistryEntry ,
reference_block_ids : & BTreeSet < String > ,
query : Option <& str > ,
) -> Option < Value > {
let sidecar_path = sidecar_blocks_path ( entry ) ? ;
let content = fs ::read_to_string ( & sidecar_path ). ok () ? ;
let query_terms = query
. map ( normalize_text_for_match )
. map ( | value | query_match_terms ( & value . to_ascii_lowercase ()))
. unwrap_or_default ()
. into_iter ()
. filter ( | term | term . chars (). count () >= 2 )
. collect ::< Vec < _ >> ();
let mut sections = Vec ::< DocumentStructureSection > ::new ();
let mut current : Option < DocumentStructureSection > = None ;
let mut ordinal = 0 u64 ;
let mut truncated = false ;
for line in content . lines () {
let Ok ( block ) = serde_json ::from_str ::< Value > ( line ) else {
continue ;
};
if block . get ( "type" ). and_then ( Value ::as_str ) != Some ( "content" ) {
continue ;
}
let block_id = block
. get ( "blockid" )
. and_then ( Value ::as_str )
. unwrap_or_default ()
. to_string ();
let block_content = block
. get ( "content" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
let positions = block
. get ( "positions" )
. and_then ( Value ::as_array )
. map ( | positions | parse_lightrag_locator_positions ( positions . as_slice ()))
. unwrap_or_default ();
let heading_title = heading_title_from_block ( block_content );
if let Some ( title ) = heading_title {
if let Some ( section ) = current . take () {
sections . push ( section );
if sections . len () >= DOCUMENT_STRUCTURE_INDEX_MAX_SECTIONS_PER_DOC {
truncated = true ;
break ;
}
}
let mut heading_path = heading_path_for_sidecar_block ( Some ( & block ));
if heading_path . last (). is_none_or ( | last | last != & title ) {
heading_path . push ( title . clone ());
}
let level = block . get ( "level" ). and_then ( Value ::as_u64 ). unwrap_or ( 1 );
current = Some ( DocumentStructureSection {
section_id : format ! ( "{}#block-{}" , entry . source_id , block_id ),
title ,
level ,
heading_path ,
start_block_ordinal : ordinal ,
end_block_ordinal : ordinal ,
block_count : 1 ,
start_paragraph_ordinal : positions . paragraph_ordinal ,
end_paragraph_ordinal : positions . paragraph_ordinal ,
start_block_id : block_id . clone (),
end_block_id : block_id . clone (),
query_match_count : query_match_count ( block_content , & query_terms ),
matched_reference_count : u64 ::from ( reference_block_ids . contains ( & block_id )),
sample : String ::new (),
});
} else {
let section = current . get_or_insert_with ( || DocumentStructureSection {
section_id : format ! ( "{}#body" , entry . source_id ),
title : "正文" . into (),
level : 0 ,
heading_path : Vec ::new (),
start_block_ordinal : ordinal ,
end_block_ordinal : ordinal ,
block_count : 0 ,
start_paragraph_ordinal : positions . paragraph_ordinal ,
end_paragraph_ordinal : positions . paragraph_ordinal ,
start_block_id : block_id . clone (),
end_block_id : block_id . clone (),
query_match_count : 0 ,
matched_reference_count : 0 ,
sample : String ::new (),
});
section . end_block_ordinal = ordinal ;
section . block_count += 1 ;
section . end_paragraph_ordinal = positions
. paragraph_ordinal
. or ( section . end_paragraph_ordinal );
section . end_block_id = block_id . clone ();
section . query_match_count += query_match_count ( block_content , & query_terms );
if reference_block_ids . contains ( & block_id ) {
section . matched_reference_count += 1 ;
}
if section . sample . is_empty () {
section . sample = clean_lightrag_text_for_display ( block_content )
. chars ()
. take ( 220 )
. collect ();
}
}
ordinal += 1 ;
}
if ! truncated {
if let Some ( section ) = current . take () {
sections . push ( section );
}
}
let query_matched_sections = sections
. iter ()
. filter ( | section | section . query_match_count > 0 )
. count ();
let reference_matched_sections = sections
. iter ()
. filter ( | section | section . matched_reference_count > 0 )
. count ();
Some ( json! ({
"sourceId" : entry . source_id ,
"sourceRootRelativePath" : entry . source_root_relative_path ,
"lightRagDocId" : entry . light_rag_doc_id ,
"lightRagFilePath" : entry . light_rag_file_path ,
"sidecarAvailable" : true ,
"sidecarPath" : root_relative_path ( root_path , & sidecar_path ). unwrap_or_else ( | _ | sidecar_path . display (). to_string ()),
"sectionCount" : sections . len (),
"queryMatchedSections" : query_matched_sections ,
"referenceMatchedSections" : reference_matched_sections ,
"truncated" : truncated ,
"sections" : sections ,
}))
}
fn heading_title_from_block ( content : & str ) -> Option < String > {
let first = content . lines (). find ( | line | ! line . trim (). is_empty ()) ? . trim ();
if ! first . starts_with ( '#' ) {
return None ;
}
let title = clean_lightrag_text_for_display ( first . trim_start_matches ( '#' ). trim ());
( ! title . is_empty ()). then_some ( title )
}
fn query_match_count ( content : & str , query_terms : & [ String ]) -> u64 {
if query_terms . is_empty () {
return 0 ;
}
let normalized = normalize_text_for_match ( content ). to_ascii_lowercase ();
query_terms
. iter ()
. filter ( | term | block_matches_query_term ( & normalized , term ))
. count () as u64
}
2026-06-08 20:35:49 +08:00
#[derive(Debug, Clone, Default)]
struct SidecarTextStats {
blocks : usize ,
meaningful_blocks : usize ,
}
fn sidecar_text_stats ( path : & Path ) -> Result < SidecarTextStats , WebError > {
let content = fs ::read_to_string ( path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_sidecar_read_failed" ,
format! ( "无法读取 LightRAG sidecar {} : {error} " , path . display ()),
)
}) ? ;
let mut stats = SidecarTextStats ::default ();
for line in content . lines () {
let Ok ( block ) = serde_json ::from_str ::< Value > ( line ) else {
continue ;
};
if block . get ( "type" ). and_then ( Value ::as_str ) != Some ( "content" ) {
continue ;
}
stats . blocks += 1 ;
let content = block
. get ( "content" )
. and_then ( Value ::as_str )
. unwrap_or_default ();
if quote_content_diagnostics ( Some ( content ), "sidecar" )
. get ( "ocrTextExposed" )
. and_then ( Value ::as_bool )
. unwrap_or ( false )
{
stats . meaningful_blocks += 1 ;
}
}
Ok ( stats )
}
2026-06-09 18:40:48 +08:00
fn lightrag_chunk_content_for_reference_ids (
source_chunk_id : & Value ,
chunk_id : & Value ,
) -> Option < String > {
lightrag_chunk_id_candidates ( source_chunk_id , chunk_id )
. into_iter ()
. find_map ( | id | lightrag_chunk_content_for_id ( & id ))
}
fn lightrag_chunk_sidecar_for_reference_ids (
source_chunk_id : & Value ,
chunk_id : & Value ,
) -> Option < Value > {
lightrag_chunk_id_candidates ( source_chunk_id , chunk_id )
. into_iter ()
. find_map ( | id | {
lightrag_chunk_value_for_id ( & id ). and_then ( | chunk | chunk . get ( "sidecar" ). cloned ())
})
}
fn lightrag_chunk_id_candidates ( source_chunk_id : & Value , chunk_id : & Value ) -> Vec < String > {
let mut ids = Vec ::new ();
for value in [ source_chunk_id , chunk_id ] {
let Some ( raw ) = value
. as_str ()
. map ( str ::trim )
. filter ( | value | ! value . is_empty ())
else {
continue ;
};
ids . push ( raw . to_string ());
if let Some (( base , _ )) = raw . split_once ( "#match-" ) {
ids . push ( base . to_string ());
}
}
ids . sort ();
ids . dedup ();
ids
}
fn lightrag_chunk_value_for_id ( chunk_id : & str ) -> Option < Value > {
2026-06-07 01:10:31 +08:00
let content =
fs ::read_to_string ( lightrag_working_dir (). join ( "kv_store_text_chunks.json" )). ok () ? ;
let chunks = serde_json ::from_str ::< Value > ( & content ). ok () ? ;
2026-06-09 18:40:48 +08:00
chunks . get ( chunk_id ). cloned ()
}
fn lightrag_chunk_content_for_id ( chunk_id : & str ) -> Option < String > {
lightrag_chunk_value_for_id ( chunk_id )
. as_ref ()
2026-06-07 01:10:31 +08:00
. and_then ( | chunk | chunk . get ( "content" ))
. and_then ( Value ::as_str )
. map ( ToOwned ::to_owned )
}
fn read_registry (
root_path : & Path ,
workspace_id : & str ,
root_uri : & str ,
) -> Result < KnowledgeRagSourceRegistry , WebError > {
let path = registry_path ( root_path );
if ! path . exists () {
return Ok ( KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : workspace_id . to_string (),
root_uri : root_uri . to_string (),
updated_at_ms : now_ms (),
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : Vec ::new (),
});
}
let content = fs ::read_to_string ( & path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_registry_read_failed" ,
format! ( "无法读取 LightRAG source registry: {error} " ),
)
}) ? ;
serde_json ::from_str ( & content ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_registry_invalid" ,
format! ( "LightRAG source registry JSON 无效: {error} " ),
)
})
}
fn write_registry (
root_path : & Path ,
registry : & mut KnowledgeRagSourceRegistry ,
) -> Result < (), WebError > {
registry . updated_at_ms = now_ms ();
let path = registry_path ( root_path );
if let Some ( parent ) = path . parent () {
fs ::create_dir_all ( parent ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_registry_dir_failed" ,
format! ( "无法创建 LightRAG source registry 目录: {error} " ),
)
}) ? ;
}
let content = serde_json ::to_string_pretty ( registry ). map_err ( | error | {
WebError ::internal ( format! ( "无法序列化 LightRAG source registry: {error} " ))
}) ? ;
fs ::write ( & path , format! ( " {content} \n " )). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_registry_write_failed" ,
format! ( "无法写入 LightRAG source registry: {error} " ),
)
})
}
fn upsert_registry_entry (
registry : & mut KnowledgeRagSourceRegistry ,
next : KnowledgeRagSourceRegistryEntry ,
) {
if let Some ( existing ) = registry
. entries
. iter_mut ()
. find ( | entry | entry . source_path == next . source_path )
{
* existing = next ;
return ;
}
registry . entries . push ( next );
}
2026-06-08 20:35:49 +08:00
fn upsert_indexed_root_for_request (
registry : & mut KnowledgeRagSourceRegistry ,
root_path : & Path ,
source_path : & str ,
context : & RequestContext ,
) -> Result < (), WebError > {
let canonical = ensure_source_path_in_root ( root_path , source_path , context ) ? ;
let relative = root_relative_path ( root_path , & canonical ) ? ;
let normalized = relative . trim_matches ( '/' ). replace ( '\\' , "/" );
let recursive = canonical . is_dir ();
let now = now_ms ();
if let Some ( existing ) = registry
. indexed_roots
. iter_mut ()
. find ( | root | root . root_relative_path == normalized )
{
existing . recursive = recursive ;
existing . run_on_change . get_or_insert ( true );
existing . updated_at_ms = now ;
return Ok (());
}
registry . indexed_roots . push ( KnowledgeRagIndexedRoot {
root_relative_path : normalized ,
recursive ,
exclude_patterns : Vec ::new (),
run_on_change : Some ( true ),
updated_at_ms : now ,
});
registry
. indexed_roots
. sort_by ( | left , right | left . root_relative_path . cmp ( & right . root_relative_path ));
Ok (())
}
2026-06-07 01:10:31 +08:00
fn registry_path ( root_path : & Path ) -> PathBuf {
root_path
. join ( ".mnote" )
. join ( "index" )
. join ( "lightrag-source-registry.json" )
}
2026-06-13 22:20:01 +08:00
fn document_structure_index_path ( root_path : & Path ) -> PathBuf {
root_path
. join ( ".mnote" )
. join ( "index" )
. join ( "lightrag-document-structure-index.json" )
}
fn refresh_document_structure_index (
root_path : & Path ,
registry : & KnowledgeRagSourceRegistry ,
) -> Result < (), WebError > {
let payload = document_structure_index_payload (
root_path ,
registry ,
& [],
& [],
None ,
DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PERSISTED ,
) ? ;
let path = document_structure_index_path ( root_path );
if let Some ( parent ) = path . parent () {
fs ::create_dir_all ( parent ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_structure_index_dir_failed" ,
format! ( "无法创建 LightRAG document_structure_index 目录: {error} " ),
)
}) ? ;
}
let content = serde_json ::to_string_pretty ( & payload ). map_err ( | error | {
WebError ::internal ( format! (
"无法序列化 LightRAG document_structure_index: {error} "
))
}) ? ;
fs ::write ( & path , format! ( " {content} \n " )). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_structure_index_write_failed" ,
format! ( "无法写入 LightRAG document_structure_index: {error} " ),
)
})
}
2026-06-07 01:10:31 +08:00
#[derive(Debug, Clone)]
struct ResolvedKnowledgeRagSource {
canonical_path : PathBuf ,
requested_path : String ,
source_kind : & 'static str ,
}
fn resolve_knowledge_rag_sources (
root_path : & Path ,
source_path : & str ,
context : & RequestContext ,
) -> Result < Vec < ResolvedKnowledgeRagSource > , WebError > {
let canonical = ensure_source_path_in_root ( root_path , source_path , context ) ? ;
if canonical . is_file () {
if ! knowledge_rag_source_supported_file ( & canonical ) {
return Err ( WebError ::bad_request_code (
"knowledge_rag_source_unsupported" ,
"资料库 source 文件类型暂不支持" ,
)
. with_context ( context ));
}
return Ok ( vec! [ ResolvedKnowledgeRagSource {
canonical_path : canonical ,
requested_path : source_path . trim (). to_string (),
source_kind : "file" ,
}]);
}
if canonical . is_dir () {
let mut files = collect_knowledge_rag_directory_sources ( root_path , & canonical , context ) ? ;
if files . is_empty () {
return Err ( WebError ::bad_request_code (
"knowledge_rag_source_directory_empty" ,
"资料库目录中没有可索引文件" ,
)
. with_context ( context ));
}
files . sort ();
files . dedup ();
return Ok ( files
. into_iter ()
. map ( | path | ResolvedKnowledgeRagSource {
canonical_path : path ,
requested_path : source_path . trim (). to_string (),
source_kind : "directory" ,
})
. collect ());
}
Err ( WebError ::bad_request_code (
"knowledge_rag_source_not_file_or_directory" ,
"资料库 source 必须是文件或目录" ,
)
. with_context ( context ))
}
fn collect_knowledge_rag_directory_sources (
root_path : & Path ,
directory : & Path ,
context : & RequestContext ,
) -> Result < Vec < PathBuf > , WebError > {
let mut files = Vec ::new ();
let mut stack = vec! [ directory . to_path_buf ()];
while let Some ( current_dir ) = stack . pop () {
let mut entries = fs ::read_dir ( & current_dir )
. map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_source_directory_read_failed" ,
format! ( "无法读取资料库目录 {} : {error} " , current_dir . display ()),
)
. with_context ( context )
}) ?
. filter_map ( Result ::ok )
. collect ::< Vec < _ >> ();
entries . sort_by_key ( | entry | entry . path ());
for entry in entries {
let path = entry . path ();
let canonical = path . canonicalize (). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_source_unavailable" ,
format! ( "无法访问资料库 source: {error} " ),
)
. with_context ( context )
}) ? ;
if ! canonical . starts_with ( root_path ) {
continue ;
}
if canonical . is_dir () {
if should_skip_knowledge_rag_directory ( & canonical ) {
continue ;
}
stack . push ( canonical );
continue ;
}
if canonical . is_file () && knowledge_rag_source_supported_file ( & canonical ) {
files . push ( canonical );
if files . len () > MAX_INGEST_SOURCES_PER_REQUEST {
return Err ( WebError ::bad_request_code (
"knowledge_rag_source_limit_exceeded" ,
format! (
"单次资料库索引最多支持 {MAX_INGEST_SOURCES_PER_REQUEST} 个文件,请缩小目录范围"
),
)
. with_context ( context ));
}
}
}
}
Ok ( files )
}
fn should_skip_knowledge_rag_directory ( path : & Path ) -> bool {
let name = path
. file_name ()
. and_then ( | value | value . to_str ())
. unwrap_or_default ();
matches! (
name ,
".git" | ".mnote" | "node_modules" | "target" | "__parsed__" | ".venv"
)
}
fn knowledge_rag_source_supported_file ( path : & Path ) -> bool {
path . extension ()
. and_then ( | value | value . to_str ())
. map ( | value | {
KNOWLEDGE_RAG_SOURCE_EXTENSIONS
. iter ()
. any ( | extension | value . eq_ignore_ascii_case ( extension ))
})
. unwrap_or ( false )
}
fn lightrag_scan_supported_file ( path : & Path ) -> bool {
path . extension ()
. and_then ( | value | value . to_str ())
. map ( | value | {
LIGHTRAG_SCAN_SOURCE_EXTENSIONS
. iter ()
. any ( | extension | value . eq_ignore_ascii_case ( extension ))
})
. unwrap_or ( false )
}
struct StagedLightRagSource {
light_rag_file_path : String ,
staged_path : PathBuf ,
}
fn stage_lightrag_source (
canonical : & Path ,
file_name : & str ,
parser_hint : Option <& str > ,
input_dir : & Path ,
context : & RequestContext ,
) -> Result < StagedLightRagSource , WebError > {
if lightrag_scan_supported_file ( canonical ) {
let light_rag_file_path = lightrag_symlink_name ( canonical , file_name , parser_hint );
let symlink_path = input_dir . join ( & light_rag_file_path );
replace_lightrag_input_file ( & symlink_path , context ) ? ;
create_symlink ( canonical , & symlink_path , context ) ? ;
return Ok ( StagedLightRagSource {
light_rag_file_path ,
staged_path : symlink_path ,
});
}
let asset_file_path = lightrag_symlink_name ( canonical , file_name , None );
let asset_symlink_path = input_dir . join ( & asset_file_path );
replace_lightrag_input_file ( & asset_symlink_path , context ) ? ;
create_symlink ( canonical , & asset_symlink_path , context ) ? ;
let wrapper_file_path = format! ( " {asset_file_path} .md" );
let wrapper_path = input_dir . join ( & wrapper_file_path );
replace_lightrag_input_file ( & wrapper_path , context ) ? ;
let wrapper_markdown = format! ( "# {file_name} \n\n  \n " );
fs ::write ( & wrapper_path , wrapper_markdown ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_wrapper_write_failed" ,
format! ( "无法写入 LightRAG 图片包装 Markdown: {error} " ),
)
. with_context ( context )
}) ? ;
Ok ( StagedLightRagSource {
light_rag_file_path : wrapper_file_path ,
staged_path : wrapper_path ,
})
}
fn replace_lightrag_input_file ( path : & Path , context : & RequestContext ) -> Result < (), WebError > {
if path . exists () || path . symlink_metadata (). is_ok () {
fs ::remove_file ( path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_symlink_replace_failed" ,
format! ( "无法替换 LightRAG source 文件: {error} " ),
)
. with_context ( context )
}) ? ;
}
Ok (())
}
fn ensure_source_path_in_root (
root_path : & Path ,
source_path : & str ,
context : & RequestContext ,
) -> Result < PathBuf , WebError > {
let raw_path = source_path
. trim ()
. strip_prefix ( "file://" )
. map ( PathBuf ::from )
. unwrap_or_else ( || PathBuf ::from ( source_path . trim ()));
let target = if raw_path . is_absolute () {
raw_path
} else {
root_path . join ( raw_path )
};
let canonical = target . canonicalize (). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_source_unavailable" ,
format! ( "无法访问资料库 source: {error} " ),
)
. with_context ( context )
}) ? ;
if ! canonical . starts_with ( root_path ) {
return Err ( WebError ::bad_request_code (
"knowledge_rag_source_root_escape" ,
"资料库 source 不能越过授权目录" ,
)
. with_context ( context ));
}
Ok ( canonical )
}
fn root_relative_path ( root_path : & Path , source_path : & Path ) -> Result < String , WebError > {
let relative = source_path . strip_prefix ( root_path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_source_relative_failed" ,
format! ( "无法生成资料库 source 相对路径: {error} " ),
)
}) ? ;
Ok ( relative . to_string_lossy (). replace ( '\\' , "/" ))
}
fn create_symlink (
source : & Path ,
symlink_path : & Path ,
context : & RequestContext ,
) -> Result < (), WebError > {
#[cfg(unix)]
{
std ::os ::unix ::fs ::symlink ( source , symlink_path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_symlink_failed" ,
format! ( "无法创建 LightRAG source symlink: {error} " ),
)
. with_context ( context )
})
}
#[cfg(not(unix))]
{
fs ::copy ( source , symlink_path ). map ( | _ | ()). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_staging_copy_failed" ,
format! ( "无法创建 LightRAG source staging 文件: {error} " ),
)
. with_context ( context )
})
}
}
fn source_hash ( path : & Path ) -> Result < String , WebError > {
let bytes = fs ::read ( path ). map_err ( | error | {
WebError ::bad_request_code (
"knowledge_rag_source_hash_failed" ,
format! ( "无法读取资料库 source 以计算 hash: {error} " ),
)
}) ? ;
Ok ( format! ( "mnote-fnv64: {:016x} " , fnv64 ( & bytes )))
}
fn fnv64 ( bytes : & [ u8 ]) -> u64 {
let mut hash = 0xcbf29ce484222325 u64 ;
for byte in bytes {
hash ^= u64 ::from ( * byte );
hash = hash . wrapping_mul ( 0x100000001b3 );
}
hash
}
fn lightrag_symlink_name ( path : & Path , file_name : & str , parser_hint : Option <& str > ) -> String {
let prefix = format! ( "mnote- {} -" , short_hash ( & path . display (). to_string ()));
let Some ( parser_hint ) = parser_hint else {
return format! ( " {prefix}{file_name} " );
};
if strip_one_supported_parser_hint ( file_name ). is_some () {
return format! ( " {prefix}{file_name} " );
}
let path = Path ::new ( file_name );
let stem = path
. file_stem ()
. and_then ( | value | value . to_str ())
. unwrap_or ( file_name );
let extension = path . extension (). and_then ( | value | value . to_str ());
match extension {
Some ( extension ) if ! extension . is_empty () => {
format! ( " {prefix}{stem} .[ {parser_hint} ]. {extension} " )
}
_ => format! ( " {prefix}{stem} .[ {parser_hint} ]" ),
}
}
fn normalize_parser_hint (
value : Option <& str > ,
context : & RequestContext ,
) -> Result < Option < String > , WebError > {
let Some ( value ) = value . map ( str ::trim ). filter ( | value | ! value . is_empty ()) else {
return Ok ( None );
};
2026-06-08 20:35:49 +08:00
let hint = value . trim_matches ( '[' ). trim_matches ( ']' ). trim ();
if let Some ( normalized ) = normalize_supported_parser_hint ( hint ) {
return Ok ( Some ( normalized ));
2026-06-07 01:10:31 +08:00
}
2026-06-08 20:35:49 +08:00
Err ( WebError ::bad_request_code (
"knowledge_rag_parser_hint_invalid" ,
"LightRAG parserHint 仅支持 legacy/native/mineru/docling 及 native-P 这类 engine-options" ,
)
. with_context ( context ))
}
fn default_lightrag_parser_hint_for_source ( path : & Path , file_name : & str ) -> Option < String > {
let extension = path
. extension ()
. and_then ( | value | value . to_str ())
. map ( | value | value . to_ascii_lowercase ());
2026-06-13 22:20:01 +08:00
let large_source = path
. metadata ()
. map ( | metadata | metadata . len () >= LARGE_DOCUMENT_SKIP_KG_MIN_BYTES )
. unwrap_or ( false );
2026-06-08 20:35:49 +08:00
let lower_name = file_name . to_ascii_lowercase ();
2026-06-13 22:20:01 +08:00
let is_ocr_layered_docx = extension . as_deref () == Some ( "docx" )
&& lower_name . contains ( "[ocr]" )
&& lower_name . contains ( ".layered" );
if large_source && matches! ( extension . as_deref (), Some ( "docx" | "pdf" )) {
if is_ocr_layered_docx {
return Some ( "native-P!" . into ());
}
return Some ( "-P!" . into ());
}
if is_ocr_layered_docx {
2026-06-08 20:35:49 +08:00
return Some ( "native-P" . into ());
}
None
}
fn normalize_supported_parser_hint ( value : & str ) -> Option < String > {
let trimmed = value . trim ();
if trimmed . is_empty () {
return None ;
}
if let Some ( options ) = trimmed . strip_prefix ( '-' ) {
return valid_lightrag_parser_options ( options ). then ( || format! ( "- {options} " ));
}
let ( engine , options ) = trimmed
. split_once ( '-' )
. map ( | ( engine , options ) | ( engine , Some ( options )))
. unwrap_or (( trimmed , None ));
let engine = engine . to_ascii_lowercase ();
if ! matches! ( engine . as_str (), "legacy" | "native" | "mineru" | "docling" ) {
return None ;
}
match options {
Some ( options ) if valid_lightrag_parser_options ( options ) => {
Some ( format! ( " {engine} - {options} " ))
}
Some ( _ ) => None ,
None => Some ( engine ),
}
}
fn valid_lightrag_parser_options ( options : & str ) -> bool {
! options . is_empty ()
&& options
. chars ()
2026-06-13 22:20:01 +08:00
. all ( | value | matches! ( value , 'i' | 't' | 'e' | '!' | 'R' | 'F' | 'P' ))
2026-06-07 01:10:31 +08:00
}
fn document_for_registry_entry < 'a > (
by_file_path : & 'a BTreeMap < String , Value > ,
entry : & KnowledgeRagSourceRegistryEntry ,
) -> Option <& 'a Value > {
by_file_path . get ( & entry . light_rag_file_path ). or_else ( || {
strip_one_supported_parser_hint ( & entry . light_rag_file_path )
. as_ref ()
. and_then ( | path | by_file_path . get ( path ))
})
}
fn lightrag_file_path_matches ( entry : & KnowledgeRagSourceRegistryEntry , file_path : & str ) -> bool {
entry . light_rag_file_path == file_path
|| strip_one_supported_parser_hint ( & entry . light_rag_file_path ). as_deref () == Some ( file_path )
}
fn strip_one_supported_parser_hint ( file_name : & str ) -> Option < String > {
2026-06-08 20:35:49 +08:00
let mut search_start = 0 usize ;
while let Some ( offset ) = file_name . get ( search_start .. ) ? . find ( ".[" ) {
let start = search_start + offset ;
let hint_start = start + 2 ;
let Some ( end_offset ) = file_name . get ( hint_start .. ) ? . find ( ']' ) else {
break ;
};
let hint_end = hint_start + end_offset ;
let hint = file_name . get ( hint_start .. hint_end ) ? ;
if normalize_supported_parser_hint ( hint ). is_some () {
let mut stripped = String ::new ();
stripped . push_str ( file_name . get ( .. start ) ? );
stripped . push_str ( file_name . get ( hint_end + 1 .. ) ? );
return Some ( stripped );
2026-06-07 01:10:31 +08:00
}
2026-06-08 20:35:49 +08:00
search_start = hint_end + 1 ;
2026-06-07 01:10:31 +08:00
}
None
}
fn short_hash ( value : & str ) -> String {
let mut hasher = DefaultHasher ::new ();
value . hash ( & mut hasher );
format! ( " {:016x} " , hasher . finish ())
}
2026-06-09 09:20:56 +08:00
fn short_citation_id (
source_id : & str ,
chunk_id : & str ,
block_id : & str ,
occurrence_index : u64 ,
fingerprint : & str ,
) -> String {
short_hash ( & format! (
"lightrag \n {source_id} \n {chunk_id} \n {block_id} \n {occurrence_index} \n {fingerprint} "
))
. chars ()
. take ( 4 )
. collect ()
}
2026-06-07 01:10:31 +08:00
fn lightrag_endpoint () -> String {
env ::var ( "MNOTE_LIGHTRAG_ENDPOINT" )
. or_else ( | _ | env ::var ( "LIGHTRAG_ENDPOINT" ))
. unwrap_or_else ( | _ | DEFAULT_LIGHTRAG_ENDPOINT . into ())
. trim ()
. trim_end_matches ( '/' )
. to_string ()
}
fn lightrag_dashboard_url () -> String {
env ::var ( "MNOTE_LIGHTRAG_DASHBOARD_URL" ). unwrap_or_else ( | _ | lightrag_endpoint ())
}
fn lightrag_input_dir () -> PathBuf {
if let Ok ( value ) = env ::var ( "MNOTE_LIGHTRAG_INPUT_DIR" ) {
return PathBuf ::from ( value );
}
if let Some ( value ) = read_lightrag_dotenv_value ( "INPUT_DIR" ) {
return PathBuf ::from ( value );
}
env ::var ( "INPUT_DIR" )
. map ( PathBuf ::from )
. unwrap_or_else ( | _ | PathBuf ::from ( DEFAULT_LIGHTRAG_INPUT_DIR ))
}
fn lightrag_working_dir () -> PathBuf {
if let Ok ( value ) = env ::var ( "MNOTE_LIGHTRAG_WORKING_DIR" ) {
return PathBuf ::from ( value );
}
if let Some ( value ) = read_lightrag_dotenv_value ( "WORKING_DIR" ) {
return PathBuf ::from ( value );
}
env ::var ( "WORKING_DIR" )
. map ( PathBuf ::from )
. unwrap_or_else ( | _ | PathBuf ::from ( DEFAULT_LIGHTRAG_WORKING_DIR ))
}
fn lightrag_api_key () -> Option < String > {
env ::var ( "MNOTE_LIGHTRAG_API_KEY" )
. ok ()
. map ( | value | value . trim (). trim_matches ( '"' ). to_string ())
. filter ( | value | ! value . is_empty ())
. or_else ( || read_lightrag_dotenv_value ( "LIGHTRAG_API_KEY" ))
. or_else ( || env ::var ( "LIGHTRAG_API_KEY" ). ok ())
. map ( | value | value . trim (). trim_matches ( '"' ). to_string ())
. filter ( | value | ! value . is_empty ())
}
fn read_lightrag_dotenv_value ( key : & str ) -> Option < String > {
let path = env ::var ( "MNOTE_LIGHTRAG_ENV_FILE" )
. map ( PathBuf ::from )
. unwrap_or_else ( | _ | PathBuf ::from ( "/mnt/Data1T/Mnote_data/lightrag/LightRAG/.env" ));
let content = fs ::read_to_string ( path ). ok () ? ;
for line in content . lines () {
let line = line . trim ();
if line . is_empty () || line . starts_with ( '#' ) {
continue ;
}
let Some (( name , value )) = line . split_once ( '=' ) else {
continue ;
};
if name . trim () != key {
continue ;
}
let value = value . trim (). trim_matches ( '"' ). to_string ();
if ! value . is_empty () {
return Some ( value );
}
}
None
}
fn effective_workspace_id ( input : Option <& str > , root_uri : & str ) -> String {
input
. map ( str ::trim )
. filter ( | value | ! value . is_empty ())
. map ( ToOwned ::to_owned )
. unwrap_or_else ( || {
local_folder_source ::local_workspace_id_from_root_uri ( root_uri )
. unwrap_or_else ( | _ | format! ( "local-ws: {} " , short_hash ( root_uri )))
})
}
fn now_ms () -> u128 {
SystemTime ::now ()
. duration_since ( UNIX_EPOCH )
. map ( | value | value . as_millis ())
. unwrap_or_default ()
}
#[cfg(test)]
mod tests {
use super ::* ;
use axum ::http ::{ HeaderMap , Method , Uri };
use std ::sync ::{ Mutex , OnceLock };
fn env_lock () -> & 'static Mutex < () > {
static LOCK : OnceLock < Mutex < () >> = OnceLock ::new ();
LOCK . get_or_init ( || Mutex ::new (()))
}
fn test_context () -> RequestContext {
RequestContext ::from_http_parts ( & Method ::GET , & Uri ::from_static ( "/" ), & HeaderMap ::new ())
}
fn temp_root ( label : & str ) -> PathBuf {
let root = std ::env ::temp_dir (). join ( format! ( " {label} - {} " , now_ms ()));
fs ::create_dir_all ( & root ). expect ( "temp root" );
root
}
fn test_registry_entry (
root : & Path ,
relative : & str ,
doc_id : Option <& str > ,
provider_status : Option <& str > ,
indexed_at_ms : Option < u128 > ,
deleted_at_ms : Option < u128 > ,
stale : bool ,
) -> KnowledgeRagSourceRegistryEntry {
KnowledgeRagSourceRegistryEntry {
source_id : format ! ( "src-{relative}" ),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
source_path : root . join ( relative ). display (). to_string (),
source_root_relative_path : relative . into (),
source_hash : "mnote-fnv64:test" . into (),
light_rag_doc_id : doc_id . map ( ToOwned ::to_owned ),
light_rag_status : provider_status . map ( ToOwned ::to_owned ),
light_rag_file_path : relative . into (),
symlink_path : format ! ( "/tmp/{relative}" ),
parser_hint : None ,
indexed_at_ms ,
deleted_at_ms ,
stale ,
updated_at_ms : 2 ,
}
}
#[tokio::test]
async fn retired_local_ocr_endpoint_points_to_lightrag_replacements () {
let ( status , Json ( payload )) = retired_local_ocr_endpoint ( Extension ( test_context ())). await ;
assert_eq! ( status , StatusCode ::GONE );
assert_eq! ( payload [ "code" ], "mnote_local_ocr_retired" );
assert_eq! ( payload [ "replacement" ][ "provider" ], "lightrag" );
assert_eq! (
payload [ "replacement" ][ "ingest" ],
"/api/knowledge-rag/ingest"
);
}
#[test]
fn lightrag_document_summaries_expose_status_groups_for_ui_sync () {
let docs = json! ({
"statuses" : {
"processed" : [
{ "id" : "doc-ok" , "file_path" : "ok.pdf" , "status" : "processed" , "summary" : "ok" }
],
"failed" : [
{ "id" : "doc-fail" , "file_path" : "fail.pdf" , "status" : "failed" , "summary" : "bad" }
]
}
});
let counts = lightrag_document_status_group_counts ( & docs );
assert_eq! ( counts . get ( "processed" ), Some ( & 1 ));
assert_eq! ( counts . get ( "failed" ), Some ( & 1 ));
let summaries = lightrag_document_summaries ( & docs );
assert_eq! ( summaries . len (), 2 );
assert! ( summaries . iter (). any ( | item | item [ "filePath" ] == "ok.pdf" ));
assert! ( summaries . iter (). any ( | item | item [ "statusGroup" ] == "failed" ));
let by_path = lightrag_documents_by_file_path ( & docs );
assert! ( by_path . contains_key ( "ok.pdf" ));
assert! ( by_path . contains_key ( "fail.pdf" ));
}
2026-06-08 20:35:49 +08:00
#[test]
fn lightrag_pipeline_summary_extracts_chunk_progress_for_ui () {
let status = json! ({
"busy" : true ,
"scanning" : true ,
2026-06-09 22:12:09 +08:00
"destructive_busy" : false ,
"pending_enqueues" : 1 ,
"docs" : 1 ,
"batchs" : 2 ,
"cur_batch" : 1 ,
2026-06-08 20:35:49 +08:00
"job_name" : "book.docx" ,
"latest_message" : "Chunk 212 of 333 extracted 4 Ent + 0 Rel doc-ff0b60997a285a85e5704a114d7b3ffa-chunk-212"
});
let summary = lightrag_pipeline_status_summary ( & status );
assert_eq! ( summary [ "busy" ], true );
assert_eq! ( summary [ "scanning" ], true );
2026-06-09 22:12:09 +08:00
assert_eq! ( summary [ "pendingEnqueues" ], 1 );
assert_eq! ( summary [ "docs" ], 1 );
assert_eq! ( summary [ "batches" ], 2 );
assert_eq! ( summary [ "currentBatch" ], 1 );
2026-06-08 20:35:49 +08:00
assert_eq! ( summary [ "progress" ][ "current" ], 212 );
assert_eq! ( summary [ "progress" ][ "total" ], 333 );
assert_eq! (
summary [ "progress" ][ "docId" ],
"doc-ff0b60997a285a85e5704a114d7b3ffa"
);
}
2026-06-09 22:12:09 +08:00
#[test]
fn lightrag_pipeline_summary_extracts_chunk_progress_from_history () {
let status = json! ({
"busy" : true ,
"scanning" : false ,
"job_name" : "book.pdf" ,
"latest_message" : "Merging stage 1/1: book.pdf" ,
"history_messages" : [
"Analyzing multimodal: doc-e51400269a2c41a2fcb22f1071193fad" ,
"Chunk 5 of 21 extracted 2 Ent + 1 Rel doc-e51400269a2c41a2fcb22f1071193fad-mm-drawing-001" ,
"Merging stage 1/1: book.pdf"
]
});
let summary = lightrag_pipeline_status_summary ( & status );
assert_eq! ( summary [ "progress" ][ "current" ], 5 );
assert_eq! ( summary [ "progress" ][ "total" ], 21 );
assert_eq! (
summary [ "progress" ][ "docId" ],
"doc-e51400269a2c41a2fcb22f1071193fad"
);
assert_eq! ( summary [ "historyMessages" ]. as_array (). unwrap (). len (), 3 );
}
2026-06-08 20:35:49 +08:00
#[test]
fn old_registry_json_defaults_indexed_roots () {
let registry : KnowledgeRagSourceRegistry = serde_json ::from_value ( json! ({
"schema" : REGISTRY_SCHEMA ,
"workspaceId" : "ws" ,
"rootUri" : "file:///tmp/root" ,
"updatedAtMs" : 1 ,
"entries" : []
}))
. expect ( "registry" );
assert! ( registry . indexed_roots . is_empty ());
}
#[test]
fn ingest_request_records_explicit_indexed_root_scope () {
let root = temp_root ( "mnote-knowledge-rag-indexed-root" );
fs ::create_dir_all ( root . join ( "papers" )). expect ( "papers" );
fs ::write ( root . join ( "papers" ). join ( "a.pdf" ), b "pdf" ). expect ( "pdf" );
let mut registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : Vec ::new (),
};
upsert_indexed_root_for_request ( & mut registry , & root , "papers" , & test_context ())
. expect ( "indexed root" );
assert_eq! ( registry . indexed_roots . len (), 1 );
assert_eq! ( registry . indexed_roots [ 0 ]. root_relative_path , "papers" );
assert! ( registry . indexed_roots [ 0 ]. recursive );
assert_eq! ( registry . indexed_roots [ 0 ]. run_on_change , Some ( true ));
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn references_only_search_result_is_clickable_source_hit () {
let reference = json! ({
"provider" : "lightrag" ,
"sourceId" : "src-paper" ,
"sourceRootRelativePath" : "papers/a.pdf" ,
"chunkId" : "chunk-1" ,
"quote" : "Canterbury corpus compression ratio appears in this paragraph." ,
"locatorDegraded" : false ,
"locator" : {
"rootUri" : "file:///tmp/root" ,
"ownerDocumentId" : "local-md:papers~2FHost.md" ,
"resourcePath" : "papers/a.pdf" ,
"resourceKind" : "pdf" ,
"page" : 3
},
"citationUrl" : "/documents/local-md:papers~2FHost.md?resourcePath=papers%2Fa.pdf" ,
"citationMarkdown" : "[a.pdf · p.3](/documents/local-md:papers~2FHost.md)"
});
let result = knowledge_rag_search_result ( & reference , 0 , "file:///tmp/root" , "三甲基硅基" );
assert_eq! ( result [ "provider" ], "lightrag" );
assert_eq! ( result [ "query" ], "三甲基硅基" );
assert_eq! ( result [ "matchSource" ], "lightrag_reference" );
assert_eq! ( result [ "resourceType" ], "pdf" );
assert_eq! ( result [ "path" ], "papers/a.pdf" );
assert_eq! ( result [ "locator" ][ "page" ], 3 );
assert_eq! ( result [ "citationUrl" ], reference [ "citationUrl" ]);
}
2026-06-07 01:10:31 +08:00
#[test]
fn reference_mapping_uses_registry_source_path () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/books/a.pdf" . into (),
source_root_relative_path : "books/a.pdf" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-a.pdf" . into (),
symlink_path : "/tmp/input/mnote-hash-a.pdf" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let mapped = map_reference_plan (
& json! ({ "file_path" :"mnote-hash-a.pdf" , "chunk_id" :"chunk1" }),
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 01:10:31 +08:00
);
assert_eq! ( mapped [ "sourceRootRelativePath" ], "books/a.pdf" );
assert_eq! ( mapped [ "locatorDegraded" ], true );
assert_eq! ( mapped [ "openAction" ][ "params" ][ "path" ], "books/a.pdf" );
}
#[test]
fn missing_registry_reference_is_degraded () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [],
};
let mapped = map_reference_plan (
& json! ({ "file_path" :"unknown.pdf" }),
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 01:10:31 +08:00
);
assert_eq! ( mapped [ "locatorDegraded" ], true );
assert! ( mapped [ "citationMarkdown" ]
. as_str ()
. unwrap ()
. contains ( "来源定位降级" ));
}
#[test]
fn reference_mapping_does_not_match_deleted_entry_when_doc_id_missing () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "deleted" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/deleted.md" . into (),
source_root_relative_path : "deleted.md" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : None ,
light_rag_status : None ,
light_rag_file_path : "deleted.md" . into (),
symlink_path : "/tmp/input/deleted.md" . into (),
parser_hint : None ,
indexed_at_ms : None ,
deleted_at_ms : Some ( 2 ),
stale : true ,
updated_at_ms : 2 ,
}],
};
let mapped = map_reference_plan (
& json! ({ "file_path" :"active.md" , "chunk_id" :"chunk-active" }),
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 01:10:31 +08:00
);
assert! ( mapped [ "sourceRootRelativePath" ]. is_null ());
assert_eq! ( mapped [ "stale" ], false );
assert_eq! ( mapped [ "deleted" ], false );
}
#[test]
fn degraded_known_resource_keeps_clickable_resource_tab_url () {
let root = temp_root ( "mnote-knowledge-rag-docx-fallback" );
fs ::create_dir_all ( root . join ( "docs" )). expect ( "docs" );
fs ::write ( root . join ( "docs" ). join ( "Host.md" ), "# Host \n " ). expect ( "host" );
fs ::write ( root . join ( "docs" ). join ( "a.docx" ), b "docx" ). expect ( "docx" );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : root . join ( "docs" ). join ( "a.docx" ). display (). to_string (),
source_root_relative_path : "docs/a.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-a.docx" . into (),
symlink_path : "/tmp/input/mnote-hash-a.docx" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let mapped = map_reference_plan (
& json! ({ "file_path" :"mnote-hash-a.docx" , "chunk_id" :"chunk1" }),
& registry ,
"file:///tmp/root" ,
& root ,
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 01:10:31 +08:00
);
assert_eq! ( mapped [ "locatorDegraded" ], true );
let citation_url = mapped [ "citationUrl" ]. as_str (). expect ( "citation url" );
assert! ( citation_url . starts_with ( "/documents/local-md:docs~2FHost.md?" ));
assert! ( citation_url . contains ( "resourceTab=" ));
assert! ( citation_url . contains ( "resourcePath=docs%2Fa.docx" ));
assert! ( mapped [ "citationMarkdown" ]
. as_str ()
. unwrap ()
. contains ( "来源定位降级" ));
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn source_state_marks_deleted_and_changed_entries_stale () {
let root = temp_root ( "mnote-knowledge-rag-source-state" );
fs ::create_dir_all ( root . join ( "docs" )). expect ( "docs" );
let changed_path = root . join ( "docs" ). join ( "changed.md" );
fs ::write ( & changed_path , "new content" ). expect ( "changed" );
let missing_path = root . join ( "docs" ). join ( "missing.md" );
let mut registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [
KnowledgeRagSourceRegistryEntry {
source_id : "missing" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : missing_path . display (). to_string (),
source_root_relative_path : "docs/missing.md" . into (),
source_hash : "mnote-fnv64:old" . into (),
light_rag_doc_id : Some ( "doc-missing" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "missing.md" . into (),
symlink_path : "/tmp/input/missing.md" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
},
KnowledgeRagSourceRegistryEntry {
source_id : "changed" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : changed_path . display (). to_string (),
source_root_relative_path : "docs/changed.md" . into (),
source_hash : "mnote-fnv64:old" . into (),
light_rag_doc_id : Some ( "doc-changed" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "changed.md" . into (),
symlink_path : "/tmp/input/changed.md" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
},
],
};
let stale_doc_ids = sync_registry_source_state ( & mut registry , 42 ). expect ( "sync" );
assert_eq! ( stale_doc_ids , vec! [ "doc-changed" , "doc-missing" ]);
assert_eq! ( registry . entries [ 0 ]. deleted_at_ms , Some ( 42 ));
2026-06-07 10:35:21 +08:00
assert_eq! (
registry . entries [ 0 ]. light_rag_doc_id . as_deref (),
Some ( "doc-missing" )
);
assert_eq! (
registry . entries [ 0 ]. light_rag_status . as_deref (),
Some ( "delete_submitted" )
);
2026-06-07 01:10:31 +08:00
assert_eq! ( registry . entries [ 0 ]. indexed_at_ms , None );
assert! ( registry . entries [ 0 ]. stale );
assert_eq! ( registry . entries [ 1 ]. deleted_at_ms , None );
2026-06-07 10:35:21 +08:00
assert_eq! (
registry . entries [ 1 ]. light_rag_doc_id . as_deref (),
Some ( "doc-changed" )
);
assert_eq! (
registry . entries [ 1 ]. light_rag_status . as_deref (),
Some ( "delete_submitted" )
);
2026-06-07 01:10:31 +08:00
assert_eq! ( registry . entries [ 1 ]. indexed_at_ms , None );
assert! ( registry . entries [ 1 ]. stale );
assert_ne! ( registry . entries [ 1 ]. source_hash , "mnote-fnv64:old" );
let _ = fs ::remove_dir_all ( root );
}
2026-06-07 10:35:21 +08:00
#[test]
fn provider_delete_confirmation_clears_doc_id_for_deleted_or_stale_entries () {
let root = temp_root ( "mnote-knowledge-rag-delete-confirmed" );
let mut deleted = test_registry_entry (
& root ,
"deleted.pdf" ,
Some ( "doc-deleted" ),
Some ( "delete_submitted" ),
Some ( 2 ),
Some ( 3 ),
true ,
);
let mut changed = test_registry_entry (
& root ,
"changed.pdf" ,
Some ( "doc-changed" ),
Some ( "delete_submitted" ),
Some ( 2 ),
None ,
true ,
);
let active = test_registry_entry (
& root ,
"active.pdf" ,
Some ( "doc-active" ),
Some ( "processed" ),
Some ( 2 ),
None ,
false ,
);
assert! ( knowledge_rag_provider_delete_confirmed ( & deleted ));
mark_registry_entry_delete_completed ( & mut deleted , 42 );
assert_eq! ( deleted . light_rag_doc_id , None );
assert_eq! ( deleted . indexed_at_ms , None );
assert_eq! (
deleted . light_rag_status . as_deref (),
Some ( "delete_completed" )
);
assert! ( knowledge_rag_provider_delete_confirmed ( & changed ));
mark_registry_entry_delete_completed ( & mut changed , 43 );
assert_eq! ( changed . light_rag_doc_id , None );
assert_eq! (
changed . light_rag_status . as_deref (),
Some ( "delete_completed" )
);
assert! ( ! knowledge_rag_provider_delete_confirmed ( & active ));
assert_eq! ( active . light_rag_doc_id . as_deref (), Some ( "doc-active" ));
let _ = fs ::remove_dir_all ( root );
}
2026-06-07 01:10:31 +08:00
#[test]
fn source_statuses_distinguish_indexed_processing_failed_and_deleted () {
let root = temp_root ( "mnote-knowledge-rag-source-statuses" );
fs ::write ( root . join ( "indexed.pdf" ), b "indexed" ). expect ( "indexed" );
fs ::write ( root . join ( "submitted.pdf" ), b "submitted" ). expect ( "submitted" );
fs ::write ( root . join ( "failed.pdf" ), b "failed" ). expect ( "failed" );
fs ::write ( root . join ( "deleting.pdf" ), b "deleting" ). expect ( "deleting" );
fs ::write ( root . join ( "removed.pdf" ), b "removed" ). expect ( "removed" );
let root_uri = format! ( "file:// {} " , root . display ());
let mut registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : root_uri . clone (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [
test_registry_entry (
& root ,
"indexed.pdf" ,
Some ( "doc-indexed" ),
Some ( "processed" ),
Some ( 2 ),
None ,
false ,
),
test_registry_entry (
& root ,
"submitted.pdf" ,
None ,
Some ( "submitted" ),
None ,
None ,
false ,
),
test_registry_entry ( & root , "failed.pdf" , None , Some ( "failed" ), None , None , false ),
test_registry_entry (
& root ,
"deleting.pdf" ,
Some ( "doc-deleting" ),
Some ( "delete_submitted" ),
Some ( 2 ),
Some ( 3 ),
true ,
),
test_registry_entry (
& root ,
"removed.pdf" ,
None ,
Some ( "delete_completed" ),
None ,
Some ( 3 ),
true ,
),
],
};
write_registry ( & root , & mut registry ). expect ( "write registry" );
let statuses = knowledge_rag_source_statuses ( & root , "ws" , & root_uri ). expect ( "statuses" );
assert! ( statuses . indexed_paths . contains ( "indexed.pdf" ));
assert! ( statuses . indexing_paths . contains ( "submitted.pdf" ));
assert! ( statuses . indexing_paths . contains ( "deleting.pdf" ));
assert! ( statuses . failed_paths . contains ( "failed.pdf" ));
assert! ( ! statuses . indexed_paths . contains ( "removed.pdf" ));
assert! ( ! statuses . indexing_paths . contains ( "removed.pdf" ));
assert! ( ! statuses . failed_paths . contains ( "removed.pdf" ));
let _ = fs ::remove_dir_all ( root );
}
#[test]
2026-06-08 20:35:49 +08:00
fn image_source_is_staged_directly_for_lightrag_scan () {
let root = temp_root ( "mnote-knowledge-rag-image-direct-scan" );
2026-06-07 01:10:31 +08:00
let input_dir = root . join ( "inputs" );
fs ::create_dir_all ( & input_dir ). expect ( "input dir" );
let image_path = root . join ( "image copy 6.png" );
fs ::write ( & image_path , b "png" ). expect ( "image" );
let staged = stage_lightrag_source (
& image_path ,
"image copy 6.png" ,
None ,
& input_dir ,
& test_context (),
)
. expect ( "stage image" );
2026-06-08 20:35:49 +08:00
assert! ( staged . light_rag_file_path . ends_with ( ".png" ));
assert! ( staged . staged_path . symlink_metadata (). is_ok ());
#[cfg(unix)]
assert! ( staged
. staged_path
. symlink_metadata ()
. expect ( "staged metadata" )
. file_type ()
. is_symlink ());
2026-06-07 01:10:31 +08:00
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn registry_prune_predicate_keeps_inflight_delete_until_confirmed () {
let root = temp_root ( "mnote-knowledge-rag-prune-predicate" );
let active = test_registry_entry (
& root ,
"active.pdf" ,
Some ( "doc-active" ),
Some ( "processed" ),
Some ( 2 ),
None ,
false ,
);
let deleting = test_registry_entry (
& root ,
"deleting.pdf" ,
Some ( "doc-deleting" ),
Some ( "delete_submitted" ),
Some ( 2 ),
Some ( 3 ),
true ,
);
let removed = test_registry_entry (
& root ,
"removed.pdf" ,
None ,
Some ( "delete_completed" ),
None ,
Some ( 3 ),
true ,
);
let failed =
test_registry_entry ( & root , "failed.pdf" , None , Some ( "failed" ), None , None , false );
2026-06-07 10:35:21 +08:00
let retry = test_registry_entry (
& root ,
"retry.pdf" ,
Some ( "doc-retry" ),
Some ( "delete_retry_required" ),
Some ( 2 ),
Some ( 3 ),
true ,
);
let stale_with_doc = test_registry_entry (
& root ,
"stale.pdf" ,
Some ( "doc-stale" ),
Some ( "processed" ),
Some ( 2 ),
None ,
true ,
);
2026-06-07 01:10:31 +08:00
assert! ( ! knowledge_rag_registry_entry_prunable ( & active ));
assert! ( ! knowledge_rag_registry_entry_prunable ( & deleting ));
2026-06-07 10:35:21 +08:00
assert! ( ! knowledge_rag_registry_entry_prunable ( & retry ));
assert! ( ! knowledge_rag_registry_entry_prunable ( & stale_with_doc ));
2026-06-07 01:10:31 +08:00
assert! ( knowledge_rag_registry_entry_prunable ( & removed ));
assert! ( knowledge_rag_registry_entry_prunable ( & failed ));
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn source_scope_filters_mapped_references_by_file_or_directory () {
let mut references = vec! [
json! ({ "sourceRootRelativePath" : "books/a.pdf" }),
json! ({ "sourceRootRelativePath" : "papers/one/b.pdf" }),
json! ({ "sourceRootRelativePath" : "papers/two/c.pdf" }),
];
let scope = normalize_source_scope ( Some ( & [
"books/a.pdf" . to_string (),
"papers/one" . to_string (),
"papers/one" . to_string (),
]));
assert_eq! ( scope , vec! [ "books/a.pdf" , "papers/one" ]);
filter_mapped_references_by_source_scope ( & mut references , & scope );
let paths = references
. iter ()
. filter_map ( | reference | reference [ "sourceRootRelativePath" ]. as_str ())
. collect ::< Vec < _ >> ();
assert_eq! ( paths , vec! [ "books/a.pdf" , "papers/one/b.pdf" ]);
}
2026-06-08 20:35:49 +08:00
#[test]
fn search_result_reports_paragraph_locator_precision_without_bbox () {
let reference = json! ({
"sourceRootRelativePath" : "docs/book.docx" ,
"sourceId" : "src1" ,
"chunkId" : "chunk-1" ,
2026-06-09 09:20:56 +08:00
"quote" : "raw should not be used" ,
"displayQuote" : "吡咯烷,5 h,90%" ,
"locatorEvidenceText" : "如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。" ,
"citationId" : "a1b2" ,
"normalizedFingerprint" : "吡咯烷5h90" ,
2026-06-08 20:35:49 +08:00
"locator" : {
"blockId" : "block-pyrrolidine" ,
"openAction" : {
"params" : {
"evidenceText" : "吡咯烷,5 h,90%"
}
}
},
"locatorDegraded" : true
});
let result = knowledge_rag_search_result ( & reference , 0 , "file:///tmp/root" , "吡咯烷" );
assert_eq! ( result [ "locatorPrecision" ], "paragraph" );
assert_eq! ( result [ "locatorDegraded" ], true );
2026-06-09 09:20:56 +08:00
assert_eq! ( result [ "snippet" ], "吡咯烷,5 h,90%" );
assert_eq! ( result [ "citationId" ], "a1b2" );
assert_eq! (
result [ "locatorEvidenceText" ],
"如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。"
);
}
#[test]
fn citation_text_bundle_strips_equation_and_drawing_for_display () {
let bundle = citation_text_bundle (
Some (
r #"三乙基硅 <equation format="latex">{\left( {C}_{2}{H}_{5} \right)}_{3}</equation> <drawing id="im1" format="jpeg" /> </e"# ,
),
Some ( "三乙基硅" ),
None ,
"chunk" ,
);
assert! ( bundle . display_quote . contains ( "三乙基硅" ));
assert! ( ! bundle . display_quote . contains ( "equation" ));
assert! ( ! bundle . display_quote . contains ( "latex" ));
assert! ( ! bundle . display_quote . contains ( "drawing" ));
assert! ( ! bundle . display_quote . contains ( "</e" ));
assert! ( bundle . display_cleaned );
}
#[test]
fn citation_text_bundle_preserves_query_for_locator () {
let raw = "前置无关内容 \n 吡咯烷,5 h, 90% \n 如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。 \n 参考文献" ;
let bundle = citation_text_bundle ( Some ( raw ), Some ( "吡咯烷" ), Some ( raw ), "sidecar" );
assert! ( bundle . locator_evidence_text . contains ( "吡咯烷" ));
assert! ( bundle . locator_evidence_text . contains ( "BOC 亚乙基" ));
assert_eq! ( bundle . search_query , "吡咯烷" );
assert_eq! ( bundle . locator_text_source , "sidecar_block" );
assert! ( ! bundle . normalized_fingerprint . is_empty ());
}
#[test]
fn knowledge_rag_citation_contains_unified_display_and_locator_text () {
let reference = json! ({
"provider" : "lightrag" ,
"citationId" : "c0de" ,
"sourceId" : "src1" ,
"sourceRootRelativePath" : "docs/book.docx" ,
"filePath" : "book.docx" ,
"chunkId" : "chunk-1" ,
"displayQuote" : "吡咯烷,5 h,90%" ,
"rawQuote" : "<equation>raw</equation>" ,
"locatorEvidenceText" : "如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。" ,
"searchQuery" : "吡咯烷" ,
"locatorPrecision" : "paragraph" ,
"locatorDegraded" : true ,
"locator" : { "blockId" : "block-pyrrolidine" },
"citationUrl" : "/documents/local-md:docs~2FHost.md" ,
"citationMarkdown" : "[book.docx](/documents/local-md:docs~2FHost.md)" ,
"citationDiagnostics" : { "displayCleaned" : true }
});
let citation = knowledge_rag_citation ( & reference , 0 ). expect ( "citation" );
assert_eq! ( citation [ "citationId" ], "c0de" );
assert_eq! ( citation [ "displayQuote" ], "吡咯烷,5 h,90%" );
assert_eq! (
citation [ "locatorEvidenceText" ],
"如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。"
);
assert_eq! ( citation [ "locatorPrecision" ], "paragraph" );
}
#[test]
fn heading_path_for_sidecar_block_strips_lightrag_markup () {
let block = json! ({
"parent_headings" : [ "保护" , "<equation format= \" latex \" >{R}_{2}</equation>" ],
"heading" : "二乙基胺加成物: <equation format= \" latex \" >{NEt}_{2}</equation>"
});
let heading_path = heading_path_for_sidecar_block ( Some ( & block ));
assert_eq! ( heading_path [ 0 ], "保护" );
assert! ( ! heading_path . join ( " " ). contains ( "equation" ));
assert! ( ! heading_path . join ( " " ). contains ( "latex" ));
assert! ( heading_path . join ( " " ). contains ( "二乙基胺加成物" ));
}
#[test]
fn provider_rerank_disabled_is_reported_in_status () {
let health = json! ({
"configuration" : {
"enable_rerank" : false ,
"rerank_binding" : "null" ,
"rerank_model" : Value ::Null ,
"min_rerank_score" : 0.0
},
"rerank_queue_status" : {
"available" : false
}
});
let status = lightrag_rerank_status_summary ( & health );
assert_eq! ( status [ "enabled" ], false );
assert_eq! ( status [ "available" ], false );
assert_eq! ( status [ "providerRerankEnabled" ], false );
assert_eq! ( status [ "providerRerankAvailable" ], false );
assert_eq! ( status [ "status" ], "disabled" );
assert_eq! ( status [ "implementation" ], "provider_status_only" );
2026-06-08 20:35:49 +08:00
}
#[test]
fn lightrag_search_dedupes_same_paragraph_block () {
let mut references = vec! [
json! ({
"sourceRootRelativePath" : "docs/book.docx" ,
"chunkId" : "chunk-056#match-0" ,
"matchSource" : "lightrag_search" ,
"quote" : "吡咯烷,5 h,90%" ,
"locator" : { "blockId" : "same-block" }
}),
json! ({
"sourceRootRelativePath" : "docs/book.docx" ,
"chunkId" : "chunk-056#match-1" ,
"matchSource" : "lightrag_search" ,
"quote" : "吡咯烷,5 h,90%" ,
"locator" : { "blockId" : "same-block" }
}),
json! ({
"sourceRootRelativePath" : "docs/book.docx" ,
"chunkId" : "chunk-057#match-0" ,
"matchSource" : "lightrag_search" ,
"quote" : "另一段吡咯烷" ,
"locator" : { "blockId" : "other-block" }
}),
];
dedupe_mapped_references_by_locator ( & mut references );
assert_eq! ( references . len (), 2 );
assert_eq! ( references [ 0 ][ "locator" ][ "blockId" ], "same-block" );
assert_eq! ( references [ 1 ][ "locator" ][ "blockId" ], "other-block" );
}
2026-06-07 01:10:31 +08:00
#[test]
fn chunks_without_references_become_reference_candidates () {
let raw = json! ({
"data" : {
"chunks" : [
{ "file_path" : "a.md" , "reference_id" : "1" , "chunk_id" : "chunk-a" , "content" : "A" },
{ "file_path" : "b.md" , "reference_id" : "2" , "chunk_id" : "chunk-b" , "content" : "B" }
]
}
});
let references = reference_array ( & raw );
assert_eq! ( references . len (), 2 );
assert_eq! ( references [ 0 ][ "file_path" ]. as_str (), Some ( "a.md" ));
assert! ( references [ 0 ][ "chunks" ]
. as_array ()
. is_some_and ( | items | items . len () == 1 ));
}
#[test]
fn mapped_references_include_chunk_quote () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 01:10:31 +08:00
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/papers/a.pdf" . into (),
source_root_relative_path : "papers/a.pdf" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-a.pdf" . into (),
symlink_path : "/tmp/input/mnote-hash-a.pdf" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let raw = json! ({
"data" : {
"references" : [{ "reference_id" :"1" , "file_path" :"mnote-hash-a.pdf" }],
"chunks" : [{
"reference_id" :"1" ,
"chunk_id" :"doc1-chunk-000" ,
"file_path" :"mnote-hash-a.pdf" ,
"content" :"chunk quote from LightRAG"
}]
}
});
2026-06-08 20:35:49 +08:00
let mapped = mapped_references (
& raw ,
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
None ,
);
2026-06-07 01:10:31 +08:00
assert_eq! ( mapped [ 0 ][ "chunkId" ], "doc1-chunk-000" );
assert_eq! ( mapped [ 0 ][ "quote" ], "chunk quote from LightRAG" );
2026-06-08 20:35:49 +08:00
assert_eq! ( mapped [ 0 ][ "quoteSource" ], "chunk" );
assert_eq! (
mapped [ 0 ][ "contentDiagnostics" ][ "ocrTextExposed" ]. as_bool (),
Some ( true )
);
2026-06-07 01:10:31 +08:00
assert_eq! (
mapped [ 0 ][ "openAction" ][ "params" ][ "chunkId" ],
"doc1-chunk-000"
);
}
2026-06-08 20:35:49 +08:00
#[test]
fn mapped_references_expand_lightrag_chunks_in_same_source () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/books/a.docx" . into (),
source_root_relative_path : "books/a.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-a.docx" . into (),
symlink_path : "/tmp/input/mnote-hash-a.docx" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let raw = json! ({
"data" : {
"references" : [{ "reference_id" :"1" , "file_path" :"mnote-hash-a.docx" }],
"chunks" : [
{
"reference_id" :"1" ,
"chunk_id" :"doc1-chunk-001" ,
"file_path" :"mnote-hash-a.docx" ,
"content" :"第一处 三甲基硅酯 内容"
},
{
"reference_id" :"1" ,
"chunk_id" :"doc1-chunk-002" ,
"file_path" :"mnote-hash-a.docx" ,
"content" :"第二处 三甲基硅基 内容"
}
]
}
});
let mut mapped = mapped_references (
& raw ,
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
Some ( "三甲基硅" ),
);
filter_mapped_references_by_search_query ( & mut mapped , "三甲基硅" );
rank_mapped_references_for_query ( & mut mapped , "三甲基硅" );
assert_eq! ( mapped . len (), 2 );
assert_eq! ( mapped [ 0 ][ "chunkId" ], "doc1-chunk-001" );
assert_eq! ( mapped [ 1 ][ "chunkId" ], "doc1-chunk-002" );
assert! ( mapped [ 0 ][ "quote" ]. as_str (). unwrap (). contains ( "三甲基硅" ));
assert! ( mapped [ 1 ][ "quote" ]. as_str (). unwrap (). contains ( "三甲基硅" ));
}
#[test]
fn search_query_filter_drops_unrelated_semantic_reference () {
let mut references = vec! [
json! ({
"sourceRootRelativePath" : "books/protecting-groups.docx" ,
"quote" : "# 三乙基硅酯(TES): RCOOSi"
}),
json! ({
"sourceRootRelativePath" : "images/image copy 6.png" ,
"quote" : "CodePilot Bridge 处理过程截图,包含终端输出和飞书回复"
}),
];
filter_mapped_references_by_search_query ( & mut references , "三乙基硅酯" );
assert_eq! ( references . len (), 1 );
assert_eq! (
references [ 0 ][ "sourceRootRelativePath" ]. as_str (),
Some ( "books/protecting-groups.docx" )
);
}
2026-06-09 09:20:56 +08:00
#[test]
fn search_query_filter_keeps_cjk_natural_language_keyword_match () {
let mut references = vec! [
json! ({
"sourceRootRelativePath" : "books/protecting-groups.docx" ,
"quote" : "在此过程中吗啉被用作烯丙基的清除剂。"
}),
json! ({
"sourceRootRelativePath" : "books/other.docx" ,
"quote" : "保护基资料包含用途说明,但没有目标化合物。"
}),
];
filter_mapped_references_by_search_query ( & mut references , "吗啉 用途" );
assert_eq! ( references . len (), 1 );
assert! ( references [ 0 ][ "quote" ]
. as_str ()
. unwrap_or_default ()
. contains ( "吗啉" ));
}
2026-06-08 20:35:49 +08:00
#[test]
fn mapped_image_placeholder_reference_marks_ocr_text_missing () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/docs/image.png" . into (),
source_root_relative_path : "docs/image.png" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-image.png.md" . into (),
symlink_path : "/tmp/input/mnote-hash-image.png.md" . into (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let mapped = map_reference_plan (
& json! ({
"file_path" : "mnote-hash-image.png.md" ,
"chunk_id" : "doc1-chunk-000" ,
"chunks" : [{
"chunk_id" : "doc1-chunk-000" ,
"content" : "# image.png \n\n "
}]
}),
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
None ,
);
assert_eq! (
mapped [ "contentDiagnostics" ][ "quoteOnlyImagePlaceholder" ]. as_bool (),
Some ( true )
);
assert_eq! (
mapped [ "contentDiagnostics" ][ "ocrTextExposed" ]. as_bool (),
Some ( false )
);
assert_eq! ( mapped [ "quoteSource" ], "chunk" );
}
#[test]
fn mapped_reference_uses_chunk_query_window_for_late_cjk_match () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-late-cjk-query" );
fs ::create_dir_all ( root . join ( "docs" )). expect ( "docs" );
fs ::write ( root . join ( "docs" ). join ( "book.docx" ), b "docx" ). expect ( "docx" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir . join ( "__parsed__" ). join ( "book.docx.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "book.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":2}"# ,
r ##"{"type":"content","blockid":"wrong-index","content":"# 5.1.8 取代苄酯, 775 三苯甲基,775 二(邻硝基苯基)甲基,779","positions":[{"type":"bbox","anchor":"1","range":[1.0,2.0,3.0,4.0]}]}"## ,
r ##"{"type":"content","blockid":"tes","content":"# 三乙基硅酯(TES): RCOOSi <equation format=\"latex\">{\\left( {C}_{2}{H}_{5} \\right)}_{3}</equation>","positions":[{"type":"bbox","anchor":"2","range":[10.0,20.0,30.0,40.0]}]}"## ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : root . join ( "docs" ). join ( "book.docx" ). display (). to_string (),
source_root_relative_path : "docs/book.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "book.docx" . into (),
symlink_path : input_dir . join ( "book.docx" ). display (). to_string (),
parser_hint : Some ( "native-P" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let chunk_content = format! (
" {} \n # 三乙基硅酯(TES): RCOOSi <equation format= \" latex \" > {{\\ left( {{ C }} _ {{ 2 }}{{ H }} _ {{ 5 }} \\ right) }} _ {{ 3 }} </equation> \n # 保护" ,
"前置内容。" . repeat ( 140 )
);
assert! ( query_centered_quote ( & chunk_content , "三乙基硅" , 500 )
. unwrap ()
. contains ( "三乙基硅酯" ));
let mapped = map_reference_plan (
& json! ({
"file_path" : "book.docx" ,
"chunk_id" : "doc1-chunk-256" ,
"chunks" : [{
"chunk_id" : "doc1-chunk-256" ,
2026-06-09 18:40:48 +08:00
"content" : chunk_content ,
"sidecar" : {
"type" : "block" ,
"refs" : [
{ "type" : "block" , "id" : "wrong-index" },
{ "type" : "block" , "id" : "tes" }
]
}
2026-06-08 20:35:49 +08:00
}]
}),
& registry ,
"file:///tmp/root" ,
& root ,
Some ( "三乙基硅" ),
);
assert! ( mapped [ "quote" ]. as_str (). unwrap (). contains ( "三乙基硅酯" ));
assert! ( ! mapped [ "quote" ]. as_str (). unwrap (). contains ( "取代苄酯" ));
assert_eq! ( mapped [ "quoteSource" ], "chunk" );
assert_eq! ( mapped [ "locator" ][ "blockId" ], "tes" );
assert_eq! ( mapped [ "locator" ][ "bbox" ][ "x0" ]. as_f64 (), Some ( 10.0 ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
let _ = fs ::remove_dir_all ( root );
}
2026-06-13 22:20:01 +08:00
#[test]
fn mapped_reference_uses_kv_store_sidecar_refs_for_book_locator () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-kv-sidecar-locator" );
fs ::create_dir_all ( root . join ( "docs" )). expect ( "docs" );
fs ::write ( root . join ( "docs" ). join ( "book.docx" ), b "docx" ). expect ( "docx" );
let input_dir = root . join ( "inputs" );
let working_dir = root . join ( "rag_storage" );
let parsed_dir = input_dir . join ( "__parsed__" ). join ( "book.docx.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::create_dir_all ( & working_dir ). expect ( "working dir" );
fs ::write (
parsed_dir . join ( "book.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":3}"# ,
r ##"{"type":"content","blockid":"heading","content":"羟基-去-胺化","positions":[{"type":"paraid","anchor":10,"range":[null,null]}]}"## ,
r ##"{"type":"content","blockid":"amide-hydrolysis","content":"无取代的酰胺能在酸或碱催化条件下水解,产物分别是游离的酸和铵根离子。","positions":[{"type":"paraid","anchor":11,"range":["A","A"]},{"type":"text_fingerprint","anchor":"fp-amide"}]}"## ,
r ##"{"type":"content","blockid":"water-only","content":"仅仅用水难以水解绝大多数酰胺。","positions":[{"type":"paraid","anchor":12,"range":["B","B"]}]}"## ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
fs ::write (
working_dir . join ( "kv_store_text_chunks.json" ),
serde_json ::to_string ( & json! ({
"doc-book-chunk-847" : {
"content" : "羟基-去-胺化 \n\n 无取代的酰胺能在酸或碱催化条件下水解,产物分别是游离的酸和铵根离子。 \n\n 仅仅用水难以水解绝大多数酰胺。" ,
"sidecar" : {
"type" : "block" ,
"id" : "heading" ,
"refs" : [
{ "type" : "block" , "id" : "heading" },
{ "type" : "block" , "id" : "amide-hydrolysis" },
{ "type" : "block" , "id" : "water-only" }
]
}
}
}))
. expect ( "kv json" ),
)
. expect ( "kv store" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
std ::env ::set_var ( "MNOTE_LIGHTRAG_WORKING_DIR" , & working_dir );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src-book" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : root . join ( "docs" ). join ( "book.docx" ). display (). to_string (),
source_root_relative_path : "docs/book.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc-book" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "book.docx" . into (),
symlink_path : input_dir . join ( "book.docx" ). display (). to_string (),
parser_hint : Some ( "-P!" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let mapped = map_reference_plan (
& json! ({
"file_path" : "book.docx" ,
"chunk_id" : "doc-book-chunk-847" ,
"chunks" : [{
"chunk_id" : "doc-book-chunk-847" ,
"file_path" : "book.docx" ,
"content" : "羟基-去-胺化 \n\n 无取代的酰胺能在酸或碱催化条件下水解,产物分别是游离的酸和铵根离子。"
}]
}),
& registry ,
"file:///tmp/root" ,
& root ,
Some ( "酰胺 水解" ),
);
assert_eq! ( mapped [ "quoteSource" ], "chunk" );
assert_eq! ( mapped [ "citationDiagnostics" ][ "sidecarBlockMapped" ], true );
assert_eq! ( mapped [ "locator" ][ "blockId" ], "amide-hydrolysis" );
assert_eq! (
mapped [ "locator" ][ "openAction" ][ "params" ][ "paragraphOrdinal" ]. as_u64 (),
Some ( 11 )
);
assert_eq! (
mapped [ "locator" ][ "openAction" ][ "params" ][ "textFingerprint" ]. as_str (),
Some ( "fp-amide" )
);
assert_eq! ( mapped [ "locatorPrecision" ], "paragraph" );
assert! ( ! mapped [ "citationMarkdown" ]
. as_str ()
. unwrap_or_default ()
. contains ( "来源定位降级" ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
std ::env ::remove_var ( "MNOTE_LIGHTRAG_WORKING_DIR" );
let _ = fs ::remove_dir_all ( root );
}
2026-06-08 20:35:49 +08:00
#[test]
fn source_content_diagnostics_reports_sidecar_ocr_text () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-source-diagnostics" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir
. join ( "__parsed__" )
. join ( "mnote-hash-image.png.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "mnote-hash-image.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":1}"# ,
r ##"{"type":"content","blockid":"block1","content":"线程作用域 OCR text","positions":[{"type":"bbox","anchor":"1","range":[1.0,2.0,3.0,4.0]}]}"## ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : root . join ( "image.png" ). display (). to_string (),
source_root_relative_path : "image.png" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-image.png" . into (),
symlink_path : input_dir . join ( "mnote-hash-image.png" ). display (). to_string (),
parser_hint : Some ( "mineru" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let diagnostics = knowledge_rag_source_content_diagnostics ( & registry );
assert_eq! ( diagnostics [ 0 ][ "ocrTextExposed" ]. as_bool (), Some ( true ));
assert_eq! ( diagnostics [ 0 ][ "sidecarMeaningfulBlocks" ]. as_u64 (), Some ( 1 ));
assert_eq! ( diagnostics [ 0 ][ "directImageScan" ]. as_bool (), Some ( true ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
let _ = fs ::remove_dir_all ( root );
}
2026-06-13 22:20:01 +08:00
#[test]
fn document_structure_index_uses_sidecar_headings_and_paragraph_ranges () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-structure-index" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir . join ( "__parsed__" ). join ( "book.docx.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "book.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":4}"# ,
r ##"{"type":"content","blockid":"h1","content":"# 第 1 章 酰胺","level":1,"parent_headings":[],"positions":[{"type":"paraid","anchor":0,"range":[null,null]}]}"## ,
r ##"{"type":"content","blockid":"p1","content":"酰胺的水解可在酸性条件下进行。","level":1,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":1,"range":[null,null]}]}"## ,
r ###"{"type":"content","blockid":"h2","content":"## 碱性水解","level":2,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":2,"range":[null,null]}]}"### ,
r ###"{"type":"content","blockid":"p2","content":"也可用碱性水解或酶促水解。","level":2,"parent_headings":["第 1 章 酰胺","碱性水解"],"positions":[{"type":"paraid","anchor":3,"range":[null,null]}]}"### ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src-book" . into (),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
source_path : root . join ( "book.docx" ). display (). to_string (),
source_root_relative_path : "book.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc-book" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "book.docx" . into (),
symlink_path : input_dir . join ( "book.docx" ). display (). to_string (),
parser_hint : Some ( "-P!" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let payload = document_structure_index_payload (
& root ,
& registry ,
& [],
& [ json! ({ "sourceId" :"src-book" , "locator" :{ "blockId" :"p2" }})],
Some ( "酰胺 水解" ),
8 ,
)
. expect ( "structure index" );
let sections = payload [ "documents" ][ 0 ][ "sections" ]. as_array (). unwrap ();
assert_eq! ( sections . len (), 2 );
assert_eq! ( sections [ 0 ][ "title" ], "第 1 章 酰胺" );
assert_eq! ( sections [ 0 ][ "startParagraphOrdinal" ], 0 );
assert_eq! ( sections [ 0 ][ "endParagraphOrdinal" ], 1 );
assert_eq! ( sections [ 1 ][ "title" ], "碱性水解" );
assert_eq! ( sections [ 1 ][ "matchedReferenceCount" ], 1 );
assert! ( sections [ 1 ][ "queryMatchCount" ]. as_u64 (). unwrap () > 0 );
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn section_context_reads_limited_sidecar_blocks_for_range () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-section-context" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir . join ( "__parsed__" ). join ( "book.docx.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "book.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":5}"# ,
r ##"{"type":"content","blockid":"h1","content":"# 第 1 章 酰胺","level":1,"parent_headings":[],"positions":[{"type":"paraid","anchor":0,"range":[null,null]}]}"## ,
r ##"{"type":"content","blockid":"p1","content":"酰胺的水解可在酸性条件下进行。","level":1,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":1,"range":["p1","p1"]}]}"## ,
r ###"{"type":"content","blockid":"h2","content":"## 碱性水解","level":2,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":2,"range":[null,null]}]}"### ,
r ###"{"type":"content","blockid":"p2","content":"也可用碱性水解或酶促水解。","level":2,"parent_headings":["第 1 章 酰胺","碱性水解"],"positions":[{"type":"paraid","anchor":3,"range":["p2","p2"]}]}"### ,
r ###"{"type":"content","blockid":"p3","content":"这个段落不应在 maxBlocks=2 时返回。","level":2,"parent_headings":["第 1 章 酰胺","碱性水解"],"positions":[{"type":"paraid","anchor":4,"range":["p3","p3"]}]}"### ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src-book" . into (),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
source_path : root . join ( "book.docx" ). display (). to_string (),
source_root_relative_path : "book.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc-book" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "book.docx" . into (),
symlink_path : input_dir . join ( "book.docx" ). display (). to_string (),
parser_hint : Some ( "-P!" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let payload = sidecar_section_context_payload (
& root ,
& registry ,
& KnowledgeRagSectionContextRequest {
workspace_id : Some ( "ws" . into ()),
root_uri : format ! ( "file://{}" , root . display ()),
source_path : Some ( "book.docx" . into ()),
source_id : None ,
light_rag_doc_id : None ,
file_path : None ,
section_id : Some ( "src-book#block-h2" . into ()),
start_block_ordinal : Some ( 2 ),
end_block_ordinal : Some ( 4 ),
start_paragraph_ordinal : None ,
end_paragraph_ordinal : None ,
context_before : Some ( 1 ),
context_after : Some ( 0 ),
max_blocks : Some ( 2 ),
max_chars : Some ( 2_000 ),
},
)
. expect ( "section context" );
assert_eq! ( payload [ "blocks" ]. as_array (). unwrap (). len (), 2 );
assert_eq! ( payload [ "blocks" ][ 0 ][ "blockOrdinal" ], 1 );
assert_eq! ( payload [ "blocks" ][ 1 ][ "blockOrdinal" ], 2 );
assert_eq! ( payload [ "chunks" ]. as_array (). unwrap (). len (), 1 );
assert_eq! ( payload [ "chunks" ][ 0 ][ "startBlockOrdinal" ], 1 );
assert_eq! ( payload [ "limits" ][ "truncated" ], true );
assert! ( payload [ "text" ]
. as_str ()
. unwrap ()
. contains ( "酰胺的水解可在酸性条件下进行" ));
assert! ( ! payload [ "text" ]. as_str (). unwrap (). contains ( "这个段落不应" ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
let _ = fs ::remove_dir_all ( root );
}
2026-06-07 10:35:21 +08:00
#[test]
fn mapped_references_filters_unmapped_provider_references () {
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 10:35:21 +08:00
entries : vec ! [],
};
let raw = json! ({
"data" : {
"references" : [{ "reference_id" :"1" , "file_path" :"orphan.pdf" }],
"chunks" : [{
"reference_id" :"1" ,
"chunk_id" :"orphan-chunk" ,
"file_path" :"orphan.pdf" ,
"content" :"orphan provider chunk"
}]
}
});
2026-06-08 20:35:49 +08:00
let mapped = mapped_references (
& raw ,
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
None ,
);
2026-06-07 10:35:21 +08:00
assert! (
mapped . is_empty (),
"unmapped provider references must not become MNote citations"
);
let plan = map_reference_plan (
& json! ({ "file_path" :"orphan.pdf" , "chunk_id" :"orphan-chunk" }),
& registry ,
"file:///tmp/root" ,
Path ::new ( "/tmp/root" ),
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 10:35:21 +08:00
);
assert_eq! ( plan [ "unmapped" ], true );
assert_eq! ( plan [ "locatorDegraded" ], true );
}
2026-06-07 01:10:31 +08:00
#[test]
fn query_ranking_prefers_exact_source_and_quote_match () {
let mut references = vec! [
json! ({
"sourceRootRelativePath" : "knowledge-rag-fixtures-7-50/paper-brotli-comparison.pdf" ,
"filePath" : "mnote-paper-brotli-comparison.pdf" ,
"quote" : "brotli compression table" ,
"citationMarkdown" : "[paper-brotli-comparison.pdf · p.3](/)"
}),
json! ({
"sourceRootRelativePath" : "knowledge-rag-fixtures-7-50/scan-image-start.pdf" ,
"filePath" : "mnote-scan-image-start.pdf" ,
"quote" : "# START: This is the first image in PDF" ,
"citationMarkdown" : "[scan-image-start.pdf · p.1](/)"
}),
];
rank_mapped_references_for_query ( & mut references , "scan image start" );
assert_eq! (
references [ 0 ][ "sourceRootRelativePath" ],
"knowledge-rag-fixtures-7-50/scan-image-start.pdf"
);
}
#[test]
2026-06-09 09:20:56 +08:00
fn query_mode_preserves_lightrag_native_modes () {
assert_eq! ( normalize_lightrag_query_mode ( Some ( "hybrid" )), "hybrid" );
2026-06-07 01:10:31 +08:00
assert_eq! ( normalize_lightrag_query_mode ( Some ( "global" )), "global" );
2026-06-09 09:20:56 +08:00
assert_eq! ( normalize_lightrag_query_mode ( Some ( "exact" )), "naive" );
2026-06-07 01:10:31 +08:00
assert_eq! ( normalize_lightrag_query_mode ( None ), "mix" );
2026-06-09 09:20:56 +08:00
assert_eq! ( normalize_knowledge_rag_search_mode ( Some ( "exact" )), "exact" );
assert_eq! ( normalize_knowledge_rag_search_mode ( Some ( "vector" )), "naive" );
assert_eq! (
normalize_knowledge_rag_search_mode ( Some ( "hybrid" )),
"hybrid"
);
2026-06-09 09:32:36 +08:00
assert_eq! ( normalize_knowledge_rag_search_mode ( None ), "exact" );
2026-06-07 01:10:31 +08:00
}
2026-06-13 22:20:01 +08:00
#[test]
fn skip_kg_scoped_query_forces_naive_mode_for_book_sources () {
let root = temp_root ( "mnote-knowledge-rag-skip-kg-query-mode" );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : format ! ( "file://{}" , root . display ()),
updated_at_ms : 1 ,
indexed_roots : Vec ::new (),
entries : vec ! [
test_registry_entry (
& root ,
"books/organic.docx" ,
Some ( "doc-book" ),
Some ( "processed" ),
Some ( 2 ),
None ,
false ,
),
test_registry_entry (
& root ,
"papers/small.pdf" ,
Some ( "doc-paper" ),
Some ( "processed" ),
Some ( 2 ),
None ,
false ,
),
],
};
let mut registry = registry ;
registry . entries [ 0 ]. parser_hint = Some ( "-P!" . into ());
registry . entries [ 0 ]. light_rag_file_path = "mnote-book.[-P!].docx" . into ();
registry . entries [ 1 ]. parser_hint = None ;
registry . entries [ 1 ]. light_rag_file_path = "mnote-paper.pdf" . into ();
let scoped_book = normalize_source_scope ( Some ( & [ "books/organic.docx" . to_string ()]));
let decision = resolve_lightrag_query_mode_for_scope ( & registry , & scoped_book , "mix" );
assert_eq! ( decision . mode , "naive" );
assert_eq! ( decision . reason , "source_scope_skip_kg_document" );
let search_decision =
resolve_knowledge_rag_search_mode_for_scope ( & registry , & scoped_book , "hybrid" );
assert_eq! ( search_decision . mode , "naive" );
assert_eq! ( search_decision . reason , "source_scope_skip_kg_document" );
let exact_search_decision =
resolve_knowledge_rag_search_mode_for_scope ( & registry , & scoped_book , "exact" );
assert_eq! ( exact_search_decision . mode , "exact" );
assert_eq! ( exact_search_decision . reason , "requested_mode" );
let naive_decision =
resolve_lightrag_query_mode_for_scope ( & registry , & scoped_book , "naive" );
assert_eq! ( naive_decision . mode , "naive" );
assert_eq! ( naive_decision . reason , "requested_mode" );
let scoped_paper = normalize_source_scope ( Some ( & [ "papers/small.pdf" . to_string ()]));
let paper_decision = resolve_lightrag_query_mode_for_scope ( & registry , & scoped_paper , "mix" );
assert_eq! ( paper_decision . mode , "mix" );
assert_eq! ( paper_decision . reason , "requested_mode" );
let _ = fs ::remove_dir_all ( root );
}
2026-06-07 01:10:31 +08:00
#[test]
fn symlink_name_preserves_parser_hint_before_extension () {
let name =
lightrag_symlink_name ( Path ::new ( "/tmp/books/demo.pdf" ), "demo.pdf" , Some ( "mineru" ));
assert! ( name . ends_with ( "-demo.[mineru].pdf" ));
}
2026-06-08 20:35:49 +08:00
#[test]
fn symlink_name_preserves_parser_hint_with_options_before_extension () {
let name = lightrag_symlink_name (
Path ::new ( "/tmp/books/demo.docx" ),
"demo.docx" ,
Some ( "native-P" ),
);
assert! ( name . ends_with ( "-demo.[native-P].docx" ));
assert! ( strip_one_supported_parser_hint ( & name )
. as_deref ()
. unwrap_or_default ()
. ends_with ( "-demo.docx" ));
}
#[test]
fn ocr_layered_docx_defaults_to_native_paragraph_strategy () {
let file_name = "[OCR]_有机合成中的保护基_20250201.layered_删减-2025-02-04 18-59-42.docx" ;
assert_eq! (
default_lightrag_parser_hint_for_source ( Path ::new ( file_name ), file_name ). as_deref (),
Some ( "native-P" )
);
assert_eq! (
default_lightrag_parser_hint_for_source ( Path ::new ( "ordinary.docx" ), "ordinary.docx" ),
None
);
}
2026-06-13 22:20:01 +08:00
#[test]
fn large_docx_and_pdf_default_to_paragraph_skip_kg_strategy () {
let root = temp_root ( "mnote-knowledge-rag-large-skip-kg" );
let docx = root . join ( "organic-book.docx" );
let pdf = root . join ( "organic-book.pdf" );
fs ::File ::create ( & docx )
. expect ( "docx" )
. set_len ( LARGE_DOCUMENT_SKIP_KG_MIN_BYTES )
. expect ( "large docx" );
fs ::File ::create ( & pdf )
. expect ( "pdf" )
. set_len ( LARGE_DOCUMENT_SKIP_KG_MIN_BYTES )
. expect ( "large pdf" );
assert_eq! (
default_lightrag_parser_hint_for_source ( & docx , "organic-book.docx" ). as_deref (),
Some ( "-P!" )
);
assert_eq! (
default_lightrag_parser_hint_for_source ( & pdf , "organic-book.pdf" ). as_deref (),
Some ( "-P!" )
);
assert! ( valid_lightrag_parser_options ( "P!" ));
let layered = root . join ( "[OCR]_book.layered.docx" );
fs ::File ::create ( & layered )
. expect ( "layered" )
. set_len ( LARGE_DOCUMENT_SKIP_KG_MIN_BYTES )
. expect ( "large layered" );
assert_eq! (
default_lightrag_parser_hint_for_source ( & layered , "[OCR]_book.layered.docx" ). as_deref (),
Some ( "native-P!" )
);
let _ = fs ::remove_dir_all ( root );
}
2026-06-07 01:10:31 +08:00
#[test]
fn symlink_name_does_not_duplicate_existing_parser_hint () {
let name = lightrag_symlink_name (
Path ::new ( "/tmp/books/demo.[native].pdf" ),
"demo.[native].pdf" ,
Some ( "native" ),
);
assert! ( name . ends_with ( "-demo.[native].pdf" ));
assert! ( ! name . contains ( ".[native].[native]." ));
}
2026-06-08 20:35:49 +08:00
#[test]
fn symlink_name_does_not_duplicate_existing_parser_hint_with_options () {
let name = lightrag_symlink_name (
Path ::new ( "/tmp/books/demo.[native-P].docx" ),
"demo.[native-P].docx" ,
Some ( "native-P" ),
);
assert! ( name . ends_with ( "-demo.[native-P].docx" ));
assert! ( ! name . contains ( ".[native-P].[native-P]." ));
}
2026-06-07 01:10:31 +08:00
#[test]
fn registry_file_path_matches_lightrag_hint_canonicalization () {
let entry = KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/scan.[mineru].pdf" . into (),
source_root_relative_path : "scan.[mineru].pdf" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : None ,
light_rag_status : None ,
light_rag_file_path : "mnote-hash-scan.[mineru].[mineru].pdf" . into (),
symlink_path : "/tmp/input/mnote-hash-scan.[mineru].[mineru].pdf" . into (),
parser_hint : Some ( "mineru" . into ()),
indexed_at_ms : None ,
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
};
assert! ( lightrag_file_path_matches (
& entry ,
"mnote-hash-scan.[mineru].pdf"
));
}
#[test]
fn locator_uses_stripped_parser_hint_sidecar_path () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-sidecar" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir
. join ( "__parsed__" )
. join ( "mnote-hash-scan.pdf.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "mnote-hash-scan.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":1}"# ,
r ##"{"type":"content","blockid":"block1","content":"# START: This is the first image in PDF\nThis is text BEFORE the image.","positions":[{"type":"bbox","anchor":"1","range":[171.0,126.0,648.0,152.0]}]}"## ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let entry = KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/scan.pdf" . into (),
source_root_relative_path : "scan.pdf" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-scan.[mineru].pdf" . into (),
symlink_path : input_dir
. join ( "mnote-hash-scan.[mineru].pdf" )
. display ()
. to_string (),
parser_hint : Some ( "mineru" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
};
let locator = lightrag_locator_for_reference (
Path ::new ( "/tmp/root" ),
"file:///tmp/root" ,
& entry ,
& json! ( "doc1-chunk-000" ),
2026-06-08 20:35:49 +08:00
None ,
2026-06-09 09:20:56 +08:00
None ,
2026-06-09 18:40:48 +08:00
& json! ( 0 ),
Some ( & json! ({ "type" :"block" , "refs" :[{ "type" :"block" , "id" :"block1" }]})),
2026-06-07 01:10:31 +08:00
)
. expect ( "locator" );
assert_eq! ( locator . page , Some ( 1 ));
assert_eq! ( locator . bbox . as_ref (). map ( | bbox | bbox . x0 ), Some ( 171.0 ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
2026-06-09 09:20:56 +08:00
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn locator_exports_docx_paraid_positions_for_office_preview () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-docx-paraid" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir
. join ( "__parsed__" )
. join ( "mnote-hash-report.docx.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "mnote-hash-report.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":1}"# ,
r ##"{"type":"content","blockid":"block-docx-1","content":"1.1 监管政策变革\n化妆品新原料监管政策变革内容。","positions":[{"type":"paraid","anchor":4,"range":["6692C49B","6692C49B"]},{"type":"text_fingerprint","anchor":"1454ec0b0ffd9fc0"}]}"## ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let entry = KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : "/tmp/root/report.docx" . into (),
source_root_relative_path : "report.docx" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-report.docx" . into (),
symlink_path : input_dir
. join ( "mnote-hash-report.docx" )
. display ()
. to_string (),
parser_hint : Some ( "native" . into ()),
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
};
let locator = lightrag_locator_for_reference (
Path ::new ( "/tmp/root" ),
"file:///tmp/root" ,
& entry ,
& json! ( "doc1-chunk-000" ),
Some ( "监管政策变革" ),
None ,
2026-06-09 18:40:48 +08:00
& json! ( 0 ),
Some ( & json! ({ "type" :"block" , "refs" :[{ "type" :"block" , "id" :"block-docx-1" }]})),
2026-06-09 09:20:56 +08:00
)
. expect ( "locator" );
assert_eq! ( locator . resource_kind , EvidenceResourceKind ::Office );
assert_eq! ( locator . page , None );
assert_eq! ( locator . bbox , None );
assert_eq! ( locator . block_id . as_deref (), Some ( "block-docx-1" ));
assert_eq! (
locator . open_action . params [ "paragraphOrdinal" ]. as_u64 (),
Some ( 4 )
);
assert_eq! (
locator . open_action . params [ "paraIdStart" ]. as_str (),
Some ( "6692C49B" )
);
assert_eq! (
locator . open_action . params [ "textFingerprint" ]. as_str (),
Some ( "1454ec0b0ffd9fc0" )
);
let citation_url = crate ::routes ::evidence ::citation_url_for_locator ( & locator );
assert! ( citation_url . contains ( "paragraphOrdinal=4" ));
assert! ( citation_url . contains ( "paraIdStart=6692C49B" ));
assert! ( citation_url . contains ( "textFingerprint=1454ec0b0ffd9fc0" ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
2026-06-07 10:35:21 +08:00
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn locator_degrades_when_sidecar_quote_does_not_match () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-sidecar-no-match" );
fs ::create_dir_all ( root . join ( "docs" )). expect ( "docs" );
fs ::write ( root . join ( "docs" ). join ( "Host.md" ), "# Host \n " ). expect ( "host" );
fs ::write ( root . join ( "docs" ). join ( "scan.pdf" ), b "pdf" ). expect ( "pdf" );
let input_dir = root . join ( "inputs" );
let parsed_dir = input_dir
. join ( "__parsed__" )
. join ( "mnote-hash-scan.pdf.parsed" );
fs ::create_dir_all ( & parsed_dir ). expect ( "parsed dir" );
fs ::write (
parsed_dir . join ( "mnote-hash-scan.blocks.jsonl" ),
[
r #"{"type":"meta","blocks":1}"# ,
r ##"{"type":"content","blockid":"block1","content":"This block is not the returned quote.","positions":[{"type":"bbox","anchor":"9","range":[1.0,2.0,3.0,4.0]}]}"## ,
]
. join ( " \n " ),
)
. expect ( "blocks" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_INPUT_DIR" , & input_dir );
let registry = KnowledgeRagSourceRegistry {
schema : REGISTRY_SCHEMA . to_string (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
updated_at_ms : 1 ,
2026-06-08 20:35:49 +08:00
indexed_roots : Vec ::new (),
2026-06-07 10:35:21 +08:00
entries : vec ! [ KnowledgeRagSourceRegistryEntry {
source_id : "src1" . into (),
workspace_id : "ws" . into (),
root_uri : "file:///tmp/root" . into (),
source_path : root . join ( "docs" ). join ( "scan.pdf" ). display (). to_string (),
source_root_relative_path : "docs/scan.pdf" . into (),
source_hash : "mnote-fnv64:1" . into (),
light_rag_doc_id : Some ( "doc1" . into ()),
light_rag_status : Some ( "processed" . into ()),
light_rag_file_path : "mnote-hash-scan.pdf" . into (),
symlink_path : input_dir . join ( "mnote-hash-scan.pdf" ). display (). to_string (),
parser_hint : None ,
indexed_at_ms : Some ( 2 ),
deleted_at_ms : None ,
stale : false ,
updated_at_ms : 2 ,
}],
};
let mapped = map_reference_plan (
& json! ({
"file_path" : "mnote-hash-scan.pdf" ,
"chunk_id" : "doc1-chunk-000" ,
"chunks" : [{ "chunk_id" : "doc1-chunk-000" , "content" : "A different quote should not get page or bbox." }]
}),
& registry ,
"file:///tmp/root" ,
& root ,
2026-06-08 20:35:49 +08:00
None ,
2026-06-07 10:35:21 +08:00
);
assert! ( mapped [ "locator" ]. is_null ());
assert_eq! ( mapped [ "locatorDegraded" ], true );
assert! ( mapped [ "citationUrl" ]
. as_str ()
. is_some_and ( | url | url . contains ( "resourceTab=" )));
assert! ( mapped [ "citationMarkdown" ]
. as_str ()
. unwrap ()
. contains ( "来源定位降级" ));
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
2026-06-07 01:10:31 +08:00
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn lightrag_paths_prefer_source_env_over_legacy_process_env () {
let _guard = env_lock (). lock (). expect ( "env lock" );
let root = temp_root ( "mnote-knowledge-rag-env" );
let source_input = root . join ( "source-inputs" );
let source_working = root . join ( "source-storage" );
let legacy_input = root . join ( "legacy-inputs" );
let legacy_working = root . join ( "legacy-storage" );
let env_file = root . join ( "lightrag.env" );
fs ::write (
& env_file ,
format! (
"INPUT_DIR= {} \n WORKING_DIR= {} \n " ,
source_input . display (),
source_working . display ()
),
)
. expect ( "env file" );
std ::env ::set_var ( "MNOTE_LIGHTRAG_ENV_FILE" , & env_file );
std ::env ::set_var ( "INPUT_DIR" , & legacy_input );
std ::env ::set_var ( "WORKING_DIR" , & legacy_working );
std ::env ::remove_var ( "MNOTE_LIGHTRAG_INPUT_DIR" );
std ::env ::remove_var ( "MNOTE_LIGHTRAG_WORKING_DIR" );
assert_eq! ( lightrag_input_dir (), source_input );
assert_eq! ( lightrag_working_dir (), source_working );
std ::env ::remove_var ( "MNOTE_LIGHTRAG_ENV_FILE" );
std ::env ::remove_var ( "INPUT_DIR" );
std ::env ::remove_var ( "WORKING_DIR" );
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn directory_source_expands_supported_files_and_skips_internal_dirs () {
let root = temp_root ( "mnote-knowledge-rag-directory" );
fs ::create_dir_all ( root . join ( "docs" ). join ( "nested" )). expect ( "nested" );
fs ::create_dir_all ( root . join ( ".mnote" ). join ( "index" )). expect ( "mnote" );
fs ::write ( root . join ( "docs" ). join ( "a.pdf" ), b "%PDF" ). expect ( "pdf" );
fs ::write ( root . join ( "docs" ). join ( "nested" ). join ( "b.md" ), "# B" ). expect ( "md" );
fs ::write ( root . join ( "docs" ). join ( "ignored.tmp" ), "tmp" ). expect ( "tmp" );
fs ::write (
root . join ( ".mnote" ). join ( "index" ). join ( "hidden.pdf" ),
b "%PDF" ,
)
. expect ( "hidden" );
let sources =
resolve_knowledge_rag_sources ( & root , "docs" , & test_context ()). expect ( "sources" );
let rel = sources
. iter ()
. map ( | source | root_relative_path ( & root , & source . canonical_path ). expect ( "relative" ))
. collect ::< Vec < _ >> ();
assert_eq! ( rel , vec! [ "docs/a.pdf" , "docs/nested/b.md" ]);
assert! ( sources
. iter ()
. all ( | source | source . source_kind == "directory" ));
let _ = fs ::remove_dir_all ( root );
}
#[test]
fn file_source_rejects_unsupported_extension () {
let root = temp_root ( "mnote-knowledge-rag-unsupported" );
fs ::write ( root . join ( "notes.tmp" ), "tmp" ). expect ( "tmp" );
let error = resolve_knowledge_rag_sources ( & root , "notes.tmp" , & test_context ())
. expect_err ( "unsupported" );
assert_eq! ( error . code (), "knowledge_rag_source_unsupported" );
let _ = fs ::remove_dir_all ( root );
}
}