use crate::app::AppState; use crate::context::RequestContext; use crate::error::WebError; use crate::routes::evidence::{citation_markdown_for_locator, citation_url_for_locator}; use crate::routes::local_folder_source; use axum::body::Body; use axum::extract::{Extension, Json, Query, State}; use axum::http::{header, StatusCode}; use axum::response::Response; use core_protocol::evidence::{ EvidenceBBox, EvidenceLocator, EvidenceOpenAction, EvidenceResourceKind, }; use futures_util::TryStreamExt; use serde::{Deserialize, Serialize}; use serde_json::{json, Value}; use std::collections::hash_map::DefaultHasher; use std::collections::{BTreeMap, BTreeSet}; use std::env; use std::fs; use std::hash::{Hash, Hasher}; use std::path::{Path, PathBuf}; use std::time::{Duration, SystemTime, UNIX_EPOCH}; const REGISTRY_SCHEMA: &str = "mnote.knowledge_rag.source_registry.v1"; const REFERENCE_SCHEMA: &str = "mnote.knowledge_rag.reference.v1"; const DEFAULT_LIGHTRAG_ENDPOINT: &str = "http://127.0.0.1:9621"; const DEFAULT_LIGHTRAG_INPUT_DIR: &str = "/mnt/Data1T/Mnote_data/lightrag/inputs"; const DEFAULT_LIGHTRAG_WORKING_DIR: &str = "/mnt/Data1T/Mnote_data/lightrag/rag_storage"; const MAX_INGEST_SOURCES_PER_REQUEST: usize = 200; const LIGHTRAG_PROVIDER_MIN_QUERY_CHARS: usize = 2; const LARGE_DOCUMENT_SKIP_KG_MIN_BYTES: u64 = 10 * 1024 * 1024; const DOCUMENT_STRUCTURE_INDEX_SCHEMA: &str = "mnote.knowledge_rag.document_structure_index.v1"; const DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PER_QUERY: usize = 8; const DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PERSISTED: usize = 256; const DOCUMENT_STRUCTURE_INDEX_MAX_SECTIONS_PER_DOC: usize = 2_000; const SOURCE_SCOPE_MODE_POST_FILTER: &str = "post_filter_mapped_references"; const KNOWLEDGE_RAG_SOURCE_EXTENSIONS: &[&str] = &[ "md", "markdown", "txt", "pdf", "doc", "docx", "ppt", "pptx", "xls", "xlsx", "csv", "png", "jpg", "jpeg", "webp", "gif", "bmp", "tif", "tiff", ]; const LIGHTRAG_SCAN_SOURCE_EXTENSIONS: &[&str] = &[ "md", "markdown", "mdx", "txt", "pdf", "docx", "pptx", "xlsx", "rtf", "odt", "tex", "epub", "html", "htm", "png", "jpg", "jpeg", "webp", "gif", "bmp", "tif", "tiff", ]; #[derive(Debug, Clone, Default)] struct CitationTextBundle { raw_quote: String, display_quote: String, locator_evidence_text: String, search_query: String, normalized_fingerprint: String, locator_text_source: &'static str, display_cleaned: bool, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagStatusQuery { pub workspace_id: Option, pub root_uri: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagIngestRequest { pub workspace_id: Option, pub root_uri: String, pub sources: Vec, pub force: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagSourceInput { pub source_path: Option, pub path: Option, pub parser_hint: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagQueryRequest { pub workspace_id: Option, pub root_uri: String, pub question: Option, pub query: Option, pub mode: Option, pub top_k: Option, pub chunk_top_k: Option, pub include_chunk_content: Option, pub source_paths: Option>, pub include_document_structure_index: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagSearchRequest { pub workspace_id: Option, pub root_uri: String, pub query: String, pub mode: Option, pub top_k: Option, pub chunk_top_k: Option, pub include_chunk_content: Option, pub source_paths: Option>, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagOpenReferenceRequest { pub workspace_id: Option, pub root_uri: String, pub reference: Option, pub reference_id: Option, pub file_path: Option, pub chunk_id: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagSectionContextRequest { pub workspace_id: Option, pub root_uri: String, pub source_path: Option, pub source_id: Option, pub light_rag_doc_id: Option, pub file_path: Option, pub section_id: Option, pub start_block_ordinal: Option, pub end_block_ordinal: Option, pub start_paragraph_ordinal: Option, pub end_paragraph_ordinal: Option, pub context_before: Option, pub context_after: Option, pub max_blocks: Option, pub max_chars: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagDeleteSourceRequest { pub workspace_id: Option, pub root_uri: String, pub source_path: Option, pub light_rag_doc_id: Option, } #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] pub struct KnowledgeRagPruneRegistryRequest { pub workspace_id: Option, pub root_uri: String, } #[derive(Debug, Clone, Serialize, Deserialize, Default)] #[serde(rename_all = "camelCase")] struct KnowledgeRagSourceRegistry { schema: String, workspace_id: String, root_uri: String, updated_at_ms: u128, #[serde(default)] indexed_roots: Vec, entries: Vec, } #[derive(Debug, Clone, Serialize, Deserialize, Default)] #[serde(rename_all = "camelCase")] struct KnowledgeRagIndexedRoot { root_relative_path: String, recursive: bool, #[serde(default)] exclude_patterns: Vec, run_on_change: Option, updated_at_ms: u128, } #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] struct KnowledgeRagSourceRegistryEntry { source_id: String, workspace_id: String, root_uri: String, source_path: String, source_root_relative_path: String, source_hash: String, light_rag_doc_id: Option, #[serde(default, skip_serializing_if = "Option::is_none")] light_rag_status: Option, light_rag_file_path: String, symlink_path: String, parser_hint: Option, indexed_at_ms: Option, deleted_at_ms: Option, stale: bool, updated_at_ms: u128, } #[derive(Debug, Clone, Default)] pub(crate) struct KnowledgeRagSourceStatuses { pub(crate) indexed_paths: BTreeSet, pub(crate) indexing_paths: BTreeSet, pub(crate) failed_paths: BTreeSet, } pub(crate) fn knowledge_rag_source_statuses( root_path: &Path, workspace_id: &str, root_uri: &str, ) -> Result { let registry = read_registry(root_path, workspace_id, root_uri)?; let mut statuses = KnowledgeRagSourceStatuses::default(); for entry in registry.entries { let path = entry.source_root_relative_path.trim(); if path.is_empty() { continue; } let provider_status = entry.light_rag_status.as_deref().unwrap_or_default(); if provider_status == "delete_completed" { continue; } if provider_status == "failed" { statuses.failed_paths.insert(path.to_string()); } else if provider_status == "delete_retry_required" { statuses.failed_paths.insert(path.to_string()); } else if provider_status == "delete_submitted" || (entry.deleted_at_ms.is_some() && entry.light_rag_doc_id.is_some()) { statuses.indexing_paths.insert(path.to_string()); } else if entry.stale { statuses.failed_paths.insert(path.to_string()); } else if entry.deleted_at_ms.is_some() && entry.light_rag_doc_id.is_none() { continue; } else if entry.indexed_at_ms.is_some() && entry.light_rag_doc_id.is_some() { statuses.indexed_paths.insert(path.to_string()); } else { statuses.indexing_paths.insert(path.to_string()); } } Ok(statuses) } pub async fn retired_local_ocr_endpoint( Extension(context): Extension, ) -> (StatusCode, Json) { ( StatusCode::GONE, Json(json!({ "ok": false, "code": "mnote_local_ocr_retired", "message": "本地 OCR sidecar 已退役;图片、PDF、Office 与索引统一交给 LightRAG 资料库处理。", "replacement": { "provider": "lightrag", "status": "/api/knowledge-rag/status", "ingest": "/api/knowledge-rag/ingest", "deleteSource": "/api/knowledge-rag/delete-source" }, "requestId": context.trace.request_id, "traceId": context.trace.trace_id, })), ) } pub async fn status( State(state): State, Extension(context): Extension, Query(query): Query, ) -> Result, WebError> { let (registry, registry_diagnostics) = if let Some(root_uri) = query.root_uri.as_deref().filter(|value| !value.is_empty()) { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(query.workspace_id.as_deref(), root_uri); let mut registry = read_registry(&root_path, &workspace_id, root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let diagnostics = knowledge_rag_source_content_diagnostics(®istry); (Some(registry), diagnostics) } else { (None, Value::Array(Vec::new())) }; let documents = match lightrag_json(reqwest::Method::GET, "/documents", None, true, &context).await { Ok(value) => json!({ "ok": true, "rawStatusGroups": lightrag_document_status_group_counts(&value), "documents": lightrag_document_summaries(&value), }), Err(error) => json!({ "ok": false, "code": error.code(), "message": error.message(), "rawStatusGroups": {}, "documents": [], }), }; let pipeline = match lightrag_json( reqwest::Method::GET, "/documents/pipeline_status", None, true, &context, ) .await { Ok(value) => lightrag_pipeline_status_summary(&value), Err(error) => json!({ "ok": false, "code": error.code(), "message": error.message(), }), }; let endpoint = lightrag_endpoint(); let health_raw = lightrag_json(reqwest::Method::GET, "/health", None, false, &context).await; let rerank_status = health_raw .as_ref() .ok() .map(lightrag_rerank_status_summary) .unwrap_or_else(|| { json!({ "enabled": false, "available": false, "binding": Value::Null, "model": Value::Null, "status": "unknown", }) }); let health = match health_raw { Ok(value) => json!({ "ok": true, "health": value, }), Err(error) => json!({ "ok": false, "code": error.code(), "message": error.message(), }), }; Ok(Json(json!({ "ok": true, "schema": "mnote.knowledge_rag.provider_status.v1", "provider": "lightrag", "endpoint": endpoint, "dashboardUrl": lightrag_dashboard_url(), "inputDir": lightrag_input_dir().display().to_string(), "health": health, "rerank": rerank_status, "documents": documents, "pipeline": pipeline, "registry": registry, "registryDiagnostics": registry_diagnostics, }))) } pub async fn pipeline_events( Extension(context): Extension, ) -> Result { let endpoint = lightrag_endpoint(); let url = format!( "{}/documents/pipeline_status/events", endpoint.trim_end_matches('/') ); let client = reqwest::Client::builder() .connect_timeout(Duration::from_secs(10)) .build() .map_err(|error| { WebError::internal(format!("LightRAG SSE client 初始化失败: {error}")) .with_context(&context) })?; let mut request = client.get(&url).header("accept", "text/event-stream"); if let Some(api_key) = lightrag_api_key() { request = request.header("X-API-Key", api_key); } let upstream = request.send().await.map_err(|error| { WebError::bad_gateway_code( "knowledge_rag_lightrag_events_unreachable", format!("无法连接 LightRAG pipeline 事件流: {error}"), ) .with_context(&context) })?; let upstream_status = upstream.status(); if !upstream_status.is_success() { let text = upstream.text().await.unwrap_or_default(); return Err(WebError::bad_gateway_code( "knowledge_rag_lightrag_events_error", format!("LightRAG pipeline 事件流返回 HTTP {upstream_status}: {text}"), ) .with_context(&context)); } let stream = upstream.bytes_stream().map_err(|error| { std::io::Error::new( std::io::ErrorKind::Other, format!("LightRAG pipeline 事件流读取失败: {error}"), ) }); Response::builder() .status(StatusCode::OK) .header(header::CONTENT_TYPE, "text/event-stream; charset=utf-8") .header(header::CACHE_CONTROL, "no-cache, no-transform") .header("x-accel-buffering", "no") .body(Body::from_stream(stream)) .map_err(|error| { WebError::internal(format!("LightRAG pipeline 事件响应构造失败: {error}")) .with_context(&context) }) } pub(crate) async fn sync_registry_for_root( state: &AppState, context: &RequestContext, root_uri: &str, workspace_id: Option<&str>, ) -> Result<(), WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( state, context, root_uri, ) .map_err(|error| error.with_context(context))?; let workspace_id = effective_workspace_id(workspace_id, root_uri); let mut registry = read_registry(&root_path, &workspace_id, root_uri)?; sync_registry_with_documents(&root_path, &mut registry, context).await } pub async fn ingest( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let input_dir = lightrag_input_dir(); fs::create_dir_all(&input_dir).map_err(|error| { WebError::bad_request_code( "knowledge_rag_input_dir_unavailable", format!("无法创建 LightRAG input 目录: {error}"), ) .with_context(&context) })?; let mut configured = Vec::new(); let mut expanded_count = 0usize; let mut seen_sources = BTreeSet::::new(); let force = body.force.unwrap_or(false); let mut requested_sources = body.sources; if requested_sources.is_empty() { requested_sources = registry .indexed_roots .iter() .map(|root| KnowledgeRagSourceInput { source_path: Some(root.root_relative_path.clone()), path: None, parser_hint: None, }) .collect(); } if requested_sources.is_empty() { return Err(WebError::bad_request_code( "knowledge_rag_source_required", "资料库索引缺少 sourcePath,且没有可重建的已保存索引范围", ) .with_context(&context)); } for source in requested_sources { let source_path = source.source_path.or(source.path).ok_or_else(|| { WebError::bad_request_code("knowledge_rag_source_required", "资料库索引缺少 sourcePath") .with_context(&context) })?; upsert_indexed_root_for_request(&mut registry, &root_path, &source_path, &context)?; let requested_parser_hint = normalize_parser_hint(source.parser_hint.as_deref(), &context)?; let resolved_sources = resolve_knowledge_rag_sources(&root_path, &source_path, &context)?; for resolved in resolved_sources { expanded_count += 1; if expanded_count > MAX_INGEST_SOURCES_PER_REQUEST { return Err(WebError::bad_request_code( "knowledge_rag_source_limit_exceeded", format!( "单次资料库索引最多支持 {MAX_INGEST_SOURCES_PER_REQUEST} 个文件,请缩小目录范围" ), ) .with_context(&context)); } let canonical = resolved.canonical_path; let canonical_key = canonical.display().to_string(); if !seen_sources.insert(canonical_key.clone()) { configured.push(json!({ "sourcePath": canonical_key, "requestedSourcePath": resolved.requested_path, "sourceKind": resolved.source_kind, "skipped": true, "skipReason": "duplicate_request", })); continue; } let relative = root_relative_path(&root_path, &canonical)?; let source_hash = source_hash(&canonical)?; let file_name = canonical .file_name() .and_then(|value| value.to_str()) .unwrap_or("source"); let parser_hint = requested_parser_hint .clone() .or_else(|| default_lightrag_parser_hint_for_source(&canonical, file_name)); let direct_scan_source = lightrag_scan_supported_file(&canonical); let reingest_doc_ids = if force { registry .entries .iter() .filter(|entry| { entry.source_path == canonical_key && entry.deleted_at_ms.is_none() && !entry.stale }) .filter_map(|entry| entry.light_rag_doc_id.clone()) .collect::>() } else { Vec::new() }; let provider_delete = if reingest_doc_ids.is_empty() { Value::Null } else { delete_lightrag_documents_for_reingest(reingest_doc_ids, &context).await? }; if !force { if let Some(existing) = registry.entries.iter().find(|entry| { entry.source_path == canonical_key && entry.source_hash == source_hash && entry.deleted_at_ms.is_none() && !entry.stale && (entry.indexed_at_ms.is_some() || matches!( entry.light_rag_status.as_deref(), Some("processing" | "pending" | "submitted" | "parsed") )) }) { configured.push(json!({ "sourceId": existing.source_id, "sourcePath": existing.source_path, "sourceRootRelativePath": existing.source_root_relative_path, "requestedSourcePath": resolved.requested_path, "sourceKind": resolved.source_kind, "lightRagFilePath": existing.light_rag_file_path, "lightRagDocId": existing.light_rag_doc_id, "lightRagStatus": existing.light_rag_status, "skipped": true, "skipReason": "already_registered", })); continue; } } let staged_source = stage_lightrag_source( &canonical, file_name, parser_hint.as_deref(), &input_dir, &context, )?; let light_rag_file_path = staged_source.light_rag_file_path.clone(); let symlink_path = staged_source.staged_path.clone(); let now = now_ms(); let source_id = format!( "lightrag-source-{}", short_hash(&canonical.display().to_string()) ); upsert_registry_entry( &mut registry, KnowledgeRagSourceRegistryEntry { source_id: source_id.clone(), workspace_id: workspace_id.clone(), root_uri: body.root_uri.clone(), source_path: canonical_key.clone(), source_root_relative_path: relative.clone(), source_hash: source_hash.clone(), light_rag_doc_id: None, light_rag_status: Some("submitted".into()), light_rag_file_path: light_rag_file_path.clone(), symlink_path: symlink_path.display().to_string(), parser_hint: parser_hint.clone(), indexed_at_ms: None, deleted_at_ms: None, stale: false, updated_at_ms: now, }, ); configured.push(json!({ "sourceId": source_id, "sourcePath": canonical_key, "sourceRootRelativePath": relative, "requestedSourcePath": resolved.requested_path, "sourceKind": resolved.source_kind, "lightRagFilePath": light_rag_file_path, "symlinkPath": symlink_path.display().to_string(), "parserHint": parser_hint, "providerDelete": provider_delete, "scanMode": if direct_scan_source { "direct" } else { "markdown_wrapper" }, "lightRagStatus": "submitted", })); } } write_registry(&root_path, &mut registry)?; let scan = lightrag_json( reqwest::Method::POST, "/documents/scan", None, true, &context, ) .await?; let scan_status = scan.get("status").and_then(Value::as_str).unwrap_or(""); let retry_required = scan_status == "scanning_skipped_pipeline_busy"; Ok(Json(json!({ "ok": !retry_required, "schema": "mnote.knowledge_rag.ingest_result.v1", "provider": "lightrag", "force": body.force.unwrap_or(false), "configuredSources": configured, "scan": scan, "retryRequired": retry_required, "registry": registry, }))) } pub async fn query_rag( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let query = body.question.or(body.query).ok_or_else(|| { WebError::bad_request_code("knowledge_rag_query_required", "资料库问答缺少 query") .with_context(&context) })?; validate_lightrag_provider_query_length(&query, "knowledge_rag_query_too_short", &context)?; let source_scope = normalize_source_scope(body.source_paths.as_deref()); let requested_mode = normalize_lightrag_query_mode(body.mode.as_deref()); let mode_decision = resolve_lightrag_query_mode_for_scope(®istry, &source_scope, &requested_mode); let mode = mode_decision.mode.clone(); let raw = lightrag_json( reqwest::Method::POST, "/query/data", Some(json!({ "query": query.clone(), "mode": mode.clone(), "top_k": body.top_k, "chunk_top_k": body.chunk_top_k, "include_references": true, "include_chunk_content": body.include_chunk_content.unwrap_or(true), })), true, &context, ) .await?; let mut references = mapped_references(&raw, ®istry, &body.root_uri, &root_path, Some(&query)); filter_mapped_references_by_source_scope(&mut references, &source_scope); let include_document_structure_index = body .include_document_structure_index .unwrap_or_else(|| mode_decision.reason == "source_scope_skip_kg_document"); let document_structure_index = if include_document_structure_index { document_structure_index_payload( &root_path, ®istry, &source_scope, &references, Some(&query), DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PER_QUERY, )? } else { Value::Null }; Ok(Json(json!({ "ok": true, "schema": "mnote.knowledge_rag.query_result.v1", "provider": "lightrag", "requestedRetrievalMode": requested_mode, "retrievalMode": mode, "effectiveRetrievalMode": mode_decision.mode, "retrievalModeReason": mode_decision.reason, "documentStructureIndexIncluded": include_document_structure_index, "documentStructureIndexReason": if include_document_structure_index { "requested_or_skip_kg_source_scope" } else { "not_requested" }, "sourceScope": source_scope, "sourceScopeMode": SOURCE_SCOPE_MODE_POST_FILTER, "rawScopeFiltered": false, "raw": raw, "citations": knowledge_rag_citations(&references), "references": references, "documentStructureIndex": document_structure_index, }))) } pub async fn search( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let query = body.query.trim().to_string(); if query.is_empty() { return Err(WebError::bad_request_code( "knowledge_rag_search_query_required", "资料库检索缺少 query", ) .with_context(&context)); } validate_lightrag_provider_query_length( &query, "knowledge_rag_search_query_too_short", &context, )?; let source_scope = normalize_source_scope(body.source_paths.as_deref()); let requested_search_mode = normalize_knowledge_rag_search_mode(body.mode.as_deref()); let mode_decision = resolve_knowledge_rag_search_mode_for_scope( ®istry, &source_scope, &requested_search_mode, ); let search_mode = mode_decision.mode.clone(); let raw = if search_mode == "exact" { lightrag_json( reqwest::Method::POST, "/query/search", Some(json!({ "query": query.clone(), "limit": body.top_k.or(body.chunk_top_k).unwrap_or(50), "max_per_chunk": 12, "include_chunk_content": body.include_chunk_content.unwrap_or(true), "include_sidecar": true, })), true, &context, ) .await? } else { lightrag_json( reqwest::Method::POST, "/query/data", Some(json!({ "query": query.clone(), "mode": search_mode.clone(), "top_k": body.top_k, "chunk_top_k": body.chunk_top_k, "include_references": true, "include_chunk_content": body.include_chunk_content.unwrap_or(true), })), true, &context, ) .await? }; let mut references = mapped_references(&raw, ®istry, &body.root_uri, &root_path, Some(&query)); filter_mapped_references_by_source_scope(&mut references, &source_scope); if search_mode == "exact" { filter_mapped_references_by_search_query(&mut references, &query); rank_mapped_references_for_query(&mut references, &query); } dedupe_mapped_references_by_locator(&mut references); let results = references .iter() .enumerate() .map(|(index, reference)| { knowledge_rag_search_result(reference, index, &body.root_uri, &query) }) .collect::>(); Ok(Json(json!({ "ok": true, "schema": "mnote.knowledge_rag.search_results.v1", "provider": "lightrag", "query": query, "requestedRetrievalMode": requested_search_mode, "retrievalMode": search_mode, "effectiveRetrievalMode": mode_decision.mode, "retrievalModeReason": mode_decision.reason, "rawRetrievalMode": raw .get("metadata") .and_then(|metadata| metadata.get("query_mode")) .cloned() .unwrap_or(Value::Null), "sourceScope": source_scope, "sourceScopeMode": SOURCE_SCOPE_MODE_POST_FILTER, "rawScopeFiltered": false, "results": results, "citations": knowledge_rag_citations(&references), "references": references, "registry": { "schema": registry.schema, "workspaceId": registry.workspace_id, "rootUri": registry.root_uri, "updatedAtMs": registry.updated_at_ms, "indexedRoots": registry.indexed_roots, "sourceCount": registry.entries.len(), }, }))) } fn knowledge_rag_search_result( reference: &Value, index: usize, root_uri: &str, query: &str, ) -> Value { let source_path = reference .get("sourceRootRelativePath") .and_then(Value::as_str) .or_else(|| reference.get("filePath").and_then(Value::as_str)) .unwrap_or_default(); let title = Path::new(source_path) .file_name() .and_then(|value| value.to_str()) .unwrap_or(source_path); let chunk_id = reference .get("chunkId") .and_then(Value::as_str) .or_else(|| { reference .get("chunkId") .and_then(|value| value.get("id")) .and_then(Value::as_str) }) .unwrap_or_default(); let source_id = reference .get("sourceId") .and_then(Value::as_str) .unwrap_or_default(); let id = if !source_id.is_empty() || !chunk_id.is_empty() { format!("knowledge-rag:{source_id}:{chunk_id}") } else { format!("knowledge-rag:reference:{index}") }; let quote = reference_display_quote(reference); let locator_precision = locator_precision_for_reference(reference); json!({ "id": id, "documentId": id, "citationId": reference.get("citationId").cloned().unwrap_or(Value::Null), "citationLabel": reference.get("citationLabel").cloned().unwrap_or(Value::Null), "title": title, "path": source_path, "resourceType": resource_type_for_path(source_path), "sourceKind": "local_folder", "rootUri": root_uri, "snippet": quote.chars().take(220).collect::(), "quote": quote, "rawQuote": reference.get("rawQuote").cloned().unwrap_or(Value::Null), "displayQuote": reference.get("displayQuote").cloned().unwrap_or_else(|| json!(quote)), "locatorEvidenceText": reference.get("locatorEvidenceText").cloned().unwrap_or(Value::Null), "normalizedFingerprint": reference.get("normalizedFingerprint").cloned().unwrap_or(Value::Null), "citationDiagnostics": reference.get("citationDiagnostics").cloned().unwrap_or(Value::Null), "matchSource": reference.get("matchSource").and_then(Value::as_str).unwrap_or("lightrag_reference"), "occurrenceIndex": reference.get("occurrenceIndex").cloned().unwrap_or(Value::Null), "provider": "lightrag", "query": query, "hasOcr": true, "locator": reference.get("locator").cloned().unwrap_or(Value::Null), "locatorDegraded": reference.get("locatorDegraded").cloned().unwrap_or_else(|| json!(true)), "locatorPrecision": locator_precision, "citationUrl": reference.get("citationUrl").cloned().unwrap_or(Value::Null), "citationMarkdown": reference.get("citationMarkdown").cloned().unwrap_or(Value::Null), "publicPath": reference.get("citationUrl").cloned().unwrap_or(Value::Null), "openAction": reference.get("openAction").cloned().unwrap_or(Value::Null), "source": { "locator": reference.get("locator").cloned().unwrap_or(Value::Null), "reference": reference, } }) } fn knowledge_rag_citations(references: &[Value]) -> Vec { references .iter() .enumerate() .filter_map(|(index, reference)| knowledge_rag_citation(reference, index)) .collect() } fn knowledge_rag_citation(reference: &Value, index: usize) -> Option { let citation_url = reference.get("citationUrl").cloned().unwrap_or(Value::Null); if citation_url.is_null() && reference.get("locator").is_none_or(Value::is_null) { return None; } let citation_id = reference .get("citationId") .and_then(Value::as_str) .map(ToOwned::to_owned) .unwrap_or_else(|| { short_citation_id( reference .get("sourceId") .and_then(Value::as_str) .unwrap_or_default(), reference .get("chunkId") .and_then(Value::as_str) .unwrap_or_default(), reference .get("locator") .and_then(|locator| locator.get("blockId")) .and_then(Value::as_str) .unwrap_or_default(), index as u64, reference .get("normalizedFingerprint") .and_then(Value::as_str) .unwrap_or_default(), ) }); Some(json!({ "schema": "mnote.knowledge_rag.citation.v1", "provider": reference.get("provider").cloned().unwrap_or_else(|| json!("lightrag")), "citationId": citation_id.clone(), "citationLabel": format!("[{}]", citation_id), "sourceId": reference.get("sourceId").cloned().unwrap_or(Value::Null), "sourcePath": reference.get("sourcePath").cloned().unwrap_or(Value::Null), "sourceRootRelativePath": reference.get("sourceRootRelativePath").cloned().unwrap_or(Value::Null), "lightRagFilePath": reference.get("filePath").cloned().unwrap_or(Value::Null), "lightRagChunkId": reference.get("chunkId").cloned().unwrap_or(Value::Null), "blockId": reference .get("locator") .and_then(|locator| locator.get("blockId")) .cloned() .unwrap_or(Value::Null), "headingPath": reference.get("headingPath").cloned().unwrap_or_else(|| json!([])), "rawQuote": reference.get("rawQuote").cloned().unwrap_or(Value::Null), "displayQuote": reference.get("displayQuote").cloned().unwrap_or_else(|| json!(reference_display_quote(reference))), "locatorEvidenceText": reference.get("locatorEvidenceText").cloned().unwrap_or_else(|| json!(reference_locator_evidence_text(reference))), "searchQuery": reference.get("searchQuery").cloned().unwrap_or(Value::Null), "quoteSource": reference.get("quoteSource").cloned().unwrap_or(Value::Null), "locatorPrecision": reference.get("locatorPrecision").cloned().unwrap_or_else(|| json!(locator_precision_for_reference(reference))), "locatorDegraded": reference.get("locatorDegraded").cloned().unwrap_or_else(|| json!(true)), "locator": reference.get("locator").cloned().unwrap_or(Value::Null), "citationUrl": citation_url, "citationMarkdown": reference.get("citationMarkdown").cloned().unwrap_or(Value::Null), "relevanceScore": reference.get("relevanceScore").cloned().unwrap_or(Value::Null), "diagnostics": reference.get("citationDiagnostics").cloned().unwrap_or(Value::Null), })) } fn heading_path_for_sidecar_block(block: Option<&Value>) -> Vec { let Some(block) = block else { return Vec::new(); }; let mut headings = block .get("parent_headings") .or_else(|| block.get("parentHeadings")) .and_then(Value::as_array) .map(|items| { items .iter() .filter_map(Value::as_str) .map(clean_lightrag_text_for_display) .filter(|value| !value.is_empty()) .collect::>() }) .unwrap_or_default(); if let Some(heading) = block .get("heading") .and_then(Value::as_str) .map(clean_lightrag_text_for_display) .filter(|value| !value.is_empty()) { if headings.last().is_none_or(|last| last != &heading) { headings.push(heading); } } headings } fn locator_precision_for_reference(reference: &Value) -> &'static str { let locator = reference.get("locator").unwrap_or(&Value::Null); locator_precision_for_locator_value(locator) } fn locator_precision_for_locator_value(locator: &Value) -> &'static str { if locator.is_null() { return "file"; } let has_bbox = locator.get("bbox").is_some_and(|value| !value.is_null()); let has_page = locator .get("page") .and_then(Value::as_u64) .is_some_and(|page| page > 0); if has_bbox && has_page { return "bbox"; } if has_page { return "page"; } let has_block = locator .get("blockId") .and_then(Value::as_str) .is_some_and(|value| !value.trim().is_empty()); if has_block { return "paragraph"; } "file" } fn locator_precision_for_locator(locator: &EvidenceLocator) -> &'static str { if locator.page.is_some() && locator.bbox.is_some() { return "bbox"; } if locator.page.is_some() { return "page"; } if locator .block_id .as_deref() .is_some_and(|value| !value.trim().is_empty()) { return "paragraph"; } "file" } fn non_whitespace_char_count(query: &str) -> usize { query.chars().filter(|ch| !ch.is_whitespace()).count() } fn validate_lightrag_provider_query_length( query: &str, code: &'static str, context: &RequestContext, ) -> Result<(), WebError> { if non_whitespace_char_count(query) >= LIGHTRAG_PROVIDER_MIN_QUERY_CHARS { return Ok(()); } Err(WebError::bad_request_code(code, "请输入至少 2 个字再搜索").with_context(context)) } fn normalize_source_scope(source_paths: Option<&[String]>) -> Vec { source_paths .unwrap_or(&[]) .iter() .map(|value| value.trim().trim_matches('/').replace('\\', "/")) .filter(|value| !value.is_empty() && value != ".") .collect::>() .into_iter() .collect() } fn filter_mapped_references_by_source_scope(references: &mut Vec, source_scope: &[String]) { if source_scope.is_empty() { return; } references.retain(|reference| { let source_path = reference .get("sourceRootRelativePath") .and_then(Value::as_str) .unwrap_or_default() .trim_matches('/') .replace('\\', "/"); source_scope .iter() .any(|scope| source_path == *scope || source_path.starts_with(&format!("{scope}/"))) }); } fn filter_mapped_references_by_search_query(references: &mut Vec, query: &str) { let query_normalized = normalize_text_for_match(query).to_ascii_lowercase(); if query_normalized.is_empty() { return; } references .retain(|reference| mapped_reference_matches_search_query(reference, &query_normalized)); } fn mapped_reference_matches_search_query(reference: &Value, query_normalized: &str) -> bool { let quote = reference_display_quote(reference); let source_path = reference .get("sourceRootRelativePath") .and_then(Value::as_str) .unwrap_or_default(); let combined = normalize_text_for_match(&format!("{source_path}\n{quote}")).to_ascii_lowercase(); if combined.contains(query_normalized) { return true; } let terms = query_match_terms(query_normalized); if query_normalized.chars().any(is_cjk_char) { if query_is_single_cjk_lookup(query_normalized) { return false; } return terms .iter() .filter(|term| significant_cjk_query_term(term)) .any(|term| block_matches_query_term(&combined, term)); } !terms.is_empty() && terms .iter() .all(|term| block_matches_query_term(&combined, term)) } fn dedupe_mapped_references_by_locator(references: &mut Vec) { let mut seen = BTreeSet::::new(); references.retain(|reference| { let source = reference .get("sourceRootRelativePath") .and_then(Value::as_str) .or_else(|| reference.get("filePath").and_then(Value::as_str)) .unwrap_or_default(); let block_id = reference .get("locator") .and_then(|locator| locator.get("blockId")) .and_then(Value::as_str) .unwrap_or_default(); let chunk_id = reference .get("chunkId") .and_then(Value::as_str) .unwrap_or_default(); let fingerprint = reference .get("normalizedFingerprint") .and_then(Value::as_str) .filter(|value| !value.trim().is_empty()) .map(ToOwned::to_owned) .unwrap_or_else(|| normalized_quote_fingerprint(reference_display_quote(reference))); let key = if !block_id.is_empty() { format!("{source}\nblock:{block_id}") } else if !chunk_id.is_empty() { format!("{source}\nchunk:{chunk_id}\nfingerprint:{fingerprint}") } else { format!("{source}\nquote:{fingerprint}") }; seen.insert(key) }); } fn rank_mapped_references_for_query(references: &mut Vec, query: &str) { let mut indexed = references .drain(..) .enumerate() .map(|(index, reference)| { let score = reference_query_score(&reference, query); (index, score, reference) }) .collect::>(); indexed.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(&right.0))); references.extend(indexed.into_iter().map(|(_, _, reference)| reference)); } fn reference_query_score(reference: &Value, query: &str) -> i64 { let query_normalized = normalize_text_for_match(query).to_ascii_lowercase(); if query_normalized.is_empty() { return 0; } let tokens = query_match_terms(&query_normalized); let source_path = reference .get("sourceRootRelativePath") .and_then(Value::as_str) .unwrap_or_default() .to_ascii_lowercase(); let file_path = reference .get("filePath") .and_then(Value::as_str) .unwrap_or_default() .to_ascii_lowercase(); let quote = reference_display_quote(reference).to_ascii_lowercase(); let citation = reference .get("citationMarkdown") .and_then(Value::as_str) .unwrap_or_default() .to_ascii_lowercase(); let mut score = 0; if quote.contains(&query_normalized) { score += 100; } if source_path.contains(&query_normalized) || file_path.contains(&query_normalized) { score += 60; } for token in tokens { if source_path.contains(&token) { score += 20; } if file_path.contains(&token) { score += 12; } if quote.contains(&token) { score += 8; } if citation.contains(&token) { score += 4; } } score } fn normalize_lightrag_query_mode(mode: Option<&str>) -> String { match mode .map(|value| value.trim().to_ascii_lowercase()) .filter(|value| !value.is_empty()) .as_deref() { Some("local") => "local".into(), Some("global") => "global".into(), Some("hybrid") => "hybrid".into(), Some("naive") => "naive".into(), Some("mix") => "mix".into(), Some("bypass") => "bypass".into(), Some("exact" | "keyword" | "full_text" | "full-text") => "naive".into(), Some(_) => "mix".into(), None => "mix".into(), } } #[derive(Debug, Clone, PartialEq, Eq)] struct LightRagQueryModeDecision { mode: String, reason: String, } fn resolve_lightrag_query_mode_for_scope( registry: &KnowledgeRagSourceRegistry, source_scope: &[String], requested_mode: &str, ) -> LightRagQueryModeDecision { if lightrag_query_scope_has_skip_kg_document(registry, source_scope) && !matches!(requested_mode, "naive" | "bypass") { return LightRagQueryModeDecision { mode: "naive".into(), reason: "source_scope_skip_kg_document".into(), }; } LightRagQueryModeDecision { mode: requested_mode.to_string(), reason: "requested_mode".into(), } } fn lightrag_query_scope_has_skip_kg_document( registry: &KnowledgeRagSourceRegistry, source_scope: &[String], ) -> bool { if source_scope.is_empty() { return false; } registry.entries.iter().any(|entry| { !entry.stale && entry.deleted_at_ms.is_none() && entry_matches_source_scope(entry, source_scope) && registry_entry_has_skip_kg_parser_hint(entry) }) } fn entry_matches_source_scope( entry: &KnowledgeRagSourceRegistryEntry, source_scope: &[String], ) -> bool { let source_path = entry .source_root_relative_path .trim_matches('/') .replace('\\', "/"); source_scope .iter() .any(|scope| source_path == *scope || source_path.starts_with(&format!("{scope}/"))) } fn registry_entry_has_skip_kg_parser_hint(entry: &KnowledgeRagSourceRegistryEntry) -> bool { entry .parser_hint .as_deref() .is_some_and(parser_hint_contains_skip_kg) || parser_hint_segment_from_file_name(&entry.light_rag_file_path) .as_deref() .is_some_and(parser_hint_contains_skip_kg) } fn parser_hint_contains_skip_kg(hint: &str) -> bool { normalize_supported_parser_hint(hint) .as_deref() .is_some_and(|normalized| { normalized .split_once('-') .map(|(_, options)| options) .or_else(|| normalized.strip_prefix('-')) .is_some_and(|options| options.contains('!')) }) } fn parser_hint_segment_from_file_name(file_name: &str) -> Option { let mut search_start = 0usize; while let Some(offset) = file_name.get(search_start..)?.find(".[") { let start = search_start + offset; let hint_start = start + 2; let Some(end_offset) = file_name.get(hint_start..)?.find(']') else { break; }; let hint_end = hint_start + end_offset; let hint = file_name.get(hint_start..hint_end)?; if normalize_supported_parser_hint(hint).is_some() { return Some(hint.to_string()); } search_start = hint_end + 1; } None } fn normalize_knowledge_rag_search_mode(mode: Option<&str>) -> String { match mode .map(|value| value.trim().to_ascii_lowercase()) .filter(|value| !value.is_empty()) .as_deref() { Some("exact" | "keyword" | "keywords" | "full_text" | "full-text") => "exact".into(), Some("local") => "local".into(), Some("global") => "global".into(), Some("hybrid") => "hybrid".into(), Some("naive" | "vector") => "naive".into(), Some("mix" | "mixed") => "mix".into(), Some(_) => "mix".into(), None => "exact".into(), } } fn resolve_knowledge_rag_search_mode_for_scope( registry: &KnowledgeRagSourceRegistry, source_scope: &[String], requested_mode: &str, ) -> LightRagQueryModeDecision { if lightrag_query_scope_has_skip_kg_document(registry, source_scope) && !matches!(requested_mode, "exact" | "naive" | "bypass") { return LightRagQueryModeDecision { mode: "naive".into(), reason: "source_scope_skip_kg_document".into(), }; } LightRagQueryModeDecision { mode: requested_mode.to_string(), reason: "requested_mode".into(), } } pub async fn open_reference( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let file_path = body .file_path .or_else(|| { body.reference .as_ref() .and_then(|value| value.get("file_path")) .and_then(Value::as_str) .map(ToOwned::to_owned) }) .ok_or_else(|| { WebError::bad_request_code( "knowledge_rag_reference_file_path_required", "资料库引用打开缺少 filePath", ) .with_context(&context) })?; let reference = map_reference_plan( &json!({ "reference_id": body.reference_id, "file_path": file_path, "chunk_id": body.chunk_id, }), ®istry, &body.root_uri, &root_path, None, ); Ok(Json(json!({ "ok": true, "schema": "mnote.knowledge_rag.open_reference_result.v1", "reference": reference, "registry": registry, }))) } pub async fn section_context( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let payload = sidecar_section_context_payload(&root_path, ®istry, &body) .map_err(|error| error.with_context(&context))?; Ok(Json(payload)) } pub async fn delete_source( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let source_path = body.source_path.as_deref().map(str::trim); let doc_id = body.light_rag_doc_id.as_deref().map(str::trim); let now = now_ms(); let mut matched_doc_ids = Vec::new(); for entry in &mut registry.entries { let source_matches = source_path .filter(|value| !value.is_empty()) .is_some_and(|value| { value == entry.source_path || value == entry.source_root_relative_path }); let doc_matches = doc_id .filter(|value| !value.is_empty()) .is_some_and(|value| entry.light_rag_doc_id.as_deref() == Some(value)); if source_matches || doc_matches { if let Some(doc_id) = entry.light_rag_doc_id.clone() { matched_doc_ids.push(doc_id); } entry.stale = true; entry.deleted_at_ms = Some(now); entry.light_rag_status = Some( if entry.light_rag_doc_id.is_some() { "delete_submitted" } else { "delete_completed" } .into(), ); entry.updated_at_ms = now; } } matched_doc_ids.sort(); matched_doc_ids.dedup(); let delete_result = if matched_doc_ids.is_empty() { json!({ "status": "no_lightrag_doc", "message": "没有可删除的 LightRAG doc id;registry 已按 source 标记 stale", }) } else { lightrag_json( reqwest::Method::DELETE, "/documents/delete_document", Some(json!({ "doc_ids": matched_doc_ids, "delete_file": false, "delete_llm_cache": false, })), true, &context, ) .await? }; write_registry(&root_path, &mut registry)?; Ok(Json(json!({ "ok": true, "schema": "mnote.knowledge_rag.delete_source_result.v1", "provider": "lightrag", "deleteResult": delete_result, "registry": registry, }))) } pub async fn prune_registry( State(state): State, Extension(context): Extension, Json(body): Json, ) -> Result, WebError> { let root_path = local_folder_source::ensure_local_workspace_read_access_with_state( &state, &context, &body.root_uri, ) .map_err(|error| error.with_context(&context))?; let workspace_id = effective_workspace_id(body.workspace_id.as_deref(), &body.root_uri); let mut registry = read_registry(&root_path, &workspace_id, &body.root_uri)?; sync_registry_with_documents(&root_path, &mut registry, &context).await?; let before = registry.entries.len(); registry .entries .retain(|entry| !knowledge_rag_registry_entry_prunable(entry)); let removed = before.saturating_sub(registry.entries.len()); write_registry(&root_path, &mut registry)?; Ok(Json(json!({ "ok": true, "schema": "mnote.knowledge_rag.prune_registry_result.v1", "provider": "lightrag", "removed": removed, "registry": registry, }))) } fn knowledge_rag_registry_entry_prunable(entry: &KnowledgeRagSourceRegistryEntry) -> bool { if matches!( entry.light_rag_status.as_deref(), Some("delete_submitted" | "delete_retry_required") ) { return false; } if entry.light_rag_doc_id.is_some() && (entry.deleted_at_ms.is_some() || entry.stale) { return false; } entry.deleted_at_ms.is_some() || entry.stale || matches!( entry.light_rag_status.as_deref(), Some("delete_completed" | "failed") ) } fn knowledge_rag_provider_delete_confirmed(entry: &KnowledgeRagSourceRegistryEntry) -> bool { entry.light_rag_doc_id.is_some() && (entry.deleted_at_ms.is_some() || entry.stale || matches!( entry.light_rag_status.as_deref(), Some("delete_submitted" | "delete_retry_required") )) } fn mark_registry_entry_delete_completed(entry: &mut KnowledgeRagSourceRegistryEntry, now: u128) { entry.light_rag_doc_id = None; entry.indexed_at_ms = None; entry.light_rag_status = Some("delete_completed".into()); entry.updated_at_ms = now; } async fn sync_registry_with_documents( root_path: &Path, registry: &mut KnowledgeRagSourceRegistry, context: &RequestContext, ) -> Result<(), WebError> { if registry.entries.is_empty() { return Ok(()); } let docs = lightrag_json(reqwest::Method::GET, "/documents", None, true, context).await?; let by_file_path = lightrag_documents_by_file_path(&docs); let now = now_ms(); let mut changed = false; let mut retry_doc_ids = Vec::new(); for entry in &mut registry.entries { if let Some(doc) = document_for_registry_entry(&by_file_path, entry) { if let Some(id) = doc.get("id").and_then(Value::as_str) { entry.light_rag_doc_id = Some(id.to_string()); } let delete_pending = matches!( entry.light_rag_status.as_deref(), Some("delete_submitted" | "delete_retry_required") ) || entry.deleted_at_ms.is_some(); if delete_pending { if let Some(doc_id) = entry.light_rag_doc_id.clone() { retry_doc_ids.push(doc_id); } if entry.light_rag_status.is_none() { entry.light_rag_status = Some("delete_submitted".into()); } } else if let Some(status) = doc.get("status").and_then(Value::as_str) { entry.light_rag_status = Some(status.to_string()); } if !entry.stale && entry.deleted_at_ms.is_none() && doc.get("status").and_then(Value::as_str) == Some("processed") { entry.indexed_at_ms.get_or_insert(now); entry.stale = !Path::new(&entry.source_path).exists(); } entry.updated_at_ms = now; changed = true; } else if knowledge_rag_provider_delete_confirmed(entry) { mark_registry_entry_delete_completed(entry, now); changed = true; } else if entry.deleted_at_ms.is_some() && entry.light_rag_doc_id.is_none() && entry.light_rag_status.as_deref() != Some("delete_completed") { entry.indexed_at_ms = None; entry.light_rag_status = Some("delete_completed".into()); entry.updated_at_ms = now; changed = true; } else if !Path::new(&entry.source_path).exists() { entry.stale = true; entry.updated_at_ms = now; changed = true; } } let mut stale_doc_ids = sync_registry_source_state(registry, now)?; stale_doc_ids.extend(retry_doc_ids); stale_doc_ids.sort(); stale_doc_ids.dedup(); if !stale_doc_ids.is_empty() { let delete_result = lightrag_json( reqwest::Method::DELETE, "/documents/delete_document", Some(json!({ "doc_ids": stale_doc_ids, "delete_file": false, "delete_llm_cache": false, })), true, context, ) .await; for entry in &mut registry.entries { if entry .light_rag_doc_id .as_deref() .is_some_and(|doc_id| stale_doc_ids.iter().any(|item| item == doc_id)) { entry.light_rag_status = Some( if delete_result.is_err() { "delete_retry_required" } else { "delete_submitted" } .into(), ); entry.updated_at_ms = now; } } changed = true; } if changed { write_registry(root_path, registry)?; refresh_document_structure_index(root_path, registry)?; } Ok(()) } async fn delete_lightrag_documents_for_reingest( doc_ids: Vec, context: &RequestContext, ) -> Result { let mut doc_ids = doc_ids .into_iter() .map(|value| value.trim().to_string()) .filter(|value| !value.is_empty()) .collect::>(); doc_ids.sort(); doc_ids.dedup(); if doc_ids.is_empty() { return Ok(Value::Null); } let delete_result = lightrag_json( reqwest::Method::DELETE, "/documents/delete_document", Some(json!({ "doc_ids": doc_ids, "delete_file": false, "delete_llm_cache": false, })), true, context, ) .await?; if delete_result.get("status").and_then(Value::as_str) == Some("busy") { return Err(WebError::bad_request_code( "knowledge_rag_force_reingest_provider_busy", "LightRAG 当前仍在处理旧任务,不能安全强制重建;请等待 pipeline idle 后重试", ) .with_context(context)); } wait_lightrag_pipeline_idle_for_reingest(context).await?; Ok(delete_result) } async fn wait_lightrag_pipeline_idle_for_reingest( context: &RequestContext, ) -> Result<(), WebError> { for _ in 0..180 { let status = lightrag_json( reqwest::Method::GET, "/documents/pipeline_status", None, true, context, ) .await?; let busy = status.get("busy").and_then(Value::as_bool).unwrap_or(false); let scanning = status .get("scanning") .and_then(Value::as_bool) .unwrap_or(false); let destructive_busy = status .get("destructive_busy") .and_then(Value::as_bool) .unwrap_or(false); if !busy && !scanning && !destructive_busy { return Ok(()); } tokio::time::sleep(Duration::from_secs(1)).await; } Err(WebError::bad_gateway_code( "knowledge_rag_force_reingest_delete_timeout", "LightRAG 删除旧 doc 后 pipeline 未在 180 秒内恢复 idle,已停止本次重建以避免复用旧索引", ) .with_context(context)) } fn lightrag_document_status_group_counts(docs: &Value) -> BTreeMap { let mut counts = BTreeMap::new(); for (status, items) in docs .get("statuses") .and_then(Value::as_object) .into_iter() .flat_map(|map| map.iter()) { counts.insert(status.clone(), items.as_array().map_or(0, Vec::len)); } counts } fn lightrag_document_summaries(docs: &Value) -> Vec { let mut items = Vec::new(); for (status_group, doc) in docs .get("statuses") .and_then(Value::as_object) .into_iter() .flat_map(|map| map.iter()) .flat_map(|(status, items)| { items .as_array() .into_iter() .flat_map(move |docs| docs.iter().map(move |doc| (status, doc))) }) { items.push(json!({ "id": doc.get("id").and_then(Value::as_str).unwrap_or_default(), "filePath": doc.get("file_path").and_then(Value::as_str).unwrap_or_default(), "status": doc.get("status").and_then(Value::as_str).unwrap_or(status_group), "statusGroup": status_group, "summary": doc.get("summary").and_then(Value::as_str).unwrap_or_default(), "chunksCount": doc.get("chunks_count").or_else(|| doc.get("chunks")).cloned().unwrap_or(Value::Null), "createdAt": doc.get("created_at").or_else(|| doc.get("created")).cloned().unwrap_or(Value::Null), "updatedAt": doc.get("updated_at").or_else(|| doc.get("updated")).cloned().unwrap_or(Value::Null), })); } items } fn lightrag_pipeline_status_summary(value: &Value) -> Value { let latest_message = value .get("latest_message") .and_then(Value::as_str) .unwrap_or_default(); let history_messages = lightrag_pipeline_history_messages(value); let progress = parse_lightrag_chunk_progress(latest_message).or_else(|| { history_messages .iter() .rev() .find_map(|message| parse_lightrag_chunk_progress(message)) }); json!({ "ok": true, "busy": value.get("busy").and_then(Value::as_bool).unwrap_or(false), "destructiveBusy": value.get("destructive_busy").and_then(Value::as_bool).unwrap_or(false), "scanning": value.get("scanning").and_then(Value::as_bool).unwrap_or(false), "scanningExclusive": value.get("scanning_exclusive").and_then(Value::as_bool).unwrap_or(false), "requestPending": value.get("request_pending").and_then(Value::as_bool).unwrap_or(false), "pendingEnqueues": value.get("pending_enqueues").and_then(Value::as_u64).unwrap_or(0), "pendingRequests": value.get("pending_requests").and_then(Value::as_bool).unwrap_or(false), "docs": value.get("docs").and_then(Value::as_u64).unwrap_or(0), "batches": value.get("batchs").and_then(Value::as_u64).unwrap_or(0), "currentBatch": value.get("cur_batch").and_then(Value::as_u64).unwrap_or(0), "jobName": value.get("job_name").and_then(Value::as_str).unwrap_or_default(), "jobStart": value.get("job_start").cloned().unwrap_or(Value::Null), "latestMessage": latest_message, "historyMessages": history_messages, "cancellationRequested": value.get("cancellation_requested").and_then(Value::as_bool).unwrap_or(false), "cancellationReason": value.get("cancellation_reason").cloned().unwrap_or(Value::Null), "progress": progress.map(|progress| json!({ "current": progress.current, "total": progress.total, "docId": progress.doc_id, })).unwrap_or(Value::Null), }) } fn lightrag_pipeline_history_messages(value: &Value) -> Vec { value .get("history_messages") .and_then(Value::as_array) .map(|messages| { messages .iter() .filter_map(Value::as_str) .rev() .take(80) .map(ToOwned::to_owned) .collect::>() .into_iter() .rev() .collect() }) .unwrap_or_default() } fn lightrag_rerank_status_summary(value: &Value) -> Value { let configuration = value.get("configuration").unwrap_or(&Value::Null); let queue = value.get("rerank_queue_status").unwrap_or(&Value::Null); let enabled = configuration .get("enable_rerank") .and_then(Value::as_bool) .unwrap_or(false); let available = queue .get("available") .and_then(Value::as_bool) .unwrap_or(false); let binding = configuration .get("rerank_binding") .cloned() .unwrap_or(Value::Null); let model = configuration .get("rerank_model") .cloned() .unwrap_or(Value::Null); json!({ "enabled": enabled, "available": available, "providerRerankEnabled": enabled, "providerRerankAvailable": available, "rerankModel": model.clone(), "binding": binding, "model": model, "minScore": configuration.get("min_rerank_score").cloned().unwrap_or(Value::Null), "status": if enabled && available { "available" } else if enabled { "unavailable" } else { "disabled" }, "implementation": "provider_status_only", }) } #[derive(Debug, Clone, PartialEq, Eq)] struct LightRagChunkProgress { current: u64, total: u64, doc_id: String, } fn parse_lightrag_chunk_progress(message: &str) -> Option { let rest = message.trim().strip_prefix("Chunk ")?; let (current, rest) = rest.split_once(" of ")?; let current = current.trim().parse::().ok()?; let (total, _) = rest.split_once(' ')?; let total = total.trim().parse::().ok()?; let doc_token = message.split_whitespace().find(|token| { token.starts_with("doc-") && (token.contains("-chunk-") || token.contains("-mm-")) })?; let doc_id = doc_token .rsplit_once("-chunk-") .or_else(|| doc_token.rsplit_once("-mm-")) .map(|(doc_id, _)| doc_id) .unwrap_or(doc_token) .trim_end_matches(|ch: char| !ch.is_ascii_alphanumeric()) .to_string(); if total == 0 || current == 0 || doc_id.is_empty() { return None; } Some(LightRagChunkProgress { current: current.min(total), total, doc_id, }) } fn lightrag_documents_by_file_path(docs: &Value) -> BTreeMap { let mut by_file_path = BTreeMap::::new(); for doc in docs .get("statuses") .and_then(Value::as_object) .into_iter() .flat_map(|map| map.values()) .filter_map(Value::as_array) .flat_map(|items| items.iter()) { if let Some(file_path) = doc.get("file_path").and_then(Value::as_str) { by_file_path.insert(file_path.to_string(), doc.clone()); } } by_file_path } fn sync_registry_source_state( registry: &mut KnowledgeRagSourceRegistry, now: u128, ) -> Result, WebError> { let mut stale_doc_ids = Vec::new(); for entry in &mut registry.entries { if entry.deleted_at_ms.is_some() { continue; } let Some(doc_id) = entry.light_rag_doc_id.clone() else { continue; }; let source_path = Path::new(&entry.source_path); if !source_path.exists() { entry.stale = true; entry.deleted_at_ms = Some(now); entry.indexed_at_ms = None; entry.light_rag_status = Some("delete_submitted".into()); entry.updated_at_ms = now; stale_doc_ids.push(doc_id); continue; } let current_hash = source_hash(source_path)?; if current_hash != entry.source_hash { entry.stale = true; entry.source_hash = current_hash; entry.indexed_at_ms = None; entry.light_rag_status = Some("delete_submitted".into()); entry.updated_at_ms = now; stale_doc_ids.push(doc_id); } } stale_doc_ids.sort(); stale_doc_ids.dedup(); Ok(stale_doc_ids) } async fn lightrag_json( method: reqwest::Method, path: &str, body: Option, use_api_key: bool, context: &RequestContext, ) -> Result { let endpoint = lightrag_endpoint(); let url = format!("{}{}", endpoint.trim_end_matches('/'), path); let client = reqwest::Client::builder() .timeout(Duration::from_secs(180)) .build() .map_err(|error| { WebError::internal(format!("LightRAG HTTP client 初始化失败: {error}")) .with_context(context) })?; let api_key = use_api_key.then(lightrag_api_key).flatten(); for attempt in 0..2 { let mut request = client.request(method.clone(), &url); if let Some(api_key) = api_key.as_deref() { request = request.header("X-API-Key", api_key); } if let Some(body) = body.as_ref() { request = request.json(body); } let response = request.send().await.map_err(|error| { WebError::bad_gateway_code( "knowledge_rag_lightrag_unreachable", format!("无法访问 LightRAG provider: {error}"), ) .with_context(context) })?; let status = response.status(); let text = response.text().await.unwrap_or_default(); if !status.is_success() { if attempt == 0 && should_retry_lightrag_provider_error(status, &text) { tokio::time::sleep(Duration::from_millis(700)).await; continue; } return Err(WebError::bad_gateway_code( "knowledge_rag_lightrag_error", format!("LightRAG provider 返回 HTTP {status}: {text}"), ) .with_context(context)); } return serde_json::from_str(&text).map_err(|error| { WebError::bad_gateway_code( "knowledge_rag_lightrag_json_invalid", format!("LightRAG provider 返回非 JSON 响应: {error}"), ) .with_context(context) }); } unreachable!("lightrag retry loop returns on every branch") } fn should_retry_lightrag_provider_error(status: StatusCode, body: &str) -> bool { status.is_server_error() && (body.contains("RetryError") || body.contains("InvalidResponseError") || body.contains("Received empty content")) } fn mapped_references( raw: &Value, registry: &KnowledgeRagSourceRegistry, root_uri: &str, root_path: &Path, query: Option<&str>, ) -> Vec { reference_array(raw) .into_iter() .flat_map(|reference| { let enriched = enrich_reference_with_chunks(raw, &reference); expand_reference_by_chunks(&enriched) .into_iter() .map(|candidate| { map_reference_plan(&candidate, registry, root_uri, root_path, query) }) .collect::>() }) .filter(|reference| { !reference .get("stale") .and_then(Value::as_bool) .unwrap_or(false) && !reference .get("deleted") .and_then(Value::as_bool) .unwrap_or(false) && !reference .get("unmapped") .and_then(Value::as_bool) .unwrap_or(false) }) .collect() } fn expand_reference_by_chunks(reference: &Value) -> Vec { let Some(chunks) = reference.get("chunks").and_then(Value::as_array) else { return vec![reference.clone()]; }; if chunks.is_empty() { return vec![reference.clone()]; } chunks .iter() .enumerate() .map(|(index, chunk)| { let mut candidate = reference.clone(); if let Some(map) = candidate.as_object_mut() { map.insert("chunks".into(), Value::Array(vec![chunk.clone()])); if let Some(chunk_id) = chunk.get("chunk_id").cloned() { map.insert("chunk_id".into(), chunk_id); } map.insert("chunkIndex".into(), json!(index)); } candidate }) .collect() } fn reference_array(raw: &Value) -> Vec { if let Some(references) = raw .get("references") .or_else(|| raw.get("data").and_then(|value| value.get("references"))) .and_then(Value::as_array) { return references.clone(); } let mut by_key = BTreeMap::::new(); for chunk in chunk_array(raw) { let file_path = chunk.get("file_path").and_then(Value::as_str).unwrap_or(""); if file_path.is_empty() { continue; } let reference_id = chunk .get("reference_id") .and_then(Value::as_str) .unwrap_or(file_path); let key = format!("{file_path}\n{reference_id}"); by_key.entry(key).or_insert_with(|| { json!({ "file_path": file_path, "reference_id": reference_id, "chunks": [chunk.clone()], }) }); } by_key.into_values().collect() } fn chunk_array(raw: &Value) -> Vec<&Value> { raw.get("chunks") .or_else(|| raw.get("data").and_then(|value| value.get("chunks"))) .and_then(Value::as_array) .map(|items| items.iter().collect()) .unwrap_or_default() } fn enrich_reference_with_chunks(raw: &Value, reference: &Value) -> Value { let reference_id = reference.get("reference_id").and_then(Value::as_str); let file_path = reference.get("file_path").and_then(Value::as_str); let chunks = chunk_array(raw) .into_iter() .filter(|chunk| { let chunk_reference_id = chunk.get("reference_id").and_then(Value::as_str); let chunk_file_path = chunk.get("file_path").and_then(Value::as_str); reference_id .zip(chunk_reference_id) .is_some_and(|(left, right)| left == right) || file_path .zip(chunk_file_path) .is_some_and(|(left, right)| left == right) }) .cloned() .collect::>(); if chunks.is_empty() { return reference.clone(); } let mut enriched = reference.clone(); if let Some(map) = enriched.as_object_mut() { map.insert("chunks".into(), Value::Array(chunks)); } enriched } fn map_reference_plan( reference: &Value, registry: &KnowledgeRagSourceRegistry, root_uri: &str, root_path: &Path, query: Option<&str>, ) -> Value { let file_path = reference .get("file_path") .and_then(Value::as_str) .unwrap_or_default(); let doc_id = reference.get("doc_id").and_then(Value::as_str); let entry = registry.entries.iter().find(|entry| { lightrag_file_path_matches(entry, file_path) || doc_id.is_some_and(|doc_id| entry.light_rag_doc_id.as_deref() == Some(doc_id)) }); let locator_degraded = entry.is_none_or(|entry| entry.stale || entry.deleted_at_ms.is_some()); let source_path = entry.map(|entry| entry.source_path.clone()); let source_root_relative_path = entry.map(|entry| entry.source_root_relative_path.clone()); let primary_chunk = reference .get("chunks") .and_then(Value::as_array) .and_then(|chunks| chunks.first()); let chunk_id = reference .get("chunk_id") .or_else(|| primary_chunk.and_then(|chunk| chunk.get("chunk_id"))) .cloned() .unwrap_or(Value::Null); let source_chunk_id = reference .get("source_chunk_id") .or_else(|| primary_chunk.and_then(|chunk| chunk.get("source_chunk_id"))) .cloned() .or_else(|| { chunk_id .as_str() .and_then(|value| value.split_once("#match-").map(|(base, _)| json!(base))) }) .unwrap_or(Value::Null); let occurrence_index = reference .get("occurrence_index") .or_else(|| primary_chunk.and_then(|chunk| chunk.get("occurrence_index"))) .cloned() .unwrap_or(Value::Null); let chunk_sidecar = lightrag_reference_sidecar(reference, primary_chunk, &source_chunk_id, &chunk_id); let (quote, quote_source) = primary_chunk .and_then(|chunk| chunk.get("content")) .and_then(Value::as_str) .map(|value| (value.to_owned(), "chunk")) .or_else(|| { lightrag_chunk_content_for_reference_ids(&source_chunk_id, &chunk_id) .map(|value| (value, "kv_store")) }) .map(|(value, source)| { let quote = query .and_then(|query| query_centered_quote(&value, query, 500)) .unwrap_or_else(|| value.chars().take(500).collect::()); (quote, source) }) .map_or((None, "missing"), |(value, source)| (Some(value), source)); let locator_sidecar_block = entry.and_then(|entry| { find_lightrag_sidecar_block(entry, query, &occurrence_index, chunk_sidecar.as_ref()) }); let locator_sidecar_text = locator_sidecar_block .as_ref() .and_then(|block| block.get("content")) .and_then(Value::as_str) .map(ToOwned::to_owned); let heading_path = heading_path_for_sidecar_block(locator_sidecar_block.as_ref()); let text_bundle = citation_text_bundle( quote.as_deref(), query, locator_sidecar_text.as_deref(), quote_source, ); let content_diagnostics = quote_content_diagnostics(quote.as_deref(), quote_source); let locator = entry.and_then(|entry| { lightrag_locator_for_reference( root_path, root_uri, entry, &chunk_id, query, Some(&text_bundle.locator_evidence_text), &occurrence_index, chunk_sidecar.as_ref(), ) }); let locator_precision = locator .as_ref() .map(locator_precision_for_locator) .unwrap_or("file"); let locator_degraded = locator_degraded || matches!(locator_precision, "file"); let fallback_citation_url = entry.and_then(|entry| fallback_resource_citation_url(root_path, root_uri, entry)); let citation_url = locator .as_ref() .map(citation_url_for_locator) .or(fallback_citation_url); let citation_markdown = locator .as_ref() .map(citation_markdown_for_locator) .or_else(|| { citation_url.as_ref().map(|url| { format!( "[来源定位降级:{}]({})", markdown_link_label_escape(file_path), url.replace(')', "%29") ) }) }); let citation_id = short_citation_id( entry .map(|entry| entry.source_id.as_str()) .unwrap_or_default(), chunk_id.as_str().unwrap_or_default(), locator .as_ref() .and_then(|locator| locator.block_id.as_deref()) .unwrap_or_default(), occurrence_index.as_u64().unwrap_or(0), &text_bundle.normalized_fingerprint, ); json!({ "schema": REFERENCE_SCHEMA, "provider": "lightrag", "citationId": citation_id.clone(), "citationLabel": format!("[{}]", citation_id), "matchSource": if occurrence_index.is_null() { "lightrag_reference" } else { "lightrag_search" }, "reference": reference, "filePath": file_path, "chunkId": chunk_id, "sourceChunkId": source_chunk_id, "occurrenceIndex": occurrence_index, "rawQuote": text_bundle.raw_quote.clone(), "displayQuote": text_bundle.display_quote.clone(), "locatorEvidenceText": text_bundle.locator_evidence_text.clone(), "searchQuery": text_bundle.search_query.clone(), "normalizedFingerprint": text_bundle.normalized_fingerprint.clone(), "headingPath": heading_path, "quote": text_bundle.display_quote.clone(), "quoteSource": quote_source, "contentDiagnostics": content_diagnostics, "citationDiagnostics": { "quoteSource": quote_source, "rawReferenceMapped": entry.is_some(), "sidecarBlockMapped": locator_sidecar_text.is_some(), "displayCleaned": text_bundle.display_cleaned, "locatorTextSource": text_bundle.locator_text_source, }, "locator": locator, "locatorPrecision": locator_precision, "citationUrl": citation_url, "sourceId": entry.map(|entry| entry.source_id.clone()), "sourcePath": source_path, "sourceRootRelativePath": source_root_relative_path, "unmapped": entry.is_none(), "stale": entry.is_some_and(|entry| entry.stale), "deleted": entry.is_some_and(|entry| entry.deleted_at_ms.is_some()), "locatorDegraded": locator_degraded, "openAction": { "kind": "mnote.local_resource.open", "params": { "rootUri": root_uri, "path": entry.map(|entry| entry.source_root_relative_path.clone()), "provider": "lightrag", "filePath": file_path, "chunkId": chunk_id, "occurrenceIndex": occurrence_index, "searchQuery": query.unwrap_or_default(), "evidenceText": text_bundle.locator_evidence_text.clone(), "displayQuote": text_bundle.display_quote.clone(), }, }, "citationMarkdown": citation_markdown.unwrap_or_else(|| format!("[来源定位降级:{}](#)", file_path)), }) } fn fallback_resource_citation_url( root_path: &Path, root_uri: &str, entry: &KnowledgeRagSourceRegistryEntry, ) -> Option { if entry.stale || entry.deleted_at_ms.is_some() { return None; } let owner_document_id = fallback_owner_document_id(root_path, &entry.source_root_relative_path)?; let mut url = format!("/documents/{owner_document_id}"); append_query_param(&mut url, "sourceKind", "local_folder"); append_query_param(&mut url, "rootUri", root_uri); append_query_param( &mut url, "resourceTab", &format!( "resource:file:{}:{}", root_uri, entry.source_root_relative_path ), ); append_query_param(&mut url, "resourcePath", &entry.source_root_relative_path); Some(url) } fn fallback_owner_document_id(root_path: &Path, relative_path: &str) -> Option { if is_markdown_like_path(relative_path) { return Some(local_markdown_document_id(relative_path)); } let source_path = Path::new(relative_path); if let Some(parent) = source_path .parent() .filter(|parent| !parent.as_os_str().is_empty()) { let parent_path = root_path.join(parent); if let Ok(entries) = fs::read_dir(parent_path) { let mut candidates = entries .filter_map(Result::ok) .filter_map(|entry| { let path = entry.path(); if !path.is_file() || !is_markdown_like_path(&path.to_string_lossy()) { return None; } let file_name = path.file_name()?.to_string_lossy(); if file_name.starts_with('.') { return None; } let rel = parent.join(file_name.as_ref()); Some(rel.to_string_lossy().replace('\\', "/")) }) .collect::>(); candidates.sort(); if let Some(candidate) = candidates.first() { return Some(local_markdown_document_id(candidate)); } } } for root_candidate in ["README.md", "index.md"] { if root_path.join(root_candidate).is_file() { return Some(local_markdown_document_id(root_candidate)); } } None } fn is_markdown_like_path(value: &str) -> bool { Path::new(value) .extension() .and_then(|extension| extension.to_str()) .is_some_and(|extension| { extension.eq_ignore_ascii_case("md") || extension.eq_ignore_ascii_case("markdown") }) } fn local_markdown_document_id(relative_path: &str) -> String { format!( "local-md:{}", encode_local_document_id_segment(relative_path) ) } fn append_query_param(url: &mut String, key: &str, value: &str) { let value = value.trim(); if value.is_empty() { return; } let separator = if url.contains('?') { '&' } else { '?' }; url.push(separator); url.push_str(&encode_query_component(key)); url.push('='); url.push_str(&encode_query_component(value)); } fn encode_query_component(value: &str) -> String { let mut encoded = String::with_capacity(value.len()); for byte in value.as_bytes() { if byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'.' | b'~') { encoded.push(*byte as char); } else { encoded.push_str(&format!("%{byte:02X}")); } } encoded } fn encode_local_document_id_segment(relative_path: &str) -> String { relative_path .replace('\\', "/") .bytes() .flat_map(|byte| match byte { b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' => { vec![byte as char] } b'/' => "~2F".chars().collect::>(), _ => format!("~{byte:02X}").chars().collect::>(), }) .collect() } fn markdown_link_label_escape(value: &str) -> String { value.replace('[', "\\[").replace(']', "\\]") } fn quote_content_diagnostics(quote: Option<&str>, quote_source: &str) -> Value { let quote = quote.unwrap_or_default(); let non_empty_lines = quote .lines() .map(str::trim) .filter(|line| !line.is_empty()) .collect::>(); let non_heading_lines = non_empty_lines .iter() .copied() .filter(|line| !line.starts_with('#')) .collect::>(); let quote_only_image_placeholder = !non_heading_lines.is_empty() && non_heading_lines .iter() .all(|line| is_markdown_image_placeholder_line(line)); let ocr_text_exposed = !quote.trim().is_empty() && !quote_only_image_placeholder; json!({ "quoteEmpty": quote.trim().is_empty(), "quoteOnlyImagePlaceholder": quote_only_image_placeholder, "ocrTextExposed": ocr_text_exposed, "quoteSource": quote_source, }) } fn citation_text_bundle( raw_quote: Option<&str>, search_query: Option<&str>, locator_source_text: Option<&str>, quote_source: &'static str, ) -> CitationTextBundle { let raw_quote = raw_quote.unwrap_or_default().trim().to_string(); let search_query = search_query.unwrap_or_default().trim().to_string(); let locator_raw = locator_source_text .filter(|value| !value.trim().is_empty()) .unwrap_or(&raw_quote); let locator_window = if !search_query.is_empty() { query_centered_quote(locator_raw, &search_query, 700) .or_else(|| query_centered_quote(&raw_quote, &search_query, 700)) .unwrap_or_else(|| locator_raw.chars().take(700).collect::()) } else { locator_raw.chars().take(700).collect::() }; let display_source = locator_source_text .filter(|value| !value.trim().is_empty()) .unwrap_or(&raw_quote); let display_quote = clean_lightrag_text_for_display(display_source); let locator_evidence_text = query_centered_clean_window( &clean_lightrag_text_for_locator(&locator_window), &search_query, 700, ); CitationTextBundle { raw_quote: raw_quote.clone(), display_cleaned: display_quote != normalize_text_for_match(&raw_quote), normalized_fingerprint: normalized_quote_fingerprint(&display_quote), display_quote, locator_evidence_text, search_query, locator_text_source: if locator_source_text.is_some() { "sidecar_block" } else { quote_source }, } } fn clean_lightrag_text_for_display(value: &str) -> String { let without_tags = strip_lightrag_markup(value); normalize_latex_plain_text(&without_tags) } fn clean_lightrag_text_for_locator(value: &str) -> String { clean_lightrag_text_for_display(value) } fn strip_lightrag_markup(value: &str) -> String { let mut output = String::new(); let mut chars = value.chars().peekable(); while let Some(ch) = chars.next() { if ch != '<' { output.push(ch); continue; } let mut tag = String::new(); let mut closed = false; for next in chars.by_ref() { if next == '>' { closed = true; break; } tag.push(next); if tag.chars().count() > 240 { break; } } output.push(' '); if !closed { break; } } output } fn normalize_latex_plain_text(value: &str) -> String { let mut output = String::new(); let mut chars = value.chars().peekable(); while let Some(ch) = chars.next() { if ch == '\\' { let mut command = String::new(); while let Some(next) = chars.peek().copied() { if next.is_ascii_alphabetic() { command.push(next); chars.next(); } else { break; } } match command.as_str() { "gt" => output.push('>'), "lt" => output.push('<'), "sim" => output.push('∼'), "circ" => output.push('∘'), "alpha" => output.push('α'), "beta" => output.push('β'), "gamma" => output.push('γ'), "delta" => output.push('δ'), "Delta" => output.push('△'), "left" | "right" | "mathrm" | "text" | "operatorname" => output.push(' '), "" => output.push(' '), _ => output.push(' '), } continue; } match ch { '{' | '}' | '_' | '^' => output.push(' '), _ => output.push(ch), } } normalize_text_for_match(&output) } fn normalized_quote_fingerprint(value: &str) -> String { normalize_text_for_match(value) .chars() .filter(|ch| ch.is_alphanumeric() || is_cjk_char(*ch)) .take(160) .collect::() .to_ascii_lowercase() } fn query_centered_clean_window(value: &str, query: &str, max_chars: usize) -> String { let text = normalize_text_for_match(value); let query = normalize_text_for_match(query); if text.is_empty() || query.is_empty() { return text.chars().take(max_chars).collect(); } let lower = text.to_ascii_lowercase(); let query_lower = query.to_ascii_lowercase(); let start = lower.find(&query_lower).unwrap_or(0); text.get(start..) .unwrap_or(&text) .chars() .take(max_chars) .collect() } fn reference_display_quote(reference: &Value) -> &str { reference .get("displayQuote") .and_then(Value::as_str) .or_else(|| reference.get("quote").and_then(Value::as_str)) .unwrap_or_default() } fn reference_locator_evidence_text(reference: &Value) -> &str { reference .get("locatorEvidenceText") .and_then(Value::as_str) .or_else(|| reference.get("quote").and_then(Value::as_str)) .unwrap_or_default() } fn query_centered_quote(content: &str, query: &str, max_chars: usize) -> Option { let query_normalized = normalize_text_for_match(query).to_ascii_lowercase(); if query_normalized.is_empty() { return None; } let lines = content.lines().collect::>(); for (index, line) in lines.iter().enumerate() { let line_normalized = normalize_text_for_match(line).to_ascii_lowercase(); if !line_normalized.contains(&query_normalized) { continue; } let mut quote = String::new(); if index > 0 { let previous = lines[index - 1].trim(); if previous.starts_with('#') { quote.push_str(previous); quote.push('\n'); } } for line in lines.iter().skip(index) { let line = line.trim(); if line.is_empty() { if !quote.is_empty() { quote.push('\n'); } continue; } let next_len = quote.chars().count() + line.chars().count() + 1; if next_len > max_chars && !quote.trim().is_empty() { break; } if !quote.is_empty() && !quote.ends_with('\n') { quote.push('\n'); } quote.push_str(line); if quote.chars().count() >= max_chars { break; } } let quote = quote.trim(); if !quote.is_empty() { return Some(quote.chars().take(max_chars).collect()); } } let content_lower = content.to_ascii_lowercase(); let byte_index = content_lower.find(&query_normalized)?; Some(char_window_around_byte(content, byte_index, max_chars)) } fn char_window_around_byte(content: &str, byte_index: usize, max_chars: usize) -> String { let target_char_index = content[..byte_index].chars().count(); let before = max_chars / 3; let start = target_char_index.saturating_sub(before); content .chars() .skip(start) .take(max_chars) .collect::() .trim() .to_string() } fn is_markdown_image_placeholder_line(line: &str) -> bool { line.starts_with("![") && line.contains("](") && line.ends_with(')') } fn lightrag_locator_for_reference( root_path: &Path, root_uri: &str, entry: &KnowledgeRagSourceRegistryEntry, chunk_id: &Value, query: Option<&str>, locator_evidence_text: Option<&str>, occurrence_index: &Value, chunk_sidecar: Option<&Value>, ) -> Option { let block = find_lightrag_sidecar_block(entry, query, occurrence_index, chunk_sidecar)?; let positions = block .get("positions") .and_then(Value::as_array) .map(|positions| parse_lightrag_locator_positions(positions.as_slice())) .unwrap_or_default(); let resource_path = entry.source_root_relative_path.clone(); let block_content = block .get("content") .and_then(Value::as_str) .unwrap_or_default(); let evidence_text = locator_evidence_text .filter(|value| !value.trim().is_empty()) .map(ToOwned::to_owned) .or_else(|| { query .and_then(|query| query_centered_quote(block_content, query, 700)) .map(|value| clean_lightrag_text_for_locator(&value)) }) .unwrap_or_else(|| clean_lightrag_text_for_locator(block_content)); let mut open_params = json!({ "rootUri": root_uri, "resourcePath": resource_path, "provider": "lightrag", "chunkId": chunk_id, "searchQuery": query.unwrap_or_default(), "query": evidence_text, "evidenceText": evidence_text, }); if let Some(map) = open_params.as_object_mut() { if let Some(paragraph_ordinal) = positions.paragraph_ordinal { map.insert("paragraphOrdinal".into(), json!(paragraph_ordinal)); } if let Some(para_id_start) = positions.para_id_start.as_deref() { map.insert("paraIdStart".into(), json!(para_id_start)); } if let Some(para_id_end) = positions.para_id_end.as_deref() { map.insert("paraIdEnd".into(), json!(para_id_end)); } if let Some(text_fingerprint) = positions.text_fingerprint.as_deref() { map.insert("textFingerprint".into(), json!(text_fingerprint)); } } let mut locator = EvidenceLocator::new( root_uri, fallback_owner_document_id(root_path, &resource_path).unwrap_or_default(), &resource_path, evidence_resource_kind_for_path(&resource_path), EvidenceOpenAction { action_type: "mnote.open_resource_locator".into(), url: "/".into(), params: open_params, }, ); locator.resource_path = Some(entry.source_root_relative_path.clone()); if let Some(position) = positions.bbox { locator.page = Some(position.page); locator.bbox = Some(position.bbox); } locator.block_id = block .get("blockid") .and_then(Value::as_str) .map(ToOwned::to_owned) .or_else(|| chunk_id.as_str().map(ToOwned::to_owned)); locator.source_map_path = sidecar_blocks_path(entry).and_then(|path| root_relative_path(root_path, &path).ok()); Some(locator) } fn evidence_resource_kind_for_path(path: &str) -> EvidenceResourceKind { match Path::new(path) .extension() .and_then(|value| value.to_str()) .map(|value| value.to_ascii_lowercase()) .as_deref() { Some("pdf") => EvidenceResourceKind::Pdf, Some("png" | "jpg" | "jpeg" | "gif" | "webp" | "bmp" | "svg") => { EvidenceResourceKind::Image } Some("md" | "markdown" | "txt") => EvidenceResourceKind::Markdown, Some("doc" | "docx" | "xls" | "xlsx" | "ppt" | "pptx" | "csv") => { EvidenceResourceKind::Office } _ => EvidenceResourceKind::RawFile, } } #[derive(Debug, Clone)] struct LightRagBboxPosition { page: u32, bbox: EvidenceBBox, } #[derive(Debug, Clone, Default)] struct LightRagLocatorPositions { bbox: Option, paragraph_ordinal: Option, para_id_start: Option, para_id_end: Option, text_fingerprint: Option, } fn parse_lightrag_locator_positions(positions: &[Value]) -> LightRagLocatorPositions { let mut parsed = LightRagLocatorPositions::default(); for position in positions { match position.get("type").and_then(Value::as_str) { Some("bbox") if parsed.bbox.is_none() => { parsed.bbox = parse_bbox_position(position); } Some("paraid") => { if parsed.paragraph_ordinal.is_none() { parsed.paragraph_ordinal = position.get("anchor").and_then(value_to_u32); } if parsed.para_id_start.is_none() || parsed.para_id_end.is_none() { if let Some(range) = position.get("range").and_then(Value::as_array) { parsed.para_id_start = range .first() .and_then(Value::as_str) .map(|value| value.trim().to_string()) .filter(|value| !value.is_empty()); parsed.para_id_end = range .get(1) .and_then(Value::as_str) .map(|value| value.trim().to_string()) .filter(|value| !value.is_empty()) .or_else(|| parsed.para_id_start.clone()); } } } Some("text_fingerprint") if parsed.text_fingerprint.is_none() => { parsed.text_fingerprint = position .get("anchor") .and_then(Value::as_str) .map(|value| value.trim().to_string()) .filter(|value| !value.is_empty()); } _ => {} } } parsed } fn parse_bbox_position(value: &Value) -> Option { let page = value.get("anchor").and_then(value_to_u32)?; let range = value.get("range")?.as_array()?; if range.len() != 4 { return None; } Some(LightRagBboxPosition { page, bbox: EvidenceBBox { x0: range.first()?.as_f64()?, y0: range.get(1)?.as_f64()?, x1: range.get(2)?.as_f64()?, y1: range.get(3)?.as_f64()?, }, }) } fn value_to_u32(value: &Value) -> Option { value .as_u64() .and_then(|value| u32::try_from(value).ok()) .or_else(|| { value .as_str() .and_then(|value| value.trim().parse::().ok()) }) } fn find_lightrag_sidecar_block( entry: &KnowledgeRagSourceRegistryEntry, query: Option<&str>, occurrence_index: &Value, chunk_sidecar: Option<&Value>, ) -> Option { let ref_ids = lightrag_sidecar_ref_ids(chunk_sidecar?); if ref_ids.is_empty() { return None; } let blocks = lightrag_sidecar_blocks_by_id(entry, &ref_ids)?; let query = query .map(str::trim) .filter(|value| !value.is_empty()) .map(ToOwned::to_owned); let Some(query) = query else { return ref_ids.into_iter().find_map(|id| blocks.get(&id).cloned()); }; let target_occurrence = occurrence_index.as_u64().unwrap_or(0); let mut seen_occurrences = 0u64; for id in ref_ids { let Some(block) = blocks.get(&id) else { continue; }; let count = block .get("content") .and_then(Value::as_str) .map(|content| exact_match_count(content, &query)) .unwrap_or(0); if count == 0 { continue; } if seen_occurrences + count > target_occurrence { return Some(block.clone()); } seen_occurrences += count; } let query_terms = locator_query_match_terms(&query); if query_terms.is_empty() { return None; } let ref_ids = lightrag_sidecar_ref_ids(chunk_sidecar?); if query_terms.len() <= 6 { for id in &ref_ids { let Some(block) = blocks.get(id) else { continue; }; let block_normalized = block .get("content") .and_then(Value::as_str) .map(|content| normalize_text_for_match(content).to_lowercase()) .unwrap_or_default(); if !block_normalized.is_empty() && query_terms .iter() .all(|term| block_matches_query_term(&block_normalized, term)) { return Some(block.clone()); } } } for id in &ref_ids { let Some(block) = blocks.get(id) else { continue; }; let block_normalized = block .get("content") .and_then(Value::as_str) .map(|content| normalize_text_for_match(content).to_lowercase()) .unwrap_or_default(); if !block_normalized.is_empty() && query_terms .iter() .any(|term| block_matches_query_term(&block_normalized, term)) { return Some(block.clone()); } } None } fn lightrag_sidecar_ref_ids(sidecar: &Value) -> Vec { let mut ids = Vec::new(); if let Some(id) = sidecar.get("id").and_then(Value::as_str) { ids.push(id.trim().to_string()); } if let Some(refs) = sidecar.get("refs").and_then(Value::as_array) { for item in refs { if let Some(id) = item.get("id").and_then(Value::as_str) { ids.push(id.trim().to_string()); } } } ids.retain(|id| !id.is_empty()); ids.dedup(); ids } fn lightrag_sidecar_blocks_by_id( entry: &KnowledgeRagSourceRegistryEntry, ref_ids: &[String], ) -> Option> { let wanted = ref_ids.iter().cloned().collect::>(); let path = sidecar_blocks_path(entry)?; let content = fs::read_to_string(path).ok()?; let mut blocks = BTreeMap::new(); for line in content.lines() { let block = serde_json::from_str::(line).ok()?; let Some(block_id) = block.get("blockid").and_then(Value::as_str) else { continue; }; if wanted.contains(block_id) && block.get("positions").and_then(Value::as_array).is_some() { blocks.insert(block_id.to_string(), block); } if blocks.len() == wanted.len() { break; } } Some(blocks) } fn exact_match_count(content: &str, query: &str) -> u64 { if query.is_empty() { return 0; } let content_lower = content.to_lowercase(); let query_lower = query.to_lowercase(); let mut count = 0u64; let mut start = 0usize; while let Some(index) = content_lower[start..].find(&query_lower) { count += 1; start += index + query_lower.len().max(1); } count } fn lightrag_reference_sidecar( reference: &Value, primary_chunk: Option<&Value>, source_chunk_id: &Value, chunk_id: &Value, ) -> Option { reference .get("sidecar") .cloned() .or_else(|| { primary_chunk .and_then(|chunk| chunk.get("sidecar")) .cloned() }) .or_else(|| lightrag_chunk_sidecar_for_reference_ids(source_chunk_id, chunk_id)) } fn block_matches_query_term(block_normalized: &str, term: &str) -> bool { if block_normalized.contains(term) { return true; } let term_chars = term.chars().collect::>(); if term_chars.len() < 2 || !term_chars.iter().all(|ch| is_cjk_char(*ch)) { return false; } let mut index = 0usize; let mut gap = 0usize; for ch in block_normalized.chars() { if ch == term_chars[index] { index += 1; gap = 0; if index == term_chars.len() { return true; } continue; } if index > 0 { gap += 1; if gap > 4 { index = 0; gap = 0; } } } false } fn query_match_terms(query_normalized: &str) -> Vec { let mut terms = query_normalized .split(|ch: char| ch.is_whitespace() || ch.is_ascii_punctuation()) .map(str::trim) .filter(|term| !term.is_empty()) .map(ToOwned::to_owned) .collect::>(); for term in terms.clone() { let chars = term.chars().collect::>(); if chars.len() < 4 || !chars.iter().any(|ch| is_cjk_char(*ch)) { continue; } for window in chars.windows(2) { terms.push(window.iter().collect::()); } for window in chars.windows(3) { terms.push(window.iter().collect::()); } } terms.sort(); terms.dedup(); terms } fn locator_query_match_terms(query: &str) -> Vec { let query_normalized = normalize_text_for_match(query).to_lowercase(); let mut terms = query_match_terms(&query_normalized) .into_iter() .filter(|term| { let chars = term.chars().collect::>(); if chars.iter().any(|ch| is_cjk_char(*ch)) { return chars.len() <= 4 && chars.iter().all(|ch| is_cjk_char(*ch)) && significant_cjk_query_term(term); } chars.len() >= 3 }) .collect::>(); terms.sort(); terms.dedup(); terms } fn query_is_single_cjk_lookup(query_normalized: &str) -> bool { let chars = query_normalized.chars().collect::>(); !chars.is_empty() && chars.len() <= 8 && chars .iter() .all(|ch| is_cjk_char(*ch) || ch.is_ascii_alphanumeric()) } fn significant_cjk_query_term(term: &str) -> bool { let chars = term.chars().collect::>(); if chars.len() < 2 || !chars.iter().all(|ch| is_cjk_char(*ch)) { return false; } !matches!( term, "请用" | "使用" | "资料" | "资料库" | "知识" | "知识库" | "检索" | "搜索" | "回答" | "说明" | "解释" | "总结" | "用途" | "应用" | "来源" | "出处" | "证据" | "定位" | "相关" | "什么" | "哪些" | "怎么" | "如何" | "中的" | "有什么" ) } fn is_cjk_char(ch: char) -> bool { ('\u{4e00}'..='\u{9fff}').contains(&ch) || ('\u{3400}'..='\u{4dbf}').contains(&ch) || ('\u{f900}'..='\u{faff}').contains(&ch) } fn normalize_text_for_match(value: &str) -> String { value .split_whitespace() .collect::>() .join(" ") .chars() .collect::() } fn sidecar_blocks_path(entry: &KnowledgeRagSourceRegistryEntry) -> Option { for file_path in lightrag_sidecar_file_path_candidates(entry) { let stem = Path::new(&file_path).file_stem()?.to_string_lossy(); let path = lightrag_input_dir() .join("__parsed__") .join(format!("{file_path}.parsed")) .join(format!("{stem}.blocks.jsonl")); if path.exists() { return Some(path); } } None } fn lightrag_sidecar_file_path_candidates(entry: &KnowledgeRagSourceRegistryEntry) -> Vec { let mut candidates = vec![entry.light_rag_file_path.clone()]; if let Some(stripped) = strip_one_supported_parser_hint(&entry.light_rag_file_path) { candidates.push(stripped); } if let Some(asset_path) = image_wrapper_asset_file_path(&entry.light_rag_file_path) { candidates.push(asset_path); } candidates.sort(); candidates.dedup(); candidates } fn image_wrapper_asset_file_path(file_path: &str) -> Option { let asset_path = file_path.strip_suffix(".md")?; if resource_type_for_path(asset_path) == "image" { return Some(asset_path.to_string()); } None } fn resource_type_for_path(path: &str) -> &'static str { match Path::new(path) .extension() .and_then(|value| value.to_str()) .map(|value| value.to_ascii_lowercase()) .as_deref() { Some("png" | "jpg" | "jpeg" | "webp" | "gif" | "bmp" | "tif" | "tiff") => "image", Some("pdf") => "pdf", Some("doc" | "docx") => "docx", Some("ppt" | "pptx") => "pptx", Some("xls" | "xlsx") => "xlsx", _ => "resource", } } fn knowledge_rag_source_content_diagnostics(registry: &KnowledgeRagSourceRegistry) -> Value { Value::Array( registry .entries .iter() .map(|entry| { let sidecar_path = sidecar_blocks_path(entry); let sidecar_stats = sidecar_path .as_ref() .and_then(|path| sidecar_text_stats(path).ok()); json!({ "sourceId": entry.source_id, "sourceRootRelativePath": entry.source_root_relative_path, "lightRagDocId": entry.light_rag_doc_id, "lightRagFilePath": entry.light_rag_file_path, "lightRagStatus": entry.light_rag_status, "directImageScan": resource_type_for_path(&entry.light_rag_file_path) == "image", "sidecarExists": sidecar_path.is_some(), "sidecarPath": sidecar_path.map(|path| path.display().to_string()), "ocrTextExposed": sidecar_stats .as_ref() .map(|stats| stats.meaningful_blocks > 0) .unwrap_or(false), "sidecarBlocks": sidecar_stats.as_ref().map(|stats| stats.blocks).unwrap_or(0), "sidecarMeaningfulBlocks": sidecar_stats .as_ref() .map(|stats| stats.meaningful_blocks) .unwrap_or(0), }) }) .collect(), ) } #[derive(Debug, Clone)] struct SidecarContextBlock { block_ordinal: u64, block_id: String, text: String, heading_path: Vec, positions: LightRagLocatorPositions, } fn sidecar_section_context_payload( root_path: &Path, registry: &KnowledgeRagSourceRegistry, request: &KnowledgeRagSectionContextRequest, ) -> Result { let entry = resolve_section_context_entry(registry, request)?; let sidecar_path = sidecar_blocks_path(entry).ok_or_else(|| { WebError::bad_request_code( "knowledge_rag_section_sidecar_missing", "该资料没有可读取的 LightRAG native sidecar blocks", ) })?; let blocks = read_sidecar_context_blocks(&sidecar_path)?; let (selected_start, selected_end) = section_context_selected_range(&blocks, request, entry, root_path)?; let context_before = request.context_before.unwrap_or(1).min(20); let context_after = request.context_after.unwrap_or(1).min(20); let range_start = selected_start.saturating_sub(context_before); let range_end = selected_end.saturating_add(context_after); let max_blocks = request.max_blocks.unwrap_or(24).clamp(1, 80); let max_chars = request.max_chars.unwrap_or(12_000).clamp(500, 40_000); let mut remaining_chars = max_chars; let mut returned_blocks = Vec::new(); let mut text_parts = Vec::new(); let mut truncated = false; for block in blocks .iter() .filter(|block| block.block_ordinal >= range_start && block.block_ordinal <= range_end) { if returned_blocks.len() >= max_blocks { truncated = true; break; } if remaining_chars == 0 { truncated = true; break; } let (display_text, text_truncated) = take_chars_with_truncation(&block.text, remaining_chars); remaining_chars = remaining_chars.saturating_sub(display_text.chars().count()); if text_truncated { truncated = true; } text_parts.push(display_text.clone()); returned_blocks.push(json!({ "blockOrdinal": block.block_ordinal, "blockId": block.block_id.clone(), "paragraphOrdinal": block.positions.paragraph_ordinal, "headingPath": block.heading_path.clone(), "text": display_text, "textTruncated": text_truncated, "locator": sidecar_context_locator(&block.positions), })); if text_truncated { break; } } let chunks = section_context_chunks(&returned_blocks, 2_400); Ok(json!({ "ok": true, "schema": "mnote.knowledge_rag.section_context.v1", "provider": "lightrag", "sourceId": entry.source_id, "sourceRootRelativePath": entry.source_root_relative_path, "lightRagDocId": entry.light_rag_doc_id, "lightRagFilePath": entry.light_rag_file_path, "sidecarPath": root_relative_path(root_path, &sidecar_path).unwrap_or_else(|_| sidecar_path.display().to_string()), "section": { "sectionId": request.section_id, "startBlockOrdinal": request.start_block_ordinal, "endBlockOrdinal": request.end_block_ordinal, "startParagraphOrdinal": request.start_paragraph_ordinal, "endParagraphOrdinal": request.end_paragraph_ordinal, "selectedStartBlockOrdinal": selected_start, "selectedEndBlockOrdinal": selected_end, }, "limits": { "contextBefore": context_before, "contextAfter": context_after, "maxBlocks": max_blocks, "maxChars": max_chars, "returnedBlocks": returned_blocks.len(), "returnedChars": max_chars.saturating_sub(remaining_chars), "truncated": truncated, }, "blocks": returned_blocks, "chunks": chunks, "text": text_parts.join("\n\n"), })) } fn resolve_section_context_entry<'a>( registry: &'a KnowledgeRagSourceRegistry, request: &KnowledgeRagSectionContextRequest, ) -> Result<&'a KnowledgeRagSourceRegistryEntry, WebError> { let source_id = request .source_id .as_deref() .map(str::trim) .filter(|value| !value.is_empty()) .or_else(|| { request .section_id .as_deref() .and_then(|value| value.split_once('#').map(|(source_id, _)| source_id.trim())) .filter(|value| !value.is_empty()) }); let source_path = request .source_path .as_deref() .map(|value| value.trim().trim_matches('/').replace('\\', "/")) .filter(|value| !value.is_empty()); let light_rag_doc_id = request .light_rag_doc_id .as_deref() .map(str::trim) .filter(|value| !value.is_empty()); let file_path = request .file_path .as_deref() .map(str::trim) .filter(|value| !value.is_empty()); if source_id.is_none() && source_path.is_none() && light_rag_doc_id.is_none() && file_path.is_none() { return Err(WebError::bad_request_code( "knowledge_rag_section_source_required", "按章节读取资料上下文需要 sourcePath、sourceId、lightRagDocId 或 filePath", )); } registry .entries .iter() .find(|entry| { !entry.stale && entry.deleted_at_ms.is_none() && entry .light_rag_status .as_deref() .is_none_or(|status| status == "processed") && source_id.is_none_or(|value| entry.source_id == value) && source_path .as_deref() .is_none_or(|value| entry.source_root_relative_path == value) && light_rag_doc_id .is_none_or(|value| entry.light_rag_doc_id.as_deref() == Some(value)) && file_path.is_none_or(|value| lightrag_file_path_matches(entry, value)) }) .ok_or_else(|| { WebError::bad_request_code( "knowledge_rag_section_source_not_found", "没有找到匹配且已处理完成的资料来源", ) }) } fn read_sidecar_context_blocks(path: &Path) -> Result, WebError> { let content = fs::read_to_string(path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_sidecar_read_failed", format!("无法读取 LightRAG sidecar {}: {error}", path.display()), ) })?; let mut blocks = Vec::new(); let mut ordinal = 0u64; for line in content.lines() { let Ok(block) = serde_json::from_str::(line) else { continue; }; if block.get("type").and_then(Value::as_str) != Some("content") { continue; } let raw_text = block .get("content") .and_then(Value::as_str) .unwrap_or_default(); let positions = block .get("positions") .and_then(Value::as_array) .map(|positions| parse_lightrag_locator_positions(positions.as_slice())) .unwrap_or_default(); blocks.push(SidecarContextBlock { block_ordinal: ordinal, block_id: block .get("blockid") .and_then(Value::as_str) .unwrap_or_default() .to_string(), text: clean_lightrag_text_for_display(raw_text), heading_path: heading_path_for_sidecar_block(Some(&block)), positions, }); ordinal += 1; } Ok(blocks) } fn section_context_selected_range( blocks: &[SidecarContextBlock], request: &KnowledgeRagSectionContextRequest, entry: &KnowledgeRagSourceRegistryEntry, root_path: &Path, ) -> Result<(u64, u64), WebError> { if blocks.is_empty() { return Err(WebError::bad_request_code( "knowledge_rag_section_sidecar_empty", "该资料的 LightRAG sidecar 没有 content blocks", )); } if let Some(start) = request.start_block_ordinal { let end = request.end_block_ordinal.unwrap_or(start); if start > end { return Err(WebError::bad_request_code( "knowledge_rag_section_block_range_invalid", "startBlockOrdinal 不能大于 endBlockOrdinal", )); } return Ok((start, end)); } if let Some(start) = request.start_paragraph_ordinal { let end = request.end_paragraph_ordinal.unwrap_or(start); if start > end { return Err(WebError::bad_request_code( "knowledge_rag_section_paragraph_range_invalid", "startParagraphOrdinal 不能大于 endParagraphOrdinal", )); } let matching = blocks .iter() .filter(|block| { block .positions .paragraph_ordinal .is_some_and(|ordinal| ordinal >= start && ordinal <= end) }) .map(|block| block.block_ordinal) .collect::>(); let first = matching.first().copied().ok_or_else(|| { WebError::bad_request_code( "knowledge_rag_section_paragraph_range_not_found", "没有在 sidecar blocks 中找到对应 paragraph ordinal 范围", ) })?; let last = matching.last().copied().unwrap_or(first); return Ok((first, last)); } if let Some(section_id) = request .section_id .as_deref() .map(str::trim) .filter(|value| !value.is_empty()) { if let Some((start, end)) = section_context_range_for_section_id(root_path, entry, section_id) { return Ok((start, end)); } } Err(WebError::bad_request_code( "knowledge_rag_section_range_required", "按章节读取资料上下文需要 sectionId、startBlockOrdinal 或 startParagraphOrdinal", )) } fn section_context_range_for_section_id( root_path: &Path, entry: &KnowledgeRagSourceRegistryEntry, section_id: &str, ) -> Option<(u64, u64)> { let structure = document_structure_for_entry(root_path, entry, &BTreeSet::new(), None)?; let sections = structure.get("sections").and_then(Value::as_array)?; sections.iter().find_map(|section| { let matches_id = section .get("sectionId") .and_then(Value::as_str) .is_some_and(|value| value == section_id); if !matches_id { return None; } let start = section.get("startBlockOrdinal").and_then(Value::as_u64)?; let end = section .get("endBlockOrdinal") .and_then(Value::as_u64) .unwrap_or(start); Some((start, end)) }) } fn sidecar_context_locator(positions: &LightRagLocatorPositions) -> Value { let bbox = positions.bbox.as_ref().map(|position| { json!({ "page": position.page, "bbox": { "x0": position.bbox.x0, "y0": position.bbox.y0, "x1": position.bbox.x1, "y1": position.bbox.y1, } }) }); json!({ "paragraphOrdinal": positions.paragraph_ordinal, "paraIdStart": positions.para_id_start.clone(), "paraIdEnd": positions.para_id_end.clone(), "textFingerprint": positions.text_fingerprint.clone(), "bbox": bbox, }) } fn section_context_chunks(blocks: &[Value], max_chunk_chars: usize) -> Vec { let mut chunks = Vec::new(); let mut current_text = String::new(); let mut current_block_ids = Vec::new(); let mut start_block_ordinal: Option = None; let mut end_block_ordinal: Option = None; let flush_current = |chunks: &mut Vec, current_text: &mut String, current_block_ids: &mut Vec, start_block_ordinal: &mut Option, end_block_ordinal: &mut Option| { if current_text.trim().is_empty() { current_block_ids.clear(); *start_block_ordinal = None; *end_block_ordinal = None; return; } let chunk_ordinal = chunks.len(); let text = current_text.trim().to_string(); let block_ids = current_block_ids.clone(); let start = *start_block_ordinal; let end = *end_block_ordinal; chunks.push(json!({ "chunkOrdinal": chunk_ordinal, "startBlockOrdinal": start, "endBlockOrdinal": end, "blockIds": block_ids, "text": text, })); current_text.clear(); current_block_ids.clear(); *start_block_ordinal = None; *end_block_ordinal = None; }; for block in blocks { let text = block .get("text") .and_then(Value::as_str) .unwrap_or_default(); if text.trim().is_empty() { continue; } let block_ordinal = block.get("blockOrdinal").and_then(Value::as_u64); let block_id = block .get("blockId") .and_then(Value::as_str) .unwrap_or_default(); let projected_chars = current_text.chars().count() + text.chars().count() + 2; if !current_text.is_empty() && projected_chars > max_chunk_chars { flush_current( &mut chunks, &mut current_text, &mut current_block_ids, &mut start_block_ordinal, &mut end_block_ordinal, ); } if start_block_ordinal.is_none() { start_block_ordinal = block_ordinal; } end_block_ordinal = block_ordinal.or(end_block_ordinal); if !block_id.is_empty() { current_block_ids.push(block_id.to_string()); } if !current_text.is_empty() { current_text.push_str("\n\n"); } current_text.push_str(text); } flush_current( &mut chunks, &mut current_text, &mut current_block_ids, &mut start_block_ordinal, &mut end_block_ordinal, ); chunks } fn take_chars_with_truncation(value: &str, max_chars: usize) -> (String, bool) { let mut output = String::new(); let mut truncated = false; for (index, ch) in value.chars().enumerate() { if index >= max_chars { truncated = true; break; } output.push(ch); } (output, truncated) } #[derive(Debug, Clone, Default, Serialize)] #[serde(rename_all = "camelCase")] struct DocumentStructureSection { section_id: String, title: String, level: u64, heading_path: Vec, start_block_ordinal: u64, end_block_ordinal: u64, block_count: u64, start_paragraph_ordinal: Option, end_paragraph_ordinal: Option, start_block_id: String, end_block_id: String, query_match_count: u64, matched_reference_count: u64, sample: String, } fn document_structure_index_payload( root_path: &Path, registry: &KnowledgeRagSourceRegistry, source_scope: &[String], references: &[Value], query: Option<&str>, max_documents: usize, ) -> Result { let reference_source_ids = references .iter() .filter_map(|reference| reference.get("sourceId").and_then(Value::as_str)) .map(ToOwned::to_owned) .collect::>(); let mut entries = registry .entries .iter() .filter(|entry| { !entry.stale && entry.deleted_at_ms.is_none() && entry .light_rag_status .as_deref() .is_none_or(|status| status == "processed") }) .filter(|entry| { if source_scope.is_empty() { reference_source_ids.is_empty() || reference_source_ids.contains(&entry.source_id) } else { source_scope.iter().any(|scope| { entry.source_root_relative_path == *scope || entry .source_root_relative_path .starts_with(&format!("{scope}/")) }) } }) .collect::>(); entries.sort_by(|left, right| { left.source_root_relative_path .cmp(&right.source_root_relative_path) }); entries.truncate(max_documents); let reference_block_ids = references .iter() .filter_map(|reference| { reference .get("locator") .and_then(|locator| locator.get("blockId")) .and_then(Value::as_str) }) .map(ToOwned::to_owned) .collect::>(); let documents = entries .into_iter() .map(|entry| { document_structure_for_entry(root_path, entry, &reference_block_ids, query) .unwrap_or_else(|| { json!({ "sourceId": entry.source_id, "sourceRootRelativePath": entry.source_root_relative_path, "lightRagDocId": entry.light_rag_doc_id, "lightRagFilePath": entry.light_rag_file_path, "sidecarAvailable": false, "sections": [], }) }) }) .collect::>(); Ok(json!({ "schema": DOCUMENT_STRUCTURE_INDEX_SCHEMA, "generatedAtMs": now_ms(), "mode": "sidecar_heading_sections", "sourceScope": source_scope, "referenceCount": references.len(), "documents": documents, })) } fn document_structure_for_entry( root_path: &Path, entry: &KnowledgeRagSourceRegistryEntry, reference_block_ids: &BTreeSet, query: Option<&str>, ) -> Option { let sidecar_path = sidecar_blocks_path(entry)?; let content = fs::read_to_string(&sidecar_path).ok()?; let query_terms = query .map(normalize_text_for_match) .map(|value| query_match_terms(&value.to_ascii_lowercase())) .unwrap_or_default() .into_iter() .filter(|term| term.chars().count() >= 2) .collect::>(); let mut sections = Vec::::new(); let mut current: Option = None; let mut ordinal = 0u64; let mut truncated = false; for line in content.lines() { let Ok(block) = serde_json::from_str::(line) else { continue; }; if block.get("type").and_then(Value::as_str) != Some("content") { continue; } let block_id = block .get("blockid") .and_then(Value::as_str) .unwrap_or_default() .to_string(); let block_content = block .get("content") .and_then(Value::as_str) .unwrap_or_default(); let positions = block .get("positions") .and_then(Value::as_array) .map(|positions| parse_lightrag_locator_positions(positions.as_slice())) .unwrap_or_default(); let heading_title = heading_title_from_block(block_content); if let Some(title) = heading_title { if let Some(section) = current.take() { sections.push(section); if sections.len() >= DOCUMENT_STRUCTURE_INDEX_MAX_SECTIONS_PER_DOC { truncated = true; break; } } let mut heading_path = heading_path_for_sidecar_block(Some(&block)); if heading_path.last().is_none_or(|last| last != &title) { heading_path.push(title.clone()); } let level = block.get("level").and_then(Value::as_u64).unwrap_or(1); current = Some(DocumentStructureSection { section_id: format!("{}#block-{}", entry.source_id, block_id), title, level, heading_path, start_block_ordinal: ordinal, end_block_ordinal: ordinal, block_count: 1, start_paragraph_ordinal: positions.paragraph_ordinal, end_paragraph_ordinal: positions.paragraph_ordinal, start_block_id: block_id.clone(), end_block_id: block_id.clone(), query_match_count: query_match_count(block_content, &query_terms), matched_reference_count: u64::from(reference_block_ids.contains(&block_id)), sample: String::new(), }); } else { let section = current.get_or_insert_with(|| DocumentStructureSection { section_id: format!("{}#body", entry.source_id), title: "正文".into(), level: 0, heading_path: Vec::new(), start_block_ordinal: ordinal, end_block_ordinal: ordinal, block_count: 0, start_paragraph_ordinal: positions.paragraph_ordinal, end_paragraph_ordinal: positions.paragraph_ordinal, start_block_id: block_id.clone(), end_block_id: block_id.clone(), query_match_count: 0, matched_reference_count: 0, sample: String::new(), }); section.end_block_ordinal = ordinal; section.block_count += 1; section.end_paragraph_ordinal = positions .paragraph_ordinal .or(section.end_paragraph_ordinal); section.end_block_id = block_id.clone(); section.query_match_count += query_match_count(block_content, &query_terms); if reference_block_ids.contains(&block_id) { section.matched_reference_count += 1; } if section.sample.is_empty() { section.sample = clean_lightrag_text_for_display(block_content) .chars() .take(220) .collect(); } } ordinal += 1; } if !truncated { if let Some(section) = current.take() { sections.push(section); } } let query_matched_sections = sections .iter() .filter(|section| section.query_match_count > 0) .count(); let reference_matched_sections = sections .iter() .filter(|section| section.matched_reference_count > 0) .count(); Some(json!({ "sourceId": entry.source_id, "sourceRootRelativePath": entry.source_root_relative_path, "lightRagDocId": entry.light_rag_doc_id, "lightRagFilePath": entry.light_rag_file_path, "sidecarAvailable": true, "sidecarPath": root_relative_path(root_path, &sidecar_path).unwrap_or_else(|_| sidecar_path.display().to_string()), "sectionCount": sections.len(), "queryMatchedSections": query_matched_sections, "referenceMatchedSections": reference_matched_sections, "truncated": truncated, "sections": sections, })) } fn heading_title_from_block(content: &str) -> Option { let first = content.lines().find(|line| !line.trim().is_empty())?.trim(); if !first.starts_with('#') { return None; } let title = clean_lightrag_text_for_display(first.trim_start_matches('#').trim()); (!title.is_empty()).then_some(title) } fn query_match_count(content: &str, query_terms: &[String]) -> u64 { if query_terms.is_empty() { return 0; } let normalized = normalize_text_for_match(content).to_ascii_lowercase(); query_terms .iter() .filter(|term| block_matches_query_term(&normalized, term)) .count() as u64 } #[derive(Debug, Clone, Default)] struct SidecarTextStats { blocks: usize, meaningful_blocks: usize, } fn sidecar_text_stats(path: &Path) -> Result { let content = fs::read_to_string(path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_sidecar_read_failed", format!("无法读取 LightRAG sidecar {}: {error}", path.display()), ) })?; let mut stats = SidecarTextStats::default(); for line in content.lines() { let Ok(block) = serde_json::from_str::(line) else { continue; }; if block.get("type").and_then(Value::as_str) != Some("content") { continue; } stats.blocks += 1; let content = block .get("content") .and_then(Value::as_str) .unwrap_or_default(); if quote_content_diagnostics(Some(content), "sidecar") .get("ocrTextExposed") .and_then(Value::as_bool) .unwrap_or(false) { stats.meaningful_blocks += 1; } } Ok(stats) } fn lightrag_chunk_content_for_reference_ids( source_chunk_id: &Value, chunk_id: &Value, ) -> Option { lightrag_chunk_id_candidates(source_chunk_id, chunk_id) .into_iter() .find_map(|id| lightrag_chunk_content_for_id(&id)) } fn lightrag_chunk_sidecar_for_reference_ids( source_chunk_id: &Value, chunk_id: &Value, ) -> Option { lightrag_chunk_id_candidates(source_chunk_id, chunk_id) .into_iter() .find_map(|id| { lightrag_chunk_value_for_id(&id).and_then(|chunk| chunk.get("sidecar").cloned()) }) } fn lightrag_chunk_id_candidates(source_chunk_id: &Value, chunk_id: &Value) -> Vec { let mut ids = Vec::new(); for value in [source_chunk_id, chunk_id] { let Some(raw) = value .as_str() .map(str::trim) .filter(|value| !value.is_empty()) else { continue; }; ids.push(raw.to_string()); if let Some((base, _)) = raw.split_once("#match-") { ids.push(base.to_string()); } } ids.sort(); ids.dedup(); ids } fn lightrag_chunk_value_for_id(chunk_id: &str) -> Option { let content = fs::read_to_string(lightrag_working_dir().join("kv_store_text_chunks.json")).ok()?; let chunks = serde_json::from_str::(&content).ok()?; chunks.get(chunk_id).cloned() } fn lightrag_chunk_content_for_id(chunk_id: &str) -> Option { lightrag_chunk_value_for_id(chunk_id) .as_ref() .and_then(|chunk| chunk.get("content")) .and_then(Value::as_str) .map(ToOwned::to_owned) } fn read_registry( root_path: &Path, workspace_id: &str, root_uri: &str, ) -> Result { let path = registry_path(root_path); if !path.exists() { return Ok(KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: workspace_id.to_string(), root_uri: root_uri.to_string(), updated_at_ms: now_ms(), indexed_roots: Vec::new(), entries: Vec::new(), }); } let content = fs::read_to_string(&path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_registry_read_failed", format!("无法读取 LightRAG source registry: {error}"), ) })?; serde_json::from_str(&content).map_err(|error| { WebError::bad_request_code( "knowledge_rag_registry_invalid", format!("LightRAG source registry JSON 无效: {error}"), ) }) } fn write_registry( root_path: &Path, registry: &mut KnowledgeRagSourceRegistry, ) -> Result<(), WebError> { registry.updated_at_ms = now_ms(); let path = registry_path(root_path); if let Some(parent) = path.parent() { fs::create_dir_all(parent).map_err(|error| { WebError::bad_request_code( "knowledge_rag_registry_dir_failed", format!("无法创建 LightRAG source registry 目录: {error}"), ) })?; } let content = serde_json::to_string_pretty(registry).map_err(|error| { WebError::internal(format!("无法序列化 LightRAG source registry: {error}")) })?; fs::write(&path, format!("{content}\n")).map_err(|error| { WebError::bad_request_code( "knowledge_rag_registry_write_failed", format!("无法写入 LightRAG source registry: {error}"), ) }) } fn upsert_registry_entry( registry: &mut KnowledgeRagSourceRegistry, next: KnowledgeRagSourceRegistryEntry, ) { if let Some(existing) = registry .entries .iter_mut() .find(|entry| entry.source_path == next.source_path) { *existing = next; return; } registry.entries.push(next); } fn upsert_indexed_root_for_request( registry: &mut KnowledgeRagSourceRegistry, root_path: &Path, source_path: &str, context: &RequestContext, ) -> Result<(), WebError> { let canonical = ensure_source_path_in_root(root_path, source_path, context)?; let relative = root_relative_path(root_path, &canonical)?; let normalized = relative.trim_matches('/').replace('\\', "/"); let recursive = canonical.is_dir(); let now = now_ms(); if let Some(existing) = registry .indexed_roots .iter_mut() .find(|root| root.root_relative_path == normalized) { existing.recursive = recursive; existing.run_on_change.get_or_insert(true); existing.updated_at_ms = now; return Ok(()); } registry.indexed_roots.push(KnowledgeRagIndexedRoot { root_relative_path: normalized, recursive, exclude_patterns: Vec::new(), run_on_change: Some(true), updated_at_ms: now, }); registry .indexed_roots .sort_by(|left, right| left.root_relative_path.cmp(&right.root_relative_path)); Ok(()) } fn registry_path(root_path: &Path) -> PathBuf { root_path .join(".mnote") .join("index") .join("lightrag-source-registry.json") } fn document_structure_index_path(root_path: &Path) -> PathBuf { root_path .join(".mnote") .join("index") .join("lightrag-document-structure-index.json") } fn refresh_document_structure_index( root_path: &Path, registry: &KnowledgeRagSourceRegistry, ) -> Result<(), WebError> { let payload = document_structure_index_payload( root_path, registry, &[], &[], None, DOCUMENT_STRUCTURE_INDEX_MAX_DOCUMENTS_PERSISTED, )?; let path = document_structure_index_path(root_path); if let Some(parent) = path.parent() { fs::create_dir_all(parent).map_err(|error| { WebError::bad_request_code( "knowledge_rag_structure_index_dir_failed", format!("无法创建 LightRAG document_structure_index 目录: {error}"), ) })?; } let content = serde_json::to_string_pretty(&payload).map_err(|error| { WebError::internal(format!( "无法序列化 LightRAG document_structure_index: {error}" )) })?; fs::write(&path, format!("{content}\n")).map_err(|error| { WebError::bad_request_code( "knowledge_rag_structure_index_write_failed", format!("无法写入 LightRAG document_structure_index: {error}"), ) }) } #[derive(Debug, Clone)] struct ResolvedKnowledgeRagSource { canonical_path: PathBuf, requested_path: String, source_kind: &'static str, } fn resolve_knowledge_rag_sources( root_path: &Path, source_path: &str, context: &RequestContext, ) -> Result, WebError> { let canonical = ensure_source_path_in_root(root_path, source_path, context)?; if canonical.is_file() { if !knowledge_rag_source_supported_file(&canonical) { return Err(WebError::bad_request_code( "knowledge_rag_source_unsupported", "资料库 source 文件类型暂不支持", ) .with_context(context)); } return Ok(vec![ResolvedKnowledgeRagSource { canonical_path: canonical, requested_path: source_path.trim().to_string(), source_kind: "file", }]); } if canonical.is_dir() { let mut files = collect_knowledge_rag_directory_sources(root_path, &canonical, context)?; if files.is_empty() { return Err(WebError::bad_request_code( "knowledge_rag_source_directory_empty", "资料库目录中没有可索引文件", ) .with_context(context)); } files.sort(); files.dedup(); return Ok(files .into_iter() .map(|path| ResolvedKnowledgeRagSource { canonical_path: path, requested_path: source_path.trim().to_string(), source_kind: "directory", }) .collect()); } Err(WebError::bad_request_code( "knowledge_rag_source_not_file_or_directory", "资料库 source 必须是文件或目录", ) .with_context(context)) } fn collect_knowledge_rag_directory_sources( root_path: &Path, directory: &Path, context: &RequestContext, ) -> Result, WebError> { let mut files = Vec::new(); let mut stack = vec![directory.to_path_buf()]; while let Some(current_dir) = stack.pop() { let mut entries = fs::read_dir(¤t_dir) .map_err(|error| { WebError::bad_request_code( "knowledge_rag_source_directory_read_failed", format!("无法读取资料库目录 {}: {error}", current_dir.display()), ) .with_context(context) })? .filter_map(Result::ok) .collect::>(); entries.sort_by_key(|entry| entry.path()); for entry in entries { let path = entry.path(); let canonical = path.canonicalize().map_err(|error| { WebError::bad_request_code( "knowledge_rag_source_unavailable", format!("无法访问资料库 source: {error}"), ) .with_context(context) })?; if !canonical.starts_with(root_path) { continue; } if canonical.is_dir() { if should_skip_knowledge_rag_directory(&canonical) { continue; } stack.push(canonical); continue; } if canonical.is_file() && knowledge_rag_source_supported_file(&canonical) { files.push(canonical); if files.len() > MAX_INGEST_SOURCES_PER_REQUEST { return Err(WebError::bad_request_code( "knowledge_rag_source_limit_exceeded", format!( "单次资料库索引最多支持 {MAX_INGEST_SOURCES_PER_REQUEST} 个文件,请缩小目录范围" ), ) .with_context(context)); } } } } Ok(files) } fn should_skip_knowledge_rag_directory(path: &Path) -> bool { let name = path .file_name() .and_then(|value| value.to_str()) .unwrap_or_default(); matches!( name, ".git" | ".mnote" | "node_modules" | "target" | "__parsed__" | ".venv" ) } fn knowledge_rag_source_supported_file(path: &Path) -> bool { path.extension() .and_then(|value| value.to_str()) .map(|value| { KNOWLEDGE_RAG_SOURCE_EXTENSIONS .iter() .any(|extension| value.eq_ignore_ascii_case(extension)) }) .unwrap_or(false) } fn lightrag_scan_supported_file(path: &Path) -> bool { path.extension() .and_then(|value| value.to_str()) .map(|value| { LIGHTRAG_SCAN_SOURCE_EXTENSIONS .iter() .any(|extension| value.eq_ignore_ascii_case(extension)) }) .unwrap_or(false) } struct StagedLightRagSource { light_rag_file_path: String, staged_path: PathBuf, } fn stage_lightrag_source( canonical: &Path, file_name: &str, parser_hint: Option<&str>, input_dir: &Path, context: &RequestContext, ) -> Result { if lightrag_scan_supported_file(canonical) { let light_rag_file_path = lightrag_symlink_name(canonical, file_name, parser_hint); let symlink_path = input_dir.join(&light_rag_file_path); replace_lightrag_input_file(&symlink_path, context)?; create_symlink(canonical, &symlink_path, context)?; return Ok(StagedLightRagSource { light_rag_file_path, staged_path: symlink_path, }); } let asset_file_path = lightrag_symlink_name(canonical, file_name, None); let asset_symlink_path = input_dir.join(&asset_file_path); replace_lightrag_input_file(&asset_symlink_path, context)?; create_symlink(canonical, &asset_symlink_path, context)?; let wrapper_file_path = format!("{asset_file_path}.md"); let wrapper_path = input_dir.join(&wrapper_file_path); replace_lightrag_input_file(&wrapper_path, context)?; let wrapper_markdown = format!("# {file_name}\n\n![{file_name}](<{asset_file_path}>)\n"); fs::write(&wrapper_path, wrapper_markdown).map_err(|error| { WebError::bad_request_code( "knowledge_rag_wrapper_write_failed", format!("无法写入 LightRAG 图片包装 Markdown: {error}"), ) .with_context(context) })?; Ok(StagedLightRagSource { light_rag_file_path: wrapper_file_path, staged_path: wrapper_path, }) } fn replace_lightrag_input_file(path: &Path, context: &RequestContext) -> Result<(), WebError> { if path.exists() || path.symlink_metadata().is_ok() { fs::remove_file(path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_symlink_replace_failed", format!("无法替换 LightRAG source 文件: {error}"), ) .with_context(context) })?; } Ok(()) } fn ensure_source_path_in_root( root_path: &Path, source_path: &str, context: &RequestContext, ) -> Result { let raw_path = source_path .trim() .strip_prefix("file://") .map(PathBuf::from) .unwrap_or_else(|| PathBuf::from(source_path.trim())); let target = if raw_path.is_absolute() { raw_path } else { root_path.join(raw_path) }; let canonical = target.canonicalize().map_err(|error| { WebError::bad_request_code( "knowledge_rag_source_unavailable", format!("无法访问资料库 source: {error}"), ) .with_context(context) })?; if !canonical.starts_with(root_path) { return Err(WebError::bad_request_code( "knowledge_rag_source_root_escape", "资料库 source 不能越过授权目录", ) .with_context(context)); } Ok(canonical) } fn root_relative_path(root_path: &Path, source_path: &Path) -> Result { let relative = source_path.strip_prefix(root_path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_source_relative_failed", format!("无法生成资料库 source 相对路径: {error}"), ) })?; Ok(relative.to_string_lossy().replace('\\', "/")) } fn create_symlink( source: &Path, symlink_path: &Path, context: &RequestContext, ) -> Result<(), WebError> { #[cfg(unix)] { std::os::unix::fs::symlink(source, symlink_path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_symlink_failed", format!("无法创建 LightRAG source symlink: {error}"), ) .with_context(context) }) } #[cfg(not(unix))] { fs::copy(source, symlink_path).map(|_| ()).map_err(|error| { WebError::bad_request_code( "knowledge_rag_staging_copy_failed", format!("无法创建 LightRAG source staging 文件: {error}"), ) .with_context(context) }) } } fn source_hash(path: &Path) -> Result { let bytes = fs::read(path).map_err(|error| { WebError::bad_request_code( "knowledge_rag_source_hash_failed", format!("无法读取资料库 source 以计算 hash: {error}"), ) })?; Ok(format!("mnote-fnv64:{:016x}", fnv64(&bytes))) } fn fnv64(bytes: &[u8]) -> u64 { let mut hash = 0xcbf29ce484222325u64; for byte in bytes { hash ^= u64::from(*byte); hash = hash.wrapping_mul(0x100000001b3); } hash } fn lightrag_symlink_name(path: &Path, file_name: &str, parser_hint: Option<&str>) -> String { let prefix = format!("mnote-{}-", short_hash(&path.display().to_string())); let Some(parser_hint) = parser_hint else { return format!("{prefix}{file_name}"); }; if strip_one_supported_parser_hint(file_name).is_some() { return format!("{prefix}{file_name}"); } let path = Path::new(file_name); let stem = path .file_stem() .and_then(|value| value.to_str()) .unwrap_or(file_name); let extension = path.extension().and_then(|value| value.to_str()); match extension { Some(extension) if !extension.is_empty() => { format!("{prefix}{stem}.[{parser_hint}].{extension}") } _ => format!("{prefix}{stem}.[{parser_hint}]"), } } fn normalize_parser_hint( value: Option<&str>, context: &RequestContext, ) -> Result, WebError> { let Some(value) = value.map(str::trim).filter(|value| !value.is_empty()) else { return Ok(None); }; let hint = value.trim_matches('[').trim_matches(']').trim(); if let Some(normalized) = normalize_supported_parser_hint(hint) { return Ok(Some(normalized)); } Err(WebError::bad_request_code( "knowledge_rag_parser_hint_invalid", "LightRAG parserHint 仅支持 legacy/native/mineru/docling 及 native-P 这类 engine-options", ) .with_context(context)) } fn default_lightrag_parser_hint_for_source(path: &Path, file_name: &str) -> Option { let extension = path .extension() .and_then(|value| value.to_str()) .map(|value| value.to_ascii_lowercase()); let large_source = path .metadata() .map(|metadata| metadata.len() >= LARGE_DOCUMENT_SKIP_KG_MIN_BYTES) .unwrap_or(false); let lower_name = file_name.to_ascii_lowercase(); let is_ocr_layered_docx = extension.as_deref() == Some("docx") && lower_name.contains("[ocr]") && lower_name.contains(".layered"); if large_source && matches!(extension.as_deref(), Some("docx" | "pdf")) { if is_ocr_layered_docx { return Some("native-P!".into()); } return Some("-P!".into()); } if is_ocr_layered_docx { return Some("native-P".into()); } None } fn normalize_supported_parser_hint(value: &str) -> Option { let trimmed = value.trim(); if trimmed.is_empty() { return None; } if let Some(options) = trimmed.strip_prefix('-') { return valid_lightrag_parser_options(options).then(|| format!("-{options}")); } let (engine, options) = trimmed .split_once('-') .map(|(engine, options)| (engine, Some(options))) .unwrap_or((trimmed, None)); let engine = engine.to_ascii_lowercase(); if !matches!(engine.as_str(), "legacy" | "native" | "mineru" | "docling") { return None; } match options { Some(options) if valid_lightrag_parser_options(options) => { Some(format!("{engine}-{options}")) } Some(_) => None, None => Some(engine), } } fn valid_lightrag_parser_options(options: &str) -> bool { !options.is_empty() && options .chars() .all(|value| matches!(value, 'i' | 't' | 'e' | '!' | 'R' | 'F' | 'P')) } fn document_for_registry_entry<'a>( by_file_path: &'a BTreeMap, entry: &KnowledgeRagSourceRegistryEntry, ) -> Option<&'a Value> { by_file_path.get(&entry.light_rag_file_path).or_else(|| { strip_one_supported_parser_hint(&entry.light_rag_file_path) .as_ref() .and_then(|path| by_file_path.get(path)) }) } fn lightrag_file_path_matches(entry: &KnowledgeRagSourceRegistryEntry, file_path: &str) -> bool { entry.light_rag_file_path == file_path || strip_one_supported_parser_hint(&entry.light_rag_file_path).as_deref() == Some(file_path) } fn strip_one_supported_parser_hint(file_name: &str) -> Option { let mut search_start = 0usize; while let Some(offset) = file_name.get(search_start..)?.find(".[") { let start = search_start + offset; let hint_start = start + 2; let Some(end_offset) = file_name.get(hint_start..)?.find(']') else { break; }; let hint_end = hint_start + end_offset; let hint = file_name.get(hint_start..hint_end)?; if normalize_supported_parser_hint(hint).is_some() { let mut stripped = String::new(); stripped.push_str(file_name.get(..start)?); stripped.push_str(file_name.get(hint_end + 1..)?); return Some(stripped); } search_start = hint_end + 1; } None } fn short_hash(value: &str) -> String { let mut hasher = DefaultHasher::new(); value.hash(&mut hasher); format!("{:016x}", hasher.finish()) } fn short_citation_id( source_id: &str, chunk_id: &str, block_id: &str, occurrence_index: u64, fingerprint: &str, ) -> String { short_hash(&format!( "lightrag\n{source_id}\n{chunk_id}\n{block_id}\n{occurrence_index}\n{fingerprint}" )) .chars() .take(4) .collect() } fn lightrag_endpoint() -> String { env::var("MNOTE_LIGHTRAG_ENDPOINT") .or_else(|_| env::var("LIGHTRAG_ENDPOINT")) .unwrap_or_else(|_| DEFAULT_LIGHTRAG_ENDPOINT.into()) .trim() .trim_end_matches('/') .to_string() } fn lightrag_dashboard_url() -> String { env::var("MNOTE_LIGHTRAG_DASHBOARD_URL").unwrap_or_else(|_| lightrag_endpoint()) } fn lightrag_input_dir() -> PathBuf { if let Ok(value) = env::var("MNOTE_LIGHTRAG_INPUT_DIR") { return PathBuf::from(value); } if let Some(value) = read_lightrag_dotenv_value("INPUT_DIR") { return PathBuf::from(value); } env::var("INPUT_DIR") .map(PathBuf::from) .unwrap_or_else(|_| PathBuf::from(DEFAULT_LIGHTRAG_INPUT_DIR)) } fn lightrag_working_dir() -> PathBuf { if let Ok(value) = env::var("MNOTE_LIGHTRAG_WORKING_DIR") { return PathBuf::from(value); } if let Some(value) = read_lightrag_dotenv_value("WORKING_DIR") { return PathBuf::from(value); } env::var("WORKING_DIR") .map(PathBuf::from) .unwrap_or_else(|_| PathBuf::from(DEFAULT_LIGHTRAG_WORKING_DIR)) } fn lightrag_api_key() -> Option { env::var("MNOTE_LIGHTRAG_API_KEY") .ok() .map(|value| value.trim().trim_matches('"').to_string()) .filter(|value| !value.is_empty()) .or_else(|| read_lightrag_dotenv_value("LIGHTRAG_API_KEY")) .or_else(|| env::var("LIGHTRAG_API_KEY").ok()) .map(|value| value.trim().trim_matches('"').to_string()) .filter(|value| !value.is_empty()) } fn read_lightrag_dotenv_value(key: &str) -> Option { let path = env::var("MNOTE_LIGHTRAG_ENV_FILE") .map(PathBuf::from) .unwrap_or_else(|_| PathBuf::from("/mnt/Data1T/Mnote_data/lightrag/LightRAG/.env")); let content = fs::read_to_string(path).ok()?; for line in content.lines() { let line = line.trim(); if line.is_empty() || line.starts_with('#') { continue; } let Some((name, value)) = line.split_once('=') else { continue; }; if name.trim() != key { continue; } let value = value.trim().trim_matches('"').to_string(); if !value.is_empty() { return Some(value); } } None } fn effective_workspace_id(input: Option<&str>, root_uri: &str) -> String { input .map(str::trim) .filter(|value| !value.is_empty()) .map(ToOwned::to_owned) .unwrap_or_else(|| { local_folder_source::local_workspace_id_from_root_uri(root_uri) .unwrap_or_else(|_| format!("local-ws:{}", short_hash(root_uri))) }) } fn now_ms() -> u128 { SystemTime::now() .duration_since(UNIX_EPOCH) .map(|value| value.as_millis()) .unwrap_or_default() } #[cfg(test)] mod tests { use super::*; use axum::http::{HeaderMap, Method, Uri}; use std::sync::{Mutex, OnceLock}; fn env_lock() -> &'static Mutex<()> { static LOCK: OnceLock> = OnceLock::new(); LOCK.get_or_init(|| Mutex::new(())) } fn test_context() -> RequestContext { RequestContext::from_http_parts(&Method::GET, &Uri::from_static("/"), &HeaderMap::new()) } fn temp_root(label: &str) -> PathBuf { let root = std::env::temp_dir().join(format!("{label}-{}", now_ms())); fs::create_dir_all(&root).expect("temp root"); root } fn test_registry_entry( root: &Path, relative: &str, doc_id: Option<&str>, provider_status: Option<&str>, indexed_at_ms: Option, deleted_at_ms: Option, stale: bool, ) -> KnowledgeRagSourceRegistryEntry { KnowledgeRagSourceRegistryEntry { source_id: format!("src-{relative}"), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), source_path: root.join(relative).display().to_string(), source_root_relative_path: relative.into(), source_hash: "mnote-fnv64:test".into(), light_rag_doc_id: doc_id.map(ToOwned::to_owned), light_rag_status: provider_status.map(ToOwned::to_owned), light_rag_file_path: relative.into(), symlink_path: format!("/tmp/{relative}"), parser_hint: None, indexed_at_ms, deleted_at_ms, stale, updated_at_ms: 2, } } #[tokio::test] async fn retired_local_ocr_endpoint_points_to_lightrag_replacements() { let (status, Json(payload)) = retired_local_ocr_endpoint(Extension(test_context())).await; assert_eq!(status, StatusCode::GONE); assert_eq!(payload["code"], "mnote_local_ocr_retired"); assert_eq!(payload["replacement"]["provider"], "lightrag"); assert_eq!( payload["replacement"]["ingest"], "/api/knowledge-rag/ingest" ); } #[test] fn lightrag_document_summaries_expose_status_groups_for_ui_sync() { let docs = json!({ "statuses": { "processed": [ {"id": "doc-ok", "file_path": "ok.pdf", "status": "processed", "summary": "ok"} ], "failed": [ {"id": "doc-fail", "file_path": "fail.pdf", "status": "failed", "summary": "bad"} ] } }); let counts = lightrag_document_status_group_counts(&docs); assert_eq!(counts.get("processed"), Some(&1)); assert_eq!(counts.get("failed"), Some(&1)); let summaries = lightrag_document_summaries(&docs); assert_eq!(summaries.len(), 2); assert!(summaries.iter().any(|item| item["filePath"] == "ok.pdf")); assert!(summaries.iter().any(|item| item["statusGroup"] == "failed")); let by_path = lightrag_documents_by_file_path(&docs); assert!(by_path.contains_key("ok.pdf")); assert!(by_path.contains_key("fail.pdf")); } #[test] fn lightrag_pipeline_summary_extracts_chunk_progress_for_ui() { let status = json!({ "busy": true, "scanning": true, "destructive_busy": false, "pending_enqueues": 1, "docs": 1, "batchs": 2, "cur_batch": 1, "job_name": "book.docx", "latest_message": "Chunk 212 of 333 extracted 4 Ent + 0 Rel doc-ff0b60997a285a85e5704a114d7b3ffa-chunk-212" }); let summary = lightrag_pipeline_status_summary(&status); assert_eq!(summary["busy"], true); assert_eq!(summary["scanning"], true); assert_eq!(summary["pendingEnqueues"], 1); assert_eq!(summary["docs"], 1); assert_eq!(summary["batches"], 2); assert_eq!(summary["currentBatch"], 1); assert_eq!(summary["progress"]["current"], 212); assert_eq!(summary["progress"]["total"], 333); assert_eq!( summary["progress"]["docId"], "doc-ff0b60997a285a85e5704a114d7b3ffa" ); } #[test] fn lightrag_pipeline_summary_extracts_chunk_progress_from_history() { let status = json!({ "busy": true, "scanning": false, "job_name": "book.pdf", "latest_message": "Merging stage 1/1: book.pdf", "history_messages": [ "Analyzing multimodal: doc-e51400269a2c41a2fcb22f1071193fad", "Chunk 5 of 21 extracted 2 Ent + 1 Rel doc-e51400269a2c41a2fcb22f1071193fad-mm-drawing-001", "Merging stage 1/1: book.pdf" ] }); let summary = lightrag_pipeline_status_summary(&status); assert_eq!(summary["progress"]["current"], 5); assert_eq!(summary["progress"]["total"], 21); assert_eq!( summary["progress"]["docId"], "doc-e51400269a2c41a2fcb22f1071193fad" ); assert_eq!(summary["historyMessages"].as_array().unwrap().len(), 3); } #[test] fn old_registry_json_defaults_indexed_roots() { let registry: KnowledgeRagSourceRegistry = serde_json::from_value(json!({ "schema": REGISTRY_SCHEMA, "workspaceId": "ws", "rootUri": "file:///tmp/root", "updatedAtMs": 1, "entries": [] })) .expect("registry"); assert!(registry.indexed_roots.is_empty()); } #[test] fn ingest_request_records_explicit_indexed_root_scope() { let root = temp_root("mnote-knowledge-rag-indexed-root"); fs::create_dir_all(root.join("papers")).expect("papers"); fs::write(root.join("papers").join("a.pdf"), b"pdf").expect("pdf"); let mut registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), updated_at_ms: 1, indexed_roots: Vec::new(), entries: Vec::new(), }; upsert_indexed_root_for_request(&mut registry, &root, "papers", &test_context()) .expect("indexed root"); assert_eq!(registry.indexed_roots.len(), 1); assert_eq!(registry.indexed_roots[0].root_relative_path, "papers"); assert!(registry.indexed_roots[0].recursive); assert_eq!(registry.indexed_roots[0].run_on_change, Some(true)); let _ = fs::remove_dir_all(root); } #[test] fn references_only_search_result_is_clickable_source_hit() { let reference = json!({ "provider": "lightrag", "sourceId": "src-paper", "sourceRootRelativePath": "papers/a.pdf", "chunkId": "chunk-1", "quote": "Canterbury corpus compression ratio appears in this paragraph.", "locatorDegraded": false, "locator": { "rootUri": "file:///tmp/root", "ownerDocumentId": "local-md:papers~2FHost.md", "resourcePath": "papers/a.pdf", "resourceKind": "pdf", "page": 3 }, "citationUrl": "/documents/local-md:papers~2FHost.md?resourcePath=papers%2Fa.pdf", "citationMarkdown": "[a.pdf · p.3](/documents/local-md:papers~2FHost.md)" }); let result = knowledge_rag_search_result(&reference, 0, "file:///tmp/root", "三甲基硅基"); assert_eq!(result["provider"], "lightrag"); assert_eq!(result["query"], "三甲基硅基"); assert_eq!(result["matchSource"], "lightrag_reference"); assert_eq!(result["resourceType"], "pdf"); assert_eq!(result["path"], "papers/a.pdf"); assert_eq!(result["locator"]["page"], 3); assert_eq!(result["citationUrl"], reference["citationUrl"]); } #[test] fn reference_mapping_uses_registry_source_path() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/books/a.pdf".into(), source_root_relative_path: "books/a.pdf".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-a.pdf".into(), symlink_path: "/tmp/input/mnote-hash-a.pdf".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let mapped = map_reference_plan( &json!({"file_path":"mnote-hash-a.pdf","chunk_id":"chunk1"}), ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert_eq!(mapped["sourceRootRelativePath"], "books/a.pdf"); assert_eq!(mapped["locatorDegraded"], true); assert_eq!(mapped["openAction"]["params"]["path"], "books/a.pdf"); } #[test] fn missing_registry_reference_is_degraded() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![], }; let mapped = map_reference_plan( &json!({"file_path":"unknown.pdf"}), ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert_eq!(mapped["locatorDegraded"], true); assert!(mapped["citationMarkdown"] .as_str() .unwrap() .contains("来源定位降级")); } #[test] fn reference_mapping_does_not_match_deleted_entry_when_doc_id_missing() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "deleted".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/deleted.md".into(), source_root_relative_path: "deleted.md".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: None, light_rag_status: None, light_rag_file_path: "deleted.md".into(), symlink_path: "/tmp/input/deleted.md".into(), parser_hint: None, indexed_at_ms: None, deleted_at_ms: Some(2), stale: true, updated_at_ms: 2, }], }; let mapped = map_reference_plan( &json!({"file_path":"active.md","chunk_id":"chunk-active"}), ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert!(mapped["sourceRootRelativePath"].is_null()); assert_eq!(mapped["stale"], false); assert_eq!(mapped["deleted"], false); } #[test] fn degraded_known_resource_keeps_clickable_resource_tab_url() { let root = temp_root("mnote-knowledge-rag-docx-fallback"); fs::create_dir_all(root.join("docs")).expect("docs"); fs::write(root.join("docs").join("Host.md"), "# Host\n").expect("host"); fs::write(root.join("docs").join("a.docx"), b"docx").expect("docx"); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: root.join("docs").join("a.docx").display().to_string(), source_root_relative_path: "docs/a.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-a.docx".into(), symlink_path: "/tmp/input/mnote-hash-a.docx".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let mapped = map_reference_plan( &json!({"file_path":"mnote-hash-a.docx","chunk_id":"chunk1"}), ®istry, "file:///tmp/root", &root, None, ); assert_eq!(mapped["locatorDegraded"], true); let citation_url = mapped["citationUrl"].as_str().expect("citation url"); assert!(citation_url.starts_with("/documents/local-md:docs~2FHost.md?")); assert!(citation_url.contains("resourceTab=")); assert!(citation_url.contains("resourcePath=docs%2Fa.docx")); assert!(mapped["citationMarkdown"] .as_str() .unwrap() .contains("来源定位降级")); let _ = fs::remove_dir_all(root); } #[test] fn source_state_marks_deleted_and_changed_entries_stale() { let root = temp_root("mnote-knowledge-rag-source-state"); fs::create_dir_all(root.join("docs")).expect("docs"); let changed_path = root.join("docs").join("changed.md"); fs::write(&changed_path, "new content").expect("changed"); let missing_path = root.join("docs").join("missing.md"); let mut registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![ KnowledgeRagSourceRegistryEntry { source_id: "missing".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: missing_path.display().to_string(), source_root_relative_path: "docs/missing.md".into(), source_hash: "mnote-fnv64:old".into(), light_rag_doc_id: Some("doc-missing".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "missing.md".into(), symlink_path: "/tmp/input/missing.md".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }, KnowledgeRagSourceRegistryEntry { source_id: "changed".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: changed_path.display().to_string(), source_root_relative_path: "docs/changed.md".into(), source_hash: "mnote-fnv64:old".into(), light_rag_doc_id: Some("doc-changed".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "changed.md".into(), symlink_path: "/tmp/input/changed.md".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }, ], }; let stale_doc_ids = sync_registry_source_state(&mut registry, 42).expect("sync"); assert_eq!(stale_doc_ids, vec!["doc-changed", "doc-missing"]); assert_eq!(registry.entries[0].deleted_at_ms, Some(42)); assert_eq!( registry.entries[0].light_rag_doc_id.as_deref(), Some("doc-missing") ); assert_eq!( registry.entries[0].light_rag_status.as_deref(), Some("delete_submitted") ); assert_eq!(registry.entries[0].indexed_at_ms, None); assert!(registry.entries[0].stale); assert_eq!(registry.entries[1].deleted_at_ms, None); assert_eq!( registry.entries[1].light_rag_doc_id.as_deref(), Some("doc-changed") ); assert_eq!( registry.entries[1].light_rag_status.as_deref(), Some("delete_submitted") ); assert_eq!(registry.entries[1].indexed_at_ms, None); assert!(registry.entries[1].stale); assert_ne!(registry.entries[1].source_hash, "mnote-fnv64:old"); let _ = fs::remove_dir_all(root); } #[test] fn provider_delete_confirmation_clears_doc_id_for_deleted_or_stale_entries() { let root = temp_root("mnote-knowledge-rag-delete-confirmed"); let mut deleted = test_registry_entry( &root, "deleted.pdf", Some("doc-deleted"), Some("delete_submitted"), Some(2), Some(3), true, ); let mut changed = test_registry_entry( &root, "changed.pdf", Some("doc-changed"), Some("delete_submitted"), Some(2), None, true, ); let active = test_registry_entry( &root, "active.pdf", Some("doc-active"), Some("processed"), Some(2), None, false, ); assert!(knowledge_rag_provider_delete_confirmed(&deleted)); mark_registry_entry_delete_completed(&mut deleted, 42); assert_eq!(deleted.light_rag_doc_id, None); assert_eq!(deleted.indexed_at_ms, None); assert_eq!( deleted.light_rag_status.as_deref(), Some("delete_completed") ); assert!(knowledge_rag_provider_delete_confirmed(&changed)); mark_registry_entry_delete_completed(&mut changed, 43); assert_eq!(changed.light_rag_doc_id, None); assert_eq!( changed.light_rag_status.as_deref(), Some("delete_completed") ); assert!(!knowledge_rag_provider_delete_confirmed(&active)); assert_eq!(active.light_rag_doc_id.as_deref(), Some("doc-active")); let _ = fs::remove_dir_all(root); } #[test] fn source_statuses_distinguish_indexed_processing_failed_and_deleted() { let root = temp_root("mnote-knowledge-rag-source-statuses"); fs::write(root.join("indexed.pdf"), b"indexed").expect("indexed"); fs::write(root.join("submitted.pdf"), b"submitted").expect("submitted"); fs::write(root.join("failed.pdf"), b"failed").expect("failed"); fs::write(root.join("deleting.pdf"), b"deleting").expect("deleting"); fs::write(root.join("removed.pdf"), b"removed").expect("removed"); let root_uri = format!("file://{}", root.display()); let mut registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: root_uri.clone(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![ test_registry_entry( &root, "indexed.pdf", Some("doc-indexed"), Some("processed"), Some(2), None, false, ), test_registry_entry( &root, "submitted.pdf", None, Some("submitted"), None, None, false, ), test_registry_entry(&root, "failed.pdf", None, Some("failed"), None, None, false), test_registry_entry( &root, "deleting.pdf", Some("doc-deleting"), Some("delete_submitted"), Some(2), Some(3), true, ), test_registry_entry( &root, "removed.pdf", None, Some("delete_completed"), None, Some(3), true, ), ], }; write_registry(&root, &mut registry).expect("write registry"); let statuses = knowledge_rag_source_statuses(&root, "ws", &root_uri).expect("statuses"); assert!(statuses.indexed_paths.contains("indexed.pdf")); assert!(statuses.indexing_paths.contains("submitted.pdf")); assert!(statuses.indexing_paths.contains("deleting.pdf")); assert!(statuses.failed_paths.contains("failed.pdf")); assert!(!statuses.indexed_paths.contains("removed.pdf")); assert!(!statuses.indexing_paths.contains("removed.pdf")); assert!(!statuses.failed_paths.contains("removed.pdf")); let _ = fs::remove_dir_all(root); } #[test] fn image_source_is_staged_directly_for_lightrag_scan() { let root = temp_root("mnote-knowledge-rag-image-direct-scan"); let input_dir = root.join("inputs"); fs::create_dir_all(&input_dir).expect("input dir"); let image_path = root.join("image copy 6.png"); fs::write(&image_path, b"png").expect("image"); let staged = stage_lightrag_source( &image_path, "image copy 6.png", None, &input_dir, &test_context(), ) .expect("stage image"); assert!(staged.light_rag_file_path.ends_with(".png")); assert!(staged.staged_path.symlink_metadata().is_ok()); #[cfg(unix)] assert!(staged .staged_path .symlink_metadata() .expect("staged metadata") .file_type() .is_symlink()); let _ = fs::remove_dir_all(root); } #[test] fn registry_prune_predicate_keeps_inflight_delete_until_confirmed() { let root = temp_root("mnote-knowledge-rag-prune-predicate"); let active = test_registry_entry( &root, "active.pdf", Some("doc-active"), Some("processed"), Some(2), None, false, ); let deleting = test_registry_entry( &root, "deleting.pdf", Some("doc-deleting"), Some("delete_submitted"), Some(2), Some(3), true, ); let removed = test_registry_entry( &root, "removed.pdf", None, Some("delete_completed"), None, Some(3), true, ); let failed = test_registry_entry(&root, "failed.pdf", None, Some("failed"), None, None, false); let retry = test_registry_entry( &root, "retry.pdf", Some("doc-retry"), Some("delete_retry_required"), Some(2), Some(3), true, ); let stale_with_doc = test_registry_entry( &root, "stale.pdf", Some("doc-stale"), Some("processed"), Some(2), None, true, ); assert!(!knowledge_rag_registry_entry_prunable(&active)); assert!(!knowledge_rag_registry_entry_prunable(&deleting)); assert!(!knowledge_rag_registry_entry_prunable(&retry)); assert!(!knowledge_rag_registry_entry_prunable(&stale_with_doc)); assert!(knowledge_rag_registry_entry_prunable(&removed)); assert!(knowledge_rag_registry_entry_prunable(&failed)); let _ = fs::remove_dir_all(root); } #[test] fn source_scope_filters_mapped_references_by_file_or_directory() { let mut references = vec![ json!({"sourceRootRelativePath": "books/a.pdf"}), json!({"sourceRootRelativePath": "papers/one/b.pdf"}), json!({"sourceRootRelativePath": "papers/two/c.pdf"}), ]; let scope = normalize_source_scope(Some(&[ "books/a.pdf".to_string(), "papers/one".to_string(), "papers/one".to_string(), ])); assert_eq!(scope, vec!["books/a.pdf", "papers/one"]); filter_mapped_references_by_source_scope(&mut references, &scope); let paths = references .iter() .filter_map(|reference| reference["sourceRootRelativePath"].as_str()) .collect::>(); assert_eq!(paths, vec!["books/a.pdf", "papers/one/b.pdf"]); } #[test] fn search_result_reports_paragraph_locator_precision_without_bbox() { let reference = json!({ "sourceRootRelativePath": "docs/book.docx", "sourceId": "src1", "chunkId": "chunk-1", "quote": "raw should not be used", "displayQuote": "吡咯烷,5 h,90%", "locatorEvidenceText": "如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。", "citationId": "a1b2", "normalizedFingerprint": "吡咯烷5h90", "locator": { "blockId": "block-pyrrolidine", "openAction": { "params": { "evidenceText": "吡咯烷,5 h,90%" } } }, "locatorDegraded": true }); let result = knowledge_rag_search_result(&reference, 0, "file:///tmp/root", "吡咯烷"); assert_eq!(result["locatorPrecision"], "paragraph"); assert_eq!(result["locatorDegraded"], true); assert_eq!(result["snippet"], "吡咯烷,5 h,90%"); assert_eq!(result["citationId"], "a1b2"); assert_eq!( result["locatorEvidenceText"], "如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。" ); } #[test] fn citation_text_bundle_strips_equation_and_drawing_for_display() { let bundle = citation_text_bundle( Some( r#"三乙基硅 {\left( {C}_{2}{H}_{5} \right)}_{3} raw", "locatorEvidenceText": "如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。", "searchQuery": "吡咯烷", "locatorPrecision": "paragraph", "locatorDegraded": true, "locator": {"blockId": "block-pyrrolidine"}, "citationUrl": "/documents/local-md:docs~2FHost.md", "citationMarkdown": "[book.docx](/documents/local-md:docs~2FHost.md)", "citationDiagnostics": {"displayCleaned": true} }); let citation = knowledge_rag_citation(&reference, 0).expect("citation"); assert_eq!(citation["citationId"], "c0de"); assert_eq!(citation["displayQuote"], "吡咯烷,5 h,90%"); assert_eq!( citation["locatorEvidenceText"], "如果叔丁基用 TFA 裂解, 吡咯烷将不会去除 BOC 亚乙基。" ); assert_eq!(citation["locatorPrecision"], "paragraph"); } #[test] fn heading_path_for_sidecar_block_strips_lightrag_markup() { let block = json!({ "parent_headings": ["保护", "{R}_{2}"], "heading": "二乙基胺加成物: {NEt}_{2}" }); let heading_path = heading_path_for_sidecar_block(Some(&block)); assert_eq!(heading_path[0], "保护"); assert!(!heading_path.join(" ").contains("equation")); assert!(!heading_path.join(" ").contains("latex")); assert!(heading_path.join(" ").contains("二乙基胺加成物")); } #[test] fn provider_rerank_disabled_is_reported_in_status() { let health = json!({ "configuration": { "enable_rerank": false, "rerank_binding": "null", "rerank_model": Value::Null, "min_rerank_score": 0.0 }, "rerank_queue_status": { "available": false } }); let status = lightrag_rerank_status_summary(&health); assert_eq!(status["enabled"], false); assert_eq!(status["available"], false); assert_eq!(status["providerRerankEnabled"], false); assert_eq!(status["providerRerankAvailable"], false); assert_eq!(status["status"], "disabled"); assert_eq!(status["implementation"], "provider_status_only"); } #[test] fn lightrag_search_dedupes_same_paragraph_block() { let mut references = vec![ json!({ "sourceRootRelativePath": "docs/book.docx", "chunkId": "chunk-056#match-0", "matchSource": "lightrag_search", "quote": "吡咯烷,5 h,90%", "locator": {"blockId": "same-block"} }), json!({ "sourceRootRelativePath": "docs/book.docx", "chunkId": "chunk-056#match-1", "matchSource": "lightrag_search", "quote": "吡咯烷,5 h,90%", "locator": {"blockId": "same-block"} }), json!({ "sourceRootRelativePath": "docs/book.docx", "chunkId": "chunk-057#match-0", "matchSource": "lightrag_search", "quote": "另一段吡咯烷", "locator": {"blockId": "other-block"} }), ]; dedupe_mapped_references_by_locator(&mut references); assert_eq!(references.len(), 2); assert_eq!(references[0]["locator"]["blockId"], "same-block"); assert_eq!(references[1]["locator"]["blockId"], "other-block"); } #[test] fn chunks_without_references_become_reference_candidates() { let raw = json!({ "data": { "chunks": [ {"file_path": "a.md", "reference_id": "1", "chunk_id": "chunk-a", "content": "A"}, {"file_path": "b.md", "reference_id": "2", "chunk_id": "chunk-b", "content": "B"} ] } }); let references = reference_array(&raw); assert_eq!(references.len(), 2); assert_eq!(references[0]["file_path"].as_str(), Some("a.md")); assert!(references[0]["chunks"] .as_array() .is_some_and(|items| items.len() == 1)); } #[test] fn mapped_references_include_chunk_quote() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/papers/a.pdf".into(), source_root_relative_path: "papers/a.pdf".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-a.pdf".into(), symlink_path: "/tmp/input/mnote-hash-a.pdf".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let raw = json!({ "data": { "references": [{"reference_id":"1","file_path":"mnote-hash-a.pdf"}], "chunks": [{ "reference_id":"1", "chunk_id":"doc1-chunk-000", "file_path":"mnote-hash-a.pdf", "content":"chunk quote from LightRAG" }] } }); let mapped = mapped_references( &raw, ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert_eq!(mapped[0]["chunkId"], "doc1-chunk-000"); assert_eq!(mapped[0]["quote"], "chunk quote from LightRAG"); assert_eq!(mapped[0]["quoteSource"], "chunk"); assert_eq!( mapped[0]["contentDiagnostics"]["ocrTextExposed"].as_bool(), Some(true) ); assert_eq!( mapped[0]["openAction"]["params"]["chunkId"], "doc1-chunk-000" ); } #[test] fn mapped_references_expand_lightrag_chunks_in_same_source() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/books/a.docx".into(), source_root_relative_path: "books/a.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-a.docx".into(), symlink_path: "/tmp/input/mnote-hash-a.docx".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let raw = json!({ "data": { "references": [{"reference_id":"1","file_path":"mnote-hash-a.docx"}], "chunks": [ { "reference_id":"1", "chunk_id":"doc1-chunk-001", "file_path":"mnote-hash-a.docx", "content":"第一处 三甲基硅酯 内容" }, { "reference_id":"1", "chunk_id":"doc1-chunk-002", "file_path":"mnote-hash-a.docx", "content":"第二处 三甲基硅基 内容" } ] } }); let mut mapped = mapped_references( &raw, ®istry, "file:///tmp/root", Path::new("/tmp/root"), Some("三甲基硅"), ); filter_mapped_references_by_search_query(&mut mapped, "三甲基硅"); rank_mapped_references_for_query(&mut mapped, "三甲基硅"); assert_eq!(mapped.len(), 2); assert_eq!(mapped[0]["chunkId"], "doc1-chunk-001"); assert_eq!(mapped[1]["chunkId"], "doc1-chunk-002"); assert!(mapped[0]["quote"].as_str().unwrap().contains("三甲基硅")); assert!(mapped[1]["quote"].as_str().unwrap().contains("三甲基硅")); } #[test] fn search_query_filter_drops_unrelated_semantic_reference() { let mut references = vec![ json!({ "sourceRootRelativePath": "books/protecting-groups.docx", "quote": "# 三乙基硅酯(TES): RCOOSi" }), json!({ "sourceRootRelativePath": "images/image copy 6.png", "quote": "CodePilot Bridge 处理过程截图,包含终端输出和飞书回复" }), ]; filter_mapped_references_by_search_query(&mut references, "三乙基硅酯"); assert_eq!(references.len(), 1); assert_eq!( references[0]["sourceRootRelativePath"].as_str(), Some("books/protecting-groups.docx") ); } #[test] fn search_query_filter_keeps_cjk_natural_language_keyword_match() { let mut references = vec![ json!({ "sourceRootRelativePath": "books/protecting-groups.docx", "quote": "在此过程中吗啉被用作烯丙基的清除剂。" }), json!({ "sourceRootRelativePath": "books/other.docx", "quote": "保护基资料包含用途说明,但没有目标化合物。" }), ]; filter_mapped_references_by_search_query(&mut references, "吗啉 用途"); assert_eq!(references.len(), 1); assert!(references[0]["quote"] .as_str() .unwrap_or_default() .contains("吗啉")); } #[test] fn mapped_image_placeholder_reference_marks_ocr_text_missing() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/docs/image.png".into(), source_root_relative_path: "docs/image.png".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-image.png.md".into(), symlink_path: "/tmp/input/mnote-hash-image.png.md".into(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let mapped = map_reference_plan( &json!({ "file_path": "mnote-hash-image.png.md", "chunk_id": "doc1-chunk-000", "chunks": [{ "chunk_id": "doc1-chunk-000", "content": "# image.png\n\n![image.png]()" }] }), ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert_eq!( mapped["contentDiagnostics"]["quoteOnlyImagePlaceholder"].as_bool(), Some(true) ); assert_eq!( mapped["contentDiagnostics"]["ocrTextExposed"].as_bool(), Some(false) ); assert_eq!(mapped["quoteSource"], "chunk"); } #[test] fn mapped_reference_uses_chunk_query_window_for_late_cjk_match() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-late-cjk-query"); fs::create_dir_all(root.join("docs")).expect("docs"); fs::write(root.join("docs").join("book.docx"), b"docx").expect("docx"); let input_dir = root.join("inputs"); let parsed_dir = input_dir.join("__parsed__").join("book.docx.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("book.blocks.jsonl"), [ r#"{"type":"meta","blocks":2}"#, r##"{"type":"content","blockid":"wrong-index","content":"# 5.1.8 取代苄酯, 775 三苯甲基,775 二(邻硝基苯基)甲基,779","positions":[{"type":"bbox","anchor":"1","range":[1.0,2.0,3.0,4.0]}]}"##, r##"{"type":"content","blockid":"tes","content":"# 三乙基硅酯(TES): RCOOSi {\\left( {C}_{2}{H}_{5} \\right)}_{3}","positions":[{"type":"bbox","anchor":"2","range":[10.0,20.0,30.0,40.0]}]}"##, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: root.join("docs").join("book.docx").display().to_string(), source_root_relative_path: "docs/book.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "book.docx".into(), symlink_path: input_dir.join("book.docx").display().to_string(), parser_hint: Some("native-P".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let chunk_content = format!( "{}\n# 三乙基硅酯(TES): RCOOSi {{\\left( {{C}}_{{2}}{{H}}_{{5}} \\right)}}_{{3}}\n# 保护", "前置内容。".repeat(140) ); assert!(query_centered_quote(&chunk_content, "三乙基硅", 500) .unwrap() .contains("三乙基硅酯")); let mapped = map_reference_plan( &json!({ "file_path": "book.docx", "chunk_id": "doc1-chunk-256", "chunks": [{ "chunk_id": "doc1-chunk-256", "content": chunk_content, "sidecar": { "type": "block", "refs": [ {"type": "block", "id": "wrong-index"}, {"type": "block", "id": "tes"} ] } }] }), ®istry, "file:///tmp/root", &root, Some("三乙基硅"), ); assert!(mapped["quote"].as_str().unwrap().contains("三乙基硅酯")); assert!(!mapped["quote"].as_str().unwrap().contains("取代苄酯")); assert_eq!(mapped["quoteSource"], "chunk"); assert_eq!(mapped["locator"]["blockId"], "tes"); assert_eq!(mapped["locator"]["bbox"]["x0"].as_f64(), Some(10.0)); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn mapped_reference_uses_kv_store_sidecar_refs_for_book_locator() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-kv-sidecar-locator"); fs::create_dir_all(root.join("docs")).expect("docs"); fs::write(root.join("docs").join("book.docx"), b"docx").expect("docx"); let input_dir = root.join("inputs"); let working_dir = root.join("rag_storage"); let parsed_dir = input_dir.join("__parsed__").join("book.docx.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::create_dir_all(&working_dir).expect("working dir"); fs::write( parsed_dir.join("book.blocks.jsonl"), [ r#"{"type":"meta","blocks":3}"#, r##"{"type":"content","blockid":"heading","content":"羟基-去-胺化","positions":[{"type":"paraid","anchor":10,"range":[null,null]}]}"##, r##"{"type":"content","blockid":"amide-hydrolysis","content":"无取代的酰胺能在酸或碱催化条件下水解,产物分别是游离的酸和铵根离子。","positions":[{"type":"paraid","anchor":11,"range":["A","A"]},{"type":"text_fingerprint","anchor":"fp-amide"}]}"##, r##"{"type":"content","blockid":"water-only","content":"仅仅用水难以水解绝大多数酰胺。","positions":[{"type":"paraid","anchor":12,"range":["B","B"]}]}"##, ] .join("\n"), ) .expect("blocks"); fs::write( working_dir.join("kv_store_text_chunks.json"), serde_json::to_string(&json!({ "doc-book-chunk-847": { "content": "羟基-去-胺化\n\n无取代的酰胺能在酸或碱催化条件下水解,产物分别是游离的酸和铵根离子。\n\n仅仅用水难以水解绝大多数酰胺。", "sidecar": { "type": "block", "id": "heading", "refs": [ {"type": "block", "id": "heading"}, {"type": "block", "id": "amide-hydrolysis"}, {"type": "block", "id": "water-only"} ] } } })) .expect("kv json"), ) .expect("kv store"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); std::env::set_var("MNOTE_LIGHTRAG_WORKING_DIR", &working_dir); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src-book".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: root.join("docs").join("book.docx").display().to_string(), source_root_relative_path: "docs/book.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc-book".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "book.docx".into(), symlink_path: input_dir.join("book.docx").display().to_string(), parser_hint: Some("-P!".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let mapped = map_reference_plan( &json!({ "file_path": "book.docx", "chunk_id": "doc-book-chunk-847", "chunks": [{ "chunk_id": "doc-book-chunk-847", "file_path": "book.docx", "content": "羟基-去-胺化\n\n无取代的酰胺能在酸或碱催化条件下水解,产物分别是游离的酸和铵根离子。" }] }), ®istry, "file:///tmp/root", &root, Some("酰胺 水解"), ); assert_eq!(mapped["quoteSource"], "chunk"); assert_eq!(mapped["citationDiagnostics"]["sidecarBlockMapped"], true); assert_eq!(mapped["locator"]["blockId"], "amide-hydrolysis"); assert_eq!( mapped["locator"]["openAction"]["params"]["paragraphOrdinal"].as_u64(), Some(11) ); assert_eq!( mapped["locator"]["openAction"]["params"]["textFingerprint"].as_str(), Some("fp-amide") ); assert_eq!(mapped["locatorPrecision"], "paragraph"); assert!(!mapped["citationMarkdown"] .as_str() .unwrap_or_default() .contains("来源定位降级")); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); std::env::remove_var("MNOTE_LIGHTRAG_WORKING_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn source_content_diagnostics_reports_sidecar_ocr_text() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-source-diagnostics"); let input_dir = root.join("inputs"); let parsed_dir = input_dir .join("__parsed__") .join("mnote-hash-image.png.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("mnote-hash-image.blocks.jsonl"), [ r#"{"type":"meta","blocks":1}"#, r##"{"type":"content","blockid":"block1","content":"线程作用域 OCR text","positions":[{"type":"bbox","anchor":"1","range":[1.0,2.0,3.0,4.0]}]}"##, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: root.join("image.png").display().to_string(), source_root_relative_path: "image.png".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-image.png".into(), symlink_path: input_dir.join("mnote-hash-image.png").display().to_string(), parser_hint: Some("mineru".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let diagnostics = knowledge_rag_source_content_diagnostics(®istry); assert_eq!(diagnostics[0]["ocrTextExposed"].as_bool(), Some(true)); assert_eq!(diagnostics[0]["sidecarMeaningfulBlocks"].as_u64(), Some(1)); assert_eq!(diagnostics[0]["directImageScan"].as_bool(), Some(true)); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn document_structure_index_uses_sidecar_headings_and_paragraph_ranges() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-structure-index"); let input_dir = root.join("inputs"); let parsed_dir = input_dir.join("__parsed__").join("book.docx.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("book.blocks.jsonl"), [ r#"{"type":"meta","blocks":4}"#, r##"{"type":"content","blockid":"h1","content":"# 第 1 章 酰胺","level":1,"parent_headings":[],"positions":[{"type":"paraid","anchor":0,"range":[null,null]}]}"##, r##"{"type":"content","blockid":"p1","content":"酰胺的水解可在酸性条件下进行。","level":1,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":1,"range":[null,null]}]}"##, r###"{"type":"content","blockid":"h2","content":"## 碱性水解","level":2,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":2,"range":[null,null]}]}"###, r###"{"type":"content","blockid":"p2","content":"也可用碱性水解或酶促水解。","level":2,"parent_headings":["第 1 章 酰胺","碱性水解"],"positions":[{"type":"paraid","anchor":3,"range":[null,null]}]}"###, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src-book".into(), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), source_path: root.join("book.docx").display().to_string(), source_root_relative_path: "book.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc-book".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "book.docx".into(), symlink_path: input_dir.join("book.docx").display().to_string(), parser_hint: Some("-P!".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let payload = document_structure_index_payload( &root, ®istry, &[], &[json!({"sourceId":"src-book","locator":{"blockId":"p2"}})], Some("酰胺 水解"), 8, ) .expect("structure index"); let sections = payload["documents"][0]["sections"].as_array().unwrap(); assert_eq!(sections.len(), 2); assert_eq!(sections[0]["title"], "第 1 章 酰胺"); assert_eq!(sections[0]["startParagraphOrdinal"], 0); assert_eq!(sections[0]["endParagraphOrdinal"], 1); assert_eq!(sections[1]["title"], "碱性水解"); assert_eq!(sections[1]["matchedReferenceCount"], 1); assert!(sections[1]["queryMatchCount"].as_u64().unwrap() > 0); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn section_context_reads_limited_sidecar_blocks_for_range() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-section-context"); let input_dir = root.join("inputs"); let parsed_dir = input_dir.join("__parsed__").join("book.docx.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("book.blocks.jsonl"), [ r#"{"type":"meta","blocks":5}"#, r##"{"type":"content","blockid":"h1","content":"# 第 1 章 酰胺","level":1,"parent_headings":[],"positions":[{"type":"paraid","anchor":0,"range":[null,null]}]}"##, r##"{"type":"content","blockid":"p1","content":"酰胺的水解可在酸性条件下进行。","level":1,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":1,"range":["p1","p1"]}]}"##, r###"{"type":"content","blockid":"h2","content":"## 碱性水解","level":2,"parent_headings":["第 1 章 酰胺"],"positions":[{"type":"paraid","anchor":2,"range":[null,null]}]}"###, r###"{"type":"content","blockid":"p2","content":"也可用碱性水解或酶促水解。","level":2,"parent_headings":["第 1 章 酰胺","碱性水解"],"positions":[{"type":"paraid","anchor":3,"range":["p2","p2"]}]}"###, r###"{"type":"content","blockid":"p3","content":"这个段落不应在 maxBlocks=2 时返回。","level":2,"parent_headings":["第 1 章 酰胺","碱性水解"],"positions":[{"type":"paraid","anchor":4,"range":["p3","p3"]}]}"###, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src-book".into(), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), source_path: root.join("book.docx").display().to_string(), source_root_relative_path: "book.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc-book".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "book.docx".into(), symlink_path: input_dir.join("book.docx").display().to_string(), parser_hint: Some("-P!".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let payload = sidecar_section_context_payload( &root, ®istry, &KnowledgeRagSectionContextRequest { workspace_id: Some("ws".into()), root_uri: format!("file://{}", root.display()), source_path: Some("book.docx".into()), source_id: None, light_rag_doc_id: None, file_path: None, section_id: Some("src-book#block-h2".into()), start_block_ordinal: Some(2), end_block_ordinal: Some(4), start_paragraph_ordinal: None, end_paragraph_ordinal: None, context_before: Some(1), context_after: Some(0), max_blocks: Some(2), max_chars: Some(2_000), }, ) .expect("section context"); assert_eq!(payload["blocks"].as_array().unwrap().len(), 2); assert_eq!(payload["blocks"][0]["blockOrdinal"], 1); assert_eq!(payload["blocks"][1]["blockOrdinal"], 2); assert_eq!(payload["chunks"].as_array().unwrap().len(), 1); assert_eq!(payload["chunks"][0]["startBlockOrdinal"], 1); assert_eq!(payload["limits"]["truncated"], true); assert!(payload["text"] .as_str() .unwrap() .contains("酰胺的水解可在酸性条件下进行")); assert!(!payload["text"].as_str().unwrap().contains("这个段落不应")); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn mapped_references_filters_unmapped_provider_references() { let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![], }; let raw = json!({ "data": { "references": [{"reference_id":"1","file_path":"orphan.pdf"}], "chunks": [{ "reference_id":"1", "chunk_id":"orphan-chunk", "file_path":"orphan.pdf", "content":"orphan provider chunk" }] } }); let mapped = mapped_references( &raw, ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert!( mapped.is_empty(), "unmapped provider references must not become MNote citations" ); let plan = map_reference_plan( &json!({"file_path":"orphan.pdf","chunk_id":"orphan-chunk"}), ®istry, "file:///tmp/root", Path::new("/tmp/root"), None, ); assert_eq!(plan["unmapped"], true); assert_eq!(plan["locatorDegraded"], true); } #[test] fn query_ranking_prefers_exact_source_and_quote_match() { let mut references = vec![ json!({ "sourceRootRelativePath": "knowledge-rag-fixtures-7-50/paper-brotli-comparison.pdf", "filePath": "mnote-paper-brotli-comparison.pdf", "quote": "brotli compression table", "citationMarkdown": "[paper-brotli-comparison.pdf · p.3](/)" }), json!({ "sourceRootRelativePath": "knowledge-rag-fixtures-7-50/scan-image-start.pdf", "filePath": "mnote-scan-image-start.pdf", "quote": "# START: This is the first image in PDF", "citationMarkdown": "[scan-image-start.pdf · p.1](/)" }), ]; rank_mapped_references_for_query(&mut references, "scan image start"); assert_eq!( references[0]["sourceRootRelativePath"], "knowledge-rag-fixtures-7-50/scan-image-start.pdf" ); } #[test] fn query_mode_preserves_lightrag_native_modes() { assert_eq!(normalize_lightrag_query_mode(Some("hybrid")), "hybrid"); assert_eq!(normalize_lightrag_query_mode(Some("global")), "global"); assert_eq!(normalize_lightrag_query_mode(Some("exact")), "naive"); assert_eq!(normalize_lightrag_query_mode(None), "mix"); assert_eq!(normalize_knowledge_rag_search_mode(Some("exact")), "exact"); assert_eq!(normalize_knowledge_rag_search_mode(Some("vector")), "naive"); assert_eq!( normalize_knowledge_rag_search_mode(Some("hybrid")), "hybrid" ); assert_eq!(normalize_knowledge_rag_search_mode(None), "exact"); } #[test] fn skip_kg_scoped_query_forces_naive_mode_for_book_sources() { let root = temp_root("mnote-knowledge-rag-skip-kg-query-mode"); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: format!("file://{}", root.display()), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![ test_registry_entry( &root, "books/organic.docx", Some("doc-book"), Some("processed"), Some(2), None, false, ), test_registry_entry( &root, "papers/small.pdf", Some("doc-paper"), Some("processed"), Some(2), None, false, ), ], }; let mut registry = registry; registry.entries[0].parser_hint = Some("-P!".into()); registry.entries[0].light_rag_file_path = "mnote-book.[-P!].docx".into(); registry.entries[1].parser_hint = None; registry.entries[1].light_rag_file_path = "mnote-paper.pdf".into(); let scoped_book = normalize_source_scope(Some(&["books/organic.docx".to_string()])); let decision = resolve_lightrag_query_mode_for_scope(®istry, &scoped_book, "mix"); assert_eq!(decision.mode, "naive"); assert_eq!(decision.reason, "source_scope_skip_kg_document"); let search_decision = resolve_knowledge_rag_search_mode_for_scope(®istry, &scoped_book, "hybrid"); assert_eq!(search_decision.mode, "naive"); assert_eq!(search_decision.reason, "source_scope_skip_kg_document"); let exact_search_decision = resolve_knowledge_rag_search_mode_for_scope(®istry, &scoped_book, "exact"); assert_eq!(exact_search_decision.mode, "exact"); assert_eq!(exact_search_decision.reason, "requested_mode"); let naive_decision = resolve_lightrag_query_mode_for_scope(®istry, &scoped_book, "naive"); assert_eq!(naive_decision.mode, "naive"); assert_eq!(naive_decision.reason, "requested_mode"); let scoped_paper = normalize_source_scope(Some(&["papers/small.pdf".to_string()])); let paper_decision = resolve_lightrag_query_mode_for_scope(®istry, &scoped_paper, "mix"); assert_eq!(paper_decision.mode, "mix"); assert_eq!(paper_decision.reason, "requested_mode"); let _ = fs::remove_dir_all(root); } #[test] fn symlink_name_preserves_parser_hint_before_extension() { let name = lightrag_symlink_name(Path::new("/tmp/books/demo.pdf"), "demo.pdf", Some("mineru")); assert!(name.ends_with("-demo.[mineru].pdf")); } #[test] fn symlink_name_preserves_parser_hint_with_options_before_extension() { let name = lightrag_symlink_name( Path::new("/tmp/books/demo.docx"), "demo.docx", Some("native-P"), ); assert!(name.ends_with("-demo.[native-P].docx")); assert!(strip_one_supported_parser_hint(&name) .as_deref() .unwrap_or_default() .ends_with("-demo.docx")); } #[test] fn ocr_layered_docx_defaults_to_native_paragraph_strategy() { let file_name = "[OCR]_有机合成中的保护基_20250201.layered_删减-2025-02-04 18-59-42.docx"; assert_eq!( default_lightrag_parser_hint_for_source(Path::new(file_name), file_name).as_deref(), Some("native-P") ); assert_eq!( default_lightrag_parser_hint_for_source(Path::new("ordinary.docx"), "ordinary.docx"), None ); } #[test] fn large_docx_and_pdf_default_to_paragraph_skip_kg_strategy() { let root = temp_root("mnote-knowledge-rag-large-skip-kg"); let docx = root.join("organic-book.docx"); let pdf = root.join("organic-book.pdf"); fs::File::create(&docx) .expect("docx") .set_len(LARGE_DOCUMENT_SKIP_KG_MIN_BYTES) .expect("large docx"); fs::File::create(&pdf) .expect("pdf") .set_len(LARGE_DOCUMENT_SKIP_KG_MIN_BYTES) .expect("large pdf"); assert_eq!( default_lightrag_parser_hint_for_source(&docx, "organic-book.docx").as_deref(), Some("-P!") ); assert_eq!( default_lightrag_parser_hint_for_source(&pdf, "organic-book.pdf").as_deref(), Some("-P!") ); assert!(valid_lightrag_parser_options("P!")); let layered = root.join("[OCR]_book.layered.docx"); fs::File::create(&layered) .expect("layered") .set_len(LARGE_DOCUMENT_SKIP_KG_MIN_BYTES) .expect("large layered"); assert_eq!( default_lightrag_parser_hint_for_source(&layered, "[OCR]_book.layered.docx").as_deref(), Some("native-P!") ); let _ = fs::remove_dir_all(root); } #[test] fn symlink_name_does_not_duplicate_existing_parser_hint() { let name = lightrag_symlink_name( Path::new("/tmp/books/demo.[native].pdf"), "demo.[native].pdf", Some("native"), ); assert!(name.ends_with("-demo.[native].pdf")); assert!(!name.contains(".[native].[native].")); } #[test] fn symlink_name_does_not_duplicate_existing_parser_hint_with_options() { let name = lightrag_symlink_name( Path::new("/tmp/books/demo.[native-P].docx"), "demo.[native-P].docx", Some("native-P"), ); assert!(name.ends_with("-demo.[native-P].docx")); assert!(!name.contains(".[native-P].[native-P].")); } #[test] fn registry_file_path_matches_lightrag_hint_canonicalization() { let entry = KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/scan.[mineru].pdf".into(), source_root_relative_path: "scan.[mineru].pdf".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: None, light_rag_status: None, light_rag_file_path: "mnote-hash-scan.[mineru].[mineru].pdf".into(), symlink_path: "/tmp/input/mnote-hash-scan.[mineru].[mineru].pdf".into(), parser_hint: Some("mineru".into()), indexed_at_ms: None, deleted_at_ms: None, stale: false, updated_at_ms: 2, }; assert!(lightrag_file_path_matches( &entry, "mnote-hash-scan.[mineru].pdf" )); } #[test] fn locator_uses_stripped_parser_hint_sidecar_path() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-sidecar"); let input_dir = root.join("inputs"); let parsed_dir = input_dir .join("__parsed__") .join("mnote-hash-scan.pdf.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("mnote-hash-scan.blocks.jsonl"), [ r#"{"type":"meta","blocks":1}"#, r##"{"type":"content","blockid":"block1","content":"# START: This is the first image in PDF\nThis is text BEFORE the image.","positions":[{"type":"bbox","anchor":"1","range":[171.0,126.0,648.0,152.0]}]}"##, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let entry = KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/scan.pdf".into(), source_root_relative_path: "scan.pdf".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-scan.[mineru].pdf".into(), symlink_path: input_dir .join("mnote-hash-scan.[mineru].pdf") .display() .to_string(), parser_hint: Some("mineru".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }; let locator = lightrag_locator_for_reference( Path::new("/tmp/root"), "file:///tmp/root", &entry, &json!("doc1-chunk-000"), None, None, &json!(0), Some(&json!({"type":"block","refs":[{"type":"block","id":"block1"}]})), ) .expect("locator"); assert_eq!(locator.page, Some(1)); assert_eq!(locator.bbox.as_ref().map(|bbox| bbox.x0), Some(171.0)); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn locator_exports_docx_paraid_positions_for_office_preview() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-docx-paraid"); let input_dir = root.join("inputs"); let parsed_dir = input_dir .join("__parsed__") .join("mnote-hash-report.docx.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("mnote-hash-report.blocks.jsonl"), [ r#"{"type":"meta","blocks":1}"#, r##"{"type":"content","blockid":"block-docx-1","content":"1.1 监管政策变革\n化妆品新原料监管政策变革内容。","positions":[{"type":"paraid","anchor":4,"range":["6692C49B","6692C49B"]},{"type":"text_fingerprint","anchor":"1454ec0b0ffd9fc0"}]}"##, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let entry = KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: "/tmp/root/report.docx".into(), source_root_relative_path: "report.docx".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-report.docx".into(), symlink_path: input_dir .join("mnote-hash-report.docx") .display() .to_string(), parser_hint: Some("native".into()), indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }; let locator = lightrag_locator_for_reference( Path::new("/tmp/root"), "file:///tmp/root", &entry, &json!("doc1-chunk-000"), Some("监管政策变革"), None, &json!(0), Some(&json!({"type":"block","refs":[{"type":"block","id":"block-docx-1"}]})), ) .expect("locator"); assert_eq!(locator.resource_kind, EvidenceResourceKind::Office); assert_eq!(locator.page, None); assert_eq!(locator.bbox, None); assert_eq!(locator.block_id.as_deref(), Some("block-docx-1")); assert_eq!( locator.open_action.params["paragraphOrdinal"].as_u64(), Some(4) ); assert_eq!( locator.open_action.params["paraIdStart"].as_str(), Some("6692C49B") ); assert_eq!( locator.open_action.params["textFingerprint"].as_str(), Some("1454ec0b0ffd9fc0") ); let citation_url = crate::routes::evidence::citation_url_for_locator(&locator); assert!(citation_url.contains("paragraphOrdinal=4")); assert!(citation_url.contains("paraIdStart=6692C49B")); assert!(citation_url.contains("textFingerprint=1454ec0b0ffd9fc0")); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn locator_degrades_when_sidecar_quote_does_not_match() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-sidecar-no-match"); fs::create_dir_all(root.join("docs")).expect("docs"); fs::write(root.join("docs").join("Host.md"), "# Host\n").expect("host"); fs::write(root.join("docs").join("scan.pdf"), b"pdf").expect("pdf"); let input_dir = root.join("inputs"); let parsed_dir = input_dir .join("__parsed__") .join("mnote-hash-scan.pdf.parsed"); fs::create_dir_all(&parsed_dir).expect("parsed dir"); fs::write( parsed_dir.join("mnote-hash-scan.blocks.jsonl"), [ r#"{"type":"meta","blocks":1}"#, r##"{"type":"content","blockid":"block1","content":"This block is not the returned quote.","positions":[{"type":"bbox","anchor":"9","range":[1.0,2.0,3.0,4.0]}]}"##, ] .join("\n"), ) .expect("blocks"); std::env::set_var("MNOTE_LIGHTRAG_INPUT_DIR", &input_dir); let registry = KnowledgeRagSourceRegistry { schema: REGISTRY_SCHEMA.to_string(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), updated_at_ms: 1, indexed_roots: Vec::new(), entries: vec![KnowledgeRagSourceRegistryEntry { source_id: "src1".into(), workspace_id: "ws".into(), root_uri: "file:///tmp/root".into(), source_path: root.join("docs").join("scan.pdf").display().to_string(), source_root_relative_path: "docs/scan.pdf".into(), source_hash: "mnote-fnv64:1".into(), light_rag_doc_id: Some("doc1".into()), light_rag_status: Some("processed".into()), light_rag_file_path: "mnote-hash-scan.pdf".into(), symlink_path: input_dir.join("mnote-hash-scan.pdf").display().to_string(), parser_hint: None, indexed_at_ms: Some(2), deleted_at_ms: None, stale: false, updated_at_ms: 2, }], }; let mapped = map_reference_plan( &json!({ "file_path": "mnote-hash-scan.pdf", "chunk_id": "doc1-chunk-000", "chunks": [{"chunk_id": "doc1-chunk-000", "content": "A different quote should not get page or bbox."}] }), ®istry, "file:///tmp/root", &root, None, ); assert!(mapped["locator"].is_null()); assert_eq!(mapped["locatorDegraded"], true); assert!(mapped["citationUrl"] .as_str() .is_some_and(|url| url.contains("resourceTab="))); assert!(mapped["citationMarkdown"] .as_str() .unwrap() .contains("来源定位降级")); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn lightrag_paths_prefer_source_env_over_legacy_process_env() { let _guard = env_lock().lock().expect("env lock"); let root = temp_root("mnote-knowledge-rag-env"); let source_input = root.join("source-inputs"); let source_working = root.join("source-storage"); let legacy_input = root.join("legacy-inputs"); let legacy_working = root.join("legacy-storage"); let env_file = root.join("lightrag.env"); fs::write( &env_file, format!( "INPUT_DIR={}\nWORKING_DIR={}\n", source_input.display(), source_working.display() ), ) .expect("env file"); std::env::set_var("MNOTE_LIGHTRAG_ENV_FILE", &env_file); std::env::set_var("INPUT_DIR", &legacy_input); std::env::set_var("WORKING_DIR", &legacy_working); std::env::remove_var("MNOTE_LIGHTRAG_INPUT_DIR"); std::env::remove_var("MNOTE_LIGHTRAG_WORKING_DIR"); assert_eq!(lightrag_input_dir(), source_input); assert_eq!(lightrag_working_dir(), source_working); std::env::remove_var("MNOTE_LIGHTRAG_ENV_FILE"); std::env::remove_var("INPUT_DIR"); std::env::remove_var("WORKING_DIR"); let _ = fs::remove_dir_all(root); } #[test] fn directory_source_expands_supported_files_and_skips_internal_dirs() { let root = temp_root("mnote-knowledge-rag-directory"); fs::create_dir_all(root.join("docs").join("nested")).expect("nested"); fs::create_dir_all(root.join(".mnote").join("index")).expect("mnote"); fs::write(root.join("docs").join("a.pdf"), b"%PDF").expect("pdf"); fs::write(root.join("docs").join("nested").join("b.md"), "# B").expect("md"); fs::write(root.join("docs").join("ignored.tmp"), "tmp").expect("tmp"); fs::write( root.join(".mnote").join("index").join("hidden.pdf"), b"%PDF", ) .expect("hidden"); let sources = resolve_knowledge_rag_sources(&root, "docs", &test_context()).expect("sources"); let rel = sources .iter() .map(|source| root_relative_path(&root, &source.canonical_path).expect("relative")) .collect::>(); assert_eq!(rel, vec!["docs/a.pdf", "docs/nested/b.md"]); assert!(sources .iter() .all(|source| source.source_kind == "directory")); let _ = fs::remove_dir_all(root); } #[test] fn file_source_rejects_unsupported_extension() { let root = temp_root("mnote-knowledge-rag-unsupported"); fs::write(root.join("notes.tmp"), "tmp").expect("tmp"); let error = resolve_knowledge_rag_sources(&root, "notes.tmp", &test_context()) .expect_err("unsupported"); assert_eq!(error.code(), "knowledge_rag_source_unsupported"); let _ = fs::remove_dir_all(root); } }