From 0eb15c5b6f29a77e838d1ea7820a79fe09b5b0fb Mon Sep 17 00:00:00 2001 From: MightZero <1904769228@qq.com> Date: Mon, 29 Jun 2026 18:31:25 +0800 Subject: [PATCH] =?UTF-8?q?[refactor]=20=E5=BD=BB=E5=BA=95=E7=A7=BB?= =?UTF-8?q?=E9=99=A4clean=E5=8F=82=E6=95=B0=EF=BC=8C=E7=AE=80=E5=8C=96=20i?= =?UTF-8?q?ngest=20=E5=92=8C=20update=20=E5=87=BD=E6=95=B0=E6=8E=A5?= =?UTF-8?q?=E5=8F=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scripts/ingest.py | 27 +++++++++------------------ scripts/update.py | 16 +++------------- src/ingestion/loader.py | 10 ++++------ src/mcp/server.py | 8 ++------ src/mcp/tools.py | 9 +++------ 5 files changed, 21 insertions(+), 49 deletions(-) diff --git a/scripts/ingest.py b/scripts/ingest.py index 7d65068..d3bb0df 100644 --- a/scripts/ingest.py +++ b/scripts/ingest.py @@ -103,7 +103,7 @@ def _save_to_stores(file_name: str, filepath: str, content: str, ) -def ingest_local(file_path: Path, clean: bool = False, tag: bool = True, +def ingest_local(file_path: Path, tag: bool = True, verbose: bool = False, type: str | None = None, author: str | None = None) -> IngestResult: """Import a single local .md file into three-layer storage. @@ -113,7 +113,6 @@ def ingest_local(file_path: Path, clean: bool = False, tag: bool = True, Args: file_path: Path to the local .md file. - clean: Run LLM-based text cleaning. tag: Extract tags via LLM. verbose: Print processing details to stdout. type: Note type (note/solution/template). @@ -122,17 +121,14 @@ def ingest_local(file_path: Path, clean: bool = False, tag: bool = True, Returns: An ``IngestResult`` with metadata of the imported note. """ - if verbose and clean: - print(" 清洗: ") - print(" > ", end="") saved_path, content = load_local( - file_path, _file_store, clean, stream=verbose) + file_path, _file_store, stream=verbose) return _save_to_stores(saved_path.name, str(saved_path.resolve()), content, source_url=None, type=type, author=author, tag=tag, verbose=verbose) -def ingest_locals(dir_path: Path, clean: bool = False, tag: bool = True, +def ingest_locals(dir_path: Path, tag: bool = True, verbose: bool = False, type: str | None = None, author: str | None = None) -> List[IngestResult]: """Import all .md files from a directory. @@ -142,7 +138,6 @@ def ingest_locals(dir_path: Path, clean: bool = False, tag: bool = True, Args: dir_path: Directory containing .md files. - clean: Run LLM-based text cleaning on each file. tag: Extract tags via LLM. verbose: Print per-file details. type: Note type (note/solution/template). @@ -157,7 +152,7 @@ def ingest_locals(dir_path: Path, clean: bool = False, tag: bool = True, results = [] for idx, file_path in enumerate(note_files, 1): print(f"[{idx}/{files_number}] {file_path.name}") - results.append(ingest_local(file_path, clean, tag, verbose, + results.append(ingest_local(file_path, tag, verbose, type=type, author=author)) print(f"导入完成: {files_number} 个文件", end="") elapsed = time.time() - start_time @@ -166,7 +161,7 @@ def ingest_locals(dir_path: Path, clean: bool = False, tag: bool = True, return results -def ingest_web(url: str, clean: bool = True, tag: bool = True, +def ingest_web(url: str, tag: bool = True, verbose: bool = False, type: str | None = None, author: str | None = None) -> IngestResult: """Import a web page into three-layer storage. @@ -176,7 +171,6 @@ def ingest_web(url: str, clean: bool = True, tag: bool = True, Args: url: The web page URL to import. - clean: Run LLM-based text cleaning on the fetched content. tag: Extract tags via LLM. verbose: Print processing details to stdout. type: Note type (note/solution/template). @@ -186,10 +180,10 @@ def ingest_web(url: str, clean: bool = True, tag: bool = True, An ``IngestResult`` with metadata of the imported note. """ start_time = time.time() - if verbose and clean: + if verbose: print(" 清洗: ") print(" > ", end="") - saved_path, content = load_web(url, _file_store, clean, stream=verbose) + saved_path, content = load_web(url, _file_store, stream=verbose) result = _save_to_stores(saved_path.name, str(saved_path.resolve()), content, source_url=url, type=type, author=author, tag=tag, verbose=verbose) @@ -211,9 +205,6 @@ if __name__ == "__main__": help="网页 URL 地址") # Must one and no more - source - - parser.add_argument("--clean", action="store_true", - help="执行 LLM 清洗") parser.add_argument("--no-tag", action="store_true", help="跳过打 tag") parser.add_argument("-v", "--verbose", action="store_true", @@ -226,10 +217,10 @@ if __name__ == "__main__": if args.path: if args.path.is_file(): - result = ingest_local(args.path, clean=args.clean, + result = ingest_local(args.path, tag=tag, verbose=args.verbose) elif args.path.is_dir(): - results = ingest_locals(args.path, clean=args.clean, + results = ingest_locals(args.path, tag=tag, verbose=args.verbose) else: print(f"❌ 路径不存在: {args.path}") diff --git a/scripts/update.py b/scripts/update.py index 5b8b9e9..b6dd2b0 100644 --- a/scripts/update.py +++ b/scripts/update.py @@ -84,7 +84,7 @@ def update_metadata(filename: str, title: str | None = None, def update_note(file_name: str, file_path: Path | None = None, - clean: bool = False, tag: bool = True, + tag: bool = True, verbose: bool = False, type: str | None = None, author: str | None = None) -> UpdateResult: """Update a note across all three storage layers. @@ -95,7 +95,6 @@ def update_note(file_name: str, file_path: Path | None = None, Args: file_name: The note filename to update (e.g. "fenwick.md"). file_path: Path to new local file (optional). - clean: Run LLM-based text cleaning. tag: Extract tags via LLM. verbose: Print processing details. type: Note type (note/solution/template). @@ -112,23 +111,16 @@ def update_note(file_name: str, file_path: Path | None = None, source_url = sql_record.get("source_url") if file_path: - if verbose and clean: - print(" 清洗: ") - print(" > ", end="") content = Path(file_path).read_text(encoding="utf-8") - if clean: - content = clean_text(content, verbose) - _file_store.update(file_name, content) elif source_url: - if verbose and clean: + if verbose: print(" 清洗: ") print(" > ", end="") from langchain_community.document_loaders import WebBaseLoader from src.ingestion.loader import _HEADERS docs = WebBaseLoader(source_url, header_template=_HEADERS).load() content = docs[0].page_content - if clean: - content = clean_text(content, verbose) + content = clean_text(content, verbose) _file_store.update(file_name, content) else: logger.error(f"Cannot update {file_name}: no --file and no source_url") @@ -178,7 +170,6 @@ if __name__ == "__main__": parser.add_argument("filename", help="笔记文件名") parser.add_argument("--file", type=Path, help="新的本地文件路径") - parser.add_argument("--clean", action="store_true", help="执行 LLM 清洗") parser.add_argument("--no-tag", action="store_true", help="跳过打 tag") parser.add_argument("-v", "--verbose", action="store_true", help="详细输出") parser.add_argument("--json", type=Path, metavar="FILE", @@ -199,7 +190,6 @@ if __name__ == "__main__": result = update_note( args.filename, file_path=args.file, - clean=args.clean, tag=tag, verbose=args.verbose, ) diff --git a/src/ingestion/loader.py b/src/ingestion/loader.py index 1c919d2..a2e41cf 100644 --- a/src/ingestion/loader.py +++ b/src/ingestion/loader.py @@ -52,13 +52,12 @@ def load_text(content: str, file_name: str, file_store: FileStore, clean: bool = return Path(saved_path), content -def load_local(filepath: str | Path, file_store: FileStore, clean: bool = False, stream: bool = False) -> tuple[Path, str]: +def load_local(filepath: str | Path, file_store: FileStore, stream: bool = False) -> tuple[Path, str]: """Load a local .md file and save to FileStore. Args: filepath: Path to the local markdown file. file_store: FileStore instance for saving. - clean: Run LLM-based text cleaning before saving. stream: Stream cleaning output to stdout. Returns: @@ -71,13 +70,13 @@ def load_local(filepath: str | Path, file_store: FileStore, clean: bool = False, filename = path.name content = path.read_text(encoding="utf-8") saved_path, content = load_text( - content, filename, file_store, clean, stream) + content, filename, file_store, False, stream) logger.info(f"Loaded local file: {filename} -> {saved_path}") return saved_path, content -def load_web(url: str, file_store: FileStore, clean: bool = True, stream: bool = False) -> tuple[Path, str]: +def load_web(url: str, file_store: FileStore, stream: bool = False) -> tuple[Path, str]: """Load a web page, optionally clean it, and save to FileStore. Uses a Chrome 120 browser header template to avoid blocking. @@ -85,7 +84,6 @@ def load_web(url: str, file_store: FileStore, clean: bool = True, stream: bool = Args: url: The URL of the web page. file_store: FileStore instance for saving. - clean: If True, clean the raw text with LLM before saving. stream: Stream cleaning output to stdout. Returns: @@ -104,7 +102,7 @@ def load_web(url: str, file_store: FileStore, clean: bool = True, stream: bool = filename = _url_to_filename(url) - saved_path, content = load_text(raw_text, filename, file_store, clean, stream) + saved_path, content = load_text(raw_text, filename, file_store, True, stream) logger.info(f"Saved web content: {filename} -> {saved_path}") return saved_path, content diff --git a/src/mcp/server.py b/src/mcp/server.py index 88f37a8..c33211d 100644 --- a/src/mcp/server.py +++ b/src/mcp/server.py @@ -41,7 +41,6 @@ def get_mcp(host: str = "127.0.0.1", port: int = 8000) -> FastMCP: def ingest( path: str | None = None, url: str | None = None, - clean: bool = False, tag: bool = True, type: str = "note", author: str | None = None, @@ -54,12 +53,11 @@ def ingest( Args: path: Local file or directory path to import. url: Web page URL to import. - clean: Run LLM-based text cleaning before storing. tag: Extract tags via LLM (default: true). type: Note type (note/solution/template, default: note). author: Note author. """ - return ingest_tool(path=path, url=url, clean=clean, tag=tag, + return ingest_tool(path=path, url=url, tag=tag, type=type, author=author) @@ -67,7 +65,6 @@ def ingest( def update( filename: str, file_path: str | None = None, - clean: bool = False, tag: bool = True, ) -> dict: """Update an existing note across all three storage layers. @@ -78,10 +75,9 @@ def update( Args: filename: Note filename to update (e.g. "fenwick.md"). file_path: Path to new local file (optional). - clean: Run LLM-based text cleaning. tag: Extract tags via LLM. """ - return update_tool(filename=filename, file_path=file_path, clean=clean, tag=tag) + return update_tool(filename=filename, file_path=file_path, tag=tag) @get_mcp().tool() diff --git a/src/mcp/tools.py b/src/mcp/tools.py index 14e0b90..db81736 100644 --- a/src/mcp/tools.py +++ b/src/mcp/tools.py @@ -13,7 +13,6 @@ from src.store.vector_store import get_vector_store def ingest_tool( path: str | None = None, url: str | None = None, - clean: bool = False, tag: bool = True, type: str = "note", author: str | None = None, @@ -27,7 +26,7 @@ def ingest_tool( try: if url: - result = ingest_web(url, clean=clean, tag=tag, verbose=False, + result = ingest_web(url, tag=tag, verbose=False, type=type, author=author) return { "file_name": result.file_name, @@ -39,7 +38,7 @@ def ingest_tool( file_path = Path(path) if file_path.is_file(): - result = ingest_local(file_path, clean=clean, tag=tag, verbose=False, + result = ingest_local(file_path, tag=tag, verbose=False, type=type, author=author) return { "file_name": result.file_name, @@ -49,7 +48,7 @@ def ingest_tool( "note_id": result.note_id, } elif file_path.is_dir(): - results = ingest_locals(file_path, clean=clean, tag=tag, verbose=False, + results = ingest_locals(file_path, tag=tag, verbose=False, type=type, author=author) return [ { @@ -70,7 +69,6 @@ def ingest_tool( def update_tool( filename: str, file_path: str | None = None, - clean: bool = False, tag: bool = True, ) -> dict[str, Any]: sql_store = get_sql_store() @@ -90,7 +88,6 @@ def update_tool( result = update_note( filename, file_path=Path(file_path) if file_path else None, - clean=clean, tag=tag, verbose=False, ) -- 2.34.1