"""Read-mostly global assistant knowledge API and admin import actions.""" from __future__ import annotations import asyncio import base64 import json import logging import re from collections.abc import AsyncIterable, AsyncIterator from datetime import UTC, datetime from pathlib import Path from typing import Any from uuid import uuid4 from fastapi import APIRouter, File, Form, HTTPException, Query, Request, UploadFile from fastapi.responses import FileResponse from pydantic import BaseModel, Field from app.gateway.deps import get_optional_user_from_request from deerflow.assistant_knowledge.archive import PackageArchive from deerflow.assistant_knowledge.package_io import PACKAGE_VERSION, package_path, write_jsonl_record from deerflow.config.runtime_paths import runtime_home from deerflow.integrations.weknora.client import WeKnoraError from deerflow.integrations.weknora.runtime import build_weknora_client, get_resolved_llmwiki_runtime router = APIRouter(prefix="/api/assistant-knowledge", tags=["assistant-knowledge"]) export_router = APIRouter(tags=["assistant-knowledge"]) logger = logging.getLogger(__name__) def _package_root() -> Path: return runtime_home() / "assistant-knowledge" / "packages" def _job_package_url(job_id: str) -> str: return f"/api/assistant-knowledge/import-jobs/{job_id}/package" def _weknora_export_package_url(mapping_id: str, job_id: str) -> str: return f"/api/llmwiki/knowledge-bases/{mapping_id}/export-package/{job_id}/download" def _weknora_export_status_url(mapping_id: str, job_id: str) -> str: return f"/api/llmwiki/knowledge-bases/{mapping_id}/export-package/{job_id}" def _export_manifest_path(job_id: str) -> Path: return package_path(_package_root(), job_id).with_suffix(".manifest.json") def _now_iso() -> str: return datetime.now(UTC).isoformat() def _normalize_export_manifest(payload: dict[str, Any]) -> dict[str, Any]: job_id = str(payload.get("id") or "") mapping_id = str(payload.get("mapping_id") or "") path = package_path(_package_root(), job_id) if job_id else None file_exists = bool(path and path.is_file()) status = str(payload.get("status") or "") normalized = { **payload, "id": job_id, "mapping_id": mapping_id, "status": status, "phase": str(payload.get("phase") or status or "unknown"), "counts": payload.get("counts") if isinstance(payload.get("counts"), dict) else {}, "package_download_url": ( _weknora_export_package_url(mapping_id, job_id) if mapping_id and job_id else payload.get("package_download_url") ), "status_url": ( _weknora_export_status_url(mapping_id, job_id) if mapping_id and job_id else payload.get("status_url") ), "file_exists": file_exists, "file_size_bytes": int(path.stat().st_size) if file_exists and path else int(payload.get("file_size_bytes") or 0), } return normalized def _write_export_manifest( *, job_id: str, mapping_id: str, status: str, phase: str, counts: dict[str, Any] | None = None, error: str | None = None, ) -> dict[str, Any]: path = _export_manifest_path(job_id) existing: dict[str, Any] = {} if path.is_file(): try: value = json.loads(path.read_text(encoding="utf-8")) existing = value if isinstance(value, dict) else {} except (OSError, ValueError, TypeError): existing = {} now = _now_iso() payload = { **existing, "id": job_id, "mapping_id": mapping_id, "status": status, "phase": phase, "counts": counts if counts is not None else existing.get("counts") or {}, "package_download_url": _weknora_export_package_url(mapping_id, job_id), "status_url": _weknora_export_status_url(mapping_id, job_id), "error": error, "created_at": existing.get("created_at") or now, "updated_at": now, "completed_at": now if status in {"completed", "failed"} else existing.get("completed_at"), } path.parent.mkdir(parents=True, exist_ok=True) payload = _normalize_export_manifest(payload) temp = path.with_suffix(".tmp") temp.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") temp.replace(path) return payload def _read_export_manifest(job_id: str) -> dict[str, Any] | None: path = _export_manifest_path(job_id) if not path.is_file(): return None try: value = json.loads(path.read_text(encoding="utf-8")) except (OSError, ValueError, TypeError): return None return _normalize_export_manifest(value) if isinstance(value, dict) else None def _list_export_manifests(mapping_id: str, *, limit: int) -> list[dict[str, Any]]: root = _package_root() if not root.is_dir(): return [] rows: list[dict[str, Any]] = [] for path in root.glob("*.manifest.json"): try: value = json.loads(path.read_text(encoding="utf-8")) except (OSError, ValueError, TypeError): continue if not isinstance(value, dict) or str(value.get("mapping_id") or "") != mapping_id: continue rows.append(_normalize_export_manifest(value)) rows.sort(key=lambda row: str(row.get("created_at") or row.get("updated_at") or ""), reverse=True) return rows[:limit] def _delete_base_package_files(base_id: str, import_job_ids: list[str]) -> None: """Delete package artifacts owned by one assistant base only.""" import shutil from app.gateway.assistant_file_ingest import file_job_dir for job_id in import_job_ids: # UUID validation confines cleanup to exactly one job directory. try: directory = file_job_dir(job_id) except ValueError: continue if directory.is_dir() and not directory.is_symlink(): shutil.rmtree(directory) root = _package_root() if not root.is_dir(): return job_ids = {str(value) for value in import_job_ids if value} scope = f"assistant:{base_id}" for manifest_path in root.glob("*.manifest.json"): try: value = json.loads(manifest_path.read_text(encoding="utf-8")) except (OSError, ValueError, TypeError): continue if not isinstance(value, dict) or str(value.get("mapping_id") or "") != scope: continue job_id = str(value.get("id") or "") if job_id: job_ids.add(job_id) manifest_path.unlink(missing_ok=True) for job_id in job_ids: path = package_path(root, job_id) path.unlink(missing_ok=True) path.with_suffix(".tmp").unlink(missing_ok=True) path.with_suffix(".manifest.json").unlink(missing_ok=True) async def _actor(request: Request, *, admin: bool = False) -> str | None: user = await get_optional_user_from_request(request) if user is None: return None if admin and getattr(user, "system_role", None) != "admin": raise HTTPException(status_code=403, detail="该操作仅管理员可用") return str(user.id) def _store(request: Request): store = getattr(request.app.state, "assistant_knowledge_store", None) if store is None: raise HTTPException(status_code=503, detail="助手知识库存储不可用") return store class InitializeRequest(BaseModel): name: str = Field(default="知识梳理总库", min_length=1, max_length=255) class BaseCreateRequest(BaseModel): name: str = Field(min_length=1, max_length=255) description: str = Field(default="", max_length=2048) class BaseUpdateRequest(BaseModel): name: str = Field(min_length=1, max_length=255) description: str = Field(default="", max_length=2048) class SkillImportRequest(BaseModel): skill_names: list[str] = Field(min_length=1, max_length=200) force: bool = False class ExportToAssistantRequest(BaseModel): assistant_base_id: str | None = Field(default=None, max_length=128) @router.get("/base") async def get_base(request: Request) -> dict[str, Any]: await _actor(request) return {"base": await _store(request).get_base()} @router.post("/base/initialize", status_code=201) async def initialize(request: Request, body: InitializeRequest) -> dict[str, Any]: actor = await _actor(request, admin=True) return {"base": await _store(request).initialize(created_by=actor, name=body.name)} @router.get("/bases") async def list_bases(request: Request) -> dict[str, Any]: await _actor(request) bases = await _store(request).list_bases() return {"bases": bases, "total": len(bases)} @router.post("/bases", status_code=201) async def create_base(request: Request, body: BaseCreateRequest) -> dict[str, Any]: actor = await _actor(request, admin=True) return { "base": await _store(request).create_base( created_by=actor, name=body.name, description=body.description, ) } @router.get("/bases/{base_id}") async def get_base_by_id(request: Request, base_id: str) -> dict[str, Any]: await _actor(request) base = await _store(request).get_base(base_id) if base is None: raise HTTPException(status_code=404, detail="助手知识库不存在") return {"base": base} @router.patch("/bases/{base_id}") async def update_base(request: Request, base_id: str, body: BaseUpdateRequest) -> dict[str, Any]: await _actor(request, admin=True) base = await _store(request).update_base( base_id, name=body.name, description=body.description, ) if base is None: raise HTTPException(status_code=404, detail="助手知识库不存在") return {"base": base} @router.delete("/bases/{base_id}", status_code=204) async def delete_base(request: Request, base_id: str) -> None: await _actor(request, admin=True) store = _store(request) base = await store.get_base(base_id) if base is None: raise HTTPException(status_code=404, detail="助手知识库不存在") if base.get("is_global"): raise HTTPException(status_code=409, detail="知识梳理总库承担全库沉淀,不能删除") jobs = await store.list_jobs(limit=500, base_id=base_id) if any(job.get("status") in {"queued", "running"} for job in jobs): raise HTTPException(status_code=409, detail="知识库正在导入数据,请等待任务完成后再删除") exports = await asyncio.to_thread( _list_export_manifests, f"assistant:{base_id}", limit=10_000, ) if any(job.get("status") in {"queued", "running"} for job in exports): raise HTTPException(status_code=409, detail="知识库正在生成导出包,请等待任务完成后再删除") try: deleted = await store.delete_base(base_id) except ValueError as exc: raise HTTPException(status_code=409, detail=str(exc)) from exc if not deleted: raise HTTPException(status_code=404, detail="助手知识库不存在") await asyncio.to_thread( _delete_base_package_files, base_id, [str(job.get("id") or "") for job in jobs], ) @router.get("/wiki-pages") async def list_wiki_pages( request: Request, base_id: str | None = Query(default=None), q: str | None = Query(default=None), page_type: str | None = Query(default=None), limit: int = Query(default=100, ge=1, le=500), offset: int = Query(default=0, ge=0), ) -> dict[str, Any]: await _actor(request) pages = await _store(request).list_pages(base_id=base_id, query=q, page_type=page_type, limit=limit + 1, offset=offset) return {"pages": pages[:limit], "total": min(len(pages), limit), "next_offset": offset + limit if len(pages) > limit else None} @router.get("/wiki-pages/{slug:path}/revisions/compare") async def compare_revisions(request: Request, slug: str, left: str = Query(...), right: str = Query(...)) -> dict[str, Any]: await _actor(request) result = await _store(request).compare_revisions(left, right) if result is None: raise HTTPException(status_code=404, detail="版本不存在或不属于同一页面") return result @router.get("/wiki-pages/{slug:path}/revisions") async def list_revisions(request: Request, slug: str, base_id: str | None = Query(default=None)) -> dict[str, Any]: await _actor(request) return {"revisions": await _store(request).list_revisions(slug, base_id=base_id)} @router.post("/wiki-pages/{slug:path}/revisions/{revision_id}/rollback") async def rollback_revision( request: Request, slug: str, revision_id: str, base_id: str | None = Query(default=None), ) -> dict[str, Any]: actor = await _actor(request, admin=True) row = await _store(request).rollback(slug=slug, revision_id=revision_id, actor=actor, base_id=base_id) if row is None: raise HTTPException(status_code=404, detail="页面或版本不存在") return row @router.get("/wiki-pages/{slug:path}") async def get_wiki_page(request: Request, slug: str, base_id: str | None = Query(default=None)) -> dict[str, Any]: await _actor(request) row = await _store(request).get_page(slug, base_id=base_id) if row is None: raise HTTPException(status_code=404, detail="Wiki 页面不存在") return row @router.get("/search") async def search( request: Request, q: str = Query(..., min_length=1), base_id: str | None = Query(default=None), page_type: str | None = Query(default=None), ) -> dict[str, Any]: await _actor(request) store = _store(request) base = await store.get_base(base_id) embedding = getattr(request.app.state, "llmwiki_embedding", None) rows = await store.search_pages( base_ids=[base["id"]] if base else [], query=q, limit=20, include_chunks=True, embedding_client=embedding, ) return {"results": rows, "total": len(rows), "retrieval_mode": rows[0].get("retrieval_mode", "keyword") if rows else "empty"} @router.get("/sources") async def list_sources(request: Request, base_id: str | None = Query(default=None)) -> dict[str, Any]: await _actor(request) return {"sources": await _store(request).list_sources(base_id=base_id)} @router.get("/entities/{entity_id}") async def get_entity(request: Request, entity_id: str) -> dict[str, Any]: await _actor(request) row = await _store(request).get_entity(entity_id) if row is None: raise HTTPException(status_code=404, detail="实体不存在") return row @router.get("/relations/{relation_id}") async def get_relation(request: Request, relation_id: str) -> dict[str, Any]: await _actor(request) row = await _store(request).get_relation(relation_id) if row is None: raise HTTPException(status_code=404, detail="关系不存在") return row @router.delete("/sources/{source_id}", status_code=204) async def delete_source(request: Request, source_id: str) -> None: await _actor(request, admin=True) if not await _store(request).delete_source(source_id): raise HTTPException(status_code=404, detail="来源不存在") @router.get("/import-jobs") async def list_import_jobs( request: Request, base_id: str | None = Query(default=None), limit: int = Query(default=100, ge=1, le=500), ) -> dict[str, Any]: await _actor(request) jobs = await _store(request).list_jobs(limit, base_id=base_id) for job in jobs: job["package_download_url"] = _job_package_url(str(job["id"])) return {"jobs": jobs} @router.get("/import-jobs/{job_id}") async def get_import_job(request: Request, job_id: str) -> dict[str, Any]: await _actor(request) row = await _store(request).get_job(job_id) if row is None: raise HTTPException(status_code=404, detail="导入任务不存在") row["package_download_url"] = _job_package_url(job_id) return row @router.get("/import-jobs/{job_id}/package") async def download_import_package(request: Request, job_id: str) -> FileResponse: await _actor(request, admin=True) row = await _store(request).get_job(job_id) if row is None: raise HTTPException(status_code=404, detail="导入任务不存在") path = package_path(_package_root(), job_id) if not path.is_file(): raise HTTPException(status_code=404, detail="导出包尚未生成或已被清理") return FileResponse( path, media_type="application/x-ndjson", filename=f"weknora-assistant-knowledge-{job_id}.jsonl", ) @router.post("/imports/skills", status_code=202) async def import_skills(request: Request, _body: SkillImportRequest) -> dict[str, Any]: await _actor(request, admin=True) raise HTTPException(status_code=410, detail="不支持技能直接归纳到助手知识库,请先归纳到 WeKnora 普通知识库后再导入助手知识库") @router.post("/imports/files", status_code=202) async def import_files( request: Request, file: UploadFile = File(...), assistant_base_id: str | None = Form(default=None), ) -> dict[str, Any]: import hashlib from app.gateway.assistant_file_ingest import ALLOWED_EXTENSIONS, MAX_FILE_BYTES, file_job_dir, save_json, start_file_job actor = await _actor(request, admin=True) filename = (file.filename or "document").replace("\\", "/").rsplit("/", 1)[-1][:240] extension = Path(filename).suffix.lower() if extension not in ALLOWED_EXTENSIONS: await file.close() raise HTTPException(status_code=415, detail="支持 Markdown、TXT、CSV、PDF、Word、Excel、PowerPoint 文件") job = await _store(request).create_queued_import_job( source_type="file", source_key=f"file-upload:{uuid4()}", source_name=filename, trigger="local_file_upload", created_by=actor, base_id=assistant_base_id, metadata={"type": "file", "filename": filename}, phase="uploading", ) root = file_job_dir(job["id"]) await asyncio.to_thread(root.mkdir, parents=True, exist_ok=True) path = root / f"original{extension}" size, digest = 0, hashlib.sha256() try: with path.open("wb") as handle: while chunk := await file.read(1024 * 1024): size += len(chunk) if size > MAX_FILE_BYTES: raise HTTPException(status_code=413, detail="单个文件最大 100 MB,请拆分后上传") digest.update(chunk) await asyncio.to_thread(handle.write, chunk) if not size: raise HTTPException(status_code=422, detail="上传文件为空") await asyncio.to_thread(save_json, root / "manifest.json", { "filename": filename, "stored_name": path.name, "digest": digest.hexdigest(), "actor": actor, }) job = await _store(request).update_import_job(job["id"], phase="queued", counts={"uploaded_bytes": size}) start_file_job(request.app, job["id"]) return job except Exception as exc: await _store(request).update_import_job(job["id"], status="failed", phase="upload_failed", error=str(getattr(exc, "detail", exc)), completed=True) await asyncio.to_thread(path.unlink, missing_ok=True) raise finally: await file.close() @router.post("/import-jobs/{job_id}/retry-file", status_code=202) async def retry_file_import(request: Request, job_id: str) -> dict[str, Any]: from app.gateway.assistant_file_ingest import file_job_dir, retry_file_job await _actor(request, admin=True) store = _store(request) job = await store.get_job(job_id) if not job or job.get("trigger") != "local_file_upload": raise HTTPException(status_code=404, detail="文件处理任务不存在") if job["status"] == "completed": raise HTTPException(status_code=409, detail="文件已处理完成") if not (file_job_dir(job_id) / "manifest.json").exists(): raise HTTPException(status_code=409, detail="文件上传未完成,请重新上传") if not await retry_file_job(request.app, job_id): raise HTTPException(status_code=409, detail="文件仍在处理中") return await store.get_job(job_id) @router.post("/imports/weknora-package", status_code=202) async def import_weknora_package_file( request: Request, file: UploadFile = File(...), assistant_base_id: str | None = Form(default=None), ) -> dict[str, Any]: actor = await _actor(request, admin=True) upload_id = str(uuid4()) filename = (file.filename or "weknora-assistant-knowledge.jsonl").strip() job = await _store(request).create_queued_import_job( source_type="weknora", source_key=f"upload:{upload_id}", source_name=filename, trigger="manual_package_upload", created_by=actor, metadata={ "type": "weknora_export_package", "filename": filename, "upload_id": upload_id, }, upload_batch_id=upload_id, base_id=assistant_base_id, phase="uploading", ) path = package_path(_package_root(), str(job["id"])) path.parent.mkdir(parents=True, exist_ok=True) size = 0 try: with path.open("wb") as handle: while True: chunk = await file.read(1024 * 1024) if not chunk: break size += len(chunk) await asyncio.to_thread(handle.write, chunk) except Exception as exc: await _store(request).update_import_job(str(job["id"]), status="failed", phase="upload_failed", error=str(exc), completed=True) raise finally: await file.close() if size <= 0: await _store(request).update_import_job( str(job["id"]), status="failed", phase="failed", error="上传文件为空", completed=True, ) raise HTTPException(status_code=422, detail="上传文件为空") await _store(request).update_import_job( str(job["id"]), status="queued", phase="queued", counts={"uploaded_bytes": size}, ) asyncio.create_task( _import_package_file_for_job( request=request, job_id=str(job["id"]), path=path, actor=actor, ) ) job["package_download_url"] = _job_package_url(str(job["id"])) return job async def _write_weknora_export_package( client: Any, remote_id: str, source: dict[str, Any], path: Path, *, vectors: AsyncIterable[dict[str, Any]] | None = None, ) -> dict[str, int]: path.parent.mkdir(parents=True, exist_ok=True) tmp_path = path.with_suffix(path.suffix + ".tmp") counts: dict[str, int] = { "wiki_page": 0, "document": 0, "chunk": 0, "vector": 0, "entity": 0, "relation": 0, "graph": 0, } with tmp_path.open("w", encoding="utf-8", newline="\n") as handle: write_jsonl_record( handle, "source", { **source, "provider": "weknora", "package_version": PACKAGE_VERSION, "export_format": "assistant-knowledge-jsonl", }, ) # Wiki pages first: these are the human-readable pages that the # assistant knowledge base presents in wiki form. page_no = 1 graph_nodes = [] graph_edges = set() while True: page = await client.list_wiki_pages(remote_id, page=page_no, page_size=100) batch = page.get("pages") or [] for row in batch: if isinstance(row, dict): payload = row slug = str(row.get("slug") or row.get("wiki_slug") or row.get("path") or "") if slug: try: detail = await client.get_wiki_page(remote_id, slug) if isinstance(detail, dict): payload = {**row, **detail} except WeKnoraError: # A list summary is not a replacement for the full # article. An incomplete package must fail visibly. raise write_jsonl_record(handle, "wiki_page", payload) counts["wiki_page"] += 1 graph_nodes.append({"slug": slug, "title": payload.get("title"), "page_type": payload.get("page_type"), "aliases": payload.get("aliases") or []}) for target in payload.get("out_links") or []: graph_edges.add((slug, str(target))) for match in re.finditer(r"\[\[([^\]|]+)(?:\|[^\]]+)?\]\]", str(payload.get("content") or "")): graph_edges.add((slug, match.group(1).strip())) if counts["wiki_page"] >= int(page.get("total") or counts["wiki_page"]) or not batch: break page_no += 1 # Then export WeKnora's processed knowledge and chunks. Chunks are the # important boundary for large RAG imports: the assistant side should # consume the already-parsed/segmented material instead of re-reading # arbitrary source files. document_count = 0 page_no = 1 while True: page = await client.list_documents(remote_id, page=page_no, page_size=100) batch = page.get("items") or [] for document in batch: if not isinstance(document, dict): continue write_jsonl_record(handle, "document", document) counts["document"] += 1 document_count += 1 document_id = str(document.get("id") or document.get("knowledge_id") or "") if not document_id: continue chunk_page = 1 while True: values = await client.list_chunks(document_id, page=chunk_page, page_size=100) chunk_batch = values.get("items") or [] for row in chunk_batch: if isinstance(row, dict): write_jsonl_record(handle, "chunk", {**row, "source_document_id": document_id}) counts["chunk"] += 1 if len(chunk_batch) < 100: break chunk_page += 1 if document_count >= int(page.get("total") or document_count) or not batch: break page_no += 1 if vectors is not None: async for row in vectors: if isinstance(row, dict): write_jsonl_record(handle, "vector", row) counts["vector"] += 1 # Build the complete link graph from the paginated full pages. The # provider's overview graph endpoint is capped and can omit nodes. known_slugs = {row["slug"] for row in graph_nodes} graph = {"nodes": graph_nodes, "edges": [{"source": a, "target": b} for a, b in sorted(graph_edges) if a in known_slugs and b in known_slugs]} nodes = [row for row in graph.get("nodes") or [] if isinstance(row, dict)] edges = [row for row in graph.get("edges") or [] if isinstance(row, dict)] write_jsonl_record(handle, "graph", {**graph, "nodes": nodes, "edges": edges}) counts["graph"] += 1 for row in nodes: if (row.get("slug") or row.get("id")) and (row.get("title") or row.get("name")): write_jsonl_record( handle, "entity", { "id": row.get("slug") or row.get("id"), "name": row.get("title") or row.get("name"), "type": row.get("page_type") or row.get("type") or "concept", "source_wiki_slug": row.get("slug") or row.get("id"), "aliases": row.get("aliases") or [], "confidence": 1.0, }, ) counts["entity"] += 1 for row in edges: write_jsonl_record( handle, "relation", { "source_entity_id": row.get("source"), "target_entity_id": row.get("target"), "predicate": row.get("predicate") or "references", "source_wiki_slug": row.get("source"), "confidence": 1.0, "evidence": [{"source": "weknora_graph"}], }, ) counts["relation"] += 1 tmp_path.replace(path) return counts async def _local_wiki_vector_context(request: Request, mapping_id: str) -> tuple[Any | None, str, str]: store = getattr(request.app.state, "llmwiki_index_store", None) if store is None: return None, "", "" embedding = getattr(request.app.state, "llmwiki_embedding", None) fingerprint = str(getattr(embedding, "fingerprint", "") or "") if not fingerprint: try: _revision, state_fingerprint, _state = await store.get_index_revision(mapping_id) fingerprint = str(state_fingerprint or "") except Exception: # noqa: BLE001 fingerprint = "" if not fingerprint: return store, "", "" model = str(getattr(getattr(embedding, "config", None), "model", "") or "") return store, fingerprint, model def _local_wiki_vector_export_payload( row: dict[str, Any], *, model: str, fingerprint: str, ) -> dict[str, Any] | None: blob = row.get("vector_blob") if isinstance(blob, memoryview): blob = blob.tobytes() if not isinstance(blob, (bytes, bytearray)): return None page = row.get("page") if isinstance(row.get("page"), dict) else {} return { "id": row.get("id"), "wiki_slug": str(page.get("slug") or "").strip("/"), "wiki_page_id": page.get("id") or page.get("remote_page_id"), "title": page.get("title") or "", "section_index": int(row.get("section_index") or 0), "heading": row.get("heading"), "section_content": row.get("section_content") or "", "content_hash": row.get("content_hash") or "", "embedding_model": model, "embedding_fingerprint": row.get("embedding_fingerprint") or fingerprint, "embedding_dimensions": int(row.get("embedding_dimensions") or 0), "vector_encoding": "float32-le-normalized-base64", "vector_blob_base64": base64.b64encode(bytes(blob)).decode("ascii"), } async def _iter_local_wiki_vectors_for_export( request: Request, mapping_id: str, *, batch_size: int = 1000, ) -> AsyncIterator[dict[str, Any]]: store, fingerprint, model = await _local_wiki_vector_context(request, mapping_id) if store is None or not fingerprint: return try: async for row in store.iter_vector_snapshot_rows(mapping_id, fingerprint, batch_size=batch_size): if not isinstance(row, dict): continue payload = _local_wiki_vector_export_payload(row, model=model, fingerprint=fingerprint) if payload is not None: yield payload except Exception: logger.exception("Wiki vector export failed for %s", mapping_id) raise async def _has_local_wiki_vectors_for_export(request: Request, mapping_id: str) -> bool: async for _row in _iter_local_wiki_vectors_for_export(request, mapping_id, batch_size=1): return True return False async def _load_local_wiki_vectors_for_export(request: Request, mapping_id: str) -> list[dict[str, Any]]: return [row async for row in _iter_local_wiki_vectors_for_export(request, mapping_id)] async def _ensure_local_wiki_vectors_for_export( request: Request, mapping: dict[str, Any], ) -> None: mapping_id = str(mapping.get("id") or mapping.get("knowledge_base_id") or "") service = getattr(request.app.state, "llmwiki_sync_service", None) if service is None or not mapping_id: raise RuntimeError("请先配置 Wiki 编码模型并启用本地向量索引,再生成向量数据包") # Refresh remote metadata/content hashes even when the previous generation # was complete. Unchanged articles reuse their existing vectors. sync_mapping = dict(mapping) sync_mapping.setdefault("id", mapping_id) sync_mapping.setdefault( "weknora_id", mapping.get("remote_knowledge_base_id") or mapping.get("weknora_id") or "", ) if not sync_mapping.get("weknora_id"): return result = await service.sync_mapping(sync_mapping, force=False, notify=False) if result.get("status") != "completed": raise RuntimeError("Wiki 向量化尚未全部完成,请查看向量化进度后重试") async def _import_package_file_for_job( *, request: Request, job_id: str, path: Path, actor: str | None, ) -> None: store = _store(request) package = None try: await store.update_import_job(job_id, status="running", phase="loading_package") package = await asyncio.to_thread(PackageArchive, path) await store.update_import_job(job_id, status="running", phase="importing") await store.import_package_for_job( job_id, package=package, created_by=actor, mirror_to_global=True, ) except Exception as exc: # noqa: BLE001 logger.exception("Assistant knowledge package import failed: job=%s", job_id) await store.update_import_job( job_id, status="failed", phase="failed", error=str(exc), completed=True, ) finally: if package is not None: await asyncio.to_thread(package.close) async def _export_weknora_then_import_for_job( *, request: Request, job_id: str, mapping: dict[str, Any], remote_id: str, source: dict[str, Any], actor: str | None, index_fresh: bool = False, ) -> None: store = _store(request) path = package_path(_package_root(), job_id) try: await store.update_import_job(job_id, status="running", phase="exporting") runtime = get_resolved_llmwiki_runtime(request.app.state.config) if not runtime.weknora_enabled: raise RuntimeError("未配置普通知识库服务") mapping_id = str(mapping.get("id") or mapping.get("knowledge_base_id") or "") if not index_fresh: await _ensure_local_wiki_vectors_for_export(request, mapping) counts = await _write_weknora_export_package( build_weknora_client(runtime), remote_id, source, path, vectors=_iter_local_wiki_vectors_for_export(request, mapping_id), ) await store.update_import_job(job_id, status="running", phase="package_ready", counts=counts) await _import_package_file_for_job(request=request, job_id=job_id, path=path, actor=actor) except Exception as exc: # noqa: BLE001 logger.exception("Assistant knowledge WeKnora export failed: job=%s", job_id) await store.update_import_job( job_id, status="failed", phase="failed", error=str(exc), completed=True, ) async def _export_weknora_package(client: Any, remote_id: str, source: dict[str, Any]) -> dict[str, Any]: """Legacy in-memory export shape, kept for tests and older callers.""" pages: list[dict[str, Any]] = [] page_no = 1 while True: page = await client.list_wiki_pages(remote_id, page=page_no, page_size=100) batch = page.get("pages") or [] pages.extend(batch) if len(pages) >= int(page.get("total") or len(pages)) or not batch: break page_no += 1 documents: list[dict[str, Any]] = [] chunks: list[dict[str, Any]] = [] page_no = 1 while True: page = await client.list_documents(remote_id, page=page_no, page_size=100) batch = page.get("items") or [] documents.extend(batch) for document in batch: document_id = str(document.get("id") or document.get("knowledge_id") or "") if not document_id: continue chunk_page = 1 while True: values = await client.list_chunks(document_id, page=chunk_page, page_size=100) chunk_batch = values.get("items") or [] chunks.extend([{**row, "source_document_id": document_id} for row in chunk_batch]) if len(chunk_batch) < 100: break chunk_page += 1 if len(documents) >= int(page.get("total") or len(documents)) or not batch: break page_no += 1 try: graph = await client.get_wiki_graph(remote_id, limit=10000) except WeKnoraError: graph = {"nodes": [], "edges": [], "capability_warning": "graph API unavailable"} entities = [{"id": row.get("id"), "name": row.get("title") or row.get("name"), "type": row.get("type") or "concept", "confidence": 1.0} for row in graph.get("nodes") or [] if row.get("id") and (row.get("title") or row.get("name"))] relations = [ {"source_entity_id": row.get("source"), "target_entity_id": row.get("target"), "predicate": row.get("predicate") or "references", "confidence": 1.0, "evidence": [{"source": "weknora_graph"}]} for row in graph.get("edges") or [] ] return {"source": source, "wiki_pages": pages, "documents": documents, "chunks": chunks, "entities": entities, "relations": relations, "graph": graph} async def _export_weknora_package_only_job( *, request: Request, job_id: str, mapping: dict[str, Any], mapping_id: str, remote_id: str, source: dict[str, Any], ) -> None: try: _write_export_manifest(job_id=job_id, mapping_id=mapping_id, status="running", phase="exporting") runtime = get_resolved_llmwiki_runtime(request.app.state.config) if not runtime.weknora_enabled: raise RuntimeError("未配置普通知识库服务") await _ensure_local_wiki_vectors_for_export(request, mapping) counts = await _write_weknora_export_package( build_weknora_client(runtime), remote_id, source, package_path(_package_root(), job_id), vectors=_iter_local_wiki_vectors_for_export(request, mapping_id), ) _write_export_manifest( job_id=job_id, mapping_id=mapping_id, status="completed", phase="package_ready", counts=counts, ) except Exception as exc: # noqa: BLE001 logger.exception("WeKnora vectorized export package failed: mapping=%s job=%s", mapping_id, job_id) _write_export_manifest( job_id=job_id, mapping_id=mapping_id, status="failed", phase="failed", error=str(exc), ) @export_router.get("/api/llmwiki/knowledge-bases/{mapping_id}/export-package") async def list_weknora_export_packages( request: Request, mapping_id: str, limit: int = Query(default=50, ge=1, le=200), ) -> dict[str, Any]: actor = await _actor(request, admin=True) mapping = await request.app.state.llmwiki_store.get_authorized( mapping_id, actor or "system", write=False, is_admin=True, ) if mapping is None: raise HTTPException(status_code=404, detail="普通知识库不存在") rows = _list_export_manifests(mapping_id, limit=limit) return {"jobs": rows, "total": len(rows)} @export_router.post("/api/llmwiki/knowledge-bases/{mapping_id}/export-package", status_code=202) async def create_weknora_export_package(request: Request, mapping_id: str) -> dict[str, Any]: actor = await _actor(request, admin=True) runtime = get_resolved_llmwiki_runtime(request.app.state.config) if not runtime.weknora_enabled: raise HTTPException(status_code=503, detail="未配置普通知识库服务") mapping = await request.app.state.llmwiki_store.get_authorized( mapping_id, actor or "system", write=False, is_admin=True, ) if mapping is None: raise HTTPException(status_code=404, detail="普通知识库不存在") job_id = str(uuid4()) source = { "type": "weknora", "knowledge_base_id": mapping_id, "remote_knowledge_base_id": mapping["weknora_id"], "knowledge_base_name": mapping["name"], "package_kind": "weknora_vectorized_export", } payload = _write_export_manifest(job_id=job_id, mapping_id=mapping_id, status="queued", phase="queued") asyncio.create_task( _export_weknora_package_only_job( request=request, job_id=job_id, mapping=mapping, mapping_id=mapping_id, remote_id=str(mapping["weknora_id"]), source=source, ) ) return payload @export_router.get("/api/llmwiki/knowledge-bases/{mapping_id}/export-package/{job_id}") async def get_weknora_export_package_status(request: Request, mapping_id: str, job_id: str) -> dict[str, Any]: await _actor(request, admin=True) payload = _read_export_manifest(job_id) if payload is None or str(payload.get("mapping_id") or "") != mapping_id: raise HTTPException(status_code=404, detail="导出任务不存在") return payload @export_router.get("/api/llmwiki/knowledge-bases/{mapping_id}/export-package/{job_id}/download") async def download_weknora_export_package(request: Request, mapping_id: str, job_id: str) -> FileResponse: await _actor(request, admin=True) payload = _read_export_manifest(job_id) if payload is None or str(payload.get("mapping_id") or "") != mapping_id: raise HTTPException(status_code=404, detail="导出任务不存在") if str(payload.get("status") or "") != "completed": raise HTTPException(status_code=409, detail="导出包尚未生成完成") path = package_path(_package_root(), job_id) if not path.is_file(): raise HTTPException(status_code=404, detail="导出包已不存在") return FileResponse( path, media_type="application/x-ndjson", filename=f"weknora-vectorized-knowledge-{mapping_id}-{job_id}.jsonl", ) @export_router.post("/api/llmwiki/knowledge-bases/{mapping_id}/export-to-assistant", status_code=202) async def export_to_assistant( request: Request, mapping_id: str, body: ExportToAssistantRequest | None = None, ) -> dict[str, Any]: actor = await _actor(request, admin=True) runtime = get_resolved_llmwiki_runtime(request.app.state.config) if not runtime.weknora_enabled: raise HTTPException(status_code=503, detail="未配置普通知识库服务") mapping = await request.app.state.llmwiki_store.get_authorized(mapping_id, actor or "system", write=False, is_admin=True) if mapping is None: raise HTTPException(status_code=404, detail="普通知识库不存在") source = { "type": "weknora", "knowledge_base_id": mapping_id, "remote_knowledge_base_id": mapping["weknora_id"], "knowledge_base_name": mapping["name"], } try: job = await _store(request).create_queued_import_job( source_type="weknora", source_key=mapping_id, source_name=str(mapping["name"]), trigger="manual_from_weknora", created_by=actor, metadata=source, mapping_id=mapping_id, base_id=body.assistant_base_id if body else None, phase="exporting", ) except ValueError as exc: raise HTTPException(status_code=404, detail=str(exc)) from exc job["package_download_url"] = _job_package_url(str(job["id"])) asyncio.create_task( _export_weknora_then_import_for_job( request=request, job_id=str(job["id"]), mapping=mapping, remote_id=str(mapping["weknora_id"]), source=source, actor=actor, ) ) return job @router.post("/sources/{source_id}/reimport", status_code=202) async def reimport_source(request: Request, source_id: str) -> dict[str, Any]: actor = await _actor(request, admin=True) store = _store(request) source = await store.get_source(source_id) if source is None: raise HTTPException(status_code=404, detail="来源不存在") if source["source_type"] == "skill": raise HTTPException(status_code=410, detail="不支持技能来源直接重新导入助手知识库,请重新从 WeKnora 普通知识库导入") if source["source_type"] != "weknora": raise HTTPException(status_code=409, detail="本地文件来源未保留原文件,请重新上传") mapping_id = str(source.get("knowledge_base_mapping_id") or source["source_key"]) if not mapping_id or mapping_id.startswith("upload:"): raise HTTPException(status_code=409, detail="上传包来源不能直接重新导入,请重新上传 WeKnora 导出包") runtime = get_resolved_llmwiki_runtime(request.app.state.config) if not runtime.weknora_enabled: raise HTTPException(status_code=503, detail="未配置普通知识库服务") mapping = await request.app.state.llmwiki_store.get_authorized( mapping_id, actor or "system", write=False, is_admin=True, ) if mapping is None: raise HTTPException(status_code=404, detail="来源普通知识库不存在") source_payload = { "type": "weknora", "knowledge_base_id": mapping_id, "remote_knowledge_base_id": mapping["weknora_id"], "knowledge_base_name": mapping["name"], } try: job = await store.create_queued_import_job( source_type="weknora", source_key=mapping_id, source_name=str(mapping["name"]), trigger="source_reimport", created_by=actor, metadata=source_payload, mapping_id=mapping_id, base_id=str(source["base_id"]), phase="exporting", ) except ValueError as exc: raise HTTPException(status_code=404, detail=str(exc)) from exc job["package_download_url"] = _job_package_url(str(job["id"])) asyncio.create_task( _export_weknora_then_import_for_job( request=request, job_id=str(job["id"]), mapping=mapping, remote_id=str(mapping["weknora_id"]), source=source_payload, actor=actor, ) ) return job