init: 初始化项目
This commit is contained in:
@@ -0,0 +1,46 @@
|
||||
import os
|
||||
import json
|
||||
from docx import Document
|
||||
from PyPDF2 import PdfReader
|
||||
from app.services.file_storage import get_file_content
|
||||
|
||||
|
||||
async def parse_reference_file(file_path: str) -> str:
|
||||
ext = os.path.splitext(file_path)[1].lower()
|
||||
content = await get_file_content(file_path)
|
||||
|
||||
if ext == ".txt":
|
||||
return content.decode("utf-8", errors="ignore")
|
||||
if ext == ".docx":
|
||||
from io import BytesIO
|
||||
doc = Document(BytesIO(content))
|
||||
return "\n".join(p.text for p in doc.paragraphs if p.text.strip())
|
||||
if ext == ".pdf":
|
||||
from io import BytesIO
|
||||
reader = PdfReader(BytesIO(content))
|
||||
return "\n".join(page.extract_text() or "" for page in reader.pages)
|
||||
|
||||
raise ValueError(f"不支持的文件格式: {ext}")
|
||||
|
||||
|
||||
async def parse_reference_files(ref_file_path: str | None) -> str:
|
||||
if not ref_file_path:
|
||||
return ""
|
||||
|
||||
try:
|
||||
paths = json.loads(ref_file_path)
|
||||
if isinstance(paths, list):
|
||||
texts = []
|
||||
for path in paths:
|
||||
try:
|
||||
texts.append(await parse_reference_file(path))
|
||||
except Exception:
|
||||
texts.append(f"[无法解析文件: {path}]")
|
||||
return "\n\n".join(texts)
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
try:
|
||||
return await parse_reference_file(ref_file_path)
|
||||
except Exception:
|
||||
return f"[无法解析文件: {ref_file_path}]"
|
||||
Reference in New Issue
Block a user