feat: OpenMesh 基础平台与 MD/PDF 转换技能
- 后端: coworker 智能体框架, WS API, 文件上传, 附件处理 - 前端: Open WebUI, 文件全量走 upload API (含 MD/TXT/JSON 等文本类) - 技能: md-to-office (pandoc + wkhtmltopdf) - 修复: 上传文件路径丢失, Agent 搜索浪费, 输出文件跑到 uploads/ - 打包: PyInstaller one-dir, 预打包 pandoc/wkhtmltopdf/chromium
This commit is contained in:
344
coworker/file_upload.py
Normal file
344
coworker/file_upload.py
Normal file
@@ -0,0 +1,344 @@
|
||||
"""File upload handling and text extraction for Office documents and PDFs.
|
||||
|
||||
Uploaded files are saved to <workspace>/uploads/ with a timestamp prefix to avoid
|
||||
name collisions. For supported document types (PDF, DOCX, PPTX, XLSX), text content
|
||||
is extracted automatically so the model can read them immediately.
|
||||
|
||||
The raw file stays on disk so the model (or skills) can operate on it further
|
||||
(e.g. generate charts from an Excel file, reformat a Word doc, etc.).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
# Maximum text extraction output per file (in characters). Beyond this the text
|
||||
# is truncated so we don't blow out the context window with a single huge file.
|
||||
MAX_EXTRACTED_CHARS = 200_000
|
||||
|
||||
|
||||
def _safe_filename(name: str) -> str:
|
||||
"""Sanitize a filename — keep alphanumerics, dots, dashes, underscores, Chinese chars."""
|
||||
name = Path(name).name # strip any path components
|
||||
# Remove dangerous characters but keep CJK and common safe chars
|
||||
name = re.sub(r'[\\/:*?"<>|\x00-\x1f]', "_", name)
|
||||
name = name.strip().strip(".")
|
||||
if not name:
|
||||
name = "upload"
|
||||
return name[:200] # reasonable length cap
|
||||
|
||||
|
||||
def upload_dir(workspace: str | Path) -> Path:
|
||||
"""Return (and create if needed) the uploads directory for a workspace."""
|
||||
d = Path(workspace) / "uploads"
|
||||
d.mkdir(parents=True, exist_ok=True)
|
||||
return d
|
||||
|
||||
|
||||
def save_upload(workspace: str | Path, filename: str, data: bytes) -> Path:
|
||||
"""Save an uploaded file to the workspace uploads directory with a timestamp prefix.
|
||||
|
||||
Returns the absolute path of the saved file.
|
||||
"""
|
||||
uploads = upload_dir(workspace)
|
||||
safe = _safe_filename(filename)
|
||||
ts = int(time.time() * 1000) # millisecond timestamp
|
||||
# Insert timestamp before the extension so the file still has a recognisable name
|
||||
stem = Path(safe).stem
|
||||
suffix = Path(safe).suffix
|
||||
target = uploads / f"{ts}_{stem}{suffix}"
|
||||
# If somehow it already exists, add a counter
|
||||
counter = 1
|
||||
while target.exists():
|
||||
target = uploads / f"{ts}_{stem}_{counter}{suffix}"
|
||||
counter += 1
|
||||
target.write_bytes(data)
|
||||
return target.resolve()
|
||||
|
||||
|
||||
def detect_file_kind(path: str | Path) -> str:
|
||||
"""Detect the document kind from the filename extension.
|
||||
|
||||
Returns one of: "pdf", "docx", "pptx", "xlsx", "doc", "ppt", "xls", "md", "text", "other"
|
||||
"""
|
||||
ext = Path(path).suffix.lower()
|
||||
mapping = {
|
||||
".pdf": "pdf",
|
||||
".docx": "docx",
|
||||
".doc": "doc",
|
||||
".pptx": "pptx",
|
||||
".ppt": "ppt",
|
||||
".xlsx": "xlsx",
|
||||
".xls": "xls",
|
||||
".md": "md",
|
||||
".markdown": "md",
|
||||
}
|
||||
if ext in mapping:
|
||||
return mapping[ext]
|
||||
# 所有常见纯文本文件类型统一归入 "text"
|
||||
text_exts = {
|
||||
".txt", ".csv", ".tsv", ".json", ".yml", ".yaml", ".log", ".ini", ".toml",
|
||||
".py", ".js", ".ts", ".tsx", ".jsx", ".rs", ".go", ".java", ".c", ".h", ".cpp",
|
||||
".cs", ".rb", ".php", ".swift", ".kt", ".scala", ".r", ".m", ".mm",
|
||||
".sh", ".bash", ".zsh", ".ps1", ".bat", ".cmd",
|
||||
".html", ".htm", ".css", ".scss", ".less", ".sql", ".xml",
|
||||
".cfg", ".conf", ".properties", ".env", ".gitignore", ".dockerfile", ".makefile",
|
||||
".tex", ".rst", ".asciidoc", ".wiki", ".org",
|
||||
}
|
||||
if ext in text_exts:
|
||||
return "text"
|
||||
return "other"
|
||||
|
||||
|
||||
def extract_text(path: str | Path, *, kind: Optional[str] = None) -> str:
|
||||
"""Extract text content from a document file.
|
||||
|
||||
Tries the appropriate library for each file type. If the library isn't available
|
||||
or extraction fails, returns an empty string.
|
||||
|
||||
Supported: PDF (.pdf), Word (.docx), PowerPoint (.pptx), Excel (.xlsx), Markdown (.md)
|
||||
Legacy formats (.doc, .ppt, .xls) are not supported (require external tools).
|
||||
"""
|
||||
p = Path(path)
|
||||
if not p.is_file():
|
||||
return ""
|
||||
if kind is None:
|
||||
kind = detect_file_kind(p)
|
||||
|
||||
try:
|
||||
if kind == "pdf":
|
||||
return _extract_pdf(p)
|
||||
elif kind == "docx":
|
||||
return _extract_docx(p)
|
||||
elif kind == "pptx":
|
||||
return _extract_pptx(p)
|
||||
elif kind == "xlsx":
|
||||
return _extract_xlsx(p)
|
||||
elif kind == "md":
|
||||
return _extract_md(p)
|
||||
elif kind == "text":
|
||||
# 通用纯文本文件(.txt, .json, .py, .yaml, .log 等)
|
||||
return _extract_plain_text(p)
|
||||
except Exception:
|
||||
# Extraction failures are non-fatal — the file is still on disk for the
|
||||
# model to handle via skills if needed.
|
||||
pass
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_plain_text(path: Path) -> str:
|
||||
"""Extract text from any plain-text file."""
|
||||
# 尝试多种编码,UTF-8 优先
|
||||
for enc in ("utf-8", "utf-8-sig", "gbk", "latin-1"):
|
||||
try:
|
||||
text = path.read_text(encoding=enc)
|
||||
return _truncate(text)
|
||||
except (UnicodeDecodeError, UnicodeError):
|
||||
continue
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_md(path: Path) -> str:
|
||||
"""Extract text from a Markdown .md file."""
|
||||
text = path.read_text(encoding="utf-8")
|
||||
return _truncate(text)
|
||||
|
||||
|
||||
def _truncate(text: str) -> str:
|
||||
if len(text) <= MAX_EXTRACTED_CHARS:
|
||||
return text
|
||||
return text[:MAX_EXTRACTED_CHARS] + f"\n\n[... truncated, {len(text)} total chars]"
|
||||
|
||||
|
||||
def _extract_pdf(path: Path) -> str:
|
||||
"""Extract text from a PDF using pdfplumber (preferred) or PyPDF2 (fallback)."""
|
||||
try:
|
||||
import pdfplumber
|
||||
|
||||
out: list[str] = []
|
||||
with pdfplumber.open(str(path)) as pdf:
|
||||
for i, page in enumerate(pdf.pages):
|
||||
t = page.extract_text() or ""
|
||||
if t.strip():
|
||||
out.append(f"--- Page {i + 1} ---\n{t}")
|
||||
return _truncate("\n\n".join(out))
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
try:
|
||||
from PyPDF2 import PdfReader
|
||||
|
||||
reader = PdfReader(str(path))
|
||||
out: list[str] = []
|
||||
for i, page in enumerate(reader.pages):
|
||||
t = page.extract_text() or ""
|
||||
if t.strip():
|
||||
out.append(f"--- Page {i + 1} ---\n{t}")
|
||||
return _truncate("\n\n".join(out))
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_docx(path: Path) -> str:
|
||||
"""Extract text from a Word .docx file using python-docx."""
|
||||
try:
|
||||
from docx import Document
|
||||
|
||||
doc = Document(str(path))
|
||||
out: list[str] = []
|
||||
|
||||
# Paragraphs
|
||||
for para in doc.paragraphs:
|
||||
if para.text.strip():
|
||||
out.append(para.text)
|
||||
|
||||
# Tables
|
||||
for table_idx, table in enumerate(doc.tables):
|
||||
out.append(f"\n--- Table {table_idx + 1} ---")
|
||||
for row in table.rows:
|
||||
cells = [cell.text.strip() for cell in row.cells]
|
||||
out.append(" | ".join(cells))
|
||||
|
||||
return _truncate("\n".join(out))
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_pptx(path: Path) -> str:
|
||||
"""Extract text from a PowerPoint .pptx file using python-pptx."""
|
||||
try:
|
||||
from pptx import Presentation
|
||||
|
||||
prs = Presentation(str(path))
|
||||
out: list[str] = []
|
||||
for i, slide in enumerate(prs.slides):
|
||||
slide_text: list[str] = []
|
||||
for shape in slide.shapes:
|
||||
if hasattr(shape, "text") and shape.text.strip():
|
||||
slide_text.append(shape.text.strip())
|
||||
# Also check tables
|
||||
for shape in slide.shapes:
|
||||
if shape.has_table:
|
||||
for row in shape.table.rows:
|
||||
cells = [cell.text.strip() for cell in row.cells]
|
||||
slide_text.append(" | ".join(cells))
|
||||
if slide_text:
|
||||
out.append(f"--- Slide {i + 1} ---\n" + "\n".join(slide_text))
|
||||
|
||||
# Notes
|
||||
for i, slide in enumerate(prs.slides):
|
||||
if slide.has_notes_slide:
|
||||
notes = slide.notes_slide.notes_text_frame.text.strip()
|
||||
if notes:
|
||||
out.append(f"--- Slide {i + 1} Notes ---\n{notes}")
|
||||
|
||||
return _truncate("\n\n".join(out))
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_xlsx(path: Path) -> str:
|
||||
"""Extract text from an Excel .xlsx file using openpyxl."""
|
||||
try:
|
||||
from openpyxl import load_workbook
|
||||
|
||||
wb = load_workbook(str(path), read_only=True, data_only=True)
|
||||
out: list[str] = []
|
||||
for sheet_name in wb.sheetnames:
|
||||
ws = wb[sheet_name]
|
||||
out.append(f"--- Sheet: {sheet_name} ---")
|
||||
row_count = 0
|
||||
for row in ws.iter_rows(values_only=True):
|
||||
cells = [
|
||||
str(cell) if cell is not None else ""
|
||||
for cell in row
|
||||
]
|
||||
# Skip completely empty rows
|
||||
if any(c.strip() for c in cells):
|
||||
out.append(" | ".join(cells))
|
||||
row_count += 1
|
||||
if row_count > 500: # safety cap per sheet
|
||||
out.append("[... more rows omitted]")
|
||||
break
|
||||
out.append("") # blank line between sheets
|
||||
|
||||
wb.close()
|
||||
return _truncate("\n".join(out))
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def upload_result_dict(
|
||||
file_path: Path,
|
||||
*,
|
||||
original_name: str,
|
||||
extracted_text: Optional[str] = None,
|
||||
kind: Optional[str] = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build the result dict returned by the upload API.
|
||||
|
||||
The attachment shape uses `kind: "text"` so it's compatible with the existing
|
||||
attachment pipeline — the model sees the extracted text inline. Additional
|
||||
fields (`file_path`, `file_kind`, `file_size`) carry metadata the model or
|
||||
skills can use to operate on the raw file.
|
||||
"""
|
||||
if kind is None:
|
||||
kind = detect_file_kind(file_path)
|
||||
|
||||
size = file_path.stat().st_size if file_path.is_file() else 0
|
||||
|
||||
# Prefix the extracted text with file metadata so the model knows where the
|
||||
# raw file lives and can use skills to operate on it further.
|
||||
# Note: build_user_content() in attachments.py already adds its own
|
||||
# "[Attached file: <name>]" prefix, so we start with the path/type info
|
||||
# right after the filename to avoid double-wrapping.
|
||||
text_body = extracted_text or ""
|
||||
if text_body:
|
||||
header_lines = [
|
||||
"--- UPLOADED FILE INFO ---",
|
||||
f"📎 File: {original_name} | Type: {kind} | Size: {size} bytes",
|
||||
f"📁 ABSOLUTE PATH (USE THIS DIRECTLY — do NOT search for the file):",
|
||||
f" {file_path}",
|
||||
"--- END UPLOADED FILE INFO ---",
|
||||
"",
|
||||
"--- Extracted text content ---",
|
||||
"",
|
||||
]
|
||||
text_body = "\n".join(header_lines) + text_body
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"kind": "text", # compatible with existing attachment pipeline
|
||||
"name": original_name,
|
||||
"mime": _mime_for_kind(kind),
|
||||
"text": text_body,
|
||||
"file_path": str(file_path),
|
||||
"file_kind": kind,
|
||||
"file_size": size,
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def _mime_for_kind(kind: str) -> str:
|
||||
mapping = {
|
||||
"pdf": "application/pdf",
|
||||
"docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"doc": "application/msword",
|
||||
"pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"ppt": "application/vnd.ms-powerpoint",
|
||||
"xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"xls": "application/vnd.ms-excel",
|
||||
"md": "text/markdown",
|
||||
}
|
||||
return mapping.get(kind, "application/octet-stream")
|
||||
Reference in New Issue
Block a user