Files
OpenMesh/coworker/file_upload.py
zhaolei 6f402ffcee
Some checks failed
CI / pytest (push) Has been cancelled
CI / gui-unit (push) Has been cancelled
CI / gui-e2e (push) Has been cancelled
feat: OpenMesh 基础平台与 MD/PDF 转换技能
- 后端: coworker 智能体框架, WS API, 文件上传, 附件处理
- 前端: Open WebUI, 文件全量走 upload API (含 MD/TXT/JSON 等文本类)
- 技能: md-to-office (pandoc + wkhtmltopdf)
- 修复: 上传文件路径丢失, Agent 搜索浪费, 输出文件跑到 uploads/
- 打包: PyInstaller one-dir, 预打包 pandoc/wkhtmltopdf/chromium
2026-09-13 23:41:04 +08:00

345 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""File upload handling and text extraction for Office documents and PDFs.
Uploaded files are saved to <workspace>/uploads/ with a timestamp prefix to avoid
name collisions. For supported document types (PDF, DOCX, PPTX, XLSX), text content
is extracted automatically so the model can read them immediately.
The raw file stays on disk so the model (or skills) can operate on it further
(e.g. generate charts from an Excel file, reformat a Word doc, etc.).
"""
from __future__ import annotations
import re
import time
from pathlib import Path
from typing import Any, Optional
# Maximum text extraction output per file (in characters). Beyond this the text
# is truncated so we don't blow out the context window with a single huge file.
MAX_EXTRACTED_CHARS = 200_000
def _safe_filename(name: str) -> str:
"""Sanitize a filename — keep alphanumerics, dots, dashes, underscores, Chinese chars."""
name = Path(name).name # strip any path components
# Remove dangerous characters but keep CJK and common safe chars
name = re.sub(r'[\\/:*?"<>|\x00-\x1f]', "_", name)
name = name.strip().strip(".")
if not name:
name = "upload"
return name[:200] # reasonable length cap
def upload_dir(workspace: str | Path) -> Path:
"""Return (and create if needed) the uploads directory for a workspace."""
d = Path(workspace) / "uploads"
d.mkdir(parents=True, exist_ok=True)
return d
def save_upload(workspace: str | Path, filename: str, data: bytes) -> Path:
"""Save an uploaded file to the workspace uploads directory with a timestamp prefix.
Returns the absolute path of the saved file.
"""
uploads = upload_dir(workspace)
safe = _safe_filename(filename)
ts = int(time.time() * 1000) # millisecond timestamp
# Insert timestamp before the extension so the file still has a recognisable name
stem = Path(safe).stem
suffix = Path(safe).suffix
target = uploads / f"{ts}_{stem}{suffix}"
# If somehow it already exists, add a counter
counter = 1
while target.exists():
target = uploads / f"{ts}_{stem}_{counter}{suffix}"
counter += 1
target.write_bytes(data)
return target.resolve()
def detect_file_kind(path: str | Path) -> str:
"""Detect the document kind from the filename extension.
Returns one of: "pdf", "docx", "pptx", "xlsx", "doc", "ppt", "xls", "md", "text", "other"
"""
ext = Path(path).suffix.lower()
mapping = {
".pdf": "pdf",
".docx": "docx",
".doc": "doc",
".pptx": "pptx",
".ppt": "ppt",
".xlsx": "xlsx",
".xls": "xls",
".md": "md",
".markdown": "md",
}
if ext in mapping:
return mapping[ext]
# 所有常见纯文本文件类型统一归入 "text"
text_exts = {
".txt", ".csv", ".tsv", ".json", ".yml", ".yaml", ".log", ".ini", ".toml",
".py", ".js", ".ts", ".tsx", ".jsx", ".rs", ".go", ".java", ".c", ".h", ".cpp",
".cs", ".rb", ".php", ".swift", ".kt", ".scala", ".r", ".m", ".mm",
".sh", ".bash", ".zsh", ".ps1", ".bat", ".cmd",
".html", ".htm", ".css", ".scss", ".less", ".sql", ".xml",
".cfg", ".conf", ".properties", ".env", ".gitignore", ".dockerfile", ".makefile",
".tex", ".rst", ".asciidoc", ".wiki", ".org",
}
if ext in text_exts:
return "text"
return "other"
def extract_text(path: str | Path, *, kind: Optional[str] = None) -> str:
"""Extract text content from a document file.
Tries the appropriate library for each file type. If the library isn't available
or extraction fails, returns an empty string.
Supported: PDF (.pdf), Word (.docx), PowerPoint (.pptx), Excel (.xlsx), Markdown (.md)
Legacy formats (.doc, .ppt, .xls) are not supported (require external tools).
"""
p = Path(path)
if not p.is_file():
return ""
if kind is None:
kind = detect_file_kind(p)
try:
if kind == "pdf":
return _extract_pdf(p)
elif kind == "docx":
return _extract_docx(p)
elif kind == "pptx":
return _extract_pptx(p)
elif kind == "xlsx":
return _extract_xlsx(p)
elif kind == "md":
return _extract_md(p)
elif kind == "text":
# 通用纯文本文件(.txt, .json, .py, .yaml, .log 等)
return _extract_plain_text(p)
except Exception:
# Extraction failures are non-fatal — the file is still on disk for the
# model to handle via skills if needed.
pass
return ""
def _extract_plain_text(path: Path) -> str:
"""Extract text from any plain-text file."""
# 尝试多种编码UTF-8 优先
for enc in ("utf-8", "utf-8-sig", "gbk", "latin-1"):
try:
text = path.read_text(encoding=enc)
return _truncate(text)
except (UnicodeDecodeError, UnicodeError):
continue
return ""
def _extract_md(path: Path) -> str:
"""Extract text from a Markdown .md file."""
text = path.read_text(encoding="utf-8")
return _truncate(text)
def _truncate(text: str) -> str:
if len(text) <= MAX_EXTRACTED_CHARS:
return text
return text[:MAX_EXTRACTED_CHARS] + f"\n\n[... truncated, {len(text)} total chars]"
def _extract_pdf(path: Path) -> str:
"""Extract text from a PDF using pdfplumber (preferred) or PyPDF2 (fallback)."""
try:
import pdfplumber
out: list[str] = []
with pdfplumber.open(str(path)) as pdf:
for i, page in enumerate(pdf.pages):
t = page.extract_text() or ""
if t.strip():
out.append(f"--- Page {i + 1} ---\n{t}")
return _truncate("\n\n".join(out))
except ImportError:
pass
try:
from PyPDF2 import PdfReader
reader = PdfReader(str(path))
out: list[str] = []
for i, page in enumerate(reader.pages):
t = page.extract_text() or ""
if t.strip():
out.append(f"--- Page {i + 1} ---\n{t}")
return _truncate("\n\n".join(out))
except ImportError:
pass
return ""
def _extract_docx(path: Path) -> str:
"""Extract text from a Word .docx file using python-docx."""
try:
from docx import Document
doc = Document(str(path))
out: list[str] = []
# Paragraphs
for para in doc.paragraphs:
if para.text.strip():
out.append(para.text)
# Tables
for table_idx, table in enumerate(doc.tables):
out.append(f"\n--- Table {table_idx + 1} ---")
for row in table.rows:
cells = [cell.text.strip() for cell in row.cells]
out.append(" | ".join(cells))
return _truncate("\n".join(out))
except ImportError:
pass
return ""
def _extract_pptx(path: Path) -> str:
"""Extract text from a PowerPoint .pptx file using python-pptx."""
try:
from pptx import Presentation
prs = Presentation(str(path))
out: list[str] = []
for i, slide in enumerate(prs.slides):
slide_text: list[str] = []
for shape in slide.shapes:
if hasattr(shape, "text") and shape.text.strip():
slide_text.append(shape.text.strip())
# Also check tables
for shape in slide.shapes:
if shape.has_table:
for row in shape.table.rows:
cells = [cell.text.strip() for cell in row.cells]
slide_text.append(" | ".join(cells))
if slide_text:
out.append(f"--- Slide {i + 1} ---\n" + "\n".join(slide_text))
# Notes
for i, slide in enumerate(prs.slides):
if slide.has_notes_slide:
notes = slide.notes_slide.notes_text_frame.text.strip()
if notes:
out.append(f"--- Slide {i + 1} Notes ---\n{notes}")
return _truncate("\n\n".join(out))
except ImportError:
pass
return ""
def _extract_xlsx(path: Path) -> str:
"""Extract text from an Excel .xlsx file using openpyxl."""
try:
from openpyxl import load_workbook
wb = load_workbook(str(path), read_only=True, data_only=True)
out: list[str] = []
for sheet_name in wb.sheetnames:
ws = wb[sheet_name]
out.append(f"--- Sheet: {sheet_name} ---")
row_count = 0
for row in ws.iter_rows(values_only=True):
cells = [
str(cell) if cell is not None else ""
for cell in row
]
# Skip completely empty rows
if any(c.strip() for c in cells):
out.append(" | ".join(cells))
row_count += 1
if row_count > 500: # safety cap per sheet
out.append("[... more rows omitted]")
break
out.append("") # blank line between sheets
wb.close()
return _truncate("\n".join(out))
except ImportError:
pass
return ""
def upload_result_dict(
file_path: Path,
*,
original_name: str,
extracted_text: Optional[str] = None,
kind: Optional[str] = None,
) -> dict[str, Any]:
"""Build the result dict returned by the upload API.
The attachment shape uses `kind: "text"` so it's compatible with the existing
attachment pipeline — the model sees the extracted text inline. Additional
fields (`file_path`, `file_kind`, `file_size`) carry metadata the model or
skills can use to operate on the raw file.
"""
if kind is None:
kind = detect_file_kind(file_path)
size = file_path.stat().st_size if file_path.is_file() else 0
# Prefix the extracted text with file metadata so the model knows where the
# raw file lives and can use skills to operate on it further.
# Note: build_user_content() in attachments.py already adds its own
# "[Attached file: <name>]" prefix, so we start with the path/type info
# right after the filename to avoid double-wrapping.
text_body = extracted_text or ""
if text_body:
header_lines = [
"--- UPLOADED FILE INFO ---",
f"📎 File: {original_name} | Type: {kind} | Size: {size} bytes",
f"📁 ABSOLUTE PATH (USE THIS DIRECTLY — do NOT search for the file):",
f" {file_path}",
"--- END UPLOADED FILE INFO ---",
"",
"--- Extracted text content ---",
"",
]
text_body = "\n".join(header_lines) + text_body
result: dict[str, Any] = {
"kind": "text", # compatible with existing attachment pipeline
"name": original_name,
"mime": _mime_for_kind(kind),
"text": text_body,
"file_path": str(file_path),
"file_kind": kind,
"file_size": size,
}
return result
def _mime_for_kind(kind: str) -> str:
mapping = {
"pdf": "application/pdf",
"docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"doc": "application/msword",
"pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
"ppt": "application/vnd.ms-powerpoint",
"xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"xls": "application/vnd.ms-excel",
"md": "text/markdown",
}
return mapping.get(kind, "application/octet-stream")