feat: 优化输出产物存储

This commit is contained in:
hp0912 2026-07-28 08:17:44 +08:00
parent b3193cb0c7
commit 8b99623429
11 changed files with 112 additions and 57 deletions

View File

@ -15,7 +15,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
- 每次检查脚本返回 JSON;只有 `ok` 为 `true` 时才继续。`validate_document.py` 还必须返回 `status: valid`。
- 只在需要读取图片、截图或扫描页中的文字时调用 `ocr_document.py`。只使用 `pages[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。
- 远程地址只交给 `download_document.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
- 不覆盖用户提供的源文件。最终结果写入 `output/docx/`,中间产物写入 `tmp/docx/<任务名>/`。
- 不覆盖用户提供的源文件。Word 最终文件一律写入 `/usr/local/src/word/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/word/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。
## 脚本清单
@ -55,7 +55,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
调用 `scripts/download_document.py`:
```text
--url 'https://example.com/report.docx?signature=...' --output 'tmp/docx/<任务名>/source.docx'
--url 'https://example.com/report.docx?signature=...' --output '/usr/local/src/word/tmp/<任务名>/source.docx'
```
可选参数:
@ -123,7 +123,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
调用 `scripts/create_document.py`:
```text
--output 'output/docx/result.docx' --spec '<JSON对象>'
--output '/usr/local/src/word/result.docx' --spec '<JSON对象>'
```
内容较长时先把 JSON 写到任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。
@ -230,7 +230,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
调用 `scripts/edit_document.py`:
```text
--input 'source.docx' --output 'output/docx/edited.docx' --spec '<JSON对象>'
--input 'source.docx' --output '/usr/local/src/word/edited.docx' --spec '<JSON对象>'
```
JSON 顶层只有 `operations`。支持:
@ -260,7 +260,7 @@ JSON 顶层只有 `operations`。支持:
调用 `scripts/add_comment.py`:
```text
--input 'source.docx' --output 'output/docx/commented.docx' --find '费用上限' --comment '请确认该上限是否含税' --author '审阅人' --initials 'SR'
--input 'source.docx' --output '/usr/local/src/word/commented.docx' --find '费用上限' --comment '请确认该上限是否含税' --author '审阅人' --initials 'SR'
```
可选:
@ -276,7 +276,7 @@ JSON 顶层只有 `operations`。支持:
调用 `scripts/accept_changes.py`:
```text
--input 'redlined.docx' --output 'output/docx/clean.docx'
--input 'redlined.docx' --output '/usr/local/src/word/clean.docx'
```
脚本只执行固定的“接受全部修订”宏,不能运行用户提供的宏。必须检查:
@ -290,7 +290,7 @@ JSON 顶层只有 `operations`。支持:
调用 `scripts/convert_document.py`:
```text
--input 'legacy.doc' --output 'tmp/docx/task/source.docx'
--input 'legacy.doc' --output '/usr/local/src/word/tmp/task/source.docx'
```
支持:
@ -317,7 +317,7 @@ JSON 顶层只有 `operations`。支持:
调用 `scripts/validate_document.py`:
```text
--input 'output/docx/result.docx' --check-convert
--input '/usr/local/src/word/result.docx' --check-convert
```
必须满足:
@ -336,7 +336,7 @@ JSON 顶层只有 `operations`。支持:
调用 `scripts/render_document.py`:
```text
--input 'output/docx/result.docx' --output-dir 'tmp/docx/task/rendered'
--input '/usr/local/src/word/result.docx' --output-dir '/usr/local/src/word/tmp/task/rendered'
```
默认 150 DPI、单次最多 20 页。可传:

View File

@ -20,6 +20,7 @@ DOCX_INPUT_SUFFIXES = {".docx", ".dotx"}
WORD_INPUT_SUFFIXES = DOCX_INPUT_SUFFIXES | {".doc"}
DOCX_OUTPUT_SUFFIXES = {".docx", ".dotx"}
DOCUMENT_OUTPUT_SUFFIXES = DOCX_OUTPUT_SUFFIXES | {".pdf", ".txt", ".md"}
WORD_OUTPUT_ROOT = Path("/usr/local/src/word")
MAX_ARCHIVE_MEMBERS = 20_000
MAX_ARCHIVE_UNCOMPRESSED_BYTES = 512 * 1024 * 1024
MAX_MEMBER_BYTES = 128 * 1024 * 1024
@ -90,13 +91,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path:
return path
def ensure_output_path(path: Path) -> Path:
resolved = path.expanduser().resolve()
try:
resolved.relative_to(WORD_OUTPUT_ROOT)
except ValueError as exc:
raise ValueError(
f"Word 产物必须输出到 {WORD_OUTPUT_ROOT} 目录下:{resolved}"
) from exc
return resolved
def output_file(
value: str,
suffixes: Optional[set[str]] = None,
*,
overwrite: bool = False,
) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if suffixes is not None and path.suffix.lower() not in suffixes:
expected = "、".join(sorted(suffixes))
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
@ -109,7 +121,7 @@ def output_file(
def output_directory(value: str) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if path.exists() and not path.is_dir():
raise ValueError(f"输出路径不是目录:{path}")
path.mkdir(parents=True, exist_ok=True)

View File

@ -9,6 +9,7 @@ from typing import Any
from _docx_common import (
DOCX_INPUT_SUFFIXES,
SkillArgumentParser,
ensure_output_path,
input_file,
safe_extract_docx,
run_cli,
@ -27,7 +28,7 @@ def build_parser() -> argparse.ArgumentParser:
def main() -> dict[str, Any]:
args = build_parser().parse_args()
source = input_file(args.input, DOCX_INPUT_SUFFIXES)
destination = Path(args.output_dir).expanduser().resolve()
destination = ensure_output_path(Path(args.output_dir))
if destination.exists():
if not destination.is_dir():
raise ValueError(f"输出路径不是目录:{destination}")

View File

@ -16,6 +16,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。
- 收到 `ok: false` 时,依据 `error` 调整合法参数或向用户说明失败原因,不要把参数改传给其他脚本碰运气。
- 阅读或总结时只使用 `pages[]` 中 `usable_for_summary: true` 的文本。`needs_ocr: false` 时不得为了“常规检查”继续 OCR、渲染或调用图片识别。
- PDF 最终文件一律写入 `/usr/local/src/pdf/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/pdf/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
- 不把 PDF 密码作为脚本参数;工具调用参数可能进入运行日志。
## 脚本清单
@ -36,14 +37,14 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
## 标准流程
1. 为任务选择简短目录名,把中间文件放在 `tmp/pdfs/<任务名>/`。
1. 为任务选择简短目录名,把中间文件放在 `/usr/local/src/pdf/tmp/<任务名>/`。
2. 远程 HTTPS 链接先调用 `download_pdf.py`;本地文件直接进入下一步。
3. 调用 `inspect_pdf.py` 检查文件。遇到加密 PDF 时停止处理,请用户提供已解密副本;当前固定脚本不接收密码。
4. 阅读或总结时调用 `extract_text.py`。结果为 `usable_for_summary: true` 时使用可靠页文本并根据游标继续;同时为 `needs_ocr: false` 时直接回答,不调用 OCR、渲染或图片识别。
5. 只有 `extract_text.py` 返回 `needs_ocr: true` 时,才对 `text_quality.suspect_pages` 调用 `ocr_text.py`。原生可靠文本优先,OCR 只补齐可疑页,不重复识别正常页。
6. `ocr_text.py` 会在脚本内部临时渲染指定页面并交给本地 RapidOCR,完成后自动删除 PNG;普通扫描件解析不调用大模型识图,也不需要先调用 `render_pdf.py`。
7. 仅在用户明确要求检查视觉版式,或任务涉及创建/修改 PDF 时调用 `render_pdf.py`。
8. 创建或修改后的最终 PDF 写入 `output/pdf/`,重新执行检查、文本提取和全部页面渲染。
8. 创建或修改后的最终 PDF 写入 `/usr/local/src/pdf/`,重新执行检查、文本提取和全部页面渲染。
9. 最终产物位于临时目录之外且不再需要缓存时,调用 `cleanup_pdf_temp.py` 清理本次任务目录。
## 下载远程 PDF
@ -53,7 +54,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
调用 `scripts/download_pdf.py`:
```text
--url 'https://example.com/document.pdf' --output 'tmp/pdfs/<任务名>/source.pdf'
--url 'https://example.com/document.pdf' --output '/usr/local/src/pdf/tmp/<任务名>/source.pdf'
```
可选参数:
@ -69,7 +70,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
调用 `scripts/inspect_pdf.py`:
```text
--input 'tmp/pdfs/<任务名>/source.pdf'
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf'
```
使用返回的 `page_count`、`encrypted`、`metadata`、`page_layouts` 和 `form_field_count` 判断后续处理方式。不要直接调用 `pdfinfo`。
@ -79,7 +80,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
首次调用 `scripts/extract_text.py`:
```text
--input 'tmp/pdfs/<任务名>/source.pdf'
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf'
```
默认使用 `auto` 引擎:先由 Poppler `pdftotext` 提取;结果不可用或命令不可用时自动尝试 `pdfplumber`,并可逐页选择质量更好的结果。脚本使用 `pypdf` 获取标准页数,并拒绝把页数不一致的提取结果当作成功。不要直接执行 `pdftotext`。
@ -106,7 +107,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
仅当 `extract_text.py` 返回 `needs_ocr: true` 时调用 `scripts/ocr_text.py`。`--pages` 必须明确指定 `text_quality.suspect_pages` 中要读取的页,单次最多 4 页:
```text
--input 'tmp/pdfs/<任务名>/source.pdf' --pages '2,5-6'
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --pages '2,5-6'
```
默认以 260 DPI 临时渲染,并使用镜像中预置的 RapidOCR 与 ONNX Runtime 在本地识别。脚本不会联网下载模型,不会保留渲染图片,也不会调用大模型视觉能力。可选参数:
@ -131,7 +132,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
调用 `scripts/extract_tables.py`:
```text
--input 'tmp/pdfs/<任务名>/source.pdf' --start-page 1
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --start-page 1
```
默认单次最多处理 5 页、20 个表格和 2000 个单元格。可用 `--end-page`、`--start-table`、`--max-pages`、`--max-tables`、`--max-cells` 调整。若 `has_more: true`,把 `next_page` 传给 `--start-page`、`next_table` 传给 `--start-table` 后继续,并保留首次调用的 `--end-page`(如果指定)及其他提取选项。
@ -146,7 +147,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
不要因为输入是 PDF、需要总结、需要 OCR 或需要检查首页就自动调用本脚本;OCR 的临时渲染由 `ocr_text.py` 内部完成。调用脚本时不要直接执行 `pdftoppm`:
```text
--input 'tmp/pdfs/<任务名>/source.pdf' --output-dir 'tmp/pdfs/<任务名>/rendered' --start-page 1
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --output-dir '/usr/local/src/pdf/tmp/<任务名>/rendered' --start-page 1
```
默认 150 DPI、单次最多 10 页。可使用 `--end-page`、`--max-pages`、`--dpi`、`--timeout` 和 `--overwrite`。若 `has_more: true`,使用 `next_page` 继续,并保留首次调用的 `--end-page`(如果指定)、输出目录及其他渲染选项。脚本返回标准化的 `page-0001.png` 文件路径。
@ -158,7 +159,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
先使用 `write_file` 把内容写为 UTF-8 `.txt` 或 `.md` 文件,再调用 `scripts/create_pdf.py`:
```text
--input 'tmp/pdfs/<任务名>/content.md' --output 'output/pdf/<文件名>.pdf' --title '文档标题'
--input '/usr/local/src/pdf/tmp/<任务名>/content.md' --output '/usr/local/src/pdf/<文件名>.pdf' --title '文档标题'
```
脚本支持 Markdown 标题、项目符号和简单表格,自动选择可嵌入的 Unicode 字体并添加页码。可选参数:
@ -178,13 +179,13 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
合并:
```text
merge --input 'a.pdf' --input 'b.pdf' --output 'output/pdf/merged.pdf'
merge --input 'a.pdf' --input 'b.pdf' --output '/usr/local/src/pdf/merged.pdf'
```
拆分指定范围:
```text
split --input 'source.pdf' --output-dir 'output/pdf/split' --range 1-3 --range 4-6
split --input 'source.pdf' --output-dir '/usr/local/src/pdf/split' --range 1-3 --range 4-6
```
不传 `--range` 时每页生成一个 PDF。
@ -192,7 +193,7 @@ split --input 'source.pdf' --output-dir 'output/pdf/split' --range 1-3 --range 4
旋转指定页面:
```text
rotate --input 'source.pdf' --output 'output/pdf/rotated.pdf' --pages '1,3-5' --degrees 90
rotate --input 'source.pdf' --output '/usr/local/src/pdf/rotated.pdf' --pages '1,3-5' --degrees 90
```
`--degrees` 只能是 `90`、`180` 或 `270`;不传 `--pages` 时旋转全部页面。目标已存在且确认可覆盖时添加 `--overwrite`。
@ -202,10 +203,10 @@ rotate --input 'source.pdf' --output 'output/pdf/rotated.pdf' --pages '1,3-5' --
调用 `scripts/cleanup_pdf_temp.py`:
```text
--task-dir 'tmp/pdfs/<任务名>'
--task-dir '/usr/local/src/pdf/tmp/<任务名>'
```
脚本只允许删除本 Skill 的 `tmp/pdfs/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。
脚本只允许删除 `/usr/local/src/pdf/tmp/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。
## 质量要求

View File

@ -11,6 +11,9 @@ from pathlib import Path
from typing import Any, Callable, NoReturn, Optional
PDF_OUTPUT_ROOT = Path("/usr/local/src/pdf")
def quiet_pdf_library_logs() -> None:
"""Keep third-party recovery warnings out of the JSON tool response."""
logging.getLogger("pdfminer").setLevel(logging.ERROR)
@ -62,8 +65,19 @@ def input_pdf(value: str) -> Path:
return path
def ensure_output_path(path: Path) -> Path:
resolved = path.expanduser().resolve()
try:
resolved.relative_to(PDF_OUTPUT_ROOT)
except ValueError as exc:
raise ValueError(
f"PDF 产物必须输出到 {PDF_OUTPUT_ROOT} 目录下:{resolved}"
) from exc
return resolved
def output_pdf(value: str, overwrite: bool) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if path.suffix.lower() != ".pdf":
raise ValueError("PDF 输出路径必须以 .pdf 结尾")
if path.exists() and not overwrite:
@ -73,7 +87,7 @@ def output_pdf(value: str, overwrite: bool) -> Path:
def output_directory(value: str) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if path.exists() and not path.is_dir():
raise ValueError(f"输出路径不是目录:{path}")
path.mkdir(parents=True, exist_ok=True)

View File

@ -7,7 +7,7 @@ import sys
from pathlib import Path
from typing import Any
from _pdf_common import SkillArgumentParser, run_cli
from _pdf_common import PDF_OUTPUT_ROOT, SkillArgumentParser, run_cli
def _parse_args(argv: list[str]):
@ -15,28 +15,29 @@ def _parse_args(argv: list[str]):
parser.add_argument(
"--task-dir",
required=True,
help="仅允许删除本 Skill 的 tmp/pdfs/ 下某个具体任务目录",
help=f"仅允许删除 {PDF_OUTPUT_ROOT}/tmp/ 下某个具体任务目录",
)
return parser.parse_args(argv)
def _cleanup(value: str) -> dict[str, Any]:
skill_root = Path(__file__).resolve().parents[1]
allowed_root = (skill_root / "tmp" / "pdfs").resolve()
allowed_root = (PDF_OUTPUT_ROOT / "tmp").resolve()
candidate = Path(value).expanduser()
target = (
candidate.resolve()
if candidate.is_absolute()
else (skill_root / candidate).resolve()
else (allowed_root / candidate).resolve()
)
try:
relative = target.relative_to(allowed_root)
except ValueError as exc:
raise ValueError(f"只能清理 {allowed_root} 下的任务目录") from exc
if not relative.parts:
raise ValueError("不能删除 tmp/pdfs 根目录")
raise ValueError(f"不能删除 {allowed_root} 根目录")
if len(relative.parts) != 1:
raise ValueError("task-dir 必须直接指向 tmp/pdfs 下的单个任务目录")
raise ValueError(
f"task-dir 必须直接指向 {allowed_root} 下的单个任务目录"
)
if not target.is_dir():
raise FileNotFoundError(f"任务临时目录不存在:{target}")
shutil.rmtree(target)

View File

@ -15,6 +15,8 @@ import urllib.request
from pathlib import Path
from typing import NoReturn, Optional
from _pdf_common import ensure_output_path
DEFAULT_TIMEOUT_SECONDS = 60
DEFAULT_MAX_BYTES = 100 * 1024 * 1024
CHUNK_SIZE = 1024 * 1024
@ -100,7 +102,7 @@ def _parse_args(argv: list[str]) -> argparse.Namespace:
output = Path(args.output).expanduser()
if output.suffix.lower() != ".pdf":
raise ValueError("output 必须以 .pdf 结尾")
args.output = output.resolve()
args.output = ensure_output_path(output)
return args

View File

@ -14,7 +14,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染
- PptxGenJS、React Icons、Sharp、LibreOffice、Poppler 和 ZIP 操作只允许由固定脚本在内部调用。
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`validate_presentation.py` 还必须返回 `status: valid`、`issue_count: 0`。
- 只在需要读取图片、截图或视觉图表中的文字时调用 `ocr_presentation.py`。只使用 `slides[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。
- 不覆盖用户提供的源文件。最终结果写入 `output/pptx/`,中间产物写入 `tmp/pptx/<任务名>/`。
- 不覆盖用户提供的源文件。PPT 最终文件一律写入 `/usr/local/src/ppt/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/ppt/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
- 远程地址只交给 `download_presentation.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
- 环境已预置全部依赖,不安装软件包,也不提示用户安装依赖。
@ -54,7 +54,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染
只接受 HTTPS 地址。完整保留 URL 及查询参数传给脚本,但不要在回复或输出文件名中暴露查询参数。
```text
--url 'https://example.com/deck.pptx?signature=...' --output 'tmp/pptx/<任务名>/source.pptx'
--url 'https://example.com/deck.pptx?signature=...' --output '/usr/local/src/ppt/tmp/<任务名>/source.pptx'
```
可选参数:
@ -84,7 +84,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染
需要连续正文时调用:
```text
--input 'source.pptx' --output 'tmp/pptx/<任务名>/content.md'
--input 'source.pptx' --output '/usr/local/src/ppt/tmp/<任务名>/content.md'
```
Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
@ -114,7 +114,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
调用:
```text
--output 'output/pptx/result.pptx' --spec '<JSON对象>'
--output '/usr/local/src/ppt/result.pptx' --spec '<JSON对象>'
```
内容较长时先把 JSON 写入任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。
@ -236,7 +236,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
需要图标时先调用:
```text
--library fi --name FiTrendingUp --color 2563EB --size 256 --output 'tmp/pptx/<任务名>/trend.png'
--library fi --name FiTrendingUp --color 2563EB --size 256 --output '/usr/local/src/ppt/tmp/<任务名>/trend.png'
```
允许的图标库:`fa6`、`fi`、`hi2`、`io5`、`lu`、`md`、`ri`、`tb`。把生成的 PNG 作为普通图片元素插入。
@ -302,7 +302,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
调用:
```text
--input 'source.pptx' --output 'output/pptx/edited.pptx' --spec '<JSON对象>'
--input 'source.pptx' --output '/usr/local/src/ppt/edited.pptx' --spec '<JSON对象>'
```
JSON 顶层只有 `operations`,按数组顺序执行:
@ -341,7 +341,7 @@ JSON 顶层只有 `operations`,按数组顺序执行:
只对 `.pptx` 使用:
```text
--input 'template.pptx' --output 'tmp/pptx/<任务名>/expanded.pptx' --slide 2 --after 4
--input 'template.pptx' --output '/usr/local/src/ppt/tmp/<任务名>/expanded.pptx' --slide 2 --after 4
```
`slide` 是复制来源,`after` 是插入位置;省略 `after` 时紧跟来源页插入。脚本会更新页面清单、关系和内容类型,并移除不能安全共享的备注/批注关系。
@ -381,13 +381,13 @@ JSON 顶层只有 `operations`,按数组顺序执行:
结构校验:
```text
--input 'output/pptx/result.pptx' --check-render
--input '/usr/local/src/ppt/result.pptx' --check-render
```
模板派生结果:
```text
--input 'output/pptx/result.pptx' --original 'template.pptx' --check-render
--input '/usr/local/src/ppt/result.pptx' --original 'template.pptx' --check-render
```
必须满足:
@ -402,7 +402,7 @@ JSON 顶层只有 `operations`,按数组顺序执行:
渲染全部页面:
```text
--input 'output/pptx/result.pptx' --output-dir 'tmp/pptx/<任务名>/rendered' --contact-sheet --include-pdf
--input '/usr/local/src/ppt/result.pptx' --output-dir '/usr/local/src/ppt/tmp/<任务名>/rendered' --contact-sheet --include-pdf
```
默认 150 DPI、单次最多 30 页。可用 `--start-slide/--end-slide/--max-slides` 分批,复杂图表或小字可把 `--dpi` 提高到 180–220。联系表用于快速检查整体节奏,逐页 PNG 用于最终 QA。

View File

@ -21,6 +21,7 @@ OOXML_PRESENTATION_SUFFIXES = {".pptx", ".potx", ".ppsx"}
PRESENTATION_INPUT_SUFFIXES = OOXML_PRESENTATION_SUFFIXES | {".ppt"}
PRESENTATION_OUTPUT_SUFFIXES = {".pptx", ".potx", ".pdf"}
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp"}
PPT_OUTPUT_ROOT = Path("/usr/local/src/ppt")
MAX_ARCHIVE_MEMBERS = 20_000
MAX_ARCHIVE_UNCOMPRESSED_BYTES = 1_073_741_824
@ -92,13 +93,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path:
return path
def ensure_output_path(path: Path) -> Path:
resolved = path.expanduser().resolve()
try:
resolved.relative_to(PPT_OUTPUT_ROOT)
except ValueError as exc:
raise ValueError(
f"PPT 产物必须输出到 {PPT_OUTPUT_ROOT} 目录下:{resolved}"
) from exc
return resolved
def output_file(
value: str,
suffixes: Optional[set[str]] = None,
*,
overwrite: bool = False,
) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if suffixes is not None and path.suffix.lower() not in suffixes:
expected = "、".join(sorted(suffixes))
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
@ -111,7 +123,7 @@ def output_file(
def output_directory(value: str) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if path.exists() and not path.is_dir():
raise ValueError(f"输出路径不是目录:{path}")
path.mkdir(parents=True, exist_ok=True)

View File

@ -14,7 +14,7 @@ description: "创建、读取、编辑、修复、转换、重算、校验和渲
- 外部程序只允许由固定脚本在内部以无 shell 参数数组方式调用。
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`status: errors_found` 虽然表示脚本成功运行,但工作簿不合格,必须修复。
- 远程地址只交给 `download_workbook.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
- 不覆盖用户提供的源文件。创建或编辑结果写入 `output/xlsx/`,中间产物写入 `tmp/xlsx/<任务名>/`。
- 不覆盖用户提供的源文件。Excel 最终文件一律写入 `/usr/local/src/excel/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/excel/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。
## 脚本清单
@ -46,7 +46,7 @@ description: "创建、读取、编辑、修复、转换、重算、校验和渲
调用 `scripts/download_workbook.py`:
```text
--url 'https://example.com/report.xlsx?signature=...' --output 'tmp/xlsx/<任务名>/source.xlsx'
--url 'https://example.com/report.xlsx?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/source.xlsx'
```
可选参数:
@ -87,13 +87,13 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
调用 `scripts/apply_workbook.py`,新建时省略 `--input`,编辑时提供源文件:
```text
--output 'output/xlsx/result.xlsx' --spec '<JSON对象>'
--output '/usr/local/src/excel/result.xlsx' --spec '<JSON对象>'
```
或:
```text
--input 'source.xlsx' --output 'output/xlsx/result.xlsx' --spec-file 'tmp/xlsx/task/operations.json'
--input 'source.xlsx' --output '/usr/local/src/excel/result.xlsx' --spec-file '/usr/local/src/excel/tmp/task/operations.json'
```
目标已存在且确认是本次任务的旧产物时才传 `--overwrite`。输入含外部链接时脚本默认拒绝保存;只有用户明确接受缓存值可能丢失的风险时才传 `--allow-external-links`。`.xlsm` 必须继续输出 `.xlsm` 才能保留宏;只有用户明确同意丢弃宏时才输出 `.xlsx` 并传 `--drop-macros`。
@ -167,7 +167,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
调用 `scripts/convert_workbook.py`:
```text
--input 'legacy.xls' --output 'tmp/xlsx/task/source.xlsx'
--input 'legacy.xls' --output '/usr/local/src/excel/tmp/task/source.xlsx'
```
常见用法:
@ -182,7 +182,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
含公式的工作簿必须调用 `scripts/recalculate_workbook.py`:
```text
--input 'output/xlsx/result.xlsx' --output 'output/xlsx/result-recalculated.xlsx'
--input '/usr/local/src/excel/result.xlsx' --output '/usr/local/src/excel/result-recalculated.xlsx'
```
检查返回值:
@ -198,7 +198,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
调用 `scripts/render_workbook.py`:
```text
--input 'output/xlsx/result-recalculated.xlsx' --output-dir 'tmp/xlsx/task/rendered'
--input '/usr/local/src/excel/result-recalculated.xlsx' --output-dir '/usr/local/src/excel/tmp/task/rendered'
```
默认 150 DPI、单次最多 20 页。可传:

View File

@ -19,6 +19,7 @@ from urllib.parse import quote
EXCEL_INPUT_SUFFIXES = {".xlsx", ".xlsm", ".xltx", ".xltm"}
TABULAR_INPUT_SUFFIXES = EXCEL_INPUT_SUFFIXES | {".xls", ".csv", ".tsv"}
EXCEL_OUTPUT_SUFFIXES = {".xlsx", ".xlsm"}
EXCEL_OUTPUT_ROOT = Path("/usr/local/src/excel")
FORMULA_ERROR_VALUES = {
"#NULL!",
"#DIV/0!",
@ -101,13 +102,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path:
return path
def ensure_output_path(path: Path) -> Path:
resolved = path.expanduser().resolve()
try:
resolved.relative_to(EXCEL_OUTPUT_ROOT)
except ValueError as exc:
raise ValueError(
f"Excel 产物必须输出到 {EXCEL_OUTPUT_ROOT} 目录下:{resolved}"
) from exc
return resolved
def output_file(
value: str,
suffixes: Optional[set[str]] = None,
*,
overwrite: bool = False,
) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if suffixes is not None and path.suffix.lower() not in suffixes:
expected = "、".join(sorted(suffixes))
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
@ -120,7 +132,7 @@ def output_file(
def output_directory(value: str) -> Path:
path = Path(value).expanduser().resolve()
path = ensure_output_path(Path(value))
if path.exists() and not path.is_dir():
raise ValueError(f"输出路径不是目录:{path}")
path.mkdir(parents=True, exist_ok=True)