From 8b9962342942f60da9ea8f8a2cfc3612f4b5fe48 Mon Sep 17 00:00:00 2001 From: hp0912 <809211365@qq.com> Date: Tue, 28 Jul 2026 08:17:44 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E4=BC=98=E5=8C=96=E8=BE=93=E5=87=BA?= =?UTF-8?q?=E4=BA=A7=E7=89=A9=E5=AD=98=E5=82=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- skills/docx/SKILL.md | 18 ++++++++-------- skills/docx/scripts/_docx_common.py | 16 ++++++++++++-- skills/docx/scripts/unpack_document.py | 3 ++- skills/pdf/SKILL.md | 29 +++++++++++++------------- skills/pdf/scripts/_pdf_common.py | 18 ++++++++++++++-- skills/pdf/scripts/cleanup_pdf_temp.py | 15 ++++++------- skills/pdf/scripts/download_pdf.py | 4 +++- skills/pptx/SKILL.md | 20 +++++++++--------- skills/pptx/scripts/_pptx_common.py | 16 ++++++++++++-- skills/xlsx/SKILL.md | 14 ++++++------- skills/xlsx/scripts/_xlsx_common.py | 16 ++++++++++++-- 11 files changed, 112 insertions(+), 57 deletions(-) diff --git a/skills/docx/SKILL.md b/skills/docx/SKILL.md index 534df2a..49f0b63 100644 --- a/skills/docx/SKILL.md +++ b/skills/docx/SKILL.md @@ -15,7 +15,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验 - 每次检查脚本返回 JSON;只有 `ok` 为 `true` 时才继续。`validate_document.py` 还必须返回 `status: valid`。 - 只在需要读取图片、截图或扫描页中的文字时调用 `ocr_document.py`。只使用 `pages[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。 - 远程地址只交给 `download_document.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。 -- 不覆盖用户提供的源文件。最终结果写入 `output/docx/`,中间产物写入 `tmp/docx/<任务名>/`。 +- 不覆盖用户提供的源文件。Word 最终文件一律写入 `/usr/local/src/word/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/word/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。 - 环境已预置依赖,不安装软件包,也不提示用户安装依赖。 ## 脚本清单 @@ -55,7 +55,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验 调用 `scripts/download_document.py`: ```text ---url 'https://example.com/report.docx?signature=...' --output 'tmp/docx/<任务名>/source.docx' +--url 'https://example.com/report.docx?signature=...' --output '/usr/local/src/word/tmp/<任务名>/source.docx' ``` 可选参数: @@ -123,7 +123,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验 调用 `scripts/create_document.py`: ```text ---output 'output/docx/result.docx' --spec '' +--output '/usr/local/src/word/result.docx' --spec '' ``` 内容较长时先把 JSON 写到任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。 @@ -230,7 +230,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验 调用 `scripts/edit_document.py`: ```text ---input 'source.docx' --output 'output/docx/edited.docx' --spec '' +--input 'source.docx' --output '/usr/local/src/word/edited.docx' --spec '' ``` JSON 顶层只有 `operations`。支持: @@ -260,7 +260,7 @@ JSON 顶层只有 `operations`。支持: 调用 `scripts/add_comment.py`: ```text ---input 'source.docx' --output 'output/docx/commented.docx' --find '费用上限' --comment '请确认该上限是否含税' --author '审阅人' --initials 'SR' +--input 'source.docx' --output '/usr/local/src/word/commented.docx' --find '费用上限' --comment '请确认该上限是否含税' --author '审阅人' --initials 'SR' ``` 可选: @@ -276,7 +276,7 @@ JSON 顶层只有 `operations`。支持: 调用 `scripts/accept_changes.py`: ```text ---input 'redlined.docx' --output 'output/docx/clean.docx' +--input 'redlined.docx' --output '/usr/local/src/word/clean.docx' ``` 脚本只执行固定的“接受全部修订”宏,不能运行用户提供的宏。必须检查: @@ -290,7 +290,7 @@ JSON 顶层只有 `operations`。支持: 调用 `scripts/convert_document.py`: ```text ---input 'legacy.doc' --output 'tmp/docx/task/source.docx' +--input 'legacy.doc' --output '/usr/local/src/word/tmp/task/source.docx' ``` 支持: @@ -317,7 +317,7 @@ JSON 顶层只有 `operations`。支持: 调用 `scripts/validate_document.py`: ```text ---input 'output/docx/result.docx' --check-convert +--input '/usr/local/src/word/result.docx' --check-convert ``` 必须满足: @@ -336,7 +336,7 @@ JSON 顶层只有 `operations`。支持: 调用 `scripts/render_document.py`: ```text ---input 'output/docx/result.docx' --output-dir 'tmp/docx/task/rendered' +--input '/usr/local/src/word/result.docx' --output-dir '/usr/local/src/word/tmp/task/rendered' ``` 默认 150 DPI、单次最多 20 页。可传: diff --git a/skills/docx/scripts/_docx_common.py b/skills/docx/scripts/_docx_common.py index f1982df..2e49a2c 100644 --- a/skills/docx/scripts/_docx_common.py +++ b/skills/docx/scripts/_docx_common.py @@ -20,6 +20,7 @@ DOCX_INPUT_SUFFIXES = {".docx", ".dotx"} WORD_INPUT_SUFFIXES = DOCX_INPUT_SUFFIXES | {".doc"} DOCX_OUTPUT_SUFFIXES = {".docx", ".dotx"} DOCUMENT_OUTPUT_SUFFIXES = DOCX_OUTPUT_SUFFIXES | {".pdf", ".txt", ".md"} +WORD_OUTPUT_ROOT = Path("/usr/local/src/word") MAX_ARCHIVE_MEMBERS = 20_000 MAX_ARCHIVE_UNCOMPRESSED_BYTES = 512 * 1024 * 1024 MAX_MEMBER_BYTES = 128 * 1024 * 1024 @@ -90,13 +91,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path: return path +def ensure_output_path(path: Path) -> Path: + resolved = path.expanduser().resolve() + try: + resolved.relative_to(WORD_OUTPUT_ROOT) + except ValueError as exc: + raise ValueError( + f"Word 产物必须输出到 {WORD_OUTPUT_ROOT} 目录下:{resolved}" + ) from exc + return resolved + + def output_file( value: str, suffixes: Optional[set[str]] = None, *, overwrite: bool = False, ) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if suffixes is not None and path.suffix.lower() not in suffixes: expected = "、".join(sorted(suffixes)) raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}") @@ -109,7 +121,7 @@ def output_file( def output_directory(value: str) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if path.exists() and not path.is_dir(): raise ValueError(f"输出路径不是目录:{path}") path.mkdir(parents=True, exist_ok=True) diff --git a/skills/docx/scripts/unpack_document.py b/skills/docx/scripts/unpack_document.py index 90e810c..9fc261a 100644 --- a/skills/docx/scripts/unpack_document.py +++ b/skills/docx/scripts/unpack_document.py @@ -9,6 +9,7 @@ from typing import Any from _docx_common import ( DOCX_INPUT_SUFFIXES, SkillArgumentParser, + ensure_output_path, input_file, safe_extract_docx, run_cli, @@ -27,7 +28,7 @@ def build_parser() -> argparse.ArgumentParser: def main() -> dict[str, Any]: args = build_parser().parse_args() source = input_file(args.input, DOCX_INPUT_SUFFIXES) - destination = Path(args.output_dir).expanduser().resolve() + destination = ensure_output_path(Path(args.output_dir)) if destination.exists(): if not destination.is_dir(): raise ValueError(f"输出路径不是目录:{destination}") diff --git a/skills/pdf/SKILL.md b/skills/pdf/SKILL.md index 13f5d05..d852e88 100644 --- a/skills/pdf/SKILL.md +++ b/skills/pdf/SKILL.md @@ -16,6 +16,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全 - 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。 - 收到 `ok: false` 时,依据 `error` 调整合法参数或向用户说明失败原因,不要把参数改传给其他脚本碰运气。 - 阅读或总结时只使用 `pages[]` 中 `usable_for_summary: true` 的文本。`needs_ocr: false` 时不得为了“常规检查”继续 OCR、渲染或调用图片识别。 +- PDF 最终文件一律写入 `/usr/local/src/pdf/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/pdf/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。 - 不把 PDF 密码作为脚本参数;工具调用参数可能进入运行日志。 ## 脚本清单 @@ -36,14 +37,14 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全 ## 标准流程 -1. 为任务选择简短目录名,把中间文件放在 `tmp/pdfs/<任务名>/`。 +1. 为任务选择简短目录名,把中间文件放在 `/usr/local/src/pdf/tmp/<任务名>/`。 2. 远程 HTTPS 链接先调用 `download_pdf.py`;本地文件直接进入下一步。 3. 调用 `inspect_pdf.py` 检查文件。遇到加密 PDF 时停止处理,请用户提供已解密副本;当前固定脚本不接收密码。 4. 阅读或总结时调用 `extract_text.py`。结果为 `usable_for_summary: true` 时使用可靠页文本并根据游标继续;同时为 `needs_ocr: false` 时直接回答,不调用 OCR、渲染或图片识别。 5. 只有 `extract_text.py` 返回 `needs_ocr: true` 时,才对 `text_quality.suspect_pages` 调用 `ocr_text.py`。原生可靠文本优先,OCR 只补齐可疑页,不重复识别正常页。 6. `ocr_text.py` 会在脚本内部临时渲染指定页面并交给本地 RapidOCR,完成后自动删除 PNG;普通扫描件解析不调用大模型识图,也不需要先调用 `render_pdf.py`。 7. 仅在用户明确要求检查视觉版式,或任务涉及创建/修改 PDF 时调用 `render_pdf.py`。 -8. 创建或修改后的最终 PDF 写入 `output/pdf/`,重新执行检查、文本提取和全部页面渲染。 +8. 创建或修改后的最终 PDF 写入 `/usr/local/src/pdf/`,重新执行检查、文本提取和全部页面渲染。 9. 最终产物位于临时目录之外且不再需要缓存时,调用 `cleanup_pdf_temp.py` 清理本次任务目录。 ## 下载远程 PDF @@ -53,7 +54,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全 调用 `scripts/download_pdf.py`: ```text ---url 'https://example.com/document.pdf' --output 'tmp/pdfs/<任务名>/source.pdf' +--url 'https://example.com/document.pdf' --output '/usr/local/src/pdf/tmp/<任务名>/source.pdf' ``` 可选参数: @@ -69,7 +70,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全 调用 `scripts/inspect_pdf.py`: ```text ---input 'tmp/pdfs/<任务名>/source.pdf' +--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' ``` 使用返回的 `page_count`、`encrypted`、`metadata`、`page_layouts` 和 `form_field_count` 判断后续处理方式。不要直接调用 `pdfinfo`。 @@ -79,7 +80,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全 首次调用 `scripts/extract_text.py`: ```text ---input 'tmp/pdfs/<任务名>/source.pdf' +--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' ``` 默认使用 `auto` 引擎:先由 Poppler `pdftotext` 提取;结果不可用或命令不可用时自动尝试 `pdfplumber`,并可逐页选择质量更好的结果。脚本使用 `pypdf` 获取标准页数,并拒绝把页数不一致的提取结果当作成功。不要直接执行 `pdftotext`。 @@ -106,7 +107,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全 仅当 `extract_text.py` 返回 `needs_ocr: true` 时调用 `scripts/ocr_text.py`。`--pages` 必须明确指定 `text_quality.suspect_pages` 中要读取的页,单次最多 4 页: ```text ---input 'tmp/pdfs/<任务名>/source.pdf' --pages '2,5-6' +--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --pages '2,5-6' ``` 默认以 260 DPI 临时渲染,并使用镜像中预置的 RapidOCR 与 ONNX Runtime 在本地识别。脚本不会联网下载模型,不会保留渲染图片,也不会调用大模型视觉能力。可选参数: @@ -131,7 +132,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se 调用 `scripts/extract_tables.py`: ```text ---input 'tmp/pdfs/<任务名>/source.pdf' --start-page 1 +--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --start-page 1 ``` 默认单次最多处理 5 页、20 个表格和 2000 个单元格。可用 `--end-page`、`--start-table`、`--max-pages`、`--max-tables`、`--max-cells` 调整。若 `has_more: true`,把 `next_page` 传给 `--start-page`、`next_table` 传给 `--start-table` 后继续,并保留首次调用的 `--end-page`(如果指定)及其他提取选项。 @@ -146,7 +147,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se 不要因为输入是 PDF、需要总结、需要 OCR 或需要检查首页就自动调用本脚本;OCR 的临时渲染由 `ocr_text.py` 内部完成。调用脚本时不要直接执行 `pdftoppm`: ```text ---input 'tmp/pdfs/<任务名>/source.pdf' --output-dir 'tmp/pdfs/<任务名>/rendered' --start-page 1 +--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --output-dir '/usr/local/src/pdf/tmp/<任务名>/rendered' --start-page 1 ``` 默认 150 DPI、单次最多 10 页。可使用 `--end-page`、`--max-pages`、`--dpi`、`--timeout` 和 `--overwrite`。若 `has_more: true`,使用 `next_page` 继续,并保留首次调用的 `--end-page`(如果指定)、输出目录及其他渲染选项。脚本返回标准化的 `page-0001.png` 文件路径。 @@ -158,7 +159,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se 先使用 `write_file` 把内容写为 UTF-8 `.txt` 或 `.md` 文件,再调用 `scripts/create_pdf.py`: ```text ---input 'tmp/pdfs/<任务名>/content.md' --output 'output/pdf/<文件名>.pdf' --title '文档标题' +--input '/usr/local/src/pdf/tmp/<任务名>/content.md' --output '/usr/local/src/pdf/<文件名>.pdf' --title '文档标题' ``` 脚本支持 Markdown 标题、项目符号和简单表格,自动选择可嵌入的 Unicode 字体并添加页码。可选参数: @@ -178,13 +179,13 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se 合并: ```text -merge --input 'a.pdf' --input 'b.pdf' --output 'output/pdf/merged.pdf' +merge --input 'a.pdf' --input 'b.pdf' --output '/usr/local/src/pdf/merged.pdf' ``` 拆分指定范围: ```text -split --input 'source.pdf' --output-dir 'output/pdf/split' --range 1-3 --range 4-6 +split --input 'source.pdf' --output-dir '/usr/local/src/pdf/split' --range 1-3 --range 4-6 ``` 不传 `--range` 时每页生成一个 PDF。 @@ -192,7 +193,7 @@ split --input 'source.pdf' --output-dir 'output/pdf/split' --range 1-3 --range 4 旋转指定页面: ```text -rotate --input 'source.pdf' --output 'output/pdf/rotated.pdf' --pages '1,3-5' --degrees 90 +rotate --input 'source.pdf' --output '/usr/local/src/pdf/rotated.pdf' --pages '1,3-5' --degrees 90 ``` `--degrees` 只能是 `90`、`180` 或 `270`;不传 `--pages` 时旋转全部页面。目标已存在且确认可覆盖时添加 `--overwrite`。 @@ -202,10 +203,10 @@ rotate --input 'source.pdf' --output 'output/pdf/rotated.pdf' --pages '1,3-5' -- 调用 `scripts/cleanup_pdf_temp.py`: ```text ---task-dir 'tmp/pdfs/<任务名>' +--task-dir '/usr/local/src/pdf/tmp/<任务名>' ``` -脚本只允许删除本 Skill 的 `tmp/pdfs/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。 +脚本只允许删除 `/usr/local/src/pdf/tmp/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。 ## 质量要求 diff --git a/skills/pdf/scripts/_pdf_common.py b/skills/pdf/scripts/_pdf_common.py index e1e2d3c..87f595e 100644 --- a/skills/pdf/scripts/_pdf_common.py +++ b/skills/pdf/scripts/_pdf_common.py @@ -11,6 +11,9 @@ from pathlib import Path from typing import Any, Callable, NoReturn, Optional +PDF_OUTPUT_ROOT = Path("/usr/local/src/pdf") + + def quiet_pdf_library_logs() -> None: """Keep third-party recovery warnings out of the JSON tool response.""" logging.getLogger("pdfminer").setLevel(logging.ERROR) @@ -62,8 +65,19 @@ def input_pdf(value: str) -> Path: return path +def ensure_output_path(path: Path) -> Path: + resolved = path.expanduser().resolve() + try: + resolved.relative_to(PDF_OUTPUT_ROOT) + except ValueError as exc: + raise ValueError( + f"PDF 产物必须输出到 {PDF_OUTPUT_ROOT} 目录下:{resolved}" + ) from exc + return resolved + + def output_pdf(value: str, overwrite: bool) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if path.suffix.lower() != ".pdf": raise ValueError("PDF 输出路径必须以 .pdf 结尾") if path.exists() and not overwrite: @@ -73,7 +87,7 @@ def output_pdf(value: str, overwrite: bool) -> Path: def output_directory(value: str) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if path.exists() and not path.is_dir(): raise ValueError(f"输出路径不是目录:{path}") path.mkdir(parents=True, exist_ok=True) diff --git a/skills/pdf/scripts/cleanup_pdf_temp.py b/skills/pdf/scripts/cleanup_pdf_temp.py index 5d865d3..bc0c698 100644 --- a/skills/pdf/scripts/cleanup_pdf_temp.py +++ b/skills/pdf/scripts/cleanup_pdf_temp.py @@ -7,7 +7,7 @@ import sys from pathlib import Path from typing import Any -from _pdf_common import SkillArgumentParser, run_cli +from _pdf_common import PDF_OUTPUT_ROOT, SkillArgumentParser, run_cli def _parse_args(argv: list[str]): @@ -15,28 +15,29 @@ def _parse_args(argv: list[str]): parser.add_argument( "--task-dir", required=True, - help="仅允许删除本 Skill 的 tmp/pdfs/ 下某个具体任务目录", + help=f"仅允许删除 {PDF_OUTPUT_ROOT}/tmp/ 下某个具体任务目录", ) return parser.parse_args(argv) def _cleanup(value: str) -> dict[str, Any]: - skill_root = Path(__file__).resolve().parents[1] - allowed_root = (skill_root / "tmp" / "pdfs").resolve() + allowed_root = (PDF_OUTPUT_ROOT / "tmp").resolve() candidate = Path(value).expanduser() target = ( candidate.resolve() if candidate.is_absolute() - else (skill_root / candidate).resolve() + else (allowed_root / candidate).resolve() ) try: relative = target.relative_to(allowed_root) except ValueError as exc: raise ValueError(f"只能清理 {allowed_root} 下的任务目录") from exc if not relative.parts: - raise ValueError("不能删除 tmp/pdfs 根目录") + raise ValueError(f"不能删除 {allowed_root} 根目录") if len(relative.parts) != 1: - raise ValueError("task-dir 必须直接指向 tmp/pdfs 下的单个任务目录") + raise ValueError( + f"task-dir 必须直接指向 {allowed_root} 下的单个任务目录" + ) if not target.is_dir(): raise FileNotFoundError(f"任务临时目录不存在:{target}") shutil.rmtree(target) diff --git a/skills/pdf/scripts/download_pdf.py b/skills/pdf/scripts/download_pdf.py index c27e773..f21d205 100644 --- a/skills/pdf/scripts/download_pdf.py +++ b/skills/pdf/scripts/download_pdf.py @@ -15,6 +15,8 @@ import urllib.request from pathlib import Path from typing import NoReturn, Optional +from _pdf_common import ensure_output_path + DEFAULT_TIMEOUT_SECONDS = 60 DEFAULT_MAX_BYTES = 100 * 1024 * 1024 CHUNK_SIZE = 1024 * 1024 @@ -100,7 +102,7 @@ def _parse_args(argv: list[str]) -> argparse.Namespace: output = Path(args.output).expanduser() if output.suffix.lower() != ".pdf": raise ValueError("output 必须以 .pdf 结尾") - args.output = output.resolve() + args.output = ensure_output_path(output) return args diff --git a/skills/pptx/SKILL.md b/skills/pptx/SKILL.md index 60b8e7e..9700053 100644 --- a/skills/pptx/SKILL.md +++ b/skills/pptx/SKILL.md @@ -14,7 +14,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染 - PptxGenJS、React Icons、Sharp、LibreOffice、Poppler 和 ZIP 操作只允许由固定脚本在内部调用。 - 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`validate_presentation.py` 还必须返回 `status: valid`、`issue_count: 0`。 - 只在需要读取图片、截图或视觉图表中的文字时调用 `ocr_presentation.py`。只使用 `slides[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。 -- 不覆盖用户提供的源文件。最终结果写入 `output/pptx/`,中间产物写入 `tmp/pptx/<任务名>/`。 +- 不覆盖用户提供的源文件。PPT 最终文件一律写入 `/usr/local/src/ppt/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/ppt/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。 - 远程地址只交给 `download_presentation.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。 - 环境已预置全部依赖,不安装软件包,也不提示用户安装依赖。 @@ -54,7 +54,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染 只接受 HTTPS 地址。完整保留 URL 及查询参数传给脚本,但不要在回复或输出文件名中暴露查询参数。 ```text ---url 'https://example.com/deck.pptx?signature=...' --output 'tmp/pptx/<任务名>/source.pptx' +--url 'https://example.com/deck.pptx?signature=...' --output '/usr/local/src/ppt/tmp/<任务名>/source.pptx' ``` 可选参数: @@ -84,7 +84,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染 需要连续正文时调用: ```text ---input 'source.pptx' --output 'tmp/pptx/<任务名>/content.md' +--input 'source.pptx' --output '/usr/local/src/ppt/tmp/<任务名>/content.md' ``` Markdown 适合检查遗漏、错字和顺序,不代表页面版式。 @@ -114,7 +114,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。 调用: ```text ---output 'output/pptx/result.pptx' --spec '' +--output '/usr/local/src/ppt/result.pptx' --spec '' ``` 内容较长时先把 JSON 写入任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。 @@ -236,7 +236,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。 需要图标时先调用: ```text ---library fi --name FiTrendingUp --color 2563EB --size 256 --output 'tmp/pptx/<任务名>/trend.png' +--library fi --name FiTrendingUp --color 2563EB --size 256 --output '/usr/local/src/ppt/tmp/<任务名>/trend.png' ``` 允许的图标库:`fa6`、`fi`、`hi2`、`io5`、`lu`、`md`、`ri`、`tb`。把生成的 PNG 作为普通图片元素插入。 @@ -302,7 +302,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。 调用: ```text ---input 'source.pptx' --output 'output/pptx/edited.pptx' --spec '' +--input 'source.pptx' --output '/usr/local/src/ppt/edited.pptx' --spec '' ``` JSON 顶层只有 `operations`,按数组顺序执行: @@ -341,7 +341,7 @@ JSON 顶层只有 `operations`,按数组顺序执行: 只对 `.pptx` 使用: ```text ---input 'template.pptx' --output 'tmp/pptx/<任务名>/expanded.pptx' --slide 2 --after 4 +--input 'template.pptx' --output '/usr/local/src/ppt/tmp/<任务名>/expanded.pptx' --slide 2 --after 4 ``` `slide` 是复制来源,`after` 是插入位置;省略 `after` 时紧跟来源页插入。脚本会更新页面清单、关系和内容类型,并移除不能安全共享的备注/批注关系。 @@ -381,13 +381,13 @@ JSON 顶层只有 `operations`,按数组顺序执行: 结构校验: ```text ---input 'output/pptx/result.pptx' --check-render +--input '/usr/local/src/ppt/result.pptx' --check-render ``` 模板派生结果: ```text ---input 'output/pptx/result.pptx' --original 'template.pptx' --check-render +--input '/usr/local/src/ppt/result.pptx' --original 'template.pptx' --check-render ``` 必须满足: @@ -402,7 +402,7 @@ JSON 顶层只有 `operations`,按数组顺序执行: 渲染全部页面: ```text ---input 'output/pptx/result.pptx' --output-dir 'tmp/pptx/<任务名>/rendered' --contact-sheet --include-pdf +--input '/usr/local/src/ppt/result.pptx' --output-dir '/usr/local/src/ppt/tmp/<任务名>/rendered' --contact-sheet --include-pdf ``` 默认 150 DPI、单次最多 30 页。可用 `--start-slide/--end-slide/--max-slides` 分批,复杂图表或小字可把 `--dpi` 提高到 180–220。联系表用于快速检查整体节奏,逐页 PNG 用于最终 QA。 diff --git a/skills/pptx/scripts/_pptx_common.py b/skills/pptx/scripts/_pptx_common.py index 9d38592..f67607e 100644 --- a/skills/pptx/scripts/_pptx_common.py +++ b/skills/pptx/scripts/_pptx_common.py @@ -21,6 +21,7 @@ OOXML_PRESENTATION_SUFFIXES = {".pptx", ".potx", ".ppsx"} PRESENTATION_INPUT_SUFFIXES = OOXML_PRESENTATION_SUFFIXES | {".ppt"} PRESENTATION_OUTPUT_SUFFIXES = {".pptx", ".potx", ".pdf"} IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp"} +PPT_OUTPUT_ROOT = Path("/usr/local/src/ppt") MAX_ARCHIVE_MEMBERS = 20_000 MAX_ARCHIVE_UNCOMPRESSED_BYTES = 1_073_741_824 @@ -92,13 +93,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path: return path +def ensure_output_path(path: Path) -> Path: + resolved = path.expanduser().resolve() + try: + resolved.relative_to(PPT_OUTPUT_ROOT) + except ValueError as exc: + raise ValueError( + f"PPT 产物必须输出到 {PPT_OUTPUT_ROOT} 目录下:{resolved}" + ) from exc + return resolved + + def output_file( value: str, suffixes: Optional[set[str]] = None, *, overwrite: bool = False, ) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if suffixes is not None and path.suffix.lower() not in suffixes: expected = "、".join(sorted(suffixes)) raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}") @@ -111,7 +123,7 @@ def output_file( def output_directory(value: str) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if path.exists() and not path.is_dir(): raise ValueError(f"输出路径不是目录:{path}") path.mkdir(parents=True, exist_ok=True) diff --git a/skills/xlsx/SKILL.md b/skills/xlsx/SKILL.md index 8009a41..9716bef 100644 --- a/skills/xlsx/SKILL.md +++ b/skills/xlsx/SKILL.md @@ -14,7 +14,7 @@ description: "创建、读取、编辑、修复、转换、重算、校验和渲 - 外部程序只允许由固定脚本在内部以无 shell 参数数组方式调用。 - 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`status: errors_found` 虽然表示脚本成功运行,但工作簿不合格,必须修复。 - 远程地址只交给 `download_workbook.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。 -- 不覆盖用户提供的源文件。创建或编辑结果写入 `output/xlsx/`,中间产物写入 `tmp/xlsx/<任务名>/`。 +- 不覆盖用户提供的源文件。Excel 最终文件一律写入 `/usr/local/src/excel/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/excel/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。 - 环境已预置依赖,不安装软件包,也不提示用户安装依赖。 ## 脚本清单 @@ -46,7 +46,7 @@ description: "创建、读取、编辑、修复、转换、重算、校验和渲 调用 `scripts/download_workbook.py`: ```text ---url 'https://example.com/report.xlsx?signature=...' --output 'tmp/xlsx/<任务名>/source.xlsx' +--url 'https://example.com/report.xlsx?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/source.xlsx' ``` 可选参数: @@ -87,13 +87,13 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。 调用 `scripts/apply_workbook.py`,新建时省略 `--input`,编辑时提供源文件: ```text ---output 'output/xlsx/result.xlsx' --spec '' +--output '/usr/local/src/excel/result.xlsx' --spec '' ``` 或: ```text ---input 'source.xlsx' --output 'output/xlsx/result.xlsx' --spec-file 'tmp/xlsx/task/operations.json' +--input 'source.xlsx' --output '/usr/local/src/excel/result.xlsx' --spec-file '/usr/local/src/excel/tmp/task/operations.json' ``` 目标已存在且确认是本次任务的旧产物时才传 `--overwrite`。输入含外部链接时脚本默认拒绝保存;只有用户明确接受缓存值可能丢失的风险时才传 `--allow-external-links`。`.xlsm` 必须继续输出 `.xlsm` 才能保留宏;只有用户明确同意丢弃宏时才输出 `.xlsx` 并传 `--drop-macros`。 @@ -167,7 +167,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。 调用 `scripts/convert_workbook.py`: ```text ---input 'legacy.xls' --output 'tmp/xlsx/task/source.xlsx' +--input 'legacy.xls' --output '/usr/local/src/excel/tmp/task/source.xlsx' ``` 常见用法: @@ -182,7 +182,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。 含公式的工作簿必须调用 `scripts/recalculate_workbook.py`: ```text ---input 'output/xlsx/result.xlsx' --output 'output/xlsx/result-recalculated.xlsx' +--input '/usr/local/src/excel/result.xlsx' --output '/usr/local/src/excel/result-recalculated.xlsx' ``` 检查返回值: @@ -198,7 +198,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。 调用 `scripts/render_workbook.py`: ```text ---input 'output/xlsx/result-recalculated.xlsx' --output-dir 'tmp/xlsx/task/rendered' +--input '/usr/local/src/excel/result-recalculated.xlsx' --output-dir '/usr/local/src/excel/tmp/task/rendered' ``` 默认 150 DPI、单次最多 20 页。可传: diff --git a/skills/xlsx/scripts/_xlsx_common.py b/skills/xlsx/scripts/_xlsx_common.py index 6f56dc5..269d704 100644 --- a/skills/xlsx/scripts/_xlsx_common.py +++ b/skills/xlsx/scripts/_xlsx_common.py @@ -19,6 +19,7 @@ from urllib.parse import quote EXCEL_INPUT_SUFFIXES = {".xlsx", ".xlsm", ".xltx", ".xltm"} TABULAR_INPUT_SUFFIXES = EXCEL_INPUT_SUFFIXES | {".xls", ".csv", ".tsv"} EXCEL_OUTPUT_SUFFIXES = {".xlsx", ".xlsm"} +EXCEL_OUTPUT_ROOT = Path("/usr/local/src/excel") FORMULA_ERROR_VALUES = { "#NULL!", "#DIV/0!", @@ -101,13 +102,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path: return path +def ensure_output_path(path: Path) -> Path: + resolved = path.expanduser().resolve() + try: + resolved.relative_to(EXCEL_OUTPUT_ROOT) + except ValueError as exc: + raise ValueError( + f"Excel 产物必须输出到 {EXCEL_OUTPUT_ROOT} 目录下:{resolved}" + ) from exc + return resolved + + def output_file( value: str, suffixes: Optional[set[str]] = None, *, overwrite: bool = False, ) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if suffixes is not None and path.suffix.lower() not in suffixes: expected = "、".join(sorted(suffixes)) raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}") @@ -120,7 +132,7 @@ def output_file( def output_directory(value: str) -> Path: - path = Path(value).expanduser().resolve() + path = ensure_output_path(Path(value)) if path.exists() and not path.is_dir(): raise ValueError(f"输出路径不是目录:{path}") path.mkdir(parents=True, exist_ok=True)