feat: 优化输出产物存储
This commit is contained in:
parent
b3193cb0c7
commit
8b99623429
@ -15,7 +15,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
|
|||||||
- 每次检查脚本返回 JSON;只有 `ok` 为 `true` 时才继续。`validate_document.py` 还必须返回 `status: valid`。
|
- 每次检查脚本返回 JSON;只有 `ok` 为 `true` 时才继续。`validate_document.py` 还必须返回 `status: valid`。
|
||||||
- 只在需要读取图片、截图或扫描页中的文字时调用 `ocr_document.py`。只使用 `pages[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。
|
- 只在需要读取图片、截图或扫描页中的文字时调用 `ocr_document.py`。只使用 `pages[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。
|
||||||
- 远程地址只交给 `download_document.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
|
- 远程地址只交给 `download_document.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
|
||||||
- 不覆盖用户提供的源文件。最终结果写入 `output/docx/`,中间产物写入 `tmp/docx/<任务名>/`。
|
- 不覆盖用户提供的源文件。Word 最终文件一律写入 `/usr/local/src/word/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/word/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
|
||||||
- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。
|
- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。
|
||||||
|
|
||||||
## 脚本清单
|
## 脚本清单
|
||||||
@ -55,7 +55,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
|
|||||||
调用 `scripts/download_document.py`:
|
调用 `scripts/download_document.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--url 'https://example.com/report.docx?signature=...' --output 'tmp/docx/<任务名>/source.docx'
|
--url 'https://example.com/report.docx?signature=...' --output '/usr/local/src/word/tmp/<任务名>/source.docx'
|
||||||
```
|
```
|
||||||
|
|
||||||
可选参数:
|
可选参数:
|
||||||
@ -123,7 +123,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
|
|||||||
调用 `scripts/create_document.py`:
|
调用 `scripts/create_document.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--output 'output/docx/result.docx' --spec '<JSON对象>'
|
--output '/usr/local/src/word/result.docx' --spec '<JSON对象>'
|
||||||
```
|
```
|
||||||
|
|
||||||
内容较长时先把 JSON 写到任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。
|
内容较长时先把 JSON 写到任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。
|
||||||
@ -230,7 +230,7 @@ description: "创建、读取、编辑、转换、批注、接受修订、校验
|
|||||||
调用 `scripts/edit_document.py`:
|
调用 `scripts/edit_document.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'source.docx' --output 'output/docx/edited.docx' --spec '<JSON对象>'
|
--input 'source.docx' --output '/usr/local/src/word/edited.docx' --spec '<JSON对象>'
|
||||||
```
|
```
|
||||||
|
|
||||||
JSON 顶层只有 `operations`。支持:
|
JSON 顶层只有 `operations`。支持:
|
||||||
@ -260,7 +260,7 @@ JSON 顶层只有 `operations`。支持:
|
|||||||
调用 `scripts/add_comment.py`:
|
调用 `scripts/add_comment.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'source.docx' --output 'output/docx/commented.docx' --find '费用上限' --comment '请确认该上限是否含税' --author '审阅人' --initials 'SR'
|
--input 'source.docx' --output '/usr/local/src/word/commented.docx' --find '费用上限' --comment '请确认该上限是否含税' --author '审阅人' --initials 'SR'
|
||||||
```
|
```
|
||||||
|
|
||||||
可选:
|
可选:
|
||||||
@ -276,7 +276,7 @@ JSON 顶层只有 `operations`。支持:
|
|||||||
调用 `scripts/accept_changes.py`:
|
调用 `scripts/accept_changes.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'redlined.docx' --output 'output/docx/clean.docx'
|
--input 'redlined.docx' --output '/usr/local/src/word/clean.docx'
|
||||||
```
|
```
|
||||||
|
|
||||||
脚本只执行固定的“接受全部修订”宏,不能运行用户提供的宏。必须检查:
|
脚本只执行固定的“接受全部修订”宏,不能运行用户提供的宏。必须检查:
|
||||||
@ -290,7 +290,7 @@ JSON 顶层只有 `operations`。支持:
|
|||||||
调用 `scripts/convert_document.py`:
|
调用 `scripts/convert_document.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'legacy.doc' --output 'tmp/docx/task/source.docx'
|
--input 'legacy.doc' --output '/usr/local/src/word/tmp/task/source.docx'
|
||||||
```
|
```
|
||||||
|
|
||||||
支持:
|
支持:
|
||||||
@ -317,7 +317,7 @@ JSON 顶层只有 `operations`。支持:
|
|||||||
调用 `scripts/validate_document.py`:
|
调用 `scripts/validate_document.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/docx/result.docx' --check-convert
|
--input '/usr/local/src/word/result.docx' --check-convert
|
||||||
```
|
```
|
||||||
|
|
||||||
必须满足:
|
必须满足:
|
||||||
@ -336,7 +336,7 @@ JSON 顶层只有 `operations`。支持:
|
|||||||
调用 `scripts/render_document.py`:
|
调用 `scripts/render_document.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/docx/result.docx' --output-dir 'tmp/docx/task/rendered'
|
--input '/usr/local/src/word/result.docx' --output-dir '/usr/local/src/word/tmp/task/rendered'
|
||||||
```
|
```
|
||||||
|
|
||||||
默认 150 DPI、单次最多 20 页。可传:
|
默认 150 DPI、单次最多 20 页。可传:
|
||||||
|
|||||||
@ -20,6 +20,7 @@ DOCX_INPUT_SUFFIXES = {".docx", ".dotx"}
|
|||||||
WORD_INPUT_SUFFIXES = DOCX_INPUT_SUFFIXES | {".doc"}
|
WORD_INPUT_SUFFIXES = DOCX_INPUT_SUFFIXES | {".doc"}
|
||||||
DOCX_OUTPUT_SUFFIXES = {".docx", ".dotx"}
|
DOCX_OUTPUT_SUFFIXES = {".docx", ".dotx"}
|
||||||
DOCUMENT_OUTPUT_SUFFIXES = DOCX_OUTPUT_SUFFIXES | {".pdf", ".txt", ".md"}
|
DOCUMENT_OUTPUT_SUFFIXES = DOCX_OUTPUT_SUFFIXES | {".pdf", ".txt", ".md"}
|
||||||
|
WORD_OUTPUT_ROOT = Path("/usr/local/src/word")
|
||||||
MAX_ARCHIVE_MEMBERS = 20_000
|
MAX_ARCHIVE_MEMBERS = 20_000
|
||||||
MAX_ARCHIVE_UNCOMPRESSED_BYTES = 512 * 1024 * 1024
|
MAX_ARCHIVE_UNCOMPRESSED_BYTES = 512 * 1024 * 1024
|
||||||
MAX_MEMBER_BYTES = 128 * 1024 * 1024
|
MAX_MEMBER_BYTES = 128 * 1024 * 1024
|
||||||
@ -90,13 +91,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path:
|
|||||||
return path
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_output_path(path: Path) -> Path:
|
||||||
|
resolved = path.expanduser().resolve()
|
||||||
|
try:
|
||||||
|
resolved.relative_to(WORD_OUTPUT_ROOT)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"Word 产物必须输出到 {WORD_OUTPUT_ROOT} 目录下:{resolved}"
|
||||||
|
) from exc
|
||||||
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
def output_file(
|
def output_file(
|
||||||
value: str,
|
value: str,
|
||||||
suffixes: Optional[set[str]] = None,
|
suffixes: Optional[set[str]] = None,
|
||||||
*,
|
*,
|
||||||
overwrite: bool = False,
|
overwrite: bool = False,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if suffixes is not None and path.suffix.lower() not in suffixes:
|
if suffixes is not None and path.suffix.lower() not in suffixes:
|
||||||
expected = "、".join(sorted(suffixes))
|
expected = "、".join(sorted(suffixes))
|
||||||
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
|
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
|
||||||
@ -109,7 +121,7 @@ def output_file(
|
|||||||
|
|
||||||
|
|
||||||
def output_directory(value: str) -> Path:
|
def output_directory(value: str) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if path.exists() and not path.is_dir():
|
if path.exists() and not path.is_dir():
|
||||||
raise ValueError(f"输出路径不是目录:{path}")
|
raise ValueError(f"输出路径不是目录:{path}")
|
||||||
path.mkdir(parents=True, exist_ok=True)
|
path.mkdir(parents=True, exist_ok=True)
|
||||||
|
|||||||
@ -9,6 +9,7 @@ from typing import Any
|
|||||||
from _docx_common import (
|
from _docx_common import (
|
||||||
DOCX_INPUT_SUFFIXES,
|
DOCX_INPUT_SUFFIXES,
|
||||||
SkillArgumentParser,
|
SkillArgumentParser,
|
||||||
|
ensure_output_path,
|
||||||
input_file,
|
input_file,
|
||||||
safe_extract_docx,
|
safe_extract_docx,
|
||||||
run_cli,
|
run_cli,
|
||||||
@ -27,7 +28,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|||||||
def main() -> dict[str, Any]:
|
def main() -> dict[str, Any]:
|
||||||
args = build_parser().parse_args()
|
args = build_parser().parse_args()
|
||||||
source = input_file(args.input, DOCX_INPUT_SUFFIXES)
|
source = input_file(args.input, DOCX_INPUT_SUFFIXES)
|
||||||
destination = Path(args.output_dir).expanduser().resolve()
|
destination = ensure_output_path(Path(args.output_dir))
|
||||||
if destination.exists():
|
if destination.exists():
|
||||||
if not destination.is_dir():
|
if not destination.is_dir():
|
||||||
raise ValueError(f"输出路径不是目录:{destination}")
|
raise ValueError(f"输出路径不是目录:{destination}")
|
||||||
|
|||||||
@ -16,6 +16,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
|
|||||||
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。
|
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。
|
||||||
- 收到 `ok: false` 时,依据 `error` 调整合法参数或向用户说明失败原因,不要把参数改传给其他脚本碰运气。
|
- 收到 `ok: false` 时,依据 `error` 调整合法参数或向用户说明失败原因,不要把参数改传给其他脚本碰运气。
|
||||||
- 阅读或总结时只使用 `pages[]` 中 `usable_for_summary: true` 的文本。`needs_ocr: false` 时不得为了“常规检查”继续 OCR、渲染或调用图片识别。
|
- 阅读或总结时只使用 `pages[]` 中 `usable_for_summary: true` 的文本。`needs_ocr: false` 时不得为了“常规检查”继续 OCR、渲染或调用图片识别。
|
||||||
|
- PDF 最终文件一律写入 `/usr/local/src/pdf/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/pdf/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
|
||||||
- 不把 PDF 密码作为脚本参数;工具调用参数可能进入运行日志。
|
- 不把 PDF 密码作为脚本参数;工具调用参数可能进入运行日志。
|
||||||
|
|
||||||
## 脚本清单
|
## 脚本清单
|
||||||
@ -36,14 +37,14 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
|
|||||||
|
|
||||||
## 标准流程
|
## 标准流程
|
||||||
|
|
||||||
1. 为任务选择简短目录名,把中间文件放在 `tmp/pdfs/<任务名>/`。
|
1. 为任务选择简短目录名,把中间文件放在 `/usr/local/src/pdf/tmp/<任务名>/`。
|
||||||
2. 远程 HTTPS 链接先调用 `download_pdf.py`;本地文件直接进入下一步。
|
2. 远程 HTTPS 链接先调用 `download_pdf.py`;本地文件直接进入下一步。
|
||||||
3. 调用 `inspect_pdf.py` 检查文件。遇到加密 PDF 时停止处理,请用户提供已解密副本;当前固定脚本不接收密码。
|
3. 调用 `inspect_pdf.py` 检查文件。遇到加密 PDF 时停止处理,请用户提供已解密副本;当前固定脚本不接收密码。
|
||||||
4. 阅读或总结时调用 `extract_text.py`。结果为 `usable_for_summary: true` 时使用可靠页文本并根据游标继续;同时为 `needs_ocr: false` 时直接回答,不调用 OCR、渲染或图片识别。
|
4. 阅读或总结时调用 `extract_text.py`。结果为 `usable_for_summary: true` 时使用可靠页文本并根据游标继续;同时为 `needs_ocr: false` 时直接回答,不调用 OCR、渲染或图片识别。
|
||||||
5. 只有 `extract_text.py` 返回 `needs_ocr: true` 时,才对 `text_quality.suspect_pages` 调用 `ocr_text.py`。原生可靠文本优先,OCR 只补齐可疑页,不重复识别正常页。
|
5. 只有 `extract_text.py` 返回 `needs_ocr: true` 时,才对 `text_quality.suspect_pages` 调用 `ocr_text.py`。原生可靠文本优先,OCR 只补齐可疑页,不重复识别正常页。
|
||||||
6. `ocr_text.py` 会在脚本内部临时渲染指定页面并交给本地 RapidOCR,完成后自动删除 PNG;普通扫描件解析不调用大模型识图,也不需要先调用 `render_pdf.py`。
|
6. `ocr_text.py` 会在脚本内部临时渲染指定页面并交给本地 RapidOCR,完成后自动删除 PNG;普通扫描件解析不调用大模型识图,也不需要先调用 `render_pdf.py`。
|
||||||
7. 仅在用户明确要求检查视觉版式,或任务涉及创建/修改 PDF 时调用 `render_pdf.py`。
|
7. 仅在用户明确要求检查视觉版式,或任务涉及创建/修改 PDF 时调用 `render_pdf.py`。
|
||||||
8. 创建或修改后的最终 PDF 写入 `output/pdf/`,重新执行检查、文本提取和全部页面渲染。
|
8. 创建或修改后的最终 PDF 写入 `/usr/local/src/pdf/`,重新执行检查、文本提取和全部页面渲染。
|
||||||
9. 最终产物位于临时目录之外且不再需要缓存时,调用 `cleanup_pdf_temp.py` 清理本次任务目录。
|
9. 最终产物位于临时目录之外且不再需要缓存时,调用 `cleanup_pdf_temp.py` 清理本次任务目录。
|
||||||
|
|
||||||
## 下载远程 PDF
|
## 下载远程 PDF
|
||||||
@ -53,7 +54,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
|
|||||||
调用 `scripts/download_pdf.py`:
|
调用 `scripts/download_pdf.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--url 'https://example.com/document.pdf' --output 'tmp/pdfs/<任务名>/source.pdf'
|
--url 'https://example.com/document.pdf' --output '/usr/local/src/pdf/tmp/<任务名>/source.pdf'
|
||||||
```
|
```
|
||||||
|
|
||||||
可选参数:
|
可选参数:
|
||||||
@ -69,7 +70,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
|
|||||||
调用 `scripts/inspect_pdf.py`:
|
调用 `scripts/inspect_pdf.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'tmp/pdfs/<任务名>/source.pdf'
|
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf'
|
||||||
```
|
```
|
||||||
|
|
||||||
使用返回的 `page_count`、`encrypted`、`metadata`、`page_layouts` 和 `form_field_count` 判断后续处理方式。不要直接调用 `pdfinfo`。
|
使用返回的 `page_count`、`encrypted`、`metadata`、`page_layouts` 和 `form_field_count` 判断后续处理方式。不要直接调用 `pdfinfo`。
|
||||||
@ -79,7 +80,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
|
|||||||
首次调用 `scripts/extract_text.py`:
|
首次调用 `scripts/extract_text.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'tmp/pdfs/<任务名>/source.pdf'
|
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf'
|
||||||
```
|
```
|
||||||
|
|
||||||
默认使用 `auto` 引擎:先由 Poppler `pdftotext` 提取;结果不可用或命令不可用时自动尝试 `pdfplumber`,并可逐页选择质量更好的结果。脚本使用 `pypdf` 获取标准页数,并拒绝把页数不一致的提取结果当作成功。不要直接执行 `pdftotext`。
|
默认使用 `auto` 引擎:先由 Poppler `pdftotext` 提取;结果不可用或命令不可用时自动尝试 `pdfplumber`,并可逐页选择质量更好的结果。脚本使用 `pypdf` 获取标准页数,并拒绝把页数不一致的提取结果当作成功。不要直接执行 `pdftotext`。
|
||||||
@ -106,7 +107,7 @@ description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,包括安全
|
|||||||
仅当 `extract_text.py` 返回 `needs_ocr: true` 时调用 `scripts/ocr_text.py`。`--pages` 必须明确指定 `text_quality.suspect_pages` 中要读取的页,单次最多 4 页:
|
仅当 `extract_text.py` 返回 `needs_ocr: true` 时调用 `scripts/ocr_text.py`。`--pages` 必须明确指定 `text_quality.suspect_pages` 中要读取的页,单次最多 4 页:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'tmp/pdfs/<任务名>/source.pdf' --pages '2,5-6'
|
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --pages '2,5-6'
|
||||||
```
|
```
|
||||||
|
|
||||||
默认以 260 DPI 临时渲染,并使用镜像中预置的 RapidOCR 与 ONNX Runtime 在本地识别。脚本不会联网下载模型,不会保留渲染图片,也不会调用大模型视觉能力。可选参数:
|
默认以 260 DPI 临时渲染,并使用镜像中预置的 RapidOCR 与 ONNX Runtime 在本地识别。脚本不会联网下载模型,不会保留渲染图片,也不会调用大模型视觉能力。可选参数:
|
||||||
@ -131,7 +132,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
|
|||||||
调用 `scripts/extract_tables.py`:
|
调用 `scripts/extract_tables.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'tmp/pdfs/<任务名>/source.pdf' --start-page 1
|
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --start-page 1
|
||||||
```
|
```
|
||||||
|
|
||||||
默认单次最多处理 5 页、20 个表格和 2000 个单元格。可用 `--end-page`、`--start-table`、`--max-pages`、`--max-tables`、`--max-cells` 调整。若 `has_more: true`,把 `next_page` 传给 `--start-page`、`next_table` 传给 `--start-table` 后继续,并保留首次调用的 `--end-page`(如果指定)及其他提取选项。
|
默认单次最多处理 5 页、20 个表格和 2000 个单元格。可用 `--end-page`、`--start-table`、`--max-pages`、`--max-tables`、`--max-cells` 调整。若 `has_more: true`,把 `next_page` 传给 `--start-page`、`next_table` 传给 `--start-table` 后继续,并保留首次调用的 `--end-page`(如果指定)及其他提取选项。
|
||||||
@ -146,7 +147,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
|
|||||||
不要因为输入是 PDF、需要总结、需要 OCR 或需要检查首页就自动调用本脚本;OCR 的临时渲染由 `ocr_text.py` 内部完成。调用脚本时不要直接执行 `pdftoppm`:
|
不要因为输入是 PDF、需要总结、需要 OCR 或需要检查首页就自动调用本脚本;OCR 的临时渲染由 `ocr_text.py` 内部完成。调用脚本时不要直接执行 `pdftoppm`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'tmp/pdfs/<任务名>/source.pdf' --output-dir 'tmp/pdfs/<任务名>/rendered' --start-page 1
|
--input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --output-dir '/usr/local/src/pdf/tmp/<任务名>/rendered' --start-page 1
|
||||||
```
|
```
|
||||||
|
|
||||||
默认 150 DPI、单次最多 10 页。可使用 `--end-page`、`--max-pages`、`--dpi`、`--timeout` 和 `--overwrite`。若 `has_more: true`,使用 `next_page` 继续,并保留首次调用的 `--end-page`(如果指定)、输出目录及其他渲染选项。脚本返回标准化的 `page-0001.png` 文件路径。
|
默认 150 DPI、单次最多 10 页。可使用 `--end-page`、`--max-pages`、`--dpi`、`--timeout` 和 `--overwrite`。若 `has_more: true`,使用 `next_page` 继续,并保留首次调用的 `--end-page`(如果指定)、输出目录及其他渲染选项。脚本返回标准化的 `page-0001.png` 文件路径。
|
||||||
@ -158,7 +159,7 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
|
|||||||
先使用 `write_file` 把内容写为 UTF-8 `.txt` 或 `.md` 文件,再调用 `scripts/create_pdf.py`:
|
先使用 `write_file` 把内容写为 UTF-8 `.txt` 或 `.md` 文件,再调用 `scripts/create_pdf.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'tmp/pdfs/<任务名>/content.md' --output 'output/pdf/<文件名>.pdf' --title '文档标题'
|
--input '/usr/local/src/pdf/tmp/<任务名>/content.md' --output '/usr/local/src/pdf/<文件名>.pdf' --title '文档标题'
|
||||||
```
|
```
|
||||||
|
|
||||||
脚本支持 Markdown 标题、项目符号和简单表格,自动选择可嵌入的 Unicode 字体并添加页码。可选参数:
|
脚本支持 Markdown 标题、项目符号和简单表格,自动选择可嵌入的 Unicode 字体并添加页码。可选参数:
|
||||||
@ -178,13 +179,13 @@ OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_se
|
|||||||
合并:
|
合并:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
merge --input 'a.pdf' --input 'b.pdf' --output 'output/pdf/merged.pdf'
|
merge --input 'a.pdf' --input 'b.pdf' --output '/usr/local/src/pdf/merged.pdf'
|
||||||
```
|
```
|
||||||
|
|
||||||
拆分指定范围:
|
拆分指定范围:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
split --input 'source.pdf' --output-dir 'output/pdf/split' --range 1-3 --range 4-6
|
split --input 'source.pdf' --output-dir '/usr/local/src/pdf/split' --range 1-3 --range 4-6
|
||||||
```
|
```
|
||||||
|
|
||||||
不传 `--range` 时每页生成一个 PDF。
|
不传 `--range` 时每页生成一个 PDF。
|
||||||
@ -192,7 +193,7 @@ split --input 'source.pdf' --output-dir 'output/pdf/split' --range 1-3 --range 4
|
|||||||
旋转指定页面:
|
旋转指定页面:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
rotate --input 'source.pdf' --output 'output/pdf/rotated.pdf' --pages '1,3-5' --degrees 90
|
rotate --input 'source.pdf' --output '/usr/local/src/pdf/rotated.pdf' --pages '1,3-5' --degrees 90
|
||||||
```
|
```
|
||||||
|
|
||||||
`--degrees` 只能是 `90`、`180` 或 `270`;不传 `--pages` 时旋转全部页面。目标已存在且确认可覆盖时添加 `--overwrite`。
|
`--degrees` 只能是 `90`、`180` 或 `270`;不传 `--pages` 时旋转全部页面。目标已存在且确认可覆盖时添加 `--overwrite`。
|
||||||
@ -202,10 +203,10 @@ rotate --input 'source.pdf' --output 'output/pdf/rotated.pdf' --pages '1,3-5' --
|
|||||||
调用 `scripts/cleanup_pdf_temp.py`:
|
调用 `scripts/cleanup_pdf_temp.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--task-dir 'tmp/pdfs/<任务名>'
|
--task-dir '/usr/local/src/pdf/tmp/<任务名>'
|
||||||
```
|
```
|
||||||
|
|
||||||
脚本只允许删除本 Skill 的 `tmp/pdfs/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。
|
脚本只允许删除 `/usr/local/src/pdf/tmp/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。
|
||||||
|
|
||||||
## 质量要求
|
## 质量要求
|
||||||
|
|
||||||
|
|||||||
@ -11,6 +11,9 @@ from pathlib import Path
|
|||||||
from typing import Any, Callable, NoReturn, Optional
|
from typing import Any, Callable, NoReturn, Optional
|
||||||
|
|
||||||
|
|
||||||
|
PDF_OUTPUT_ROOT = Path("/usr/local/src/pdf")
|
||||||
|
|
||||||
|
|
||||||
def quiet_pdf_library_logs() -> None:
|
def quiet_pdf_library_logs() -> None:
|
||||||
"""Keep third-party recovery warnings out of the JSON tool response."""
|
"""Keep third-party recovery warnings out of the JSON tool response."""
|
||||||
logging.getLogger("pdfminer").setLevel(logging.ERROR)
|
logging.getLogger("pdfminer").setLevel(logging.ERROR)
|
||||||
@ -62,8 +65,19 @@ def input_pdf(value: str) -> Path:
|
|||||||
return path
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_output_path(path: Path) -> Path:
|
||||||
|
resolved = path.expanduser().resolve()
|
||||||
|
try:
|
||||||
|
resolved.relative_to(PDF_OUTPUT_ROOT)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"PDF 产物必须输出到 {PDF_OUTPUT_ROOT} 目录下:{resolved}"
|
||||||
|
) from exc
|
||||||
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
def output_pdf(value: str, overwrite: bool) -> Path:
|
def output_pdf(value: str, overwrite: bool) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if path.suffix.lower() != ".pdf":
|
if path.suffix.lower() != ".pdf":
|
||||||
raise ValueError("PDF 输出路径必须以 .pdf 结尾")
|
raise ValueError("PDF 输出路径必须以 .pdf 结尾")
|
||||||
if path.exists() and not overwrite:
|
if path.exists() and not overwrite:
|
||||||
@ -73,7 +87,7 @@ def output_pdf(value: str, overwrite: bool) -> Path:
|
|||||||
|
|
||||||
|
|
||||||
def output_directory(value: str) -> Path:
|
def output_directory(value: str) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if path.exists() and not path.is_dir():
|
if path.exists() and not path.is_dir():
|
||||||
raise ValueError(f"输出路径不是目录:{path}")
|
raise ValueError(f"输出路径不是目录:{path}")
|
||||||
path.mkdir(parents=True, exist_ok=True)
|
path.mkdir(parents=True, exist_ok=True)
|
||||||
|
|||||||
@ -7,7 +7,7 @@ import sys
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
from _pdf_common import SkillArgumentParser, run_cli
|
from _pdf_common import PDF_OUTPUT_ROOT, SkillArgumentParser, run_cli
|
||||||
|
|
||||||
|
|
||||||
def _parse_args(argv: list[str]):
|
def _parse_args(argv: list[str]):
|
||||||
@ -15,28 +15,29 @@ def _parse_args(argv: list[str]):
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--task-dir",
|
"--task-dir",
|
||||||
required=True,
|
required=True,
|
||||||
help="仅允许删除本 Skill 的 tmp/pdfs/ 下某个具体任务目录",
|
help=f"仅允许删除 {PDF_OUTPUT_ROOT}/tmp/ 下某个具体任务目录",
|
||||||
)
|
)
|
||||||
return parser.parse_args(argv)
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
|
|
||||||
def _cleanup(value: str) -> dict[str, Any]:
|
def _cleanup(value: str) -> dict[str, Any]:
|
||||||
skill_root = Path(__file__).resolve().parents[1]
|
allowed_root = (PDF_OUTPUT_ROOT / "tmp").resolve()
|
||||||
allowed_root = (skill_root / "tmp" / "pdfs").resolve()
|
|
||||||
candidate = Path(value).expanduser()
|
candidate = Path(value).expanduser()
|
||||||
target = (
|
target = (
|
||||||
candidate.resolve()
|
candidate.resolve()
|
||||||
if candidate.is_absolute()
|
if candidate.is_absolute()
|
||||||
else (skill_root / candidate).resolve()
|
else (allowed_root / candidate).resolve()
|
||||||
)
|
)
|
||||||
try:
|
try:
|
||||||
relative = target.relative_to(allowed_root)
|
relative = target.relative_to(allowed_root)
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
raise ValueError(f"只能清理 {allowed_root} 下的任务目录") from exc
|
raise ValueError(f"只能清理 {allowed_root} 下的任务目录") from exc
|
||||||
if not relative.parts:
|
if not relative.parts:
|
||||||
raise ValueError("不能删除 tmp/pdfs 根目录")
|
raise ValueError(f"不能删除 {allowed_root} 根目录")
|
||||||
if len(relative.parts) != 1:
|
if len(relative.parts) != 1:
|
||||||
raise ValueError("task-dir 必须直接指向 tmp/pdfs 下的单个任务目录")
|
raise ValueError(
|
||||||
|
f"task-dir 必须直接指向 {allowed_root} 下的单个任务目录"
|
||||||
|
)
|
||||||
if not target.is_dir():
|
if not target.is_dir():
|
||||||
raise FileNotFoundError(f"任务临时目录不存在:{target}")
|
raise FileNotFoundError(f"任务临时目录不存在:{target}")
|
||||||
shutil.rmtree(target)
|
shutil.rmtree(target)
|
||||||
|
|||||||
@ -15,6 +15,8 @@ import urllib.request
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import NoReturn, Optional
|
from typing import NoReturn, Optional
|
||||||
|
|
||||||
|
from _pdf_common import ensure_output_path
|
||||||
|
|
||||||
DEFAULT_TIMEOUT_SECONDS = 60
|
DEFAULT_TIMEOUT_SECONDS = 60
|
||||||
DEFAULT_MAX_BYTES = 100 * 1024 * 1024
|
DEFAULT_MAX_BYTES = 100 * 1024 * 1024
|
||||||
CHUNK_SIZE = 1024 * 1024
|
CHUNK_SIZE = 1024 * 1024
|
||||||
@ -100,7 +102,7 @@ def _parse_args(argv: list[str]) -> argparse.Namespace:
|
|||||||
output = Path(args.output).expanduser()
|
output = Path(args.output).expanduser()
|
||||||
if output.suffix.lower() != ".pdf":
|
if output.suffix.lower() != ".pdf":
|
||||||
raise ValueError("output 必须以 .pdf 结尾")
|
raise ValueError("output 必须以 .pdf 结尾")
|
||||||
args.output = output.resolve()
|
args.output = ensure_output_path(output)
|
||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@ -14,7 +14,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染
|
|||||||
- PptxGenJS、React Icons、Sharp、LibreOffice、Poppler 和 ZIP 操作只允许由固定脚本在内部调用。
|
- PptxGenJS、React Icons、Sharp、LibreOffice、Poppler 和 ZIP 操作只允许由固定脚本在内部调用。
|
||||||
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`validate_presentation.py` 还必须返回 `status: valid`、`issue_count: 0`。
|
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`validate_presentation.py` 还必须返回 `status: valid`、`issue_count: 0`。
|
||||||
- 只在需要读取图片、截图或视觉图表中的文字时调用 `ocr_presentation.py`。只使用 `slides[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。
|
- 只在需要读取图片、截图或视觉图表中的文字时调用 `ocr_presentation.py`。只使用 `slides[]` 中 `usable_for_summary: true` 的 `text`;低置信度结果不得作为可靠正文。
|
||||||
- 不覆盖用户提供的源文件。最终结果写入 `output/pptx/`,中间产物写入 `tmp/pptx/<任务名>/`。
|
- 不覆盖用户提供的源文件。PPT 最终文件一律写入 `/usr/local/src/ppt/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/ppt/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
|
||||||
- 远程地址只交给 `download_presentation.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
|
- 远程地址只交给 `download_presentation.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
|
||||||
- 环境已预置全部依赖,不安装软件包,也不提示用户安装依赖。
|
- 环境已预置全部依赖,不安装软件包,也不提示用户安装依赖。
|
||||||
|
|
||||||
@ -54,7 +54,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染
|
|||||||
只接受 HTTPS 地址。完整保留 URL 及查询参数传给脚本,但不要在回复或输出文件名中暴露查询参数。
|
只接受 HTTPS 地址。完整保留 URL 及查询参数传给脚本,但不要在回复或输出文件名中暴露查询参数。
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--url 'https://example.com/deck.pptx?signature=...' --output 'tmp/pptx/<任务名>/source.pptx'
|
--url 'https://example.com/deck.pptx?signature=...' --output '/usr/local/src/ppt/tmp/<任务名>/source.pptx'
|
||||||
```
|
```
|
||||||
|
|
||||||
可选参数:
|
可选参数:
|
||||||
@ -84,7 +84,7 @@ description: "创建、读取、编辑、复制页面、转换、校验和渲染
|
|||||||
需要连续正文时调用:
|
需要连续正文时调用:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'source.pptx' --output 'tmp/pptx/<任务名>/content.md'
|
--input 'source.pptx' --output '/usr/local/src/ppt/tmp/<任务名>/content.md'
|
||||||
```
|
```
|
||||||
|
|
||||||
Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
|
Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
|
||||||
@ -114,7 +114,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
|
|||||||
调用:
|
调用:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--output 'output/pptx/result.pptx' --spec '<JSON对象>'
|
--output '/usr/local/src/ppt/result.pptx' --spec '<JSON对象>'
|
||||||
```
|
```
|
||||||
|
|
||||||
内容较长时先把 JSON 写入任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。
|
内容较长时先把 JSON 写入任务临时目录,再传 `--spec-file`。目标是本次任务旧产物且确认可覆盖时才传 `--overwrite`。
|
||||||
@ -236,7 +236,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
|
|||||||
需要图标时先调用:
|
需要图标时先调用:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--library fi --name FiTrendingUp --color 2563EB --size 256 --output 'tmp/pptx/<任务名>/trend.png'
|
--library fi --name FiTrendingUp --color 2563EB --size 256 --output '/usr/local/src/ppt/tmp/<任务名>/trend.png'
|
||||||
```
|
```
|
||||||
|
|
||||||
允许的图标库:`fa6`、`fi`、`hi2`、`io5`、`lu`、`md`、`ri`、`tb`。把生成的 PNG 作为普通图片元素插入。
|
允许的图标库:`fa6`、`fi`、`hi2`、`io5`、`lu`、`md`、`ri`、`tb`。把生成的 PNG 作为普通图片元素插入。
|
||||||
@ -302,7 +302,7 @@ Markdown 适合检查遗漏、错字和顺序,不代表页面版式。
|
|||||||
调用:
|
调用:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'source.pptx' --output 'output/pptx/edited.pptx' --spec '<JSON对象>'
|
--input 'source.pptx' --output '/usr/local/src/ppt/edited.pptx' --spec '<JSON对象>'
|
||||||
```
|
```
|
||||||
|
|
||||||
JSON 顶层只有 `operations`,按数组顺序执行:
|
JSON 顶层只有 `operations`,按数组顺序执行:
|
||||||
@ -341,7 +341,7 @@ JSON 顶层只有 `operations`,按数组顺序执行:
|
|||||||
只对 `.pptx` 使用:
|
只对 `.pptx` 使用:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'template.pptx' --output 'tmp/pptx/<任务名>/expanded.pptx' --slide 2 --after 4
|
--input 'template.pptx' --output '/usr/local/src/ppt/tmp/<任务名>/expanded.pptx' --slide 2 --after 4
|
||||||
```
|
```
|
||||||
|
|
||||||
`slide` 是复制来源,`after` 是插入位置;省略 `after` 时紧跟来源页插入。脚本会更新页面清单、关系和内容类型,并移除不能安全共享的备注/批注关系。
|
`slide` 是复制来源,`after` 是插入位置;省略 `after` 时紧跟来源页插入。脚本会更新页面清单、关系和内容类型,并移除不能安全共享的备注/批注关系。
|
||||||
@ -381,13 +381,13 @@ JSON 顶层只有 `operations`,按数组顺序执行:
|
|||||||
结构校验:
|
结构校验:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/pptx/result.pptx' --check-render
|
--input '/usr/local/src/ppt/result.pptx' --check-render
|
||||||
```
|
```
|
||||||
|
|
||||||
模板派生结果:
|
模板派生结果:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/pptx/result.pptx' --original 'template.pptx' --check-render
|
--input '/usr/local/src/ppt/result.pptx' --original 'template.pptx' --check-render
|
||||||
```
|
```
|
||||||
|
|
||||||
必须满足:
|
必须满足:
|
||||||
@ -402,7 +402,7 @@ JSON 顶层只有 `operations`,按数组顺序执行:
|
|||||||
渲染全部页面:
|
渲染全部页面:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/pptx/result.pptx' --output-dir 'tmp/pptx/<任务名>/rendered' --contact-sheet --include-pdf
|
--input '/usr/local/src/ppt/result.pptx' --output-dir '/usr/local/src/ppt/tmp/<任务名>/rendered' --contact-sheet --include-pdf
|
||||||
```
|
```
|
||||||
|
|
||||||
默认 150 DPI、单次最多 30 页。可用 `--start-slide/--end-slide/--max-slides` 分批,复杂图表或小字可把 `--dpi` 提高到 180–220。联系表用于快速检查整体节奏,逐页 PNG 用于最终 QA。
|
默认 150 DPI、单次最多 30 页。可用 `--start-slide/--end-slide/--max-slides` 分批,复杂图表或小字可把 `--dpi` 提高到 180–220。联系表用于快速检查整体节奏,逐页 PNG 用于最终 QA。
|
||||||
|
|||||||
@ -21,6 +21,7 @@ OOXML_PRESENTATION_SUFFIXES = {".pptx", ".potx", ".ppsx"}
|
|||||||
PRESENTATION_INPUT_SUFFIXES = OOXML_PRESENTATION_SUFFIXES | {".ppt"}
|
PRESENTATION_INPUT_SUFFIXES = OOXML_PRESENTATION_SUFFIXES | {".ppt"}
|
||||||
PRESENTATION_OUTPUT_SUFFIXES = {".pptx", ".potx", ".pdf"}
|
PRESENTATION_OUTPUT_SUFFIXES = {".pptx", ".potx", ".pdf"}
|
||||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp"}
|
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp"}
|
||||||
|
PPT_OUTPUT_ROOT = Path("/usr/local/src/ppt")
|
||||||
|
|
||||||
MAX_ARCHIVE_MEMBERS = 20_000
|
MAX_ARCHIVE_MEMBERS = 20_000
|
||||||
MAX_ARCHIVE_UNCOMPRESSED_BYTES = 1_073_741_824
|
MAX_ARCHIVE_UNCOMPRESSED_BYTES = 1_073_741_824
|
||||||
@ -92,13 +93,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path:
|
|||||||
return path
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_output_path(path: Path) -> Path:
|
||||||
|
resolved = path.expanduser().resolve()
|
||||||
|
try:
|
||||||
|
resolved.relative_to(PPT_OUTPUT_ROOT)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"PPT 产物必须输出到 {PPT_OUTPUT_ROOT} 目录下:{resolved}"
|
||||||
|
) from exc
|
||||||
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
def output_file(
|
def output_file(
|
||||||
value: str,
|
value: str,
|
||||||
suffixes: Optional[set[str]] = None,
|
suffixes: Optional[set[str]] = None,
|
||||||
*,
|
*,
|
||||||
overwrite: bool = False,
|
overwrite: bool = False,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if suffixes is not None and path.suffix.lower() not in suffixes:
|
if suffixes is not None and path.suffix.lower() not in suffixes:
|
||||||
expected = "、".join(sorted(suffixes))
|
expected = "、".join(sorted(suffixes))
|
||||||
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
|
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
|
||||||
@ -111,7 +123,7 @@ def output_file(
|
|||||||
|
|
||||||
|
|
||||||
def output_directory(value: str) -> Path:
|
def output_directory(value: str) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if path.exists() and not path.is_dir():
|
if path.exists() and not path.is_dir():
|
||||||
raise ValueError(f"输出路径不是目录:{path}")
|
raise ValueError(f"输出路径不是目录:{path}")
|
||||||
path.mkdir(parents=True, exist_ok=True)
|
path.mkdir(parents=True, exist_ok=True)
|
||||||
|
|||||||
@ -14,7 +14,7 @@ description: "创建、读取、编辑、修复、转换、重算、校验和渲
|
|||||||
- 外部程序只允许由固定脚本在内部以无 shell 参数数组方式调用。
|
- 外部程序只允许由固定脚本在内部以无 shell 参数数组方式调用。
|
||||||
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`status: errors_found` 虽然表示脚本成功运行,但工作簿不合格,必须修复。
|
- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`status: errors_found` 虽然表示脚本成功运行,但工作簿不合格,必须修复。
|
||||||
- 远程地址只交给 `download_workbook.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
|
- 远程地址只交给 `download_workbook.py`;不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。
|
||||||
- 不覆盖用户提供的源文件。创建或编辑结果写入 `output/xlsx/`,中间产物写入 `tmp/xlsx/<任务名>/`。
|
- 不覆盖用户提供的源文件。Excel 最终文件一律写入 `/usr/local/src/excel/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/excel/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。
|
||||||
- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。
|
- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。
|
||||||
|
|
||||||
## 脚本清单
|
## 脚本清单
|
||||||
@ -46,7 +46,7 @@ description: "创建、读取、编辑、修复、转换、重算、校验和渲
|
|||||||
调用 `scripts/download_workbook.py`:
|
调用 `scripts/download_workbook.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--url 'https://example.com/report.xlsx?signature=...' --output 'tmp/xlsx/<任务名>/source.xlsx'
|
--url 'https://example.com/report.xlsx?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/source.xlsx'
|
||||||
```
|
```
|
||||||
|
|
||||||
可选参数:
|
可选参数:
|
||||||
@ -87,13 +87,13 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
|
|||||||
调用 `scripts/apply_workbook.py`,新建时省略 `--input`,编辑时提供源文件:
|
调用 `scripts/apply_workbook.py`,新建时省略 `--input`,编辑时提供源文件:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--output 'output/xlsx/result.xlsx' --spec '<JSON对象>'
|
--output '/usr/local/src/excel/result.xlsx' --spec '<JSON对象>'
|
||||||
```
|
```
|
||||||
|
|
||||||
或:
|
或:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'source.xlsx' --output 'output/xlsx/result.xlsx' --spec-file 'tmp/xlsx/task/operations.json'
|
--input 'source.xlsx' --output '/usr/local/src/excel/result.xlsx' --spec-file '/usr/local/src/excel/tmp/task/operations.json'
|
||||||
```
|
```
|
||||||
|
|
||||||
目标已存在且确认是本次任务的旧产物时才传 `--overwrite`。输入含外部链接时脚本默认拒绝保存;只有用户明确接受缓存值可能丢失的风险时才传 `--allow-external-links`。`.xlsm` 必须继续输出 `.xlsm` 才能保留宏;只有用户明确同意丢弃宏时才输出 `.xlsx` 并传 `--drop-macros`。
|
目标已存在且确认是本次任务的旧产物时才传 `--overwrite`。输入含外部链接时脚本默认拒绝保存;只有用户明确接受缓存值可能丢失的风险时才传 `--allow-external-links`。`.xlsm` 必须继续输出 `.xlsm` 才能保留宏;只有用户明确同意丢弃宏时才输出 `.xlsx` 并传 `--drop-macros`。
|
||||||
@ -167,7 +167,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
|
|||||||
调用 `scripts/convert_workbook.py`:
|
调用 `scripts/convert_workbook.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'legacy.xls' --output 'tmp/xlsx/task/source.xlsx'
|
--input 'legacy.xls' --output '/usr/local/src/excel/tmp/task/source.xlsx'
|
||||||
```
|
```
|
||||||
|
|
||||||
常见用法:
|
常见用法:
|
||||||
@ -182,7 +182,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
|
|||||||
含公式的工作簿必须调用 `scripts/recalculate_workbook.py`:
|
含公式的工作簿必须调用 `scripts/recalculate_workbook.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/xlsx/result.xlsx' --output 'output/xlsx/result-recalculated.xlsx'
|
--input '/usr/local/src/excel/result.xlsx' --output '/usr/local/src/excel/result-recalculated.xlsx'
|
||||||
```
|
```
|
||||||
|
|
||||||
检查返回值:
|
检查返回值:
|
||||||
@ -198,7 +198,7 @@ CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。
|
|||||||
调用 `scripts/render_workbook.py`:
|
调用 `scripts/render_workbook.py`:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
--input 'output/xlsx/result-recalculated.xlsx' --output-dir 'tmp/xlsx/task/rendered'
|
--input '/usr/local/src/excel/result-recalculated.xlsx' --output-dir '/usr/local/src/excel/tmp/task/rendered'
|
||||||
```
|
```
|
||||||
|
|
||||||
默认 150 DPI、单次最多 20 页。可传:
|
默认 150 DPI、单次最多 20 页。可传:
|
||||||
|
|||||||
@ -19,6 +19,7 @@ from urllib.parse import quote
|
|||||||
EXCEL_INPUT_SUFFIXES = {".xlsx", ".xlsm", ".xltx", ".xltm"}
|
EXCEL_INPUT_SUFFIXES = {".xlsx", ".xlsm", ".xltx", ".xltm"}
|
||||||
TABULAR_INPUT_SUFFIXES = EXCEL_INPUT_SUFFIXES | {".xls", ".csv", ".tsv"}
|
TABULAR_INPUT_SUFFIXES = EXCEL_INPUT_SUFFIXES | {".xls", ".csv", ".tsv"}
|
||||||
EXCEL_OUTPUT_SUFFIXES = {".xlsx", ".xlsm"}
|
EXCEL_OUTPUT_SUFFIXES = {".xlsx", ".xlsm"}
|
||||||
|
EXCEL_OUTPUT_ROOT = Path("/usr/local/src/excel")
|
||||||
FORMULA_ERROR_VALUES = {
|
FORMULA_ERROR_VALUES = {
|
||||||
"#NULL!",
|
"#NULL!",
|
||||||
"#DIV/0!",
|
"#DIV/0!",
|
||||||
@ -101,13 +102,24 @@ def input_file(value: str, suffixes: Optional[set[str]] = None) -> Path:
|
|||||||
return path
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_output_path(path: Path) -> Path:
|
||||||
|
resolved = path.expanduser().resolve()
|
||||||
|
try:
|
||||||
|
resolved.relative_to(EXCEL_OUTPUT_ROOT)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"Excel 产物必须输出到 {EXCEL_OUTPUT_ROOT} 目录下:{resolved}"
|
||||||
|
) from exc
|
||||||
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
def output_file(
|
def output_file(
|
||||||
value: str,
|
value: str,
|
||||||
suffixes: Optional[set[str]] = None,
|
suffixes: Optional[set[str]] = None,
|
||||||
*,
|
*,
|
||||||
overwrite: bool = False,
|
overwrite: bool = False,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if suffixes is not None and path.suffix.lower() not in suffixes:
|
if suffixes is not None and path.suffix.lower() not in suffixes:
|
||||||
expected = "、".join(sorted(suffixes))
|
expected = "、".join(sorted(suffixes))
|
||||||
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
|
raise ValueError(f"不支持的输出格式 {path.suffix};允许:{expected}")
|
||||||
@ -120,7 +132,7 @@ def output_file(
|
|||||||
|
|
||||||
|
|
||||||
def output_directory(value: str) -> Path:
|
def output_directory(value: str) -> Path:
|
||||||
path = Path(value).expanduser().resolve()
|
path = ensure_output_path(Path(value))
|
||||||
if path.exists() and not path.is_dir():
|
if path.exists() and not path.is_dir():
|
||||||
raise ValueError(f"输出路径不是目录:{path}")
|
raise ValueError(f"输出路径不是目录:{path}")
|
||||||
path.mkdir(parents=True, exist_ok=True)
|
path.mkdir(parents=True, exist_ok=True)
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user