From 67351ccbddc459c5f6f2ba02a38b0f268a235911 Mon Sep 17 00:00:00 2001 From: hp0912 <809211365@qq.com> Date: Wed, 9 Sep 2026 12:30:35 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=A2=9E=E5=BC=BA=20pdf=20=E6=8A=80?= =?UTF-8?q?=E8=83=BD?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- skills/pdf/SKILL.md | 280 +++++------------------ skills/pdf/agents/openai.yaml | 6 +- skills/pdf/assets/design.css | 74 ++++++ skills/pdf/assets/report.html | 26 +++ skills/pdf/references/conversion.md | 37 +++ skills/pdf/references/creation.md | 48 ++++ skills/pdf/references/dependencies.md | 40 ++++ skills/pdf/references/design.md | 67 ++++++ skills/pdf/references/editing.md | 54 +++++ skills/pdf/references/operations.md | 164 +++++++++++++ skills/pdf/references/reading-and-ocr.md | 44 ++++ skills/pdf/scripts/_pdf_common.py | 9 + skills/pdf/scripts/_render_html.cjs | 139 +++++++++++ skills/pdf/scripts/compile_latex.py | 57 +++++ skills/pdf/scripts/convert_to_pdf.py | 57 +++++ skills/pdf/scripts/create_design_pdf.py | 107 +++++++++ skills/pdf/scripts/edit_pdf.py | 246 ++++++++++++++++++++ skills/pdf/scripts/ocr_image.py | 47 ++++ skills/pdf/scripts/ocr_text.py | 33 ++- skills/pdf/scripts/render_pdf.py | 2 + skills/pdf/tests/test_pdf_workflows.py | 246 ++++++++++++++++++++ 21 files changed, 1547 insertions(+), 236 deletions(-) create mode 100644 skills/pdf/assets/design.css create mode 100644 skills/pdf/assets/report.html create mode 100644 skills/pdf/references/conversion.md create mode 100644 skills/pdf/references/creation.md create mode 100644 skills/pdf/references/dependencies.md create mode 100644 skills/pdf/references/design.md create mode 100644 skills/pdf/references/editing.md create mode 100644 skills/pdf/references/operations.md create mode 100644 skills/pdf/references/reading-and-ocr.md create mode 100644 skills/pdf/scripts/_render_html.cjs create mode 100644 skills/pdf/scripts/compile_latex.py create mode 100644 skills/pdf/scripts/convert_to_pdf.py create mode 100644 skills/pdf/scripts/create_design_pdf.py create mode 100644 skills/pdf/scripts/edit_pdf.py create mode 100644 skills/pdf/scripts/ocr_image.py create mode 100644 skills/pdf/tests/test_pdf_workflows.py diff --git a/skills/pdf/SKILL.md b/skills/pdf/SKILL.md index 46295b3..4e7fa2b 100644 --- a/skills/pdf/SKILL.md +++ b/skills/pdf/SKILL.md @@ -1,233 +1,55 @@ --- name: pdf -description: "处理本地 PDF 文件或远程 HTTPS PDF 链接,并下载 PDF 任务所需且不超过 25 MiB 的图片、音视频、压缩包和其他 HTTPS 附件;包括源 PDF 安全下载、元数据与页面检查、多引擎分段文本提取和质量检测、扫描页本地 OCR、表格提取、按需页面 PNG 渲染、从文本创建 PDF、合并、拆分、旋转及最终质量校验。当用户提供 .pdf 文件或 HTTPS PDF 地址,或要求总结、读取、识别扫描件、生成、编辑、转换或审阅 PDF 时使用。" +description: "读取、OCR、创建、设计排版、转换和处理 PDF。支持本地文件与 HTTPS PDF、扫描件和任务图像,以原生文本提取及本地 OCR 为主,模型辅助疑难复核;提供 HTML/CSS 出版排版、文本生成、Office/LaTeX 导出、表单、页面与元数据操作。用户要求阅读或总结 PDF、识别扫描件、制作报告/简历/提案 PDF 或编辑现有 PDF 时使用;Office 主文档编辑由对应 skill 处理。" --- -# PDF 处理 - -## 强制执行规则 - -当前智能体不能执行 shell、任意 Python 代码或系统命令。只能通过 `execute_skill_script` 调用本 Skill 中真实存在的固定脚本。 - -- 只调用下表列出的可执行脚本。 -- 不执行 `scripts/` 目录,不执行内部模块 `scripts/_pdf_common.py`。 -- 不传 `-c`、`python3`、`ls`、`pdftoppm` 或其他 shell/系统命令作为脚本参数。 -- 不创建或猜测脚本清单以外的文件。 -- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。 -- 收到 `ok: false` 时,依据 `error` 调整合法参数或向用户说明失败原因,不要把参数改传给其他脚本碰运气。 -- 阅读或总结时只使用 `pages[]` 中 `usable_for_summary: true` 的文本。`needs_ocr: false` 时不得为了“常规检查”继续 OCR、渲染或调用图片识别。 -- 远程 PDF 源文件只交给 `download_pdf.py`;任务所需的远程图片、视频、音频、压缩包或其他附件只交给 `download_attachment.py`。不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。 -- PDF 最终文件一律写入 `/usr/local/src/pdf/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/pdf/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。 -- 不把 PDF 密码作为脚本参数;工具调用参数可能进入运行日志。 - -## 脚本清单 - -| 脚本 | 用途 | 底层能力 | -| --- | --- | --- | -| `scripts/download_pdf.py` | 下载并校验远程 HTTPS PDF | `urllib`、`pypdf` | -| `scripts/download_attachment.py` | 下载图片、音视频、压缩包等通用 HTTPS 附件 | `urllib`、HEAD 大小探测、流式硬限制 | -| `scripts/inspect_pdf.py` | 检查页数、加密、元数据、页面尺寸和表单数量 | `pypdf` | -| `scripts/extract_text.py` | 多引擎提取、质量检测并分段返回正文 | Poppler `pdftotext`、`pdfplumber`;`pypdf` 校验 | -| `scripts/ocr_text.py` | 对指定扫描页执行离线 OCR 并返回可靠文字 | RapidOCR、ONNX Runtime、Poppler `pdftoppm` | -| `scripts/extract_tables.py` | 按页提取表格 | `pdfplumber` | -| `scripts/render_pdf.py` | 把指定页面渲染为 PNG | Poppler `pdftoppm` | -| `scripts/create_pdf.py` | 从 UTF-8 文本或 Markdown 创建 PDF | `reportlab`、`pypdf` | -| `scripts/manage_pdf.py` | 合并、拆分或旋转 PDF | `pypdf` | -| `scripts/cleanup_pdf_temp.py` | 安全删除本次任务临时目录 | Python 文件 API | - -环境已预置所有依赖。不要安装依赖,也不要提示用户安装依赖。 - -## 标准流程 - -1. 为任务选择简短目录名,把中间文件放在 `/usr/local/src/pdf/tmp/<任务名>/`。 -2. 远程 HTTPS 链接先调用 `download_pdf.py`;本地文件直接进入下一步。 -3. 调用 `inspect_pdf.py` 检查文件。遇到加密 PDF 时停止处理,请用户提供已解密副本;当前固定脚本不接收密码。 -4. 阅读或总结时调用 `extract_text.py`。结果为 `usable_for_summary: true` 时使用可靠页文本并根据游标继续;同时为 `needs_ocr: false` 时直接回答,不调用 OCR、渲染或图片识别。 -5. 只有 `extract_text.py` 返回 `needs_ocr: true` 时,才对 `text_quality.suspect_pages` 调用 `ocr_text.py`。原生可靠文本优先,OCR 只补齐可疑页,不重复识别正常页。 -6. `ocr_text.py` 会在脚本内部临时渲染指定页面并交给本地 RapidOCR,完成后自动删除 PNG;普通扫描件解析不调用大模型识图,也不需要先调用 `render_pdf.py`。 -7. 仅在用户明确要求检查视觉版式,或任务涉及创建/修改 PDF 时调用 `render_pdf.py`。 -8. 创建或修改后的最终 PDF 写入 `/usr/local/src/pdf/`,重新执行检查、文本提取和全部页面渲染。 -9. 最终产物位于临时目录之外且不再需要缓存时,调用 `cleanup_pdf_temp.py` 清理本次任务目录。 - -## 下载远程 PDF - -只接受 HTTPS 地址。完整保留 URL 及查询参数,不在回复、日志摘要或文件名中复述敏感参数。 - -调用 `scripts/download_pdf.py`: - -```text ---url 'https://example.com/document.pdf' --output '/usr/local/src/pdf/tmp/<任务名>/source.pdf' -``` - -可选参数: - -- `--timeout <秒>`:默认 `60`。 -- `--max-bytes <字节数>`:默认且最高 `26214400`(25 MiB),只允许设置更小的限制。 -- `--overwrite`:仅在目标是本次任务生成的缓存时使用。 - -脚本会创建父目录、流式下载、阻止 HTTPS 重定向降级到 HTTP,并验证 PDF。成功结果包含 `path`、`size_bytes`、`page_count` 和 `encrypted`。 - -## 下载通用附件 - -需要下载作为 PDF 任务素材的图片、视频、音频、压缩包或其他文件时,调用 `scripts/download_attachment.py`: - -```text ---url 'https://example.com/asset.bin?signature=...' --output '/usr/local/src/pdf/tmp/<任务名>/asset.bin' -``` - -只接受 HTTPS 地址,`output` 可使用任意附件扩展名。可选参数只有 `--timeout <1-600>`(默认 `60`)和 `--overwrite`。附件上限固定为 25 MiB(26214400 字节),不可调高:脚本先用 HEAD 探测远端声明大小,再检查 GET 响应声明,并在流式接收时持续兜底计数;任一阶段发现超限都会返回 `ok: false` 和明确的“已拒绝下载”错误,且不会发布部分文件。 - -成功结果包含 `path`、实际 `size_bytes`、`declared_size_bytes`、`size_limit_bytes`、`size_probe` 和 `content_type`。本脚本不校验文件业务格式;远程 PDF 源文件仍使用 `download_pdf.py`。 - -## 检查 PDF - -调用 `scripts/inspect_pdf.py`: - -```text ---input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' -``` - -使用返回的 `page_count`、`encrypted`、`metadata`、`page_layouts` 和 `form_field_count` 判断后续处理方式。不要直接调用 `pdfinfo`。 - -## 提取正文 - -首次调用 `scripts/extract_text.py`: - -```text ---input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' -``` - -默认使用 `auto` 引擎:先由 Poppler `pdftotext` 提取;结果不可用或命令不可用时自动尝试 `pdfplumber`,并可逐页选择质量更好的结果。脚本使用 `pypdf` 获取标准页数,并拒绝把页数不一致的提取结果当作成功。不要直接执行 `pdftotext`。 - -默认单次最多处理 8 页、返回 24000 个字符。可使用: - -- `--start-page <页码>`、`--end-page <页码>`:页码从 `1` 开始。 -- `--start-offset <字符偏移>`:继续读取被字符上限截断的同一页;大于 `0` 时同时传入上次返回的 `next_engine`。 -- `--max-pages <页数>`、`--max-chars <字符数>`:控制单次输出。 -- `--layout`:仅在需要尽量保留版面空格时使用。 -- `--engine `:首次及跨页提取保持 `auto`;同页字符续读时传入上次返回的 `next_engine`。 -- `--timeout <秒>`:Poppler 提取超时,默认 `120`。 - -先检查 `usable_for_summary` 和 `text_quality.status`: - -- `usable_for_summary: true`:只使用 `pages[]` 中同样标为 `usable_for_summary: true` 的 `text`;可疑页的文本会被置空。如果 `has_more: true`,始终传回 `next_page` 和 `next_offset`。仅当 `next_offset` 大于 `0` 时,把非空的 `next_engine` 传给 `--engine` 以固定同页字符游标;这种调用只续读当前页。当前页完成后返回的 `next_offset` 为 `0`,此时不要传 `--engine`,让下一页重新使用 `auto`。保留首次调用的 `--end-page`(如果指定)及其他选项,直至 `has_more: false`。 -- `usable_for_summary: false`:本批次没有可靠文本,不要使用返回内容。查看 `engine_attempts`、`text_quality.reasons`、`text_quality.suspect_pages` 和 `needs_ocr`;若 `has_more: true`,仍按跨页游标继续检查后续批次,避免漏掉后续可搜索文本。 -- `needs_ocr: true`:一个或多个页面未得到可靠文本。把 `text_quality.suspect_pages` 中实际需要阅读的页码传给 `ocr_text.py`;不要先调用 `render_pdf.py`,也不要把临时图片交给大模型。 - -`complete_text_coverage: true` 表示本批次所有页面均有可靠文本。`text_quality` 按页检测空白或过少文本、页面实际可见图像覆盖过大但文字不足、`(cid:...)`、Unicode 替换字符、异常控制字符及外观像汉字的部首字符;`pages[].extractor` 表示该页最终采用的引擎。`status: mixed` 表示同一批次同时包含可靠页和可疑页:可先使用可靠页文本,同时只核验 `suspect_pages`。不要只根据“肉眼看起来能读”判定提取结果可靠。 - -## 本地 OCR 扫描页 - -仅当 `extract_text.py` 返回 `needs_ocr: true` 时调用 `scripts/ocr_text.py`。`--pages` 必须明确指定 `text_quality.suspect_pages` 中要读取的页,单次最多 4 页: - -```text ---input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --pages '2,5-6' -``` - -默认以 260 DPI 临时渲染,并使用镜像中预置的 RapidOCR 与 ONNX Runtime 在本地识别。脚本不会联网下载模型,不会保留渲染图片,也不会调用大模型视觉能力。可选参数: - -- `--dpi <150-400>`:文字过小或识别质量不足时适度提高,默认 `260`。 -- `--max-chars <字符数>`:默认 `24000`,最大 `60000`。 -- `--timeout <秒>`:每页 Poppler 渲染超时,默认 `180`。 -- `--start-offset <字符偏移>`:续读被字符上限截断的单页;使用时 `--pages` 只能包含该页。 - -只使用 `pages[]` 中 `usable_for_summary: true` 的 `text`。`status: empty`、`sparse` 或 `low_confidence` 的页面文本会被置空,并通过 `needs_review: true` 提醒人工检查。 - -如果 `has_more: true`: - -- `next_offset > 0`:用 `--pages --start-offset ` 续读同一页。 -- `next_offset = 0`:用返回的 `remaining_pages` 继续下一批。 -- 同页续读完成后,再处理先前返回的其他 `remaining_pages`。 - -OCR 结果中的 `mean_confidence`、`line_count`、`render_seconds` 和 `ocr_seconds` 仅用于判断质量与性能。原生提取成功的页面始终采用 `extract_text.py` 结果,不用 OCR 覆盖。 - -## 提取表格 - -调用 `scripts/extract_tables.py`: - -```text ---input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --start-page 1 -``` - -默认单次最多处理 5 页、20 个表格和 2000 个单元格。可用 `--end-page`、`--start-table`、`--max-pages`、`--max-tables`、`--max-cells` 调整。若 `has_more: true`,把 `next_page` 传给 `--start-page`、`next_table` 传给 `--start-table` 后继续,并保留首次调用的 `--end-page`(如果指定)及其他提取选项。 - -## 渲染页面 - -只有满足以下任一条件时才调用 `scripts/render_pdf.py`: - -- 用户明确要求审阅版式、图表、印章、公式或页面外观; -- 创建或修改 PDF 后进行最终视觉检查。 - -不要因为输入是 PDF、需要总结、需要 OCR 或需要检查首页就自动调用本脚本;OCR 的临时渲染由 `ocr_text.py` 内部完成。调用脚本时不要直接执行 `pdftoppm`: - -```text ---input '/usr/local/src/pdf/tmp/<任务名>/source.pdf' --output-dir '/usr/local/src/pdf/tmp/<任务名>/rendered' --start-page 1 -``` - -默认 150 DPI、单次最多 10 页。可使用 `--end-page`、`--max-pages`、`--dpi`、`--timeout` 和 `--overwrite`。若 `has_more: true`,使用 `next_page` 继续,并保留首次调用的 `--end-page`(如果指定)、输出目录及其他渲染选项。脚本返回标准化的 `page-0001.png` 文件路径。 - -对文字较小或图表密集的页面提高 DPI。使用可用的图像查看工具检查返回的 PNG,不要尝试把图片路径交给下载脚本。 - -## 创建 PDF - -先使用 `write_file` 把内容写为 UTF-8 `.txt` 或 `.md` 文件,再调用 `scripts/create_pdf.py`: - -```text ---input '/usr/local/src/pdf/tmp/<任务名>/content.md' --output '/usr/local/src/pdf/<文件名>.pdf' --title '文档标题' -``` - -脚本支持 Markdown 标题、项目符号和简单表格,自动选择可嵌入的 Unicode 字体并添加页码。可选参数: - -- `--page-size ` -- `--font-path ` -- `--font-size <字号>` -- `--margin ` -- `--overwrite` - -输入内容只使用 ASCII 连字符 `-`;脚本也会把常见 Unicode 横线规范化为 ASCII 连字符。 - -## 合并、拆分与旋转 - -调用 `scripts/manage_pdf.py`,第一个参数必须是操作名。 - -合并: - -```text -merge --input 'a.pdf' --input 'b.pdf' --output '/usr/local/src/pdf/merged.pdf' -``` - -拆分指定范围: - -```text -split --input 'source.pdf' --output-dir '/usr/local/src/pdf/split' --range 1-3 --range 4-6 -``` - -不传 `--range` 时每页生成一个 PDF。 - -旋转指定页面: - -```text -rotate --input 'source.pdf' --output '/usr/local/src/pdf/rotated.pdf' --pages '1,3-5' --degrees 90 -``` - -`--degrees` 只能是 `90`、`180` 或 `270`;不传 `--pages` 时旋转全部页面。目标已存在且确认可覆盖时添加 `--overwrite`。 - -## 清理临时目录 - -调用 `scripts/cleanup_pdf_temp.py`: - -```text ---task-dir '/usr/local/src/pdf/tmp/<任务名>' -``` - -脚本只允许删除 `/usr/local/src/pdf/tmp/` 下一级任务目录,拒绝删除根目录、仓库目录或其他路径。 - -## 质量要求 - -- 不覆盖用户提供的源文件。 -- 创建或修改后重新检查页数、页面尺寸、加密状态和文本可读性。 -- 扫描件先做原生文字检测,再只 OCR 可疑页;不得把低置信度 OCR 文本当作可靠正文。 -- 逐页确认没有裁切、重叠、溢出、乱码、黑方块、错误分页或异常空白页。 -- 检查标题层级、段落间距、页边距、表格、图表、图片、页码及章节衔接。 -- 引用和参考文献必须可读,不得残留工具令牌、占位符或临时路径。 -- 只有最新渲染结果不存在可见缺陷时才交付创建或修改后的 PDF。 +# PDF 读取、设计与处理 + +## 运行约定 + +- 当前机器人不能直接运行 Bash、Python、Node.js 或系统命令。只通过 `execute_skill_script` 调用下列真实存在的固定脚本;参数是文件、文本、页码或受控 JSON,不能传 shell 命令、`-c`、`eval`、代码片段或解释器命令。固定脚本可在内部调用预置引擎,调用者不直接执行底层程序。 +- `scripts/_pdf_common.py`、`scripts/_render_html.cjs` 是内部实现,不直接执行;不调用原 pdf-skill 的 shell 安装器、通用 Python/Node CLI,不创建临时可执行脚本。 +- 依赖只在基础镜像构建时安装。任务中不运行 pip/npm/apt、不下载浏览器、TeX 包或 OCR 模型;缺失时报告具体依赖和镜像需更新,不能假装处理成功。 +- 最终 PDF 写入 `/usr/local/src/pdf/`;缓存、源 HTML/JSON、预览放在 `/usr/local/src/pdf/tmp/<任务名>/`。传绝对路径,保留用户源文件。`--overwrite` 仅用于本任务已生成的旧产物。 +- 每次检查返回 JSON;`ok: false` 先处理原因。`requires_visual_review`、`needs_review` 或 warnings 需要实际核验,程序运行成功不代表内容和版式通过。 +- HTTPS PDF 用 `download_pdf.py`,其他素材用 `download_attachment.py`。不在回复中复述敏感 URL 查询参数;加密 PDF 请用户提供已解密副本,密码不进入工具参数。 + +## 按任务读取 + +| 任务 | 指南 | +| --- | --- | +| 阅读、总结、扫描页、图片文字、图表辅助识别 | [读取与 OCR](references/reading-and-ocr.md) | +| 创建报告、提案、简历、学术或品牌 PDF | [设计规范](references/design.md) + [HTML 创建接口](references/creation.md) | +| Office/LaTeX 导出,PDF 内容重建为 Office | [转换](references/conversion.md) | +| 表单填写、裁剪、嵌入图片、元数据 | [编辑](references/editing.md) | +| 下载、分段提取、表格、简单文本 PDF、合并/拆分/旋转、渲染与清理 | [原有操作接口](references/operations.md) | +| 镜像缺包或能力边界 | [依赖说明](references/dependencies.md) | + +## 固定脚本 + +| 脚本 | 用途 | +| --- | --- | +| `scripts/download_pdf.py` | 安全下载并校验不超过 25 MiB 的 HTTPS PDF | +| `scripts/download_attachment.py` | 下载不超过 25 MiB 的任务附件 | +| `scripts/inspect_pdf.py` | 页数、尺寸、加密、元数据与表单数量 | +| `scripts/extract_text.py` | 多引擎正文提取、逐页质量检测和字符游标 | +| `scripts/ocr_text.py` | 对指定 PDF 页执行离线 OCR,标记局部疑难区域 | +| `scripts/ocr_image.py` | 对单张任务图片执行离线 OCR | +| `scripts/extract_tables.py` | 原生 PDF 表格分页提取 | +| `scripts/render_pdf.py` | 按页输出 PNG,供版式检查或疑难辅助复核 | +| `scripts/create_pdf.py` | 简单文本/Markdown 生成 PDF | +| `scripts/create_design_pdf.py` | 静态 HTML/CSS、图表和公式设计排版 | +| `scripts/convert_to_pdf.py` | Office 文件导出 PDF | +| `scripts/compile_latex.py` | 仅用镜像缓存资源编译 LaTeX | +| `scripts/edit_pdf.py` | 表单、裁剪、元数据和嵌入图片 | +| `scripts/manage_pdf.py` | 合并、拆分、旋转 | +| `scripts/cleanup_pdf_temp.py` | 清理本任务临时目录 | + +## 工作原则 + +1. 阅读先检查 PDF,再提取可靠原生文本;扫描页或图片文字以本地 OCR 为主。只有低置信度、手写、复杂表格/公式、阅读顺序冲突或非文本图形理解需要时,才用大模型复核相关页/区域。不要把整个扫描件直接交给大模型代替 OCR。 +2. 创建设计先确认读者、用途、内容与输出限制,按需选择封面、配色和字体层级。用户模板、品牌、大纲、语言与篇幅优先;不强加独立封面,不为凑页数填充或删除内容。 +3. 简单文字选 `create_pdf.py`;需要封面、图文、页眉页脚、交叉引用、数学公式时选 `create_design_pdf.py`。通过 `write_file` 写静态内容文件,不写可执行代码。HTML 禁止脚本、事件处理程序和外部资源;公式、流程图由固定引擎本地处理。 +4. 编辑现有 PDF 保留内容与结构。裁剪不等于脱敏;表单字段值写入不等于外观正确;PDF 转 Office 应按提取/OCR 后重建来规划,不能承诺无损逆转换。 +5. 创建、转换或修改后重新检查页数、文本与关键数字,再渲染全部相关页逐页核验封面、字体、表格、公式、图表、页码、裁切和空白页。要求精确页数时使用 `--expected-pages`。修正后检查最新产物,才交付。 +6. 内容引用可核验。用户提供的材料可直接引用;新增时效、专业或不确定事实使用当前可用搜索工具查证,不编造统计、论文或参考文献。图片识别的猜测与原文分开标记。 diff --git a/skills/pdf/agents/openai.yaml b/skills/pdf/agents/openai.yaml index f337328..0f7c8aa 100644 --- a/skills/pdf/agents/openai.yaml +++ b/skills/pdf/agents/openai.yaml @@ -1,4 +1,4 @@ interface: - display_name: "PDF 处理" - short_description: "读取、创建和审阅本地或远程 PDF,按需执行本地 OCR" - default_prompt: "使用 $pdf 下载或读取这个 PDF,优先提取可靠文本,并只对扫描页执行本地 OCR。" + display_name: "PDF 读取与设计" + short_description: "本地 OCR 优先识别,设计排版、转换与编辑 PDF,并完成逐页校验" + default_prompt: "使用 $pdf 读取或制作 PDF;扫描图像先做本地 OCR,疑难内容辅助复核,按内容设计版式并检查最终页面。" diff --git a/skills/pdf/assets/design.css b/skills/pdf/assets/design.css new file mode 100644 index 0000000..11235b4 --- /dev/null +++ b/skills/pdf/assets/design.css @@ -0,0 +1,74 @@ +/* Publication defaults. Author CSS overrides these tokens and styles. */ +:root { + --page-width: 210mm; --page-height: 297mm; + --accent: #8a3a2a; --accent-light: #f5ece7; --cover-bg: #30231f; --cover-text: #faf5ef; + --ink: #202124; --muted: #62666a; --rule: #d8dadd; + --font-display: 'PingFang SC', 'Microsoft YaHei', 'Noto Sans CJK SC', Arial, sans-serif; + --font-body: 'SimSun', 'Noto Serif CJK SC', 'Times New Roman', serif; + --font-sans: 'PingFang SC', 'Microsoft YaHei', 'Noto Sans CJK SC', Arial, sans-serif; +} +@page { size: A4; margin: 24mm 24mm 22mm; + @top-left { content: string(chapter); font: 8pt var(--font-sans); color: #62666a; } + @bottom-right { content: counter(page); font: 8pt var(--font-sans); color: #62666a; } +} +@page cover { size: A4; margin: 0; + @top-left { content: none; } @bottom-right { content: none; } +} +* { box-sizing: border-box; } +html, body { margin: 0; padding: 0; } +body { color: var(--ink); font: 10.5pt/1.65 var(--font-body); overflow-wrap: break-word; } +h1, h2, h3 { font-family: var(--font-display); color: var(--accent); break-after: avoid; line-height: 1.3; } +h1 { font-size: 22pt; margin: 24pt 0 12pt; string-set: chapter content(text); } +h2 { font-size: 15pt; margin: 18pt 0 8pt; } +h3 { font-size: 11.5pt; margin: 12pt 0 6pt; } +p { margin: 0 0 8pt; orphans: 3; widows: 3; } +a { color: var(--accent); text-decoration: underline; } +strong { font-family: var(--font-sans); } +.section-start { break-before: page; } +.eyebrow { font: 8.5pt/1.4 var(--font-sans); letter-spacing: .1em; } +.lead { font-size: 13pt; line-height: 1.7; } +.cover { page: cover; break-after: page; width: var(--page-width); height: var(--page-height); + padding: 30mm 27mm; position: relative; display: flex; flex-direction: column; justify-content: center; + background: var(--cover-bg); color: var(--cover-text); font-family: var(--font-sans); } +.cover h1 { margin: 16pt 0; color: inherit; font: 700 40pt/1.2 var(--font-display); string-set: none; } +.cover .subtitle { font-size: 14pt; line-height: 1.6; max-width: 140mm; } +.cover .metadata { margin-top: 25mm; font-size: 10pt; line-height: 1.8; } +.cover .accent-line { width: 24mm; border-top: 3pt solid var(--accent); margin: 14pt 0; } +.cover-fullbleed::before { content: ''; position: absolute; top: 0; right: 20mm; width: 12mm; height: 28mm; background: var(--accent); } +.cover-split { background: #f7f5f1; color: var(--ink); padding-left: 101mm; padding-right: 18mm; } +.cover-split::before { content: ''; position: absolute; inset: 0 auto 0 0; width: 42%; background: var(--cover-bg); border-right: 2mm solid var(--accent); } +.cover-split h1 { font-size: 28pt; } +.cover-typographic, .cover-minimal, .cover-frame { background: #fafaf7; color: var(--ink); } +.cover-typographic h1 { font-size: 48pt; } +.cover-minimal { border-left: 3mm solid var(--accent); } +.cover-minimal h1 { font-weight: 400; } +.cover-frame::before { content: ''; position: absolute; inset: 11mm; border: 1pt solid var(--accent); pointer-events: none; } +.cover-frame { text-align: center; align-items: center; } +.cover-editorial::before { content: attr(data-mark); position: absolute; top: 12mm; right: 10mm; opacity: .07; font: 180pt/1 var(--font-display); } +.cover-editorial h1 { font-size: 48pt; } +figure { margin: 14pt 0; break-inside: avoid; } +img, svg { max-width: 100%; } +img { height: auto; } +figcaption, .caption { font: 8.5pt/1.5 var(--font-sans); color: var(--muted); margin-top: 6pt; } +.mermaid { text-align: center; margin: 12pt 0; break-inside: avoid; } +.mermaid svg { max-height: 180mm; } +.math-display { margin: 12pt 0; text-align: center; break-inside: avoid; } +.katex { font-size: 1.05em; } +table { width: 100%; border-collapse: collapse; margin: 12pt 0; font: 9.5pt/1.5 var(--font-sans); } +thead { display: table-header-group; } +tr { break-inside: avoid; } +th { text-align: left; background: var(--accent); color: white; border-top: 1.5pt solid var(--accent); } +th, td { padding: 7pt 8pt; overflow-wrap: anywhere; vertical-align: top; } +td { border-bottom: .5pt solid var(--rule); } +tbody tr:nth-child(even) { background: var(--accent-light); } +tbody tr:last-child td { border-bottom: 1.5pt solid var(--accent); } +.three-line th { background: transparent; color: var(--ink); border-top: 1.5pt solid var(--ink); border-bottom: .8pt solid var(--ink); } +.three-line td { border: 0; } .three-line tbody tr { background: transparent; } +.three-line tbody tr:last-child td { border-bottom: 1.5pt solid var(--ink); } +.numeric { text-align: right; font-variant-numeric: tabular-nums; } +blockquote, .callout, .theorem { margin: 12pt 0; padding: 8pt 12pt; border-left: 2pt solid var(--accent); background: var(--accent-light); } +pre { font: 9pt/1.5 'Courier New', monospace; padding: 10pt; background: #f5f5f5; white-space: pre-wrap; overflow-wrap: anywhere; } +ul, ol { padding-left: 20pt; } li { margin-bottom: 4pt; } +.references { font-size: 9pt; line-height: 1.6; } +.toc a { display: block; margin: 6pt 0; text-decoration: none; } +.toc a::after { content: ' ' target-counter(attr(href), page); float: right; } diff --git a/skills/pdf/assets/report.html b/skills/pdf/assets/report.html new file mode 100644 index 0000000..5f315d2 --- /dev/null +++ b/skills/pdf/assets/report.html @@ -0,0 +1,26 @@ + + +报告标题 + + + +
+

报告类型 · 年份

+

报告标题

+
+

一句说明报告对象、范围和阅读目的的副标题。

+ +
+
+

核心结论

+

用可核对的证据说明主要结论,并区分事实与判断。

+

数据与依据

+ +
指标观察来源
示例指标替换为真实数据或明确标注的示例填写可验证来源
+
适用范围

说明时间范围、样本、计算口径与不确定性。

+

建议与下一步

+

建议应能追溯到证据;需要决策的事项写清楚条件与影响。

+
+ diff --git a/skills/pdf/references/conversion.md b/skills/pdf/references/conversion.md new file mode 100644 index 0000000..ca101f4 --- /dev/null +++ b/skills/pdf/references/conversion.md @@ -0,0 +1,37 @@ +# Office、LaTeX 与 PDF 转换 + +只调用固定脚本。不能直接运行 LibreOffice、Tectonic、Bash、Python 或 Node,也不在任务中安装依赖。 + +## Office → PDF + +Office 主文档的编辑、公式计算、图表和版式由 docx/xlsx/pptx 等对应 skill 完成。已有完成的源文件可调用 `scripts/convert_to_pdf.py`: + +```text +--input '/usr/local/src/pdf/tmp/task/report.docx' --output '/usr/local/src/pdf/report.pdf' +``` + +支持 DOCX/DOC/ODT/RTF、PPTX/PPT/ODP、XLSX/XLS/ODS,最大 25 MiB。`--timeout` 默认 180 秒,1–600;可用 `--overwrite` 覆盖本任务旧产物。每次转换使用独立 LibreOffice profile 和临时副本,不修改源文件。 + +转换前确认 Excel 公式已重算、打印范围正确;核对字体替换、图表、分页和页数。转换可能有版式差异,必须渲染检查。CSV 和 HTML 分别先走 xlsx 或 HTML 设计接口,不能含糊地自动推断编码和布局。 + +## PDF → Office + +PDF 是固定版面,不能承诺用 LibreOffice 直接得到结构完整的 Word、Excel 或 PPT。先用原生提取/OCR 得到可靠文本和表格,再由对应 skill 的固定写入接口重建;明确哪些结构可编辑、哪些需要保留图片。扫描件不会因为换扩展名就变成可编辑文本。 + +需要高保真还原时先确认重点是视觉一致还是编辑结构;保留原 PDF 对照。不得把每页截图贴入 Word 后声称正文可编辑。 + +## LaTeX → PDF + +用户明确提供 LaTeX 模板或源文件时调用 `scripts/compile_latex.py`: + +```text +--input '/usr/local/src/pdf/tmp/task/main.tex' --output '/usr/local/src/pdf/paper.pdf' +``` + +输入 `.tex` 最大 2 MiB,相关图片和被引用的 `.tex` 放在本任务目录;超时默认 180 秒,范围 1–600。固定脚本内部调用 Tectonic,禁用 shell escape,只使用镜像构建时缓存的 TeX 资源。缺失包、未缓存模板、编译失败或超时都明确报错,不能安装或偷偷切换到联网编译。 + +基础镜像预热 ctex、常用数学、表格、图片、几何和超链接包;特殊模板或包可能还需维护者补充镜像。Tectonic 自行处理常见重跑与引用,不把同一编译无意义重复多次。 + +保留用户模板的语言、字体、章节与引用规范。中文模板可使用 `ctexart` 和已缓存的 Fandol 字体;新的模板需确认所用字体实际存在。若只有少量数学公式而没有 LaTeX 模板,HTML + KaTeX 即可,无需编写完整 LaTeX 文档。 + +返回页数、警告和视觉复核标记。检查 undefined references、缺字、Overfull 等具体问题;输出成功后仍需提取与渲染检查,不把“有 PDF 文件”作为通过标准。 diff --git a/skills/pdf/references/creation.md b/skills/pdf/references/creation.md new file mode 100644 index 0000000..3852e57 --- /dev/null +++ b/skills/pdf/references/creation.md @@ -0,0 +1,48 @@ +# HTML/CSS 设计排版接口 + +先读 [设计规范](design.md)。简单文本使用 `create_pdf.py`;图文报告、封面、页眉页脚、目录、公式和流程图使用 `create_design_pdf.py`。 + +## 调用 + +通过 `write_file` 将内容写为 UTF-8 `.html`,以 `` 声明标准模式,并包含 ``。通过 `execute_skill_script` 调用 `scripts/create_design_pdf.py`,不直接调用 Node 或 Chromium: + +```text +--input '/usr/local/src/pdf/tmp/task/report.html' --output '/usr/local/src/pdf/report.pdf' +``` + +可选 `--css <本地CSS>`、`--page-size A4|LETTER`、`--expected-pages <1–200>`、`--timeout <10–600>`(默认 180)和 `--overwrite`。HTML/CSS 单文件不超过 2 MiB。先读取 `assets/report.html` 了解结构,按实际内容编写;模板中的示例文字不可直接交付。 + +`assets/design.css` 由固定渲染器自动注入,提供变量、标题、六种封面、三线表、代码、引用、图注与目录。用户 HTML 内的 CSS 可覆盖默认值,`--css` 最后应用。不需要手动加载该样式或任何 JS 库。 + +## 静态内容与本地资源 + +- 只写静态 HTML/CSS/SVG。禁止 `", '', + '', + 'x', + '', + '']: + with self.subTest(html=html), self.assertRaises(ValueError): + create_design_pdf.StaticHTML().feed(html) + + def test_static_math_diagram_and_svg_are_accepted(self): + html = '

报告

E=mc^2

flowchart LR\nA-->B
' + create_design_pdf.StaticHTML().feed(html) + + +if __name__ == "__main__": + unittest.main()