From 33809a79e4ff89245c00ee43e24fb8a69c0b9d26 Mon Sep 17 00:00:00 2001 From: hp0912 <809211365@qq.com> Date: Wed, 9 Sep 2026 11:28:01 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E8=9E=8D=E5=90=88=E8=B1=86=E5=8C=85exc?= =?UTF-8?q?el=20skills?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- skills/xlsx/SKILL.md | 261 ++--------- skills/xlsx/agents/openai.yaml | 6 +- skills/xlsx/evals/evals.json | 35 ++ skills/xlsx/references/data-analysis.md | 112 +++++ skills/xlsx/references/dependencies.md | 23 + skills/xlsx/references/modeling.md | 103 +++++ skills/xlsx/references/reporting.md | 33 ++ skills/xlsx/references/text-analysis.md | 34 ++ skills/xlsx/references/workbook-operations.md | 198 ++++++++ skills/xlsx/scripts/_xlsx_data.py | 258 +++++++++++ skills/xlsx/scripts/analyze_workbook.py | 249 +++++++++++ skills/xlsx/scripts/apply_workbook.py | 53 ++- skills/xlsx/scripts/inspect_workbook.py | 8 +- skills/xlsx/scripts/model_workbook.py | 423 ++++++++++++++++++ skills/xlsx/scripts/recalculate_workbook.py | 4 +- skills/xlsx/tests/test_data_workflows.py | 218 +++++++++ 16 files changed, 1784 insertions(+), 234 deletions(-) create mode 100644 skills/xlsx/evals/evals.json create mode 100644 skills/xlsx/references/data-analysis.md create mode 100644 skills/xlsx/references/dependencies.md create mode 100644 skills/xlsx/references/modeling.md create mode 100644 skills/xlsx/references/reporting.md create mode 100644 skills/xlsx/references/text-analysis.md create mode 100644 skills/xlsx/references/workbook-operations.md create mode 100644 skills/xlsx/scripts/_xlsx_data.py create mode 100644 skills/xlsx/scripts/analyze_workbook.py create mode 100644 skills/xlsx/scripts/model_workbook.py create mode 100644 skills/xlsx/tests/test_data_workflows.py diff --git a/skills/xlsx/SKILL.md b/skills/xlsx/SKILL.md index 951fbae..06a27f0 100644 --- a/skills/xlsx/SKILL.md +++ b/skills/xlsx/SKILL.md @@ -1,235 +1,58 @@ --- name: xlsx -description: "创建、读取、编辑、修复、转换、重算、校验和渲染本地或远程 HTTPS Excel 工作簿及表格数据,并下载 Excel 任务所需且不超过 25 MiB 的图片、音视频、压缩包和其他 HTTPS 附件。用户提到 Excel、电子表格、工作簿、工作表、单元格、公式、图表、数据清洗,或提供 HTTPS Excel/CSV/TSV 地址、.xlsx、.xlsm、.xltx、.xls、.csv、.tsv 文件时使用;支持源文件安全下载、保留现有样式、批量写入、公式与缓存值检查、表格/图表/图片/数据验证/条件格式、旧格式转换和逐页视觉检查。最终交付物必须是电子表格文件;若主要交付物是 Word、PDF、HTML、数据库程序或在线 Google Sheets,则不要使用。" +description: "创建、读取、编辑、转换、重算和渲染 Excel/CSV/TSV,进行数据清洗、分组与交叉汇总、业务洞察、文本归类与翻译、综合评价、预测、回归分类、聚类异常检测和资源优化。用户提供电子表格文件或 HTTPS 地址,要求处理表格数据、分析工作簿、制作报表或模板时使用;也支持将口述数据整理为 Excel。独立 Word/PDF/HTML 制作、在线 Google Sheets 和与电子表格无关的通用编程使用对应工具。" --- -# Excel 工作簿处理 +# Excel 工作簿与数据分析 -## 强制执行规则 +## 运行约定 -当前智能体不能直接执行 shell、任意 Python 代码或系统命令。只能通过 `execute_skill_script` 调用本 Skill 中真实存在的固定 Python 脚本。 +- 当前机器人通过 `execute_skill_script` 调用本 skill 的固定 Python 脚本。所有脚本与资源路径相对 xlsx 根目录;不执行内部模块,不把 shell、任意 Python 代码、系统命令当作脚本或参数。 +- 工具链以 xlsx 为主:`openpyxl` 读写和图表,LibreOffice 重算与转换,Poppler 渲染;pandas 负责结构化处理。仅模型求解用 SciPy / scikit-learn。依赖在基础镜像中构建安装,任务运行时不安装;依赖缺失时明确说明镜像需更新,不能假装已完成计算。见 [依赖与能力边界](references/dependencies.md)。 +- 源文件保留。最终工作簿写入 `/usr/local/src/excel/`,下载缓存、分析 JSON、预览和中间文件放在 `/usr/local/src/excel/tmp/<任务名>/`。传绝对路径,输出根目录由脚本校验。 +- 每次检查返回的 JSON,`ok: false` 时先处理原因;`status: errors_found` 不能作为通过。`--overwrite` 只用于本次任务生成的旧产物。 +- 用户格式与处理范围优先,其次源模板。只修改完成任务必需的区域;不因排版删除内容或截断原文,不擅自重命名、删除源工作表。 +- 用户只要解释或统计结论时,可直接回复;需要文件时交付可编辑工作簿。额外报告、HTML 或压缩包按实际要求生成,不固定增加产物数量。发送文件仅使用当前环境实际可用且已获授权的工具。 -- 只调用下表列出的可执行脚本,不执行 `scripts/` 目录或内部模块 `scripts/_xlsx_common.py`。 -- 不把 `python3`、`soffice`、`libreoffice`、`pdftoppm`、`zip`、`unzip`、`rm` 或其他系统命令作为脚本参数。 -- 外部程序只允许由固定脚本在内部以无 shell 参数数组方式调用。 -- 每次检查脚本返回的 JSON;只有 `ok` 为 `true` 时才继续。`status: errors_found` 虽然表示脚本成功运行,但工作簿不合格,必须修复。 -- 远程 Excel/CSV/TSV 源文件只交给 `download_workbook.py`;任务所需的远程图片、视频、音频、压缩包或其他附件只交给 `download_attachment.py`。不要在回复、日志摘要或文件名中复述可能含敏感查询参数的完整 URL。 -- 不覆盖用户提供的源文件。Excel 最终文件一律写入 `/usr/local/src/excel/`,下载缓存、中间文件和渲染结果一律写入 `/usr/local/src/excel/tmp/<任务名>/`。始终传绝对路径;固定脚本会自动创建目录并拒绝该根目录之外的输出。 -- 环境已预置依赖,不安装软件包,也不提示用户安装依赖。 +## 按任务读取指南 -## 脚本清单 - -| 脚本 | 用途 | 底层能力 | -| --- | --- | --- | -| `scripts/download_workbook.py` | 下载并校验远程 HTTPS Excel/CSV/TSV | Python `urllib`、安全 OOXML 解析、`openpyxl` | -| `scripts/download_attachment.py` | 下载图片、音视频、压缩包等通用 HTTPS 附件 | Python `urllib`、HEAD 大小探测、流式硬限制 | -| `scripts/inspect_workbook.py` | 分段读取结构、公式、缓存值和样式 | `openpyxl`、Python `csv` | -| `scripts/apply_workbook.py` | 按受控 JSON 创建或编辑工作簿 | `openpyxl`、Pillow | -| `scripts/convert_workbook.py` | 转换 `.xls/.csv/.tsv/.xlsx/.xlsm/.xltx` | `openpyxl`、LibreOffice | -| `scripts/recalculate_workbook.py` | 重算公式并检查公式错误 | LibreOffice、`openpyxl` | -| `scripts/render_workbook.py` | 把工作簿渲染为逐页 PNG/PDF | LibreOffice、Poppler | - -## 标准流程 - -1. 输入是 HTTPS 地址时,先调用 `download_workbook.py` 下载到本次任务临时目录;本地文件直接进入下一步。 -2. 检查输入格式。旧版 `.xls` 先调用 `convert_workbook.py` 转为 `.xlsx`。 -3. 编辑现有文件前先调用 `inspect_workbook.py`;读取公式和缓存值,确认工作表名称、输入区域、合并区域、表格、图表、隐藏工作表和外部链接。 -4. 用 `apply_workbook.py` 创建或编辑新文件。只修改用户要求的单元格或结构,保留未涉及的公式和样式。 -5. 只要结果中 `formula_count > 0` 或 `requires_recalculation: true`,必须调用 `recalculate_workbook.py`,并确保 `status: success`、`total_errors: 0`。 -6. 先抽查 2–3 个关键公式的引用和计算逻辑,再调用 `inspect_workbook.py` 读取重算后的公式与缓存值;“没有公式错误”不等于“公式逻辑正确”。 -7. 创建或修改后调用 `render_workbook.py`,检查全部页面或按游标分批检查,确认没有裁切、异常分页、乱码、重叠、空白页或不可读图表。 -8. 只有结构检查、公式检查和视觉检查都通过后才交付最终工作簿。 - -## 下载远程工作簿 - -只接受 HTTPS 地址。完整保留 URL 及查询参数传给脚本,但不要在回复、日志摘要或输出文件名中复述敏感参数。 - -调用 `scripts/download_workbook.py`: - -```text ---url 'https://example.com/report.xlsx?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/source.xlsx' -``` - -可选参数: - -- `--timeout <1-600>`:连接和读取超时秒数,默认 `60`。 -- `--max-bytes <字节数>`:默认且最高 `26214400`(25 MiB),只允许设置更小的限制。 -- `--overwrite`:只在目标是本次任务生成的旧缓存时使用。 - -`output` 扩展名必须是 `.xlsx`、`.xlsm`、`.xltx`、`.xltm`、`.xls`、`.csv` 或 `.tsv`。脚本阻止 HTTPS 重定向降级到 HTTP,流式限制大小,先写同目录临时文件,再原子发布;OOXML 会检查 ZIP 路径、成员大小、内容类型并用 `openpyxl` 打开,CSV/TSV 会拒绝二进制或网页响应。实际 OOXML 格式与 `output` 扩展名不一致时,根据错误中的实际格式更正缓存扩展名,再调用同一脚本。 - -成功结果包含 `path`、`size_bytes`、`format` 和 `validation`;OOXML 还包含 `sheet_count`。后续脚本只使用返回的本地 `path`,不再访问原 URL。 - -## 下载通用附件 - -需要下载作为 Excel 任务素材的图片、视频、音频、压缩包或其他文件时,调用 `scripts/download_attachment.py`: - -```text ---url 'https://example.com/asset.bin?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/asset.bin' -``` - -只接受 HTTPS 地址,`output` 可使用任意附件扩展名。可选参数只有 `--timeout <1-600>`(默认 `60`)和 `--overwrite`。附件上限固定为 25 MiB(26214400 字节),不可调高:脚本先用 HEAD 探测远端声明大小,再检查 GET 响应声明,并在流式接收时持续兜底计数;任一阶段发现超限都会返回 `ok: false` 和明确的“已拒绝下载”错误,且不会发布部分文件。 - -成功结果包含 `path`、实际 `size_bytes`、`declared_size_bytes`、`size_limit_bytes`、`size_probe` 和 `content_type`。本脚本不校验文件业务格式;远程 Excel/CSV/TSV 源文件仍使用 `download_workbook.py`。 - -## 检查工作簿 - -调用 `scripts/inspect_workbook.py`: - -```text ---input 'source.xlsx' -``` - -可选参数: - -- `--sheet <名称>`:选择工作表;默认活动工作表。 -- `--start-row <行>`、`--start-column <列>`:读取起点,均从 `1` 开始。 -- `--max-rows <1-200>`、`--max-columns <1-100>`:限制单次输出,默认 `40 × 20`。 - -结果同时给出公式字符串和缓存值: - -- `formula`:原始公式。 -- `cached_value`:Excel/LibreOffice 上次计算后保存的结果。 -- `has_external_links`:为 `true` 时,编辑或重算可能破坏外部链接缓存;默认停止并向用户说明。 -- `selection.has_more`、`next_row`、`next_column`:用于继续读取大表,不要一次返回整本工作簿。 - -CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。 - -## 创建或编辑 - -调用 `scripts/apply_workbook.py`,新建时省略 `--input`,编辑时提供源文件: - -```text ---output '/usr/local/src/excel/result.xlsx' --spec '' -``` - -或: - -```text ---input 'source.xlsx' --output '/usr/local/src/excel/result.xlsx' --spec-file '/usr/local/src/excel/tmp/task/operations.json' -``` - -目标已存在且确认是本次任务的旧产物时才传 `--overwrite`。输入含外部链接时脚本默认拒绝保存;只有用户明确接受缓存值可能丢失的风险时才传 `--allow-external-links`。`.xlsm` 必须继续输出 `.xlsm` 才能保留宏;只有用户明确同意丢弃宏时才输出 `.xlsx` 并传 `--drop-macros`。 - -操作说明顶层字段: - -```json -{ - "properties": { - "title": "销售分析", - "creator": "示例公司" - }, - "calculation_mode": "auto", - "active_sheet": "汇总", - "operations": [] -} -``` - -支持的 `operations[].type`: - -| 类型 | 关键字段 | +| 任务 | 指南 | | --- | --- | -| `add_sheet` | `name`,可选 `index` | -| `remove_sheet` | `sheet` | -| `rename_sheet` | `sheet`、`name` | -| `set_cells` | `sheet`、`cells[]` | -| `write_rows` | `sheet`、`start_cell`、`rows[][]`,可选统一 `style` | -| `append_rows` | `sheet`、`rows[][]` | -| `style_range` | `sheet`、`range`、`style` | -| `clear_range` | `sheet`、`range`,可选 `values/styles/comments/hyperlinks` | -| `insert_rows` / `delete_rows` | `sheet`、`index`、`amount` | -| `insert_columns` / `delete_columns` | `sheet`、`index`、`amount` | -| `merge_cells` / `unmerge_cells` | `sheet`、`range` | -| `set_column_widths` | `sheet`、`widths`,如 `{"A": 18, "B:D": 12}` | -| `set_row_heights` | `sheet`、`heights`,如 `{"1": 28, "2:5": 20}` | -| `freeze_panes` | `sheet`、`cell`;传空值取消冻结 | -| `set_auto_filter` | `sheet`、`range`;传空值取消筛选 | -| `add_table` | `sheet`、`range`、`name`,可选 `style` | -| `add_chart` | `sheet`、`chart_type`、`data_range`、`anchor`;可选 `categories_range/title` | -| `add_image` | `sheet`、`path`、`anchor`;可选像素 `width/height` | -| `add_data_validation` | `sheet`、`range`、`validation_type`、`formula1` | -| `add_conditional_format` | `sheet`、`range`、`rule_type` 及对应规则参数 | -| `set_print` | `sheet`,可选 `print_area/orientation/paper_size/fit_to_width/margins` | -| `set_named_range` | `sheet`、`name`、`range` | +| 下载、检查、创建、模板回填、格式、公式、图表、转换 | [工作簿操作接口](references/workbook-operations.md) | +| 字段理解、数据质量、清洗、连接、去重、业务洞察、交叉汇总 | [数据分析](references/data-analysis.md) | +| 提炼文本、标签、标准化、规则分类、逐格翻译 | [文本处理](references/text-analysis.md) | +| 综合评价、时序预测、回归分类、聚类、异常、资源优化 | [模型接口与方法选择](references/modeling.md) | +| 图表选择、分析结论、报告与财务格式 | [展示与交付](references/reporting.md) | -`set_cells.cells[]` 中每项使用: +## 固定脚本 -```json -{ - "cell": "B2", - "formula": "=SUM(B3:B10)", - "style": { - "font": {"name": "Arial", "size": 11, "bold": true, "color": "FFFFFF"}, - "fill": {"color": "1F4E78"}, - "alignment": {"horizontal": "center", "vertical": "center", "wrap_text": true}, - "number_format": "#,##0.00", - "border": { - "bottom": {"style": "thin", "color": "808080"} - }, - "protection": {"locked": true} - }, - "comment": {"author": "AI", "text": "来源:用户提供的 2026 年预算"}, - "hyperlink": "https://example.com/source" -} -``` +| 脚本 | 用途 | +| --- | --- | +| `scripts/download_workbook.py` | 下载并校验不超过 25 MiB 的 HTTPS Excel/CSV/TSV | +| `scripts/download_attachment.py` | 下载本任务所需且不超过 25 MiB 的 HTTPS 附件 | +| `scripts/inspect_workbook.py` | 分段检查结构、公式、缓存、样式、合并区域和外部链接 | +| `scripts/apply_workbook.py` | 受控 JSON 创建/编辑、表格、图表、条件格式、自适应列宽行高 | +| `scripts/convert_workbook.py` | 旧格式、CSV/TSV、工作簿与预览 PDF 转换 | +| `scripts/recalculate_workbook.py` | 重算公式,检查缓存错误 | +| `scripts/render_workbook.py` | 输出逐页 PNG 与可选 PDF | +| `scripts/analyze_workbook.py` | 数据概况、处理、分组、交叉汇总、相关分析和规则分类,生成写入操作 JSON | +| `scripts/model_workbook.py` | 受控建模与求解,生成写入操作 JSON | -同一单元格不能同时传 `value` 和 `formula`。`formula` 必须以 `=` 开头。`write_rows.rows[][]` 可直接传值,也可在某个位置传带 `value/formula/style/comment/hyperlink` 的对象。 +`_xlsx_common.py`、`_xlsx_data.py` 是内部模块,不直接执行。 -## 转换文件 +## 工作流程 -调用 `scripts/convert_workbook.py`: +1. HTTPS 工作簿先安全下载;完整 URL 仅传给下载脚本,不在回复或日志摘要复述敏感查询参数。`.xls` 先转换;中文 CSV 可先指定编码转换成 XLSX。 +2. 编辑或分析前先检查工作簿。按游标读取相关数据范围及各相关 sheet,不能把前几行当作完整数据。确认真正表头、数据起止行、合并锚点、明细与汇总、单位、公式和缓存。 +3. 明确指标对应字段、分子分母、连接键、日期粒度及输出位置。关键缺失信息会影响结论时提出精确问题;其余说明合理假设后继续。 +4. 简单可维护计算优先使用最终工作簿中的 Excel 公式。分析/模型脚本输出结果快照和方法说明,再调用 `apply_workbook.py --spec-file <返回的 spec_path>` 写入新工作簿或源文件副本;脚本生成的模型参数不代表任意 Python 执行接口。 +5. 结果含公式或 `requires_recalculation: true` 时,用 `recalculate_workbook.py` 重算,确认 `status: success`、`total_errors: 0`;再检查关键引用、缓存与业务口径。缓存为空需分辨空字符串与未计算,不能直接当作 0。 +6. 创建/修改后用 `render_workbook.py` 逐页或分批检查全部相关页面:乱码、裁切、分页、长文本、图表与合并区域。数值检查与视觉检查通过后交付。 -```text ---input 'legacy.xls' --output '/usr/local/src/excel/tmp/task/source.xlsx' -``` +## 数据与分析约束 -常见用法: - -- CSV/TSV → XLSX:可传 `--sheet-name <名称>`;默认所有字段按文本保留,确认可以推断数字/布尔值时才传 `--infer-types`。 -- XLSX/XLSM → CSV/TSV:可传 `--sheet <名称>`;默认导出缓存结果,明确需要公式字符串时传 `--formulas`。 -- Excel → PDF:输出路径使用 `.pdf`;该 PDF 仅用于预览或用户明确要求的转换,不替代工作簿交付。 -- 中文旧系统文本可传 `--encoding gb18030`;默认 `utf-8-sig`。 - -## 公式重算 - -含公式的工作簿必须调用 `scripts/recalculate_workbook.py`: - -```text ---input '/usr/local/src/excel/result.xlsx' --output '/usr/local/src/excel/result-recalculated.xlsx' -``` - -检查返回值: - -- `status: success` 且 `total_errors: 0`:公式可被 LibreOffice 计算。 -- `status: errors_found`:根据 `error_summary` 中的工作表、单元格和公式修复,再重算。 -- `missing_cached_value_count > 0`:可能是公式结果为空字符串,也可能未正确计算;逐个抽查。 - -优先使用 Excel 2007 时代即可稳定重算的函数,如 `SUMIFS`、`INDEX`、`MATCH`、`IFERROR`、`SUMPRODUCT`。避免 `XLOOKUP`、`XMATCH`、`SORT`、`FILTER`、`UNIQUE`、`SEQUENCE` 等动态数组或新函数;脚本会提示但不能证明其结果完整。 - -## 渲染与视觉检查 - -调用 `scripts/render_workbook.py`: - -```text ---input '/usr/local/src/excel/result-recalculated.xlsx' --output-dir '/usr/local/src/excel/tmp/task/rendered' -``` - -默认 150 DPI、单次最多 20 页。可传: - -- `--start-page`、`--end-page`、`--max-pages`:分批渲染。 -- `--dpi <72-300>`:小字或复杂图表可提高到 180–220。 -- `--include-pdf`:同时保留 `workbook.pdf`。 -- `--overwrite`:只覆盖本次任务旧渲染。 - -若 `has_more: true`,用 `next_page` 继续。通过可用的图片查看工具逐页检查返回的 PNG。 - -## 质量要求 - -- 默认使用专业字体:中文使用 `Noto Sans CJK SC` 或与原文件一致的字体,拉丁文字使用 Arial;编辑现有文件时原有规范优先。 -- 表头、单位、日期、货币、百分比和负数格式必须明确;百分比按小数存储,例如 `0.15` 显示为 `15.0%`。 -- 可计算结果使用公式,不把当前结果硬编码进单元格;假设值单独放在有标签的输入单元格中。 -- 每个外部数据、假设和硬编码数字都用批注或邻近单元格说明来源。 -- 新建供他人填写的模板要包含填写说明和一行格式示例;编辑现有文件时不要擅自插入示例行。 -- 精确遵循用户指定的工作表名、表头、公式和输出格式,不擅自重构业务逻辑。 -- 合并单元格只写左上角锚点;编辑 `.xlsm` 时保留宏;不要用 `data_only=True` 读取后再保存。 -- 公式重算、关键值抽查和全部页面视觉检查全部通过后再交付。 +- 合并区域只写左上角。提取时仅对确认属于同一记录/分组的分类字段填充,不对整表盲目向前填充。 +- 负数可能是退款或冲销,超过 100% 可能是完成率;先核对业务含义,不自动取绝对值、归零、删行或改单位。 +- 缺失、零和未知分开处理;聚合的 `count` 不含空值、`size` 含空值;排除汇总行前确认其真实语义,保留来源行号和处理数量。 +- 可修改的输入、假设和计算公式应留在交付表中。分类标签、清洗映射、模型预测、优化方案可保存静态结果,并附来源、参数、有效范围与重新运行方式;不要承诺模型结果会随单元格自动更新。 +- 将事实、相关关系、模型估计和业务假设分开陈述。没有实际检验不填 p 值,没有区间计算不声称置信区间,不从特征重要性推导因果关系。 diff --git a/skills/xlsx/agents/openai.yaml b/skills/xlsx/agents/openai.yaml index 1e0d62f..91cad3c 100644 --- a/skills/xlsx/agents/openai.yaml +++ b/skills/xlsx/agents/openai.yaml @@ -1,4 +1,4 @@ interface: - display_name: "Excel 工作簿" - short_description: "安全下载、创建、编辑、重算、校验并渲染 Excel 工作簿" - default_prompt: "使用 $xlsx 创建或处理本地文件或 HTTPS 链接中的 Excel 工作簿,并完成公式与版式校验。" + display_name: "Excel 工作簿与分析" + short_description: "创建编辑 Excel,清洗汇总、文本归类、预测评价与优化,并完成公式和版式校验" + default_prompt: "使用 $xlsx 处理本地文件或 HTTPS 链接中的 Excel 数据,按需求完成分析、工作簿生成及校验。" diff --git a/skills/xlsx/evals/evals.json b/skills/xlsx/evals/evals.json new file mode 100644 index 0000000..b0d9e35 --- /dev/null +++ b/skills/xlsx/evals/evals.json @@ -0,0 +1,35 @@ +{ + "skill_name": "xlsx", + "evals": [ + { + "id": 1, + "prompt": "我有一份销售数据的Excel文件(sales_2024.xlsx),里面有产品名称、销售额、成本、地区、月份这几列,大概500行数据。帮我分析一下各地区的销售表现,找出表现最好和最差的地区,给出改进建议。", + "expected_output": "使用 xlsx 数据洞察流程;检查真实表头和汇总行,按地区汇总收入与成本,利润/比率用工作簿公式。使用既有写入、重算和渲染脚本;结论引用完整范围,保留源文件。", + "files": [] + }, + { + "id": 2, + "prompt": "我们部门有5个供应商,我把他们的评价数据口述给你:供应商A质量合格率98%、交货准时率85%、单价32元;供应商B质量合格率92%、交货准时率95%、单价28元;供应商C质量合格率95%、交货准时率90%、单价30元;供应商D质量合格率88%、交货准时率92%、单价25元;供应商E质量合格率96%、交货准时率88%、单价35元。帮我做个综合评价排名,选出最优供应商。", + "expected_output": "将用户数据写成工作簿,明确指标方向和权重口径后使用受控评价接口;不默认编造 AHP 矩阵,不把熵权视为业务重要性,不重复反转成本指标。输出可复核排名、权重和方法。", + "files": [] + }, + { + "id": 3, + "prompt": "帮我做一个项目进度跟踪表的Excel模板,需要包含:任务名称、负责人、开始日期、截止日期、完成状态、备注这几列。要求格式美观,有条件格式(逾期自动标红),还要有一个汇总sheet统计各负责人的任务完成率。", + "expected_output": "使用原 apply_workbook.py 创建模板、条件格式和汇总公式,按指定字段与模板需求完成。重算并逐页渲染;长备注换行,原文不截断。", + "files": [] + }, + { + "id": 4, + "prompt": "我有一份客户反馈的Excel(feedback.xlsx),里面有一列是客户的文字评价,大概200条。帮我把这些评价分类(正面/负面/中性),提取关键问题点,统计各类问题的数量和占比。", + "expected_output": "逐批读取全部相关评价,由模型理解正负/中性及冲突语义,按来源行号回填;明确关键词规则时可用受控分类,但不能把关键词匹配冒称完整情感理解。输出互斥分类计数与占比。", + "files": [] + }, + { + "id": 5, + "prompt": "我有过去24个月各城市的销量数据(monthly_sales.csv),帮我预测各城市未来3个月的销量,找出增长最快和下滑最严重的城市。", + "expected_output": "先按城市与月份检查连续性和重复值,按时间留出验证,比较已支持的基线方法。输出每城市未来 3 期预测、验证误差和限制,不虚构未来特征或宣称未运行的高级算法。", + "files": [] + } + ] +} diff --git a/skills/xlsx/references/data-analysis.md b/skills/xlsx/references/data-analysis.md new file mode 100644 index 0000000..5c75fa4 --- /dev/null +++ b/skills/xlsx/references/data-analysis.md @@ -0,0 +1,112 @@ +# 数据理解、处理和汇总 + +## 先确认数据口径 + +用 `scripts/inspect_workbook.py` 分段检查相关 sheet。它返回活动表、隐藏状态、合并范围、表格图表、公式、缓存、样式和行列游标;表头不清晰时向下、向右继续读取,不猜测空表头的字段含义。对多表确定各自主键与关系,不默认逐表独立分析,也不默认第一张表代表整本工作簿。 + +`analyze_workbook.py` 的 `profile` 检查指定范围的类型分布、缺失、唯一值与疑似汇总行。该检查不自动排除任何候选行;“合计成本”可能是字段名称,备注里出现“合计”也未必是汇总行。排除明细中的真实汇总行后再聚合,避免重复计数。 + +合并区域用于分类标记时,可以对确认的分类字段前向填充;数值和普通缺失记录不能跟随整表填充。ID、邮编、前导零文本先保持文本,只有指定数值字段才转换。退款、冲销、净流出等负数照业务含义保留。 + +## 统一调用 + +通过 `execute_skill_script` 调用 `scripts/analyze_workbook.py`: + +```text +--input '/usr/local/src/excel/tmp/task/source.xlsx' --spec '' --output '/usr/local/src/excel/tmp/task/analysis.json' +``` + +也支持 `--spec-file `,与 `--spec` 二选一。`profile` 可省略 `--output`,直接返回 JSON 概况;其他方法输出供原写入脚本使用的操作说明,而不是直接修改源文件。 + +```text +scripts/apply_workbook.py +--input '/usr/local/src/excel/tmp/task/source.xlsx' --output '/usr/local/src/excel/result.xlsx' --spec-file '<返回的 spec_path>' +``` + +新建独立结果文件时省略写入脚本的 `--input`。分析计划只追加结果 sheet;若目标已有同名 sheet,先选择新名称或明确规划替换区域,不默默清空已有内容。计划内的 `分析说明` 为保留名称,记录来源哈希、字段映射、筛选规则、算法参数和快照属性。 + +## 输入范围 + +所有分析/模型共用 `source` 对象: + +```json +{ + "source": { + "sheet": "销售明细", + "range": "A3:F502", + "header_row": 3, + "columns": {"地区": "B", "销售额": "E", "日期": "A"}, + "exclude_rows": [502], + "numeric": ["销售额"], + "dates": {"日期": "%Y-%m-%d"} + } +} +``` + +- 默认活动表、第一行为表头、读取整个使用区域。`range` 含表头;`header_row` 是源文件中的真实行号。 +- `columns` 是**名称 → Excel 列字母**,用于选择字段、空表头或多层表头的人工映射;未指定时,表头必须非空且唯一。 +- `exclude_rows` 指定已确认需要排除的源行号。输出 `__source_row` 始终保存原始行号,不是 DataFrame 的索引。 +- Excel 公式使用缓存值;相关区域存在未计算或错误缓存时先重算,不能把缺失缓存视作空记录。外部链接数据需先确认并固化。 +- `.xlsx/.xlsm/.xltx/.xltm/.csv/.tsv` 可直接分析;`.xls` 先转换。CSV 默认 `utf-8-sig`,可设 `source.encoding: "gb18030"`,所有字段初始按文本保留。 +- 输入文件最大 25 MiB,单次读取最多 500000 单元格;超过时选相关区域或分组处理。不得只分析前一批却宣称覆盖全表。 + +## 方法 + +| `method` | 参数与输出 | +| --- | --- | +| `profile` | 类型分布、缺失、唯一值、样本与疑似汇总行;候选最多显示 100 项并报告总数 | +| `transform` | 执行 `steps` 后输出明细及来源行号 | +| `aggregate` | `by: [分组字段]`、`metrics: {字段: 聚合方式}` | +| `pivot` | 同 aggregate,加 `columns: [列维度]`;结果为静态交叉汇总 | +| `describe` | `columns: [数值字段]`;计数、均值、标准差、分位数、极值 | +| `correlate` | `columns`,`correlation: "pearson"` 或 `"spearman"`;相关系数和每对字段的有效样本数 | +| `classify` | `column`、`rules`,详见 [文本处理](text-analysis.md) | + +非 profile 方法可用 `result_sheet` 设置首张结果表名称。聚合支持 `sum/count/size/mean/min/max/median/nunique`:`count` 排除值字段空值,`size` 统计记录数,空分类保留;全空 `sum` 保持空白,不自动写 0。均价、转化率等加权指标应从分子与分母的汇总重新计算,不能简单平均各组百分比。 + +交叉汇总的多层列名采用 JSON 数组形式,如 `["销售额","华东"]`,避免简单拼接导致不同维度重名。没有观测的交叉组合保留空白,不自动认为业务值为零。 + +## 处理步骤 + +`steps` 顺序执行,最多 50 项;返回处理前后行数。支持: + +| `type` | 字段 | +| --- | --- | +| `trim` | `columns`,仅修剪文本前后空格 | +| `replace` | `columns`、`mapping: {原值: 新值}`,显式同义词/编码映射 | +| `numeric` | `columns`,非法文本报错,不自动去掉单位或百分号 | +| `date` | `columns`、`format`,显式日期格式 | +| `fill` | `columns`,明确 `value` 或 `method: "ffill"` | +| `drop_missing` | `columns`,排除关键字段缺失记录 | +| `deduplicate` | `columns` 为判重键;`keep: "first"/"last"/false` | +| `filter` | `column`、`operator`、`value`;比较 `eq/ne/gt/ge/lt/le`、`in/not_in`、`contains`、`is_missing/not_missing` | +| `sort` | `columns`,可选 `ascending`,稳定排序 | +| `select` | `columns`,始终保留来源行号 | +| `rename` | `mapping: {旧名: 新名}`;不能重命名来源行号 | +| `merge` | `source: {path, …输入范围参数}`、`on: [连接键]`、`how: left/inner/right/outer`;`validate` 默认 many_to_one,可选 one_to_one/one_to_many | +| `concat` | `source: {path, …输入范围参数}`,相同字段纵向拼接,保留文件与行号来源 | + +连接键含空值时先处理。默认拒绝意外多对多连接,避免金额因笛卡尔积重复统计。`contains` 是字面子串匹配,不执行正则或代码;任何步骤都没有 `eval`、SQL 或 Python 执行入口。日期跨天、货币换算、百分比缩放等业务运算优先在结果工作簿写可检查的公式,不凭常识自动修改原数值。 + +## 销售汇总示例 + +```json +{ + "method": "aggregate", + "source": {"sheet": "明细", "exclude_rows": [502], "numeric": ["收入", "成本"]}, + "by": ["地区"], + "metrics": {"收入": "sum", "成本": "sum"}, + "result_sheet": "地区汇总", + "chart": {"type": "column", "category": "地区", "values": ["收入", "成本"], "title": "各地区收入与成本"} +} +``` + +`chart` 使用首张结果表,`values` 必须按顺序选择相邻数值列;支持原 xlsx 的 bar/column/line/area/pie。多层表或复杂对照图可在后续 `apply_workbook.py` 操作中明确设置引用。 + +## 透视表边界与洞察 + +这里生成的是 pandas 聚合后写入的**静态汇总表和普通图表**,不含原生 Excel PivotTable 字段拖拽、切片器或自动刷新控件。用户需要原生控件时,先说明当前固定接口的边界;不能把静态表称为已满足原生透视表要求。用户需要实时变化的简单汇总时,优先用 SUMIFS/COUNTIFS 等公式。 + +交付前核对:原始明细与模板保留、分组结果与明细合计一致、记录数与排除口径一致、图表引用实际结果且不重复包含总计。不要因图表不好看删除用户要求的结果;先调整图表或说明限制。 + +从统计到洞察时,先给结论及数值证据,再给解释和建议;将“数据表现”与“可能原因”区分。相关系数不代表因果或显著性;常量列相关系数为空,不改成零。报告组织见 [展示与交付](reporting.md)。 diff --git a/skills/xlsx/references/dependencies.md b/skills/xlsx/references/dependencies.md new file mode 100644 index 0000000..4ec082e --- /dev/null +++ b/skills/xlsx/references/dependencies.md @@ -0,0 +1,23 @@ +# 工具链与基础镜像依赖 + +目标镜像由 `/Users/zuihoudeqingyu/Git/wechat/silk-base/Dockerfile` 构建,运行时使用 `/opt/venv`。新配置只有重新构建并部署镜像后生效,skill 任务中不临时 pip/apt 安装。 + +| 能力 | 实现 | 镜像状态 | +| --- | --- | --- | +| 文件下载、路径/JSON 校验 | Python 标准库 | 已有 | +| 工作簿读写、模板保留、公式、图表、列宽行高 | openpyxl 3.1.5、Pillow | 已有 | +| 清洗、分组、交叉汇总、描述统计、规则分类 | pandas 3.0.5、NumPy | 已有 | +| 评价与基础时序外推 | NumPy | 已有,无需新库 | +| 工作簿转换、公式计算 | LibreOffice Calc | 已有 | +| 逐页 PNG/PDF 预览 | LibreOffice + Poppler | 已有 | +| 回归、分类、聚类、Isolation Forest | scikit-learn 1.9.0 | 本次补充 | +| LP/MILP 与受控二次规划 | SciPy 1.18.1(HiGHS / SLSQP) | 本次补充 | +| 文本翻译/语义抽取 | 模型读取原文,原写入脚本回填 | 不依赖外部翻译服务 | + +选择两个新增库是为了补上现有表格工具的模型训练和优化求解缺口。常规表格和图表继续使用原工具链;无需再安装 Excel COM、桌面 Excel、xlsxwriter、外部 CBC/PuLP 求解器、deep-translator、XGBoost、Prophet、Matplotlib 或浏览器展示组件来完成当前接口。 + +新增包固定版本并限定二进制 wheel 安装,PyPI 提供 Linux amd64/arm64 对应构建;镜像已有 libgomp1。SciPy/scikit-learn 与当前 pandas、OpenCV 要求的 NumPy 2.2.x 已做临时环境兼容性验证。构建步骤会实际运行最小线性模型与整数求解自检,缺包或二进制不兼容直接失败。没有在本机运行 Docker 构建。 + +参考:[SciPy 发行包](https://pypi.org/project/scipy/1.18.1/)、[scikit-learn 发行包](https://pypi.org/project/scikit-learn/1.9.0/)、[SciPy MILP](https://docs.scipy.org/doc/scipy/reference/generated/scipy.optimize.milp.html)、[模型预处理与数据泄漏](https://scikit-learn.org/stable/common_pitfalls.html)。 + +本 skill 输出静态交叉汇总表和普通图表;原生 PivotTable、切片器、任意 Python 建模、自定义神经网络或任意非线性求解不属于已提供接口,不能通过文案声称已经支持。遇到明确的额外需求时,先评估固定脚本和现有库能否扩展,再评估新增依赖。 diff --git a/skills/xlsx/references/modeling.md b/skills/xlsx/references/modeling.md new file mode 100644 index 0000000..d9c2619 --- /dev/null +++ b/skills/xlsx/references/modeling.md @@ -0,0 +1,103 @@ +# 模型方法与固定接口 + +先确定目标、对象与粒度、数据和变量、评价指标及需要解释的结论。缺少会改变模型含义的权重、约束或时间粒度时确认;不要假定用户不会回复,也不要编造限制或训练数据。简单业务计算优先工作簿公式,不为求和、比率或格式整理启动机器学习。 + +## 通用调用 + +通过 `execute_skill_script` 调用 `scripts/model_workbook.py`: + +```text +--input '/usr/local/src/excel/tmp/task/source.xlsx' --spec '' --output '/usr/local/src/excel/tmp/task/model.json' +``` + +`--spec-file` 与 `--spec` 二选一。仅 `task: "optimize"` 可省略输入文件,直接用已确认的系数建模。其他任务共享 [数据分析](data-analysis.md) 的 `source` 结构;建模数据上限 20000 行、100000 单元格。输出 JSON `spec_path` 再交给原 `apply_workbook.py`,不直接覆盖工作簿。 + +数值字段需有限;除监督学习的特征填补外,模型不自动填补缺失数据。保留异常值和来源行号,避免从数据缺口编造结论。图表可用 `chart` 字段(首结果表),或在原写入脚本中创建。 + +## 综合评价:`task: "evaluate"` + +```json +{ + "task": "evaluate", + "source": {"sheet": "供应商"}, + "entity": "供应商", + "directions": {"质量合格率": "benefit", "交货准时率": "benefit", "单价": "cost"}, + "weighting": "user", + "weights": {"质量合格率": 0.5, "交货准时率": 0.3, "单价": 0.2} +} +``` + +- `directions`:效益型 `benefit`、成本型 `cost`、中间最优型 `{"target": 7}`。先统一正向化,之后 TOPSIS 的正理想解统一取大值;**不能在已经正向化后再次反转成本方向**。 +- `weighting`:默认透明的 `equal`;用户权重用 `user`(全部指标、非负、总和大于零);差异性赋权用 `entropy`;用户给出比较矩阵时可用 `ahp`。 +- 熵权高只说明样本差异性较大,不等于业务重要性。不要在没有用户偏好的情况下编造 AHP 比较矩阵。 +- AHP 参数 `comparison_matrix`:按 directions 顺序,2–9 阶正数互反矩阵,对角为 1;CR 不小于 0.1 时拒绝输出排名,需修正比较判断。 +- 空列、非数值、不可解释编码需先处理。常量指标没有区分度;所有对象相同则并列,得分为中性 0.5,不随行顺序强行选“第一”。 + +输出:综合排名(原值、得分、并列名次、来源行号)和指标权重(方向、权重、适用时的信息熵、无区分度标记)。解释优劣势时引用原始指标与权重;不把得分当作获胜概率。必要时在合理权重范围重新运行,观察排名是否稳定。 + +## 时序预测:`task: "forecast"` + +```json +{ + "task": "forecast", + "source": {"sheet": "月度销量", "dates": {"月份": "%Y-%m-%d"}}, + "by": ["城市"], "date": "月份", "value": "销量", + "frequency": "MS", "horizon": 3, + "methods": ["naive", "linear", "moving_average"], "holdout": 6 +} +``` + +- `by` 可省略;每组独立建模。`frequency` 支持 D(日)、W(周末为周日)、MS(月初)、QS(季度初)、YS(年初)。数据需按对应周期锚点记录、连续、无重复,至少 6 期;先按口径处理重复和缺期,不能默认缺期销量为零。 +- `horizon` 为 1–120 期。`holdout` 默认最后四分之一且至少 2 期,需保留至少 3 期训练数据。 +- `naive` 延续最后值;`linear` 线性趋势;`moving_average` 递归最近最多三期均值;`seasonal_naive` 需明确 `seasonal_period` 且训练长度足够。 +- 候选方法仅使用训练时段预测验证时段,按 MAE 选择后用全历史外推。该验证集参与选型,不能称为完全独立的最终测试集。 + +输出:各组×未来日期的预测值、验证 MAE/RMSE、历史数据。不要在未建区间模型时声称置信区间,不自动把负预测裁为零;检查趋势外推是否符合业务边界并披露限制。不能宣称已经运行 ARIMA、Prophet 或 XGBoost,本接口实际只运行列出的基线方法。 + +## 回归与分类:`task: "regression" / "classification"` + +```json +{ + "task": "regression", + "source": {"sheet": "样本"}, + "features": ["价格", "促销费用", "城市"], + "categorical": ["城市"], "target": "销量", + "algorithm": "ridge", "test_fraction": 0.2 +} +``` + +- 至少 10 条记录,目标列不能缺失。`categorical` 显式列出分类特征,其余为数值特征。ID、目标列、目标派生字段不作为有效预测特征。 +- 回归 `algorithm`:`linear`(默认)、`ridge`、`forest`。分类:`logistic`(默认)、`forest`。森林使用固定种子、100 棵树和最大深度 8;不接受任意估计器或任意模型代码。 +- `test_fraction` 为 0.1–0.5,默认 0.2,留出集至少 2 条记录。普通样本固定随机划分,分类进行分层。时序/未来预测必须指定 `time_column` 或使用 forecast,按时间划分;相同时点跨切分边界会被拒绝。若同一实体的重复记录可能泄漏信息,应先设计实体隔离的数据集,不能把随机切分当作有效泛化证据。 +- 中位数填补、分类缺失标记、One-hot 与标准化都只在训练集拟合,再应用测试集。不会先在全数据预处理后假装留出测试。 +- 回归输出训练/测试 MAE、RMSE、R²;分类输出 Accuracy、macro F1,并与均值/多数类基线对比。没有实际计算的 AUC、交叉验证、p 值不填入报告。 +- `predict_source` 可指定另一个已下载文件的 `{path, …source参数}`;先保留测试结果,再用全量训练数据重新拟合并预测新样本。缺少未来特征时不能凭空生成未来回归预测。 + +输出:留出样本实际值与预测值、模型指标、特征重要性/绝对系数、可选新样本预测。系数与重要性是关联说明,不能推断因果方向;有负 R² 或测试表现不如基线时明确说明,不以高训练分数宣传可靠性。 + +## 聚类和异常检测 + +- `task: "cluster"`:`features: [数值字段]`;`algorithm: "kmeans"` 默认,`clusters` 默认 3,需至少为 2 且小于样本数;或 `algorithm: "dbscan"`,`eps` 默认 0.5,`min_samples` 默认 5。 +- `task: "anomaly"`:`features`,Isolation Forest;`contamination` 默认 0.05,范围 (0,0.5]。该比例是假设,应与业务目的相符,不表示真实异常率。 +- 数值特征标准化后分析,不自动删除异常。输出原记录、来源行号和标签;-1 为异常候选或 DBSCAN 噪声,聚类数字不代表优劣。满足条件时报告排除噪声后的轮廓系数。 +- 对类别本身或长文本的深层语义先参见文本指南;当前接口不假装运行主题模型、FP-Growth、SMOTE 或任意外部模型。相关性可用 analyze_workbook;需要额外算法时应新增明确的固定接口并单独评估依赖。 + +## 资源优化:`task: "optimize"` + +```json +{ + "task": "optimize", + "variables": ["产品A", "产品B"], "objective": [30, 20], "sense": "max", + "bounds": [[0, 100], [0, 80]], "integer": [true, true], + "constraints": [ + {"name": "可用工时", "coefficients": [2, 1], "relation": "<=", "rhs": 160} + ] +} +``` + +- 明确变量、目标、资源限制、数量下限/上限及单位,不自动放松约束。`sense` 默认 min;`bounds` 默认非负无上界,null 表示无界;0–1 决策使用 `[0,1]` 和整数标记。 +- `objective` 是按 variables 顺序的线性系数。约束 `relation` 支持 `<=`、`>=`、`==`。最多 200 个变量、1000 条约束。 +- 连续/整数/混合整数线性问题使用 SciPy 内置 HiGHS,时间上限 60 秒。仅成功状态、约束与整数性复核通过才输出方案;超时、无解、无界不能当作最优解。 +- 可选 `quadratic` 对称矩阵 Q,目标为 `c·x + 0.5*xᵀQx`;只支持连续变量、线性约束、凸最小化或凹最大化。`initial` 可指定初始点,使用 SLSQP。任意非线性函数不在接口范围内。 + +输出变量方案、目标值、约束左端值和余量。可用参数情景重新运行分析瓶颈;未求解的“增加资源收益”不能编造。所有结果是按给定系数得到的快照,不会自动随着源单元格更新。 diff --git a/skills/xlsx/references/reporting.md b/skills/xlsx/references/reporting.md new file mode 100644 index 0000000..7b52516 --- /dev/null +++ b/skills/xlsx/references/reporting.md @@ -0,0 +1,33 @@ +# 展示与交付 + +按实际需求选择说明内容,无固定字数、文件数量或章节数要求。用户只问统计结论时直接回答;交付工作簿时,可在说明 sheet 或回复中组织: + +1. 问题、来源、时间范围、有效样本量及排除口径。 +2. 关键发现与对应数值;假设、模型估计和观测事实分开。 +3. 指标计算、分类规则、模型参数、验证结果及主要限制。 +4. 基于证据的建议;没有计算支持,不虚构收益、概率、p 值或重要性。 + +需要 Word/PDF/HTML 正式报告时使用当前可用的对应技能,并复用已经核对的结果,不为了分析任务默认生成多种报告。数据表和说明留在工作簿内即可,不暴露工具日志或思考过程;必要的业务计算口径和模型参数应保留,便于复核。 + +## 图表选择 + +- 趋势用折线图;分类对比用柱形/条形图;简单占比可用饼图,但负值、多类别、差异很小的数据优先用条形图。 +- 复杂分布、关联或多维模型可先用数据表、普通图表和条件色阶表达。现有固定接口没有地图、桑基图、交互网页或专业统计绘图库,不冒称已实现。 +- 原工具链生成可编辑 Excel 图表,预览使用 LibreOffice + Poppler。图表需引用实际数值,标题、单位、图例清楚;不把“对象存在”当作“图像有效”。 +- 类别过多时按用户目标选 Top N,剩余项目是否汇总为“其他”需说明;总计不能作为普通类别重复出现在图中。 +- 柱形/面积图通常从零开始;时间趋势的局部范围可按需调整,但明确标尺,不把“所有坐标轴从零开始”当作通用规则。 + +## 字体和长文本 + +保持原模板字体;新建表中文可用 Noto Sans CJK SC,英文用 Arial。用户要求时也可用镜像已安装的宋体、微软雅黑或苹方/SF Pro。字体并不自动证明容器已更新,乱码时检查实际镜像与渲染结果。 + +`apply_workbook.py` 的 `auto_fit` 只对指定区域估算列宽、换行和行高,不修改文本内容;默认列宽 8–40,最大行高受 Excel 限制。合并单元格、长公式、特别长的说明仍需逐页视觉检查,必要时调整宽度或将说明拆为独立区域。不要用字符串切片加省略号“修复排版”。 + +## 财务与模型表格 + +- 用户模板优先。新建财务模型可用蓝字表示可调整输入、黑字表示公式、绿字表示同一工作簿跨表引用、黄色底色表示待确认假设;这些是可选格式约定,不是所有工作簿的强制颜色。 +- 货币使用数据实际币种,表头标单位;百分比按小数存储,例如 0.15 显示 15.0%。财务负数可用括号,零显示为短横线需在说明中明确。 +- 增长率、成本、权重等可变假设单独留在有标签的输入单元格;简单总计、比例、情景计算用公式引用。不要在公式内隐藏无来源的业务假设。 +- 预测、分类、优化等模型结果是可复核的快照,附来源哈希、参数和方法说明。修改输入后应重新运行模型,不把静态值宣传为公式模型。 + +重算、关键数值抽查和逐页渲染按入口流程完成。若只有部分范围处理成功,明确范围与未完成项;不以“无公式错误”证明数据逻辑、模型有效性或原生透视表功能。 diff --git a/skills/xlsx/references/text-analysis.md b/skills/xlsx/references/text-analysis.md new file mode 100644 index 0000000..7653a06 --- /dev/null +++ b/skills/xlsx/references/text-analysis.md @@ -0,0 +1,34 @@ +# 文本归类、抽取和翻译 + +先用 `inspect_workbook.py` 分批读取目标文字列及必要上下文,确定抽取字段、标签定义和输出位置。原文与原工作表保留,分类或译文写入新列、新 sheet 或副本,记录 source_sheet、source_row、source_col。不要把前 15 行的标签直接套给未读取的全部行。 + +## 模型逐行理解 + +摘要、实体抽取、语义分类和翻译由模型基于已读取原文完成,再用 `apply_workbook.py` 的 `set_cells` 或 `write_rows` 写回。按任务保留:原文、清洗文本、要点、观点、实体、类别、子类别、多标签、标准化结果、证据片段、不确定说明;不要求每次都增加全部字段。 + +- 分类体系以用户定义为准;缺少重要定义时提出精确问题,普通边界样本可以标为“未知/需复核”。多标签要注明分母是记录数还是标签次数。 +- 标准化只处理明确同义词、空白和格式,不抹去否定词、程度词、时间、主体等会改变含义的信息。 +- 词典匹配是规则判断,不能直接当作深度语义或情感理解。例如“服务不错,但物流很慢”需要保留冲突信息。 +- 翻译前确定目标语言及术语表;保留数字、产品编号、专名、公式、链接和日期。按实际读取的单元格翻译,不翻译表名或改结构,除非用户要求。 +- `value` 写入译文/标签,包括以 `=` 开头的普通文本;真正公式使用 `formula`。不要通过重建整个 DataFrame 覆盖原工作簿来翻译,避免丢失模板、公式和宏。 +- 当前流程不调用外部翻译服务,无需 deep-translator。完成后核对处理单元格数、原文对应关系、术语一致性以及未处理项,不能因为某一格成功就报告全表完成。 + +## 可重复的规则分类 + +已有明确关键词规则时,用 `analyze_workbook.py`: + +```json +{ + "method": "classify", + "source": {"sheet": "反馈", "columns": {"评价": "C"}}, + "column": "评价", + "rules": [ + {"label": "配送问题", "keywords": ["迟到", "物流慢"]}, + {"label": "服务正向", "keywords": ["服务好", "态度好"]} + ] +} +``` + +规则采用不区分大小写的字面子串匹配。单一命中输出该类别,多类别命中为“需复核”,未命中为“未知”;输出命中标签及证据关键词,并在另一张 sheet 汇总互斥类别的记录数和占比。不会自动推断未列出的同义词、否定含义或新类别。 + +后续将生成的 `spec_path` 交给原 `apply_workbook.py`。结合模型逐行复核可修正规则结果;用户要求完整语义抽取时,不以关键词命中结果代替未完成的理解工作。 diff --git a/skills/xlsx/references/workbook-operations.md b/skills/xlsx/references/workbook-operations.md new file mode 100644 index 0000000..1cceca1 --- /dev/null +++ b/skills/xlsx/references/workbook-operations.md @@ -0,0 +1,198 @@ +# 工作簿操作接口 + +以下路径相对 xlsx skill 根目录,通过 execute_skill_script 调用。 + +## 下载远程工作簿 + +只接受 HTTPS 地址。完整保留 URL 及查询参数传给脚本,但不要在回复、日志摘要或输出文件名中复述敏感参数。 + +调用 `scripts/download_workbook.py`: + +```text +--url 'https://example.com/report.xlsx?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/source.xlsx' +``` + +可选参数: + +- `--timeout <1-600>`:连接和读取超时秒数,默认 `60`。 +- `--max-bytes <字节数>`:默认且最高 `26214400`(25 MiB),只允许设置更小的限制。 +- `--overwrite`:只在目标是本次任务生成的旧缓存时使用。 + +`output` 扩展名必须是 `.xlsx`、`.xlsm`、`.xltx`、`.xltm`、`.xls`、`.csv` 或 `.tsv`。脚本阻止 HTTPS 重定向降级到 HTTP,流式限制大小,先写同目录临时文件,再原子发布;OOXML 会检查 ZIP 路径、成员大小、内容类型并用 `openpyxl` 打开,CSV/TSV 会拒绝二进制或网页响应。实际 OOXML 格式与 `output` 扩展名不一致时,根据错误中的实际格式更正缓存扩展名,再调用同一脚本。 + +成功结果包含 `path`、`size_bytes`、`format` 和 `validation`;OOXML 还包含 `sheet_count`。后续脚本只使用返回的本地 `path`,不再访问原 URL。 + +## 下载通用附件 + +需要下载作为 Excel 任务素材的图片、视频、音频、压缩包或其他文件时,调用 `scripts/download_attachment.py`: + +```text +--url 'https://example.com/asset.bin?signature=...' --output '/usr/local/src/excel/tmp/<任务名>/asset.bin' +``` + +只接受 HTTPS 地址,`output` 可使用任意附件扩展名。可选参数只有 `--timeout <1-600>`(默认 `60`)和 `--overwrite`。附件上限固定为 25 MiB(26214400 字节),不可调高:脚本先用 HEAD 探测远端声明大小,再检查 GET 响应声明,并在流式接收时持续兜底计数;任一阶段发现超限都会返回 `ok: false` 和明确的“已拒绝下载”错误,且不会发布部分文件。 + +成功结果包含 `path`、实际 `size_bytes`、`declared_size_bytes`、`size_limit_bytes`、`size_probe` 和 `content_type`。本脚本不校验文件业务格式;远程 Excel/CSV/TSV 源文件仍使用 `download_workbook.py`。 + +## 检查工作簿 + +调用 `scripts/inspect_workbook.py`: + +```text +--input 'source.xlsx' +``` + +可选参数: + +- `--sheet <名称>`:选择工作表;默认活动工作表。 +- `--start-row <行>`、`--start-column <列>`:读取起点,均从 `1` 开始。 +- `--max-rows <1-200>`、`--max-columns <1-100>`:限制单次输出,默认 `40 × 20`。 + +结果同时给出公式字符串和缓存值: + +- `formula`:原始公式。 +- `cached_value`:Excel/LibreOffice 上次计算后保存的结果。 +- `has_external_links`:为 `true` 时,编辑或重算可能破坏外部链接缓存;默认停止并向用户说明。 +- `selection.has_more`、`next_row`、`next_column`:用于继续读取大表,不要一次返回整本工作簿。 + +CSV/TSV 只返回行数据,不存在工作表。`.xls` 必须先转换。 + +## 创建或编辑 + +调用 `scripts/apply_workbook.py`,新建时省略 `--input`,编辑时提供源文件: + +```text +--output '/usr/local/src/excel/result.xlsx' --spec '' +``` + +或: + +```text +--input 'source.xlsx' --output '/usr/local/src/excel/result.xlsx' --spec-file '/usr/local/src/excel/tmp/task/operations.json' +``` + +目标已存在且确认是本次任务的旧产物时才传 `--overwrite`。输入含外部链接时脚本默认拒绝保存;只有用户明确接受缓存值可能丢失的风险时才传 `--allow-external-links`。`.xlsm` 必须继续输出 `.xlsm` 才能保留宏;只有用户明确同意丢弃宏时才输出 `.xlsx` 并传 `--drop-macros`。 + +操作说明顶层字段: + +```json +{ + "properties": { + "title": "销售分析", + "creator": "示例公司" + }, + "calculation_mode": "auto", + "active_sheet": "汇总", + "operations": [] +} +``` + +支持的 `operations[].type`: + +| 类型 | 关键字段 | +| --- | --- | +| `add_sheet` | `name`,可选 `index` | +| `remove_sheet` | `sheet` | +| `rename_sheet` | `sheet`、`name` | +| `set_cells` | `sheet`、`cells[]` | +| `write_rows` | `sheet`、`start_cell`、`rows[][]`,可选统一 `style` | +| `append_rows` | `sheet`、`rows[][]` | +| `style_range` | `sheet`、`range`、`style` | +| `clear_range` | `sheet`、`range`,可选 `values/styles/comments/hyperlinks` | +| `insert_rows` / `delete_rows` | `sheet`、`index`、`amount` | +| `insert_columns` / `delete_columns` | `sheet`、`index`、`amount` | +| `merge_cells` / `unmerge_cells` | `sheet`、`range` | +| `set_column_widths` | `sheet`、`widths`,如 `{"A": 18, "B:D": 12}` | +| `set_row_heights` | `sheet`、`heights`,如 `{"1": 28, "2:5": 20}` | +| `auto_fit` | `sheet`、`range`,可选 `min_width/max_width`(默认 8/40);调整列宽、换行和行高,保留完整文本 | +| `freeze_panes` | `sheet`、`cell`;传空值取消冻结 | +| `set_auto_filter` | `sheet`、`range`;传空值取消筛选 | +| `add_table` | `sheet`、`range`、`name`,可选 `style` | +| `add_chart` | `sheet`、`chart_type`、`data_range`、`anchor`;可选 `categories_range/title` | +| `add_image` | `sheet`、`path`、`anchor`;可选像素 `width/height` | +| `add_data_validation` | `sheet`、`range`、`validation_type`、`formula1` | +| `add_conditional_format` | `sheet`、`range`、`rule_type` 及对应规则参数 | +| `set_print` | `sheet`,可选 `print_area/orientation/paper_size/fit_to_width/margins` | +| `set_named_range` | `sheet`、`name`、`range` | + +`set_cells.cells[]` 中每项使用: + +```json +{ + "cell": "B2", + "formula": "=SUM(B3:B10)", + "style": { + "font": {"name": "Arial", "size": 11, "bold": true, "color": "FFFFFF"}, + "fill": {"color": "1F4E78"}, + "alignment": {"horizontal": "center", "vertical": "center", "wrap_text": true}, + "number_format": "#,##0.00", + "border": { + "bottom": {"style": "thin", "color": "808080"} + }, + "protection": {"locked": true} + }, + "comment": {"author": "AI", "text": "来源:用户提供的 2026 年预算"}, + "hyperlink": "https://example.com/source" +} +``` + +同一单元格不能同时传 `value` 和 `formula`。`value` 中的字符串始终按文本写入,即使以 `=` 开头;公式使用明确的 `formula` 字段。`formula` 必须以 `=` 开头。`write_rows.rows[][]` 可直接传值,也可在某个位置传带 `value/formula/style/comment/hyperlink` 的对象。 + +## 转换文件 + +调用 `scripts/convert_workbook.py`: + +```text +--input 'legacy.xls' --output '/usr/local/src/excel/tmp/task/source.xlsx' +``` + +常见用法: + +- CSV/TSV → XLSX:可传 `--sheet-name <名称>`;默认所有字段按文本保留,确认可以推断数字/布尔值时才传 `--infer-types`。 +- XLSX/XLSM → CSV/TSV:可传 `--sheet <名称>`;默认导出缓存结果,明确需要公式字符串时传 `--formulas`。 +- Excel → PDF:输出路径使用 `.pdf`;该 PDF 仅用于预览或用户明确要求的转换,不替代工作簿交付。 +- 中文旧系统文本可传 `--encoding gb18030`;默认 `utf-8-sig`。 + +## 公式重算 + +含公式的工作簿必须调用 `scripts/recalculate_workbook.py`: + +```text +--input '/usr/local/src/excel/result.xlsx' --output '/usr/local/src/excel/result-recalculated.xlsx' +``` + +检查返回值: + +- `status: success` 且 `total_errors: 0`:公式可被 LibreOffice 计算。 +- `status: errors_found`:根据 `error_summary` 中的工作表、单元格和公式修复,再重算。 +- `missing_cached_value_count > 0`:可能是公式结果为空字符串,也可能未正确计算;逐个抽查。 + +优先使用 Excel 2007 时代即可稳定重算的函数,如 `SUMIFS`、`INDEX`、`MATCH`、`IFERROR`、`SUMPRODUCT`。避免 `XLOOKUP`、`XMATCH`、`SORT`、`FILTER`、`UNIQUE`、`SEQUENCE` 等动态数组或新函数;脚本会提示但不能证明其结果完整。 + +## 渲染与视觉检查 + +调用 `scripts/render_workbook.py`: + +```text +--input '/usr/local/src/excel/result-recalculated.xlsx' --output-dir '/usr/local/src/excel/tmp/task/rendered' +``` + +默认 150 DPI、单次最多 20 页。可传: + +- `--start-page`、`--end-page`、`--max-pages`:分批渲染。 +- `--dpi <72-300>`:小字或复杂图表可提高到 180–220。 +- `--include-pdf`:同时保留 `workbook.pdf`。 +- `--overwrite`:只覆盖本次任务旧渲染。 + +若 `has_more: true`,用 `next_page` 继续。通过可用的图片查看工具逐页检查返回的 PNG。 + +## 质量要求 + +- 默认使用专业字体:中文使用 `Noto Sans CJK SC` 或与原文件一致的字体,拉丁文字使用 Arial;编辑现有文件时原有规范优先。 +- 表头、单位、日期、货币、百分比和负数格式必须明确;百分比按小数存储,例如 `0.15` 显示为 `15.0%`。 +- 可计算结果使用公式,不把当前结果硬编码进单元格;假设值单独放在有标签的输入单元格中。 +- 每个外部数据、假设和硬编码数字都用批注或邻近单元格说明来源。 +- 新建供他人填写的模板要包含填写说明和一行格式示例;编辑现有文件时不要擅自插入示例行。 +- 精确遵循用户指定的工作表名、表头、公式和输出格式,不擅自重构业务逻辑。 +- 合并单元格只写左上角锚点;编辑 `.xlsm` 时保留宏;不要用 `data_only=True` 读取后再保存。 +- 公式重算、关键值抽查和全部页面视觉检查全部通过后再交付。 diff --git a/skills/xlsx/scripts/_xlsx_data.py b/skills/xlsx/scripts/_xlsx_data.py new file mode 100644 index 0000000..61ca709 --- /dev/null +++ b/skills/xlsx/scripts/_xlsx_data.py @@ -0,0 +1,258 @@ +"""Bounded table reads and operation plans for the existing xlsx writer.""" +from __future__ import annotations + +import csv +import hashlib +import json +import math +import os +import tempfile +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from _xlsx_common import ( + EXCEL_INPUT_SUFFIXES, input_file, output_file, publish_file, + normalize_formula_error, validate_cell_range, workbook_has_external_links, +) + +MAX_DATA_CELLS = 500_000 +SOURCE_ROW = "__source_row" + + +def require_columns(frame: pd.DataFrame, names: list[str]) -> list[str]: + if not isinstance(names, list) or not names or len(set(names)) != len(names): + raise ValueError("字段必须是非空、无重复的名称列表") + missing = [name for name in names if name not in frame.columns] + if missing: + raise ValueError(f"字段不存在:{missing};可选:{list(frame.columns)}") + return names + + +def numeric(frame: pd.DataFrame, names: list[str], *, allow_missing: bool = False) -> pd.DataFrame: + result = frame[require_columns(frame, names)].apply(pd.to_numeric, errors="raise") + values = result.to_numpy(dtype=float, na_value=np.nan) + if np.isinf(values).any() or (not allow_missing and np.isnan(values).any()): + raise ValueError("数值字段含缺失值或无穷值;请先明确处理口径") + return result.astype(float) + + +def scalar(value: Any) -> Any: + if isinstance(value, np.generic): + value = value.item() + if value is None or value is pd.NA or value is pd.NaT: + return None + if isinstance(value, float): + if math.isnan(value): + return None + if not math.isfinite(value): + raise ValueError("结果含无穷值") + if hasattr(value, "isoformat"): + return value.isoformat() + if isinstance(value, (str, int, float, bool)): + return value + return str(value) + + +def read_dataset(path_value: str, spec: dict[str, Any] | None = None) -> tuple[pd.DataFrame, dict]: + from openpyxl import load_workbook + from openpyxl.utils.cell import column_index_from_string, get_column_letter, range_boundaries + + spec = spec or {} + allowed = {"sheet", "range", "header_row", "columns", "exclude_rows", "numeric", "dates", "encoding", "path"} + if set(spec) - allowed: + raise ValueError(f"source 包含未知字段:{sorted(set(spec) - allowed)}") + source = input_file(path_value, EXCEL_INPUT_SUFFIXES | {".csv", ".tsv"}) + if source.stat().st_size > 25 * 1024 * 1024: + raise ValueError("分析输入上限为 25 MiB;请先分割文件") + if workbook_has_external_links(source): + raise ValueError("分析输入含外部链接,需先确认并固化其数据") + header_row = int(spec.get("header_row", 1)) + if header_row < 1: + raise ValueError("header_row 从 1 开始") + bounds = range_boundaries(validate_cell_range(spec["range"])) if spec.get("range") else None + if bounds and not bounds[1] <= header_row <= bounds[3]: + raise ValueError("header_row 必须位于 range 内") + records = [] + cached_wb = formula_wb = None + formula_count = 0 + try: + if source.suffix.lower() in EXCEL_INPUT_SUFFIXES: + formula_wb = load_workbook(source, read_only=True, data_only=False, keep_links=False) + cached_wb = load_workbook(source, read_only=True, data_only=True, keep_links=False) + sheet_name = spec.get("sheet") or formula_wb.active.title + if sheet_name not in formula_wb.sheetnames: + raise ValueError(f"工作表不存在:{sheet_name}") + ws, cached = formula_wb[sheet_name], cached_wb[sheet_name] + min_col, _, max_col, end_row = bounds or (1, header_row, ws.max_column, ws.max_row) + if max_col < min_col or end_row < header_row or (end_row - header_row + 1) * (max_col - min_col + 1) > MAX_DATA_CELLS: + raise ValueError("读取区域超过 500000 单元格或无效;请指定较小 range") + rows = ws.iter_rows(min_row=header_row, max_row=end_row, min_col=min_col, max_col=max_col) + values = cached.iter_rows(min_row=header_row, max_row=end_row, min_col=min_col, max_col=max_col) + for row_number, (formula_row, cached_row) in enumerate(zip(rows, values), header_row): + record = [] + for cell, cached_cell in zip(formula_row, cached_row): + is_formula = cell.data_type == "f" + value = cached_cell.value if is_formula else cell.value + if is_formula: + formula_count += 1 + if value is None or cached_cell.data_type == "e": + raise ValueError(f"{sheet_name}!{cell.coordinate} 的公式缓存缺失或错误;先重算并核对") + if cell.data_type == "e": + raise ValueError(f"{sheet_name}!{cell.coordinate} 含 Excel 错误值") + record.append(value) + records.append((row_number, record)) + else: + if spec.get("sheet"): + raise ValueError("CSV/TSV 没有工作表") + sheet_name = None + min_col, _, max_col, end_row = bounds or (1, header_row, 0, 1_048_576) + with source.open(encoding=spec.get("encoding", "utf-8-sig"), newline="") as handle: + reader = csv.reader(handle, delimiter="\t" if source.suffix.lower() == ".tsv" else ",") + cells = 0 + for row_number, row in enumerate(reader, 1): + if row_number > end_row: + break + if row_number < header_row: + continue + max_col = max_col or len(row) + cells += max_col - min_col + 1 + if cells > MAX_DATA_CELLS: + raise ValueError("读取区域超过 500000 单元格;请指定较小 range") + if not bounds and len(row) > max_col: + raise ValueError(f"CSV 第 {row_number} 行比表头多列,请先校正结构") + record = row[min_col - 1:max_col] + record += [None] * (max_col - min_col + 1 - len(record)) + records.append((row_number, [None if v == "" else v for v in record])) + if not records: + raise ValueError("指定区域没有表头和数据") + aliases = spec.get("columns") + if aliases: + if not isinstance(aliases, dict) or not all(isinstance(k, str) and k for k in aliases): + raise ValueError("columns 必须为名称到 Excel 列字母的映射") + names = list(aliases) + offsets = [column_index_from_string(str(aliases[name]).upper()) - min_col for name in names] + if len(set(offsets)) != len(offsets) or any(i < 0 or i > max_col - min_col for i in offsets): + raise ValueError("columns 必须指向区域内不同的列") + else: + names = [str(v).strip() if v is not None else "" for v in records[0][1]] + offsets = list(range(len(names))) + if not all(names) or len(set(names)) != len(names): + raise ValueError("表头为空或重复;请用 columns 显式指定名称到列字母的映射") + if SOURCE_ROW in names: + raise ValueError(f"{SOURCE_ROW} 为保留字段") + excluded = set(spec.get("exclude_rows", [])) + if any(not isinstance(v, int) or v <= header_row for v in excluded): + raise ValueError("exclude_rows 必须是表头之后的原始行号") + data = [[row_number, *[row[i] for i in offsets]] for row_number, row in records[1:] if row_number not in excluded] + frame = pd.DataFrame(data, columns=[SOURCE_ROW, *names], dtype=object) + for name in spec.get("numeric", []): + frame[name] = numeric(frame, [name], allow_missing=True)[name] + for name, fmt in spec.get("dates", {}).items(): + require_columns(frame, [name]) + frame[name] = pd.to_datetime(frame[name], format=fmt, errors="raise") + metadata = { + "path": str(source), "sha256": hashlib.sha256(source.read_bytes()).hexdigest(), + "sheet": sheet_name, "header_row": header_row, + "columns": {name: get_column_letter(min_col + offset) for name, offset in zip(names, offsets)}, + "rows_read": len(records) - 1, "rows_used": len(frame), + "excluded_rows": sorted(excluded), "formula_cache_cells": formula_count, + } + return frame, metadata + finally: + if cached_wb is not None: + cached_wb.close() + if formula_wb is not None: + formula_wb.close() + + +def save_plan(tables: list[tuple[str, pd.DataFrame]], metadata: dict, destination: str, + *, overwrite: bool = False, chart: dict | None = None) -> dict: + from openpyxl.utils import get_column_letter + + target = output_file(destination, {".json"}, overwrite=overwrite) + names = [name for name, _ in tables] + if len(set(names)) != len(names) or "分析说明" in names: + raise ValueError("结果工作表名称重复,或占用了保留名称 分析说明") + note_rows = [] + + def add_note(key: str, value: Any) -> None: + if isinstance(value, dict) and value: + for child, item in value.items(): + add_note(f"{key}.{child}", item) + return + text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, default=scalar) + for start in range(0, max(1, len(text)), 80): + note_rows.append([key if start == 0 else f"{key}(续)", text[start:start + 80]]) + + for key, value in metadata.items(): + add_note(key, value) + notes = pd.DataFrame(note_rows, columns=["项目", "说明"]) + tables = [*tables, ("分析说明", notes)] + operations = [] + cell_count = 0 + for name, table in tables: + if not name or len(name) > 31 or any(ch in name for ch in '[]:*?/\\'): + raise ValueError(f"无效的工作表名称:{name}") + if table.empty and len(table.columns) == 0: + raise ValueError("结果表没有字段") + rows = [[str(col) for col in table.columns], *[[scalar(v) for v in row] for row in table.itertuples(index=False, name=None)]] + # Explicit value objects keep untrusted text out of Excel's formula parser. + rows = [[{"value": v} if isinstance(v, str) else v for v in row] for row in rows] + last_col, last_row = get_column_letter(len(table.columns)), len(rows) + area = f"A1:{last_col}{last_row}" + cell_count += len(table.columns) * len(rows) * 3 + len(table.columns) + operations += [ + {"type": "add_sheet", "name": name}, + {"type": "write_rows", "sheet": name, "rows": rows}, + {"type": "style_range", "sheet": name, "range": area, + "style": {"font": {"name": "Noto Sans CJK SC", "size": 11}, "alignment": {"vertical": "center"}}}, + {"type": "style_range", "sheet": name, "range": f"A1:{last_col}1", + "style": {"font": {"bold": True, "color": "FFFFFF"}, "fill": {"color": "1F4E78"}}}, + {"type": "auto_fit", "sheet": name, "range": area, "min_width": 12}, + {"type": "freeze_panes", "sheet": name, "cell": "A2"}, + {"type": "set_print", "sheet": name, "print_area": area, "orientation": "landscape", "fit_to_width": 1, "fit_to_height": 0, "repeat_rows": "1:1"}, + ] + for column_index, column in enumerate(table.columns, 1): + values = table[column].dropna() + if len(values) and all(isinstance(v, (int, float, np.number)) and not isinstance(v, (bool, np.bool_)) for v in values): + letter = get_column_letter(column_index) + number_format = "#,##0" if all(float(v).is_integer() for v in values) else "#,##0.####" + operations.append({"type": "style_range", "sheet": name, "range": f"{letter}2:{letter}{last_row}", + "style": {"number_format": number_format, "alignment": {"horizontal": "right", "indent": 1}}}) + cell_count += len(table) + if cell_count > 100_000: + raise ValueError("结果超过 apply_workbook 的单次处理上限;请缩小结果范围或分组处理") + if chart: + name, table = tables[0] + category = chart["category"] + values = chart["values"] + require_columns(table, [category, *values]) + indexes = [table.columns.get_loc(value) + 1 for value in values] + if indexes != list(range(min(indexes), max(indexes) + 1)): + raise ValueError("图表 values 需按顺序选择相邻的结果列") + c = get_column_letter(table.columns.get_loc(category) + 1) + end = len(table) + 1 + if end < 2: + raise ValueError("没有数据可用于图表") + operations += [{"type": "add_chart", "sheet": name, "chart_type": chart.get("type", "column"), + "title": chart.get("title", name), "data_range": f"{get_column_letter(min(indexes))}1:{get_column_letter(max(indexes))}{end}", + "categories_range": f"{c}2:{c}{end}", "anchor": f"A{end+3}", "width": 20, "height": 10, + "legend_position": "b", "x_axis_title": category, "y_axis_title": chart.get("value_title", "数值")}, + {"type": "set_print", "sheet": name, "print_area": f"A1:{get_column_letter(max(16, len(table.columns)))}{end+27}", "orientation": "landscape", "fit_to_width": 1, "fit_to_height": 0}] + plan = {"operations": operations, "active_sheet": tables[0][0]} + raw = json.dumps(plan, ensure_ascii=False, allow_nan=False, default=scalar).encode() + if len(raw) > 2 * 1024 * 1024: + raise ValueError("结果操作说明超过 2 MiB;请缩小结果或分批输出") + descriptor, temp_name = tempfile.mkstemp(dir=target.parent, suffix=".json") + temp = Path(temp_name) + try: + with os.fdopen(descriptor, "wb") as handle: + handle.write(raw) + publish_file(temp, target, overwrite=overwrite) + finally: + temp.unlink(missing_ok=True) + return {"spec_path": str(target), "tables": [{"sheet": n, "rows": len(t), "columns": len(t.columns)} for n, t in tables], + "next_script": "scripts/apply_workbook.py", "metadata": metadata} diff --git a/skills/xlsx/scripts/analyze_workbook.py b/skills/xlsx/scripts/analyze_workbook.py new file mode 100644 index 0000000..a03cddd --- /dev/null +++ b/skills/xlsx/scripts/analyze_workbook.py @@ -0,0 +1,249 @@ +#!/usr/bin/env python3 +"""Analyze table data and emit JSON operations for apply_workbook.py.""" +from __future__ import annotations + +import re +from typing import Any + +import numpy as np +import pandas as pd + +from _xlsx_common import SkillArgumentParser, load_json_argument, run_cli +from _xlsx_data import MAX_DATA_CELLS, SOURCE_ROW, numeric, read_dataset, require_columns, save_plan, scalar + + +def profile(frame: pd.DataFrame, source: dict) -> dict: + fields = [] + for name in frame.columns: + if name == SOURCE_ROW: + continue + series = frame[name] + fields.append({"name": name, "missing": int(series.isna().sum()), + "types": {str(k): int(v) for k, v in series.dropna().map(lambda x: type(x).__name__).value_counts().items()}, + "unique": int(series.nunique(dropna=True)), + "examples": [scalar(x) for x in series.dropna().head(5)]}) + marker = re.compile(r"^(?:合计|总计|小计|汇总|累计|grand total|subtotal|total)(?:\s|[::]|$)", re.I) + candidates = [] + for row in frame.itertuples(index=False, name=None): + matches = [str(x) for x in row[1:] if isinstance(x, str) and marker.search(x.strip())] + if matches: + candidates.append({"row": int(row[0]), "labels": matches[:3]}) + return {"source": source, "columns": fields, "summary_row_candidates": candidates[:100], + "summary_row_candidate_count": len(candidates), "summary_candidates_truncated": len(candidates) > 100, + "note": "候选汇总行未自动排除;先确认口径,再用 source.exclude_rows 指定原始行号。"} + + +def transform(frame: pd.DataFrame, operations: list[dict], audit: list) -> pd.DataFrame: + if not isinstance(operations, list) or len(operations) > 50: + raise ValueError("steps 必须为不超过 50 项的列表") + for op in operations: + kind = op["type"] + before = len(frame) + cols = op.get("columns", []) + if cols: + require_columns(frame, cols) + if kind == "trim": + for col in cols: + frame[col] = frame[col].map(lambda x: x.strip() if isinstance(x, str) else x) + elif kind == "replace": + frame[cols] = frame[cols].replace(op["mapping"]) + elif kind == "numeric": + frame[cols] = numeric(frame, cols, allow_missing=True) + elif kind == "date": + for col in cols: + frame[col] = pd.to_datetime(frame[col], format=op["format"], errors="raise") + elif kind == "fill": + if op.get("method") == "ffill": + frame[cols] = frame[cols].ffill() + elif "value" in op: + frame[cols] = frame[cols].fillna(op["value"]) + else: + raise ValueError("fill 需明确 method=ffill 或 value") + elif kind == "drop_missing": + frame = frame.dropna(subset=require_columns(frame, cols)) + elif kind == "deduplicate": + keep = op.get("keep", "first") + if keep not in ("first", "last", False): + raise ValueError("deduplicate.keep 仅支持 first、last 或 false") + frame = frame.drop_duplicates(subset=require_columns(frame, cols), keep=keep) + elif kind == "filter": + col = require_columns(frame, [op["column"]])[0] + series, value, predicate = frame[col], op.get("value"), op["operator"] + if predicate in {"eq", "ne", "gt", "ge", "lt", "le"}: + mask = getattr(series, predicate)(value) + elif predicate in {"in", "not_in"}: + mask = series.isin(value) + if predicate == "not_in": + mask = ~mask + elif predicate in {"is_missing", "not_missing"}: + mask = series.isna() if predicate == "is_missing" else series.notna() + elif predicate == "contains": + mask = series.astype("string").str.contains(str(value), regex=False, na=False) + else: + raise ValueError(f"不支持的 filter.operator:{predicate}") + frame = frame.loc[mask.fillna(False)].copy() + elif kind == "sort": + frame = frame.sort_values(require_columns(frame, cols), ascending=op.get("ascending", True), kind="stable") + elif kind == "select": + chosen = require_columns(frame, cols) + frame = frame[list(dict.fromkeys([SOURCE_ROW, *chosen]))] + elif kind == "rename": + mapping = op["mapping"] + require_columns(frame, list(mapping)) + if SOURCE_ROW in mapping or SOURCE_ROW in mapping.values(): + raise ValueError("不能重命名保留的原始行号列") + frame = frame.rename(columns=mapping) + elif kind in {"merge", "concat"}: + other, info = read_dataset(op["source"]["path"], {k: v for k, v in op["source"].items() if k != "path"}) + if kind == "concat": + if set(frame.columns) - {"__source_file"} != set(other.columns) - {"__source_file"}: + raise ValueError("concat 需要相同字段;请先对齐字段") + # Row numbers alone are ambiguous after stacking multiple files. + frame = frame.copy() + if "__source_file" not in frame: + frame["__source_file"] = op.get("left_source_label", "primary") + other["__source_file"] = info["path"] + frame = pd.concat([frame, other], ignore_index=True) + else: + keys = require_columns(frame, op["on"]) + require_columns(other, keys) + if frame[keys].isna().any().any() or other[keys].isna().any().any(): + raise ValueError("merge 的连接键含空值;请先处理,避免把不同空值记录相互匹配") + validate = op.get("validate", "many_to_one") + if validate not in {"one_to_one", "one_to_many", "many_to_one"}: + raise ValueError("merge.validate 仅支持 one_to_one、one_to_many、many_to_one") + how = op.get("how", "left") + if how not in {"left", "inner", "right", "outer"}: + raise ValueError("merge.how 无效") + frame = frame.merge(other, on=keys, how=how, validate=validate, suffixes=("", "_right")) + audit.append({"additional_source": info}) + else: + raise ValueError(f"不支持的处理步骤:{kind}") + if not frame.columns.is_unique: + raise ValueError("处理后出现重复字段名") + if frame.size > MAX_DATA_CELLS: + raise ValueError("处理结果超过 500000 单元格") + audit.append({"step": kind, "rows_before": before, "rows_after": len(frame)}) + return frame + + +def aggregate(frame: pd.DataFrame, spec: dict, *, pivot: bool) -> pd.DataFrame: + by = require_columns(frame, spec["by"]) + metrics = spec["metrics"] + if not isinstance(metrics, dict) or not metrics: + raise ValueError("metrics 需为字段到聚合方式的非空映射") + require_columns(frame, list(metrics)) + allowed = {"sum", "count", "size", "mean", "min", "max", "median", "nunique"} + if set(metrics.values()) - allowed: + raise ValueError(f"聚合方式仅支持:{sorted(allowed)}") + numeric_cols = [col for col, method in metrics.items() if method in {"sum", "mean", "min", "max", "median"}] + if numeric_cols: + frame = frame.copy() + frame[numeric_cols] = numeric(frame, numeric_cols, allow_missing=True) + reducers = {col: (lambda x: x.sum(min_count=1)) if method == "sum" else method for col, method in metrics.items()} + column_fields = spec.get("columns", []) if pivot else [] + if column_fields: + require_columns(frame, column_fields) + if set(by) & set(column_fields): + raise ValueError("行维度与列维度不能重复") + result = frame.groupby(by + column_fields, dropna=False, sort=False, observed=True).agg(reducers) + if column_fields: + result = result.unstack(column_fields) + result.columns = [json_label(parts) for parts in result.columns.to_flat_index()] + return result.reset_index() + + +def json_label(parts: Any) -> str: + import json + return json.dumps([scalar(x) for x in parts], ensure_ascii=False, separators=(",", ":")) + + +def classify(frame: pd.DataFrame, spec: dict) -> tuple[pd.DataFrame, pd.DataFrame]: + column = require_columns(frame, [spec["column"]])[0] + rules = spec["rules"] + if not isinstance(rules, list) or len(rules) > 100: + raise ValueError("rules 必须为不超过 100 项的规则列表") + for rule in rules: + if not rule.get("label") or not rule.get("keywords") or not all(isinstance(v, str) and v for v in rule["keywords"]): + raise ValueError("每条规则必须有 label 和非空 keywords") + records = [] + for _, row in frame.iterrows(): + raw = row[column] + cleaned = " ".join(str(raw).split()) if pd.notna(raw) else "" + matches = [(rule["label"], [word for word in rule["keywords"] if word.casefold() in cleaned.casefold()]) for rule in rules] + hits = [(label, words) for label, words in matches if words] + labels = list(dict.fromkeys(label for label, _ in hits)) + category = labels[0] if len(labels) == 1 else ("需复核" if labels else "未知") + records.append([row[SOURCE_ROW], scalar(raw), cleaned, category, + ";".join(labels), ";".join(dict.fromkeys(word for _, words in hits for word in words))]) + detail = pd.DataFrame(records, columns=[SOURCE_ROW, "原文", "清洗文本", "分类", "命中标签", "证据关键词"]) + summary = detail.groupby("分类", dropna=False, sort=False).size().rename("记录数").reset_index() + summary["占比"] = summary["记录数"] / len(detail) if len(detail) else 0 + return detail, summary + + +def analyze(path: str, spec: dict) -> tuple[list[tuple[str, pd.DataFrame]], dict]: + method = spec.get("method", "profile") + options = { + "profile": set(), "transform": set(), "aggregate": {"by", "metrics"}, + "pivot": {"by", "columns", "metrics"}, "describe": {"columns"}, + "correlate": {"columns", "correlation"}, "classify": {"column", "rules"}, + } + if method not in options: + raise ValueError(f"不支持的 method:{method}") + unknown = set(spec) - {"method", "source", "steps", "result_sheet", "chart"} - options[method] + if unknown: + raise ValueError(f"分析说明包含未知参数:{sorted(unknown)}") + frame, source = read_dataset(path, spec.get("source")) + audit: list = [] + frame = transform(frame, spec.get("steps", []), audit) + method = spec.get("method", "profile") + metadata = {"source": source, "method": method, "steps": audit, "parameters": spec, + "result_kind": "数据处理结果快照;需要随输入变化时用工作簿公式,复杂分析需按相同参数重新运行"} + if method == "profile": + return [], profile(frame, source) + if method == "transform": + result = frame + elif method in {"aggregate", "pivot"}: + result = aggregate(frame, spec, pivot=method == "pivot") + metadata["aggregation_note"] = "空分类保留;count 排除空值,size 包含空值;全空 sum 保持空白;交叉汇总是静态表,不含原生透视控件。" + elif method == "describe": + data = numeric(frame, spec["columns"], allow_missing=True) + result = data.describe().rename_axis("统计量").reset_index() + elif method == "correlate": + data = numeric(frame, spec["columns"], allow_missing=True) + correlation = spec.get("correlation", "pearson") + if correlation not in {"pearson", "spearman"}: + raise ValueError("correlation 仅支持 pearson、spearman") + result = data.corr(method=correlation, min_periods=3).rename_axis("字段").reset_index() + valid = data.notna().astype(int) + counts = (valid.T @ valid).rename_axis("字段").reset_index() + return [(spec.get("result_sheet", "相关系数"), result), ("配对样本数", counts)], metadata + elif method == "classify": + details, summary = classify(frame, spec) + metadata["classification_note"] = "仅按显式关键词匹配;多标签冲突标为需复核,未命中标为未知,不能视为自动情感理解。" + return [(spec.get("result_sheet", "分类明细"), details), ("分类统计", summary)], metadata + else: + raise ValueError(f"不支持的 method:{method}") + return [(spec.get("result_sheet", "分析结果"), result)], metadata + + +def main() -> dict: + parser = SkillArgumentParser(description="分组汇总、交叉分析、清洗、规则分类;输出 xlsx 写入操作 JSON。") + parser.add_argument("--input", required=True) + parser.add_argument("--spec") + parser.add_argument("--spec-file") + parser.add_argument("--output", help="输出 .json 操作说明,profile 可省略") + parser.add_argument("--overwrite", action="store_true") + args = parser.parse_args() + spec = load_json_argument(args.spec, args.spec_file, label="分析说明") + tables, metadata = analyze(args.input, spec) + if not tables: + return metadata + if not args.output: + raise ValueError("此方法需 --output 指定 .json 文件") + return save_plan(tables, metadata, args.output, overwrite=args.overwrite, chart=spec.get("chart")) + + +if __name__ == "__main__": + raise SystemExit(run_cli(main)) diff --git a/skills/xlsx/scripts/apply_workbook.py b/skills/xlsx/scripts/apply_workbook.py index cc07aef..9cf1cd9 100644 --- a/skills/xlsx/scripts/apply_workbook.py +++ b/skills/xlsx/scripts/apply_workbook.py @@ -241,7 +241,11 @@ def _set_cell(cell: Any, item: dict[str, Any]) -> None: raise ValueError(f"{cell.coordinate} 的 formula 必须以 = 开头") cell.value = formula elif "value" in item: + if isinstance(item["value"], str) and len(item["value"]) > 32767: + raise ValueError(f"{cell.coordinate} 的文本超过 Excel 单元格上限,不能静默截断") cell.value = item["value"] + if isinstance(item["value"], str): + cell.data_type = "s" if "style" in item: _apply_style(cell, item["style"]) if "comment" in item: @@ -362,6 +366,8 @@ def _op_write_rows(workbook: Any, op: dict[str, Any]) -> int: ): _set_cell(cell, raw_value) else: + if isinstance(raw_value, str) and len(raw_value) > 32767: + raise ValueError(f"{cell.coordinate} 的文本超过 Excel 单元格上限,不能静默截断") cell.value = raw_value if "style" in op: style = op["style"] @@ -494,6 +500,48 @@ def _op_set_row_heights(workbook: Any, op: dict[str, Any]) -> int: return changed +def _op_auto_fit(workbook: Any, op: dict[str, Any]) -> int: + import math + import unicodedata + from openpyxl.utils.cell import get_column_letter, range_boundaries + + worksheet = _sheet(workbook, op.get("sheet")) + reference = validate_cell_range(str(op.get("range", ""))) + cells = list(_iter_range_cells(worksheet, reference)) + min_col, min_row, max_col, max_row = range_boundaries(reference) + minimum, maximum = float(op.get("min_width", 8)), float(op.get("max_width", 40)) + if not 1 <= minimum <= maximum <= 100: + raise ValueError("auto_fit 列宽需满足 1 <= min_width <= max_width <= 100") + + def lines(cell: Any) -> list[float]: + if cell.data_type == "f": + return [12.] + text = "" if cell.value is None else str(cell.value) + scale = (cell.font.sz or 11) / 11 + return [sum(2 if unicodedata.east_asian_width(ch) in "WF" else 1 for ch in line) * scale for line in text.split("\n")] + + for column in range(min_col, max_col + 1): + length = max((max(lines(worksheet.cell(row, column))) for row in range(min_row, max_row + 1)), default=0) + width = min(maximum, max(minimum, length + 3)) + worksheet.column_dimensions[get_column_letter(column)].width = width + for row in range(min_row, max_row + 1): + cell = worksheet.cell(row, column) + if cell.value is not None and (max(lines(cell)) + 3 > width or "\n" in str(cell.value)): + alignment = copy(cell.alignment) + alignment.wrap_text = True + cell.alignment = alignment + for row in range(min_row, max_row + 1): + height = 30 if row == min_row else 20 + for column in range(min_col, max_col + 1): + cell = worksheet.cell(row, column) + if cell.alignment.wrap_text: + width = max(1., worksheet.column_dimensions[get_column_letter(column)].width - 3) + count = sum(max(1, math.ceil(length / width)) for length in lines(cell)) + height = max(height, math.ceil(count * ((cell.font.sz or 11) + 4) * 1.25) + 8) + worksheet.row_dimensions[row].height = min(409., height) + return len(cells) + + def _op_add_table(workbook: Any, op: dict[str, Any]) -> int: from openpyxl.worksheet.table import Table, TableStyleInfo @@ -797,6 +845,7 @@ def _apply_operation(workbook: Any, raw_op: Any) -> int: "clear_range": _op_clear_range, "set_column_widths": _op_set_column_widths, "set_row_heights": _op_set_row_heights, + "auto_fit": _op_auto_fit, "add_table": _op_add_table, "add_chart": _op_add_chart, "add_image": _op_add_image, @@ -837,9 +886,7 @@ def _scan_workbook(workbook: Any) -> dict[str, Any]: for worksheet in workbook.worksheets: for cell in worksheet._cells.values(): value = cell.value - if cell.data_type == "f" or ( - isinstance(value, str) and value.startswith("=") - ): + if cell.data_type == "f": formula_count += 1 if ( isinstance(value, str) diff --git a/skills/xlsx/scripts/inspect_workbook.py b/skills/xlsx/scripts/inspect_workbook.py index 902a04c..ac76cef 100644 --- a/skills/xlsx/scripts/inspect_workbook.py +++ b/skills/xlsx/scripts/inspect_workbook.py @@ -40,9 +40,7 @@ def _cell_payload(formula_cell: Any, cached_cell: Any) -> Optional[dict[str, Any if formula_value is None and cached_value is None and not formula_cell.has_style: return None - is_formula = formula_cell.data_type == "f" or ( - isinstance(formula_value, str) and formula_value.startswith("=") - ) + is_formula = formula_cell.data_type == "f" value = cached_value if is_formula else formula_value error = normalize_formula_error(value) payload: dict[str, Any] = { @@ -120,9 +118,7 @@ def inspect_excel( formula_count = 0 error_count = 0 for cell in worksheet._cells.values(): - if cell.data_type == "f" or ( - isinstance(cell.value, str) and cell.value.startswith("=") - ): + if cell.data_type == "f": formula_count += 1 if normalize_formula_error(cell.value): error_count += 1 diff --git a/skills/xlsx/scripts/model_workbook.py b/skills/xlsx/scripts/model_workbook.py new file mode 100644 index 0000000..79c617d --- /dev/null +++ b/skills/xlsx/scripts/model_workbook.py @@ -0,0 +1,423 @@ +#!/usr/bin/env python3 +"""Controlled modeling tasks; workbook creation stays with apply_workbook.py.""" +from __future__ import annotations + +import math +from typing import Any + +import numpy as np +import pandas as pd + +from _xlsx_common import SkillArgumentParser, load_json_argument, run_cli +from _xlsx_data import SOURCE_ROW, numeric, read_dataset, require_columns, save_plan, scalar + + +def evaluate(frame: pd.DataFrame, spec: dict) -> tuple[list, dict]: + directions = spec["directions"] + columns = require_columns(frame, list(directions)) + data = numeric(frame, columns).to_numpy() + if len(data) < 2: + raise ValueError("综合评价至少需要两个对象") + positive = np.zeros_like(data) + for j, name in enumerate(columns): + direction = directions[name] + col = data[:, j] + if direction == "benefit": + transformed = col - col.min() + elif direction == "cost": + transformed = col.max() - col + elif isinstance(direction, dict) and "target" in direction: + distance = np.abs(col - float(direction["target"])) + transformed = distance.max() - distance + else: + raise ValueError("指标方向需为 benefit、cost 或含 target 的对象") + if not np.isfinite(transformed).all(): + raise ValueError("指标正向化结果无效") + positive[:, j] = transformed / transformed.max() if transformed.max() > 0 else 0 + method = spec.get("weighting", "equal") + entropy = np.ones(len(columns)) + consistency = None + if method == "equal": + weights = np.ones(len(columns)) + elif method == "user": + if set(spec["weights"]) != set(columns): + raise ValueError("weights 必须覆盖且仅覆盖全部指标") + weights = np.array([spec["weights"][col] for col in columns], dtype=float) + elif method == "entropy": + sums = positive.sum(axis=0) + p = np.divide(positive, sums, out=np.zeros_like(positive), where=sums > 0) + logs = np.zeros_like(p) + np.log(p, out=logs, where=p > 0) + entropy = -(p * logs).sum(axis=0) / math.log(len(data)) + entropy[sums == 0] = 1 + weights = np.maximum(0, 1 - entropy) + if not weights.any(): + weights = np.ones(len(columns)) + elif method == "ahp": + matrix = np.asarray(spec["comparison_matrix"], dtype=float) + n = len(columns) + if not 2 <= n <= 9 or matrix.shape != (n, n) or not np.isfinite(matrix).all() or (matrix <= 0).any(): + raise ValueError("AHP 需 2–9 阶正数比较矩阵,顺序与 directions 相同") + if not np.allclose(np.diag(matrix), 1) or not np.allclose(matrix * matrix.T, 1, atol=1e-6): + raise ValueError("AHP 比较矩阵必须对角为 1 且互反") + values, vectors = np.linalg.eig(matrix) + index = np.argmax(values.real) + weights = np.abs(vectors[:, index].real) + ri = [0, 0, 0, .58, .90, 1.12, 1.24, 1.32, 1.41, 1.45][n] + consistency = max(0, float(values[index].real - n) / (n - 1) / ri) if ri else 0 + if consistency >= .1: + raise ValueError(f"AHP 一致性未通过:CR={consistency:.4f},需调整用户比较矩阵") + else: + raise ValueError("weighting 仅支持 equal、user、entropy、ahp") + if not np.isfinite(weights).all() or (weights < 0).any() or weights.sum() <= 0: + raise ValueError("权重必须非负、有限且总和大于 0") + weights /= weights.sum() + norms = np.linalg.norm(positive, axis=0) + weighted = np.divide(positive, norms, out=np.zeros_like(positive), where=norms > 0) * weights + # All columns are already benefit-oriented; reversing cost columns again is wrong. + d_best = np.linalg.norm(weighted - weighted.max(axis=0), axis=1) + d_worst = np.linalg.norm(weighted - weighted.min(axis=0), axis=1) + scores = np.divide(d_worst, d_best + d_worst, out=np.full(len(data), .5), where=d_best + d_worst > 0) + entity = require_columns(frame, [spec["entity"]])[0] + result = frame[[SOURCE_ROW, entity, *columns]].copy() + result["得分"] = scores + result["排名"] = result["得分"].round(12).rank(method="min", ascending=False).astype(int) + result = result.sort_values("排名", kind="stable") + weight_table = pd.DataFrame({"指标": columns, "方向": [str(directions[c]) for c in columns], "权重": weights, + "信息熵": entropy if method == "entropy" else [None] * len(columns), + "无区分度": ["是" if value == 0 else "否" for value in norms]}) + return [("综合排名", result), ("指标权重", weight_table)], {"weighting": method, "ahp_cr": consistency, + "note": "正向化后统一采用最大值作为正理想解;无区分度对象可并列,熵权反映差异性而非业务重要性。"} + + +def _forecast_values(history: np.ndarray, horizon: int, method: str, season: int) -> np.ndarray: + if method == "naive": + return np.repeat(history[-1], horizon) + if method == "linear": + slope, intercept = np.polyfit(np.arange(len(history)), history, 1) + return intercept + slope * np.arange(len(history), len(history) + horizon) + if method == "moving_average": + values = list(history) + for _ in range(horizon): + values.append(float(np.mean(values[-min(3, len(values)):]))) + return np.asarray(values[-horizon:]) + if method == "seasonal_naive" and 1 <= season <= len(history): + return np.array([history[-season + (i % season)] for i in range(horizon)]) + raise ValueError("预测方法无效,或 seasonal_period 超过训练数据长度") + + +def forecast(frame: pd.DataFrame, spec: dict) -> tuple[list, dict]: + value, date_col = require_columns(frame, [spec["value"], spec["date"]]) + groups = spec.get("by", []) + if groups: + require_columns(frame, groups) + horizon = int(spec.get("horizon", 3)) + if not 1 <= horizon <= 120: + raise ValueError("horizon 必须在 1–120 之间") + methods = spec.get("methods", ["naive", "linear"]) + if not methods or set(methods) - {"naive", "linear", "moving_average", "seasonal_naive"}: + raise ValueError("不支持的预测方法") + freq = spec.get("frequency", "MS") + if freq not in {"D", "W", "MS", "QS", "YS"}: + raise ValueError("frequency 仅支持 D、W、MS、QS、YS") + frame = frame.copy() + frame[value] = numeric(frame, [value])[value] + frame[date_col] = pd.to_datetime(frame[date_col], errors="raise") + if frame[date_col].isna().any(): + raise ValueError("预测日期列不能缺失") + predictions, validations = [], [] + iterator = frame.groupby(groups, dropna=False, sort=False) if groups else [((), frame)] + for key, group in iterator: + group = group.sort_values(date_col) + keys = list(key if isinstance(key, tuple) else (key,)) if groups else [] + dates = pd.DatetimeIndex(group[date_col]) + expected = pd.date_range(dates[0], dates[-1], freq=freq) + if len(group) < 6 or dates.has_duplicates or not dates.equals(expected): + raise ValueError(f"分组 {keys} 需至少 6 期连续、无重复的 {freq} 数据;缺期不能自动当作 0") + holdout = int(spec.get("holdout", max(2, len(group) // 4))) + if not 1 <= holdout <= len(group) - 3: + raise ValueError("holdout 需保留至少 3 个训练时点") + values = group[value].to_numpy(dtype=float) + train, test = values[:-holdout], values[-holdout:] + season = int(spec.get("seasonal_period", 1)) + candidates = [] + for method in methods: + predicted = _forecast_values(train, holdout, method, season) + mae = float(np.mean(np.abs(predicted - test))) + rmse = float(np.sqrt(np.mean((predicted - test) ** 2))) + validations.append([*keys, method, len(train), holdout, mae, rmse]) + candidates.append((mae, method)) + selected = min(candidates, key=lambda item: item[0])[1] + future = _forecast_values(values, horizon, selected, season) + future_dates = pd.date_range(dates[-1], periods=horizon + 1, freq=freq)[1:] + predictions += [[*keys, date, float(prediction), selected] for date, prediction in zip(future_dates, future)] + return [("未来预测", pd.DataFrame(predictions, columns=[*groups, date_col, "预测值", "方法"])), + ("时序验证", pd.DataFrame(validations, columns=[*groups, "方法", "训练期数", "验证期数", "MAE", "RMSE"])), + ("历史数据", frame)], {"selection": "按时间留出验证集,以 MAE 选择方法;这不是独立测试集。", + "uncertainty": "基础趋势外推,无预测区间,未自动补期、截断负预测或假定季节性。"} + + +def supervised(frame: pd.DataFrame, spec: dict, *, classification: bool) -> tuple[list, dict]: + from sklearn.compose import ColumnTransformer + from sklearn.dummy import DummyClassifier, DummyRegressor + from sklearn.ensemble import RandomForestClassifier, RandomForestRegressor + from sklearn.impute import SimpleImputer + from sklearn.linear_model import LinearRegression, LogisticRegression, Ridge + from sklearn.metrics import accuracy_score, f1_score, mean_absolute_error, mean_squared_error, r2_score + from sklearn.model_selection import train_test_split + from sklearn.pipeline import Pipeline + from sklearn.preprocessing import OneHotEncoder, StandardScaler + + features = require_columns(frame, spec["features"]) + target = require_columns(frame, [spec["target"]])[0] + if target in features or SOURCE_ROW in features: + raise ValueError("目标列和原始行号不能作为特征") + categorical = spec.get("categorical", []) + if set(categorical) - set(features): + raise ValueError("categorical 必须是 features 的子集") + numerical = [col for col in features if col not in categorical] + X = frame[features].copy() + if numerical: + X[numerical] = numeric(frame, numerical, allow_missing=True) + for col in categorical: + X[col] = X[col].map(lambda x: str(x) if pd.notna(x) else np.nan) + if frame[target].isna().any() or len(frame) < 10: + raise ValueError("监督学习至少需要 10 行且目标列不能缺失") + y = frame[target].astype(str) if classification else numeric(frame, [target])[target] + fraction = float(spec.get("test_fraction", .2)) + if not .1 <= fraction <= .5: + raise ValueError("test_fraction 必须在 0.1–0.5 之间") + test_rows = max(2, math.ceil(len(frame) * fraction)) + indexes = np.arange(len(frame)) + if spec.get("time_column"): + date_col = require_columns(frame, [spec["time_column"]])[0] + dates = pd.to_datetime(frame[date_col], errors="raise") + if dates.isna().any(): + raise ValueError("时间列不能缺失") + indexes = np.argsort(dates.to_numpy(), kind="stable") + boundary = len(frame) - test_rows + train, test = indexes[:boundary], indexes[boundary:] + if dates.iloc[train].max() >= dates.iloc[test].min(): + raise ValueError("时间切分边界有相同时点;请先按时点汇总或调整切分比例") + else: + train, test = train_test_split(indexes, test_size=test_rows, random_state=42, stratify=y if classification else None) + if classification and (y.iloc[train].nunique() < 2 or not set(y.iloc[test]) <= set(y.iloc[train])): + raise ValueError("训练集必须覆盖至少两个类别且包含测试集所有类别") + transformers = [] + if numerical: + transformers.append(("numeric", Pipeline([("impute", SimpleImputer(strategy="median", keep_empty_features=True)), ("scale", StandardScaler())]), numerical)) + if categorical: + transformers.append(("category", Pipeline([("impute", SimpleImputer(strategy="constant", fill_value="缺失", keep_empty_features=True)), + ("encode", OneHotEncoder(handle_unknown="ignore", sparse_output=False, max_categories=50))]), categorical)) + method = spec.get("algorithm", "logistic" if classification else "linear") + choices = ({"logistic": LogisticRegression(max_iter=1000, class_weight="balanced"), + "forest": RandomForestClassifier(n_estimators=100, max_depth=8, random_state=42, n_jobs=1, class_weight="balanced")} + if classification else {"linear": LinearRegression(), "ridge": Ridge(alpha=1.), + "forest": RandomForestRegressor(n_estimators=100, max_depth=8, random_state=42, n_jobs=1)}) + if method not in choices: + raise ValueError(f"algorithm 可选:{list(choices)}") + pipeline = Pipeline([("prepare", ColumnTransformer(transformers)), ("model", choices[method])]) + pipeline.fit(X.iloc[train], y.iloc[train]) + baseline = DummyClassifier(strategy="most_frequent") if classification else DummyRegressor(strategy="mean") + baseline.fit(np.zeros((len(train), 1)), y.iloc[train]) + metrics = [] + for label, subset in (("训练", train), ("测试", test)): + actual = y.iloc[subset] + for algorithm, predicted in ((method, pipeline.predict(X.iloc[subset])), ("基线", baseline.predict(np.zeros((len(subset), 1))))): + stats = {"Accuracy": accuracy_score(actual, predicted), "F1_macro": f1_score(actual, predicted, average="macro", zero_division=0)} if classification else { + "MAE": mean_absolute_error(actual, predicted), "RMSE": math.sqrt(mean_squared_error(actual, predicted)), "R2": r2_score(actual, predicted)} + metrics += [[label, algorithm, key, float(value)] for key, value in stats.items()] + predictions = frame.iloc[test][[SOURCE_ROW, *features]].copy() + predictions["实际值"] = y.iloc[test].to_numpy() + predictions["预测值"] = pipeline.predict(X.iloc[test]) + model = pipeline.named_steps["model"] + names = pipeline.named_steps["prepare"].get_feature_names_out() + importance = model.feature_importances_ if hasattr(model, "feature_importances_") else np.mean(np.abs(np.atleast_2d(model.coef_)), axis=0) + features_table = pd.DataFrame({"特征": names, "重要性或绝对系数": importance}).sort_values("重要性或绝对系数", ascending=False) + tables = [("留出预测", predictions), ("模型指标", pd.DataFrame(metrics, columns=["数据集", "模型", "指标", "值"])), ("特征说明", features_table)] + if spec.get("predict_source"): + request = spec["predict_source"] + future, provenance = read_dataset(request["path"], {k: v for k, v in request.items() if k != "path"}) + require_columns(future, features) + future_X = future[features].copy() + if numerical: + future_X[numerical] = numeric(future, numerical, allow_missing=True) + for col in categorical: + future_X[col] = future_X[col].map(lambda x: str(x) if pd.notna(x) else np.nan) + pipeline.fit(X, y) + future = future[[SOURCE_ROW, *features]].copy() + future["预测值"] = pipeline.predict(future_X) + tables.append(("新样本预测", future)) + else: + provenance = None + return tables, {"algorithm": method, "train_rows": len(train), "test_rows": len(test), "prediction_source": provenance, + "note": "训练集拟合填补、编码和标准化;固定留出集对比简单基线。特征重要性不是因果影响。新样本预测使用全量训练数据重新拟合。"} + + +def unsupervised(frame: pd.DataFrame, spec: dict, *, anomaly: bool) -> tuple[list, dict]: + from sklearn.cluster import DBSCAN, KMeans + from sklearn.ensemble import IsolationForest + from sklearn.metrics import silhouette_score + from sklearn.preprocessing import StandardScaler + + columns = require_columns(frame, spec["features"]) + data = numeric(frame, columns) + if len(data) < 3: + raise ValueError("至少需要 3 条完整数值记录") + scaled = StandardScaler().fit_transform(data) + if anomaly: + contamination = float(spec.get("contamination", .05)) + if not 0 < contamination <= .5: + raise ValueError("contamination 必须在 (0, 0.5] 之间") + model = IsolationForest(contamination=contamination, random_state=42, n_jobs=1) + elif spec.get("algorithm", "kmeans") == "kmeans": + k = int(spec.get("clusters", 3)) + if not 2 <= k < len(data): + raise ValueError("clusters 必须至少为 2 且小于样本数") + model = KMeans(n_clusters=k, n_init=10, random_state=42) + elif spec["algorithm"] == "dbscan": + model = DBSCAN(eps=float(spec.get("eps", .5)), min_samples=int(spec.get("min_samples", 5))) + else: + raise ValueError("聚类 algorithm 仅支持 kmeans、dbscan") + labels = model.fit_predict(scaled) + result = frame.copy() + result["异常标记" if anomaly else "簇编号"] = labels + if anomaly: + result["正常程度得分"] = model.decision_function(scaled) + score = None + valid = labels != -1 + if not anomaly and 1 < len(set(labels[valid])) < valid.sum(): + score = float(silhouette_score(scaled[valid], labels[valid], sample_size=min(2000, int(valid.sum())), random_state=42)) + return [("异常检测" if anomaly else "聚类结果", result)], {"silhouette_without_noise": score, + "note": "数值字段先标准化;-1 表示异常候选或 DBSCAN 噪声,不自动删除。聚类编号没有优劣顺序。"} + + +def optimize(spec: dict) -> tuple[list, dict]: + from scipy.optimize import Bounds, LinearConstraint, milp, minimize + + names = spec["variables"] + if not isinstance(names, list) or not 1 <= len(names) <= 200 or len(set(names)) != len(names): + raise ValueError("variables 需为 1–200 个唯一名称") + n = len(names) + c = np.asarray(spec["objective"], dtype=float) + if c.shape != (n,) or not np.isfinite(c).all(): + raise ValueError("objective 需为每个变量的有限数值系数") + direction = spec.get("sense", "min") + if direction not in {"min", "max"}: + raise ValueError("sense 仅支持 min、max") + sign = 1 if direction == "min" else -1 + limits = spec.get("bounds", [[0, None] for _ in names]) + if len(limits) != n or any(len(bound) != 2 for bound in limits): + raise ValueError("bounds 需给出每个变量的 [下限, 上限],null 表示无界") + lower = np.array([-np.inf if x[0] is None else x[0] for x in limits], dtype=float) + upper = np.array([np.inf if x[1] is None else x[1] for x in limits], dtype=float) + if np.isnan(lower).any() or np.isnan(upper).any() or (lower > upper).any(): + raise ValueError("变量边界无效") + constraints = spec.get("constraints", []) + if len(constraints) > 1000: + raise ValueError("约束最多 1000 项") + A, lows, highs = [], [], [] + for constraint in constraints: + row = np.asarray(constraint["coefficients"], dtype=float) + rhs, relation = float(constraint["rhs"]), constraint["relation"] + if row.shape != (n,) or not np.isfinite(row).all() or not math.isfinite(rhs) or relation not in {"<=", ">=", "=="}: + raise ValueError("约束系数、右端值或 relation 无效") + A.append(row) + lows.append(rhs if relation in {">=", "=="} else -np.inf) + highs.append(rhs if relation in {"<=", "=="} else np.inf) + A = np.asarray(A).reshape((-1, n)) + linear = LinearConstraint(A, lows, highs) if constraints else None + integer = spec.get("integer", [False] * n) + if len(integer) != n or any(type(v) is not bool for v in integer): + raise ValueError("integer 需为与变量数一致的布尔列表") + if "quadratic" in spec: + Q = np.asarray(spec["quadratic"], dtype=float) + if any(integer) or Q.shape != (n, n) or not np.isfinite(Q).all() or not np.allclose(Q, Q.T): + raise ValueError("quadratic 需为对称矩阵,二次规划仅支持连续变量") + if np.linalg.eigvalsh(sign * Q).min() < -1e-9: + raise ValueError("仅支持凸最小化或凹最大化的二次目标") + start = np.asarray(spec.get("initial", np.clip(np.zeros(n), lower, upper)), dtype=float) + if start.shape != (n,) or not np.isfinite(start).all(): + raise ValueError("initial 需为有限数值向量") + objective = lambda x: float(c @ x + .5 * x @ Q @ x) + result = minimize(lambda x: sign * objective(x), start, jac=lambda x: sign * (c + Q @ x), method="SLSQP", + bounds=Bounds(lower, upper), constraints=[linear] if linear else [], options={"maxiter": 1000, "ftol": 1e-9}) + guarantee = "凸二次规划的数值解,已检查可行性" + else: + objective = lambda x: float(c @ x) + result = milp(sign * c, integrality=np.asarray(integer, dtype=int), bounds=Bounds(lower, upper), + constraints=linear, options={"time_limit": 60., "mip_rel_gap": 0.}) + guarantee = "HiGHS 求解成功;仅在成功且可行时输出方案" + if not result.success or result.x is None: + raise ValueError(f"求解未成功,不能输出最优方案:status={result.status}; {result.message}") + x = result.x + activity = A @ x + tolerance = 1e-6 + if (x < lower - tolerance).any() or (x > upper + tolerance).any() or (activity < np.asarray(lows) - tolerance).any() or (activity > np.asarray(highs) + tolerance).any(): + raise ValueError("求解结果未通过约束可行性复核") + if any(abs(x[i] - round(x[i])) > tolerance for i in range(n) if integer[i]): + raise ValueError("整数变量未通过整数性复核") + rows = [[constraint.get("name", f"约束{i+1}"), float(activity[i]), constraint["relation"], constraint["rhs"], + float(min(activity[i] - lows[i], highs[i] - activity[i]))] for i, constraint in enumerate(constraints)] + return [("优化方案", pd.DataFrame({"变量": names, "取值": x})), + ("约束复核", pd.DataFrame(rows, columns=["约束", "左端值", "关系", "右端值", "余量"]))], { + "objective_value": objective(x), "solver_status": int(result.status), "guarantee": guarantee, + "note": "未自动放松约束;敏感性分析需明确修改参数并重新求解。"} + + +def model(path: str | None, spec: dict) -> tuple[list, dict]: + task = spec["task"] + options = { + "evaluate": {"entity", "directions", "weighting", "weights", "comparison_matrix"}, + "forecast": {"by", "date", "value", "frequency", "horizon", "methods", "holdout", "seasonal_period"}, + "regression": {"features", "categorical", "target", "algorithm", "test_fraction", "time_column", "predict_source"}, + "classification": {"features", "categorical", "target", "algorithm", "test_fraction", "time_column", "predict_source"}, + "cluster": {"features", "algorithm", "clusters", "eps", "min_samples"}, + "anomaly": {"features", "contamination"}, + "optimize": {"variables", "objective", "sense", "bounds", "integer", "constraints", "quadratic", "initial"}, + } + if task not in options: + raise ValueError(f"不支持的 task:{task}") + unknown = set(spec) - {"task", "source", "chart"} - options[task] + if unknown: + raise ValueError(f"模型说明包含未知参数:{sorted(unknown)}") + source = None + if task == "optimize": + tables, details = optimize(spec) + else: + if not path: + raise ValueError("此任务需要 --input") + frame, source = read_dataset(path, spec.get("source")) + if frame.empty: + raise ValueError("建模范围没有数据记录") + if frame.size > 100_000 or len(frame) > 20_000: + raise ValueError("建模上限为 20000 行、100000 单元格;请缩小明确的建模范围") + if task == "evaluate": + tables, details = evaluate(frame, spec) + elif task == "forecast": + tables, details = forecast(frame, spec) + elif task in {"regression", "classification"}: + tables, details = supervised(frame, spec, classification=task == "classification") + elif task in {"cluster", "anomaly"}: + tables, details = unsupervised(frame, spec, anomaly=task == "anomaly") + else: + raise ValueError(f"不支持的 task:{task}") + return tables, {"task": task, "source": source, "parameters": spec, "results": details, + "result_kind": "模型结果快照,修改输入后需按相同参数重新运行;不是可自动重算的 Excel 公式。"} + + +def main() -> dict: + parser = SkillArgumentParser(description="评价、预测、回归、分类、聚类、异常检测和受控优化;输出 xlsx 写入操作 JSON。") + parser.add_argument("--input") + parser.add_argument("--spec") + parser.add_argument("--spec-file") + parser.add_argument("--output", required=True) + parser.add_argument("--overwrite", action="store_true") + args = parser.parse_args() + spec = load_json_argument(args.spec, args.spec_file, label="模型说明") + tables, metadata = model(args.input, spec) + return save_plan(tables, metadata, args.output, overwrite=args.overwrite, chart=spec.get("chart")) + + +if __name__ == "__main__": + raise SystemExit(run_cli(main)) diff --git a/skills/xlsx/scripts/recalculate_workbook.py b/skills/xlsx/scripts/recalculate_workbook.py index 108cd63..597e1fb 100644 --- a/skills/xlsx/scripts/recalculate_workbook.py +++ b/skills/xlsx/scripts/recalculate_workbook.py @@ -37,9 +37,7 @@ def _scan_formula_results(path: Path) -> dict[str, Any]: for formula_sheet in formulas.worksheets: cached_sheet = cached[formula_sheet.title] for cell in formula_sheet._cells.values(): - is_formula = cell.data_type == "f" or ( - isinstance(cell.value, str) and cell.value.startswith("=") - ) + is_formula = cell.data_type == "f" if not is_formula: continue total_formulas += 1 diff --git a/skills/xlsx/tests/test_data_workflows.py b/skills/xlsx/tests/test_data_workflows.py new file mode 100644 index 0000000..ca3eb4a --- /dev/null +++ b/skills/xlsx/tests/test_data_workflows.py @@ -0,0 +1,218 @@ +import csv +import hashlib +import json +import sys +import unittest +import tempfile +from copy import copy +from pathlib import Path +from unittest.mock import patch + +import numpy as np +import pandas as pd +from openpyxl import Workbook, load_workbook + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / 'scripts')) +import _xlsx_common as common +import _xlsx_data as data +import analyze_workbook as analysis +import apply_workbook as writer +import inspect_workbook as inspector +import model_workbook as modeling + +_TEMP = tempfile.TemporaryDirectory(prefix='xlsx-tests-') +QA = Path(_TEMP.name).resolve() +_ORIGINAL_OUTPUT_ROOT = common.EXCEL_OUTPUT_ROOT + +def tearDownModule(): + common.EXCEL_OUTPUT_ROOT = _ORIGINAL_OUTPUT_ROOT + _TEMP.cleanup() + + +class MergeTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + common.EXCEL_OUTPUT_ROOT = QA + cls.source = QA / 'sales.xlsx' + wb = Workbook() + ws = wb.active + ws.title = '明细' + for row in [['地区', '收入', '成本', '说明'], ['华东', 100, 60, '满意'], ['华东', -20, 5, '退款'], ['华南', 80, 50, '物流慢'], ['华南', None, 20, '服务好但物流慢'], ['合计', 160, 135, None]]: + ws.append(row) + font = copy(ws['A1'].font); font.bold = True; ws['A1'].font = font + wb.save(cls.source) + cls.source_hash = hashlib.sha256(cls.source.read_bytes()).hexdigest() + cls.csv = QA / 'text.csv' + with cls.csv.open('w', encoding='utf-8-sig', newline='') as f: + csv.writer(f).writerows([['ID', '文本'], ['001', '=1+1'], ['002', '长文本' * 80]]) + + def dataset(self): + return data.read_dataset(str(self.source), {'sheet': '明细', 'exclude_rows': [6]})[0] + + def test_profile_finds_totals_without_dropping_negative_rows(self): + _, result = analysis.analyze(str(self.source), {'method': 'profile'}) + self.assertEqual(result['source']['rows_used'], 5) + self.assertEqual(result['summary_row_candidates'][0]['row'], 6) + + def test_aggregate_negative_values_and_null_count(self): + tables, _ = analysis.analyze(str(self.source), {'method': 'aggregate', 'source': {'exclude_rows': [6]}, 'by': ['地区'], 'metrics': {'收入': 'sum', '说明': 'count'}}) + result = tables[0][1].set_index('地区') + self.assertEqual(result.loc['华东', '收入'], 80) + self.assertEqual(result.loc['华南', '说明'], 2) + + def test_all_null_sum_not_zero(self): + frame = pd.DataFrame({'类别': ['A','B','B'], '值': [None, 0, None]}) + result = analysis.aggregate(frame, {'by': ['类别'], 'metrics': {'值': 'sum'}}, pivot=False).set_index('类别') + self.assertTrue(pd.isna(result.loc['A','值'])) + self.assertEqual(result.loc['B','值'], 0) + + def test_pivot_multiple_levels(self): + frame = pd.DataFrame({'地区':['A','A','B'], '年':['2025','2026','2025'], '收入':[10,20,30]}) + result = analysis.aggregate(frame, {'by':['地区'],'columns':['年'],'metrics':{'收入':'sum'}}, pivot=True).set_index('地区') + self.assertEqual(result.loc['A','["收入","2026"]'],20) + self.assertTrue(pd.isna(result.loc['B','["收入","2026"]'])) + + def test_rules_report_conflict_and_unknown(self): + detail, summary = analysis.classify(self.dataset(), {'column':'说明','rules':[{'label':'正向','keywords':['好','满意']},{'label':'物流','keywords':['慢']}]}) + self.assertEqual(detail['分类'].tolist(), ['正向','未知','物流','需复核']) + self.assertAlmostEqual(summary['占比'].sum(),1) + + def test_join_rejects_accidental_many_to_many(self): + with self.assertRaises(pd.errors.MergeError): + analysis.transform(self.dataset(), [{'type':'merge','source':{'path':str(self.source),'exclude_rows':[6]},'on':['地区']}], []) + + def test_missing_headers_can_use_coordinates(self): + wb=Workbook(); ws=wb.active + ws.append([None,'值']); ws.append(['001',5]) + path=QA/'no_header.xlsx'; wb.save(path) + with self.assertRaises(ValueError): data.read_dataset(str(path)) + frame, meta = data.read_dataset(str(path), {'columns':{'编号':'A','值':'B'}}) + self.assertEqual(frame.iloc[0]['编号'],'001') + self.assertEqual(meta['columns']['编号'],'A') + + def test_formula_cache_is_required(self): + wb=Workbook(); ws=wb.active; ws.append(['值']); ws.append(['=1+1']) + path=QA/'uncached.xlsx'; wb.save(path) + with self.assertRaisesRegex(ValueError,'公式缓存'): + data.read_dataset(str(path)) + + def test_safe_text_and_no_truncation_in_writer(self): + tables, metadata=analysis.analyze(str(self.csv), {'method':'transform'}) + plan=QA/'safe-text.json'; output=QA/'safe-text.xlsx' + data.save_plan(tables,metadata,str(plan),overwrite=True) + with patch.object(sys,'argv',['apply','--output',str(output),'--spec-file',str(plan),'--overwrite']): + result=writer.main() + self.assertEqual(result['formula_count'],0) + wb=load_workbook(output); ws=wb['分析结果'] + self.assertEqual(ws['B2'].value,'001') + self.assertEqual(ws['C2'].value,'=1+1') + self.assertEqual(ws['C2'].data_type,'s') + self.assertEqual(ws['C3'].value,'长文本'*80) + self.assertTrue(ws['C3'].alignment.wrap_text) + self.assertGreater(ws.row_dimensions[3].height,36) + self.assertEqual(inspector.inspect_excel(output,sheet_name='分析结果',start_row=1,start_column=1,max_rows=10,max_columns=10)['formula_count'],0) + + def test_append_result_preserves_source(self): + spec={'method':'aggregate','source':{'exclude_rows':[6]},'by':['地区'],'metrics':{'收入':'sum'},'chart':{'category':'地区','values':['收入'],'title':'地区收入'}} + tables,meta=analysis.analyze(str(self.source),spec) + plan=QA/'summary.json'; output=QA/'summary.xlsx' + data.save_plan(tables,meta,str(plan),overwrite=True,chart=spec['chart']) + with patch.object(sys,'argv',['apply','--input',str(self.source),'--output',str(output),'--spec-file',str(plan),'--overwrite']): writer.main() + self.assertEqual(hashlib.sha256(self.source.read_bytes()).hexdigest(),self.source_hash) + original=load_workbook(self.source); result=load_workbook(output) + self.assertEqual(list(original['明细'].values),list(result['明细'].values)) + self.assertEqual(copy(original['明细']['A1'].font),copy(result['明细']['A1'].font)) + self.assertEqual(len(result['分析结果']._charts),1) + + def test_writer_rejects_text_that_excel_would_truncate(self): + for value in ['长' * 32768, {'value': '长' * 32768}]: + with self.subTest(explicit=isinstance(value, dict)): + wb = Workbook() + with self.assertRaisesRegex(ValueError, '不能静默截断'): + writer._op_write_rows(wb, {'sheet': wb.active.title, 'rows': [[value]]}) + + def test_cost_direction_once(self): + frame=pd.DataFrame({data.SOURCE_ROW:[2,3],'对象':['便宜','贵'],'质量':[98,98],'价格':[10,30]}) + tables, _ = modeling.evaluate(frame, {'entity':'对象','directions':{'质量':'benefit','价格':'cost'}}) + rank=tables[0][1].set_index('对象') + self.assertEqual(rank.loc['便宜','排名'],1) + self.assertEqual(rank.loc['贵','排名'],2) + + def test_identical_objects_tie(self): + frame=pd.DataFrame({data.SOURCE_ROW:[2,3],'对象':['A','B'],'值':[10,10]}) + tables,_=modeling.evaluate(frame,{'entity':'对象','directions':{'值':'cost'},'weighting':'entropy'}) + self.assertEqual(tables[0][1]['排名'].tolist(),[1,1]) + self.assertEqual(tables[0][1]['得分'].tolist(),[.5,.5]) + + def test_ahp_rejects_inconsistent_matrix(self): + frame=pd.DataFrame({data.SOURCE_ROW:[2,3],'对象':['A','B'],'a':[1,2],'b':[2,1],'c':[2,3]}) + with self.assertRaisesRegex(ValueError,'一致性'): + modeling.evaluate(frame,{'entity':'对象','directions':{'a':'benefit','b':'benefit','c':'benefit'},'weighting':'ahp','comparison_matrix':[[1,9,1/9],[1/9,1,9],[9,1/9,1]]}) + + def test_forecast_by_group_and_time_holdout(self): + dates=pd.date_range('2024-01-01',periods=24,freq='MS') + frame=pd.DataFrame({'城市':['A']*24+['B']*24,'月份':list(dates)*2,'销量':list(np.arange(24)*10+100)+list(np.arange(24)*-2+100)}) + tables,_=modeling.forecast(frame,{'by':['城市'],'date':'月份','value':'销量','horizon':3,'frequency':'MS'}) + result=tables[0][1] + self.assertEqual(len(result),6) + self.assertAlmostEqual(result[result['城市']=='A'].iloc[0]['预测值'],340) + self.assertAlmostEqual(result[result['城市']=='B'].iloc[0]['预测值'],52) + self.assertTrue((tables[1][1]['训练期数']==18).all()) + + def test_forecast_rejects_missing_month(self): + frame=pd.DataFrame({'日期':pd.date_range('2024-01-01',periods=8,freq='MS').delete(3),'值':range(7)}) + with self.assertRaisesRegex(ValueError,'连续'): + modeling.forecast(frame,{'date':'日期','value':'值'}) + + def test_regression_has_holdout_and_baseline(self): + frame=pd.DataFrame({data.SOURCE_ROW:range(2,42),'x':range(40),'y':np.arange(40)*3+7}) + tables,meta=modeling.supervised(frame,{'features':['x'],'target':'y'},classification=False) + metrics=tables[1][1] + error=metrics[(metrics['数据集']=='测试')&(metrics['模型']=='linear')&(metrics['指标']=='RMSE')].iloc[0]['值'] + self.assertLess(error,1e-8) + self.assertEqual(meta['train_rows'],32) + self.assertIn('基线',metrics['模型'].tolist()) + + def test_classification_categorical_pipeline(self): + frame=pd.DataFrame({data.SOURCE_ROW:range(2,42),'x':list(range(20))*2,'组':['A']*20+['B']*20,'标签':['低']*20+['高']*20}) + tables,_=modeling.supervised(frame,{'features':['x','组'],'categorical':['组'],'target':'标签'},classification=True) + self.assertEqual(len(tables[0][1]),8) + self.assertIn('F1_macro',tables[1][1]['指标'].tolist()) + + def test_small_regression_has_two_test_rows_for_r_squared(self): + frame = pd.DataFrame({data.SOURCE_ROW: range(2, 12), 'x': range(10), 'y': np.arange(10) * 3 + 7}) + tables, metadata = modeling.supervised(frame, {'features': ['x'], 'target': 'y', 'test_fraction': .1}, classification=False) + self.assertEqual(metadata['test_rows'], 2) + self.assertTrue(np.isfinite(tables[1][1]['值']).all()) + + def test_clustering_separates_obvious_groups(self): + frame=pd.DataFrame({data.SOURCE_ROW:range(8),'x':[0,.1,.2,.3,10,10.1,10.2,10.3]}) + tables,_=modeling.unsupervised(frame,{'features':['x'],'clusters':2},anomaly=False) + labels=tables[0][1]['簇编号'].to_numpy() + self.assertTrue(np.all(labels[:4]==labels[0])) + self.assertNotEqual(labels[0],labels[-1]) + + def test_integer_optimization(self): + tables,meta=modeling.optimize({'variables':['x','y'],'objective':[3,2],'sense':'max','integer':[True,True],'constraints':[{'coefficients':[2,1],'relation':'<=','rhs':4}]}) + self.assertEqual(meta['objective_value'],8) + self.assertEqual(tables[0][1]['取值'].tolist(),[0,4]) + + def test_infeasible_and_unbounded_rejected(self): + with self.assertRaisesRegex(ValueError,'求解未成功'): + modeling.optimize({'variables':['x'],'objective':[1],'constraints':[{'coefficients':[1],'relation':'<=','rhs':-1}]}) + with self.assertRaisesRegex(ValueError,'求解未成功'): + modeling.optimize({'variables':['x'],'objective':[1],'sense':'max'}) + + def test_convex_quadratic(self): + tables,meta=modeling.optimize({'variables':['x'],'objective':[-4],'quadratic':[[2]]}) + self.assertAlmostEqual(tables[0][1].iloc[0]['取值'],2) + self.assertAlmostEqual(meta['objective_value'],-4) + + def test_output_root_enforced(self): + with self.assertRaises(ValueError): + data.save_plan([('结果',pd.DataFrame({'x':[1]}))],{},'/private/tmp/outside-plan.json') + + +if __name__ == '__main__': + unittest.main(verbosity=2)