diff --git a/data/links/all_file_attachments.json b/data/links/all_file_attachments.json new file mode 100644 index 0000000..87a6dcc --- /dev/null +++ b/data/links/all_file_attachments.json @@ -0,0 +1,149 @@ +[ + { + "name": "游戏行业核心观点梳理.pdf", + "fileId": "b9Y4gmKWrPNm30nEFjoRXP4aJGXn6lpz", + "sender": "李志健", + "time": "2026-05-15 10:28:47", + "group": "dc" + }, + { + "name": "AI视频创作作品分享罗剑豪V3.pptx", + "fileId": "yQod3RxJKGDndA07i4noekLyJkb4Mw9r", + "sender": "李志健", + "time": "2026-05-13 14:36:23", + "group": "dc" + }, + { + "name": "坦克大逃杀答辩材料:从代码搬运工到架构师助理.pptx", + "fileId": "QBnd5ExVEvEp5KZ2I2pgG5ElJyeZqMmz", + "sender": "李志健", + "time": "2026-05-13 14:36:02", + "group": "dc" + }, + { + "name": "《仙途乐逍遥》AI歌曲分享文档.xmind", + "fileId": "R4GpnMqJzGmy3dxEiaZkMN5R8Ke0xjE3", + "sender": "李志健", + "time": "2026-05-13 14:35:35", + "group": "dc" + }, + { + "name": "刘亚彬-万剑囚天镇压獓因-分享文档02.pptx", + "fileId": "DnRL6jAJMGMRdeY0iXRqqAa6WyMoPYe1", + "sender": "李志健", + "time": "2026-05-13 14:34:08", + "group": "dc" + }, + { + "name": "寻仇AI视频制作分享PPT (2).pptx", + "fileId": "0eMKjyp813zlbnexs4j4XpbgVxAZB1Gv", + "sender": "李志健", + "time": "2026-05-13 14:21:13", + "group": "dc" + }, + { + "name": "模型报告.html", + "fileId": "MyQA2dXW7eRdPjKxS5AA5P41JzlwrZgb", + "sender": "李志健", + "time": "2026-05-11 01:28:58", + "group": "dc" + }, + { + "name": "game-analyst-agent.zip", + "fileId": "mweZ92PV6M7e1qQxSKmpwOq0WxEKBD6p", + "sender": "黄静雯", + "time": "2026-05-22 09:02:50", + "group": "dc" + }, + { + "name": "game-analyst-agent-manual.html", + "fileId": "QPGYqjpJYr7lQGAXFZRQ6lnM8akx1Z5N", + "sender": "黄静雯", + "time": "2026-05-22 09:02:50", + "group": "dc" + }, + { + "name": "2026-05-yunhu-project-selection-strategy-report.html", + "fileId": "4lgGw3P8vR203ZrEHpG2gyYE85daZ90D", + "sender": "黄静雯", + "time": "2026-05-21 00:35:28", + "group": "dc" + }, + { + "name": "取败之道——关于幸存者偏差、筹码管理与生长逻辑的深度反思.md", + "fileId": "0eMKjyp813zlbnexs4d9bqrZVxAZB1Gv", + "sender": "李志健", + "time": "2026-05-20 10:22:27", + "group": "dc" + }, + { + "name": "花园世界产品策略_GOS补充信息.md", + "fileId": "NkDwLng8ZLRBbjGpCxxqXz1MVKMEvZBY", + "sender": "傅明游", + "time": "2026-05-26 16:00:45", + "group": "dc" + }, + { + "name": "2026-05-25_report_my_garden_world_us_project.html", + "fileId": "wva2dxOW4YmAb01xF0BnvLNEVbkz3BRL", + "sender": "黄静雯", + "time": "2026-05-26 09:29:12", + "group": "dc" + }, + { + "name": "主报告-我的花园世界.html", + "fileId": "y20BglGWO2NbdYyKt0peOkjl8A7depqY", + "sender": "李志健", + "time": "2026-06-01 23:36:40", + "group": "dc" + }, + { + "name": "进一步的思考.md", + "fileId": "QPGYqjpJYr7lQGAXFZ0njADy8akx1Z5N", + "sender": "李志健", + "time": "2026-06-01 23:09:45", + "group": "dc" + }, + { + "name": "花园项目对策划的启示.md", + "fileId": "7QG4Yx2JpLMZmdaECglLlO2aJ9dEq3XD", + "sender": "李志健", + "time": "2026-06-01 22:55:36", + "group": "dc" + }, + { + "name": "花园世界等相关产品25.10.28.xlsx", + "fileId": "QPGYqjpJYr7lQGAXFZ0beYjZ8akx1Z5N", + "sender": "李志健", + "time": "2026-06-01 21:42:33", + "group": "dc" + }, + { + "name": "2026-06-03_garden-social-management-us-market-assessment.html", + "fileId": "QBnd5ExVEvEp5KZ2I24Ozg7EJyeZqMmz", + "sender": "黄静雯", + "time": "2026-06-03 09:32:58", + "group": "dc" + }, + { + "name": "2026-06-04_social-platform-dependency-overseas-expansion-v2.html", + "fileId": "1OQX0akWmxrpR12EhveqkY448GlDd3mE", + "sender": "黄静雯", + "time": "2026-06-05 09:29:32", + "group": "dc" + }, + { + "name": "数据口径入门教程.md", + "fileId": "QPGYqjpJYr7lQGAXFZqd3GMM8akx1Z5N", + "sender": "李志健", + "time": "2026-06-05 21:11:44", + "group": "dc" + }, + { + "name": "工作流总结-王雨默.html", + "fileId": "1zknDm0WRamRxP60IxnN2Dwk8BQEx5rG", + "sender": "王雨默", + "time": "2026-05-22 10:57:31", + "group": "cx" + } +] \ No newline at end of file diff --git a/data/links/all_kimi_links.json b/data/links/all_kimi_links.json new file mode 100644 index 0000000..7e4c2ae --- /dev/null +++ b/data/links/all_kimi_links.json @@ -0,0 +1,62 @@ +[ + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/", + "desc": "[分享] 我的花园世界 - 资源经济深度研究", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/", + "desc": "[分享] 我的花园世界 - 资源经济深度研究", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/", + "desc": "[分享] 我的花园世界 - 资源经济深度研究", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://kblb6hdbtbfto.ok.kimi.link/", + "desc": "[分享] 《我的花园世界》商业化梳理报告——基于游戏机制以及社区舆论", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://kblb6hdbtbfto.ok.kimi.link/", + "desc": "[分享] 《我的花园世界》商业化梳理报告——基于游戏机制以及社区舆论", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://kblb6hdbtbfto.ok.kimi.link/", + "desc": "[分享] 《我的花园世界》商业化梳理报告——基于游戏机制以及社区舆论", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/#/long-term-study", + "desc": "", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/#/activity-study", + "desc": "", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/#/player-autonomy", + "desc": "", + "sender": "陈楚真", + "group": "dc" + }, + { + "url": "https://upy4l3m3qnegc.ok.kimi.link/#/healing-to-game", + "desc": "", + "sender": "陈楚真", + "group": "dc" + } +] \ No newline at end of file diff --git a/docs/how-to-replicate.md b/docs/how-to-replicate.md new file mode 100644 index 0000000..a26c801 --- /dev/null +++ b/docs/how-to-replicate.md @@ -0,0 +1,469 @@ +# 从钉钉群聊到知识库:飞书文档自动采集与结构化系统 - 复刻指南 + +> 作者:大师 | 日期:2026-06-03 + +--- + +## 一句话说清楚这个项目做了什么 + +**从钉钉群聊消息中自动抓取飞书文档链接和文件附件,下载内容,构建知识图谱和 Obsidian 知识库。** + +整个流程 = 3个核心工具 + 6个Python脚本,不需要自己写任何API对接代码。 + +--- + +## 整体架构 + +``` +钉钉群聊消息 + | + v +(1) dws CLI(悟空)-- 拉取钉钉消息、下载文件附件 + | + v +(2) Python 脚本 -- 正则提取飞书链接 + 文件ID + | + v +(3) lark-cli -- 读取飞书文档内容(Block API) + | + v +(4) Python 脚本 -- 构建知识图谱(JSON)、Obsidian 库、汇总报告 +``` + +**关键认知:你不需要申请飞书开放平台的App。** 消息来源是钉钉(通过dws),飞书文档内容获取通过lark-cli(浏览器授权登录即可)。 + +--- + +## 核心工具获取指南 + +### 工具 1:dws CLI -- 钉钉消息采集(随悟空自动安装) + +**是什么**:dws CLI 是钉钉「悟空」(Wukong) 桌面客户端自带的命令行工具,封装了钉钉 25+ 项 MCP 服务(群聊消息、文件管理、日历、通讯录、审批等),无需自己对接钉钉开放平台API。 + +**获取方式(三种,选一种即可)**: + +| 方式 | 适合谁 | 操作 | +|------|--------|------| +| **GitHub 直接下载**(推荐) | 所有人 | 从 GitHub Releases 下载对应平台的压缩包,解压即用 | +| 安装悟空客户端 | 钉钉重度用户 | 装完悟空,dws 自动在 `C:\Program Files\Wukong\<版本>\bin\dws.exe` | +| 从同事那里复制 | 最省事 | 复制一个 `dws.exe` 文件(约 5-14MB) | + +**GitHub 仓库**:https://github.com/DingTalk-Real-AI/dingtalk-workspace-cli + +这是钉钉官方开源的 CLI 工具,2000+ stars,持续更新中(最新 v1.0.33,2026-06-02 发布)。 + +```bash +# 从 GitHub Releases 直接下载(以 Windows 为例): +# https://github.com/DingTalk-Real-AI/dingtalk-workspace-cli/releases/latest +# 下载 dws-windows-amd64.zip(约 5.3MB),解压得到 dws.exe + +# macOS / Linux 也有对应版本: +# dws-darwin-amd64.tar.gz (macOS Intel) +# dws-darwin-arm64.tar.gz (macOS Apple Silicon) +# dws-linux-amd64.tar.gz (Linux x64) + +```bash +# 悟空是钉钉的AI桌面客户端(也叫钉钉Real版),安装后自动附带: +# - dws CLI(钉钉MCP服务命令行) +# - wukong-cli(悟空Agent命令行) +# - Node.js、Python、ffmpeg 等运行时 +# +# 安装路径参考: +# 主程序:C:\Program Files\Wukong\<版本号>\DingTalkReal.exe +# dws CLI:C:\Program Files\Wukong\<版本号>\bin\dws.exe +# 运行时缓存:C:\Users\<用户名>\.real\.bin\dws\bin\dws.exe +``` + +**版本信息**(本项目实际使用): + +| 项目 | 值 | +|------|-----| +| 悟空版本 | 0.9.51 | +| dws CLI 版本 | 0.2.75(悟空内置)/ 1.0.33(GitHub 最新) | +| 架构 | MCP Dynamic Aggregation | +| Go 版本 | 1.24+ | + +**认证**:首次使用需要登录认证(钉钉扫码或账号登录),Corp ID 和 User ID 会自动配置。运行 `dws auth status` 可查看登录状态。 + +**能把 dws 单独提取出来用吗?可以。** + +dws.exe 是 Go 语言静态编译的二进制文件(14MB),不依赖任何外部 DLL,可以脱离悟空独立运行。提取方法: + +```bash +# 只需要一个文件:dws.exe(14MB) +# 路径:C:\Program Files\Wukong\<版本号>\bin\dws.exe +# 或:C:\Users\<用户名>\.real\.bin\dws\bin\dws.exe + +# 1. 复制 dws.exe 到任意目录 +copy "C:\Users\admin\.real\.bin\dws\bin\dws.exe" D:\tools\dws.exe + +# 2. 首次运行,触发认证(会自动创建 .dws 数据目录) +D:\tools\dws.exe auth status +# 如果未登录,会提示 OAuth 扫码登录(用钉钉App扫码) + +# 3. 认证成功后,目录下会自动生成 .dws/ 子目录: +# .dws/identity.json -- 身份标识 +# .dws/token.json -- Token(自动续期) +# .dws/.data -- 加密凭证 +# .dws/logs/ -- 日志 + +# 4. 验证可用 +D:\tools\dws.exe chat message list --help +D:\tools\dws.exe drive download --help +``` + +**提取后的体积**: + +| 文件 | 大小 | +|------|------| +| `dws.exe` | 14 MB | +| `.dws/` 数据目录 | < 1 MB | +| **总计** | **约 14 MB** | + +**认证机制详解(dws 自带,不需要悟空)**: + +dws 内置了完整的 OAuth 认证流程,提取出来后独立就能完成登录: + +``` +运行 dws auth status(未登录状态) + -> dws 启动 Device Flow OAuth + -> 连接 login.dingtalk.com/oauth2/auth + -> 终端显示二维码 / URL + -> 用钉钉App扫码授权 + -> dws 轮询 api.dingtalk.com 获取 Token + -> Token 加密存储到 .dws/.data + -> 完成 +``` + +dws 内置了默认的 OAuth ClientID/ClientSecret,普通用户直接用就行,不需要自己申请。如果公司有自建应用,也可以通过环境变量覆盖: + +```bash +# 可选:使用自建应用的凭证(一般不需要) +export DWS_CLIENT_ID=<你的AppKey> +export DWS_CLIENT_SECRET=<你的AppSecret> +``` + +**注意事项**: +- Token 会自动续期,但长时间不用会过期,重新运行 `dws auth status` 触发重新扫码即可 +- 每个人需要用自己的钉钉账号认证,不能共用 Token +- 提取出来的 dws 功能完整,支持全部 25+ 项 MCP 服务 +- `.dws/.data` 是加密存储的凭证文件(511字节),不要分享给别人 + +**核心能力**: + +| 命令 | 用途 | 示例 | +|------|------|------| +| `dws chat message list` | 拉取群聊消息 | `dws chat message list --group <群ID> --time "2026-05-01" --format json --limit 200` | +| `dws drive download` | 下载文件附件 | `dws drive download --node --output ./files/` | + +**获取群组ID**:需要知道目标群的 `openConversationId`,可以通过以下方式获取: +- 在钉钉管理后台查看 +- 或者先用 `dws chat group list` 命令列出你所在的群 + +**分页策略**:API 每次最多返回约200条消息,超过的话需要分段拉取 + 用 `openMessageId` 去重: + +```python +# 分段拉取示例 +segments = [ + ("2026-05-01 00:00:00", "true"), # 从5月1日向前 + ("2026-05-19 00:00:00", "true"), # 从5月19日向前 + ("2026-05-24 00:00:00", "true"), # 从5月24日向前 + ("2026-06-03 00:00:00", "false"), # 从6月3日向后 +] +# forward="true" 表示从该时间点向前(更新的消息) +# forward="false" 表示从该时间点向后(更旧的消息) +``` + +--- + +### 工具 2:lark-cli -- 飞书文档内容读取 + +**是什么**:飞书官方提供的命令行工具,通过浏览器 OAuth 授权后,可以直接调用飞书 Open API 读取文档内容。 + +**安装**: + +```bash +# 需要 Node.js 18+ +npm install -g @larksuite/cli + +# 验证安装 +lark-cli --version +``` + +**认证配置**(关键步骤): + +```bash +# 1. 初始化配置 +lark-cli config init + +# 2. 浏览器授权登录(会自动打开浏览器) +lark-cli auth login --domain docs,drive,wiki --recommend + +# 3. 验证授权状态 +lark-cli auth status +``` + +**为什么能拿到飞书文档内容?** + +这是大家最关心的问题,答案是: + +1. **lark-cli 使用浏览器 OAuth 授权**:你在浏览器里登录自己的飞书账号,lark-cli 拿到你的访问令牌(token)。 +2. **用你的身份调飞书 Open API**:后续所有请求都是以你个人身份发出的,跟你在浏览器里打开文档一样。 +3. **你有权限看的文档,lark-cli 就能读**:不是破解,不是爬虫,就是正常的API调用。 + +``` +你(浏览器登录飞书)-> OAuth Token -> lark-cli -> 飞书 Open API -> 文档内容 +``` + +**核心命令**: + +```bash +# 获取文档内容(Block API,返回JSON格式) +lark-cli api GET /open-apis/docx/v1/documents//blocks --format json + +# doc_id 从飞书链接中提取: +# 例:https://dianchukeji.feishu.cn/docx/AbLed72FgoPn5hxOYN3cRpffn0E +# ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ 这就是 doc_id +``` + +**注意事项**: + +- Token 有过期时间,长时间不用需要重新授权 +- 你只能读取你有权限的飞书文档(跟在浏览器里访问一样) +- wiki 类型的链接需要先解析真实的 doc_id(wiki 链接里的ID是节点ID,不是文档ID) + +--- + +### 工具 3:Python 脚本 -- 数据处理管线 + +**依赖安装**: + +```bash +pip install beautifulsoup4 requests +``` + +**7个脚本的职责**: + +| 脚本 | 输入 | 输出 | 干什么 | +|------|------|------|--------| +| `fetch_all.py` | dws CLI | `data/raw-messages/all_messages_combined.json` | 拉取两个群的全部消息 | +| `extract_links.py` | 上一步的JSON | 终端输出(统计信息) | 正则提取飞书链接、文件附件、Kimi链接 | +| `build_graph.py` | 消息JSON | `output/knowledge-graph/knowledge_graph.json` | 构建知识图谱(人物、文档、主题、关系) | +| `write_obsidian.py` | 硬编码内容 | `output/obsidian-vault/` 目录 | 生成 Obsidian 知识库(含双向链接) | +| `write_report.py` | 硬编码内容 | `output/reports/花园世界全量汇总.md` | 生成全量汇总报告 | +| `gen_visual.py` | knowledge_graph.json | `output/knowledge-graph/` HTML+Mermaid | 生成可视化知识图谱 | +| `daily_feishu_collector.py` | dws CLI | `feishu_links_YYYYMMDD.json` | 每日定时采集(增量) | + +--- + +## 钉钉消息为什么能拿到? + +**原理**:通过 dws CLI(悟空工具)直接调用钉钉内部API。 + +``` +dws CLI -> 钉钉内部 API -> 群聊消息(含发送者、时间、内容、文件ID) +``` + +- dws CLI 是公司内部工具,已经封装好了钉钉API的认证和调用 +- 你只需要提供群组的 `openConversationId` 和时间范围 +- 返回的消息内容是 JSON 格式,包含消息文本、发送者、时间戳等 +- 文件附件可以通过 `fileId` 用 `dws drive download` 下载 + +**不需要**:申请钉钉开放平台应用、配置回调URL、处理webhook。 + +--- + +## 飞书文档为什么能拿到? + +**原理**:lark-cli 使用你的飞书账号 OAuth 授权,以你的身份调用飞书 Open API。 + +``` +你登录飞书 -> OAuth Token -> lark-cli -> /open-apis/docx/v1/documents/{id}/blocks -> 文档内容JSON +``` + +**三个前提条件**: + +1. **你有飞书账号**:公司飞书租户下的账号 +2. **你有文档访问权限**:文档对你可见(在群里分享过的文档,群成员通常都有权限) +3. **lark-cli 授权成功**:`lark-cli auth login` 一次即可,后续自动使用缓存的token + +**能读到什么**: + +- `docx` 类型文档:直接通过 doc_id 调 Block API 获取全部内容 +- `wiki` 类型文档:需要先通过 wiki API 解析节点ID得到真实 doc_id,再调 Block API +- 文件附件(XLSX/PPT/PDF等):通过 `dws drive download` 从钉钉侧下载 + +**读不到什么**: + +- 你没有权限的文档(跟浏览器一样,没权限就是没权限) +- 已被删除的文档 + +--- + +## 复刻步骤(从零开始) + +### 第一步:确认工具就绪 + +```bash +# 检查 dws CLI +dws --version +# 如果没有,找IT获取 + +# 检查 Node.js +node --version # 需要 18+ + +# 检查 Python +python --version # 需要 3.10+ + +# 安装 lark-cli +npm install -g @larksuite/cli + +# 安装 Python 依赖 +pip install beautifulsoup4 requests +``` + +### 第二步:授权 lark-cli + +```bash +lark-cli config init +lark-cli auth login --domain docs,drive,wiki --recommend +# 浏览器会自动打开,登录你的飞书账号即可 +``` + +### 第三步:获取群组ID + +你需要知道要采集的钉钉群的 `openConversationId`。获取方式: + +1. 在钉钉管理后台查看 +2. 或者用 dws 命令列出你所在的群 +3. 或者找之前已经获取过的同事要 + +### 第四步:拉取消息 + +修改 `scripts/fetch_all.py` 中的群组ID和时间范围,然后运行: + +```bash +python scripts/fetch_all.py +``` + +这会生成 `data/raw-messages/all_messages_combined.json`,包含两个群的全部消息。 + +### 第五步:提取链接 + +```bash +python scripts/extract_links.py +``` + +输出统计信息:飞书链接数、文件附件数、Kimi链接数等。 + +### 第六步:构建知识图谱 + +```bash +python scripts/build_graph.py +``` + +生成 `output/knowledge-graph/knowledge_graph.json`,包含人物、文档、主题、关系的结构化数据。 + +### 第七步:生成可视化和知识库 + +```bash +# 知识图谱可视化 +python scripts/gen_visual.py +# 输出:output/knowledge-graph/knowledge_graph.html(浏览器打开即可查看) + +# Obsidian 知识库 +python scripts/write_obsidian.py +# 输出:output/obsidian-vault/(用 Obsidian 打开此目录) + +# 汇总报告 +python scripts/write_report.py +``` + +### 第八步:设置定时采集(可选) + +```powershell +# Windows 定时任务,每天18:00执行 +schtasks /create /tn "FeishuDocCollector" /tr "python D:\path\to\scripts\daily_feishu_collector.py" /sc daily /st 18:00 +``` + +--- + +## 常见问题 + +### Q: 需要申请飞书开放平台的App吗? + +**不需要。** lark-cli 用的是浏览器 OAuth 授权(你的个人身份),不需要创建企业自建应用。 + +### Q: 需要申请钉钉开放平台的权限吗? + +**不需要。** dws CLI 是内部工具,已经封装好了认证。 + +### Q: 飞书文档有4种域名,都能读吗? + +都能读,只要你有权限。不同域名对应不同的飞书空间/租户,lark-cli 用你的账号登录后可以跨空间访问。 + +本项目涉及的4个飞书域: + +| 域名 | 说明 | +|------|------| +| `dianchukeji.feishu.cn` | 公司飞书 | +| `fcnlycv6dd0w.feishu.cn` | 研究组飞书空间 | +| `ocnmca6f1o0p.feishu.cn` | 另一飞书空间 | +| `my.feishu.cn` | 个人飞书 | + +### Q: 消息太多拉不完怎么办? + +分段拉取 + `openMessageId` 去重。见 `fetch_all.py` 中的分段策略。 + +### Q: Token 过期了怎么办? + +重新运行 `lark-cli auth login --domain docs,drive,wiki --recommend`,浏览器重新授权即可。 + +### Q: wiki 链接和 docx 链接有什么区别? + +docx 链接的ID就是文档ID,可以直接调API。wiki 链接的ID是知识库节点ID,需要先通过 wiki API 解析出真实的文档ID。 + +--- + +## 技术栈总结 + +| 层级 | 工具 | 获取方式 | 费用 | +|------|------|----------|------| +| 钉钉消息采集 | dws CLI | 悟空(Wukong)自带,装悟空就有 | 免费 | +| 钉钉文件下载 | dws drive | 同上,dws 的子命令 | 免费 | +| 飞书文档读取 | lark-cli | `npm install -g @larksuite/cli` | 免费 | +| 数据处理 | Python + BeautifulSoup | 悟空自带Python,BS4需 `pip install` | 免费 | +| 知识库管理 | Obsidian | https://obsidian.md 下载 | 免费 | + +**总成本:0元,只需要你有飞书账号和钉钉群访问权限。** + +--- + +## 产出物展示 + +| 产出 | 路径 | 用途 | +|------|------|------| +| 知识图谱(交互式) | `output/knowledge-graph/knowledge_graph.html` | 浏览器打开,查看人物、文档、主题关系 | +| Obsidian 知识库 | `output/obsidian-vault/` | 用 Obsidian 打开,双向链接浏览 | +| 汇总报告 | `output/reports/花园世界全量汇总.md` | 一份完整的 Markdown 报告 | +| 原始文档 | `output/feishu-docs/` | 155篇飞书文档的本地备份 | +| 下载文件 | `dc_files/` | PPT/PDF/XLSX 等附件 | + +--- + +## 项目数据一览 + +| 指标 | 数量 | +|------|------| +| 采集群组 | 2个(dc战略问题研究院 + 创新组) | +| 消息总数 | 377条 | +| 飞书文档 | 155篇 | +| 文件附件 | 19个 | +| 活跃人员 | 28位 | +| 时间跨度 | 2026-05-01 ~ 2026-06-03 | +| 知识主题 | 7个核心主题 | + + + + + diff --git a/output/reports/latest_summary.md b/output/reports/latest_summary.md new file mode 100644 index 0000000..6a19b73 --- /dev/null +++ b/output/reports/latest_summary.md @@ -0,0 +1,20 @@ +# 钉钉群聊飞书文档采集报告 +> 生成: 2026-06-06 11:12 + +## 数据总览 + +| 指标 | 数量 | +|------|------| +| 消息 | 7 | +| 文档 | 139 | +| 附件 | 21 | + +### dc +- 消息: 5 (2026-06-06 00:11:33 ~ 2026-06-06 09:36:25) +- 文档: 19 +- 附件: 20 + +### cx +- 消息: 2 (2026-06-06 00:35:41 ~ 2026-06-06 01:48:34) +- 文档: 7 +- 附件: 1 diff --git a/output/reports/summary_20260606.md b/output/reports/summary_20260606.md new file mode 100644 index 0000000..6a19b73 --- /dev/null +++ b/output/reports/summary_20260606.md @@ -0,0 +1,20 @@ +# 钉钉群聊飞书文档采集报告 +> 生成: 2026-06-06 11:12 + +## 数据总览 + +| 指标 | 数量 | +|------|------| +| 消息 | 7 | +| 文档 | 139 | +| 附件 | 21 | + +### dc +- 消息: 5 (2026-06-06 00:11:33 ~ 2026-06-06 09:36:25) +- 文档: 19 +- 附件: 20 + +### cx +- 消息: 2 (2026-06-06 00:35:41 ~ 2026-06-06 01:48:34) +- 文档: 7 +- 附件: 1 diff --git a/skills/dingtalk-feishu-collector/SKILL.md b/skills/dingtalk-feishu-collector/SKILL.md new file mode 100644 index 0000000..2c95f78 --- /dev/null +++ b/skills/dingtalk-feishu-collector/SKILL.md @@ -0,0 +1,180 @@ +# 钉钉群聊飞书文档采集与知识整理 SOP + +## 触发条件 + +当用户提到以下关键词时使用此技能: +- "收集钉钉消息"、"拉取群聊"、"采集飞书文档" +- "整理群里的链接"、"汇总飞书报告" +- "每日收集"、"定时采集" +- 涉及钉钉群聊 + 飞书文档的工作流 + +## 一句话概述 + +从钉钉群聊中自动采集飞书文档链接和文件附件,下载内容,生成结构化总结。 + +## 前置条件 + +| 工具 | 用途 | 获取方式 | +|------|------|----------| +| dws CLI | 钉钉消息采集 + 文件下载 | 悟空(Wukong)自带,或从 GitHub 下载 | +| lark-cli | 飞书文档内容读取 | `npm install -g @larksuite/cli` | +| Python 3.10+ | 数据处理 | 系统自带或悟空内置 | + +**认证要求**: +- dws CLI:运行 `dws auth status` 确认已登录(钉钉扫码) +- lark-cli:运行 `lark-cli auth status` 确认已授权(飞书浏览器OAuth) + +## 完整工作流(6步) + +### Step 1: 拉取钉钉群消息 + +```bash +python scripts/step1_collect_messages.py # 全量 +python scripts/step1_collect_messages.py --days 1 # 增量 +``` + +**输出**:`data/raw-messages/all_messages_combined.json` + +### Step 2: 提取链接与附件 + +```bash +python scripts/step2_extract_links.py # 全量 +python scripts/step2_extract_links.py --incremental # 增量 +``` + +正则提取飞书链接、文件附件、Kimi链接。 + +**输出**:`data/links/all_feishu_links.json` + `data/links/all_file_attachments.json` + +### Step 3: 拉取飞书文档内容 + +```bash +python scripts/step3_fetch_feishu_docs.py # 全量 +python scripts/step3_fetch_feishu_docs.py --incremental # 增量 +``` + +wiki/docx 自动识别,Block API 解析为 Markdown。 + +**输出**:`output/feishu-docs/.md` + `data/links/all_feishu_content.json` + +### Step 4: 下载文件附件 + +```bash +python scripts/step4_download_files.py # 全量 +python scripts/step4_download_files.py --incremental # 增量 +``` + +使用 `dws drive download` 下载 HTML/MD/XLSX/PPTX/PDF 等附件。 + +**输出**:`output/downloaded-files/html-md/` 和 `output/downloaded-files/other/` + +### Step 5: 生成结构化总结 + +```bash +python scripts/step5_generate_summary.py # 全量 +python scripts/step5_generate_summary.py --since today # 今日 +python scripts/step5_generate_summary.py --brief # 简要 +``` + +按群组/人物/主题聚类,生成增量报告。 + +**输出**:`output/reports/latest_summary.md` + +### Step 6: 更新知识库(可选) + +```bash +python scripts/step6_update_knowledge_base.py +``` + +同步到 Obsidian 知识库 + 知识图谱。 + +## 配置文件 + +所有可定制项在 `config.yaml`: + +```yaml +groups: + dc战略问题研究院: "cidoUneRB4Db8TAXaTrKxkQAw==" + 创新组: "cidMuM+itt5PeY7xNSWsv3M0g==" + +collection: + start_date: "2026-05-01 00:00:00" + days_back: 3 + limit: 200 + +dws_path: auto # auto | /path/to/dws.exe + +output_dir: "./output" +data_dir: "./data" +``` + +## Agent 调用约定 + +### 增量模式(日常) +```bash +python scripts/step1_collect_messages.py --days 1 +python scripts/step2_extract_links.py --incremental +python scripts/step3_fetch_feishu_docs.py --incremental +python scripts/step4_download_files.py --incremental +python scripts/step5_generate_summary.py --since today +``` + +### 全量模式(首次/重建) +```bash +python scripts/step1_collect_messages.py --full +python scripts/step2_extract_links.py +python scripts/step3_fetch_feishu_docs.py +python scripts/step4_download_files.py +python scripts/step5_generate_summary.py --full +``` + +### 快速模式(只看今天) +```bash +python scripts/step1_collect_messages.py --days 1 +python scripts/step2_extract_links.py --incremental +python scripts/step5_generate_summary.py --since today --brief +``` + +## 输出产物 + +| 目录 | 内容 | 格式 | +|------|------|------| +| `data/raw-messages/` | 钉钉原始消息 | JSON | +| `data/links/` | 链接索引 | JSON | +| `output/feishu-docs/` | 飞书文档 | .md | +| `output/downloaded-files/` | 文件附件 | 原始格式 | +| `output/reports/` | 汇总报告 | .md | +| `output/obsidian-vault/` | Obsidian 知识库 | .md | +| `output/knowledge-graph/` | 知识图谱 | JSON + HTML | + +## 常见问题 + +**dws 未登录**:`dws auth login` 扫码。悟空内置路径 `C:\Users\\.real\.bin\dws\bin\dws.exe` + +**lark-cli 过期**:`lark-cli auth login --domain docs,drive,wiki --recommend` + +**wiki 链接解析**:脚本自动处理 wiki→docx 的节点ID解析 + +**消息分页**:脚本内置 openMessageId 去重 + 分段拉取 + +## 定时任务 + +```powershell +# Windows 每天 18:00 增量采集 +schtasks /create /tn "DingTalkFeishuCollector" /tr "python \scripts\step1_collect_messages.py --days 1" /sc daily /st 18:00 +``` + +## 技术栈 + +``` +钉钉群聊 → dws CLI (Go) → JSON消息 + ↓ + Python 正则提取 → 链接索引 + ↓ + lark-cli (Node) → 飞书文档内容 + ↓ + Python 处理 → Markdown总结 + Obsidian库 + 知识图谱 +``` + +**总依赖**:dws CLI (14MB) + lark-cli (npm) + Python 3.10+ + beautifulsoup4 +**总成本**:0 元(需要飞书账号 + 钉钉群访问权限) diff --git a/skills/dingtalk-feishu-collector/config.yaml b/skills/dingtalk-feishu-collector/config.yaml new file mode 100644 index 0000000..b05531f --- /dev/null +++ b/skills/dingtalk-feishu-collector/config.yaml @@ -0,0 +1,37 @@ +# 钉钉飞书采集器配置 +# 修改此文件即可适配其他群组/项目 + +groups: + dc战略问题研究院: "cidoUneRB4Db8TAXaTrKxkQAw==" + 创新组: "cidMuM+itt5PeY7xNSWsv3M0g==" + +collection: + # 首次全量拉取的起始时间 + start_date: "2026-05-01 00:00:00" + # 增量模式:拉取最近N天 + days_back: 3 + # 每次API返回上限 + limit: 200 + # 分段拉取的时间节点(用于全量拉取时的分页) + segments: + - "2026-05-01 00:00:00" + - "2026-05-19 00:00:00" + - "2026-05-24 00:00:00" + +# dws CLI 路径查找优先级 +# auto: 环境变量 DWS_PATH > tools/dws.exe > 悟空内置 > PATH +dws_path: auto + +# lark-cli 路径(默认从 PATH 查找) +lark_cli_path: auto + +# 输出目录(相对于项目根目录) +output_dir: "./output" +data_dir: "./data" + +# 飞书域名映射(可选,用于识别不同租户) +feishu_domains: + - "dianchukeji.feishu.cn" + - "fcnlycv6dd0w.feishu.cn" + - "ocnmca6f1o0p.feishu.cn" + - "my.feishu.cn" diff --git a/skills/dingtalk-feishu-collector/scripts/paths.py b/skills/dingtalk-feishu-collector/scripts/paths.py new file mode 100644 index 0000000..5dfca80 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/paths.py @@ -0,0 +1,75 @@ +"""共享路径和配置工具""" + +import os +import sys + +def find_project_root(): + """向上查找项目根目录(包含 data/ 目录的最顶层)""" + # 从当前脚本位置开始 + script_dir = os.path.dirname(os.path.abspath(__file__)) + + # 向上查找,找到包含 data/ 目录的目录 + d = script_dir + for _ in range(5): # 最多向上5级 + if os.path.isdir(os.path.join(d, "data")): + return d + parent = os.path.dirname(d) + if parent == d: + break + d = parent + + # 如果找不到,假设项目根在 scripts/ 的上一级 + return os.path.dirname(os.path.dirname(script_dir)) + +def find_dws(): + """查找 dws CLI""" + import shutil + + # 1. 环境变量 + env = os.environ.get("DWS_PATH") + if env and os.path.isfile(env): + return env + + # 2. 项目 tools/ 目录 + root = find_project_root() + local = os.path.join(root, "tools", "dws.exe") + if os.path.isfile(local): + return local + + # 3. 悟空内置 + candidates = [ + os.path.expanduser(r"~\.real\.bin\dws\bin\dws.exe"), + r"C:\Program Files\Wukong\0.9.51-26052503\bin\dws.exe", + ] + for p in candidates: + if os.path.isfile(p): + return p + + # 4. PATH + found = shutil.which("dws") + if found: + return found + + print("ERROR: 找不到 dws CLI,请设置 DWS_PATH 环境变量或安装悟空", file=sys.stderr) + sys.exit(1) + +def find_lark_cli(): + """查找 lark-cli""" + import shutil + found = shutil.which("lark-cli") + if found: + return found + print("ERROR: 找不到 lark-cli,请运行 npm install -g @larksuite/cli", file=sys.stderr) + sys.exit(1) + +# 常用路径 +PROJECT_ROOT = find_project_root() +DATA_DIR = os.path.join(PROJECT_ROOT, "data") +RAW_DIR = os.path.join(DATA_DIR, "raw-messages") +LINKS_DIR = os.path.join(DATA_DIR, "links") +OUTPUT_DIR = os.path.join(PROJECT_ROOT, "output") +DOCS_DIR = os.path.join(OUTPUT_DIR, "feishu-docs") +REPORTS_DIR = os.path.join(OUTPUT_DIR, "reports") +DOWNLOAD_DIR = os.path.join(OUTPUT_DIR, "downloaded-files") +OBSIDIAN_DIR = os.path.join(OUTPUT_DIR, "obsidian-vault") +KG_DIR = os.path.join(OUTPUT_DIR, "knowledge-graph") diff --git a/skills/dingtalk-feishu-collector/scripts/run_all.py b/skills/dingtalk-feishu-collector/scripts/run_all.py new file mode 100644 index 0000000..6e41b16 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/run_all.py @@ -0,0 +1,60 @@ +"""一键执行完整采集流程 + +用法: + python run_all.py # 增量(最近3天) + python run_all.py --days 1 # 最近1天 + python run_all.py --full # 全量 + python run_all.py --quick # 只看今天 + python run_all.py --skip-download # 跳过文件下载 + python run_all.py --skip-kb # 跳过知识库更新 +""" + +import argparse, subprocess, sys, os, time + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) + +def run(script, args=None): + cmd = [sys.executable, os.path.join(SCRIPT_DIR, script)] + (args or []) + print(f"\n{'='*50}\n {script} {' '.join(args or [])}\n{'='*50}") + t = time.time() + r = subprocess.run(cmd) + print(f" {'OK' if r.returncode==0 else 'WARN'} ({time.time()-t:.1f}s)") + return r.returncode + +def main(): + p = argparse.ArgumentParser() + p.add_argument("--full",action="store_true"); p.add_argument("--days",type=int) + p.add_argument("--since",type=str); p.add_argument("--quick",action="store_true") + p.add_argument("--skip-download",action="store_true"); p.add_argument("--skip-kb",action="store_true") + args = p.parse_args() + t0 = time.time() + + s1 = [] + if args.full: s1.append("--full") + elif args.since: s1 += ["--since", args.since] + elif args.days: s1 += ["--days", str(args.days)] + elif args.quick: s1 += ["--days", "1"] + run("step1_collect_messages.py", s1) + + s2 = [] if args.full else ["--incremental"] + run("step2_extract_links.py", s2) + + s3 = [] if args.full else ["--incremental"] + run("step3_fetch_feishu_docs.py", s3) + + if not args.skip_download: + s4 = [] if args.full else ["--incremental"] + run("step4_download_files.py", s4) + + s5 = [] + if args.quick: s5 += ["--since", "today", "--brief"] + elif args.since: s5 += ["--since", args.since] + elif args.days: s5 += ["--days", str(args.days)] + run("step5_generate_summary.py", s5) + + if not args.skip_kb: + run("step6_update_knowledge_base.py") + + print(f"\n{'='*50}\n 全部完成! ({time.time()-t0:.1f}s)\n{'='*50}") + +if __name__ == "__main__": main() diff --git a/skills/dingtalk-feishu-collector/scripts/step1_collect_messages.py b/skills/dingtalk-feishu-collector/scripts/step1_collect_messages.py new file mode 100644 index 0000000..951e113 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/step1_collect_messages.py @@ -0,0 +1,103 @@ +"""Step 1: 从钉钉群拉取消息 + +用法: + python step1_collect_messages.py # 增量(最近3天) + python step1_collect_messages.py --days 1 # 增量(最近1天) + python step1_collect_messages.py --full # 全量拉取 + python step1_collect_messages.py --since "2026-06-05 00:00:00" +""" + +import argparse +import json +import os +import subprocess +import sys +from datetime import datetime, timedelta + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from paths import DATA_DIR, RAW_DIR, find_dws + +DWS = find_dws() + +def fetch_messages(group_id, time_str, forward="true", limit=200): + cmd = [DWS, "chat", "message", "list", "--group", group_id, + "--time", time_str, "--forward", forward, "--limit", str(limit), "--format", "json"] + try: + result = subprocess.run(cmd, capture_output=True, timeout=60) + output = result.stdout.decode("utf-8", errors="replace") + json_start = output.find("{") + if json_start < 0: return [] + data = json.loads(output[json_start:]) + return data.get("result", {}).get("messages", []) + except Exception as e: + print(f" WARN: {e}", file=sys.stderr) + return [] + +def collect_group(group_name, group_id, since): + print(f"\n=== {group_name} (since {since}) ===") + all_msgs = {} + msgs = fetch_messages(group_id, since, "true") + for m in msgs: all_msgs[m["openMessageId"]] = m + print(f" 正向: {len(msgs)} 条") + if len(msgs) >= 180: + times = sorted([m["createTime"] for m in msgs]) + if times: + msgs2 = fetch_messages(group_id, times[len(times)//2], "false") + for m in msgs2: all_msgs[m["openMessageId"]] = m + print(f" 反向补充: {len(msgs2)} 条") + result = list(all_msgs.values()) + if result: + times = sorted([m["createTime"] for m in result]) + print(f" 总计: {len(result)} 条 ({times[0]} ~ {times[-1]})") + return result + +GROUPS = { + "dc战略问题研究院": "cidoUneRB4Db8TAXaTrKxkQAw==", + "创新组": "cidMuM+itt5PeY7xNSWsv3M0g==", +} + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--days", type=int) + parser.add_argument("--since", type=str) + parser.add_argument("--full", action="store_true") + args = parser.parse_args() + + if args.since: + since = args.since + elif args.full: + since = "2026-05-01 00:00:00" + elif args.days: + since = (datetime.now() - timedelta(days=args.days)).strftime("%Y-%m-%d 00:00:00") + else: + since = (datetime.now() - timedelta(days=3)).strftime("%Y-%m-%d 00:00:00") + + print(f"dws: {DWS}\n时间: {since} ~ now") + + os.makedirs(RAW_DIR, exist_ok=True) + combined_path = os.path.join(RAW_DIR, "all_messages_combined.json") + existing = {} + if os.path.isfile(combined_path): + with open(combined_path, "r", encoding="utf-8") as f: + existing = json.load(f) + + new_count = 0 + for name, gid in GROUPS.items(): + new_msgs = collect_group(name, gid, since) + if name not in existing: existing[name] = [] + existing_ids = {m["openMessageId"] for m in existing[name]} + added = sum(1 for m in new_msgs if m["openMessageId"] not in existing_ids) + for m in new_msgs: + if m["openMessageId"] not in existing_ids: + existing[name].append(m) + existing_ids.add(m["openMessageId"]) + print(f" 新增: {added} 条 (总计: {len(existing[name])})") + new_count += added + + with open(combined_path, "w", encoding="utf-8") as f: + json.dump(existing, f, ensure_ascii=False, indent=2) + total = sum(len(v) for v in existing.values()) + print(f"\n完成! 新增 {new_count} 条,总计 {total} 条 -> {combined_path}") + +if __name__ == "__main__": + main() diff --git a/skills/dingtalk-feishu-collector/scripts/step2_extract_links.py b/skills/dingtalk-feishu-collector/scripts/step2_extract_links.py new file mode 100644 index 0000000..d8b5db2 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/step2_extract_links.py @@ -0,0 +1,51 @@ +"""Step 2: 从消息中提取飞书链接和文件附件""" +import argparse, json, os, re, sys +from datetime import datetime +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from paths import RAW_DIR, LINKS_DIR + +FEISHU_RE = re.compile(r'https?://[a-zA-Z0-9.-]+\.feishu\.cn/(?:wiki|docx)/[A-Za-z0-9]+') +FILE_RE = re.compile(r'\[文件\]\s+(.+?)\s+fileId:\s+(\S+)') +KIMI_RE = re.compile(r'https?://[a-zA-Z0-9.-]+\.ok\.kimi\.link/\S*') + +def extract(messages): + feishu, files, kimi = [], [], [] + seen_f, seen_files = set(), set() + for grp, msgs in messages.items(): + for m in msgs: + c, s, t = m.get("content",""), m.get("sender","?"), m.get("createTime","") + for url in FEISHU_RE.findall(c): + if url not in seen_f: + seen_f.add(url) + dm = re.search(r'feishu\.cn/(?:wiki|docx)/([A-Za-z0-9]+)', url) + feishu.append({"url":url,"doc_id":dm.group(1) if dm else "","type":"wiki" if "/wiki/" in url else "docx","sender":s,"time":t,"group":grp}) + for name, fid in FILE_RE.findall(c): + if fid not in seen_files: + seen_files.add(fid) + files.append({"name":name.strip(),"fileId":fid,"sender":s,"time":t,"group":grp}) + for url in KIMI_RE.findall(c): + kimi.append({"url":url,"desc":c.split("https")[0].strip()[:80],"sender":s,"group":grp}) + return {"feishu_links":feishu,"file_attachments":files,"kimi_links":kimi} + +def main(): + p = argparse.ArgumentParser(); p.add_argument("--incremental",action="store_true"); args = p.parse_args() + path = os.path.join(RAW_DIR, "all_messages_combined.json") + if not os.path.isfile(path): print(f"ERROR: {path} not found"); sys.exit(1) + with open(path,"r",encoding="utf-8") as f: messages = json.load(f) + print(f"消息: {sum(len(v) for v in messages.values())} 条") + r = extract(messages) + print(f"飞书: {len(r['feishu_links'])}, 附件: {len(r['file_attachments'])}, Kimi: {len(r['kimi_links'])}") + for grp in messages: + gl = [l for l in r['feishu_links'] if l['group']==grp] + gf = [f for f in r['file_attachments'] if f['group']==grp] + print(f" [{grp}] 飞书:{len(gl)} 附件:{len(gf)}") + os.makedirs(LINKS_DIR, exist_ok=True) + with open(os.path.join(LINKS_DIR,"all_feishu_links.json"),"w",encoding="utf-8") as f: + json.dump({"date":datetime.now().strftime("%Y-%m-%d"),"total":len(r['feishu_links']),"links":r['feishu_links']},f,ensure_ascii=False,indent=2) + with open(os.path.join(LINKS_DIR,"all_file_attachments.json"),"w",encoding="utf-8") as f: + json.dump(r['file_attachments'],f,ensure_ascii=False,indent=2) + with open(os.path.join(LINKS_DIR,"all_kimi_links.json"),"w",encoding="utf-8") as f: + json.dump(r['kimi_links'],f,ensure_ascii=False,indent=2) + print(f"保存至: {LINKS_DIR}") + +if __name__ == "__main__": main() diff --git a/skills/dingtalk-feishu-collector/scripts/step3_fetch_feishu_docs.py b/skills/dingtalk-feishu-collector/scripts/step3_fetch_feishu_docs.py new file mode 100644 index 0000000..cf9bd50 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/step3_fetch_feishu_docs.py @@ -0,0 +1,76 @@ +"""Step 3: 拉取飞书文档内容""" +import argparse, json, os, subprocess, sys, re +from datetime import datetime +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from paths import LINKS_DIR, DOCS_DIR, find_lark_cli + +def resolve_wiki(node_id, lark): + try: + r = subprocess.run([lark,"api","GET",f"/open-apis/wiki/v2/spaces/get_node?token={node_id}","--format","json"],capture_output=True,text=True,timeout=15) + if r.returncode==0: return json.loads(r.stdout).get("data",{}).get("node",{}).get("obj_token","") + except: pass + return "" + +def extract_text(obj): + if not obj: return "" + return "".join(e.get("text_run",{}).get("content","") for e in obj.get("elements",[])).strip() + +def blocks_to_md(blocks, doc_id): + lines = [f"# {doc_id}\n", f"Source: https://feishu.cn/docx/{doc_id}\n"] + for b in blocks: + bt = b.get("block_type",0) + if bt==2: t=extract_text(b.get("text",{})); [lines.append(t)] if t else None + elif bt in range(3,10): t=extract_text(b.get(f"heading{bt-2}",{}) or b.get("text",{})); lines.append(f"{'#'*(bt-2)} {t}") if t else None + elif bt==10: t=extract_text(b.get("bullet",{}) or b.get("text",{})); lines.append(f"- {t}") if t else None + elif bt==11: t=extract_text(b.get("ordered",{}) or b.get("text",{})); lines.append(f"1. {t}") if t else None + elif bt==14: t=extract_text(b.get("quote",{}) or b.get("text",{})); lines.append(f"> {t}") if t else None + elif bt==15: t=extract_text(b.get("code",{}) or b.get("text",{})); lines.append(f"```\n{t}\n```") if t else None + else: + for k in ["text","heading1","heading2","heading3","bullet","ordered","quote","code"]: + if k in b: t=extract_text(b[k]); [lines.append(t)] if t else None; break + return "\n\n".join(lines) + +def fetch_doc(doc_id, doc_type, lark): + real_id = resolve_wiki(doc_id, lark) if doc_type=="wiki" else doc_id + if not real_id: return f"# {doc_id}\n\n> wiki解析失败" + try: + r = subprocess.run([lark,"api","GET",f"/open-apis/docx/v1/documents/{real_id}/blocks","--format","json"],capture_output=True,text=True,timeout=30) + if r.returncode!=0: return f"# {doc_id}\n\n> 获取失败" + blocks = json.loads(r.stdout).get("data",{}).get("items",[]) + return blocks_to_md(blocks, doc_id) if blocks else f"# {doc_id}\n\n> 空文档" + except Exception as e: return f"# {doc_id}\n\n> {e}" + +def main(): + p = argparse.ArgumentParser(); p.add_argument("--incremental",action="store_true"); args = p.parse_args() + links_path = os.path.join(LINKS_DIR,"all_feishu_links.json") + if not os.path.isfile(links_path): print("ERROR: 先运行 step2"); sys.exit(1) + with open(links_path,"r",encoding="utf-8") as f: all_links = json.load(f).get("links",[]) + print(f"链接: {len(all_links)} 个") + + content_path = os.path.join(LINKS_DIR,"all_feishu_content.json") + existing = {} + if os.path.isfile(content_path): + with open(content_path,"r",encoding="utf-8") as f: + for d in json.load(f).get("documents",[]): existing[d.get("doc_id","")] = d + print(f"已拉取: {len(existing)}") + + lark = find_lark_cli() + os.makedirs(DOCS_DIR, exist_ok=True) + new_count = 0 + for link in all_links: + did = link.get("doc_id","") + if not did or did in existing: continue + print(f" {did} ({link.get('sender','?')})") + md = fetch_doc(did, link.get("type","docx"), lark) + with open(os.path.join(DOCS_DIR,f"{did}.md"),"w",encoding="utf-8") as f: f.write(md) + title = "" + for line in md.split("\n"): + if line.startswith("# "): title = line[2:].strip(); break + existing[did] = {"url":link.get("url",""),"doc_id":did,"title":title,"sender":link.get("sender",""),"group":link.get("group",""),"time":link.get("time",""),"content_length":len(md),"fetched":True} + new_count += 1 + + with open(content_path,"w",encoding="utf-8") as f: + json.dump({"date":datetime.now().strftime("%Y-%m-%d"),"total":len(existing),"documents":list(existing.values())},f,ensure_ascii=False,indent=2) + print(f"\n新增 {new_count},总计 {len(existing)} -> {DOCS_DIR}") + +if __name__ == "__main__": main() diff --git a/skills/dingtalk-feishu-collector/scripts/step4_download_files.py b/skills/dingtalk-feishu-collector/scripts/step4_download_files.py new file mode 100644 index 0000000..7a06a41 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/step4_download_files.py @@ -0,0 +1,31 @@ +"""Step 4: 下载钉钉群文件附件""" +import argparse, json, os, subprocess, sys +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from paths import LINKS_DIR, DOWNLOAD_DIR, find_dws + +HTML_MD = {".html",".htm",".md",".markdown",".txt"} + +def main(): + p = argparse.ArgumentParser(); p.add_argument("--incremental",action="store_true"); args = p.parse_args() + path = os.path.join(LINKS_DIR,"all_file_attachments.json") + if not os.path.isfile(path): print("ERROR: 先运行 step2"); sys.exit(1) + with open(path,"r",encoding="utf-8") as f: files = json.load(f) + print(f"附件: {len(files)} 个") + dws = find_dws() + new = 0 + for fi in files: + fid, name = fi.get("fileId",""), fi.get("name","unknown") + if not fid: continue + ext = os.path.splitext(name)[1].lower() + subdir = "html-md" if ext in HTML_MD else "other" + out = os.path.join(DOWNLOAD_DIR, subdir) + os.makedirs(out, exist_ok=True) + print(f" {name} -> {subdir}/") + try: + r = subprocess.run([dws,"drive","download","--node",fid,"--output",out],capture_output=True,timeout=120) + if r.returncode==0: new+=1 + else: print(f" FAIL") + except Exception as e: print(f" ERR: {e}") + print(f"\n下载 {new} 个 -> {DOWNLOAD_DIR}") + +if __name__ == "__main__": main() diff --git a/skills/dingtalk-feishu-collector/scripts/step5_generate_summary.py b/skills/dingtalk-feishu-collector/scripts/step5_generate_summary.py new file mode 100644 index 0000000..3eaa208 --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/step5_generate_summary.py @@ -0,0 +1,64 @@ +"""Step 5: 生成结构化总结""" +import argparse, json, os, sys +from datetime import datetime, timedelta +from collections import defaultdict +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from paths import DATA_DIR, LINKS_DIR, DOCS_DIR, REPORTS_DIR + +def load_json(path): + if not os.path.isfile(path): return None + with open(path,"r",encoding="utf-8") as f: return json.load(f) + +def main(): + p = argparse.ArgumentParser() + p.add_argument("--since",type=str); p.add_argument("--days",type=int) + p.add_argument("--full",action="store_true"); p.add_argument("--brief",action="store_true") + args = p.parse_args() + + messages = load_json(os.path.join(DATA_DIR,"raw-messages","all_messages_combined.json")) or {} + content = load_json(os.path.join(LINKS_DIR,"all_feishu_content.json")) or {} + files = load_json(os.path.join(LINKS_DIR,"all_file_attachments.json")) or [] + docs = {} + for d in content.get("documents",[]): + did = d.get("doc_id","") or d.get("url","") + if did: docs[did] = d + + if args.since: + since = datetime.now().strftime("%Y-%m-%d 00:00:00") if args.since=="today" else (datetime.now()-timedelta(days=1)).strftime("%Y-%m-%d 00:00:00") if args.since=="yesterday" else args.since + messages = {g:[m for m in msgs if m.get("createTime","")>=since] for g,msgs in messages.items()} + elif args.days: + since = (datetime.now()-timedelta(days=args.days)).strftime("%Y-%m-%d 00:00:00") + messages = {g:[m for m in msgs if m.get("createTime","")>=since] for g,msgs in messages.items()} + + lines = [f"# 钉钉群聊飞书文档采集报告", f"> 生成: {datetime.now().strftime('%Y-%m-%d %H:%M')}", ""] + total_m = sum(len(v) for v in messages.values()) + lines += ["## 数据总览","","| 指标 | 数量 |","|------|------|", + f"| 消息 | {total_m} |", f"| 文档 | {len(docs)} |", f"| 附件 | {len(files)} |", ""] + + for grp, msgs in messages.items(): + if not msgs: continue + times = sorted([m.get("createTime","") for m in msgs]) + gd = [d for d in docs.values() if d.get("group")==grp] + gf = [f for f in files if f.get("group")==grp] + lines += [f"### {grp}", f"- 消息: {len(msgs)} ({times[0]} ~ {times[-1]})", f"- 文档: {len(gd)}", f"- 附件: {len(gf)}", ""] + if args.brief: continue + senders = defaultdict(lambda: {"m":0,"d":0}) + for m in msgs: senders[m.get("sender","?")]["m"]+=1 + for d in gd: senders[d.get("sender","?")]["d"]+=1 + lines += ["| 发送者 | 消息 | 文档 |","|--------|------|------|"] + for s,st in sorted(senders.items(),key=lambda x:-(x[1]["m"]+x[1]["d"])): + lines.append(f"| {s} | {st['m']} | {st['d']} |") + lines.append("") + lines.append("**最新消息:**") + for m in sorted(msgs,key=lambda x:x.get("createTime",""),reverse=True)[:10]: + c = m.get('content','')[:80].replace('\n',' ') + lines.append(f"- [{m.get('createTime','')}] **{m.get('sender','?')}**: {c}") + lines.append("") + + os.makedirs(REPORTS_DIR, exist_ok=True) + report = "\n".join(lines) + for name in ["latest_summary.md", f"summary_{datetime.now().strftime('%Y%m%d')}.md"]: + with open(os.path.join(REPORTS_DIR,name),"w",encoding="utf-8") as f: f.write(report) + print(f"报告已生成: {REPORTS_DIR}") + +if __name__ == "__main__": main() diff --git a/skills/dingtalk-feishu-collector/scripts/step6_update_knowledge_base.py b/skills/dingtalk-feishu-collector/scripts/step6_update_knowledge_base.py new file mode 100644 index 0000000..bcf788d --- /dev/null +++ b/skills/dingtalk-feishu-collector/scripts/step6_update_knowledge_base.py @@ -0,0 +1,79 @@ +"""Step 6: 更新知识库(Obsidian + 知识图谱)""" +import json, os, sys +from datetime import datetime +from collections import defaultdict +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from paths import LINKS_DIR, DATA_DIR, DOCS_DIR, OBSIDIAN_DIR, KG_DIR + +def safe_name(n): + for c in '<>:"/\\|?*': n=n.replace(c,'_') + return n[:80] + +def w(path, content): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path,"w",encoding="utf-8") as f: f.write(content) + +def load_json(p): + if not os.path.isfile(p): return None + with open(p,"r",encoding="utf-8") as f: return json.load(f) + +def main(): + content = load_json(os.path.join(LINKS_DIR,"all_feishu_content.json")) or {} + messages = load_json(os.path.join(DATA_DIR,"raw-messages","all_messages_combined.json")) or {} + docs = content.get("documents",[]) + + print(f"文档: {len(docs)}, 消息: {sum(len(v) for v in messages.values())}") + + # Obsidian + by_group = defaultdict(list) + for d in docs: by_group[d.get("group","未分组")].append(d) + by_sender = defaultdict(list) + for d in docs: by_sender[d.get("sender","?")].append(d) + + moc = ["---","tags: [MOC, 飞书文档]",f"created: {datetime.now().strftime('%Y-%m-%d')}","---","","# 飞书文档知识库","",f"> 更新: {datetime.now().strftime('%Y-%m-%d %H:%M')}",""] + for grp, gd in sorted(by_group.items()): + moc.append(f"## {grp}") + for d in sorted(gd,key=lambda x:x.get("time",""),reverse=True): + t = d.get("title","") or d.get("doc_id","") + moc.append(f"- [[{safe_name(t)}]] ({d.get('sender','?')}, {d.get('time','')})") + moc.append("") + w(os.path.join(OBSIDIAN_DIR,"00-MOC","飞书文档索引.md"),"\n".join(moc)) + + for sender, sd in sorted(by_sender.items()): + lines = ["---",f"tags: [人物, {sender}]","---",f"# {sender}",f"\n贡献: {len(sd)} 篇\n"] + for d in sorted(sd,key=lambda x:x.get("time",""),reverse=True): + t = d.get("title","") or d.get("doc_id","") + lines.append(f"- [[{safe_name(t)}]] ({d.get('time','')})") + w(os.path.join(OBSIDIAN_DIR,"04-人物",f"{safe_name(sender)}.md"),"\n".join(lines)) + + for d in docs: + did = d.get("doc_id","") + fpath = os.path.join(DOCS_DIR,f"{did}.md") + if not os.path.isfile(fpath): continue + with open(fpath,"r",encoding="utf-8") as f: c = f.read() + t = d.get("title","") or did + fm = ["---",f"tags: [飞书文档, {d.get('group','')}]",f"sender: {d.get('sender','')}",f"date: {d.get('time','')}",f"source: {d.get('url','')}","---",""] + w(os.path.join(OBSIDIAN_DIR,"01-产品研究",f"{safe_name(t)}.md"),"\n".join(fm)+c) + print(f" Obsidian: {len(docs)} 文档, {len(by_sender)} 人物") + + # 知识图谱 + nodes, edges, nids = [], [], set() + for s in {d.get("sender","") for d in docs} | {m.get("sender","") for msgs in messages.values() for m in msgs}: + if s: nid=f"person:{s}"; nodes.append({"id":nid,"type":"person","label":s}); nids.add(nid) + for d in docs: + did,t = d.get("doc_id",""), d.get("title","") or d.get("doc_id","") + nid = f"doc:{did}" + if nid not in nids: nodes.append({"id":nid,"type":"document","label":t[:50]}); nids.add(nid) + s = d.get("sender","") + if s: edges.append({"source":f"person:{s}","target":nid,"type":"authored"}) + g = d.get("group","") + if g: + gid=f"group:{g}" + if gid not in nids: nodes.append({"id":gid,"type":"group","label":g}); nids.add(gid) + edges.append({"source":gid,"target":nid,"type":"contains"}) + os.makedirs(KG_DIR, exist_ok=True) + with open(os.path.join(KG_DIR,"knowledge_graph.json"),"w",encoding="utf-8") as f: + json.dump({"date":datetime.now().strftime("%Y-%m-%d"),"nodes":nodes,"edges":edges},f,ensure_ascii=False,indent=2) + print(f" 图谱: {len(nodes)} 节点, {len(edges)} 关系") + +if __name__ == "__main__": main()