feat: add dingtalk-feishu-collector SOP skill
Reusable 6-step pipeline for collecting DingTalk group messages, extracting Feishu doc links, fetching content, and generating summaries. Skills structure: - SKILL.md: trigger rules, workflow, config reference - config.yaml: group IDs, collection settings - scripts/paths.py: shared path resolution - scripts/step1-6: modular pipeline steps - scripts/run_all.py: one-click runner
This commit is contained in:
@@ -0,0 +1,149 @@
|
||||
[
|
||||
{
|
||||
"name": "游戏行业核心观点梳理.pdf",
|
||||
"fileId": "b9Y4gmKWrPNm30nEFjoRXP4aJGXn6lpz",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-15 10:28:47",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "AI视频创作作品分享罗剑豪V3.pptx",
|
||||
"fileId": "yQod3RxJKGDndA07i4noekLyJkb4Mw9r",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-13 14:36:23",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "坦克大逃杀答辩材料:从代码搬运工到架构师助理.pptx",
|
||||
"fileId": "QBnd5ExVEvEp5KZ2I2pgG5ElJyeZqMmz",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-13 14:36:02",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "《仙途乐逍遥》AI歌曲分享文档.xmind",
|
||||
"fileId": "R4GpnMqJzGmy3dxEiaZkMN5R8Ke0xjE3",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-13 14:35:35",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "刘亚彬-万剑囚天镇压獓因-分享文档02.pptx",
|
||||
"fileId": "DnRL6jAJMGMRdeY0iXRqqAa6WyMoPYe1",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-13 14:34:08",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "寻仇AI视频制作分享PPT (2).pptx",
|
||||
"fileId": "0eMKjyp813zlbnexs4j4XpbgVxAZB1Gv",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-13 14:21:13",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "模型报告.html",
|
||||
"fileId": "MyQA2dXW7eRdPjKxS5AA5P41JzlwrZgb",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-11 01:28:58",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "game-analyst-agent.zip",
|
||||
"fileId": "mweZ92PV6M7e1qQxSKmpwOq0WxEKBD6p",
|
||||
"sender": "黄静雯",
|
||||
"time": "2026-05-22 09:02:50",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "game-analyst-agent-manual.html",
|
||||
"fileId": "QPGYqjpJYr7lQGAXFZRQ6lnM8akx1Z5N",
|
||||
"sender": "黄静雯",
|
||||
"time": "2026-05-22 09:02:50",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "2026-05-yunhu-project-selection-strategy-report.html",
|
||||
"fileId": "4lgGw3P8vR203ZrEHpG2gyYE85daZ90D",
|
||||
"sender": "黄静雯",
|
||||
"time": "2026-05-21 00:35:28",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "取败之道——关于幸存者偏差、筹码管理与生长逻辑的深度反思.md",
|
||||
"fileId": "0eMKjyp813zlbnexs4d9bqrZVxAZB1Gv",
|
||||
"sender": "李志健",
|
||||
"time": "2026-05-20 10:22:27",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "花园世界产品策略_GOS补充信息.md",
|
||||
"fileId": "NkDwLng8ZLRBbjGpCxxqXz1MVKMEvZBY",
|
||||
"sender": "傅明游",
|
||||
"time": "2026-05-26 16:00:45",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "2026-05-25_report_my_garden_world_us_project.html",
|
||||
"fileId": "wva2dxOW4YmAb01xF0BnvLNEVbkz3BRL",
|
||||
"sender": "黄静雯",
|
||||
"time": "2026-05-26 09:29:12",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "主报告-我的花园世界.html",
|
||||
"fileId": "y20BglGWO2NbdYyKt0peOkjl8A7depqY",
|
||||
"sender": "李志健",
|
||||
"time": "2026-06-01 23:36:40",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "进一步的思考.md",
|
||||
"fileId": "QPGYqjpJYr7lQGAXFZ0njADy8akx1Z5N",
|
||||
"sender": "李志健",
|
||||
"time": "2026-06-01 23:09:45",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "花园项目对策划的启示.md",
|
||||
"fileId": "7QG4Yx2JpLMZmdaECglLlO2aJ9dEq3XD",
|
||||
"sender": "李志健",
|
||||
"time": "2026-06-01 22:55:36",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "花园世界等相关产品25.10.28.xlsx",
|
||||
"fileId": "QPGYqjpJYr7lQGAXFZ0beYjZ8akx1Z5N",
|
||||
"sender": "李志健",
|
||||
"time": "2026-06-01 21:42:33",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "2026-06-03_garden-social-management-us-market-assessment.html",
|
||||
"fileId": "QBnd5ExVEvEp5KZ2I24Ozg7EJyeZqMmz",
|
||||
"sender": "黄静雯",
|
||||
"time": "2026-06-03 09:32:58",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "2026-06-04_social-platform-dependency-overseas-expansion-v2.html",
|
||||
"fileId": "1OQX0akWmxrpR12EhveqkY448GlDd3mE",
|
||||
"sender": "黄静雯",
|
||||
"time": "2026-06-05 09:29:32",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "数据口径入门教程.md",
|
||||
"fileId": "QPGYqjpJYr7lQGAXFZqd3GMM8akx1Z5N",
|
||||
"sender": "李志健",
|
||||
"time": "2026-06-05 21:11:44",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"name": "工作流总结-王雨默.html",
|
||||
"fileId": "1zknDm0WRamRxP60IxnN2Dwk8BQEx5rG",
|
||||
"sender": "王雨默",
|
||||
"time": "2026-05-22 10:57:31",
|
||||
"group": "cx"
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,62 @@
|
||||
[
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/",
|
||||
"desc": "[分享] 我的花园世界 - 资源经济深度研究",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/",
|
||||
"desc": "[分享] 我的花园世界 - 资源经济深度研究",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/",
|
||||
"desc": "[分享] 我的花园世界 - 资源经济深度研究",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://kblb6hdbtbfto.ok.kimi.link/",
|
||||
"desc": "[分享] 《我的花园世界》商业化梳理报告——基于游戏机制以及社区舆论",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://kblb6hdbtbfto.ok.kimi.link/",
|
||||
"desc": "[分享] 《我的花园世界》商业化梳理报告——基于游戏机制以及社区舆论",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://kblb6hdbtbfto.ok.kimi.link/",
|
||||
"desc": "[分享] 《我的花园世界》商业化梳理报告——基于游戏机制以及社区舆论",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/#/long-term-study",
|
||||
"desc": "",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/#/activity-study",
|
||||
"desc": "",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/#/player-autonomy",
|
||||
"desc": "",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
},
|
||||
{
|
||||
"url": "https://upy4l3m3qnegc.ok.kimi.link/#/healing-to-game",
|
||||
"desc": "",
|
||||
"sender": "陈楚真",
|
||||
"group": "dc"
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,469 @@
|
||||
# 从钉钉群聊到知识库:飞书文档自动采集与结构化系统 - 复刻指南
|
||||
|
||||
> 作者:大师 | 日期:2026-06-03
|
||||
|
||||
---
|
||||
|
||||
## 一句话说清楚这个项目做了什么
|
||||
|
||||
**从钉钉群聊消息中自动抓取飞书文档链接和文件附件,下载内容,构建知识图谱和 Obsidian 知识库。**
|
||||
|
||||
整个流程 = 3个核心工具 + 6个Python脚本,不需要自己写任何API对接代码。
|
||||
|
||||
---
|
||||
|
||||
## 整体架构
|
||||
|
||||
```
|
||||
钉钉群聊消息
|
||||
|
|
||||
v
|
||||
(1) dws CLI(悟空)-- 拉取钉钉消息、下载文件附件
|
||||
|
|
||||
v
|
||||
(2) Python 脚本 -- 正则提取飞书链接 + 文件ID
|
||||
|
|
||||
v
|
||||
(3) lark-cli -- 读取飞书文档内容(Block API)
|
||||
|
|
||||
v
|
||||
(4) Python 脚本 -- 构建知识图谱(JSON)、Obsidian 库、汇总报告
|
||||
```
|
||||
|
||||
**关键认知:你不需要申请飞书开放平台的App。** 消息来源是钉钉(通过dws),飞书文档内容获取通过lark-cli(浏览器授权登录即可)。
|
||||
|
||||
---
|
||||
|
||||
## 核心工具获取指南
|
||||
|
||||
### 工具 1:dws CLI -- 钉钉消息采集(随悟空自动安装)
|
||||
|
||||
**是什么**:dws CLI 是钉钉「悟空」(Wukong) 桌面客户端自带的命令行工具,封装了钉钉 25+ 项 MCP 服务(群聊消息、文件管理、日历、通讯录、审批等),无需自己对接钉钉开放平台API。
|
||||
|
||||
**获取方式(三种,选一种即可)**:
|
||||
|
||||
| 方式 | 适合谁 | 操作 |
|
||||
|------|--------|------|
|
||||
| **GitHub 直接下载**(推荐) | 所有人 | 从 GitHub Releases 下载对应平台的压缩包,解压即用 |
|
||||
| 安装悟空客户端 | 钉钉重度用户 | 装完悟空,dws 自动在 `C:\Program Files\Wukong\<版本>\bin\dws.exe` |
|
||||
| 从同事那里复制 | 最省事 | 复制一个 `dws.exe` 文件(约 5-14MB) |
|
||||
|
||||
**GitHub 仓库**:https://github.com/DingTalk-Real-AI/dingtalk-workspace-cli
|
||||
|
||||
这是钉钉官方开源的 CLI 工具,2000+ stars,持续更新中(最新 v1.0.33,2026-06-02 发布)。
|
||||
|
||||
```bash
|
||||
# 从 GitHub Releases 直接下载(以 Windows 为例):
|
||||
# https://github.com/DingTalk-Real-AI/dingtalk-workspace-cli/releases/latest
|
||||
# 下载 dws-windows-amd64.zip(约 5.3MB),解压得到 dws.exe
|
||||
|
||||
# macOS / Linux 也有对应版本:
|
||||
# dws-darwin-amd64.tar.gz (macOS Intel)
|
||||
# dws-darwin-arm64.tar.gz (macOS Apple Silicon)
|
||||
# dws-linux-amd64.tar.gz (Linux x64)
|
||||
|
||||
```bash
|
||||
# 悟空是钉钉的AI桌面客户端(也叫钉钉Real版),安装后自动附带:
|
||||
# - dws CLI(钉钉MCP服务命令行)
|
||||
# - wukong-cli(悟空Agent命令行)
|
||||
# - Node.js、Python、ffmpeg 等运行时
|
||||
#
|
||||
# 安装路径参考:
|
||||
# 主程序:C:\Program Files\Wukong\<版本号>\DingTalkReal.exe
|
||||
# dws CLI:C:\Program Files\Wukong\<版本号>\bin\dws.exe
|
||||
# 运行时缓存:C:\Users\<用户名>\.real\.bin\dws\bin\dws.exe
|
||||
```
|
||||
|
||||
**版本信息**(本项目实际使用):
|
||||
|
||||
| 项目 | 值 |
|
||||
|------|-----|
|
||||
| 悟空版本 | 0.9.51 |
|
||||
| dws CLI 版本 | 0.2.75(悟空内置)/ 1.0.33(GitHub 最新) |
|
||||
| 架构 | MCP Dynamic Aggregation |
|
||||
| Go 版本 | 1.24+ |
|
||||
|
||||
**认证**:首次使用需要登录认证(钉钉扫码或账号登录),Corp ID 和 User ID 会自动配置。运行 `dws auth status` 可查看登录状态。
|
||||
|
||||
**能把 dws 单独提取出来用吗?可以。**
|
||||
|
||||
dws.exe 是 Go 语言静态编译的二进制文件(14MB),不依赖任何外部 DLL,可以脱离悟空独立运行。提取方法:
|
||||
|
||||
```bash
|
||||
# 只需要一个文件:dws.exe(14MB)
|
||||
# 路径:C:\Program Files\Wukong\<版本号>\bin\dws.exe
|
||||
# 或:C:\Users\<用户名>\.real\.bin\dws\bin\dws.exe
|
||||
|
||||
# 1. 复制 dws.exe 到任意目录
|
||||
copy "C:\Users\admin\.real\.bin\dws\bin\dws.exe" D:\tools\dws.exe
|
||||
|
||||
# 2. 首次运行,触发认证(会自动创建 .dws 数据目录)
|
||||
D:\tools\dws.exe auth status
|
||||
# 如果未登录,会提示 OAuth 扫码登录(用钉钉App扫码)
|
||||
|
||||
# 3. 认证成功后,目录下会自动生成 .dws/ 子目录:
|
||||
# .dws/identity.json -- 身份标识
|
||||
# .dws/token.json -- Token(自动续期)
|
||||
# .dws/.data -- 加密凭证
|
||||
# .dws/logs/ -- 日志
|
||||
|
||||
# 4. 验证可用
|
||||
D:\tools\dws.exe chat message list --help
|
||||
D:\tools\dws.exe drive download --help
|
||||
```
|
||||
|
||||
**提取后的体积**:
|
||||
|
||||
| 文件 | 大小 |
|
||||
|------|------|
|
||||
| `dws.exe` | 14 MB |
|
||||
| `.dws/` 数据目录 | < 1 MB |
|
||||
| **总计** | **约 14 MB** |
|
||||
|
||||
**认证机制详解(dws 自带,不需要悟空)**:
|
||||
|
||||
dws 内置了完整的 OAuth 认证流程,提取出来后独立就能完成登录:
|
||||
|
||||
```
|
||||
运行 dws auth status(未登录状态)
|
||||
-> dws 启动 Device Flow OAuth
|
||||
-> 连接 login.dingtalk.com/oauth2/auth
|
||||
-> 终端显示二维码 / URL
|
||||
-> 用钉钉App扫码授权
|
||||
-> dws 轮询 api.dingtalk.com 获取 Token
|
||||
-> Token 加密存储到 .dws/.data
|
||||
-> 完成
|
||||
```
|
||||
|
||||
dws 内置了默认的 OAuth ClientID/ClientSecret,普通用户直接用就行,不需要自己申请。如果公司有自建应用,也可以通过环境变量覆盖:
|
||||
|
||||
```bash
|
||||
# 可选:使用自建应用的凭证(一般不需要)
|
||||
export DWS_CLIENT_ID=<你的AppKey>
|
||||
export DWS_CLIENT_SECRET=<你的AppSecret>
|
||||
```
|
||||
|
||||
**注意事项**:
|
||||
- Token 会自动续期,但长时间不用会过期,重新运行 `dws auth status` 触发重新扫码即可
|
||||
- 每个人需要用自己的钉钉账号认证,不能共用 Token
|
||||
- 提取出来的 dws 功能完整,支持全部 25+ 项 MCP 服务
|
||||
- `.dws/.data` 是加密存储的凭证文件(511字节),不要分享给别人
|
||||
|
||||
**核心能力**:
|
||||
|
||||
| 命令 | 用途 | 示例 |
|
||||
|------|------|------|
|
||||
| `dws chat message list` | 拉取群聊消息 | `dws chat message list --group <群ID> --time "2026-05-01" --format json --limit 200` |
|
||||
| `dws drive download` | 下载文件附件 | `dws drive download --node <fileId> --output ./files/` |
|
||||
|
||||
**获取群组ID**:需要知道目标群的 `openConversationId`,可以通过以下方式获取:
|
||||
- 在钉钉管理后台查看
|
||||
- 或者先用 `dws chat group list` 命令列出你所在的群
|
||||
|
||||
**分页策略**:API 每次最多返回约200条消息,超过的话需要分段拉取 + 用 `openMessageId` 去重:
|
||||
|
||||
```python
|
||||
# 分段拉取示例
|
||||
segments = [
|
||||
("2026-05-01 00:00:00", "true"), # 从5月1日向前
|
||||
("2026-05-19 00:00:00", "true"), # 从5月19日向前
|
||||
("2026-05-24 00:00:00", "true"), # 从5月24日向前
|
||||
("2026-06-03 00:00:00", "false"), # 从6月3日向后
|
||||
]
|
||||
# forward="true" 表示从该时间点向前(更新的消息)
|
||||
# forward="false" 表示从该时间点向后(更旧的消息)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 工具 2:lark-cli -- 飞书文档内容读取
|
||||
|
||||
**是什么**:飞书官方提供的命令行工具,通过浏览器 OAuth 授权后,可以直接调用飞书 Open API 读取文档内容。
|
||||
|
||||
**安装**:
|
||||
|
||||
```bash
|
||||
# 需要 Node.js 18+
|
||||
npm install -g @larksuite/cli
|
||||
|
||||
# 验证安装
|
||||
lark-cli --version
|
||||
```
|
||||
|
||||
**认证配置**(关键步骤):
|
||||
|
||||
```bash
|
||||
# 1. 初始化配置
|
||||
lark-cli config init
|
||||
|
||||
# 2. 浏览器授权登录(会自动打开浏览器)
|
||||
lark-cli auth login --domain docs,drive,wiki --recommend
|
||||
|
||||
# 3. 验证授权状态
|
||||
lark-cli auth status
|
||||
```
|
||||
|
||||
**为什么能拿到飞书文档内容?**
|
||||
|
||||
这是大家最关心的问题,答案是:
|
||||
|
||||
1. **lark-cli 使用浏览器 OAuth 授权**:你在浏览器里登录自己的飞书账号,lark-cli 拿到你的访问令牌(token)。
|
||||
2. **用你的身份调飞书 Open API**:后续所有请求都是以你个人身份发出的,跟你在浏览器里打开文档一样。
|
||||
3. **你有权限看的文档,lark-cli 就能读**:不是破解,不是爬虫,就是正常的API调用。
|
||||
|
||||
```
|
||||
你(浏览器登录飞书)-> OAuth Token -> lark-cli -> 飞书 Open API -> 文档内容
|
||||
```
|
||||
|
||||
**核心命令**:
|
||||
|
||||
```bash
|
||||
# 获取文档内容(Block API,返回JSON格式)
|
||||
lark-cli api GET /open-apis/docx/v1/documents/<doc_id>/blocks --format json
|
||||
|
||||
# doc_id 从飞书链接中提取:
|
||||
# 例:https://dianchukeji.feishu.cn/docx/AbLed72FgoPn5hxOYN3cRpffn0E
|
||||
# ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ 这就是 doc_id
|
||||
```
|
||||
|
||||
**注意事项**:
|
||||
|
||||
- Token 有过期时间,长时间不用需要重新授权
|
||||
- 你只能读取你有权限的飞书文档(跟在浏览器里访问一样)
|
||||
- wiki 类型的链接需要先解析真实的 doc_id(wiki 链接里的ID是节点ID,不是文档ID)
|
||||
|
||||
---
|
||||
|
||||
### 工具 3:Python 脚本 -- 数据处理管线
|
||||
|
||||
**依赖安装**:
|
||||
|
||||
```bash
|
||||
pip install beautifulsoup4 requests
|
||||
```
|
||||
|
||||
**7个脚本的职责**:
|
||||
|
||||
| 脚本 | 输入 | 输出 | 干什么 |
|
||||
|------|------|------|--------|
|
||||
| `fetch_all.py` | dws CLI | `data/raw-messages/all_messages_combined.json` | 拉取两个群的全部消息 |
|
||||
| `extract_links.py` | 上一步的JSON | 终端输出(统计信息) | 正则提取飞书链接、文件附件、Kimi链接 |
|
||||
| `build_graph.py` | 消息JSON | `output/knowledge-graph/knowledge_graph.json` | 构建知识图谱(人物、文档、主题、关系) |
|
||||
| `write_obsidian.py` | 硬编码内容 | `output/obsidian-vault/` 目录 | 生成 Obsidian 知识库(含双向链接) |
|
||||
| `write_report.py` | 硬编码内容 | `output/reports/花园世界全量汇总.md` | 生成全量汇总报告 |
|
||||
| `gen_visual.py` | knowledge_graph.json | `output/knowledge-graph/` HTML+Mermaid | 生成可视化知识图谱 |
|
||||
| `daily_feishu_collector.py` | dws CLI | `feishu_links_YYYYMMDD.json` | 每日定时采集(增量) |
|
||||
|
||||
---
|
||||
|
||||
## 钉钉消息为什么能拿到?
|
||||
|
||||
**原理**:通过 dws CLI(悟空工具)直接调用钉钉内部API。
|
||||
|
||||
```
|
||||
dws CLI -> 钉钉内部 API -> 群聊消息(含发送者、时间、内容、文件ID)
|
||||
```
|
||||
|
||||
- dws CLI 是公司内部工具,已经封装好了钉钉API的认证和调用
|
||||
- 你只需要提供群组的 `openConversationId` 和时间范围
|
||||
- 返回的消息内容是 JSON 格式,包含消息文本、发送者、时间戳等
|
||||
- 文件附件可以通过 `fileId` 用 `dws drive download` 下载
|
||||
|
||||
**不需要**:申请钉钉开放平台应用、配置回调URL、处理webhook。
|
||||
|
||||
---
|
||||
|
||||
## 飞书文档为什么能拿到?
|
||||
|
||||
**原理**:lark-cli 使用你的飞书账号 OAuth 授权,以你的身份调用飞书 Open API。
|
||||
|
||||
```
|
||||
你登录飞书 -> OAuth Token -> lark-cli -> /open-apis/docx/v1/documents/{id}/blocks -> 文档内容JSON
|
||||
```
|
||||
|
||||
**三个前提条件**:
|
||||
|
||||
1. **你有飞书账号**:公司飞书租户下的账号
|
||||
2. **你有文档访问权限**:文档对你可见(在群里分享过的文档,群成员通常都有权限)
|
||||
3. **lark-cli 授权成功**:`lark-cli auth login` 一次即可,后续自动使用缓存的token
|
||||
|
||||
**能读到什么**:
|
||||
|
||||
- `docx` 类型文档:直接通过 doc_id 调 Block API 获取全部内容
|
||||
- `wiki` 类型文档:需要先通过 wiki API 解析节点ID得到真实 doc_id,再调 Block API
|
||||
- 文件附件(XLSX/PPT/PDF等):通过 `dws drive download` 从钉钉侧下载
|
||||
|
||||
**读不到什么**:
|
||||
|
||||
- 你没有权限的文档(跟浏览器一样,没权限就是没权限)
|
||||
- 已被删除的文档
|
||||
|
||||
---
|
||||
|
||||
## 复刻步骤(从零开始)
|
||||
|
||||
### 第一步:确认工具就绪
|
||||
|
||||
```bash
|
||||
# 检查 dws CLI
|
||||
dws --version
|
||||
# 如果没有,找IT获取
|
||||
|
||||
# 检查 Node.js
|
||||
node --version # 需要 18+
|
||||
|
||||
# 检查 Python
|
||||
python --version # 需要 3.10+
|
||||
|
||||
# 安装 lark-cli
|
||||
npm install -g @larksuite/cli
|
||||
|
||||
# 安装 Python 依赖
|
||||
pip install beautifulsoup4 requests
|
||||
```
|
||||
|
||||
### 第二步:授权 lark-cli
|
||||
|
||||
```bash
|
||||
lark-cli config init
|
||||
lark-cli auth login --domain docs,drive,wiki --recommend
|
||||
# 浏览器会自动打开,登录你的飞书账号即可
|
||||
```
|
||||
|
||||
### 第三步:获取群组ID
|
||||
|
||||
你需要知道要采集的钉钉群的 `openConversationId`。获取方式:
|
||||
|
||||
1. 在钉钉管理后台查看
|
||||
2. 或者用 dws 命令列出你所在的群
|
||||
3. 或者找之前已经获取过的同事要
|
||||
|
||||
### 第四步:拉取消息
|
||||
|
||||
修改 `scripts/fetch_all.py` 中的群组ID和时间范围,然后运行:
|
||||
|
||||
```bash
|
||||
python scripts/fetch_all.py
|
||||
```
|
||||
|
||||
这会生成 `data/raw-messages/all_messages_combined.json`,包含两个群的全部消息。
|
||||
|
||||
### 第五步:提取链接
|
||||
|
||||
```bash
|
||||
python scripts/extract_links.py
|
||||
```
|
||||
|
||||
输出统计信息:飞书链接数、文件附件数、Kimi链接数等。
|
||||
|
||||
### 第六步:构建知识图谱
|
||||
|
||||
```bash
|
||||
python scripts/build_graph.py
|
||||
```
|
||||
|
||||
生成 `output/knowledge-graph/knowledge_graph.json`,包含人物、文档、主题、关系的结构化数据。
|
||||
|
||||
### 第七步:生成可视化和知识库
|
||||
|
||||
```bash
|
||||
# 知识图谱可视化
|
||||
python scripts/gen_visual.py
|
||||
# 输出:output/knowledge-graph/knowledge_graph.html(浏览器打开即可查看)
|
||||
|
||||
# Obsidian 知识库
|
||||
python scripts/write_obsidian.py
|
||||
# 输出:output/obsidian-vault/(用 Obsidian 打开此目录)
|
||||
|
||||
# 汇总报告
|
||||
python scripts/write_report.py
|
||||
```
|
||||
|
||||
### 第八步:设置定时采集(可选)
|
||||
|
||||
```powershell
|
||||
# Windows 定时任务,每天18:00执行
|
||||
schtasks /create /tn "FeishuDocCollector" /tr "python D:\path\to\scripts\daily_feishu_collector.py" /sc daily /st 18:00
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 常见问题
|
||||
|
||||
### Q: 需要申请飞书开放平台的App吗?
|
||||
|
||||
**不需要。** lark-cli 用的是浏览器 OAuth 授权(你的个人身份),不需要创建企业自建应用。
|
||||
|
||||
### Q: 需要申请钉钉开放平台的权限吗?
|
||||
|
||||
**不需要。** dws CLI 是内部工具,已经封装好了认证。
|
||||
|
||||
### Q: 飞书文档有4种域名,都能读吗?
|
||||
|
||||
都能读,只要你有权限。不同域名对应不同的飞书空间/租户,lark-cli 用你的账号登录后可以跨空间访问。
|
||||
|
||||
本项目涉及的4个飞书域:
|
||||
|
||||
| 域名 | 说明 |
|
||||
|------|------|
|
||||
| `dianchukeji.feishu.cn` | 公司飞书 |
|
||||
| `fcnlycv6dd0w.feishu.cn` | 研究组飞书空间 |
|
||||
| `ocnmca6f1o0p.feishu.cn` | 另一飞书空间 |
|
||||
| `my.feishu.cn` | 个人飞书 |
|
||||
|
||||
### Q: 消息太多拉不完怎么办?
|
||||
|
||||
分段拉取 + `openMessageId` 去重。见 `fetch_all.py` 中的分段策略。
|
||||
|
||||
### Q: Token 过期了怎么办?
|
||||
|
||||
重新运行 `lark-cli auth login --domain docs,drive,wiki --recommend`,浏览器重新授权即可。
|
||||
|
||||
### Q: wiki 链接和 docx 链接有什么区别?
|
||||
|
||||
docx 链接的ID就是文档ID,可以直接调API。wiki 链接的ID是知识库节点ID,需要先通过 wiki API 解析出真实的文档ID。
|
||||
|
||||
---
|
||||
|
||||
## 技术栈总结
|
||||
|
||||
| 层级 | 工具 | 获取方式 | 费用 |
|
||||
|------|------|----------|------|
|
||||
| 钉钉消息采集 | dws CLI | 悟空(Wukong)自带,装悟空就有 | 免费 |
|
||||
| 钉钉文件下载 | dws drive | 同上,dws 的子命令 | 免费 |
|
||||
| 飞书文档读取 | lark-cli | `npm install -g @larksuite/cli` | 免费 |
|
||||
| 数据处理 | Python + BeautifulSoup | 悟空自带Python,BS4需 `pip install` | 免费 |
|
||||
| 知识库管理 | Obsidian | https://obsidian.md 下载 | 免费 |
|
||||
|
||||
**总成本:0元,只需要你有飞书账号和钉钉群访问权限。**
|
||||
|
||||
---
|
||||
|
||||
## 产出物展示
|
||||
|
||||
| 产出 | 路径 | 用途 |
|
||||
|------|------|------|
|
||||
| 知识图谱(交互式) | `output/knowledge-graph/knowledge_graph.html` | 浏览器打开,查看人物、文档、主题关系 |
|
||||
| Obsidian 知识库 | `output/obsidian-vault/` | 用 Obsidian 打开,双向链接浏览 |
|
||||
| 汇总报告 | `output/reports/花园世界全量汇总.md` | 一份完整的 Markdown 报告 |
|
||||
| 原始文档 | `output/feishu-docs/` | 155篇飞书文档的本地备份 |
|
||||
| 下载文件 | `dc_files/` | PPT/PDF/XLSX 等附件 |
|
||||
|
||||
---
|
||||
|
||||
## 项目数据一览
|
||||
|
||||
| 指标 | 数量 |
|
||||
|------|------|
|
||||
| 采集群组 | 2个(dc战略问题研究院 + 创新组) |
|
||||
| 消息总数 | 377条 |
|
||||
| 飞书文档 | 155篇 |
|
||||
| 文件附件 | 19个 |
|
||||
| 活跃人员 | 28位 |
|
||||
| 时间跨度 | 2026-05-01 ~ 2026-06-03 |
|
||||
| 知识主题 | 7个核心主题 |
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
# 钉钉群聊飞书文档采集报告
|
||||
> 生成: 2026-06-06 11:12
|
||||
|
||||
## 数据总览
|
||||
|
||||
| 指标 | 数量 |
|
||||
|------|------|
|
||||
| 消息 | 7 |
|
||||
| 文档 | 139 |
|
||||
| 附件 | 21 |
|
||||
|
||||
### dc
|
||||
- 消息: 5 (2026-06-06 00:11:33 ~ 2026-06-06 09:36:25)
|
||||
- 文档: 19
|
||||
- 附件: 20
|
||||
|
||||
### cx
|
||||
- 消息: 2 (2026-06-06 00:35:41 ~ 2026-06-06 01:48:34)
|
||||
- 文档: 7
|
||||
- 附件: 1
|
||||
@@ -0,0 +1,20 @@
|
||||
# 钉钉群聊飞书文档采集报告
|
||||
> 生成: 2026-06-06 11:12
|
||||
|
||||
## 数据总览
|
||||
|
||||
| 指标 | 数量 |
|
||||
|------|------|
|
||||
| 消息 | 7 |
|
||||
| 文档 | 139 |
|
||||
| 附件 | 21 |
|
||||
|
||||
### dc
|
||||
- 消息: 5 (2026-06-06 00:11:33 ~ 2026-06-06 09:36:25)
|
||||
- 文档: 19
|
||||
- 附件: 20
|
||||
|
||||
### cx
|
||||
- 消息: 2 (2026-06-06 00:35:41 ~ 2026-06-06 01:48:34)
|
||||
- 文档: 7
|
||||
- 附件: 1
|
||||
@@ -0,0 +1,180 @@
|
||||
# 钉钉群聊飞书文档采集与知识整理 SOP
|
||||
|
||||
## 触发条件
|
||||
|
||||
当用户提到以下关键词时使用此技能:
|
||||
- "收集钉钉消息"、"拉取群聊"、"采集飞书文档"
|
||||
- "整理群里的链接"、"汇总飞书报告"
|
||||
- "每日收集"、"定时采集"
|
||||
- 涉及钉钉群聊 + 飞书文档的工作流
|
||||
|
||||
## 一句话概述
|
||||
|
||||
从钉钉群聊中自动采集飞书文档链接和文件附件,下载内容,生成结构化总结。
|
||||
|
||||
## 前置条件
|
||||
|
||||
| 工具 | 用途 | 获取方式 |
|
||||
|------|------|----------|
|
||||
| dws CLI | 钉钉消息采集 + 文件下载 | 悟空(Wukong)自带,或从 GitHub 下载 |
|
||||
| lark-cli | 飞书文档内容读取 | `npm install -g @larksuite/cli` |
|
||||
| Python 3.10+ | 数据处理 | 系统自带或悟空内置 |
|
||||
|
||||
**认证要求**:
|
||||
- dws CLI:运行 `dws auth status` 确认已登录(钉钉扫码)
|
||||
- lark-cli:运行 `lark-cli auth status` 确认已授权(飞书浏览器OAuth)
|
||||
|
||||
## 完整工作流(6步)
|
||||
|
||||
### Step 1: 拉取钉钉群消息
|
||||
|
||||
```bash
|
||||
python scripts/step1_collect_messages.py # 全量
|
||||
python scripts/step1_collect_messages.py --days 1 # 增量
|
||||
```
|
||||
|
||||
**输出**:`data/raw-messages/all_messages_combined.json`
|
||||
|
||||
### Step 2: 提取链接与附件
|
||||
|
||||
```bash
|
||||
python scripts/step2_extract_links.py # 全量
|
||||
python scripts/step2_extract_links.py --incremental # 增量
|
||||
```
|
||||
|
||||
正则提取飞书链接、文件附件、Kimi链接。
|
||||
|
||||
**输出**:`data/links/all_feishu_links.json` + `data/links/all_file_attachments.json`
|
||||
|
||||
### Step 3: 拉取飞书文档内容
|
||||
|
||||
```bash
|
||||
python scripts/step3_fetch_feishu_docs.py # 全量
|
||||
python scripts/step3_fetch_feishu_docs.py --incremental # 增量
|
||||
```
|
||||
|
||||
wiki/docx 自动识别,Block API 解析为 Markdown。
|
||||
|
||||
**输出**:`output/feishu-docs/<doc_id>.md` + `data/links/all_feishu_content.json`
|
||||
|
||||
### Step 4: 下载文件附件
|
||||
|
||||
```bash
|
||||
python scripts/step4_download_files.py # 全量
|
||||
python scripts/step4_download_files.py --incremental # 增量
|
||||
```
|
||||
|
||||
使用 `dws drive download` 下载 HTML/MD/XLSX/PPTX/PDF 等附件。
|
||||
|
||||
**输出**:`output/downloaded-files/html-md/` 和 `output/downloaded-files/other/`
|
||||
|
||||
### Step 5: 生成结构化总结
|
||||
|
||||
```bash
|
||||
python scripts/step5_generate_summary.py # 全量
|
||||
python scripts/step5_generate_summary.py --since today # 今日
|
||||
python scripts/step5_generate_summary.py --brief # 简要
|
||||
```
|
||||
|
||||
按群组/人物/主题聚类,生成增量报告。
|
||||
|
||||
**输出**:`output/reports/latest_summary.md`
|
||||
|
||||
### Step 6: 更新知识库(可选)
|
||||
|
||||
```bash
|
||||
python scripts/step6_update_knowledge_base.py
|
||||
```
|
||||
|
||||
同步到 Obsidian 知识库 + 知识图谱。
|
||||
|
||||
## 配置文件
|
||||
|
||||
所有可定制项在 `config.yaml`:
|
||||
|
||||
```yaml
|
||||
groups:
|
||||
dc战略问题研究院: "cidoUneRB4Db8TAXaTrKxkQAw=="
|
||||
创新组: "cidMuM+itt5PeY7xNSWsv3M0g=="
|
||||
|
||||
collection:
|
||||
start_date: "2026-05-01 00:00:00"
|
||||
days_back: 3
|
||||
limit: 200
|
||||
|
||||
dws_path: auto # auto | /path/to/dws.exe
|
||||
|
||||
output_dir: "./output"
|
||||
data_dir: "./data"
|
||||
```
|
||||
|
||||
## Agent 调用约定
|
||||
|
||||
### 增量模式(日常)
|
||||
```bash
|
||||
python scripts/step1_collect_messages.py --days 1
|
||||
python scripts/step2_extract_links.py --incremental
|
||||
python scripts/step3_fetch_feishu_docs.py --incremental
|
||||
python scripts/step4_download_files.py --incremental
|
||||
python scripts/step5_generate_summary.py --since today
|
||||
```
|
||||
|
||||
### 全量模式(首次/重建)
|
||||
```bash
|
||||
python scripts/step1_collect_messages.py --full
|
||||
python scripts/step2_extract_links.py
|
||||
python scripts/step3_fetch_feishu_docs.py
|
||||
python scripts/step4_download_files.py
|
||||
python scripts/step5_generate_summary.py --full
|
||||
```
|
||||
|
||||
### 快速模式(只看今天)
|
||||
```bash
|
||||
python scripts/step1_collect_messages.py --days 1
|
||||
python scripts/step2_extract_links.py --incremental
|
||||
python scripts/step5_generate_summary.py --since today --brief
|
||||
```
|
||||
|
||||
## 输出产物
|
||||
|
||||
| 目录 | 内容 | 格式 |
|
||||
|------|------|------|
|
||||
| `data/raw-messages/` | 钉钉原始消息 | JSON |
|
||||
| `data/links/` | 链接索引 | JSON |
|
||||
| `output/feishu-docs/` | 飞书文档 | .md |
|
||||
| `output/downloaded-files/` | 文件附件 | 原始格式 |
|
||||
| `output/reports/` | 汇总报告 | .md |
|
||||
| `output/obsidian-vault/` | Obsidian 知识库 | .md |
|
||||
| `output/knowledge-graph/` | 知识图谱 | JSON + HTML |
|
||||
|
||||
## 常见问题
|
||||
|
||||
**dws 未登录**:`dws auth login` 扫码。悟空内置路径 `C:\Users\<user>\.real\.bin\dws\bin\dws.exe`
|
||||
|
||||
**lark-cli 过期**:`lark-cli auth login --domain docs,drive,wiki --recommend`
|
||||
|
||||
**wiki 链接解析**:脚本自动处理 wiki→docx 的节点ID解析
|
||||
|
||||
**消息分页**:脚本内置 openMessageId 去重 + 分段拉取
|
||||
|
||||
## 定时任务
|
||||
|
||||
```powershell
|
||||
# Windows 每天 18:00 增量采集
|
||||
schtasks /create /tn "DingTalkFeishuCollector" /tr "python <project>\scripts\step1_collect_messages.py --days 1" /sc daily /st 18:00
|
||||
```
|
||||
|
||||
## 技术栈
|
||||
|
||||
```
|
||||
钉钉群聊 → dws CLI (Go) → JSON消息
|
||||
↓
|
||||
Python 正则提取 → 链接索引
|
||||
↓
|
||||
lark-cli (Node) → 飞书文档内容
|
||||
↓
|
||||
Python 处理 → Markdown总结 + Obsidian库 + 知识图谱
|
||||
```
|
||||
|
||||
**总依赖**:dws CLI (14MB) + lark-cli (npm) + Python 3.10+ + beautifulsoup4
|
||||
**总成本**:0 元(需要飞书账号 + 钉钉群访问权限)
|
||||
@@ -0,0 +1,37 @@
|
||||
# 钉钉飞书采集器配置
|
||||
# 修改此文件即可适配其他群组/项目
|
||||
|
||||
groups:
|
||||
dc战略问题研究院: "cidoUneRB4Db8TAXaTrKxkQAw=="
|
||||
创新组: "cidMuM+itt5PeY7xNSWsv3M0g=="
|
||||
|
||||
collection:
|
||||
# 首次全量拉取的起始时间
|
||||
start_date: "2026-05-01 00:00:00"
|
||||
# 增量模式:拉取最近N天
|
||||
days_back: 3
|
||||
# 每次API返回上限
|
||||
limit: 200
|
||||
# 分段拉取的时间节点(用于全量拉取时的分页)
|
||||
segments:
|
||||
- "2026-05-01 00:00:00"
|
||||
- "2026-05-19 00:00:00"
|
||||
- "2026-05-24 00:00:00"
|
||||
|
||||
# dws CLI 路径查找优先级
|
||||
# auto: 环境变量 DWS_PATH > tools/dws.exe > 悟空内置 > PATH
|
||||
dws_path: auto
|
||||
|
||||
# lark-cli 路径(默认从 PATH 查找)
|
||||
lark_cli_path: auto
|
||||
|
||||
# 输出目录(相对于项目根目录)
|
||||
output_dir: "./output"
|
||||
data_dir: "./data"
|
||||
|
||||
# 飞书域名映射(可选,用于识别不同租户)
|
||||
feishu_domains:
|
||||
- "dianchukeji.feishu.cn"
|
||||
- "fcnlycv6dd0w.feishu.cn"
|
||||
- "ocnmca6f1o0p.feishu.cn"
|
||||
- "my.feishu.cn"
|
||||
@@ -0,0 +1,75 @@
|
||||
"""共享路径和配置工具"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
def find_project_root():
|
||||
"""向上查找项目根目录(包含 data/ 目录的最顶层)"""
|
||||
# 从当前脚本位置开始
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
# 向上查找,找到包含 data/ 目录的目录
|
||||
d = script_dir
|
||||
for _ in range(5): # 最多向上5级
|
||||
if os.path.isdir(os.path.join(d, "data")):
|
||||
return d
|
||||
parent = os.path.dirname(d)
|
||||
if parent == d:
|
||||
break
|
||||
d = parent
|
||||
|
||||
# 如果找不到,假设项目根在 scripts/ 的上一级
|
||||
return os.path.dirname(os.path.dirname(script_dir))
|
||||
|
||||
def find_dws():
|
||||
"""查找 dws CLI"""
|
||||
import shutil
|
||||
|
||||
# 1. 环境变量
|
||||
env = os.environ.get("DWS_PATH")
|
||||
if env and os.path.isfile(env):
|
||||
return env
|
||||
|
||||
# 2. 项目 tools/ 目录
|
||||
root = find_project_root()
|
||||
local = os.path.join(root, "tools", "dws.exe")
|
||||
if os.path.isfile(local):
|
||||
return local
|
||||
|
||||
# 3. 悟空内置
|
||||
candidates = [
|
||||
os.path.expanduser(r"~\.real\.bin\dws\bin\dws.exe"),
|
||||
r"C:\Program Files\Wukong\0.9.51-26052503\bin\dws.exe",
|
||||
]
|
||||
for p in candidates:
|
||||
if os.path.isfile(p):
|
||||
return p
|
||||
|
||||
# 4. PATH
|
||||
found = shutil.which("dws")
|
||||
if found:
|
||||
return found
|
||||
|
||||
print("ERROR: 找不到 dws CLI,请设置 DWS_PATH 环境变量或安装悟空", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
def find_lark_cli():
|
||||
"""查找 lark-cli"""
|
||||
import shutil
|
||||
found = shutil.which("lark-cli")
|
||||
if found:
|
||||
return found
|
||||
print("ERROR: 找不到 lark-cli,请运行 npm install -g @larksuite/cli", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# 常用路径
|
||||
PROJECT_ROOT = find_project_root()
|
||||
DATA_DIR = os.path.join(PROJECT_ROOT, "data")
|
||||
RAW_DIR = os.path.join(DATA_DIR, "raw-messages")
|
||||
LINKS_DIR = os.path.join(DATA_DIR, "links")
|
||||
OUTPUT_DIR = os.path.join(PROJECT_ROOT, "output")
|
||||
DOCS_DIR = os.path.join(OUTPUT_DIR, "feishu-docs")
|
||||
REPORTS_DIR = os.path.join(OUTPUT_DIR, "reports")
|
||||
DOWNLOAD_DIR = os.path.join(OUTPUT_DIR, "downloaded-files")
|
||||
OBSIDIAN_DIR = os.path.join(OUTPUT_DIR, "obsidian-vault")
|
||||
KG_DIR = os.path.join(OUTPUT_DIR, "knowledge-graph")
|
||||
@@ -0,0 +1,60 @@
|
||||
"""一键执行完整采集流程
|
||||
|
||||
用法:
|
||||
python run_all.py # 增量(最近3天)
|
||||
python run_all.py --days 1 # 最近1天
|
||||
python run_all.py --full # 全量
|
||||
python run_all.py --quick # 只看今天
|
||||
python run_all.py --skip-download # 跳过文件下载
|
||||
python run_all.py --skip-kb # 跳过知识库更新
|
||||
"""
|
||||
|
||||
import argparse, subprocess, sys, os, time
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
def run(script, args=None):
|
||||
cmd = [sys.executable, os.path.join(SCRIPT_DIR, script)] + (args or [])
|
||||
print(f"\n{'='*50}\n {script} {' '.join(args or [])}\n{'='*50}")
|
||||
t = time.time()
|
||||
r = subprocess.run(cmd)
|
||||
print(f" {'OK' if r.returncode==0 else 'WARN'} ({time.time()-t:.1f}s)")
|
||||
return r.returncode
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--full",action="store_true"); p.add_argument("--days",type=int)
|
||||
p.add_argument("--since",type=str); p.add_argument("--quick",action="store_true")
|
||||
p.add_argument("--skip-download",action="store_true"); p.add_argument("--skip-kb",action="store_true")
|
||||
args = p.parse_args()
|
||||
t0 = time.time()
|
||||
|
||||
s1 = []
|
||||
if args.full: s1.append("--full")
|
||||
elif args.since: s1 += ["--since", args.since]
|
||||
elif args.days: s1 += ["--days", str(args.days)]
|
||||
elif args.quick: s1 += ["--days", "1"]
|
||||
run("step1_collect_messages.py", s1)
|
||||
|
||||
s2 = [] if args.full else ["--incremental"]
|
||||
run("step2_extract_links.py", s2)
|
||||
|
||||
s3 = [] if args.full else ["--incremental"]
|
||||
run("step3_fetch_feishu_docs.py", s3)
|
||||
|
||||
if not args.skip_download:
|
||||
s4 = [] if args.full else ["--incremental"]
|
||||
run("step4_download_files.py", s4)
|
||||
|
||||
s5 = []
|
||||
if args.quick: s5 += ["--since", "today", "--brief"]
|
||||
elif args.since: s5 += ["--since", args.since]
|
||||
elif args.days: s5 += ["--days", str(args.days)]
|
||||
run("step5_generate_summary.py", s5)
|
||||
|
||||
if not args.skip_kb:
|
||||
run("step6_update_knowledge_base.py")
|
||||
|
||||
print(f"\n{'='*50}\n 全部完成! ({time.time()-t0:.1f}s)\n{'='*50}")
|
||||
|
||||
if __name__ == "__main__": main()
|
||||
@@ -0,0 +1,103 @@
|
||||
"""Step 1: 从钉钉群拉取消息
|
||||
|
||||
用法:
|
||||
python step1_collect_messages.py # 增量(最近3天)
|
||||
python step1_collect_messages.py --days 1 # 增量(最近1天)
|
||||
python step1_collect_messages.py --full # 全量拉取
|
||||
python step1_collect_messages.py --since "2026-06-05 00:00:00"
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from paths import DATA_DIR, RAW_DIR, find_dws
|
||||
|
||||
DWS = find_dws()
|
||||
|
||||
def fetch_messages(group_id, time_str, forward="true", limit=200):
|
||||
cmd = [DWS, "chat", "message", "list", "--group", group_id,
|
||||
"--time", time_str, "--forward", forward, "--limit", str(limit), "--format", "json"]
|
||||
try:
|
||||
result = subprocess.run(cmd, capture_output=True, timeout=60)
|
||||
output = result.stdout.decode("utf-8", errors="replace")
|
||||
json_start = output.find("{")
|
||||
if json_start < 0: return []
|
||||
data = json.loads(output[json_start:])
|
||||
return data.get("result", {}).get("messages", [])
|
||||
except Exception as e:
|
||||
print(f" WARN: {e}", file=sys.stderr)
|
||||
return []
|
||||
|
||||
def collect_group(group_name, group_id, since):
|
||||
print(f"\n=== {group_name} (since {since}) ===")
|
||||
all_msgs = {}
|
||||
msgs = fetch_messages(group_id, since, "true")
|
||||
for m in msgs: all_msgs[m["openMessageId"]] = m
|
||||
print(f" 正向: {len(msgs)} 条")
|
||||
if len(msgs) >= 180:
|
||||
times = sorted([m["createTime"] for m in msgs])
|
||||
if times:
|
||||
msgs2 = fetch_messages(group_id, times[len(times)//2], "false")
|
||||
for m in msgs2: all_msgs[m["openMessageId"]] = m
|
||||
print(f" 反向补充: {len(msgs2)} 条")
|
||||
result = list(all_msgs.values())
|
||||
if result:
|
||||
times = sorted([m["createTime"] for m in result])
|
||||
print(f" 总计: {len(result)} 条 ({times[0]} ~ {times[-1]})")
|
||||
return result
|
||||
|
||||
GROUPS = {
|
||||
"dc战略问题研究院": "cidoUneRB4Db8TAXaTrKxkQAw==",
|
||||
"创新组": "cidMuM+itt5PeY7xNSWsv3M0g==",
|
||||
}
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--days", type=int)
|
||||
parser.add_argument("--since", type=str)
|
||||
parser.add_argument("--full", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.since:
|
||||
since = args.since
|
||||
elif args.full:
|
||||
since = "2026-05-01 00:00:00"
|
||||
elif args.days:
|
||||
since = (datetime.now() - timedelta(days=args.days)).strftime("%Y-%m-%d 00:00:00")
|
||||
else:
|
||||
since = (datetime.now() - timedelta(days=3)).strftime("%Y-%m-%d 00:00:00")
|
||||
|
||||
print(f"dws: {DWS}\n时间: {since} ~ now")
|
||||
|
||||
os.makedirs(RAW_DIR, exist_ok=True)
|
||||
combined_path = os.path.join(RAW_DIR, "all_messages_combined.json")
|
||||
existing = {}
|
||||
if os.path.isfile(combined_path):
|
||||
with open(combined_path, "r", encoding="utf-8") as f:
|
||||
existing = json.load(f)
|
||||
|
||||
new_count = 0
|
||||
for name, gid in GROUPS.items():
|
||||
new_msgs = collect_group(name, gid, since)
|
||||
if name not in existing: existing[name] = []
|
||||
existing_ids = {m["openMessageId"] for m in existing[name]}
|
||||
added = sum(1 for m in new_msgs if m["openMessageId"] not in existing_ids)
|
||||
for m in new_msgs:
|
||||
if m["openMessageId"] not in existing_ids:
|
||||
existing[name].append(m)
|
||||
existing_ids.add(m["openMessageId"])
|
||||
print(f" 新增: {added} 条 (总计: {len(existing[name])})")
|
||||
new_count += added
|
||||
|
||||
with open(combined_path, "w", encoding="utf-8") as f:
|
||||
json.dump(existing, f, ensure_ascii=False, indent=2)
|
||||
total = sum(len(v) for v in existing.values())
|
||||
print(f"\n完成! 新增 {new_count} 条,总计 {total} 条 -> {combined_path}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Step 2: 从消息中提取飞书链接和文件附件"""
|
||||
import argparse, json, os, re, sys
|
||||
from datetime import datetime
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from paths import RAW_DIR, LINKS_DIR
|
||||
|
||||
FEISHU_RE = re.compile(r'https?://[a-zA-Z0-9.-]+\.feishu\.cn/(?:wiki|docx)/[A-Za-z0-9]+')
|
||||
FILE_RE = re.compile(r'\[文件\]\s+(.+?)\s+fileId:\s+(\S+)')
|
||||
KIMI_RE = re.compile(r'https?://[a-zA-Z0-9.-]+\.ok\.kimi\.link/\S*')
|
||||
|
||||
def extract(messages):
|
||||
feishu, files, kimi = [], [], []
|
||||
seen_f, seen_files = set(), set()
|
||||
for grp, msgs in messages.items():
|
||||
for m in msgs:
|
||||
c, s, t = m.get("content",""), m.get("sender","?"), m.get("createTime","")
|
||||
for url in FEISHU_RE.findall(c):
|
||||
if url not in seen_f:
|
||||
seen_f.add(url)
|
||||
dm = re.search(r'feishu\.cn/(?:wiki|docx)/([A-Za-z0-9]+)', url)
|
||||
feishu.append({"url":url,"doc_id":dm.group(1) if dm else "","type":"wiki" if "/wiki/" in url else "docx","sender":s,"time":t,"group":grp})
|
||||
for name, fid in FILE_RE.findall(c):
|
||||
if fid not in seen_files:
|
||||
seen_files.add(fid)
|
||||
files.append({"name":name.strip(),"fileId":fid,"sender":s,"time":t,"group":grp})
|
||||
for url in KIMI_RE.findall(c):
|
||||
kimi.append({"url":url,"desc":c.split("https")[0].strip()[:80],"sender":s,"group":grp})
|
||||
return {"feishu_links":feishu,"file_attachments":files,"kimi_links":kimi}
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(); p.add_argument("--incremental",action="store_true"); args = p.parse_args()
|
||||
path = os.path.join(RAW_DIR, "all_messages_combined.json")
|
||||
if not os.path.isfile(path): print(f"ERROR: {path} not found"); sys.exit(1)
|
||||
with open(path,"r",encoding="utf-8") as f: messages = json.load(f)
|
||||
print(f"消息: {sum(len(v) for v in messages.values())} 条")
|
||||
r = extract(messages)
|
||||
print(f"飞书: {len(r['feishu_links'])}, 附件: {len(r['file_attachments'])}, Kimi: {len(r['kimi_links'])}")
|
||||
for grp in messages:
|
||||
gl = [l for l in r['feishu_links'] if l['group']==grp]
|
||||
gf = [f for f in r['file_attachments'] if f['group']==grp]
|
||||
print(f" [{grp}] 飞书:{len(gl)} 附件:{len(gf)}")
|
||||
os.makedirs(LINKS_DIR, exist_ok=True)
|
||||
with open(os.path.join(LINKS_DIR,"all_feishu_links.json"),"w",encoding="utf-8") as f:
|
||||
json.dump({"date":datetime.now().strftime("%Y-%m-%d"),"total":len(r['feishu_links']),"links":r['feishu_links']},f,ensure_ascii=False,indent=2)
|
||||
with open(os.path.join(LINKS_DIR,"all_file_attachments.json"),"w",encoding="utf-8") as f:
|
||||
json.dump(r['file_attachments'],f,ensure_ascii=False,indent=2)
|
||||
with open(os.path.join(LINKS_DIR,"all_kimi_links.json"),"w",encoding="utf-8") as f:
|
||||
json.dump(r['kimi_links'],f,ensure_ascii=False,indent=2)
|
||||
print(f"保存至: {LINKS_DIR}")
|
||||
|
||||
if __name__ == "__main__": main()
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Step 3: 拉取飞书文档内容"""
|
||||
import argparse, json, os, subprocess, sys, re
|
||||
from datetime import datetime
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from paths import LINKS_DIR, DOCS_DIR, find_lark_cli
|
||||
|
||||
def resolve_wiki(node_id, lark):
|
||||
try:
|
||||
r = subprocess.run([lark,"api","GET",f"/open-apis/wiki/v2/spaces/get_node?token={node_id}","--format","json"],capture_output=True,text=True,timeout=15)
|
||||
if r.returncode==0: return json.loads(r.stdout).get("data",{}).get("node",{}).get("obj_token","")
|
||||
except: pass
|
||||
return ""
|
||||
|
||||
def extract_text(obj):
|
||||
if not obj: return ""
|
||||
return "".join(e.get("text_run",{}).get("content","") for e in obj.get("elements",[])).strip()
|
||||
|
||||
def blocks_to_md(blocks, doc_id):
|
||||
lines = [f"# {doc_id}\n", f"Source: https://feishu.cn/docx/{doc_id}\n"]
|
||||
for b in blocks:
|
||||
bt = b.get("block_type",0)
|
||||
if bt==2: t=extract_text(b.get("text",{})); [lines.append(t)] if t else None
|
||||
elif bt in range(3,10): t=extract_text(b.get(f"heading{bt-2}",{}) or b.get("text",{})); lines.append(f"{'#'*(bt-2)} {t}") if t else None
|
||||
elif bt==10: t=extract_text(b.get("bullet",{}) or b.get("text",{})); lines.append(f"- {t}") if t else None
|
||||
elif bt==11: t=extract_text(b.get("ordered",{}) or b.get("text",{})); lines.append(f"1. {t}") if t else None
|
||||
elif bt==14: t=extract_text(b.get("quote",{}) or b.get("text",{})); lines.append(f"> {t}") if t else None
|
||||
elif bt==15: t=extract_text(b.get("code",{}) or b.get("text",{})); lines.append(f"```\n{t}\n```") if t else None
|
||||
else:
|
||||
for k in ["text","heading1","heading2","heading3","bullet","ordered","quote","code"]:
|
||||
if k in b: t=extract_text(b[k]); [lines.append(t)] if t else None; break
|
||||
return "\n\n".join(lines)
|
||||
|
||||
def fetch_doc(doc_id, doc_type, lark):
|
||||
real_id = resolve_wiki(doc_id, lark) if doc_type=="wiki" else doc_id
|
||||
if not real_id: return f"# {doc_id}\n\n> wiki解析失败"
|
||||
try:
|
||||
r = subprocess.run([lark,"api","GET",f"/open-apis/docx/v1/documents/{real_id}/blocks","--format","json"],capture_output=True,text=True,timeout=30)
|
||||
if r.returncode!=0: return f"# {doc_id}\n\n> 获取失败"
|
||||
blocks = json.loads(r.stdout).get("data",{}).get("items",[])
|
||||
return blocks_to_md(blocks, doc_id) if blocks else f"# {doc_id}\n\n> 空文档"
|
||||
except Exception as e: return f"# {doc_id}\n\n> {e}"
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(); p.add_argument("--incremental",action="store_true"); args = p.parse_args()
|
||||
links_path = os.path.join(LINKS_DIR,"all_feishu_links.json")
|
||||
if not os.path.isfile(links_path): print("ERROR: 先运行 step2"); sys.exit(1)
|
||||
with open(links_path,"r",encoding="utf-8") as f: all_links = json.load(f).get("links",[])
|
||||
print(f"链接: {len(all_links)} 个")
|
||||
|
||||
content_path = os.path.join(LINKS_DIR,"all_feishu_content.json")
|
||||
existing = {}
|
||||
if os.path.isfile(content_path):
|
||||
with open(content_path,"r",encoding="utf-8") as f:
|
||||
for d in json.load(f).get("documents",[]): existing[d.get("doc_id","")] = d
|
||||
print(f"已拉取: {len(existing)}")
|
||||
|
||||
lark = find_lark_cli()
|
||||
os.makedirs(DOCS_DIR, exist_ok=True)
|
||||
new_count = 0
|
||||
for link in all_links:
|
||||
did = link.get("doc_id","")
|
||||
if not did or did in existing: continue
|
||||
print(f" {did} ({link.get('sender','?')})")
|
||||
md = fetch_doc(did, link.get("type","docx"), lark)
|
||||
with open(os.path.join(DOCS_DIR,f"{did}.md"),"w",encoding="utf-8") as f: f.write(md)
|
||||
title = ""
|
||||
for line in md.split("\n"):
|
||||
if line.startswith("# "): title = line[2:].strip(); break
|
||||
existing[did] = {"url":link.get("url",""),"doc_id":did,"title":title,"sender":link.get("sender",""),"group":link.get("group",""),"time":link.get("time",""),"content_length":len(md),"fetched":True}
|
||||
new_count += 1
|
||||
|
||||
with open(content_path,"w",encoding="utf-8") as f:
|
||||
json.dump({"date":datetime.now().strftime("%Y-%m-%d"),"total":len(existing),"documents":list(existing.values())},f,ensure_ascii=False,indent=2)
|
||||
print(f"\n新增 {new_count},总计 {len(existing)} -> {DOCS_DIR}")
|
||||
|
||||
if __name__ == "__main__": main()
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Step 4: 下载钉钉群文件附件"""
|
||||
import argparse, json, os, subprocess, sys
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from paths import LINKS_DIR, DOWNLOAD_DIR, find_dws
|
||||
|
||||
HTML_MD = {".html",".htm",".md",".markdown",".txt"}
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(); p.add_argument("--incremental",action="store_true"); args = p.parse_args()
|
||||
path = os.path.join(LINKS_DIR,"all_file_attachments.json")
|
||||
if not os.path.isfile(path): print("ERROR: 先运行 step2"); sys.exit(1)
|
||||
with open(path,"r",encoding="utf-8") as f: files = json.load(f)
|
||||
print(f"附件: {len(files)} 个")
|
||||
dws = find_dws()
|
||||
new = 0
|
||||
for fi in files:
|
||||
fid, name = fi.get("fileId",""), fi.get("name","unknown")
|
||||
if not fid: continue
|
||||
ext = os.path.splitext(name)[1].lower()
|
||||
subdir = "html-md" if ext in HTML_MD else "other"
|
||||
out = os.path.join(DOWNLOAD_DIR, subdir)
|
||||
os.makedirs(out, exist_ok=True)
|
||||
print(f" {name} -> {subdir}/")
|
||||
try:
|
||||
r = subprocess.run([dws,"drive","download","--node",fid,"--output",out],capture_output=True,timeout=120)
|
||||
if r.returncode==0: new+=1
|
||||
else: print(f" FAIL")
|
||||
except Exception as e: print(f" ERR: {e}")
|
||||
print(f"\n下载 {new} 个 -> {DOWNLOAD_DIR}")
|
||||
|
||||
if __name__ == "__main__": main()
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Step 5: 生成结构化总结"""
|
||||
import argparse, json, os, sys
|
||||
from datetime import datetime, timedelta
|
||||
from collections import defaultdict
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from paths import DATA_DIR, LINKS_DIR, DOCS_DIR, REPORTS_DIR
|
||||
|
||||
def load_json(path):
|
||||
if not os.path.isfile(path): return None
|
||||
with open(path,"r",encoding="utf-8") as f: return json.load(f)
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--since",type=str); p.add_argument("--days",type=int)
|
||||
p.add_argument("--full",action="store_true"); p.add_argument("--brief",action="store_true")
|
||||
args = p.parse_args()
|
||||
|
||||
messages = load_json(os.path.join(DATA_DIR,"raw-messages","all_messages_combined.json")) or {}
|
||||
content = load_json(os.path.join(LINKS_DIR,"all_feishu_content.json")) or {}
|
||||
files = load_json(os.path.join(LINKS_DIR,"all_file_attachments.json")) or []
|
||||
docs = {}
|
||||
for d in content.get("documents",[]):
|
||||
did = d.get("doc_id","") or d.get("url","")
|
||||
if did: docs[did] = d
|
||||
|
||||
if args.since:
|
||||
since = datetime.now().strftime("%Y-%m-%d 00:00:00") if args.since=="today" else (datetime.now()-timedelta(days=1)).strftime("%Y-%m-%d 00:00:00") if args.since=="yesterday" else args.since
|
||||
messages = {g:[m for m in msgs if m.get("createTime","")>=since] for g,msgs in messages.items()}
|
||||
elif args.days:
|
||||
since = (datetime.now()-timedelta(days=args.days)).strftime("%Y-%m-%d 00:00:00")
|
||||
messages = {g:[m for m in msgs if m.get("createTime","")>=since] for g,msgs in messages.items()}
|
||||
|
||||
lines = [f"# 钉钉群聊飞书文档采集报告", f"> 生成: {datetime.now().strftime('%Y-%m-%d %H:%M')}", ""]
|
||||
total_m = sum(len(v) for v in messages.values())
|
||||
lines += ["## 数据总览","","| 指标 | 数量 |","|------|------|",
|
||||
f"| 消息 | {total_m} |", f"| 文档 | {len(docs)} |", f"| 附件 | {len(files)} |", ""]
|
||||
|
||||
for grp, msgs in messages.items():
|
||||
if not msgs: continue
|
||||
times = sorted([m.get("createTime","") for m in msgs])
|
||||
gd = [d for d in docs.values() if d.get("group")==grp]
|
||||
gf = [f for f in files if f.get("group")==grp]
|
||||
lines += [f"### {grp}", f"- 消息: {len(msgs)} ({times[0]} ~ {times[-1]})", f"- 文档: {len(gd)}", f"- 附件: {len(gf)}", ""]
|
||||
if args.brief: continue
|
||||
senders = defaultdict(lambda: {"m":0,"d":0})
|
||||
for m in msgs: senders[m.get("sender","?")]["m"]+=1
|
||||
for d in gd: senders[d.get("sender","?")]["d"]+=1
|
||||
lines += ["| 发送者 | 消息 | 文档 |","|--------|------|------|"]
|
||||
for s,st in sorted(senders.items(),key=lambda x:-(x[1]["m"]+x[1]["d"])):
|
||||
lines.append(f"| {s} | {st['m']} | {st['d']} |")
|
||||
lines.append("")
|
||||
lines.append("**最新消息:**")
|
||||
for m in sorted(msgs,key=lambda x:x.get("createTime",""),reverse=True)[:10]:
|
||||
c = m.get('content','')[:80].replace('\n',' ')
|
||||
lines.append(f"- [{m.get('createTime','')}] **{m.get('sender','?')}**: {c}")
|
||||
lines.append("")
|
||||
|
||||
os.makedirs(REPORTS_DIR, exist_ok=True)
|
||||
report = "\n".join(lines)
|
||||
for name in ["latest_summary.md", f"summary_{datetime.now().strftime('%Y%m%d')}.md"]:
|
||||
with open(os.path.join(REPORTS_DIR,name),"w",encoding="utf-8") as f: f.write(report)
|
||||
print(f"报告已生成: {REPORTS_DIR}")
|
||||
|
||||
if __name__ == "__main__": main()
|
||||
@@ -0,0 +1,79 @@
|
||||
"""Step 6: 更新知识库(Obsidian + 知识图谱)"""
|
||||
import json, os, sys
|
||||
from datetime import datetime
|
||||
from collections import defaultdict
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from paths import LINKS_DIR, DATA_DIR, DOCS_DIR, OBSIDIAN_DIR, KG_DIR
|
||||
|
||||
def safe_name(n):
|
||||
for c in '<>:"/\\|?*': n=n.replace(c,'_')
|
||||
return n[:80]
|
||||
|
||||
def w(path, content):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path,"w",encoding="utf-8") as f: f.write(content)
|
||||
|
||||
def load_json(p):
|
||||
if not os.path.isfile(p): return None
|
||||
with open(p,"r",encoding="utf-8") as f: return json.load(f)
|
||||
|
||||
def main():
|
||||
content = load_json(os.path.join(LINKS_DIR,"all_feishu_content.json")) or {}
|
||||
messages = load_json(os.path.join(DATA_DIR,"raw-messages","all_messages_combined.json")) or {}
|
||||
docs = content.get("documents",[])
|
||||
|
||||
print(f"文档: {len(docs)}, 消息: {sum(len(v) for v in messages.values())}")
|
||||
|
||||
# Obsidian
|
||||
by_group = defaultdict(list)
|
||||
for d in docs: by_group[d.get("group","未分组")].append(d)
|
||||
by_sender = defaultdict(list)
|
||||
for d in docs: by_sender[d.get("sender","?")].append(d)
|
||||
|
||||
moc = ["---","tags: [MOC, 飞书文档]",f"created: {datetime.now().strftime('%Y-%m-%d')}","---","","# 飞书文档知识库","",f"> 更新: {datetime.now().strftime('%Y-%m-%d %H:%M')}",""]
|
||||
for grp, gd in sorted(by_group.items()):
|
||||
moc.append(f"## {grp}")
|
||||
for d in sorted(gd,key=lambda x:x.get("time",""),reverse=True):
|
||||
t = d.get("title","") or d.get("doc_id","")
|
||||
moc.append(f"- [[{safe_name(t)}]] ({d.get('sender','?')}, {d.get('time','')})")
|
||||
moc.append("")
|
||||
w(os.path.join(OBSIDIAN_DIR,"00-MOC","飞书文档索引.md"),"\n".join(moc))
|
||||
|
||||
for sender, sd in sorted(by_sender.items()):
|
||||
lines = ["---",f"tags: [人物, {sender}]","---",f"# {sender}",f"\n贡献: {len(sd)} 篇\n"]
|
||||
for d in sorted(sd,key=lambda x:x.get("time",""),reverse=True):
|
||||
t = d.get("title","") or d.get("doc_id","")
|
||||
lines.append(f"- [[{safe_name(t)}]] ({d.get('time','')})")
|
||||
w(os.path.join(OBSIDIAN_DIR,"04-人物",f"{safe_name(sender)}.md"),"\n".join(lines))
|
||||
|
||||
for d in docs:
|
||||
did = d.get("doc_id","")
|
||||
fpath = os.path.join(DOCS_DIR,f"{did}.md")
|
||||
if not os.path.isfile(fpath): continue
|
||||
with open(fpath,"r",encoding="utf-8") as f: c = f.read()
|
||||
t = d.get("title","") or did
|
||||
fm = ["---",f"tags: [飞书文档, {d.get('group','')}]",f"sender: {d.get('sender','')}",f"date: {d.get('time','')}",f"source: {d.get('url','')}","---",""]
|
||||
w(os.path.join(OBSIDIAN_DIR,"01-产品研究",f"{safe_name(t)}.md"),"\n".join(fm)+c)
|
||||
print(f" Obsidian: {len(docs)} 文档, {len(by_sender)} 人物")
|
||||
|
||||
# 知识图谱
|
||||
nodes, edges, nids = [], [], set()
|
||||
for s in {d.get("sender","") for d in docs} | {m.get("sender","") for msgs in messages.values() for m in msgs}:
|
||||
if s: nid=f"person:{s}"; nodes.append({"id":nid,"type":"person","label":s}); nids.add(nid)
|
||||
for d in docs:
|
||||
did,t = d.get("doc_id",""), d.get("title","") or d.get("doc_id","")
|
||||
nid = f"doc:{did}"
|
||||
if nid not in nids: nodes.append({"id":nid,"type":"document","label":t[:50]}); nids.add(nid)
|
||||
s = d.get("sender","")
|
||||
if s: edges.append({"source":f"person:{s}","target":nid,"type":"authored"})
|
||||
g = d.get("group","")
|
||||
if g:
|
||||
gid=f"group:{g}"
|
||||
if gid not in nids: nodes.append({"id":gid,"type":"group","label":g}); nids.add(gid)
|
||||
edges.append({"source":gid,"target":nid,"type":"contains"})
|
||||
os.makedirs(KG_DIR, exist_ok=True)
|
||||
with open(os.path.join(KG_DIR,"knowledge_graph.json"),"w",encoding="utf-8") as f:
|
||||
json.dump({"date":datetime.now().strftime("%Y-%m-%d"),"nodes":nodes,"edges":edges},f,ensure_ascii=False,indent=2)
|
||||
print(f" 图谱: {len(nodes)} 节点, {len(edges)} 关系")
|
||||
|
||||
if __name__ == "__main__": main()
|
||||
Reference in New Issue
Block a user