From 6b4189291c3cf9b596ba874a86686d74011283c8 Mon Sep 17 00:00:00 2001 From: Caihaohan Date: Sat, 10 Jan 2026 11:50:33 +0800 Subject: [PATCH] =?UTF-8?q?chore:=20=E6=B8=85=E7=90=86=E9=87=8D=E6=9E=84?= =?UTF-8?q?=E5=90=8E=E7=9A=84=E6=97=A7=E6=9E=B6=E6=9E=84=E6=AE=8B=E7=95=99?= =?UTF-8?q?=E6=96=87=E4=BB=B6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除已废弃的 agent 定义、命令配置和脚本文件,这些文件在之前的多源数据入库架构重构后已不再使用。 Co-Authored-By: Claude --- .claude/agents/database-ingestor.md | 247 --------------- .claude/agents/deduplicator.md | 221 ------------- .claude/agents/project-analyzer.md | 290 ------------------ .claude/agents/scrapers/github-trending.md | 176 ----------- .../agents/scrapers/huggingface-trending.md | 167 ---------- .claude/agents/scrapers/papers-with-code.md | 165 ---------- .claude/commands/add-trending.md | 283 ----------------- scripts/add-markdown-project.js | 181 ----------- scripts/add-project.js | 59 ---- scripts/scrape-huggingface.js | 113 ------- 10 files changed, 1902 deletions(-) delete mode 100644 .claude/agents/database-ingestor.md delete mode 100644 .claude/agents/deduplicator.md delete mode 100644 .claude/agents/project-analyzer.md delete mode 100644 .claude/agents/scrapers/github-trending.md delete mode 100644 .claude/agents/scrapers/huggingface-trending.md delete mode 100644 .claude/agents/scrapers/papers-with-code.md delete mode 100644 .claude/commands/add-trending.md delete mode 100644 scripts/add-markdown-project.js delete mode 100644 scripts/add-project.js delete mode 100644 scripts/scrape-huggingface.js diff --git a/.claude/agents/database-ingestor.md b/.claude/agents/database-ingestor.md deleted file mode 100644 index da054a4..0000000 --- a/.claude/agents/database-ingestor.md +++ /dev/null @@ -1,247 +0,0 @@ ---- -name: database-ingestor -description: | - 将分析后的项目批量入库到数据库。读取工作区的 analyzed-projects.json,验证数据格式,调用 Webhook API,处理响应并生成报告。使用此 agent 当需要将项目数据保存到数据库时。 - - 示例场景: - - 项目分析完成后,需要批量入库 - - 从外部数据源获取项目数据后需要保存 - - 输入参数:{"workspace": ".trending-workspace/..."} -model: inherit -color: green ---- - -# 批量入库器 Agent - -## 职责 - -将分析后的项目批量入库到数据库: -1. 读取分析后的项目数据 -2. 构造符合 ProjectInputSchema 的请求 -3. 调用 Webhook API -4. 处理响应并生成报告 - -## 输入参数 - -```json -{ - "workspace": ".trending-workspace/..." -} -``` - -## 执行步骤 - -### Step 1: 读取分析数据 - -从工作区读取 `analyzed-projects.json`。 - -### Step 2: 验证数据格式 - -确保每个项目符合 ProjectInputSchema: - -```typescript -interface ProjectInput { - name: string // 必填, 1-200 字符 - nameEn?: string // 可选, 最大 200 字符 - description: string // 必填, 10-500 字符 - descriptionEn?: string // 可选, 最大 500 字符 - content?: string // 可选, 最大 10000 字符 - contentEn?: string // 可选, 最大 10000 字符 - status: 'ACTIVE' | 'ARCHIVED' - source?: string // 可选, 最大 100 字符 - tags: Tag[] // 必填, 1-10 个 - links: ExternalLink[] // 必填, 1-10 个 -} - -interface Tag { - name: string // 必填, 1-50 字符 - nameEn?: string // 可选, 最大 50 字符 -} - -interface ExternalLink { - type: 'WEBSITE' | 'GITHUB' | 'HUGGINGFACE' | 'PAPER' - url: string // 必填, 有效 URL - title?: string // 可选, 最大 200 字符 -} -``` - -### Step 3: 构造批量请求 - -**API Key**(已配置):`sk_live_agent_park_webhook_key_2025` - -**重要**:必须将请求数据写入 `ingest-request.json` 文件,确保 UTF-8 编码正确。 - -请求格式: -```json -{ - "apiKey": "sk_live_agent_park_webhook_key_2025", - "projects": [ - { - "name": "LangChain", - "nameEn": "LangChain", - "description": "通过组合性构建大型语言模型应用程序的框架...", - "descriptionEn": "Building applications with LLMs through composability", - "content": "README 完整内容...", - "contentEn": "README full content...", - "status": "ACTIVE", - "source": "GITHUB_TRENDING", - "tags": [ - { "name": "LLM", "nameEn": "Large Language Model" }, - { "name": "Python", "nameEn": "Python" } - ], - "links": [ - { - "type": "GITHUB", - "url": "https://github.com/langchain-ai/langchain", - "title": "GitHub 仓库" - }, - { - "type": "WEBSITE", - "url": "https://python.langchain.com", - "title": "官方文档" - } - ] - } - ] -} -``` - -将上述内容写入工作区的 `ingest-request.json` 文件。 - -### Step 4: 调用 Webhook API - -**API Key**:`sk_live_agent_park_webhook_key_2025` - -**API 端点**:`http://localhost:3000/api/webhook/projects` - -**重要**:使用 JSON 文件而非命令行参数,避免编码问题。 - -1. **写入请求文件** `ingest-request.json`: -```bash -cat > ingest-request.json << 'EOF' -{ - "apiKey": "sk_live_agent_park_webhook_key_2025", - "projects": [ - { - "name": "LangChain", - "nameEn": "LangChain", - "description": "通过组合性构建大型语言模型应用程序的框架...", - "descriptionEn": "Building applications with LLMs through composability", - "status": "ACTIVE", - "source": "GITHUB_TRENDING", - "tags": [ - { "name": "LLM", "nameEn": "Large Language Model" }, - { "name": "Python", "nameEn": "Python" } - ], - "links": [ - { - "type": "GITHUB", - "url": "https://github.com/langchain-ai/langchain", - "title": "GitHub 仓库" - } - ] - } - ] -} -EOF -``` - -2. **发送请求**(使用 `@` 符号读取文件): -```bash -curl -X POST http://localhost:3000/api/webhook/projects \ - -H "Content-Type: application/json; charset=utf-8" \ - -d @ingest-request.json \ - -o ingestion-response.json -``` - -### Step 5: 处理响应 - -API 响应格式: - -```json -{ - "success": true, - "processed": 32, - "created": 30, - "updated": 2, - "failed": 0, - "errors": [] -} -``` - -如果有失败项目,errors 数组包含详细信息: - -```json -{ - "success": true, - "processed": 32, - "created": 30, - "updated": 1, - "failed": 1, - "errors": [ - { - "index": 15, - "field": "description", - "message": "Description must be at least 10 characters", - "value": { /* 项目数据 */ } - } - ] -} -``` - -### Step 6: 生成入库报告 - -输出 `ingestion-result.json`: - -```json -{ - "metadata": { - "ingestedAt": "2025-01-04T12:30:00Z", - "success": true - }, - "results": { - "processed": 32, - "created": 30, - "updated": 2, - "failed": 0, - "errors": [] - }, - "projects": [ - { - "index": 0, - "name": "LangChain", - "status": "created", - "projectId": "cm2x8k9d10001" - } - ] -} -``` - -## 配置 - -**API Key**(已配置):`sk_live_agent_park_webhook_key_2025` - -**API 端点**:`http://localhost:3000/api/webhook/projects` - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| analyzed-projects.json 不存在 | 错误提示 "请先运行项目分析器" | -| API 请求失败 | 记录详细错误,保存请求体到错误文件 | -| 部分项目失败 | 继续处理其他项目,记录失败项 | -| 开发服务器未启动 | 错误提示 "请先启动开发服务器: pnpm dev" | - -## 批量大小限制 - -- 单次请求最多 100 个项目 -- 如果超过 100 个,分批处理 - -## 输出 - -成功后,返回: -- 处理项目数量 -- 创建/更新/失败的数量 -- 失败项目详情(如有) -- ingestion-result.json 路径 diff --git a/.claude/agents/deduplicator.md b/.claude/agents/deduplicator.md deleted file mode 100644 index 6460cd5..0000000 --- a/.claude/agents/deduplicator.md +++ /dev/null @@ -1,221 +0,0 @@ ---- -name: deduplicator -description: | - 对来自所有数据源的原始项目进行跨数据源去重。读取工作区的 raw-projects.json,规范化 URL,执行三级去重检测(GitHub URL、Hugging Face URL、Website URL、Slug 匹配),生成新项目列表和任务队列。使用此 agent 当需要对爬取的项目进行去重时。 - - 示例场景: - - 多个数据源爬取完成后需要去重 - - 检查新项目是否已存在于数据库 - - 输入参数:{"workspace": ".trending-workspace/..."} -model: inherit -color: purple ---- - -# 统一去重器 Agent - -## 职责 - -对来自所有数据源的原始项目进行跨数据源去重: -1. 读取原始项目数据 -2. 规范化 URL -3. 三级去重检测 -4. 生成新项目列表 -5. 生成分析任务队列 - -## 输入参数 - -```json -{ - "workspace": ".trending-workspace/..." -} -``` - -## 执行步骤 - -### Step 1: 读取原始数据 - -从工作区读取 `raw-projects.json`。 - -### Step 2: URL 规范化 - -对不同数据源的 URL 进行规范化处理: - -```typescript -function normalizeUrl(url: string): string { - return url.toLowerCase() - .replace(/\/$/, '') // 移除尾部斜杠 - .replace(/^https?:\/\//, '') // 移除协议(用于比较) -} -``` - -### Step 3: 调用去重检查 API - -**重要**:使用 API 接口而非直接访问数据库。 - -#### 3.1 构造请求体 - -根据原始项目数据构造 API 请求: - -```json -{ - "apiKey": "sk_live_agent_park_webhook_key_2025", - "projects": [ - { - "githubUrl": "https://github.com/langchain-ai/langchain", - "huggingfaceUrl": null, - "websiteUrl": "https://python.langchain.com", - "slug": "langchain" - } - ] -} -``` - -**URL 提取规则**: -- GitHub 项目:`githubUrl` = 项目 URL,`slug` = generateSlug(name) -- Hugging Face 模型:`huggingfaceUrl` = 模型 URL,`slug` = generateSlug(name) -- Papers with Code:`websiteUrl` = 论文/项目 URL,`slug` = generateSlug(name) - -#### 3.2 调用 API - -使用 Bash 执行 curl 请求: - -```bash -curl -X POST http://localhost:3000/api/webhook/check-duplicates \ - -H "Content-Type: application/json" \ - -d @check-request.json \ - -o check-response.json -``` - -#### 3.3 处理响应 - -API 响应格式: - -```json -{ - "success": true, - "results": [ - { - "githubUrl": "https://github.com/langchain-ai/langchain", - "exists": true, - "matchType": "GITHUB_URL", - "projectId": "cm2x8k9d10001", - "projectName": "LangChain" - } - ], - "stats": { - "total": 45, - "exists": 10, - "new": 35, - "breakdown": { - "githubUrl": 5, - "huggingfaceUrl": 2, - "websiteUrl": 1, - "slug": 2 - } - } -} -``` - -**MatchType 说明**: -- `GITHUB_URL`: 通过 GitHub URL 匹配 -- `HUGGINGFACE_URL`: 通过 Hugging Face URL 匹配 -- `WEBSITE_URL`: 通过官网 URL 匹配 -- `SLUG`: 通过 slug 匹配 -- `NONE`: 未匹配,新项目 - -根据 `results[i].exists` 判断是否为新项目,仅保留 `exists: false` 的项目。 - -### Step 4: 生成新项目列表 - -输出 `new-projects.json`: - -```json -{ - "metadata": { - "totalRaw": 45, - "duplicates": 10, - "duplicateBreakdown": { - "githubUrl": 5, - "huggingfaceUrl": 2, - "websiteUrl": 1, - "slug": 2 - }, - "new": 35 - }, - "projects": [ - { - "source": "github", - "name": "langchain-ai/langchain", - "url": "https://github.com/langchain-ai/langchain", - "description": "Building applications with LLMs through composability", - "metadata": { /* ... */ } - } - ] -} -``` - -### Step 5: 生成分析任务队列 - -输出 `task-queue.json`: - -```json -{ - "metadata": { - "totalTasks": 35, - "createdAt": "2025-01-04T12:10:00Z" - }, - "tasks": [ - { - "id": 1, - "source": "github", - "name": "langchain-ai/langchain", - "url": "https://github.com/langchain-ai/langchain", - "status": "pending" - }, - { - "id": 2, - "source": "huggingface", - "name": "meta-llama/Llama-2-7b", - "url": "https://huggingface.co/meta-llama/Llama-2-7b", - "status": "pending" - } - ] -} -``` - -## Slug 生成规则 - -```typescript -function generateSlug(name: string): string { - return name - .toLowerCase() - .replace(/[^a-z0-9\s-]/g, '') // 移除特殊字符 - .trim() - .replace(/\s+/g, '-') // 空格转连字符 - .substring(0, 100) // 限制长度 -} -``` - -## 跨数据源去重示例 - -同一个项目可能同时出现在: -- GitHub Trending: `https://github.com/openai/whisper` -- Hugging Face: `https://huggingface.co/openai/whisper-large-v3` - -通过 GitHub URL 匹配,识别为重复项目,保留一个即可。 - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| raw-projects.json 不存在 | 错误提示 "请先运行任务派发器" | -| API 调用失败 | 错误提示并终止,保存中间结果到 `check-response.json` | -| API 返回 success: false | 错误提示 API 错误详情 | - -## 输出 - -成功后,返回: -- 新项目数量 -- 重复项目数量及原因分布 -- 任务队列路径 diff --git a/.claude/agents/project-analyzer.md b/.claude/agents/project-analyzer.md deleted file mode 100644 index 48cdb29..0000000 --- a/.claude/agents/project-analyzer.md +++ /dev/null @@ -1,290 +0,0 @@ ---- -name: project-analyzer -description: | - 对新项目进行深度分析。从任务队列获取待处理项目,访问项目页面获取详细信息,生成中英双语内容,提取标签和链接,计算质量评分。使用此 agent 当需要分析 GitHub、Hugging Face 或 Papers with Code 项目时。 - - 示例场景: - - 去重完成后需要分析新项目 - - 需要提取项目详细信息、标签和链接 - - 输入参数:{"workspace": ".trending-workspace/..."} -model: inherit -color: blue ---- - -# 项目分析器 Agent - -## 职责 - -对新项目进行深度分析: -1. 从任务队列获取待处理项目 -2. 访问项目页面获取详细信息 -3. 生成中英双语内容 -4. 提取标签和链接 -5. 计算质量评分 -6. 输出分析后的项目数据 - -## 输入参数 - -```json -{ - "workspace": ".trending-workspace/...", - "taskId": 5 -} -``` - -**参数说明**: -- `workspace`: 工作区路径 -- `taskId`: 要处理的单个任务 ID(对应 task-queue.json 中的任务编号) - -## 执行步骤 - -### Step 1: 读取任务队列 - -从工作区读取 `task-queue.json`,获取所有任务数据。 - -### Step 2: 获取指定任务 - -根据 `taskId` 参数从任务队列中找到对应的任务。 - -**验证**: -- 确认任务存在 -- 确认任务状态为 `pending`(避免重复处理) -- 如果任务状态不是 `pending`,直接退出并返回当前状态 - -### Step 3: 更新任务状态为 processing - -将任务状态从 `pending` 更新为 `processing`(防止其他实例重复处理)。 - -### Step 4: 处理单个项目 - -对指定的项目执行以下操作: - -#### 4.1 访问项目页面 - -使用 chrome-devtools-mcp 访问项目 URL: - -- **GitHub 项目**: 访问 GitHub 仓库页面 -- **Hugging Face 模型**: 访问 Hugging Face 模型页面 -- **Papers with Code**: 访问项目/论文页面 - -#### 4.2 提取详细信息 - -从页面提取以下信息: - -**GitHub 项目**: -- README.md 内容(完整 Markdown) -- GitHub Topics(标签) -- 编程语言分布 -- 许可证 -- 最新更新时间 -- 贡献者数量 -- Issues/PRs 数量 - -**Hugging Face 模型**: -- 模型描述 -- Pipeline 类型 -- 任务标签 -- 库/框架依赖 -- 使用示例 - -**Papers with Code**: -- 论文摘要 -- 相关代码仓库 -- 任务类别 -- 引用数 - -#### 4.3 理解项目价值(核心) - -**项目用途**:项目能做什么? -- 核心功能是什么? -- 解决什么问题? -- 有什么独特价值? - -**适用场景**:谁在什么情况下使用? -- 目标用户群体(开发者、研究者、企业等) -- 典型使用场景 -- 应用领域(NLP、CV、强化学习等) - -**技术特点**:如何实现? -- 使用什么技术/框架? -- 有什么技术亮点? - -从页面内容中提炼这些信息,用用户友好的语言描述。 - -#### 4.4 生成中英双语内容 - -**name / nameEn**: 项目名称翻译 -- 通常保持英文名称不变 -- 如果有中文名称,使用原名 - -**description / descriptionEn**: 简短描述(10-500 字符) -- **重点**: 用一句话说明项目能做什么 -- 格式:"[项目名] 是一个 [用途] 的 [类型],通过 [核心特点] 实现 [价值]" -- 示例:"LangChain 是一个开发 LLM 应用的框架,通过链式调用和工具集成,简化 AI 应用的构建流程" - -**content / contentEn**: 详细内容(最多 10000 字符) -- **项目用途**: 详细的能做什么描述 -- **适用场景**: 典型使用案例 -- **核心功能**: 主要功能列表 -- **技术特点**: 技术亮点 -- **使用指南**: 快速开始或使用示例 - -#### 4.5 提取结构化标签(1-10 个) - -从以下来源提取标签: -- GitHub Topics -- 编程语言 -- Pipeline 类型 -- 任务类别 -- AI/ML 相关关键词 - -标签格式: -```json -{ - "tags": [ - { "name": "LLM", "nameEn": "Large Language Model" }, - { "name": "Python", "nameEn": "Python" }, - { "name": "深度学习", "nameEn": "Deep Learning" } - ] -} -``` - -#### 4.6 构造外部链接数组(1-10 个) - -收集项目相关链接: - -**GitHub 项目**: -```json -{ - "links": [ - { "type": "GITHUB", "url": "...", "title": "GitHub 仓库" }, - { "type": "WEBSITE", "url": "...", "title": "官网" }, - { "type": "WEBSITE", "url": "...", "title": "文档" } - ] -} -``` - -**Hugging Face 模型**: -```json -{ - "links": [ - { "type": "HUGGINGFACE", "url": "...", "title": "Hugging Face" }, - { "type": "GITHUB", "url": "...", "title": "GitHub 仓库" }, - { "type": "PAPER", "url": "...", "title": "论文" } - ] -} -``` - -#### 4.7 计算质量评分 - -总分 100 分,>= 40 分通过: - -```typescript -function calculateQualityScore(project): number { - let score = 0 - - // 描述/README (0-20) - if (project.description?.length > 50) score += 10 - if (project.content?.length > 500) score += 10 - - // Stars/Likes (0-20) - if (project.stars > 1000 || project.likes > 500) score += 20 - else if (project.stars > 100 || project.likes > 50) score += 10 - - // 活跃度 (0-20) - const daysSinceUpdate = getDaysSince(project.lastUpdate) - if (daysSinceUpdate < 30) score += 20 - else if (daysSinceUpdate < 180) score += 10 - - // 文档 (0-20) - if (project.hasDocsLink) score += 10 - if (project.hasExamples) score += 10 - - // 社区 (0-20) - if (project.forks > 10 || project.downloads > 100) score += 10 - if (project.recentActivity) score += 10 - - return score -} -``` - -#### 4.8 更新任务状态 - -将任务状态从 `pending` 更新为 `completed` 或 `failed`(质量不足)。 - -### Step 5: 更新任务状态为最终状态 - -将任务状态从 `processing` 更新为: -- `completed` - 质量评分 >= 40 -- `failed` - 质量评分 < 40 或处理出错 - -### Step 6: 输出分析结果 - -输出 `analyzed-project-{taskId}.json`(避免多实例文件冲突): - -**成功情况**: -```json -{ - "taskId": 5, - "success": true, - "project": { - "source": "github", - "name": "LangChain", - "nameEn": "LangChain", - "description": "通过组合性构建大型语言模型应用程序的框架,支持链式调用、代理、工具集成等核心功能。", - "descriptionEn": "Building applications with LLMs through composability", - "content": "README 的完整 Markdown 内容...", - "contentEn": "README content...", - "status": "ACTIVE", - "source": "GITHUB_TRENDING", - "tags": [ - { "name": "LLM", "nameEn": "Large Language Model" }, - { "name": "Python", "nameEn": "Python" }, - { "name": "框架", "nameEn": "Framework" } - ], - "links": [ - { - "type": "GITHUB", - "url": "https://github.com/langchain-ai/langchain", - "title": "GitHub 仓库" - }, - { - "type": "WEBSITE", - "url": "https://python.langchain.com", - "title": "官方文档" - } - ], - "qualityScore": 85 - } -} -``` - -**失败情况**: -```json -{ - "taskId": 5, - "success": false, - "error": "质量评分不足", - "errorDetails": { - "qualityScore": 25, - "threshold": 40 - } -} -``` - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| 页面访问失败 | 标记任务为 `failed`,记录错误原因 | -| 内容提取失败 | 标记任务为 `failed`,记录错误原因 | -| 质量评分不足 | 标记任务为 `failed`,原因 `lowQuality` | - -## 输出 - -成功后,返回: -- taskId(处理的任务 ID) -- success(是否成功) -- 项目数据(成功时)或错误信息(失败时) -- analyzed-project-{taskId}.json 路径 diff --git a/.claude/agents/scrapers/github-trending.md b/.claude/agents/scrapers/github-trending.md deleted file mode 100644 index 895556c..0000000 --- a/.claude/agents/scrapers/github-trending.md +++ /dev/null @@ -1,176 +0,0 @@ ---- -name: github-trending -description: | - 从 GitHub Trending 页面爬取 AI 相关项目。访问 GitHub Trending 页面,解析项目信息,执行 AI 关键词过滤,输出结构化项目数据。使用此 agent 当需要从 GitHub 获取最新的 AI 趋势项目时。 - - 示例场景: - - 获取每日/每周/每月的 GitHub AI 趋势项目 - - 发现热门的 AI/ML 开源项目 - - 输入参数:{"period": "daily", "limit": 25, "workspace": ".trending-workspace/..."} -model: inherit -color: black ---- - -# GitHub Trending 爬虫 - -## 职责 - -从 GitHub Trending 页面爬取 AI 相关项目。 - -## 输入参数 - -```json -{ - "period": "daily", - "limit": 25, - "workspace": ".trending-workspace/..." -} -``` - -## 执行步骤 - -### Step 1: 构造 URL - -``` -https://github.com/trending?since={period} -``` - -period 参数映射: -- `daily` → `daily` -- `weekly` → `weekly` -- `monthly` → `monthly` - -### Step 2: 访问页面 - -使用 chrome-devtools-mcp 访问 GitHub Trending 页面。 - -### Step 3: 解析页面 - -从页面中提取项目信息: - -```javascript -// 选择器 -const articles = document.querySelectorAll('article.Box-row') - -for (const article of articles) { - const nameElement = article.querySelector('h2 a') - const name = nameElement?.textContent.trim() - const url = 'https://github.com' + nameElement?.getAttribute('href') - - const description = article.querySelector('p')?.textContent.trim() - - const starsElement = article.querySelector('a[href$="/stargazers"]') - const stars = parseStars(starsElement?.textContent) - - const language = article.querySelector('span[itemprop="programmingLanguage"]')?.textContent -} -``` - -### Step 4: AI 关键词过滤 - -保留包含以下 AI/ML 相关关键词的项目: - -**英文关键词**: -- ai, artificial intelligence -- ml, machine learning -- llm, large language model -- nlp, natural language -- computer vision, cv -- deep learning, neural, network -- gpt, transformer, diffusion -- agent, autonomous -- langchain, huggingface, openai -- embedding, vector -- generative, generation - -**中文关键词**: -- 人工智能, 机器学习 -- 深度学习, 神经网络 -- 大语言模型, LLM -- 自然语言, NLP -- 计算机视觉 -- 智能体, 代理 - -**过滤规则**: -- 项目名称、描述、GitHub Topics 任一包含关键词即保留 -- 区分大小写不敏感 - -### Step 5: 输出结果 - -```json -{ - "source": "github", - "count": 25, - "projects": [ - { - "source": "github", - "name": "langchain-ai/langchain", - "url": "https://github.com/langchain-ai/langchain", - "description": "Building applications with LLMs through composability", - "metadata": { - "stars": 85432, - "starsDelta": "+234 today", - "language": "Python", - "forks": 12543 - } - } - ] -} -``` - -## 页面结构参考 - -``` -article.Box-row -├── h2 -│ └── a[href="/langchain-ai/langchain"] → langchain-ai/langchain -├── p → Building applications with LLMs... -├── div -│ ├── span[itemprop="programmingLanguage"] → Python -│ ├── a[href$="/stargazers"] → 85k stars -│ └── a[href$="/forks"] → 12k forks -└── div → Forked from ... -``` - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| 页面访问失败 | 重试 3 次,指数退避(1s, 2s, 4s) | -| 页面解析失败 | 返回空项目列表,记录错误 | -| 无 AI 相关项目 | 返回空项目列表,提示 "未找到 AI 相关项目" | -| 数量不足 | 返回找到的所有项目,不报错 | - -## 输出文件 - -将结果写入工作区:`scraped-github-projects.json` - -```json -{ - "source": "github", - "count": 25, - "projects": [ - { - "source": "github", - "name": "langchain-ai/langchain", - "url": "https://github.com/langchain-ai/langchain", - "description": "Building applications with LLMs through composability", - "metadata": { - "stars": 85432, - "starsDelta": "+234 today", - "language": "Python", - "forks": 12543 - } - } - ] -} -``` - -## 输出 - -成功后,返回: -- 输出文件路径:`{workspace}/scraped-github-projects.json` -- 爬取的项目数量 -- 过滤后的项目数量 -- 项目列表 diff --git a/.claude/agents/scrapers/huggingface-trending.md b/.claude/agents/scrapers/huggingface-trending.md deleted file mode 100644 index ca87e29..0000000 --- a/.claude/agents/scrapers/huggingface-trending.md +++ /dev/null @@ -1,167 +0,0 @@ ---- -name: huggingface-trending -description: | - 从 Hugging Face Models 页面爬取热门 AI 模型。访问 Hugging Face Models 页面,解析模型信息,提取点赞数、下载量、Pipeline 类型等元数据。使用此 agent 当需要从 Hugging Face 获取热门 AI 模型时。 - - 示例场景: - - 获取最新的热门 AI 模型 - - 发现特定 Pipeline 类别的模型 - - 输入参数:{"period": "daily", "limit": 25, "workspace": ".trending-workspace/..."} -model: inherit -color: yellow ---- - -# Hugging Face Trending 爬虫 - -## 职责 - -从 Hugging Face Models 页面爬取热门 AI 模型。 - -## 输入参数 - -```json -{ - "period": "daily", - "limit": 25, - "workspace": ".trending-workspace/..." -} -``` - -**注意**: Hugging Face 不支持 period 参数,忽略该参数。 - -## 执行步骤 - -### Step 1: 构造 URL - -``` -https://huggingface.co/models -``` - -可选参数(用于筛选): -- `?pipeline_tag=text-generation` - 文本生成模型 -- `?pipeline_tag=image-classification` - 图像分类模型 -- `?pipeline_tag=automatic-speech-recognition` - 语音识别模型 - -默认不筛选,获取所有热门模型。 - -### Step 2: 访问页面 - -使用 chrome-devtools-mcp 访问 Hugging Face Models 页面。 - -### Step 3: 解析页面 - -从页面中提取模型信息: - -```javascript -// 选择器(示例,需根据实际页面调整) -const modelCards = document.querySelectorAll('[class*="modelCard"]') - -for (const card of modelCards) { - const nameElement = card.querySelector('a[href*="/models/"]') - const name = nameElement?.textContent.trim() - const url = 'https://huggingface.co' + nameElement?.getAttribute('href') - - const description = card.querySelector('[class*="description"]')?.textContent.trim() - - const likes = parseLikes(card.querySelector('button[aria-label*="Like"]')?.textContent) - const downloads = parseDownloads(card.querySelector('[class*="downloads"]')?.textContent) - - const pipeline = card.querySelector('[class*="pipeline"]')?.textContent -} -``` - -### Step 4: AI 相关性过滤 - -所有 Hugging Face 模型都是 AI 相关的,无需额外过滤。 - -可选的筛选条件: -- Pipeline 类型:text-generation, image-generation, audio 等 -- 下载量或点赞数阈值 - -### Step 5: 输出结果 - -```json -{ - "source": "huggingface", - "count": 25, - "projects": [ - { - "source": "huggingface", - "name": "meta-llama/Llama-2-7b", - "url": "https://huggingface.co/meta-llama/Llama-2-7b", - "description": "Llama 2 is a collection of pretrained and fine-tuned generative text models...", - "metadata": { - "likes": 15234, - "downloads": 5000000, - "pipeline": "text-generation", - "task": "Text Generation" - } - } - ] -} -``` - -## 页面结构参考 - -Hugging Face 页面结构可能动态变化,需要根据实际情况调整选择器。 - -常见类名模式: -- 模型卡片: `SfProFile`, `modelCard` -- 标题: `h1`, `h2`, 或链接文本 -- 描述: `summary`, `description` -- 点赞按钮: `button[aria-label*="Like"]` -- 下载数: 包含 "downloads" 的元素 - -## 常用 Pipeline 类型 - -| Pipeline | 说明 | -|----------|------| -| text-generation | 文本生成 | -| text-classification | 文本分类 | -| image-generation | 图像生成 | -| image-classification | 图像分类 | -| automatic-speech-recognition | 语音识别 | -| text-to-speech | 文本转语音 | -| translation | 翻译 | -| question-answering | 问答 | - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| 页面访问失败 | 重试 3 次,指数退避 | -| 页面解析失败 | 返回空项目列表,记录错误 | -| 无模型数据 | 返回空项目列表 | - -## 输出文件 - -将结果写入工作区:`scraped-huggingface-projects.json` - -```json -{ - "source": "huggingface", - "count": 25, - "projects": [ - { - "source": "huggingface", - "name": "meta-llama/Llama-2-7b", - "url": "https://huggingface.co/meta-llama/Llama-2-7b", - "description": "Llama 2 is a collection of pretrained and fine-tuned generative text models...", - "metadata": { - "likes": 15234, - "downloads": 5000000, - "pipeline": "text-generation", - "task": "Text Generation" - } - } - ] -} -``` - -## 输出 - -成功后,返回: -- 输出文件路径:`{workspace}/scraped-huggingface-projects.json` -- 爬取的模型数量 -- 模型列表(按点赞数/下载量排序) diff --git a/.claude/agents/scrapers/papers-with-code.md b/.claude/agents/scrapers/papers-with-code.md deleted file mode 100644 index 5984fe0..0000000 --- a/.claude/agents/scrapers/papers-with-code.md +++ /dev/null @@ -1,165 +0,0 @@ ---- -name: papers-with-code -description: | - 从 Papers with Code 网站爬取热门论文和相关项目。访问 Papers with Code 页面,解析论文信息,提取 GitHub 仓库链接、Stars 数量、任务类别等元数据。使用此 agent 当需要从 Papers with Code 获取热门 AI 论文项目时。 - - 示例场景: - - 获取最新的热门 AI 论文 - - 发现带代码实现的学术研究 - - 输入参数:{"period": "daily", "limit": 25, "workspace": ".trending-workspace/..."} -model: inherit -color: cyan ---- - -# Papers with Code 爬虫 - -## 职责 - -从 Papers with Code 网站爬取热门论文和相关项目。 - -## 输入参数 - -```json -{ - "period": "daily", - "limit": 25, - "workspace": ".trending-workspace/..." -} -``` - -**注意**: Papers with Code 不支持 period 参数,忽略该参数。 - -## 执行步骤 - -### Step 1: 构造 URL - -``` -https://paperswithcode.com/ -``` - -或直接访问热门页面: -``` -https://paperswithcode.com/trending -``` - -### Step 2: 访问页面 - -使用 chrome-devtools-mcp 访问 Papers with Code 页面。 - -### Step 3: 解析页面 - -从页面中提取论文/项目信息: - -```javascript -// 选择器(示例,需根据实际页面调整) -const paperCards = document.querySelectorAll('[class*="paper"]') - -for (const card of paperCards) { - const titleElement = card.querySelector('a[href*="/paper/"]') - const title = titleElement?.textContent.trim() - const paperUrl = 'https://paperswithcode.com' + titleElement?.getAttribute('href') - - const description = card.querySelector('[class*="abstract"]')?.textContent.trim() - - const githubLink = card.querySelector('a[href*="github.com"]') - const githubUrl = githubLink?.getAttribute('href') - - const stars = parseStars(card.querySelector('[class*="stars"]')?.textContent) - - const tasks = Array.from(card.querySelectorAll('[class*="task"]')) - .map(el => el.textContent.trim()) -} -``` - -### Step 4: 筛选条件 - -保留同时满足以下条件的论文: -- 有 GitHub 仓库链接 -- 有 Stars 数量显示 -- 任务类别属于 AI/ML 相关(Computer Vision, NLP, Reinforcement Learning 等) - -### Step 5: 输出结果 - -```json -{ - "source": "paperswithcode", - "count": 25, - "projects": [ - { - "source": "paperswithcode", - "name": "YOLOv7: Trainable bag-of-freebies sets new state-of-the-art", - "url": "https://github.com/WongKinYiu/yolov7", - "paperUrl": "https://paperswithcode.com/paper/yolov7-trainable-bag-of-freebies-sets-new", - "description": "YOLOv7 implements bag-of-freebies and bag-of-specials...", - "metadata": { - "stars": 8000, - "tasks": ["Object Detection", "Computer Vision"], - "framework": "PyTorch" - } - } - ] -} -``` - -## 页面结构参考 - -Papers with Code 页面结构可能动态变化,需要根据实际情况调整选择器。 - -常见元素: -- 论文标题: h1, h2, 或带 paper 类名的链接 -- 摘要: abstract, summary 类名的元素 -- GitHub 链接: a[href*="github.com"] -- Stars 数量: 包含 "stars" 或 "★" 的元素 -- 任务标签: task 类名的元素 - -## 常见任务类别 - -| 类别 | 说明 | -|------|------| -| Computer Vision | 计算机视觉 | -| Natural Language Processing | 自然语言处理 | -| Reinforcement Learning | 强化学习 | -| Generative Models | 生成模型 | -| Speech | 语音处理 | -| Graph Learning | 图学习 | - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| 页面访问失败 | 重试 3 次,指数退避 | -| 页面解析失败 | 返回空项目列表,记录错误 | -| 无符合条件的论文 | 返回空项目列表 | - -## 输出文件 - -将结果写入工作区:`scraped-paperswithcode-projects.json` - -```json -{ - "source": "paperswithcode", - "count": 25, - "projects": [ - { - "source": "paperswithcode", - "name": "YOLOv7: Trainable bag-of-freebies sets new state-of-the-art", - "url": "https://github.com/WongKinYiu/yolov7", - "paperUrl": "https://paperswithcode.com/paper/yolov7-trainable-bag-of-freebies-sets-new", - "description": "YOLOv7 implements bag-of-freebies and bag-of-specials...", - "metadata": { - "stars": 8000, - "tasks": ["Object Detection", "Computer Vision"], - "framework": "PyTorch" - } - } - ] -} -``` - -## 输出 - -成功后,返回: -- 输出文件路径:`{workspace}/scraped-paperswithcode-projects.json` -- 爬取的论文/项目数量 -- 项目列表(按 Stars 数量排序) diff --git a/.claude/commands/add-trending.md b/.claude/commands/add-trending.md deleted file mode 100644 index fd53d39..0000000 --- a/.claude/commands/add-trending.md +++ /dev/null @@ -1,283 +0,0 @@ ---- -description: Use the specialized agents workflow to automatically fetch AI projects from multiple data sources (GitHub Trending, Hugging Face, etc.) and ingest them into the database ---- - -## 用户输入 - -```text -$ARGUMENTS -``` - -在继续之前, 你**必须**考虑用户输入(如果不为空). - -## 概述 - -本命令使用 **专用 Agent 工作流** 来完成多源数据获取和入库任务。核心原则: -- **所有数据处理逻辑由 Agent 执行**,不在主窗口执行 -- **使用 Task 工具调用各个 Agent**,让 Agent 自主完成其职责 -- **主窗口仅负责协调 Agent 调用**,不直接处理业务逻辑 - -本命令从多个数据源并行爬取 AI 相关项目,经过去重、分析、质量评分后批量入库。 - -**命令格式**: `/add-trending [source] [period] [limit]` - -**参数说明**: -- `source`: 数据源,可选 `github` / `huggingface` / `paperswithcode` / `all`(默认) -- `period`: 时间周期,可选 `daily`(默认)/ `weekly` / `monthly` -- `limit`: 每个数据源获取的项目数量上限,默认 25 - -**示例**: -- `/add-trending` - 默认参数(all, daily, 25个) -- `/add-trending github` - 仅 GitHub Trending -- `/add-trending huggingface daily 10` - 仅 Hugging Face,日榜10个 -- `/add-trending all weekly` - 所有数据源,周榜 - -## 数据源配置 - -| 源名称 | Agent 名称 | URL | -|--------|-----------|-----| -| github | github-trending | https://github.com/trending | -| huggingface | huggingface-trending | https://huggingface.co/models | -| paperswithcode | papers-with-code | https://paperswithcode.com/ | - -## 执行流程 - -### Stage 1: 初始化工作区与并行爬取 - -1. **解析参数**: - - 从 `$ARGUMENTS` 解析 source、period 和 limit - - 默认值: source=all, period=daily, limit=25 - -2. **创建工作区**: - - 生成时间戳目录: `.trending-workspace/{YYYYMMDD-HHMMSS}/` - - 初始化 `progress.json` 文件: - ```json - { - "startTime": "2025-01-06T12:00:00Z", - "currentStage": "scraping", - "stages": { - "scraping": "pending", - "deduplicating": "pending", - "analyzing": "pending", - "ingesting": "pending" - }, - "config": { - "source": "all", - "period": "daily", - "limit": 25 - } - } - ``` - -3. **确定要调用的 scrapers**: - 根据 source 参数筛选: - - | source | 调用的 scrapers | - |--------|-----------------| - | all | github-trending, huggingface-trending, papers-with-code | - | github | github-trending | - | huggingface | huggingface-trending | - | paperswithcode | papers-with-code | - -4. **并行调用 scrapers**: - **关键**: 必须在单个消息中发送所有 Task 调用,以实现真正的并行执行。 - - 示例(source=all): - ``` - Task(github-trending, {"period": "daily", "limit": 25, "workspace": ".trending-workspace/..."}) - Task(huggingface-trending, {"period": "daily", "limit": 25, "workspace": ".trending-workspace/..."}) - Task(papers-with-code, {"period": "daily", "limit": 25, "workspace": ".trending-workspace/..."}) - ``` - -5. **等待所有 scraper 完成并汇总**: - 每个 scraper 输出到各自的文件: - - `scraped-github-projects.json` - - `scraped-huggingface-projects.json` - - `scraped-paperswithcode-projects.json` - - 读取所有输出文件,合并生成 `raw-projects.json`: - ```json - { - "metadata": { - "timestamp": "2025-01-06T12:00:00Z", - "period": "daily", - "limit": 25, - "sources": ["github", "huggingface"], - "sourceCounts": { - "github": 25, - "huggingface": 20 - }, - "totalRaw": 45 - }, - "projects": [ - { - "source": "github", - "name": "langchain-ai/langchain", - "url": "https://github.com/langchain-ai/langchain", - "description": "Building applications with LLMs through composability", - "metadata": { - "stars": 85432, - "starsDelta": "+234", - "language": "Python", - "forks": 12543 - } - } - ] - } - ``` - -6. **更新进度**: - ```json - { - "startTime": "2025-01-06T12:00:00Z", - "currentStage": "scraping", - "stages": { - "scraping": "completed", - "deduplicating": "pending", - "analyzing": "pending", - "ingesting": "pending" - }, - "config": { - "source": "all", - "period": "daily", - "limit": 25 - } - } - ``` - -### Stage 2: 统一去重 - -1. **执行去重器**: - - **使用 Task 工具调用** `deduplicator` agent - - 将 workspace 参数传递给去重器 - - 去重器读取 `raw-projects.json` - - 使用 **dbhub PostgreSQL MCP** 执行去重查询 - - 输出 `new-projects.json` 和 `task-queue.json` - -### Stage 3: 项目分析 (并行,固定 3 实例分批处理) - -1. **读取任务队列**: - - 读取 `task-queue.json` - - 提取所有 `status: "pending"` 的任务 - - 记录任务数量 N - -2. **分批并行处理**: - - **固定并行度**: 3 个实例 - - **每批处理**: 3 个任务(最后一批可能少于 3 个) - - **批次数**: `Math.ceil(N / 3)` - - **循环执行**以下步骤,直到所有任务完成: - - **对每一批**: - - 选取 3 个 `status: "pending"` 的任务(记录其 taskId) - - **使用单个消息发送 3 个 Task 工具调用**(实现并行): - ``` - Task(project-analyzer, {"workspace": "...", "taskId": 1}) - Task(project-analyzer, {"workspace": "...", "taskId": 2}) - Task(project-analyzer, {"workspace": "...", "taskId": 3}) - ``` - - **等待这 3 个实例完成** - - **检查剩余任务**:重新读取 `task-queue.json`,确认是否还有 `pending` 任务 - - **继续下一批**:如果有 pending 任务,重复上述步骤 - -3. **等待所有批次完成**: - - 确认 `task-queue.json` 中没有 `status: "pending"` 或 `processing` 的任务 - -4. **汇总结果**: - - 读取所有 `analyzed-project-{taskId}.json` 文件 - - 合并生成 `analyzed-projects.json` - - 统计通过/失败的项目数量 - -### Stage 4: 批量入库 - -1. **执行入库器**: - - **使用 Task 工具调用** `database-ingestor` agent - - 将 workspace 参数传递给入库器 - - 调用 `POST /api/webhook/projects` - - 输出 `ingestion-result.json` - -### Stage 5: 生成报告 - -1. **汇总所有阶段的结果** -2. **清理旧工作区**(删除 7 天前的) -3. **输出最终报告** - -## 工作区文件结构 - -``` -.trending-workspace/{timestamp}/ -├── scraped-github-projects.json # GitHub 原始数据 -├── scraped-huggingface-projects.json # Hugging Face 原始数据 -├── scraped-paperswithcode-projects.json # Papers with Code 原始数据 -├── raw-projects.json # 所有数据源的汇总数据 -├── new-projects.json # 去重后的新项目 -├── task-queue.json # 分析任务队列 -├── analyzed-projects.json # 分析完成的项目 -├── ingestion-result.json # 入库结果 -└── progress.json # 进度追踪 -``` - -## 关键规则 - -- **必须使用单个消息发送多个 Task 调用**:实现 scrapers 真正并行执行 -- **必须使用 Agent 工作流**:所有数据处理由专用 Agent 完成,不在主窗口执行 -- **必须使用 Task 工具**调用 Agent:让 Agent 自主完成其职责 -- **必须**先去重再分析,避免处理已存在的项目 -- **必须**进行质量评分,仅入库 >= 40 分的项目 -- **必须**保留工作区 7 天用于调试和审计 -- **必须**生成中英双语内容(name/nameEn, description/descriptionEn) - -## 错误处理 - -| 场景 | 处理方式 | -|------|---------| -| 某个数据源失败 | 其他源继续,记录失败源到 errors.json | -| 页面解析失败 | 跳过该项目,记录到 errors.json | -| 数据库连接失败 | 保存中间结果,提示用户稍后重试 | -| 部分任务失败 | 继续处理其他任务,最终汇总失败项 | -| 环境变量缺失 | 错误提示 "WEBHOOK_API_KEY 未配置" | - -## 输出格式 - -执行完成后,输出类似以下格式的报告: - -``` -🚀 多源数据自动入库启动 -⚙️ 配置: source=github, period=daily, limit=25 - -✅ Stage 1/4: 数据爬取 - 🔍 数据源: GitHub Trending - 📊 GitHub: 25 个项目 - 📦 汇总: 25 个原始项目 - -✅ Stage 2/4: 统一去重 - 🆕 新项目: 18 个 - 🔄 重复: 7 个 - -✅ Stage 3/4: 项目分析 (18 个任务) - ⭐ 通过质量评分: 16 个 - ❌ 质量不足: 2 个 - -✅ Stage 4/4: 批量入库 - ✅ 创建: 15 个 - 🔄 更新: 1 个 - -📊 最终报告 - - 原始数据: 25 个 (GitHub: 25) - - 去重过滤: 7 个 - - 质量过滤: 2 个 - - 入库成功: 16 个 - - 耗时: 约2分钟 - -📁 工作区: .trending-workspace/20250106-120000/ -``` - -**单数据源测试示例**: -``` -/add-trending github daily 5 # 仅测试 GitHub,5个项目 -/add-trending huggingface # 仅测试 Hugging Face -/add-trending paperswithcode # 仅测试 Papers with Code -``` - -## 开始执行 - -开始执行上述流程,按照各 Stage 依次完成。 diff --git a/scripts/add-markdown-project.js b/scripts/add-markdown-project.js deleted file mode 100644 index d5915d3..0000000 --- a/scripts/add-markdown-project.js +++ /dev/null @@ -1,181 +0,0 @@ -const http = require('http'); - -const markdownContent = `# AutoGen - -## 🎯 项目简介 - -**AutoGen** 是一个由微软开发的**多智能体 AI 应用程序框架**,可以创建能够自主工作或与人类协作的智能体。 - -### ✨ 核心特性 - -- **核心 API**:实现消息传递、事件驱动智能体以及本地和分布式运行时 -- **AgentChat API**:提供更简单但更具主见的 API,用于快速原型设计 -- **扩展 API**:支持 LLM 客户端的特定实现(如 OpenAI、Azure OpenAI) -- **AutoGen Studio**:用于构建多智能体应用程序的无代码 GUI -- **AutoGen Bench**:用于评估智能体性能的基准测试套件 - -## 📦 安装方式 - -\`\`\`bash -# 使用 pip 安装 -pip install -U "autogen-agentchat" "autogen-ext[openai]" - -# 安装 AutoGen Studio -pip install -U "autogenstudio" -\`\`\` - -> 💡 **提示**:AutoGen 需要 **Python 3.10 或更高版本** - -## 🚀 快速开始 - -### Hello World 示例 - -\`\`\`python -import asyncio -from autogen_agentchat.agents import AssistantAgent -from autogen_ext.models.openai import OpenAIChatCompletionClient - -async def main() -> None: - model_client = OpenAIChatCompletionClient(model="gpt-4o") - agent = AssistantAgent("assistant", model_client=model_client) - print(await agent.run(task="Say 'Hello World!'")) - await model_client.close() - -asyncio.run(main()) -\`\`\` - -## 📊 功能对比 - -| 特性 | AutoGen | LangChain | CrewAI | -|------|---------|-----------|--------| -| 多智能体协作 | ✅ | ✅ | ✅ | -| 无代码 GUI | ✅ | ❌ | ❌ | -| 分布式运行时 | ✅ | ❌ | ❌ | -| .NET 支持 | ✅ | ❌ | ❌ | -| 基准测试套件 | ✅ | ❌ | ❌ | - -## 🔧 高级用法 - -### 多智能体编排 - -使用 \`AgentTool\` 创建基本的多智能体编排设置: - -\`\`\`python -import asyncio -from autogen_agentchat.agents import AssistantAgent -from autogen_agentchat.tools import AgentTool -from autogen_ext.models.openai import OpenAIChatCompletionClient - -async def main() -> None: - model_client = OpenAIChatCompletionClient(model="gpt-4o") - - # 创建数学专家智能体 - math_agent = AssistantAgent( - "math_expert", - model_client=model_client, - system_message="You are a math expert.", - description="A math expert assistant.", - ) - - # 创建化学专家智能体 - chemistry_agent = AssistantAgent( - "chemistry_expert", - model_client=model_client, - system_message="You are a chemistry expert.", - description="A chemistry expert assistant.", - ) - - print("智能体创建成功!") -\`\`\` - -## 📚 任务清单 - -- [x] 安装 AutoGen -- [ ] 创建第一个智能体 -- [ ] 配置 OpenAI API -- [ ] 运行多智能体对话 -- [ ] 部署到生产环境 - -## 🎓 学习资源 - -1. [官方文档](https://microsoft.github.io/autogen/) -2. [GitHub 仓库](https://github.com/microsoft/autogen) -3. [API 参考](https://microsoft.github.io/autogen/docs/reference) -4. [示例代码](https://github.com/microsoft/autogen/tree/main/samples) - -## 💬 常见问题 - -### Q: AutoGen 是免费的吗? - -**A**: 是的!AutoGen 使用 MIT 许可证,完全开源免费。 - -### Q: 支持哪些 LLM 提供商? - -**A**: AutoGen 支持 OpenAI、Azure OpenAI,以及通过扩展 API 支持其他提供商。 - ---- - -## 📄 许可证 - -MIT License - 详见 [LICENSE](https://github.com/microsoft/autogen/blob/main/LICENSE) 文件 - -**Made with ❤️ by Microsoft** -`; - -const data = JSON.stringify({ - apiKey: 'sk_live_agent_park_webhook_key_2025', - projects: [{ - name: 'AutoGen', - nameEn: 'AutoGen', - description: 'Microsoft 开发的多智能体 AI 应用程序框架,支持自主或与人类协作的智能体', - descriptionEn: 'A programming framework for creating multi-agent AI applications that can act autonomously or work alongside humans', - content: markdownContent, - contentEn: markdownContent, // 使用相同内容用于测试 - status: 'ACTIVE', - source: 'GitHub', - tags: [ - { name: '多智能体', nameEn: 'Multi-Agent' }, - { name: '框架', nameEn: 'Framework' }, - { name: '微软', nameEn: 'Microsoft' }, - { name: 'Python', nameEn: 'Python' }, - { name: 'AI', nameEn: 'AI' }, - { name: 'LLM', nameEn: 'LLM' } - ], - links: [ - { type: 'GITHUB', url: 'https://github.com/microsoft/autogen', title: 'GitHub 仓库' }, - { type: 'WEBSITE', url: 'https://microsoft.github.io/autogen/', title: '官方文档' }, - { type: 'WEBSITE', url: 'https://pypi.org/project/autogen-agentchat/', title: 'PyPI 包' } - ] - }] -}); - -const options = { - hostname: '127.0.0.1', - port: 3001, - path: '/api/webhook/projects', - method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'Content-Length': Buffer.byteLength(data) - } -}; - -const req = http.request(options, (res) => { - let responseData = ''; - - res.on('data', (chunk) => { - responseData += chunk; - }); - - res.on('end', () => { - console.log('Status:', res.statusCode); - console.log('Response:', responseData); - }); -}); - -req.on('error', (error) => { - console.error('Error:', error.message); -}); - -req.write(data); -req.end(); diff --git a/scripts/add-project.js b/scripts/add-project.js deleted file mode 100644 index 0cb5dcc..0000000 --- a/scripts/add-project.js +++ /dev/null @@ -1,59 +0,0 @@ -const http = require('http'); - -const data = JSON.stringify({ - apiKey: 'sk_live_agent_park_webhook_key_2025', - projects: [{ - name: 'AutoGen', - nameEn: 'AutoGen', - description: 'Microsoft 开发的多智能体 AI 应用程序框架,支持自主或与人类协作的智能体', - descriptionEn: 'A programming framework for creating multi-agent AI applications that can act autonomously or work alongside humans', - content: 'AutoGen 是一个由微软开发的创建多智能体 AI 应用程序的框架。主要特性:\n\n1. 核心 API:实现消息传递、事件驱动智能体以及本地和分布式运行时\n2. AgentChat API:提供更简单但更具主见的 API,用于快速原型设计\n3. 扩展 API:支持 LLM 客户端的特定实现(如 OpenAI、Azure OpenAI)\n4. AutoGen Studio:用于构建多智能体应用程序的无代码 GUI\n5. AutoGen Bench:用于评估智能体性能的基准测试套件\n\n支持 Python 3.10+ 和 .NET,使用 MIT 许可证。', - contentEn: 'AutoGen is a framework for creating multi-agent AI applications. Key features: Core API, AgentChat API, Extensions API, AutoGen Studio, AutoGen Bench. Supports Python 3.10+ and .NET. MIT licensed.', - status: 'ACTIVE', - source: 'GitHub', - tags: [ - { name: '多智能体', nameEn: 'Multi-Agent' }, - { name: '框架', nameEn: 'Framework' }, - { name: '微软', nameEn: 'Microsoft' }, - { name: 'Python', nameEn: 'Python' }, - { name: 'AI', nameEn: 'AI' }, - { name: 'LLM', nameEn: 'LLM' } - ], - links: [ - { type: 'GITHUB', url: 'https://github.com/microsoft/autogen', title: 'GitHub 仓库' }, - { type: 'WEBSITE', url: 'https://microsoft.github.io/autogen/', title: '官方文档' }, - { type: 'WEBSITE', url: 'https://pypi.org/project/autogen-agentchat/', title: 'PyPI 包' } - ] - }] -}); - -const options = { - hostname: '127.0.0.1', - port: 3001, - path: '/api/webhook/projects', - method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'Content-Length': Buffer.byteLength(data) - } -}; - -const req = http.request(options, (res) => { - let responseData = ''; - - res.on('data', (chunk) => { - responseData += chunk; - }); - - res.on('end', () => { - console.log('Status:', res.statusCode); - console.log('Response:', responseData); - }); -}); - -req.on('error', (error) => { - console.error('Error:', error.message); -}); - -req.write(data); -req.end(); diff --git a/scripts/scrape-huggingface.js b/scripts/scrape-huggingface.js deleted file mode 100644 index 191cf17..0000000 --- a/scripts/scrape-huggingface.js +++ /dev/null @@ -1,113 +0,0 @@ -// Simple Hugging Face scraper -const https = require('https'); -const { HttpsProxyAgent } = require('https-proxy-agent'); - -async function scrapeHuggingFace() { - try { - console.log('Fetching Hugging Face models page...'); - - const proxyUrl = process.env.HTTPS_PROXY || process.env.HTTP_PROXY; - const agent = proxyUrl ? new HttpsProxyAgent(proxyUrl) : undefined; - - const html = await new Promise((resolve, reject) => { - const options = { - headers: { - 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', - 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', - 'Accept-Language': 'en-US,en;q=0.9', - } - }; - - if (agent) { - options.agent = agent; - } - - const req = https.get('https://huggingface.co/models', options, (res) => { - let data = ''; - res.on('data', chunk => data += chunk); - res.on('end', () => { - if (res.statusCode === 200) { - resolve(data); - } else { - reject(new Error(`HTTP ${res.statusCode}: ${res.statusMessage}`)); - } - }); - }); - - req.on('error', reject); - req.setTimeout(30000, () => { - req.destroy(); - reject(new Error('Request timeout')); - }); - }); - - console.log(`Page fetched successfully, size: ${html.length} bytes`); - - // Extract model information - const models = []; - const modelLinkRegex = /]*>/gi; - const seenModels = new Set(); - - let match; - while ((match = modelLinkRegex.exec(html)) !== null) { - const modelId = decodeURIComponent(match[1]); - - // Filter: must have org/model format - if (!modelId.includes('/')) continue; - if (modelId.includes('/discussions')) continue; - if (modelId.includes('/blob')) continue; - if (modelId.includes('/tree')) continue; - if (modelId.includes('/commit')) continue; - if (seenModels.has(modelId)) continue; - - seenModels.add(modelId); - - models.push({ - source: 'huggingface', - name: modelId, - url: `https://huggingface.co/${modelId}`, - description: modelId, - metadata: { - likes: 0, - downloads: 0, - pipeline: '' - } - }); - - if (models.length >= 100) break; - } - - console.log(`Extracted ${models.length} unique models`); - - // Take first 25 (they should be roughly ordered by popularity on the page) - const topModels = models.slice(0, 25); - - const fs = require('fs'); - const outputPath = 'D:\\Code\\AI\\agent-park-v2\\.trending-workspace\\20260106-1226\\scraped-huggingface-projects.json'; - fs.writeFileSync(outputPath, JSON.stringify(topModels, null, 2), 'utf8'); - console.log(`\nData saved to: ${outputPath}`); - - console.log('\nTop 25 Models:'); - topModels.forEach((m, i) => { - console.log(`${i + 1}. ${m.name}`); - }); - - } catch (error) { - console.error('Error:', error.message); - - // Output empty result on error - const fs = require('fs'); - const outputPath = 'D:\\Code\\AI\\agent-park-v2\\.trending-workspace\\20260106-1226\\scraped-huggingface-projects.json'; - - const errorResult = { - error: error.message, - projects: [], - timestamp: new Date().toISOString() - }; - - fs.writeFileSync(outputPath, JSON.stringify(errorResult, null, 2), 'utf8'); - console.log(`Error result saved to: ${outputPath}`); - } -} - -scrapeHuggingFace();