Files
zhuiguang-ai/scripts/daily-discover.mjs
T
ZhuiGuangAI Dev e263e28c81 全面审查优化: 43项修复(安全/性能/可靠性)+ 文档更新
🔴 P0 安全: .env入gitignore, CSP/HSTS, Logo安全(5MB限制+SVG消毒+SHA256), middleware权限收紧(VIP不能访问admin), 发布去重, GitHub认证, 删除危险API, skill-discoverer绕过审核修复, prisma默认连接修复
🔴 P0 可靠性: deepseek/github超时重试, 脚本层DeepSeek保护, 脚本启动验证, OAuth竞态修复
🔴 P0 性能: skill-discoverer N+1优化(↓94%), 后台API分页限制, getUserStats聚合优化
🟡 P1: 分类动态加载, 星级字段统一, fetch-logos并发Bug修复, 批量发布冲突不删除, check-tools重试, 前端空指针修复(13处), logo-fetcher日志, rate-limit清理, 公开API缓存
2026-05-26 00:30:36 +08:00

318 lines
13 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { PrismaClient } from "@prisma/client";
import { PrismaMariaDb } from "@prisma/adapter-mariadb";
import { Octokit } from "octokit";
import OpenAI from "openai";
import "dotenv/config";
import { withRetry } from "./lib/retry.mjs";
const base = process.env.DATABASE_URL || "mysql://localhost:3306/zhuiguang_ai?charset=utf8mb4";
const sep = base.includes("?") ? "&" : "?";
const connectionString = `${base}${sep}connection_limit=20&pool_timeout=30`;
const adapter = new PrismaMariaDb(connectionString);
const prisma = new PrismaClient({ adapter });
const octokit = new Octokit({
auth: process.env.GITHUB_TOKEN || undefined,
});
const openai = new OpenAI({
apiKey: process.env.DEEPSEEK_API_KEY,
baseURL: "https://api.deepseek.com/v1",
});
const REVIEW_PROMPT = `你是一个GitHub开源项目评测专家。请基于以下项目信息进行五维深度评测,并以JSON格式返回。务必只返回JSON,不要包含任何其他文字或Markdown格式。
项目信息:
- 仓库全名:{repoName}
- 简介:{description}
- GitHub Stars:{stars}
- 开源许可证:{license}
- README摘要(前1500字):{readmeSummary}
请严格按照以下JSON结构输出:
{
"capability": { "score": 4.5, "summary": "一句话亮点(15字内)", "detail": "80-120字场景能力分析" },
"devExp": { "score": 3.5, "summary": "上手体验总结", "detail": "80-120字,包含安装、文档、API评价" },
"costLicense": { "score": 4.0, "summary": "成本与许可总结", "detail": "80-120字,包含硬件成本、API定价、许可证风险" },
"community": { "score": 4.5, "summary": "社区健康度总结", "detail": "80-120字,包含Issue响应、PR频率、生态描述" },
"performance": { "score": 3.5, "summary": "性能稳定性总结", "detail": "80-120字,包含运行效率、稳定性、输出质量" },
"overall": 4.0,
"tags": ["标签1", "标签2", "标签3"]
}
评分要求:
- 分数为1-5之间,可使用0.5步长
- 综合分应为五个维度分数的算术平均值(保留1位小数)
- 每个维度的summary需严格控制在15字以内
- 每个维度的detail需严格控制在80-120字
- 标签从以下类别中选择最匹配的3-5个:LLM框架, Agent框架, RAG, 提示工程, 多模态, 代码助手, 模型训练, 数据工程, 部署工具, 评测工具, 安全对齐, 其他`;
const SEARCH_QUERIES = [
"ai agent framework",
"llm framework",
"rag framework",
"prompt engineering tool",
"ai code assistant",
"multimodal ai",
"ai model training",
"ai deployment tool",
"ai evaluation benchmark",
"ai safety alignment",
];
const CATEGORY_KEYWORDS = {
"AI Agent": ["agent", "autonomous", "workflow", "tool call"],
"代码生成": ["code", "programming", "developer", "copilot"],
"测试": ["test", "testing", "qa", "quality"],
"文档": ["documentation", "docs", "readme"],
"安全": ["security", "vulnerability", "safety", "alignment"],
"DevOps与部署": ["deployment", "deploy", "devops", "ci/cd", "docker", "kubernetes"],
"RAG": ["rag", "retrieval", "vector", "embedding"],
"AI/ML工程": ["training", "fine-tun", "model", "machine learning", "ml"],
"多模态": ["multimodal", "vision", "image", "video", "audio"],
"Prompt工程": ["prompt", "prompting", "instruction"],
};
const NON_AI_KEYWORDS = [
"book", "guide", "tutorial", "awesome", "interview",
"notes", "handbook", "cheatsheet", "cheat-sheet", "roadmap",
"everything you need to know", "curated list", "collection",
];
const AI_TOPICS = [
"ai", "artificial-intelligence", "machine-learning", "deep-learning",
"llm", "large-language-model", "nlp", "natural-language-processing",
"agent", "rag", "gpt", "transformer", "generative-ai", "chatgpt",
"langchain", "prompt-engineering", "multimodal", "text-generation",
"vector-database", "embedding", "model-training", "fine-tuning",
];
function isValidAIProject(repo) {
const name = (repo.full_name || "").toLowerCase();
const desc = (repo.description || "").toLowerCase();
const text = `${name} ${desc}`;
for (const kw of NON_AI_KEYWORDS) {
if (text.includes(kw)) {
console.log(
` 🚫 过滤(非AI): ${repo.full_name} (命中关键词: "${kw}")`
);
return false;
}
}
const language = (repo.language || "").toLowerCase();
if (["markdown", "html", "tex", "latex"].includes(language) && repo.stargazers_count < 5000) {
console.log(
` 🚫 过滤(文档): ${repo.full_name} (语言: ${repo.language}, ⭐${repo.stargazers_count})`
);
return false;
}
const topics = (repo.topics || []).map((t) => t.toLowerCase());
if (topics.length > 0) {
const hasAITopic = topics.some((t) =>
AI_TOPICS.some((ai) => t.includes(ai))
);
if (!hasAITopic) {
console.log(
` 🚫 过滤(无AI主题): ${repo.full_name} (topics: ${topics.join(", ").substring(0, 100)})`
);
return false;
}
}
return true;
}
function extractJSON(str) {
const trimmed = str.trim();
const codeBlockMatch = trimmed.match(/```(?:json)?\s*([\s\S]*?)```/);
let jsonStr = codeBlockMatch ? codeBlockMatch[1].trim() : trimmed;
const braceMatch = jsonStr.match(/\{[\s\S]*\}/);
if (braceMatch) {
try { return JSON.parse(braceMatch[0]); } catch {}
}
const arrayMatch = jsonStr.match(/\[[\s\S]*\]/);
if (arrayMatch) {
try { return JSON.parse(arrayMatch[0]); } catch {}
}
return JSON.parse(jsonStr);
}
async function guessCategory(name, desc, tags) {
const categories = await prisma.skillCategory.findMany({
select: { id: true, name: true },
orderBy: { sortOrder: "asc" },
});
if (categories.length === 0) {
throw new Error("数据库中没有技能分类,请先创建分类");
}
const text = `${name} ${desc} ${tags.join(" ")}`.toLowerCase();
for (const [categoryName, keywords] of Object.entries(CATEGORY_KEYWORDS)) {
for (const kw of keywords) {
if (text.includes(kw)) {
const match = categories.find((c) => c.name === categoryName);
if (match) return match.id;
}
}
}
return categories[0].id;
}
function genSlug(name) {
return name.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "").substring(0, 100);
}
async function searchRepos(query, minStars, max) {
const q = `${query} stars:>${minStars} pushed:>2024-01-01`;
const { data } = await withRetry(async () => {
return await octokit.rest.search.repos({ q, sort: "stars", order: "desc", per_page: max });
}, { maxRetries: 3, label: `GitHub Search "${query}"` });
return data.items;
}
async function getReadme(owner, repo) {
try {
const { data } = await withRetry(async () => {
return await octokit.rest.repos.getReadme({ owner, repo });
}, { maxRetries: 3, label: `GitHub README ${owner}/${repo}` });
return Buffer.from(data.content, "base64").toString("utf-8");
} catch { return ""; }
}
async function genReview(repoName, desc, stars, license, readme) {
const prompt = REVIEW_PROMPT
.replace("{repoName}", repoName).replace("{description}", desc || "暂无描述")
.replace("{stars}", stars.toString()).replace("{license}", license || "未知")
.replace("{readmeSummary}", readme.substring(0, 1500));
const { choices } = await withRetry(async () => {
return await openai.chat.completions.create({
model: process.env.DEEPSEEK_MODEL || "deepseek-v4-pro", messages: [{ role: "user", content: prompt }], temperature: 0.3,
}, { signal: AbortSignal.timeout(120000) });
}, { maxRetries: 3, label: "DeepSeek Review API" });
const content = choices[0]?.message?.content;
if (!content) throw new Error("Empty response");
let review = extractJSON(content);
if (review.data && typeof review.data === "object") review = review.data;
if (review.result && typeof review.result === "object") review = review.result;
if (!review.capability || !review.devExp || !review.costLicense || !review.community || !review.performance) {
throw new Error("评测JSON结构不完整: " + JSON.stringify(Object.keys(review)).substring(0, 200));
}
const dims = [review.capability.score, review.devExp.score, review.costLicense.score, review.community.score, review.performance.score];
const avg = dims.reduce((a, b) => a + b, 0) / 5;
if (Math.abs(review.overall - avg) > 0.3) review.overall = Math.round(avg * 10) / 10;
return review;
}
async function main() {
if (!process.env.DEEPSEEK_API_KEY) throw new Error("缺少环境变量: DEEPSEEK_API_KEY");
if (!process.env.DATABASE_URL) throw new Error("缺少环境变量: DATABASE_URL");
const startTime = new Date();
console.log(`🚀 [Task4] ${new Date().toISOString().split("T")[0]} 发现新AI技能...\n`);
const DAILY_LIMIT = 30;
let discovered = 0, skipped = 0, reviewed = 0, failed = 0;
for (const query of SEARCH_QUERIES) {
if (discovered >= DAILY_LIMIT) break;
console.log(`🔍 "${query}"`);
let repos;
try { repos = await searchRepos(query, 1000, 5); } catch (e) { console.error(` ❌ 搜索失败: ${e.message}`); failed++; continue; }
for (const repo of repos) {
if (discovered >= DAILY_LIMIT) break;
const ex = await prisma.skill.findFirst({ where: { sourceUrl: repo.html_url } });
if (ex) { console.log(` ⏭️ ${repo.full_name} 已存在`); skipped++; continue; }
const exP = await prisma.pendingSkill.findFirst({ where: { sourceUrl: repo.html_url } });
if (exP) { console.log(` ⏭️ ${repo.full_name} 待审核中`); skipped++; continue; }
const slug = genSlug(repo.full_name);
const slugEx = await prisma.skill.findUnique({ where: { slug } });
if (slugEx) { console.log(` ⏭️ ${repo.full_name} slug冲突`); skipped++; continue; }
if (!isValidAIProject(repo)) { skipped++; continue; }
console.log(`\n📦 [${discovered + 1}/${DAILY_LIMIT}] ${repo.full_name} (⭐${repo.stargazers_count})`);
try {
const [owner, name] = repo.full_name.split("/");
const readme = await getReadme(owner, name);
let reviewData = null;
try { reviewData = await genReview(repo.full_name, repo.description || "", repo.stargazers_count, repo.license?.spdx_id || "Unknown", readme); console.log(` ✅ 预评测: ${reviewData.overall}`); reviewed++; }
catch (e) { console.log(` ⚠️ 预评测失败: ${e.message}`); }
const categoryId = await guessCategory(repo.full_name, repo.description || "", reviewData?.tags || []);
const pendingSkill = await prisma.pendingSkill.create({
data: {
name: repo.full_name.substring(0, 100), slug, categoryId,
description: repo.description || "", sourceUrl: repo.html_url, sourceType: "github",
rating: reviewData?.overall || 0,
features: {
forks: repo.forks_count,
language: repo.language,
license: repo.license?.spdx_id || null,
topics: repo.topics || [],
stars: repo.stargazers_count,
reviewData: reviewData || null,
},
tags: reviewData?.tags || [],
status: "pending",
},
});
console.log(` ✅ PendingSkill #${pendingSkill.id}`);
await prisma.reviewGenerationLog.create({
data: { skillId: pendingSkill.id, status: "SUCCESS", prompt: `task4-discover: ${repo.full_name}`, response: reviewData ? JSON.stringify(reviewData) : null },
});
discovered++;
} catch (e) {
console.error(` ❌ 失败: ${e.message}`);
failed++;
try {
await prisma.reviewGenerationLog.create({
data: { skillId: 0, status: "FAILED", prompt: `task4-discover: ${repo.full_name}`, error: e.message?.substring(0, 500) || "Unknown" },
});
} catch {}
}
}
}
const endTime = new Date();
const duration = endTime.getTime() - startTime.getTime();
const result = { discovered, skipped, reviewed, failed, duration: `${(duration / 1000).toFixed(1)}s` };
console.log(`\n🏁 [Task4] 完成: 发现${discovered} / 跳过${skipped} / 评测${reviewed} / 失败${failed}`);
await prisma.taskLog.create({
data: {
taskKey: "task4-discover-skills",
taskName: "发现AI技能",
status: "completed",
startedAt: startTime,
finishedAt: endTime,
duration,
result,
error: failed > discovered * 2 ? `失败率过高: ${failed}/${discovered}` : null,
},
});
}
main().catch(async (e) => {
console.error("💥 异常:", e);
if (prisma) {
try {
await prisma.taskLog.create({
data: {
taskKey: "task4-discover-skills",
taskName: "发现AI技能",
status: "failed",
startedAt: new Date(),
finishedAt: new Date(),
duration: 0,
error: e.message,
},
});
} catch {}
await prisma.$disconnect();
}
process.exit(1);
}).finally(() => prisma.$disconnect());