feat: 首次推送到Gitea - 完整项目代码 + 安全加固 + 知识库
This commit is contained in:
+126
-76
@@ -19,7 +19,7 @@ const octokit = new Octokit({
|
||||
|
||||
const openai = new OpenAI({
|
||||
apiKey: process.env.DEEPSEEK_API_KEY,
|
||||
baseURL: "https://api.deepseek.com/v1",
|
||||
baseURL: "https://api.deepseek.com",
|
||||
timeout: 120000,
|
||||
maxRetries: 2,
|
||||
});
|
||||
@@ -51,30 +51,41 @@ const REVIEW_PROMPT = `你是一个GitHub开源项目评测专家。请基于以
|
||||
- 每个维度的detail需严格控制在80-120字
|
||||
- 标签从以下类别中选择最匹配的3-5个:LLM框架, Agent框架, RAG, 提示工程, 多模态, 代码助手, 模型训练, 数据工程, 部署工具, 评测工具, 安全对齐, 其他`;
|
||||
|
||||
// 发现策略:查询池按天轮转 + 多排序/翻页 + 滚动时间窗
|
||||
// 旧实现固定 10 条查询 × sort:stars desc × per_page 5 → 每天返回同样 5 个高星仓库,全部"已存在" → 连续多日 0 产出
|
||||
const SEARCH_QUERIES = [
|
||||
"ai agent framework",
|
||||
"llm framework",
|
||||
"rag framework",
|
||||
"prompt engineering tool",
|
||||
"ai code assistant",
|
||||
"multimodal ai",
|
||||
"ai model training",
|
||||
"ai deployment tool",
|
||||
"ai evaluation benchmark",
|
||||
"ai safety alignment",
|
||||
"ai agent framework", "llm framework", "rag framework", "prompt engineering tool",
|
||||
"ai code assistant", "multimodal ai", "ai model training", "ai deployment tool",
|
||||
"ai evaluation benchmark", "ai safety alignment",
|
||||
"llm application", "ai agent tool", "mcp server", "ai workflow automation",
|
||||
"text to speech ai", "ai image generation", "ai video generation", "vector database",
|
||||
"ai coding agent", "ai data analysis", "ai browser automation", "llm observability",
|
||||
"ai document processing", "ai speech recognition",
|
||||
];
|
||||
|
||||
const QUERIES_PER_RUN = 5; // 每次运行取 5 条查询(按天轮转,4 天覆盖全池)
|
||||
const PAGE_SIZE = 20; // 单个查询每页 20 条
|
||||
const MAX_PAGES = 2; // 热门存量最多翻 2 页
|
||||
const PUSHED_MONTHS = 18; // 只看近 18 个月有提交的仓库
|
||||
const MIN_STARS = 1000; // 存量热门门槛
|
||||
const NEW_MIN_STARS = 300; // 新项目门槛(近 12 个月创建)
|
||||
const NEW_CREATED_MONTHS = 12;
|
||||
|
||||
const CATEGORY_KEYWORDS = {
|
||||
"AI Agent": ["agent", "autonomous", "workflow", "tool call"],
|
||||
"代码生成": ["code", "programming", "developer", "copilot"],
|
||||
"测试": ["test", "testing", "qa", "quality"],
|
||||
"MCP 服务器": ["mcp", "model context protocol"],
|
||||
"RAG 系统": ["rag", "retrieval", "vector", "embedding"],
|
||||
"Prompt 工程": ["prompt", "prompting", "instruction"],
|
||||
"Agent 编排": ["agent", "autonomous", "multi-agent", "tool call", "workflow"],
|
||||
"代码生成": ["code", "coding", "programming", "developer", "copilot"],
|
||||
"测试": ["test", "testing", "qa", "benchmark", "evaluation"],
|
||||
"文档": ["documentation", "docs", "readme"],
|
||||
"安全": ["security", "vulnerability", "safety", "alignment"],
|
||||
"DevOps与部署": ["deployment", "deploy", "devops", "ci/cd", "docker", "kubernetes"],
|
||||
"RAG": ["rag", "retrieval", "vector", "embedding"],
|
||||
"AI/ML工程": ["training", "fine-tun", "model", "machine learning", "ml"],
|
||||
"多模态": ["multimodal", "vision", "image", "video", "audio"],
|
||||
"Prompt工程": ["prompt", "prompting", "instruction"],
|
||||
"DevOps 与部署": ["deployment", "deploy", "devops", "ci/cd", "docker", "kubernetes"],
|
||||
"模型训练": ["fine-tun", "finetune", "lora", "training", "pretrain"],
|
||||
"模型部署": ["inference", "serving", "vllm", "ollama", "quantization"],
|
||||
"AI/ML 工程": ["machine learning", "deep learning", "pytorch", "tensorflow", "llm", "model"],
|
||||
"架构与设计": ["architecture", "design pattern", "spec"],
|
||||
"开发工具": ["cli", "sdk", "toolkit", "browser automation", "playwright"],
|
||||
};
|
||||
|
||||
const NON_AI_KEYWORDS = [
|
||||
@@ -168,12 +179,45 @@ function genSlug(name) {
|
||||
return name.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "").substring(0, 100);
|
||||
}
|
||||
|
||||
async function searchRepos(query, minStars, max) {
|
||||
const q = `${query} stars:>${minStars} pushed:>2024-01-01`;
|
||||
function monthsAgo(n) {
|
||||
return new Date(Date.now() - n * 30 * 24 * 3600 * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
async function searchRepos(query, { minStars, sort, page, createdAfter }) {
|
||||
const parts = [query, `stars:>${minStars}`, `pushed:>${monthsAgo(PUSHED_MONTHS)}`];
|
||||
if (createdAfter) parts.push(`created:>${createdAfter}`);
|
||||
const q = parts.join(" ");
|
||||
const { data } = await withRetry(async () => {
|
||||
return await octokit.rest.search.repos({ q, sort: "stars", order: "desc", per_page: max });
|
||||
}, { maxRetries: 3, label: `GitHub Search "${query}"` });
|
||||
return data.items;
|
||||
return await octokit.rest.search.repos({ q, sort, order: "desc", per_page: PAGE_SIZE, page });
|
||||
}, { maxRetries: 3, label: `GitHub Search "${q}"` });
|
||||
return data.items || [];
|
||||
}
|
||||
|
||||
// 对每条查询做两轮搜索:① 存量热门(按 updated 排序 + 翻页,结果随时间自然浮动)
|
||||
// ② 近 12 个月创建的新兴项目(星标 300+),避免只看到固定的一批老牌高星仓库
|
||||
async function collectCandidates(queries) {
|
||||
const map = new Map();
|
||||
let searchFailed = 0;
|
||||
for (const query of queries) {
|
||||
const passes = [
|
||||
{ minStars: MIN_STARS, sort: "updated", pages: Array.from({ length: MAX_PAGES }, (_, i) => i + 1) },
|
||||
{ minStars: NEW_MIN_STARS, sort: "stars", createdAfter: monthsAgo(NEW_CREATED_MONTHS), pages: [1] },
|
||||
];
|
||||
for (const pass of passes) {
|
||||
for (const page of pass.pages) {
|
||||
try {
|
||||
const items = await searchRepos(query, { minStars: pass.minStars, sort: pass.sort, page, createdAfter: pass.createdAfter });
|
||||
for (const repo of items) {
|
||||
if (!map.has(repo.full_name)) map.set(repo.full_name, repo);
|
||||
}
|
||||
} catch (e) {
|
||||
console.error(` ❌ 搜索失败 "${query}" (${pass.sort} p${page}): ${e.message}`);
|
||||
searchFailed++;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return { candidates: [...map.values()], searchFailed };
|
||||
}
|
||||
|
||||
async function getReadme(owner, repo) {
|
||||
@@ -192,7 +236,11 @@ async function genReview(repoName, desc, stars, license, readme) {
|
||||
.replace("{readmeSummary}", readme.substring(0, 1500));
|
||||
const { choices } = await withRetry(async () => {
|
||||
return await openai.chat.completions.create({
|
||||
model: process.env.DEEPSEEK_MODEL || "deepseek-v4-pro", messages: [{ role: "user", content: prompt }], temperature: 0.3,
|
||||
model: process.env.DEEPSEEK_MODEL || "deepseek-v4-pro",
|
||||
messages: [{ role: "user", content: prompt }],
|
||||
temperature: 0.3,
|
||||
// DeepSeek V4 的 reasoning token 计入 max_tokens,预算不足会返回空 content(历史踩坑)
|
||||
max_tokens: 16384,
|
||||
}, { signal: AbortSignal.timeout(120000) });
|
||||
}, { maxRetries: 3, label: "DeepSeek Review API" });
|
||||
const content = choices[0]?.message?.content;
|
||||
@@ -218,64 +266,66 @@ async function main() {
|
||||
const DAILY_LIMIT = 30;
|
||||
let discovered = 0, skipped = 0, reviewed = 0, failed = 0;
|
||||
|
||||
for (const query of SEARCH_QUERIES) {
|
||||
const dayIndex = Math.floor(Date.now() / 86400000);
|
||||
const startIdx = (dayIndex * QUERIES_PER_RUN) % SEARCH_QUERIES.length;
|
||||
const todaysQueries = Array.from({ length: QUERIES_PER_RUN }, (_, i) => SEARCH_QUERIES[(startIdx + i) % SEARCH_QUERIES.length]);
|
||||
console.log(`🔍 本轮查询 (${todaysQueries.length}/${SEARCH_QUERIES.length},按天轮转): ${todaysQueries.join(" / ")}`);
|
||||
|
||||
const { candidates, searchFailed } = await collectCandidates(todaysQueries);
|
||||
failed += searchFailed;
|
||||
console.log(` 候选仓库 ${candidates.length} 个(已去重)\n`);
|
||||
|
||||
for (const repo of candidates) {
|
||||
if (discovered >= DAILY_LIMIT) break;
|
||||
console.log(`🔍 "${query}"`);
|
||||
let repos;
|
||||
try { repos = await searchRepos(query, 1000, 5); } catch (e) { console.error(` ❌ 搜索失败: ${e.message}`); failed++; continue; }
|
||||
const ex = await prisma.skill.findFirst({ where: { sourceUrl: repo.html_url } });
|
||||
if (ex) { skipped++; continue; }
|
||||
const exP = await prisma.pendingSkill.findFirst({ where: { sourceUrl: repo.html_url } });
|
||||
if (exP) { skipped++; continue; }
|
||||
const slug = genSlug(repo.full_name);
|
||||
const slugEx = await prisma.skill.findUnique({ where: { slug } });
|
||||
if (slugEx) { skipped++; continue; }
|
||||
if (!isValidAIProject(repo)) { skipped++; continue; }
|
||||
|
||||
for (const repo of repos) {
|
||||
if (discovered >= DAILY_LIMIT) break;
|
||||
const ex = await prisma.skill.findFirst({ where: { sourceUrl: repo.html_url } });
|
||||
if (ex) { console.log(` ⏭️ ${repo.full_name} 已存在`); skipped++; continue; }
|
||||
const exP = await prisma.pendingSkill.findFirst({ where: { sourceUrl: repo.html_url } });
|
||||
if (exP) { console.log(` ⏭️ ${repo.full_name} 待审核中`); skipped++; continue; }
|
||||
const slug = genSlug(repo.full_name);
|
||||
const slugEx = await prisma.skill.findUnique({ where: { slug } });
|
||||
if (slugEx) { console.log(` ⏭️ ${repo.full_name} slug冲突`); skipped++; continue; }
|
||||
if (!isValidAIProject(repo)) { skipped++; continue; }
|
||||
console.log(`\n📦 [${discovered + 1}/${DAILY_LIMIT}] ${repo.full_name} (⭐${repo.stargazers_count})`);
|
||||
try {
|
||||
const [owner, name] = repo.full_name.split("/");
|
||||
const readme = await getReadme(owner, name);
|
||||
let reviewData = null;
|
||||
try { reviewData = await genReview(repo.full_name, repo.description || "", repo.stargazers_count, repo.license?.spdx_id || "Unknown", readme); console.log(` ✅ 预评测: ${reviewData.overall}`); reviewed++; }
|
||||
catch (e) { console.log(` ⚠️ 预评测失败: ${e.message}`); }
|
||||
|
||||
console.log(`\n📦 [${discovered + 1}/${DAILY_LIMIT}] ${repo.full_name} (⭐${repo.stargazers_count})`);
|
||||
try {
|
||||
const [owner, name] = repo.full_name.split("/");
|
||||
const readme = await getReadme(owner, name);
|
||||
let reviewData = null;
|
||||
try { reviewData = await genReview(repo.full_name, repo.description || "", repo.stargazers_count, repo.license?.spdx_id || "Unknown", readme); console.log(` ✅ 预评测: ${reviewData.overall}`); reviewed++; }
|
||||
catch (e) { console.log(` ⚠️ 预评测失败: ${e.message}`); }
|
||||
|
||||
const categoryId = await guessCategory(repo.full_name, repo.description || "", reviewData?.tags || []);
|
||||
const pendingSkill = await prisma.pendingSkill.create({
|
||||
data: {
|
||||
name: repo.full_name.substring(0, 100), slug, categoryId,
|
||||
description: repo.description || "", sourceUrl: repo.html_url, sourceType: "github",
|
||||
rating: reviewData?.overall || 0,
|
||||
features: {
|
||||
forks: repo.forks_count,
|
||||
language: repo.language,
|
||||
license: repo.license?.spdx_id || null,
|
||||
topics: repo.topics || [],
|
||||
stars: repo.stargazers_count,
|
||||
reviewData: reviewData || null,
|
||||
},
|
||||
tags: reviewData?.tags || [],
|
||||
status: "pending",
|
||||
const categoryId = await guessCategory(repo.full_name, repo.description || "", reviewData?.tags || []);
|
||||
const pendingSkill = await prisma.pendingSkill.create({
|
||||
data: {
|
||||
name: repo.full_name.substring(0, 100), slug, categoryId,
|
||||
description: repo.description || "", sourceUrl: repo.html_url, sourceType: "github",
|
||||
rating: reviewData?.overall || 0,
|
||||
features: {
|
||||
forks: repo.forks_count,
|
||||
language: repo.language,
|
||||
license: repo.license?.spdx_id || null,
|
||||
topics: repo.topics || [],
|
||||
stars: repo.stargazers_count,
|
||||
reviewData: reviewData || null,
|
||||
},
|
||||
});
|
||||
console.log(` ✅ PendingSkill #${pendingSkill.id}`);
|
||||
tags: reviewData?.tags || [],
|
||||
status: "pending",
|
||||
},
|
||||
});
|
||||
console.log(` ✅ PendingSkill #${pendingSkill.id}`);
|
||||
|
||||
await prisma.reviewGenerationLog.create({
|
||||
data: { skillId: 0, status: "SUCCESS", prompt: `task4-discover: ${repo.full_name}`, response: reviewData ? JSON.stringify(reviewData) : null },
|
||||
});
|
||||
discovered++;
|
||||
} catch (e) {
|
||||
console.error(` ❌ 失败: ${e.message}`);
|
||||
failed++;
|
||||
try {
|
||||
await prisma.reviewGenerationLog.create({
|
||||
data: { skillId: 0, status: "SUCCESS", prompt: `task4-discover: ${repo.full_name}`, response: reviewData ? JSON.stringify(reviewData) : null },
|
||||
data: { skillId: 0, status: "FAILED", prompt: `task4-discover: ${repo.full_name}`, error: e.message?.substring(0, 500) || "Unknown" },
|
||||
});
|
||||
discovered++;
|
||||
} catch (e) {
|
||||
console.error(` ❌ 失败: ${e.message}`);
|
||||
failed++;
|
||||
try {
|
||||
await prisma.reviewGenerationLog.create({
|
||||
data: { skillId: 0, status: "FAILED", prompt: `task4-discover: ${repo.full_name}`, error: e.message?.substring(0, 500) || "Unknown" },
|
||||
});
|
||||
} catch {}
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user