438 lines
15 KiB
JavaScript
438 lines
15 KiB
JavaScript
import { PrismaClient } from "@prisma/client";
|
||
import { PrismaMariaDb } from "@prisma/adapter-mariadb";
|
||
import OpenAI from "openai";
|
||
import "dotenv/config";
|
||
import { withRetry } from "./lib/retry.mjs";
|
||
import { cacheGet, cacheSet, disconnectRedis } from "./lib/redis-cache.mjs";
|
||
|
||
const rawUrl = process.env.DATABASE_URL;
|
||
if (!rawUrl) { console.error("DATABASE_URL not set"); process.exit(1); }
|
||
const base = rawUrl.replace("mysql://", "mariadb://");
|
||
const sep = base.includes("?") ? "&" : "?";
|
||
const connectionString = `${base}${sep}connection_limit=20&pool_timeout=30`;
|
||
const adapter = new PrismaMariaDb(connectionString);
|
||
const prisma = new PrismaClient({ adapter });
|
||
|
||
const openai = new OpenAI({
|
||
apiKey: process.env.DEEPSEEK_API_KEY,
|
||
baseURL: "https://api.deepseek.com",
|
||
timeout: 120000,
|
||
maxRetries: 2,
|
||
});
|
||
|
||
async function fetchRSS(rssUrl, source) {
|
||
// 缓存 RSS 结果 30 分钟,避免重复请求
|
||
const cacheKey = `rss:${source}:${rssUrl}`;
|
||
const cached = await cacheGet(cacheKey);
|
||
if (cached) {
|
||
console.log(` 📦 ${source} RSS 命中缓存 (${cached.length} 条)`);
|
||
return cached;
|
||
}
|
||
|
||
try {
|
||
const response = await fetch(rssUrl, {
|
||
headers: { "User-Agent": "Mozilla/5.0 (compatible; NewsBot/1.0)" },
|
||
signal: AbortSignal.timeout(10000),
|
||
});
|
||
if (!response.ok) return [];
|
||
const xml = await response.text();
|
||
const items = [];
|
||
const itemRegex = /<item>([\s\S]*?)<\/item>/g;
|
||
let match;
|
||
while ((match = itemRegex.exec(xml)) !== null) {
|
||
const block = match[1];
|
||
const titleMatch = block.match(/<title><!\[CDATA\[([\s\S]*?)\]\]><\/title>/) || block.match(/<title>([\s\S]*?)<\/title>/);
|
||
const linkMatch = block.match(/<link>([\s\S]*?)<\/link>/);
|
||
const descMatch = block.match(/<description><!\[CDATA\[([\s\S]*?)\]\]><\/description>/) || block.match(/<description>([\s\S]*?)<\/description>/);
|
||
const pubDateMatch = block.match(/<pubDate>([\s\S]*?)<\/pubDate>/) || block.match(/<published>([\s\S]*?)<\/published>/);
|
||
if (titleMatch) {
|
||
items.push({
|
||
title: titleMatch[1].trim(),
|
||
url: linkMatch ? linkMatch[1].trim() : "",
|
||
content: descMatch ? descMatch[1].replace(/<[^>]+>/g, "").trim() : "",
|
||
pubDate: pubDateMatch ? pubDateMatch[1].trim() : "",
|
||
source,
|
||
});
|
||
}
|
||
}
|
||
|
||
// 缓存 30 分钟
|
||
if (items.length > 0) {
|
||
await cacheSet(cacheKey, items, 1800);
|
||
}
|
||
return items;
|
||
} catch (e) {
|
||
console.error(` ⚠️ ${source} RSS获取失败: ${e.message}`);
|
||
return [];
|
||
}
|
||
}
|
||
|
||
async function fetch36krAI() {
|
||
return fetchRSS("https://36kr.com/feed", "36kr");
|
||
}
|
||
|
||
async function fetchITHomeAI() {
|
||
return fetchRSS("https://www.ithome.com/rss/", "IT之家");
|
||
}
|
||
|
||
async function fetchOSChinaAI() {
|
||
return fetchRSS("https://www.oschina.net/news/rss", "开源中国");
|
||
}
|
||
|
||
async function fetchGoogleNewsRSS(query, lang = "zh-CN", country = "CN") {
|
||
const url = `https://news.google.com/rss/search?q=${encodeURIComponent(query)}&hl=${lang}&gl=${country}&ceid=${country}:${lang === "zh-CN" ? "zh-Hans" : "en"}`;
|
||
const results = await fetchRSS(url, `GoogleNews-${lang}`);
|
||
return results;
|
||
}
|
||
|
||
async function fetchHackerNewsTop() {
|
||
try {
|
||
const topRes = await fetch("https://hacker-news.firebaseio.com/v0/topstories.json");
|
||
const topIds = await topRes.json();
|
||
const aiIds = topIds.slice(0, 60);
|
||
const items = [];
|
||
for (const id of aiIds) {
|
||
try {
|
||
const res = await fetch(`https://hacker-news.firebaseio.com/v0/item/${id}.json`);
|
||
const story = await res.json();
|
||
if (!story || !story.title) continue;
|
||
const title = story.title.toLowerCase();
|
||
if (
|
||
title.includes("ai") ||
|
||
title.includes("llm") ||
|
||
title.includes("gpt") ||
|
||
title.includes("openai") ||
|
||
title.includes("deepseek") ||
|
||
title.includes("machine learning") ||
|
||
title.includes("neural") ||
|
||
title.includes("model") ||
|
||
title.includes("claude") ||
|
||
title.includes("gemini") ||
|
||
title.includes("agent") ||
|
||
title.includes("rag") ||
|
||
title.includes("transformer")
|
||
) {
|
||
items.push({
|
||
title: story.title,
|
||
url: story.url || `https://news.ycombinator.com/item?id=${story.id}`,
|
||
content: "",
|
||
pubDate: new Date(story.time * 1000).toISOString(),
|
||
});
|
||
}
|
||
if (items.length >= 10) break;
|
||
} catch {}
|
||
}
|
||
return items;
|
||
} catch (e) {
|
||
console.error(` ⚠️ HackerNews获取失败: ${e.message}`);
|
||
return [];
|
||
}
|
||
}
|
||
|
||
async function searchAllNews() {
|
||
console.log("🔍 第一步:搜索真实AI新闻...");
|
||
|
||
const allResults = [];
|
||
const seenTitles = new Set();
|
||
|
||
function addResults(results, source) {
|
||
for (const r of results) {
|
||
const key = r.title.substring(0, 30).toLowerCase();
|
||
if (!seenTitles.has(key) && r.title) {
|
||
seenTitles.add(key);
|
||
allResults.push({ ...r, source });
|
||
}
|
||
}
|
||
}
|
||
|
||
console.log(" 📡 36kr...");
|
||
const kr36 = await fetch36krAI();
|
||
addResults(kr36, "36kr");
|
||
console.log(` 获取 ${kr36.length} 条`);
|
||
|
||
console.log(" 📡 IT之家...");
|
||
const ithome = await fetchITHomeAI();
|
||
addResults(ithome, "IT之家");
|
||
console.log(` 获取 ${ithome.length} 条`);
|
||
|
||
console.log(" 📡 开源中国...");
|
||
const oschina = await fetchOSChinaAI();
|
||
addResults(oschina, "开源中国");
|
||
console.log(` 获取 ${oschina.length} 条`);
|
||
|
||
console.log(" 📡 Google News (中文)...");
|
||
const zhNews = await fetchGoogleNewsRSS("人工智能 AI 最新", "zh-CN", "CN");
|
||
addResults(zhNews, "GoogleNews-CN");
|
||
console.log(` 获取 ${zhNews.length} 条`);
|
||
|
||
console.log(" 📡 Google News (English)...");
|
||
const enNews = await fetchGoogleNewsRSS("artificial intelligence AI latest news", "en", "US");
|
||
addResults(enNews, "GoogleNews-EN");
|
||
console.log(` 获取 ${enNews.length} 条`);
|
||
|
||
console.log(" 📡 HackerNews Top AI...");
|
||
const hnNews = await fetchHackerNewsTop();
|
||
addResults(hnNews, "HackerNews");
|
||
console.log(` 获取 ${hnNews.length} 条`);
|
||
|
||
const AI_KEYWORDS = [
|
||
"ai", "人工智能", "大模型", "llm", "gpt", "chatgpt", "openai", "deepseek",
|
||
"claude", "gemini", "通义", "文心", "智谱", "kimi", "豆包", "百川",
|
||
"diffusion", "stable diffusion", "midjourney", "sora",
|
||
"机器学习", "深度学习", "神经网络", "transformer",
|
||
"copilot", "agent", "rag", "aigc", "生成式",
|
||
"自动驾驶", "智能体", "多模态", "语音识别",
|
||
];
|
||
|
||
function isAIRelated(title) {
|
||
const lower = title.toLowerCase();
|
||
return AI_KEYWORDS.some((kw) => lower.includes(kw));
|
||
}
|
||
|
||
const recentResults = allResults
|
||
.filter((r) => {
|
||
if (!isAIRelated(r.title)) return false;
|
||
if (!r.pubDate) return true;
|
||
try {
|
||
const pubTime = new Date(r.pubDate).getTime();
|
||
const twoDaysAgo = Date.now() - 2 * 24 * 60 * 60 * 1000;
|
||
return pubTime > twoDaysAgo;
|
||
} catch {
|
||
return true;
|
||
}
|
||
});
|
||
|
||
const finalResults = recentResults.length > 0 ? recentResults : allResults.slice(0, 30);
|
||
console.log(` ✅ 共获取 ${allResults.length} 条,近2天 ${recentResults.length} 条,使用 ${finalResults.length} 条`);
|
||
return finalResults;
|
||
}
|
||
|
||
const FORMAT_PROMPT = `你是AI行业资深分析师。以下是从互联网搜索到的今日真实AI新闻素材,请基于这些真实素材整理一份AI日报。
|
||
|
||
严格要求:
|
||
1. 只能基于提供的素材内容,绝对不能编造任何新闻!如果素材不够6条,就只输出有素材支持的条目
|
||
2. 每条新闻必须保留原始来源URL
|
||
3. 标题简洁有力(20字以内),摘要精炼(50字以内),详情深入(100-200字)
|
||
4. 优先关注中国AI行业动态,兼顾全球重要事件
|
||
5. 分类标注规则(必须使用以下key值):
|
||
- model:模型前沿(LLM/多模态新模型发布、架构创新、训练方法突破)
|
||
- product:产品落地(AI产品发布/更新、C端应用上线、B端解决方案、新功能)
|
||
- opensource:开源生态(开源项目动态、框架更新、开源社区大事件)
|
||
- business:商业产业(投融资、收购并购、财报营收、公司战略、行业趋势)
|
||
- policy:政策治理(AI法规政策、监管动态、伦理治理、国际AI治理)
|
||
- insight:深度洞察(行业领袖观点、深度分析报告、学术研究前沿)
|
||
|
||
新闻素材:
|
||
{newsData}
|
||
|
||
返回纯JSON格式:
|
||
{
|
||
"title": "AI日报:{date} 核心要点概览",
|
||
"summary": "今日AI行业整体趋势概述(100字以内)",
|
||
"items": [
|
||
{
|
||
"category": "model",
|
||
"title": "新闻标题",
|
||
"summary": "一句话摘要",
|
||
"detail": "详细内容描述",
|
||
"sourceUrl": "原始来源URL",
|
||
"hotScore": 85
|
||
}
|
||
]
|
||
}
|
||
|
||
hotScore热度评分规则(0-100):
|
||
- 90-100:全球头条级事件
|
||
- 70-89:行业重要事件
|
||
- 50-69:值得关注的动态
|
||
- 30-49:一般性新闻`;
|
||
|
||
function extractJSON(str) {
|
||
const trimmed = str.trim();
|
||
const codeBlockMatch = trimmed.match(/```(?:json)?\s*([\s\S]*?)```/);
|
||
let jsonStr = codeBlockMatch ? codeBlockMatch[1].trim() : trimmed;
|
||
const braceMatch = jsonStr.match(/\{[\s\S]*\}/);
|
||
if (braceMatch) {
|
||
try { return JSON.parse(braceMatch[0]); } catch {}
|
||
}
|
||
const arrayMatch = jsonStr.match(/\[[\s\S]*\]/);
|
||
if (arrayMatch) {
|
||
try { return JSON.parse(arrayMatch[0]); } catch {}
|
||
}
|
||
return JSON.parse(jsonStr);
|
||
}
|
||
|
||
async function generateDailyNews() {
|
||
const now = new Date();
|
||
const cnOffset = 8 * 60 * 60 * 1000;
|
||
const cnTime = new Date(now.getTime() + cnOffset);
|
||
const dateStr = cnTime.toLocaleDateString("zh-CN", {
|
||
year: "numeric",
|
||
month: "long",
|
||
day: "numeric",
|
||
timeZone: "Asia/Shanghai",
|
||
});
|
||
const dateISO = cnTime.toISOString().split("T")[0];
|
||
|
||
console.log(`📰 开始生成 ${dateStr} 的AI日报...`);
|
||
|
||
const existing = await prisma.dailyReport.findFirst({
|
||
where: { date: new Date(dateISO) },
|
||
});
|
||
if (existing) {
|
||
console.log(`⚠️ ${dateISO} 已有日报 (ID:${existing.id}),先删除旧版`);
|
||
await prisma.dailyReport.delete({ where: { id: existing.id } });
|
||
}
|
||
|
||
const newsResults = await searchAllNews();
|
||
if (newsResults.length === 0) {
|
||
throw new Error("未搜索到任何新闻素材,无法生成日报");
|
||
}
|
||
|
||
const newsDataText = newsResults
|
||
.slice(0, 14)
|
||
.map((r, i) => `[${i + 1}] 标题: ${r.title}\n 内容: ${r.content || "(无摘要)"}\n 来源: ${r.url}\n 时间: ${r.pubDate || "未知"}`)
|
||
.join("\n\n");
|
||
|
||
console.log("🤖 第二步:调用 DeepSeek 整理新闻格式...");
|
||
const prompt = FORMAT_PROMPT
|
||
.replace(/\{newsData\}/g, newsDataText)
|
||
.replace(/\{date\}/g, dateStr);
|
||
|
||
const completion = await withRetry(async () => {
|
||
return await openai.chat.completions.create({
|
||
model: process.env.DEEPSEEK_MODEL || "deepseek-v4-pro",
|
||
messages: [{ role: "user", content: prompt }],
|
||
temperature: 0.3,
|
||
// DeepSeek V4 reasoning token 计入 max_tokens 预算:预算不足会导致 content 为空(finish_reason=length)
|
||
max_tokens: 16384,
|
||
}, { signal: AbortSignal.timeout(180000) });
|
||
}, { maxRetries: 3, label: "DeepSeek News API" });
|
||
|
||
const u = completion.usage;
|
||
if (u) {
|
||
console.log(` ℹ️ tokens: prompt=${u.prompt_tokens} completion=${u.completion_tokens} reasoning=${u.completion_tokens_details?.reasoning_tokens ?? "n/a"} finish=${completion.choices[0]?.finish_reason}`);
|
||
}
|
||
const content = completion.choices[0]?.message?.content;
|
||
if (!content) {
|
||
throw new Error(`DeepSeek 返回内容为空(finish_reason=${completion.choices[0]?.finish_reason},completion_tokens=${u?.completion_tokens},疑似 reasoning 占满 max_tokens)`);
|
||
}
|
||
|
||
let parsed;
|
||
try {
|
||
parsed = extractJSON(content);
|
||
} catch (e) {
|
||
console.error("❌ JSON解析失败,原始内容前500字:");
|
||
console.error(content.substring(0, 500));
|
||
throw e;
|
||
}
|
||
|
||
if (!parsed.items || !Array.isArray(parsed.items) || parsed.items.length === 0) {
|
||
throw new Error("整理后的日报没有有效的新闻条目");
|
||
}
|
||
|
||
console.log(`✅ 整理成功,共 ${parsed.items.length} 条新闻`);
|
||
|
||
const report = await prisma.dailyReport.create({
|
||
data: {
|
||
date: new Date(dateISO),
|
||
title: parsed.title || `AI日报:${dateStr} 核心要点概览`,
|
||
summary: parsed.summary || "",
|
||
hotCount: parsed.items.length,
|
||
status: "published",
|
||
items: {
|
||
create: parsed.items.map((item, index) => ({
|
||
sortOrder: index,
|
||
category: item.category || "product",
|
||
title: item.title || "未命名新闻",
|
||
summary: item.summary || "",
|
||
detail: item.detail || "",
|
||
sourceUrl: item.sourceUrl || "",
|
||
imageUrl: item.imageUrl || "",
|
||
hotScore: item.hotScore || 50,
|
||
})),
|
||
},
|
||
},
|
||
include: { items: true },
|
||
});
|
||
|
||
console.log(`\n🎉 日报生成完成!`);
|
||
console.log(` ID: ${report.id}`);
|
||
console.log(` 标题: ${report.title}`);
|
||
console.log(` 日期: ${dateISO}`);
|
||
console.log(` 条目数: ${report.items.length}`);
|
||
console.log(` 状态: ${report.status}`);
|
||
|
||
report.items.forEach((item, i) => {
|
||
console.log(` ${i + 1}. [${item.category}] ${item.title} (热度:${item.hotScore})`);
|
||
if (item.sourceUrl) {
|
||
console.log(` 来源: ${item.sourceUrl}`);
|
||
}
|
||
});
|
||
|
||
return report;
|
||
}
|
||
|
||
async function generateWithRetry() {
|
||
for (let attempt = 0; attempt < 2; attempt++) {
|
||
try {
|
||
return await generateDailyNews();
|
||
} catch (e) {
|
||
const msg = String(e.message || e);
|
||
// 偶发中止/超时/上游 5xx 才重试;业务类错误直接抛出
|
||
const retryable = /abort|timeout|ETIMEDOUT|ECONNRESET|socket|fetch failed|502|503|429/i.test(msg);
|
||
if (!retryable || attempt === 1) throw e;
|
||
console.warn(`⚠️ 日报生成中止(${msg.slice(0, 80)}),30s 后重试 (${attempt + 1}/2)...`);
|
||
await new Promise((r) => setTimeout(r, 30000));
|
||
}
|
||
}
|
||
throw new Error("unreachable");
|
||
}
|
||
|
||
async function main() {
|
||
if (!process.env.DEEPSEEK_API_KEY) throw new Error("缺少环境变量: DEEPSEEK_API_KEY");
|
||
if (!process.env.DATABASE_URL) throw new Error("缺少环境变量: DATABASE_URL");
|
||
|
||
const startTime = new Date();
|
||
try {
|
||
const report = await generateWithRetry();
|
||
const endTime = new Date();
|
||
const duration = endTime.getTime() - startTime.getTime();
|
||
await prisma.taskLog.create({
|
||
data: {
|
||
taskKey: "daily-news",
|
||
taskName: "AI日报生成",
|
||
status: "completed",
|
||
startedAt: startTime,
|
||
finishedAt: endTime,
|
||
duration,
|
||
result: { reportId: report.id, itemCount: report.items.length, title: report.title },
|
||
},
|
||
});
|
||
} catch (error) {
|
||
console.error("❌ 日报生成失败:", error.message);
|
||
try {
|
||
await prisma.taskLog.create({
|
||
data: {
|
||
taskKey: "daily-news",
|
||
taskName: "AI日报生成",
|
||
status: "failed",
|
||
startedAt: startTime,
|
||
finishedAt: new Date(),
|
||
duration: new Date().getTime() - startTime.getTime(),
|
||
error: error.message,
|
||
},
|
||
});
|
||
} catch {}
|
||
process.exit(1);
|
||
} finally {
|
||
await prisma.$disconnect();
|
||
await disconnectRedis();
|
||
}
|
||
}
|
||
|
||
main()
|
||
.then(() => { console.log("Done"); process.exit(0); })
|
||
.catch((e) => { console.error(e); process.exit(1); })
|
||
.finally(() => prisma.$disconnect());
|