Files
zhuiguang-ai/scripts/update-knowledge.mjs

365 lines
14 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
/**
* update-knowledge.mjs — 开发完成后自动记录与迭代脚本
*
* 捕获三类信息并沉淀知识库(幂等,可重复运行):
* 1. 能力特征 → .trae/knowledge/capabilities.json (代码结构 / 依赖 / 任务 / 角色 全貌快照)
* 2. 开发过程 → .trae/knowledge/dev_process.jsonl (按提交追加,不覆盖历史)
* 3. 模型成果 → .trae/knowledge/artifacts.jsonl (本次交付物:迁移 / 脚本 / 提交统计)
* 并把上述事实增量 upsert 进知识图谱:
* - .trae/knowledge_graph.json (渲染版,权威)
* - .trae/knowledge_graph.jsonl (MCP Knowledge Graph Memory 读取的源文件)
*
* 知识目录解析顺序:
* 1. <repo>/.trae (所有 .trae 内容均在代码仓库内,跟着 git 走)
*
* 用法:
* node scripts/update-knowledge.mjs # 正常写入
* node scripts/update-knowledge.mjs --dry-run # 只预览,不写文件
* node scripts/update-knowledge.mjs --quiet # 静默(post-commit 钩子用)
*
* 设计约束:
* - 永不删除实体:同名实体整条替换(其 observations 是派生数据),未出现的旧实体原样保留
* - 任何扫描失败只告警不中断,最终仍写出文件
*/
import fs from "fs";
import path from "path";
import { execFileSync } from "child_process";
import { fileURLToPath } from "url";
import { createLogger } from "./lib/logger.mjs";
const log = createLogger("update-knowledge");
const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
const DRY_RUN = process.argv.includes("--dry-run");
const QUIET = process.argv.includes("--quiet");
const PROJECT_NAME = "追光AI";
const CAP_ENTITY = "追光AI:能力快照";
const TODAY = new Date().toISOString().slice(0, 10);
const NOW = new Date().toISOString().slice(0, 19);
const DEV_RECORD_ENTITY = `开发记录:${TODAY}`;
/* ------------------------------------------------------------------ 路径 */
const TRAE_DIR = path.join(REPO_ROOT, ".trae");
const KNOW_DIR = path.join(TRAE_DIR, "knowledge");
const KG_JSON = path.join(TRAE_DIR, "knowledge_graph.json");
const KG_JSONL = path.join(TRAE_DIR, "knowledge_graph.jsonl");
const CAP_FILE = path.join(KNOW_DIR, "capabilities.json");
const PROC_FILE = path.join(KNOW_DIR, "dev_process.jsonl");
const ART_FILE = path.join(KNOW_DIR, "artifacts.jsonl");
/* ------------------------------------------------------------- 基础工具 */
function readJson(file, fallback = null) {
try {
return JSON.parse(fs.readFileSync(file, "utf8"));
} catch (err) {
if (fs.existsSync(file)) log.warn("读取 JSON 失败", { file, error: err.message });
return fallback;
}
}
function writeJson(file, data) {
if (DRY_RUN) return;
fs.mkdirSync(path.dirname(file), { recursive: true });
fs.writeFileSync(file, JSON.stringify(data, null, 2) + "\n", "utf8");
}
function appendJsonl(file, rows) {
if (DRY_RUN || rows.length === 0) return;
fs.mkdirSync(path.dirname(file), { recursive: true });
fs.appendFileSync(file, rows.map((r) => JSON.stringify(r)).join("\n") + "\n", "utf8");
}
function readJsonl(file) {
if (!fs.existsSync(file)) return [];
return fs
.readFileSync(file, "utf8")
.split("\n")
.map((l) => l.trim())
.filter(Boolean)
.map((l) => {
try {
return JSON.parse(l);
} catch {
return null;
}
})
.filter(Boolean);
}
function walk(dir, filter) {
const out = [];
if (!fs.existsSync(dir)) return out;
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) out.push(...walk(full, filter));
else if (filter(full)) out.push(full);
}
return out;
}
function rel(p) {
return path.relative(REPO_ROOT, p).replace(/\\/g, "/");
}
function git(args) {
try {
return execFileSync("git", args, {
cwd: REPO_ROOT,
encoding: "utf8",
timeout: 15000,
stdio: ["ignore", "pipe", "ignore"],
}).trim();
} catch {
return "";
}
}
/* ------------------------------------------------------ 1. 能力特征扫描 */
function scanCapabilities() {
const apiRoutes = walk(path.join(REPO_ROOT, "src", "app", "api"), (f) => f.endsWith("route.ts"));
const pages = walk(path.join(REPO_ROOT, "src", "app"), (f) => f.endsWith("page.tsx"));
const components = walk(path.join(REPO_ROOT, "src", "components"), (f) => f.endsWith(".tsx"));
const hooks = walk(path.join(REPO_ROOT, "src", "hooks"), (f) => /\.(ts|tsx)$/.test(f));
const libFiles = walk(path.join(REPO_ROOT, "src", "lib"), (f) => /\.(ts|tsx)$/.test(f));
const scripts = walk(path.join(REPO_ROOT, "scripts"), (f) => /\.(mjs|sh)$/.test(f));
const migrations = fs.existsSync(path.join(REPO_ROOT, "prisma", "migrations"))
? fs
.readdirSync(path.join(REPO_ROOT, "prisma", "migrations"), { withFileTypes: true })
.filter((d) => d.isDirectory())
.map((d) => d.name)
.sort()
: [];
const schema = fs.existsSync(path.join(REPO_ROOT, "prisma", "schema.prisma"))
? fs.readFileSync(path.join(REPO_ROOT, "prisma", "schema.prisma"), "utf8")
: "";
const models = [...schema.matchAll(/^model\s+(\w+)/gm)].map((m) => m[1]);
const crontabPath = path.join(REPO_ROOT, "crontab.txt");
const crontabLines = fs.existsSync(crontabPath) ? fs.readFileSync(crontabPath, "utf8").split("\n") : [];
const activeCron = crontabLines.filter((l) => /^\s*\d/.test(l));
const activeCronKeys = activeCron
.map((l) => (l.match(/scripts\/([\w-]+)\.mjs/) || [])[1])
.filter(Boolean);
const botFile = path.join(REPO_ROOT, "data", "bot-characters.json");
const botData = readJson(botFile, null);
const botCharacters = Array.isArray(botData?.characters) ? botData.characters.map((c) => c.key) : [];
const pkg = readJson(path.join(REPO_ROOT, "package.json"), {}) || {};
return {
scanned_at: NOW,
version: pkg.version || "unknown",
api_routes: apiRoutes.length,
pages: pages.length,
components: components.length,
hooks: hooks.length,
lib_modules: libFiles.length,
scripts: scripts.length,
prisma_models: models.length,
prisma_model_names: models,
migrations: migrations.length,
latest_migrations: migrations.slice(-5),
cron_active_tasks: activeCron.length,
cron_active_scripts: activeCronKeys,
bot_characters: botCharacters,
tech_stack: {
dependencies: Object.keys(pkg.dependencies || {}).length,
dev_dependencies: Object.keys(pkg.devDependencies || {}).length,
next: (pkg.dependencies || {}).next || "-",
prisma: (pkg.dependencies || {}).prisma || "-",
react: (pkg.dependencies || {}).react || "-",
},
};
}
/* ------------------------------------------------------ 2. 开发过程提取 */
function collectDevProcess() {
const raw = git(["log", "-n", "5", "--pretty=format:%H%x1f%cI%x1f%s"]);
if (!raw) return [];
const rows = [];
for (const line of raw.split("\n")) {
const [hash, date, subject] = line.split("\x1f");
if (!hash) continue;
const files = git(["diff-tree", "--no-commit-id", "--name-only", "-r", "-M", hash])
.split("\n")
.filter(Boolean);
const shortstat = git(["show", "--shortstat", "--oneline", "--format=", hash]).trim();
rows.push({
ts: NOW,
kind: "commit",
hash: hash.slice(0, 8),
committed_at: date,
subject,
files_changed: files.length,
areas: [...new Set(files.map((f) => f.split("/").slice(0, 2).join("/")))].slice(0, 12),
shortstat,
});
}
return rows;
}
/* ------------------------------------------------------ 3. 模型成果捕获 */
function collectArtifacts(cap, commits) {
const head = commits[0] || {};
const artifacts = {
ts: NOW,
date: TODAY,
kind: "delivery",
head_commit: head.hash || "-",
head_subject: head.subject || "-",
files_changed: head.files_changed ?? 0,
shortstat: head.shortstat || "",
capabilities_delta: {
api_routes: cap.api_routes,
pages: cap.pages,
components: cap.components,
prisma_models: cap.prisma_models,
migrations: cap.migrations,
scripts: cap.scripts,
cron_active_tasks: cap.cron_active_tasks,
},
latest_migrations: cap.latest_migrations,
version: cap.version,
recorded_by: "scripts/update-knowledge.mjs",
};
return artifacts;
}
/* --------------------------------------------------- 知识图谱 upsert */
function buildEntities(cap, commits, artifacts) {
const capObs = [
`版本: ${cap.version}`,
`代码规模: API 路由 ${cap.api_routes} / 页面 ${cap.pages} / 组件 ${cap.components} / hooks ${cap.hooks} / lib ${cap.lib_modules}`,
`数据层: Prisma 模型 ${cap.prisma_models} 个、迁移 ${cap.migrations} 个(最新 ${cap.latest_migrations.join("、") || "无"})`,
`脚本: ${cap.scripts} 个(含定时任务与运维脚本)`,
`定时任务: 启用 ${cap.cron_active_tasks} 个 —— ${cap.cron_active_scripts.join("、")}`,
`数字人角色: ${cap.bot_characters.length} 个(${cap.bot_characters.join("、") || "无"})`,
`依赖: dependencies ${cap.tech_stack.dependencies} / devDependencies ${cap.tech_stack.dev_dependencies}(next ${cap.tech_stack.next}、prisma ${cap.tech_stack.prisma}、react ${cap.tech_stack.react})`,
`快照时间: ${cap.scanned_at}(由 scripts/update-knowledge.mjs 自动生成,勿手工编辑)`,
];
const devObs = [
`记录时间: ${NOW}`,
`本次交付: ${artifacts.head_commit} ${artifacts.head_subject}(改动文件 ${artifacts.files_changed} 个${artifacts.shortstat ? "、" + artifacts.shortstat : ""})`,
`能力基线: API ${cap.api_routes} / 页面 ${cap.pages} / 组件 ${cap.components} / 模型 ${cap.prisma_models} / 迁移 ${cap.migrations} / 脚本 ${cap.scripts} / 定时任务 ${cap.cron_active_tasks}`,
...commits.slice(0, 5).map((c) => `${c.hash} ${c.committed_at?.slice(0, 10) || "-"} ${c.subject}`),
"维护: 运行 npm run knowledge:update 刷新本记录",
];
return [
{ type: "entity", entityType: "Capability", name: CAP_ENTITY, observations: capObs },
{ type: "entity", entityType: "fact", name: DEV_RECORD_ENTITY, observations: devObs },
];
}
function upsertKnowledgeGraph(entities, relations) {
const kg = readJson(KG_JSON, null) || {
project: PROJECT_NAME,
version: "V2.1.2",
updated_at: NOW,
entities: [],
relations: [],
};
const fresh = new Map(entities.map((e) => [e.name, e]));
const kept = (kg.entities || []).filter((e) => e.name && !fresh.has(e.name));
const relSeen = new Set();
const mergedRelations = [];
for (const r of [...(kg.relations || []), ...relations]) {
if (!r?.from || !r?.to) continue;
const key = `${r.relationType}|${r.from}|${r.to}`;
if (relSeen.has(key)) continue;
relSeen.add(key);
mergedRelations.push(r);
}
const mergedEntities = [...kept, ...entities];
const nextKg = {
project: kg.project || PROJECT_NAME,
version: kg.version || "V2.1.2",
updated_at: NOW,
entities: mergedEntities,
relations: mergedRelations,
};
writeJson(KG_JSON, nextKg);
// 同步 JSONL(MCP 记忆服务读取的源文件):同名实体整条替换,其余原样保留
const oldLines = readJsonl(KG_JSONL);
const keptLines = oldLines.filter((o) => o.type === "entity" && o.name && !fresh.has(o.name));
const jsonlRows = [...keptLines, ...entities, ...mergedRelations];
if (!DRY_RUN) {
fs.mkdirSync(path.dirname(KG_JSONL), { recursive: true });
fs.writeFileSync(KG_JSONL, jsonlRows.map((o) => JSON.stringify(o)).join("\n") + "\n", "utf8");
}
return { entities: mergedEntities.length, relations: mergedRelations.length, jsonl: jsonlRows.length };
}
/* ------------------------------------------------------------------ main */
function main() {
const cap = scanCapabilities();
const commits = collectDevProcess();
const artifacts = collectArtifacts(cap, commits);
writeJson(CAP_FILE, cap);
// 模型成果:按 (日期 + head 提交) 去重,避免同日多次运行产生重复记录
const existingArtifacts = readJsonl(ART_FILE);
const alreadyRecorded = existingArtifacts.some(
(a) => a.date === artifacts.date && a.head_commit === artifacts.head_commit
);
if (!alreadyRecorded) appendJsonl(ART_FILE, [artifacts]);
// 开发过程:按提交 hash 去重(追加,不覆盖历史)
const existingProc = readJsonl(PROC_FILE);
const seenHashes = new Set(existingProc.map((r) => r.hash));
const newCommits = commits.filter((c) => !seenHashes.has(c.hash));
appendJsonl(PROC_FILE, newCommits);
const entities = buildEntities(cap, commits, artifacts);
const relations = [
{ type: "relation", relationType: "HAS_CAPABILITY", from: PROJECT_NAME, to: CAP_ENTITY },
{ type: "relation", relationType: "HAS_DEV_RECORD", from: PROJECT_NAME, to: DEV_RECORD_ENTITY },
];
const stats = upsertKnowledgeGraph(entities, relations);
if (!QUIET) {
log.info("知识库已更新", {
mode: DRY_RUN ? "dry-run" : "write",
trae_dir: TRAE_DIR,
capabilities: {
api_routes: cap.api_routes,
pages: cap.pages,
components: cap.components,
models: cap.prisma_models,
migrations: cap.migrations,
cron_active: cap.cron_active_tasks,
},
new_dev_records: newCommits.length,
artifacts_written: !alreadyRecorded,
graph: stats,
});
if (DRY_RUN) {
log.info("dry-run:未写入任何文件(含 capabilities/artifacts/knowledge_graph)");
}
}
}
try {
main();
} catch (err) {
log.error("更新知识库失败", { error: err.message, stack: err.stack });
process.exitCode = 1;
}