feat: 首次推送到Gitea - 完整项目代码 + 安全加固 + 知识库
This commit is contained in:
@@ -0,0 +1,115 @@
|
||||
// =============================================================================
|
||||
// Reply Quality Utilities
|
||||
// 共享工具:Bot 回复质量检查、反模式检测、个性化风味
|
||||
// =============================================================================
|
||||
|
||||
import { tokenize } from "./chinese-text.mjs";
|
||||
|
||||
// 最小上下文长度(低于此长度不做关联度评分)
|
||||
export const MIN_CONTEXT_LENGTH = 20;
|
||||
|
||||
// 检查回复与上下文的相关度
|
||||
// 返回 { valid, score } – valid 为 true 表示关联度达标,score 为 0-100 的量化评分
|
||||
export function validateContextRelevance(reply, context) {
|
||||
if (!context || context.length < MIN_CONTEXT_LENGTH) {
|
||||
return { valid: true, score: 70 };
|
||||
}
|
||||
|
||||
const replyGrams = tokenize(reply);
|
||||
if (replyGrams.length === 0) {
|
||||
return { valid: false, score: 0 };
|
||||
}
|
||||
|
||||
const contextSet = new Set(tokenize(context));
|
||||
const matched = replyGrams.filter(g => contextSet.has(g)).length;
|
||||
const relevance = matched / replyGrams.length;
|
||||
|
||||
return {
|
||||
valid: relevance >= 0.05,
|
||||
score: Math.min(100, Math.round(relevance * 100)),
|
||||
};
|
||||
}
|
||||
|
||||
// 反模式检测 – 识别过于通用/模板化的 AI 回复
|
||||
export function detectBadPatterns(text) {
|
||||
const patterns = [
|
||||
/很高兴[能为可以]您/,
|
||||
/必须[地得]说/,
|
||||
/值得[一关]注/,
|
||||
/这是个好(问题|话题)/,
|
||||
/希望[能对]你[有们]帮助/,
|
||||
/感谢[您的你]分享/,
|
||||
/非常有[见解启]发/,
|
||||
];
|
||||
|
||||
const matches = patterns.filter(p => p.test(text));
|
||||
|
||||
return {
|
||||
isGeneric: matches.length >= 3,
|
||||
matchCount: matches.length,
|
||||
patterns: patterns.filter(p => p.test(text)),
|
||||
};
|
||||
}
|
||||
|
||||
// 人格风味前缀 – 给回复添加个性化开场白
|
||||
const PERSONALITY_PREFIXES = {
|
||||
pragmatic: ["实际测试下来", "根据我的经验", "试过之后发现"],
|
||||
enthusiastic: ["太棒了!", "超赞!", "强烈推荐!"],
|
||||
analytical: ["从数据来看", "分析之后发现", "对比了一下数据"],
|
||||
cautious: ["客观来说", "理性分析一下", "需要说明的是"],
|
||||
};
|
||||
|
||||
export function addPersonalityFlavor(text, personaType = "pragmatic") {
|
||||
const prefixes =
|
||||
PERSONALITY_PREFIXES[personaType] || PERSONALITY_PREFIXES.pragmatic;
|
||||
|
||||
// 检查是否已包含任何前缀风味,避免重复添加
|
||||
const allPrefixes = Object.values(PERSONALITY_PREFIXES).flat();
|
||||
if (allPrefixes.some(p => text.startsWith(p))) {
|
||||
return text;
|
||||
}
|
||||
|
||||
const prefix = prefixes[Math.floor(Math.random() * prefixes.length)];
|
||||
return `${prefix},${text}`;
|
||||
}
|
||||
|
||||
// 回复多样性指数 – 检测回复中是否包含重复短语
|
||||
export function checkDiversityScore(text) {
|
||||
if (text.length < 50) return 0.8;
|
||||
|
||||
// 将文本分成句子
|
||||
const sentences = text.split(/[。!?\n]/).filter(Boolean);
|
||||
if (sentences.length < 2) return 0.6;
|
||||
|
||||
// 检查开头词的多样性
|
||||
const sentenceStarts = sentences
|
||||
.map(s => s.trim().substring(0, 2))
|
||||
.filter(Boolean);
|
||||
const uniqueStarts = new Set(sentenceStarts);
|
||||
const startVariety = uniqueStarts.size / sentenceStarts.length;
|
||||
|
||||
// 检查句长分布
|
||||
const lengths = sentences.map(s => s.length);
|
||||
const avgLen = lengths.reduce((a, b) => a + b, 0) / lengths.length;
|
||||
const variance =
|
||||
lengths.reduce((sum, l) => sum + (l - avgLen) ** 2, 0) / lengths.length;
|
||||
|
||||
// 句子开头变化率 + 句长变化率
|
||||
const lengthScore = Math.min(1, variance / 1000);
|
||||
return Math.round((startVariety * 0.6 + lengthScore * 0.4) * 100) / 100;
|
||||
}
|
||||
|
||||
// 获取质量摘要统计
|
||||
export function getQualityMetrics(reply, context, style) {
|
||||
const { score: relevance } = validateContextRelevance(reply, context);
|
||||
const { isGeneric, matchCount } = detectBadPatterns(reply);
|
||||
const diversity = checkDiversityScore(reply);
|
||||
|
||||
return {
|
||||
relevance,
|
||||
isGeneric,
|
||||
patternMatches: matchCount,
|
||||
diversity,
|
||||
style,
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user