Files

116 lines
3.8 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// =============================================================================
// Reply Quality Utilities
// 共享工具:Bot 回复质量检查、反模式检测、个性化风味
// =============================================================================
import { tokenize } from "./chinese-text.mjs";
// 最小上下文长度(低于此长度不做关联度评分)
export const MIN_CONTEXT_LENGTH = 20;
// 检查回复与上下文的相关度
// 返回 { valid, score } – valid 为 true 表示关联度达标,score 为 0-100 的量化评分
export function validateContextRelevance(reply, context) {
if (!context || context.length < MIN_CONTEXT_LENGTH) {
return { valid: true, score: 70 };
}
const replyGrams = tokenize(reply);
if (replyGrams.length === 0) {
return { valid: false, score: 0 };
}
const contextSet = new Set(tokenize(context));
const matched = replyGrams.filter(g => contextSet.has(g)).length;
const relevance = matched / replyGrams.length;
return {
valid: relevance >= 0.05,
score: Math.min(100, Math.round(relevance * 100)),
};
}
// 反模式检测 – 识别过于通用/模板化的 AI 回复
export function detectBadPatterns(text) {
const patterns = [
/很高兴[能为可以]您/,
/必须[地得]说/,
/值得[一关]注/,
/这是个好(问题|话题)/,
/希望[能对]你[有们]帮助/,
/感谢[您的你]分享/,
/非常有[见解启]发/,
];
const matches = patterns.filter(p => p.test(text));
return {
isGeneric: matches.length >= 3,
matchCount: matches.length,
patterns: patterns.filter(p => p.test(text)),
};
}
// 人格风味前缀 – 给回复添加个性化开场白
const PERSONALITY_PREFIXES = {
pragmatic: ["实际测试下来", "根据我的经验", "试过之后发现"],
enthusiastic: ["太棒了!", "超赞!", "强烈推荐!"],
analytical: ["从数据来看", "分析之后发现", "对比了一下数据"],
cautious: ["客观来说", "理性分析一下", "需要说明的是"],
};
export function addPersonalityFlavor(text, personaType = "pragmatic") {
const prefixes =
PERSONALITY_PREFIXES[personaType] || PERSONALITY_PREFIXES.pragmatic;
// 检查是否已包含任何前缀风味,避免重复添加
const allPrefixes = Object.values(PERSONALITY_PREFIXES).flat();
if (allPrefixes.some(p => text.startsWith(p))) {
return text;
}
const prefix = prefixes[Math.floor(Math.random() * prefixes.length)];
return `${prefix},${text}`;
}
// 回复多样性指数 – 检测回复中是否包含重复短语
export function checkDiversityScore(text) {
if (text.length < 50) return 0.8;
// 将文本分成句子
const sentences = text.split(/[。!?\n]/).filter(Boolean);
if (sentences.length < 2) return 0.6;
// 检查开头词的多样性
const sentenceStarts = sentences
.map(s => s.trim().substring(0, 2))
.filter(Boolean);
const uniqueStarts = new Set(sentenceStarts);
const startVariety = uniqueStarts.size / sentenceStarts.length;
// 检查句长分布
const lengths = sentences.map(s => s.length);
const avgLen = lengths.reduce((a, b) => a + b, 0) / lengths.length;
const variance =
lengths.reduce((sum, l) => sum + (l - avgLen) ** 2, 0) / lengths.length;
// 句子开头变化率 + 句长变化率
const lengthScore = Math.min(1, variance / 1000);
return Math.round((startVariety * 0.6 + lengthScore * 0.4) * 100) / 100;
}
// 获取质量摘要统计
export function getQualityMetrics(reply, context, style) {
const { score: relevance } = validateContextRelevance(reply, context);
const { isGeneric, matchCount } = detectBadPatterns(reply);
const diversity = checkDiversityScore(reply);
return {
relevance,
isGeneric,
patternMatches: matchCount,
diversity,
style,
};
}