// ═══════════════════════════════════════════════════════════════════════════ // Story Summary - Metrics Collector (v3 - Deterministic Query + Hybrid + W-RRF) // // 命名规范: // - 存储层用 L0/L1/L2/L3(StateAtom/Chunk/Event/Fact) // - 指标层用语义名称:anchor/evidence/event/constraint/arc // ═══════════════════════════════════════════════════════════════════════════ /** * 创建空的指标对象 * @returns {object} */ export function createMetrics() { return { // Query Build - 查询构建 query: { buildTime: 0, refineTime: 0, lengths: { v0Chars: 0, v1Chars: null, // null = NA rerankChars: 0, }, }, // Anchor (L0 StateAtoms) - 语义锚点 anchor: { needRecall: false, focusEntities: [], matched: 0, floorsHit: 0, topHits: [], }, // Lexical (MiniSearch) - 词法检索 lexical: { terms: [], atomHits: 0, chunkHits: 0, eventHits: 0, searchTime: 0, }, // Fusion (W-RRF) - 多路融合 fusion: { denseCount: 0, lexCount: 0, anchorCount: 0, totalUnique: 0, afterCap: 0, time: 0, }, // Constraint (L3 Facts) - 世界约束 constraint: { total: 0, filtered: 0, injected: 0, tokens: 0, samples: [], }, // Event (L2 Events) - 事件摘要 event: { inStore: 0, considered: 0, selected: 0, byRecallType: { direct: 0, related: 0, causal: 0, lexical: 0 }, similarityDistribution: { min: 0, max: 0, mean: 0, median: 0 }, entityFilter: null, causalChainDepth: 0, causalCount: 0, entitiesUsed: 0, entityNames: [], }, // Evidence (L1 Chunks) - 原文证据 evidence: { floorsFromAnchors: 0, chunkTotal: 0, denseCoarse: 0, merged: 0, mergedByType: { anchorVirtual: 0, chunkReal: 0 }, selected: 0, selectedByType: { anchorVirtual: 0, chunkReal: 0 }, contextPairsAdded: 0, tokens: 0, assemblyTime: 0, rerankApplied: false, beforeRerank: 0, afterRerank: 0, rerankTime: 0, rerankScores: null, }, // Arc - 人物弧光 arc: { injected: 0, tokens: 0, }, // Formatting - 格式化 formatting: { sectionsIncluded: [], time: 0, }, // Budget Summary - 预算 budget: { total: 0, limit: 0, utilization: 0, breakdown: { constraints: 0, events: 0, distantEvidence: 0, recentEvidence: 0, arcs: 0, }, }, // Timing - 计时 timing: { queryBuild: 0, queryRefine: 0, anchorSearch: 0, lexicalSearch: 0, fusion: 0, constraintFilter: 0, eventRetrieval: 0, evidenceRetrieval: 0, evidenceRerank: 0, evidenceAssembly: 0, formatting: 0, total: 0, }, // Quality Indicators - 质量指标 quality: { constraintCoverage: 100, eventPrecisionProxy: 0, evidenceDensity: 0, chunkRealRatio: 0, potentialIssues: [], }, }; } /** * 计算相似度分布统计 * @param {number[]} similarities * @returns {{min: number, max: number, mean: number, median: number}} */ export function calcSimilarityStats(similarities) { if (!similarities?.length) { return { min: 0, max: 0, mean: 0, median: 0 }; } const sorted = [...similarities].sort((a, b) => a - b); const sum = sorted.reduce((a, b) => a + b, 0); return { min: Number(sorted[0].toFixed(3)), max: Number(sorted[sorted.length - 1].toFixed(3)), mean: Number((sum / sorted.length).toFixed(3)), median: Number(sorted[Math.floor(sorted.length / 2)].toFixed(3)), }; } /** * 格式化指标为可读日志 * @param {object} metrics * @returns {string} */ export function formatMetricsLog(metrics) { const m = metrics; const lines = []; lines.push(''); lines.push('════════════════════════════════════════'); lines.push(' Recall Metrics Report '); lines.push('════════════════════════════════════════'); lines.push(''); // Query Length lines.push('[Query Length] 查询长度'); lines.push(`├─ query_v0_chars: ${m.query?.lengths?.v0Chars ?? 0}`); lines.push(`├─ query_v1_chars: ${m.query?.lengths?.v1Chars == null ? 'NA' : m.query.lengths.v1Chars}`); lines.push(`└─ rerank_query_chars: ${m.query?.lengths?.rerankChars ?? 0}`); lines.push(''); // Query Build lines.push('[Query] 查询构建'); lines.push(`├─ build_time: ${m.query.buildTime}ms`); lines.push(`└─ refine_time: ${m.query.refineTime}ms`); lines.push(''); // Anchor (L0 StateAtoms) lines.push('[Anchor] L0 StateAtoms - 语义锚点'); lines.push(`├─ need_recall: ${m.anchor.needRecall}`); if (m.anchor.needRecall) { lines.push(`├─ focus_entities: [${(m.anchor.focusEntities || []).join(', ')}]`); lines.push(`├─ matched: ${m.anchor.matched || 0}`); lines.push(`└─ floors_hit: ${m.anchor.floorsHit || 0}`); } lines.push(''); // Lexical (MiniSearch) lines.push('[Lexical] MiniSearch - 词法检索'); lines.push(`├─ terms: [${(m.lexical.terms || []).slice(0, 8).join(', ')}]`); lines.push(`├─ atom_hits: ${m.lexical.atomHits}`); lines.push(`├─ chunk_hits: ${m.lexical.chunkHits}`); lines.push(`├─ event_hits: ${m.lexical.eventHits}`); lines.push(`└─ search_time: ${m.lexical.searchTime}ms`); lines.push(''); // Fusion (W-RRF) lines.push('[Fusion] W-RRF - 多路融合'); lines.push(`├─ dense_count: ${m.fusion.denseCount}`); lines.push(`├─ lex_count: ${m.fusion.lexCount}`); lines.push(`├─ anchor_count: ${m.fusion.anchorCount}`); lines.push(`├─ total_unique: ${m.fusion.totalUnique}`); lines.push(`├─ after_cap: ${m.fusion.afterCap}`); lines.push(`└─ time: ${m.fusion.time}ms`); lines.push(''); // Constraint (L3 Facts) lines.push('[Constraint] L3 Facts - 世界约束'); lines.push(`├─ total: ${m.constraint.total}`); lines.push(`├─ filtered: ${m.constraint.filtered || 0}`); lines.push(`├─ injected: ${m.constraint.injected}`); lines.push(`├─ tokens: ${m.constraint.tokens}`); if (m.constraint.samples && m.constraint.samples.length > 0) { lines.push(`└─ samples: "${m.constraint.samples.slice(0, 2).join('", "')}"`); } lines.push(''); // Event (L2 Events) lines.push('[Event] L2 Events - 事件摘要'); lines.push(`├─ in_store: ${m.event.inStore}`); lines.push(`├─ considered: ${m.event.considered}`); if (m.event.entityFilter) { const ef = m.event.entityFilter; lines.push(`├─ entity_filter:`); lines.push(`│ ├─ focus_entities: [${(ef.focusEntities || []).join(', ')}]`); lines.push(`│ ├─ before: ${ef.before}`); lines.push(`│ ├─ after: ${ef.after}`); lines.push(`│ └─ filtered: ${ef.filtered}`); } lines.push(`├─ selected: ${m.event.selected}`); lines.push(`├─ by_recall_type:`); lines.push(`│ ├─ direct: ${m.event.byRecallType.direct}`); lines.push(`│ ├─ related: ${m.event.byRecallType.related}`); lines.push(`│ ├─ causal: ${m.event.byRecallType.causal}`); lines.push(`│ └─ lexical: ${m.event.byRecallType.lexical}`); const sim = m.event.similarityDistribution; if (sim && sim.max > 0) { lines.push(`├─ similarity_distribution:`); lines.push(`│ ├─ min: ${sim.min}`); lines.push(`│ ├─ max: ${sim.max}`); lines.push(`│ ├─ mean: ${sim.mean}`); lines.push(`│ └─ median: ${sim.median}`); } lines.push(`├─ causal_chain: depth=${m.event.causalChainDepth}, count=${m.event.causalCount}`); lines.push(`└─ entities_used: ${m.event.entitiesUsed} [${(m.event.entityNames || []).join(', ')}]`); lines.push(''); // Evidence (L1 Chunks) lines.push('[Evidence] L1 Chunks - 原文证据'); lines.push(`├─ floors_from_anchors: ${m.evidence.floorsFromAnchors}`); if (m.evidence.chunkTotal > 0) { lines.push(`├─ chunk_total: ${m.evidence.chunkTotal}`); lines.push(`├─ dense_coarse: ${m.evidence.denseCoarse}`); } lines.push(`├─ merged: ${m.evidence.merged}`); if (m.evidence.mergedByType) { const mt = m.evidence.mergedByType; lines.push(`│ ├─ anchor_virtual: ${mt.anchorVirtual || 0}`); lines.push(`│ └─ chunk_real: ${mt.chunkReal || 0}`); } if (m.evidence.rerankApplied) { lines.push(`├─ rerank_applied: true`); lines.push(`│ ├─ before: ${m.evidence.beforeRerank}`); lines.push(`│ ├─ after: ${m.evidence.afterRerank}`); lines.push(`│ └─ time: ${m.evidence.rerankTime}ms`); if (m.evidence.rerankScores) { const rs = m.evidence.rerankScores; lines.push(`├─ rerank_scores: min=${rs.min}, max=${rs.max}, mean=${rs.mean}`); } } else { lines.push(`├─ rerank_applied: false`); } lines.push(`├─ selected: ${m.evidence.selected}`); if (m.evidence.selectedByType) { const st = m.evidence.selectedByType; lines.push(`│ ├─ anchor_virtual: ${st.anchorVirtual || 0}`); lines.push(`│ └─ chunk_real: ${st.chunkReal || 0}`); } lines.push(`├─ context_pairs_added: ${m.evidence.contextPairsAdded}`); lines.push(`├─ tokens: ${m.evidence.tokens}`); lines.push(`└─ assembly_time: ${m.evidence.assemblyTime}ms`); lines.push(''); // Arc if (m.arc.injected > 0) { lines.push('[Arc] 人物弧光'); lines.push(`├─ injected: ${m.arc.injected}`); lines.push(`└─ tokens: ${m.arc.tokens}`); lines.push(''); } // Formatting lines.push('[Formatting] 格式化'); lines.push(`├─ sections: [${(m.formatting.sectionsIncluded || []).join(', ')}]`); lines.push(`└─ time: ${m.formatting.time}ms`); lines.push(''); // Budget Summary lines.push('[Budget] 预算'); lines.push(`├─ total_tokens: ${m.budget.total}`); lines.push(`├─ limit: ${m.budget.limit}`); lines.push(`├─ utilization: ${m.budget.utilization}%`); lines.push(`└─ breakdown:`); const bd = m.budget.breakdown || {}; lines.push(` ├─ constraints: ${bd.constraints || 0}`); lines.push(` ├─ events: ${bd.events || 0}`); lines.push(` ├─ distant_evidence: ${bd.distantEvidence || 0}`); lines.push(` ├─ recent_evidence: ${bd.recentEvidence || 0}`); lines.push(` └─ arcs: ${bd.arcs || 0}`); lines.push(''); // Timing lines.push('[Timing] 计时'); lines.push(`├─ query_build: ${m.query.buildTime}ms`); lines.push(`├─ query_refine: ${m.query.refineTime}ms`); lines.push(`├─ anchor_search: ${m.timing.anchorSearch}ms`); lines.push(`├─ lexical_search: ${m.lexical.searchTime}ms`); lines.push(`├─ fusion: ${m.fusion.time}ms`); lines.push(`├─ constraint_filter: ${m.timing.constraintFilter}ms`); lines.push(`├─ event_retrieval: ${m.timing.eventRetrieval}ms`); lines.push(`├─ evidence_retrieval: ${m.timing.evidenceRetrieval}ms`); if (m.timing.evidenceRerank > 0) { lines.push(`├─ evidence_rerank: ${m.timing.evidenceRerank}ms`); } lines.push(`├─ evidence_assembly: ${m.timing.evidenceAssembly}ms`); lines.push(`├─ formatting: ${m.timing.formatting}ms`); lines.push(`└─ total: ${m.timing.total}ms`); lines.push(''); // Quality Indicators lines.push('[Quality] 质量指标'); lines.push(`├─ constraint_coverage: ${m.quality.constraintCoverage}%`); lines.push(`├─ event_precision_proxy: ${m.quality.eventPrecisionProxy}`); lines.push(`├─ evidence_density: ${m.quality.evidenceDensity}%`); lines.push(`├─ chunk_real_ratio: ${m.quality.chunkRealRatio}%`); if (m.quality.potentialIssues && m.quality.potentialIssues.length > 0) { lines.push(`└─ potential_issues:`); m.quality.potentialIssues.forEach((issue, i) => { const prefix = i === m.quality.potentialIssues.length - 1 ? ' └─' : ' ├─'; lines.push(`${prefix} ⚠ ${issue}`); }); } else { lines.push(`└─ potential_issues: none`); } lines.push(''); lines.push('════════════════════════════════════════'); lines.push(''); return lines.join('\n'); } /** * 检测潜在问题 * @param {object} metrics * @returns {string[]} */ export function detectIssues(metrics) { const issues = []; const m = metrics; // ───────────────────────────────────────────────────────────────── // 查询构建问题 // ───────────────────────────────────────────────────────────────── if ((m.anchor.focusEntities || []).length === 0) { issues.push('No focus entities extracted - entity lexicon may be empty or messages too short'); } // ───────────────────────────────────────────────────────────────── // 锚点匹配问题 // ───────────────────────────────────────────────────────────────── if ((m.anchor.matched || 0) === 0 && m.anchor.needRecall) { issues.push('No anchors matched - may need to generate anchors'); } // ───────────────────────────────────────────────────────────────── // 词法检索问题 // ───────────────────────────────────────────────────────────────── if ((m.lexical.terms || []).length > 0 && m.lexical.atomHits === 0 && m.lexical.chunkHits === 0 && m.lexical.eventHits === 0) { issues.push('Lexical search returned zero hits - terms may not match any indexed content'); } // ───────────────────────────────────────────────────────────────── // 融合问题 // ───────────────────────────────────────────────────────────────── if (m.fusion.lexCount === 0 && m.fusion.denseCount > 0) { issues.push('No lexical candidates in fusion - hybrid retrieval not contributing'); } if (m.fusion.afterCap === 0) { issues.push('Fusion produced zero candidates - all retrieval paths may have failed'); } // ───────────────────────────────────────────────────────────────── // 事件召回问题 // ───────────────────────────────────────────────────────────────── if (m.event.considered > 0) { // 只统计 Dense 路选中(direct + related),Lexical 是额外补充不计入 const denseSelected = (m.event.byRecallType?.direct || 0) + (m.event.byRecallType?.related || 0); const denseSelectRatio = denseSelected / m.event.considered; if (denseSelectRatio < 0.1) { issues.push(`Dense event selection ratio too low (${(denseSelectRatio * 100).toFixed(1)}%) - threshold may be too high`); } if (denseSelectRatio > 0.6 && m.event.considered > 10) { issues.push(`Dense event selection ratio high (${(denseSelectRatio * 100).toFixed(1)}%) - may include noise`); } } // 实体过滤问题 if (m.event.entityFilter) { const ef = m.event.entityFilter; if (ef.filtered === 0 && ef.before > 10) { issues.push('No events filtered by entity - focus entities may be too broad or missing'); } if (ef.before > 0 && ef.filtered > ef.before * 0.8) { issues.push(`Too many events filtered (${ef.filtered}/${ef.before}) - focus may be too narrow`); } } // 相似度问题 if (m.event.similarityDistribution && m.event.similarityDistribution.min > 0 && m.event.similarityDistribution.min < 0.5) { issues.push(`Low similarity events included (min=${m.event.similarityDistribution.min})`); } // 因果链问题 if (m.event.selected > 0 && m.event.causalCount === 0 && m.event.byRecallType.direct === 0) { issues.push('No direct or causal events - query may not align with stored events'); } // ───────────────────────────────────────────────────────────────── // 证据问题 // ───────────────────────────────────────────────────────────────── // Dense 粗筛比例 if (m.evidence.chunkTotal > 0 && m.evidence.denseCoarse > 0) { const coarseFilterRatio = 1 - (m.evidence.denseCoarse / m.evidence.chunkTotal); if (coarseFilterRatio > 0.95) { issues.push(`Very high dense coarse filter ratio (${(coarseFilterRatio * 100).toFixed(0)}%) - query vector may be poorly aligned`); } } // Rerank 相关问题 if (m.evidence.rerankApplied) { if (m.evidence.beforeRerank > 0 && m.evidence.afterRerank > 0) { const filterRatio = 1 - (m.evidence.afterRerank / m.evidence.beforeRerank); if (filterRatio > 0.7) { issues.push(`High rerank filter ratio (${(filterRatio * 100).toFixed(0)}%) - many irrelevant chunks in fusion output`); } } if (m.evidence.rerankScores) { const rs = m.evidence.rerankScores; if (rs.max < 0.5) { issues.push(`Low rerank scores (max=${rs.max}) - query may be poorly matched`); } if (rs.mean < 0.3) { issues.push(`Very low average rerank score (mean=${rs.mean}) - context may be weak`); } } if (m.evidence.rerankTime > 2000) { issues.push(`Slow rerank (${m.evidence.rerankTime}ms) - may affect response time`); } } // chunk_real 比例(核心质量指标) if (m.evidence.selected > 0 && m.evidence.selectedByType) { const chunkReal = m.evidence.selectedByType.chunkReal || 0; const ratio = chunkReal / m.evidence.selected; if (ratio === 0 && m.evidence.selected > 5) { issues.push('Zero real chunks in selected evidence - only anchor virtual chunks present'); } else if (ratio < 0.2 && m.evidence.selected > 10) { issues.push(`Low real chunk ratio (${(ratio * 100).toFixed(0)}%) - may lack concrete dialogue evidence`); } } // ───────────────────────────────────────────────────────────────── // 预算问题 // ───────────────────────────────────────────────────────────────── if (m.budget.utilization > 90) { issues.push(`High budget utilization (${m.budget.utilization}%) - may be truncating content`); } // ───────────────────────────────────────────────────────────────── // 性能问题 // ───────────────────────────────────────────────────────────────── if (m.timing.total > 8000) { issues.push(`Slow recall (${m.timing.total}ms) - consider optimization`); } if (m.query.buildTime > 100) { issues.push(`Slow query build (${m.query.buildTime}ms) - entity lexicon may be too large`); } return issues; }