diff --git a/cli/src/quality.rs b/cli/src/quality.rs index 81d88b0b..c7c21ee6 100644 --- a/cli/src/quality.rs +++ b/cli/src/quality.rs @@ -4,8 +4,13 @@ use std::sync::OnceLock; static NOISY_PATTERNS: OnceLock = OnceLock::new(); static JARGON_PATTERNS: OnceLock = OnceLock::new(); +// Pre-sized capacity for line-deduplication sets const INITIAL_LINE_CAPACITY: usize = 128; +// Frontmatter delimiters emitted by the resolver's Markdown synthesiser +const FRONTMATTER_DELIMITER: &str = "---"; +const FRONTMATTER_CLOSE: &str = "\n---"; + // Quality scoring thresholds const THRESHOLD_NOISE: usize = 6; const THRESHOLD_JARGON: usize = 3; @@ -72,11 +77,25 @@ pub fn score_content(markdown: &str, links: &[String], threshold: f32) -> Qualit .count(); let jargon_heavy = jargon_count > THRESHOLD_JARGON; - let has_frontmatter = trimmed.starts_with("---") - && trimmed.contains("relevance_score:") - && trimmed.contains("intent_category:") - && trimmed.contains("token_estimate:") - && trimmed.contains("last_updated:"); + let has_frontmatter = if let Some(rest) = trimmed.strip_prefix(FRONTMATTER_DELIMITER) { + // Restrict the checks to the frontmatter block so that a body line + // cannot spoof the bonus. `rest` begins just after the opening + // delimiter, so the closing delimiter ends at + // prefix + offset + len(close). With no closing delimiter we fall back + // to scanning the whole document, matching the previous behaviour. + let header_end = rest + .find(FRONTMATTER_CLOSE) + .map(|offset| FRONTMATTER_DELIMITER.len() + offset + FRONTMATTER_CLOSE.len()) + .unwrap_or(trimmed.len()); + let header_slice = &trimmed[..header_end]; + header_slice.contains("relevance_score:") + && header_slice.contains("intent_category:") + && header_slice.contains("token_estimate:") + && header_slice.contains("last_updated:") + } else { + false + }; + let has_structural_anchors = trimmed.contains("[ANCHOR: SUMMARY]") && trimmed.contains("[ANCHOR: TECHNICAL_DETAILS]") && trimmed.contains("[ANCHOR: COMPARISON]") diff --git a/cli/src/synthesis.rs b/cli/src/synthesis.rs index e4db6a07..6568df01 100644 --- a/cli/src/synthesis.rs +++ b/cli/src/synthesis.rs @@ -21,6 +21,10 @@ use crate::types::ResolvedResult; use serde_json::json; use std::sync::LazyLock; +static LINK_REGEX: LazyLock = LazyLock::new(|| { + regex::Regex::new(r"https?://[^\s)>\]]+").expect("Invalid link regex pattern") +}); + /// Patterns that may indicate prompt injection attempts static INJECTION_PATTERNS: LazyLock> = LazyLock::new(|| { vec![ @@ -239,7 +243,7 @@ pub fn deterministic_merge(results: &[ResolvedResult]) -> String { ) } else { let mut body = String::new(); - let mut seen_lines = std::collections::HashSet::new(); + let mut seen_lines = std::collections::HashSet::with_capacity(128); for (i, res) in results.iter().enumerate() { let idx = i + 1; @@ -249,7 +253,7 @@ pub fn deterministic_merge(results: &[ResolvedResult]) -> String { let mut unique_content = String::new(); for line in content.lines() { let trimmed = line.trim(); - if !trimmed.is_empty() && seen_lines.insert(trimmed.to_string()) { + if !trimmed.is_empty() && seen_lines.insert(trimmed) { unique_content.push_str(line); unique_content.push('\n'); } else if trimmed.is_empty() { @@ -286,8 +290,7 @@ pub fn deterministic_merge(results: &[ResolvedResult]) -> String { }; // Extract links for quality scoring - let link_re = regex::Regex::new(r"https?://[^\s)>\]]+").unwrap(); - let links: Vec = link_re + let links: Vec = LINK_REGEX .find_iter(&body_content) .map(|m| m.as_str().to_string()) .take(10) diff --git a/cli/tests/quality.rs b/cli/tests/quality.rs index 5c699139..be04ae65 100644 --- a/cli/tests/quality.rs +++ b/cli/tests/quality.rs @@ -26,3 +26,43 @@ fn test_noisy_content() { let score = score_content(&noisy_content, &links, 0.65); assert!(score.noisy); } + +/// Body lines that are unique (so they do not trip duplicate detection) and +/// long enough overall to clear the minimum-length threshold. +fn padded_body() -> String { + (0..20) + .map(|i| format!("Unique body line {i} used only for length padding.\n")) + .collect() +} + +#[test] +fn test_frontmatter_bonus_requires_fields_inside_the_block() { + // Empty links apply the missing-links penalty, which keeps the base score + // below 1.0 so the +0.05 frontmatter bonus survives the final clamp. + let links: Vec = vec![]; + let body = padded_body(); + + // Genuine frontmatter: all four fields sit inside the opening/closing block. + let real_frontmatter = format!( + "---\nrelevance_score: 1.0\nintent_category: docs\n\ + token_estimate: 100\nlast_updated: 2026-10-01\n---\n{body}" + ); + + // The same four fields, but only in the body *after* the closing delimiter. + let spoofed = format!( + "---\n{body}---\nrelevance_score: 1.0\nintent_category: docs\n\ + token_estimate: 100\nlast_updated: 2026-10-01\n" + ); + + let real = score_content(&real_frontmatter, &links, 0.65); + let fake = score_content(&spoofed, &links, 0.65); + + // Both bodies are structurally identical, so the only difference is the + // frontmatter bonus: the spoofed variant must not earn it. + assert!( + (real.score - fake.score - 0.05).abs() < f32::EPSILON, + "expected frontmatter bonus only for the real block: real={} fake={}", + real.score, + fake.score + ); +}