Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 24 additions & 5 deletions cli/src/quality.rs
Original file line number Diff line number Diff line change
Expand Up @@ -4,8 +4,13 @@ use std::sync::OnceLock;
static NOISY_PATTERNS: OnceLock<Regex> = OnceLock::new();
static JARGON_PATTERNS: OnceLock<Regex> = OnceLock::new();

// Pre-sized capacity for line-deduplication sets
const INITIAL_LINE_CAPACITY: usize = 128;

// Frontmatter delimiters emitted by the resolver's Markdown synthesiser
const FRONTMATTER_DELIMITER: &str = "---";
const FRONTMATTER_CLOSE: &str = "\n---";

// Quality scoring thresholds
const THRESHOLD_NOISE: usize = 6;
const THRESHOLD_JARGON: usize = 3;
Expand Down Expand Up @@ -72,11 +77,25 @@ pub fn score_content(markdown: &str, links: &[String], threshold: f32) -> Qualit
.count();
let jargon_heavy = jargon_count > THRESHOLD_JARGON;

let has_frontmatter = trimmed.starts_with("---")
&& trimmed.contains("relevance_score:")
&& trimmed.contains("intent_category:")
&& trimmed.contains("token_estimate:")
&& trimmed.contains("last_updated:");
let has_frontmatter = if let Some(rest) = trimmed.strip_prefix(FRONTMATTER_DELIMITER) {
// Restrict the checks to the frontmatter block so that a body line
// cannot spoof the bonus. `rest` begins just after the opening
// delimiter, so the closing delimiter ends at
// prefix + offset + len(close). With no closing delimiter we fall back
// to scanning the whole document, matching the previous behaviour.
let header_end = rest
.find(FRONTMATTER_CLOSE)
.map(|offset| FRONTMATTER_DELIMITER.len() + offset + FRONTMATTER_CLOSE.len())
.unwrap_or(trimmed.len());
let header_slice = &trimmed[..header_end];
header_slice.contains("relevance_score:")
&& header_slice.contains("intent_category:")
&& header_slice.contains("token_estimate:")
&& header_slice.contains("last_updated:")
} else {
false
};

let has_structural_anchors = trimmed.contains("[ANCHOR: SUMMARY]")
&& trimmed.contains("[ANCHOR: TECHNICAL_DETAILS]")
&& trimmed.contains("[ANCHOR: COMPARISON]")
Expand Down
11 changes: 7 additions & 4 deletions cli/src/synthesis.rs
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,10 @@ use crate::types::ResolvedResult;
use serde_json::json;
use std::sync::LazyLock;

static LINK_REGEX: LazyLock<regex::Regex> = LazyLock::new(|| {
regex::Regex::new(r"https?://[^\s)>\]]+").expect("Invalid link regex pattern")
});

/// Patterns that may indicate prompt injection attempts
static INJECTION_PATTERNS: LazyLock<Vec<regex::Regex>> = LazyLock::new(|| {
vec![
Expand Down Expand Up @@ -239,7 +243,7 @@ pub fn deterministic_merge(results: &[ResolvedResult]) -> String {
)
} else {
let mut body = String::new();
let mut seen_lines = std::collections::HashSet::new();
let mut seen_lines = std::collections::HashSet::with_capacity(128);

for (i, res) in results.iter().enumerate() {
let idx = i + 1;
Expand All @@ -249,7 +253,7 @@ pub fn deterministic_merge(results: &[ResolvedResult]) -> String {
let mut unique_content = String::new();
for line in content.lines() {
let trimmed = line.trim();
if !trimmed.is_empty() && seen_lines.insert(trimmed.to_string()) {
if !trimmed.is_empty() && seen_lines.insert(trimmed) {
unique_content.push_str(line);
unique_content.push('\n');
} else if trimmed.is_empty() {
Expand Down Expand Up @@ -286,8 +290,7 @@ pub fn deterministic_merge(results: &[ResolvedResult]) -> String {
};

// Extract links for quality scoring
let link_re = regex::Regex::new(r"https?://[^\s)>\]]+").unwrap();
let links: Vec<String> = link_re
let links: Vec<String> = LINK_REGEX
.find_iter(&body_content)
.map(|m| m.as_str().to_string())
.take(10)
Expand Down
40 changes: 40 additions & 0 deletions cli/tests/quality.rs
Original file line number Diff line number Diff line change
Expand Up @@ -26,3 +26,43 @@ fn test_noisy_content() {
let score = score_content(&noisy_content, &links, 0.65);
assert!(score.noisy);
}

/// Body lines that are unique (so they do not trip duplicate detection) and
/// long enough overall to clear the minimum-length threshold.
fn padded_body() -> String {
(0..20)
.map(|i| format!("Unique body line {i} used only for length padding.\n"))
.collect()
}

#[test]
fn test_frontmatter_bonus_requires_fields_inside_the_block() {
// Empty links apply the missing-links penalty, which keeps the base score
// below 1.0 so the +0.05 frontmatter bonus survives the final clamp.
let links: Vec<String> = vec![];
let body = padded_body();

// Genuine frontmatter: all four fields sit inside the opening/closing block.
let real_frontmatter = format!(
"---\nrelevance_score: 1.0\nintent_category: docs\n\
token_estimate: 100\nlast_updated: 2026-10-01\n---\n{body}"
);

// The same four fields, but only in the body *after* the closing delimiter.
let spoofed = format!(
"---\n{body}---\nrelevance_score: 1.0\nintent_category: docs\n\
token_estimate: 100\nlast_updated: 2026-10-01\n"
);

let real = score_content(&real_frontmatter, &links, 0.65);
let fake = score_content(&spoofed, &links, 0.65);

// Both bodies are structurally identical, so the only difference is the
// frontmatter bonus: the spoofed variant must not earn it.
assert!(
(real.score - fake.score - 0.05).abs() < f32::EPSILON,
"expected frontmatter bonus only for the real block: real={} fake={}",
real.score,
fake.score
);
}
Loading