feat(agent): semantic control mapping — embedding index + region->control retrieval
CI / Check (push) Skipped
CI / Check (pull_request) Successful in 6m7s
CI / Detect Changes (pull_request) Skipped
CI / Deploy Agent (pull_request) Skipped
CI / Deploy Dashboard (pull_request) Skipped
CI / Deploy Docs (pull_request) Skipped
CI / Deploy MCP (pull_request) Skipped

The scale mechanism for the master-control corpus (which carries no CWE to LUT
on): ControlIndex embeds each control's requirement text and returns the top-K
nearest to a code region (cosine); SemanticControlChecker retrieves those K,
judges each with the grounded judge, and grounds the verdicts -> control_refs.

The region->control direction (vs the CWE-LUT's finding->control) is what scales
to ~13.6k: the LLM only ever judges a handful of retrieved candidates, and every
survivor is still anchored to real code by the grounding gate. Reuses judge +
ground gate. Stub-tested (cosine ranking, retrieve->ground, ungrounded dropped).

Not yet wired into the scan (needs the master-controls catalog live, breakpilot #129).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Sharang Parnerkar
2026-07-21 12:35:30 +02:00
co-authored by Claude Fable 5
parent 18a23403a1
commit 30487e4d8f
3 changed files with 233 additions and 0 deletions
+110
View File
@@ -0,0 +1,110 @@
//! In-memory embedding index over the control corpus, for region → control
//! retrieval.
//!
//! At master-control scale (~13.6k) findings can't be mapped by CWE (the master
//! controls carry none), so we map by *similarity*: embed each control's
//! requirement text once, then for a code region pull the top-K nearest controls
//! to hand to the grounded judge. This is the retrieval half of the semantic path.
use compliance_core::control_check::ControlCheckSpec;
use compliance_core::error::CoreError;
use crate::llm::LlmClient;
/// A control spec paired with its requirement-text embedding.
pub struct ControlIndex {
entries: Vec<(ControlCheckSpec, Vec<f64>)>,
}
impl ControlIndex {
/// Build directly from precomputed embeddings (used by tests + callers that
/// already embedded the corpus).
pub fn from_embeddings(entries: Vec<(ControlCheckSpec, Vec<f64>)>) -> Self {
Self { entries }
}
/// Build by embedding each control's requirement text.
pub async fn build(llm: &LlmClient, specs: Vec<ControlCheckSpec>) -> Result<Self, CoreError> {
if specs.is_empty() {
return Ok(Self {
entries: Vec::new(),
});
}
let texts: Vec<String> = specs.iter().map(|s| s.requirement.clone()).collect();
let embeddings = llm
.embed(texts)
.await
.map_err(|e| CoreError::Llm(e.to_string()))?;
Ok(Self {
entries: specs.into_iter().zip(embeddings).collect(),
})
}
pub fn len(&self) -> usize {
self.entries.len()
}
pub fn is_empty(&self) -> bool {
self.entries.is_empty()
}
/// The top-`k` control specs whose embedding is nearest (cosine) to `query`.
pub fn nearest(&self, query: &[f64], k: usize) -> Vec<ControlCheckSpec> {
let mut scored: Vec<(f64, &ControlCheckSpec)> = self
.entries
.iter()
.map(|(spec, emb)| (cosine(query, emb), spec))
.collect();
scored.sort_by(|a, b| b.0.total_cmp(&a.0));
scored.into_iter().take(k).map(|(_, s)| s.clone()).collect()
}
}
/// Cosine similarity; 0.0 for length-mismatched, empty, or zero vectors.
fn cosine(a: &[f64], b: &[f64]) -> f64 {
if a.len() != b.len() || a.is_empty() {
return 0.0;
}
let dot: f64 = a.iter().zip(b).map(|(x, y)| x * y).sum();
let na: f64 = a.iter().map(|x| x * x).sum();
let nb: f64 = b.iter().map(|x| x * x).sum();
if na == 0.0 || nb == 0.0 {
return 0.0;
}
dot / (na.sqrt() * nb.sqrt())
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::models::finding::Severity;
fn spec(id: &str) -> ControlCheckSpec {
ControlCheckSpec {
control_id: id.into(),
title: id.into(),
requirement: id.into(),
default_cwe: None,
severity: Severity::Medium,
}
}
#[test]
fn nearest_ranks_by_cosine() {
let index = ControlIndex::from_embeddings(vec![
(spec("a"), vec![1.0, 0.0]),
(spec("b"), vec![0.0, 1.0]),
(spec("c"), vec![0.7, 0.7]),
]);
let hits = index.nearest(&[0.9, 0.1], 2);
assert_eq!(hits.len(), 2);
assert_eq!(hits[0].control_id, "a"); // closest to [0.9,0.1]
}
#[test]
fn cosine_edges_are_zero() {
assert_eq!(cosine(&[1.0], &[1.0, 2.0]), 0.0); // length mismatch
assert_eq!(cosine(&[0.0, 0.0], &[1.0, 1.0]), 0.0); // zero vector
assert!((cosine(&[1.0, 0.0], &[1.0, 0.0]) - 1.0).abs() < 1e-9); // identical
}
}
+4
View File
@@ -6,13 +6,17 @@
//! and snapshots it locally.
mod checker;
mod index;
mod judge;
mod oscal_provider;
mod scan_triage;
mod semantic;
mod triage;
pub use checker::GroundedControlChecker;
pub use index::ControlIndex;
pub use judge::{ControlJudge, LlmControlJudge, PROMPT_VERSION};
pub use oscal_provider::OscalControlsProvider;
pub use scan_triage::triage_repo_findings;
pub use semantic::SemanticControlChecker;
pub use triage::{ControlTriage, TriageOutcome};
+119
View File
@@ -0,0 +1,119 @@
//! Semantic control mapping: retrieve the top-K controls nearest a code region,
//! then confirm each with the grounded judge.
//!
//! The `region → controls` direction (vs. the CWE-LUT's `finding → control`) is
//! what scales to the full master-control corpus: the LLM only ever judges a
//! handful of retrieved candidates, and every surviving verdict is still anchored
//! to real code by the grounding gate.
use compliance_core::control_check::{ground, CandidateRegion};
use compliance_core::models::Finding;
use super::index::ControlIndex;
use super::judge::ControlJudge;
/// Retrieve → judge → ground, generic over the judge so tests use a stub.
pub struct SemanticControlChecker<J> {
judge: J,
}
impl<J: ControlJudge> SemanticControlChecker<J> {
pub fn new(judge: J) -> Self {
Self { judge }
}
/// Map a code region to the controls it violates. `region_embedding` is the
/// region's embedding (the caller computes it via the LLM); the top-`k`
/// nearest controls in `index` are judged and grounded.
pub async fn check(
&self,
index: &ControlIndex,
region: &CandidateRegion,
region_embedding: &[f64],
k: usize,
repo_id: &str,
) -> Vec<Finding> {
let candidates = index.nearest(region_embedding, k);
let mut findings = Vec::new();
for spec in &candidates {
let verdict = self.judge.judge(spec, region).await;
if let Some(finding) = ground(spec, region, &verdict, repo_id) {
findings.push(finding);
}
}
findings
}
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::control_check::{ControlCheckSpec, LlmVerdict};
use compliance_core::models::finding::Severity;
struct StubJudge {
verdict: LlmVerdict,
}
impl ControlJudge for StubJudge {
async fn judge(&self, _s: &ControlCheckSpec, _r: &CandidateRegion) -> LlmVerdict {
self.verdict.clone()
}
}
fn spec(id: &str) -> ControlCheckSpec {
ControlCheckSpec {
control_id: id.into(),
title: id.into(),
requirement: id.into(),
default_cwe: None,
severity: Severity::Medium,
}
}
#[tokio::test]
async fn retrieves_then_grounds_the_nearest_control() {
let index = ControlIndex::from_embeddings(vec![
(spec("mc-near"), vec![1.0, 0.0]),
(spec("mc-far"), vec![0.0, 1.0]),
]);
let checker = SemanticControlChecker::new(StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "PASSWORD = \"admin\"".into(),
cwe: None,
confidence: 0.9,
},
});
let region = CandidateRegion {
file: "src/auth.py".into(),
start_line: 1,
content: "PASSWORD = \"admin\"\n".into(),
};
// Query embedding nearest to mc-near; k=1 → only mc-near is judged.
let findings = checker
.check(&index, &region, &[0.95, 0.05], 1, "repo")
.await;
assert_eq!(findings.len(), 1);
assert_eq!(findings[0].control_refs, vec!["mc-near".to_string()]);
}
#[tokio::test]
async fn ungrounded_verdict_is_dropped() {
let index = ControlIndex::from_embeddings(vec![(spec("mc-near"), vec![1.0, 0.0])]);
let checker = SemanticControlChecker::new(StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "not in the region".into(),
cwe: None,
confidence: 0.9,
},
});
let region = CandidateRegion {
file: "f".into(),
start_line: 1,
content: "real code\n".into(),
};
let findings = checker.check(&index, &region, &[1.0, 0.0], 1, "repo").await;
assert!(findings.is_empty());
}
}