Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
effc3080e0 | ||
|
|
51bab61f77 | ||
|
|
5bdc35ee92 | ||
|
|
ea516cc054 | ||
|
|
7d5c95ddb8 |
@@ -277,8 +277,10 @@ impl PipelineOrchestrator {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Dedup against existing findings and insert new ones
|
// Dedup against existing findings: insert first-seen ones, and refresh the
|
||||||
|
// control mappings on ones we've seen before.
|
||||||
let mut new_count = 0u32;
|
let mut new_count = 0u32;
|
||||||
|
let mut refreshed_count = 0u32;
|
||||||
let mut new_findings: Vec<Finding> = Vec::new();
|
let mut new_findings: Vec<Finding> = Vec::new();
|
||||||
for mut finding in all_findings {
|
for mut finding in all_findings {
|
||||||
finding.scan_run_id = Some(scan_run_id.to_string());
|
finding.scan_run_id = Some(scan_run_id.to_string());
|
||||||
@@ -293,8 +295,25 @@ impl PipelineOrchestrator {
|
|||||||
finding.id = result.inserted_id.as_object_id();
|
finding.id = result.inserted_id.as_object_id();
|
||||||
new_findings.push(finding);
|
new_findings.push(finding);
|
||||||
new_count += 1;
|
new_count += 1;
|
||||||
|
} else if !finding.control_refs.is_empty() {
|
||||||
|
// Re-scan refresh: a mapping pass (newly enabled or tuned) computed
|
||||||
|
// control_refs for a finding first seen before mapping ran. Persist
|
||||||
|
// them onto the existing row — the insert path alone never would.
|
||||||
|
self.db
|
||||||
|
.findings()
|
||||||
|
.update_one(
|
||||||
|
doc! { "fingerprint": &finding.fingerprint },
|
||||||
|
doc! { "$set": { "control_refs": finding.control_refs.clone() } },
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
refreshed_count += 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if refreshed_count > 0 {
|
||||||
|
tracing::info!(
|
||||||
|
"[{repo_id}] Refreshed control_refs on {refreshed_count} existing findings"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
// Remove stale SBOM entries for this repo before reinserting
|
// Remove stale SBOM entries for this repo before reinserting
|
||||||
if !sbom_entries.is_empty() {
|
if !sbom_entries.is_empty() {
|
||||||
@@ -567,7 +586,21 @@ impl PipelineOrchestrator {
|
|||||||
let Some(path) = ingest_set.get(&a.id).and_then(|ia| ia.working_path.clone()) else {
|
let Some(path) = ingest_set.get(&a.id).and_then(|ia| ia.working_path.clone()) else {
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
all_findings.extend(crate::pipeline::plc::analyze_tree(&path, target_id));
|
let mut source_findings = crate::pipeline::plc::analyze_tree(&path, target_id);
|
||||||
|
// Control mapping for the PLC path (run_plc_scan is separate from
|
||||||
|
// run_pipeline, which does its own mapping). PLC findings carry
|
||||||
|
// file_path/line/cwe, so the semantic pass reads each region under this
|
||||||
|
// source's `path` and stamps master-control refs. The LUT + grounded
|
||||||
|
// surface passes are code-pattern / CRA-specific and don't apply to
|
||||||
|
// IEC 61131-3 control logic, so only the semantic pass runs here.
|
||||||
|
crate::controls::semantic_stamp_findings(
|
||||||
|
&self.config,
|
||||||
|
self.llm.clone(),
|
||||||
|
&path,
|
||||||
|
&mut source_findings,
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
all_findings.extend(source_findings);
|
||||||
// Control-application SBOM: CODESYS libraries + runtime from a
|
// Control-application SBOM: CODESYS libraries + runtime from a
|
||||||
// `.projectarchive` (uploaded, or committed in the working tree).
|
// `.projectarchive` (uploaded, or committed in the working tree).
|
||||||
let archive = a
|
let archive = a
|
||||||
@@ -592,6 +625,7 @@ impl PipelineOrchestrator {
|
|||||||
);
|
);
|
||||||
|
|
||||||
let mut new_count = 0u32;
|
let mut new_count = 0u32;
|
||||||
|
let mut refreshed_count = 0u32;
|
||||||
for mut finding in all_findings {
|
for mut finding in all_findings {
|
||||||
finding.scan_run_id = Some(scan_run_id.to_string());
|
finding.scan_run_id = Some(scan_run_id.to_string());
|
||||||
if self
|
if self
|
||||||
@@ -603,8 +637,27 @@ impl PipelineOrchestrator {
|
|||||||
{
|
{
|
||||||
self.db.findings().insert_one(&finding).await?;
|
self.db.findings().insert_one(&finding).await?;
|
||||||
new_count += 1;
|
new_count += 1;
|
||||||
|
} else if !finding.control_refs.is_empty() {
|
||||||
|
// Re-scan refresh: mirror run_pipeline — persist newly-computed
|
||||||
|
// control_refs onto a PLC finding first seen before the semantic
|
||||||
|
// pass ran. The insert path alone never would, so without this a
|
||||||
|
// PLC re-scan can only pick up mappings via a delete + re-add.
|
||||||
|
self.db
|
||||||
|
.findings()
|
||||||
|
.update_one(
|
||||||
|
doc! { "fingerprint": &finding.fingerprint },
|
||||||
|
doc! { "$set": { "control_refs": finding.control_refs.clone() } },
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
refreshed_count += 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if refreshed_count > 0 {
|
||||||
|
tracing::info!(
|
||||||
|
target_id,
|
||||||
|
"Refreshed control_refs on {refreshed_count} existing PLC findings"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
if !all_sbom.is_empty() {
|
if !all_sbom.is_empty() {
|
||||||
if let Err(e) = self
|
if let Err(e) = self
|
||||||
|
|||||||
@@ -108,16 +108,25 @@ Every finding maps to its exact control family, with the most specific control o
|
|||||||
- **Generic catch-all controls co-occur.** `mc-20890 secure_development_security_code_review` appears in the top-K for many code-security findings because it is semantically near almost all of them. It's harmless (the judge grounds it, and it never crowds out the specific controls — the SQLi example didn't get it) but is a candidate for future down-weighting.
|
- **Generic catch-all controls co-occur.** `mc-20890 secure_development_security_code_review` appears in the top-K for many code-security findings because it is semantically near almost all of them. It's harmless (the judge grounds it, and it never crowds out the specific controls — the SQLi example didn't get it) but is a candidate for future down-weighting.
|
||||||
- **Corpus classification noise.** The master-controls `verification_method` classification is imperfect — e.g. a documentation control (`eu_declaration_accuracy`) is currently tagged `source_code`. That's a corpus-side data-quality issue, separate from the mapping engine.
|
- **Corpus classification noise.** The master-controls `verification_method` classification is imperfect — e.g. a documentation control (`eu_declaration_accuracy`) is currently tagged `source_code`. That's a corpus-side data-quality issue, separate from the mapping engine.
|
||||||
|
|
||||||
|
## Emitting over MCP — closing the loop
|
||||||
|
|
||||||
|
Findings don't just land in the dashboard; they flow to breakpilot-compliance as OSCAL over the scanner's MCP server, so the compliance report is assembled from real, control-tagged findings.
|
||||||
|
|
||||||
|
- The MCP server exposes an **`oscal_assessment`** tool: given a `repo_id`, it emits a standard OSCAL 1.1 assessment-results document for that repo's findings — mapped findings target their controls via the stamped `control_refs`, and unmapped findings are reported **as-is** (as observations), so nothing is lost.
|
||||||
|
- breakpilot pulls it: `POST /v1/cra/oscal-from-scanner` calls `oscal_assessment` over MCP (Streamable HTTP + bearer) and consumes the pre-computed OSCAL — rather than pulling raw findings and re-assessing.
|
||||||
|
|
||||||
|
**Operational note — tenant context over HTTP.** The MCP server is multi-tenant; the bearer token resolves a tenant whose per-tenant database the tools query. rmcp's Streamable HTTP transport runs each session's tool calls in a `tokio::spawn`ed task, and `task_local`s do **not** cross a spawn — so binding the tenant in a per-request middleware `task_local` leaves tool handlers with no context (every call fails `no tenant context`). The fix is to bind the tenant to the **per-session server instance** at creation (the factory runs in the request scope before the spawn), not to a per-request task_local. Until this was fixed, the loop silently failed over HTTP and consumers fell back to demo data.
|
||||||
|
|
||||||
## Configuration
|
## Configuration
|
||||||
|
|
||||||
| Variable | Effect |
|
| Variable | Effect |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| `BREAKPILOT_BASE_URL` | breakpilot-compliance root; enables control ingest + Stage 5b. Unset disables all control mapping. |
|
| `BREAKPILOT_BASE_URL` | breakpilot-compliance root; enables control ingest + all mapping passes. **Unset disables all control mapping** — findings are produced without `control_refs`. |
|
||||||
| `BREAKPILOT_SEMANTIC_MAPPING` | Enables Stage 5c (semantic master-controls mapping). Default off. |
|
| `BREAKPILOT_SEMANTIC_MAPPING` | Stage 5c (semantic master-controls mapping). **Default on** (validated live). |
|
||||||
| `BREAKPILOT_GROUNDED_CHECKS` | Enables Stage 5d (grounded surface checks). Default off. |
|
| `BREAKPILOT_GROUNDED_CHECKS` | Stage 5d (grounded surface checks). **Default on** (validated live). |
|
||||||
| `BREAKPILOT_SNAPSHOT_DIR` | Where OSCAL catalog snapshots and the cached control-embedding index live. |
|
| `BREAKPILOT_SNAPSHOT_DIR` | Where OSCAL catalog snapshots and the cached control-embedding index live. |
|
||||||
|
|
||||||
The semantic and grounded passes are gated because they are the heavier, less deterministic paths; they stay off until verified live against a deployed catalog. The live verification lives in `compliance-agent/tests/c5_semantic_live.rs` (ignored; run with `--ignored`).
|
The semantic and grounded passes default **on** now that both are validated live; each is still a no-op if `BREAKPILOT_BASE_URL` is unset or the catalog is unreachable, so they only ever add coverage. The live verifications live in `compliance-agent/tests/c5_semantic_live.rs` and `grounded_surface_live.rs` (ignored; run with `--ignored`).
|
||||||
|
|
||||||
## Appendix — the master-controls data pipeline
|
## Appendix — the master-controls data pipeline
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user