Compare commits

..
Author SHA1 Message Date
Sharang ParnerkarandClaude Opus 4.7 bec47f8c7d feat(dashboard): proactively refresh expired Keycloak tokens
CI / Check (pull_request) Successful in 8m7s
CI / Detect Changes (pull_request) Has been skipped
CI / Deploy Agent (pull_request) Has been skipped
CI / Deploy Dashboard (pull_request) Has been skipped
CI / Deploy Docs (pull_request) Has been skipped
CI / Deploy MCP (pull_request) Has been skipped
The dashboard stored a refresh_token in the session at login (auth.rs)
but never used it. Once the access_token's 5-minute lifespan ran out,
every subsequent agent call failed with 401 ExpiredSignature. The UI
showed "unable to load X" until the user logged out and back in.

Fix: before attaching the bearer, decode the JWT's `exp` claim and
proactively refresh via the stored refresh_token if the token is
expired or within REFRESH_SKEW_SECS (30s) of expiry. Updates the
session with the new access_token (and rotated refresh_token if KC
sends one). Refresh failures fall through with the stale token so the
agent's 401 surfaces to the UI rather than failing the request at the
dashboard layer.

Why "proactive" instead of "retry on 401"
- Saves a wasted round-trip on every agent call once the token has
  aged past 5 min.
- Doesn't require cloning RequestBuilder bodies for retry.
- Same end state — fresh token reaches the agent.

Test plan
- cargo test -p compliance-dashboard --features server
  --no-default-features infrastructure::agent_client::tests — 5 pass:
    * expired JWT → refresh
    * near-expiry within skew window → refresh
    * fresh JWT → no refresh
    * malformed/empty JWT → refresh (defensive)
    * JWT without exp claim → refresh (defensive)
- Manual after deploy: dashboard works past the 5-min token lifespan
  without manual re-login.

Note
- The refresh code addresses the ExpiredSignature failure mode. The
  separate "JWT is missing tenant_id claim" 401 is a Keycloak realm
  config issue (the user logging in lacks the M7.1 attributes that
  the protocol mappers consume) and is fixed by realm/attribute
  config, not by this PR.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-06-17 21:38:06 +02:00
44 changed files with 73 additions and 4108 deletions
-13
View File
@@ -7,17 +7,4 @@ ignore = [
# not a realistic attack surface here. Revisit when mongodb bumps hickory.
"RUSTSEC-2026-0118", # NSEC3 loop, no fix available upstream
"RUSTSEC-2026-0119", # O(n²) name compression, fixed in hickory-proto >=0.26.1
# rmcp 0.16.0 — DNS rebinding in Streamable HTTP server transport (missing
# Host header validation). Patched in rmcp >= 1.4.0, which is a major API
# version jump from our pin; rmcp shipped 0.x → 1.x → 2.x in three months
# and the migration touches every tool handler + the auth middleware we
# just landed in #92. Threat model in our deployment: the MCP server is
# exposed at a public hostname (comp-mcp-dev.meghsakha.com) behind orca's
# TLS-terminating ingress with per-tenant bearer auth — the attack model
# (browser DNS-rebinding into localhost MCP server) doesn't directly apply.
# Defense-in-depth Host-header check is still a worthwhile follow-up.
# FOLLOW-UP: bump rmcp to 2.x in a dedicated PR (M7.3 follow-up, sized
# multi-hour due to API surface change).
"RUSTSEC-2026-0189",
]
-15
View File
@@ -13,9 +13,6 @@ env:
# both --features server and --features web shares common crate work.
RUSTC_WRAPPER: /usr/local/bin/sccache
SCCACHE_DIR: /tmp/sccache
# compliance-agent depends on tramiton-core via git; use the system git so the
# credential rewrite below (see "Configure git auth ...") is honored on fetch.
CARGO_NET_GIT_FETCH_WITH_CLI: "true"
# Cancel in-progress runs for the same branch/PR
concurrency:
@@ -49,18 +46,6 @@ jobs:
env:
RUSTC_WRAPPER: ""
# compliance-agent has a git dependency on tramiton-core (a private repo on
# this Gitea instance). Rewrite its SSH URL to HTTPS + a read token so the
# runner can fetch it. Requires a repo secret TRAMITON_FETCH_TOKEN — a
# Gitea PAT for a user with read access to sharang/tramiton.
- name: Configure git auth for private tramiton dependency
run: |
git config --global \
url."https://sharang:${{ secrets.TRAMITON_FETCH_TOKEN }}@gitea.meghsakha.com/".insteadOf \
"ssh://git@gitea.meghsakha.com:22222/"
env:
RUSTC_WRAPPER: ""
# Format (no compilation needed)
- name: Format
run: cargo fmt --all --check
Generated
+6 -73
View File
@@ -676,7 +676,6 @@ dependencies = [
"jsonwebtoken",
"mongodb",
"octocrab",
"rand 0.9.2",
"regex",
"reqwest",
"secrecy",
@@ -692,7 +691,6 @@ dependencies = [
"tower-http",
"tracing",
"tracing-subscriber",
"tramiton-core",
"urlencoding",
"uuid",
"walkdir",
@@ -820,15 +818,12 @@ dependencies = [
"bson",
"chrono",
"compliance-core",
"dashmap",
"dotenvy",
"hex",
"mongodb",
"rmcp",
"schemars 1.2.1",
"serde",
"serde_json",
"sha2",
"thiserror 2.0.18",
"tokio",
"tower-http",
@@ -1119,9 +1114,9 @@ dependencies = [
[[package]]
name = "crossbeam-epoch"
version = "0.9.20"
version = "0.9.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f"
checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e"
dependencies = [
"crossbeam-utils",
]
@@ -4198,7 +4193,7 @@ version = "3.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "219cb19e96be00ab2e37d6e299658a0cfa83e52429179969b0f0121b4ac46983"
dependencies = [
"toml_edit 0.23.10+spec-1.0.0",
"toml_edit",
]
[[package]]
@@ -4283,9 +4278,9 @@ dependencies = [
[[package]]
name = "quinn-proto"
version = "0.11.15"
version = "0.11.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4fcb935c5bec503c2f0e306bdd3e58bb9029dcb14fa8d9ac76e3a5256ac0763e"
checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098"
dependencies = [
"bytes",
"getrandom 0.3.4",
@@ -4997,15 +4992,6 @@ dependencies = [
"syn",
]
[[package]]
name = "serde_spanned"
version = "0.6.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf41e0cfaf7226dca15e8197172c295a782857fcb97fad1808a166870dee75a3"
dependencies = [
"serde",
]
[[package]]
name = "serde_urlencoded"
version = "0.7.1"
@@ -5820,27 +5806,6 @@ dependencies = [
"tokio",
]
[[package]]
name = "toml"
version = "0.8.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dc1beb996b9d83529a9e75c17a1686767d148d70663143c7854d8b4a09ced362"
dependencies = [
"serde",
"serde_spanned",
"toml_datetime 0.6.11",
"toml_edit 0.22.27",
]
[[package]]
name = "toml_datetime"
version = "0.6.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "22cddaf88f4fbc13c51aebbf5f8eceb5c7c5a9da2ac40a13519eb5b0a0e8f11c"
dependencies = [
"serde",
]
[[package]]
name = "toml_datetime"
version = "0.7.5+spec-1.1.0"
@@ -5850,20 +5815,6 @@ dependencies = [
"serde_core",
]
[[package]]
name = "toml_edit"
version = "0.22.27"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a"
dependencies = [
"indexmap 2.13.0",
"serde",
"serde_spanned",
"toml_datetime 0.6.11",
"toml_write",
"winnow",
]
[[package]]
name = "toml_edit"
version = "0.23.10+spec-1.0.0"
@@ -5871,7 +5822,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "84c8b9f757e028cee9fa244aea147aab2a9ec09d5325a9b01e0a49730c2b5269"
dependencies = [
"indexmap 2.13.0",
"toml_datetime 0.7.5+spec-1.1.0",
"toml_datetime",
"toml_parser",
"winnow",
]
@@ -5885,12 +5836,6 @@ dependencies = [
"winnow",
]
[[package]]
name = "toml_write"
version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801"
[[package]]
name = "tonic"
version = "0.12.3"
@@ -6137,18 +6082,6 @@ dependencies = [
"wasm-bindgen",
]
[[package]]
name = "tramiton-core"
version = "0.4.0"
source = "git+ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git?tag=v0.4.0#e3dc1bf7027a2f6d7b1fe43043d6dfa887ce4af3"
dependencies = [
"serde",
"tempfile",
"thiserror 1.0.69",
"toml",
"walkdir",
]
[[package]]
name = "tree-sitter"
version = "0.24.7"
-2
View File
@@ -34,5 +34,3 @@ zip = { version = "2", features = ["aes-crypto", "deflate"] }
dashmap = "6"
tokio-stream = { version = "0.1", features = ["sync"] }
aes-gcm = "0.10"
rand = "0.9"
base64 = "0.22"
-6
View File
@@ -10,11 +10,6 @@ workspace = true
compliance-core = { workspace = true, features = ["mongodb", "telemetry", "axum"] }
compliance-graph = { path = "../compliance-graph" }
compliance-dast = { path = "../compliance-dast" }
# Native firmware build/target detection for bare-metal & RTOS artifacts.
# Same-company IP, used directly (not via CLI) so the whole tramiton suite is
# available to the onboarding classifier. NOTE: CI must be able to fetch this
# private repo (see the git-auth step in .gitea/workflows/ci.yml).
tramiton-core = { git = "ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git", tag = "v0.4.0" }
serde = { workspace = true }
serde_json = { workspace = true }
tokio = { workspace = true }
@@ -47,7 +42,6 @@ tokio-tungstenite = { version = "0.26", features = ["rustls-tls-webpki-roots"] }
futures-core = "0.3"
dashmap = { workspace = true }
tokio-stream = { workspace = true }
rand = { workspace = true }
[dev-dependencies]
compliance-core = { workspace = true, features = ["mongodb", "axum"] }
-115
View File
@@ -1,115 +0,0 @@
//! Cross-tenant admin endpoints (`/api/v1/admin/*`).
//!
//! Operator-only. Auth is a **static bearer token** (`ADMIN_API_TOKEN`
//! env on the agent) — explicitly NOT a Keycloak JWT, because the
//! whole point of these endpoints is to operate ACROSS tenants. A
//! customer JWT (which always carries a single tenant_id) has no
//! business mounting them.
//!
//! Routes are only registered when `ADMIN_API_TOKEN` is set. With no
//! token, the endpoints don't exist at all (404), which is a stronger
//! guarantee than "401 if you guess the path".
//!
//! Operations:
//! - `GET /api/v1/admin/tenants` — list tenant DBs
//! - `DELETE /api/v1/admin/tenants/{tenant_id}` — GDPR delete
//!
//! Tenant ids in URLs are passed as-is to `DatabasePool::drop_tenant`,
//! which sanitises them the same way it does for creation. Listing
//! returns the raw DB names from `list_tenant_db_names` — operators
//! can reverse-derive the tenant_id from the prefix.
use axum::extract::{Extension, Path, Request};
use axum::http::{header, StatusCode};
use axum::middleware::Next;
use axum::response::{IntoResponse, Response};
use axum::Json;
use secrecy::ExposeSecret;
use serde::Serialize;
use super::dto::AgentExt;
#[derive(Serialize)]
pub struct ListTenantDbsResponse {
pub tenant_db_names: Vec<String>,
}
#[tracing::instrument(skip_all)]
pub async fn list_tenant_dbs(
Extension(agent): AgentExt,
) -> Result<Json<ListTenantDbsResponse>, StatusCode> {
let names = agent.db_pool.list_tenant_db_names().await.map_err(|e| {
tracing::error!("admin: list_tenant_db_names failed: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
Ok(Json(ListTenantDbsResponse {
tenant_db_names: names,
}))
}
#[tracing::instrument(skip_all, fields(tenant_id = %tenant_id))]
pub async fn drop_tenant_db(
Extension(agent): AgentExt,
Path(tenant_id): Path<String>,
) -> Result<Json<serde_json::Value>, StatusCode> {
agent.db_pool.drop_tenant(&tenant_id).await.map_err(|e| {
tracing::error!("admin: drop_tenant failed: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
Ok(Json(serde_json::json!({ "status": "dropped" })))
}
/// Constant-time-ish comparison of the configured admin token against
/// the incoming bearer. Uses `subtle`-style byte equality so timing
/// attacks can't probe the token character by character.
fn tokens_eq(a: &str, b: &str) -> bool {
if a.len() != b.len() {
return false;
}
let mut diff = 0u8;
for (x, y) in a.bytes().zip(b.bytes()) {
diff |= x ^ y;
}
diff == 0
}
/// Middleware enforcing the static `ADMIN_API_TOKEN`. Mounted only on
/// the admin sub-router, so this never runs on customer routes.
pub async fn require_admin_token(
Extension(agent): AgentExt,
request: Request,
next: Next,
) -> Response {
let Some(expected) = agent.config.admin_api_token.as_ref() else {
// Belt-and-braces — if the routes were somehow mounted without
// a token configured, refuse rather than no-op-pass.
return (StatusCode::NOT_FOUND, "admin disabled").into_response();
};
let presented = request
.headers()
.get(header::AUTHORIZATION)
.and_then(|v| v.to_str().ok())
.and_then(|s| s.strip_prefix("Bearer "))
.map(|s| s.trim());
let Some(presented) = presented.filter(|s| !s.is_empty()) else {
return (StatusCode::UNAUTHORIZED, "Missing bearer token").into_response();
};
if !tokens_eq(presented, expected.expose_secret()) {
return (StatusCode::UNAUTHORIZED, "Invalid admin token").into_response();
}
next.run(request).await
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn tokens_eq_basic() {
assert!(tokens_eq("abc", "abc"));
assert!(!tokens_eq("abc", "abd"));
assert!(!tokens_eq("abc", "abcd"));
assert!(!tokens_eq("", "x"));
assert!(tokens_eq("", ""));
}
}
@@ -1,186 +0,0 @@
//! `/api/v1/mcp-tokens` — per-tenant API tokens for the MCP server.
//!
//! These are opaque static bearers issued via the dashboard (or a
//! direct curl with a KC JWT) and copied into LLM clients (Claude
//! Desktop / Cursor / ChatGPT). The MCP server hashes incoming bearers
//! and looks them up in the cross-tenant `<prefix>__admin.mcp_tokens`
//! collection to derive the tenant_id for routing.
//!
//! The raw token is shown to the caller exactly once at creation; the
//! database only ever stores the SHA-256 hash. Revocation is a soft
//! delete (sets `revoked: true`) so the audit log keeps the record.
use axum::extract::{Extension, Path};
use axum::http::StatusCode;
use axum::Json;
use base64::{engine::general_purpose::URL_SAFE_NO_PAD, Engine as _};
use compliance_core::models::{McpToken, McpTokenView};
use compliance_core::tenant_ctx::TenantCtx;
use mongodb::bson::doc;
use rand::RngCore;
use sha2::{Digest, Sha256};
use super::dto::{AgentExt, ApiResponse};
/// Mongo collection name inside the admin DB.
const COLLECTION: &str = "mcp_tokens";
/// Token prefix the MCP server expects on every bearer.
const TOKEN_PREFIX: &str = "mcpt_";
/// Bytes of randomness behind each token. 32 → ~256 bits.
/// Encoded as URL-safe base64 without padding → 43 chars.
/// Combined with `mcpt_` → 48-char tokens.
const TOKEN_RAND_BYTES: usize = 32;
#[derive(serde::Deserialize)]
pub struct CreateMcpTokenRequest {
pub name: String,
}
/// Returned exactly once at creation. The `token` field is gone from
/// the listing endpoint — the user must save it now.
#[derive(serde::Serialize)]
pub struct CreateMcpTokenResponse {
pub token: String,
pub view: McpTokenView,
}
/// `POST /api/v1/mcp-tokens` — mint a new token for the caller's tenant.
#[tracing::instrument(skip_all)]
pub async fn create_mcp_token(
Extension(agent): AgentExt,
tenant: TenantCtx,
Json(req): Json<CreateMcpTokenRequest>,
) -> Result<Json<CreateMcpTokenResponse>, StatusCode> {
if req.name.trim().is_empty() {
return Err(StatusCode::BAD_REQUEST);
}
let raw = generate_token();
let token_hash = sha256_hex(&raw);
let token_prefix: String = raw.chars().take(12).collect();
let mut token = McpToken {
id: None,
token_hash,
token_prefix,
tenant_id: tenant.0.tenant_id.clone(),
name: req.name.trim().to_string(),
created_by: tenant.0.user_id.clone(),
created_at: chrono::Utc::now(),
last_used_at: None,
revoked: false,
};
let col = agent.db_pool.admin_db().collection::<McpToken>(COLLECTION);
let res = col.insert_one(&token).await.map_err(|e| {
tracing::error!("Failed to insert MCP token: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
token.id = res.inserted_id.as_object_id();
Ok(Json(CreateMcpTokenResponse {
view: McpTokenView::from(&token),
token: raw,
}))
}
/// `GET /api/v1/mcp-tokens` — list tokens for the caller's tenant.
/// Hash is never returned; only metadata + the 12-char prefix so the
/// user can identify which row is which.
#[tracing::instrument(skip_all)]
pub async fn list_mcp_tokens(
Extension(agent): AgentExt,
tenant: TenantCtx,
) -> Result<Json<ApiResponse<Vec<McpTokenView>>>, StatusCode> {
let col = agent.db_pool.admin_db().collection::<McpToken>(COLLECTION);
let mut cursor = col
.find(doc! { "tenant_id": &tenant.0.tenant_id })
.sort(doc! { "created_at": -1 })
.await
.map_err(|e| {
tracing::error!("Failed to list MCP tokens: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
let mut out = Vec::new();
while cursor.advance().await.map_err(|e| {
tracing::warn!("MCP tokens cursor advance failed: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})? {
match cursor.deserialize_current() {
Ok(t) => out.push(McpTokenView::from(&t)),
Err(e) => tracing::warn!("Failed to deserialize MCP token: {e}"),
}
}
Ok(Json(ApiResponse {
data: out,
total: None,
page: None,
}))
}
/// `DELETE /api/v1/mcp-tokens/{id}` — revoke (soft delete).
/// Scoped to the caller's tenant: a user can't revoke another tenant's
/// token even if they guess its id.
#[tracing::instrument(skip_all, fields(id = %id))]
pub async fn revoke_mcp_token(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<serde_json::Value>, StatusCode> {
let oid = mongodb::bson::oid::ObjectId::parse_str(&id).map_err(|_| StatusCode::BAD_REQUEST)?;
let col = agent.db_pool.admin_db().collection::<McpToken>(COLLECTION);
let result = col
.update_one(
doc! { "_id": oid, "tenant_id": &tenant.0.tenant_id },
doc! { "$set": { "revoked": true } },
)
.await
.map_err(|e| {
tracing::error!("Failed to revoke MCP token: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
if result.matched_count == 0 {
return Err(StatusCode::NOT_FOUND);
}
Ok(Json(serde_json::json!({ "status": "revoked" })))
}
/// 32 bytes random → URL-safe base64 → 43 chars, no padding.
/// Prefixed with `mcpt_` so the MCP server can sniff the format
/// before bothering with the DB lookup.
fn generate_token() -> String {
let mut bytes = [0u8; TOKEN_RAND_BYTES];
rand::rng().fill_bytes(&mut bytes);
format!("{TOKEN_PREFIX}{}", URL_SAFE_NO_PAD.encode(bytes))
}
fn sha256_hex(s: &str) -> String {
let mut h = Sha256::new();
h.update(s.as_bytes());
hex::encode(h.finalize())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn generated_tokens_are_unique_and_prefixed() {
let a = generate_token();
let b = generate_token();
assert_ne!(a, b);
assert!(a.starts_with(TOKEN_PREFIX));
assert!(b.starts_with(TOKEN_PREFIX));
// 5 + 43 = 48 chars
assert_eq!(a.len(), 5 + 43);
}
#[test]
fn sha256_is_stable_and_64_hex() {
let h = sha256_hex("mcpt_abc");
assert_eq!(h.len(), 64);
assert!(h.chars().all(|c| c.is_ascii_hexdigit()));
assert_eq!(sha256_hex("mcpt_abc"), h);
}
}
-2
View File
@@ -1,4 +1,3 @@
pub mod admin;
pub mod chat;
pub mod dast;
pub mod dto;
@@ -7,7 +6,6 @@ pub mod graph;
pub mod health;
pub mod help_chat;
pub mod issues;
pub mod mcp_tokens;
pub mod notifications;
pub mod pentest_handlers;
pub use pentest_handlers as pentest;
+14 -15
View File
@@ -2,6 +2,7 @@ use axum::routing::{delete, get, patch, post};
use axum::Router;
use crate::api::handlers;
use crate::webhooks;
pub fn build_router() -> Router {
Router::new()
@@ -46,15 +47,6 @@ pub fn build_router() -> Router {
.route("/api/v1/sbom/diff", get(handlers::sbom_diff))
.route("/api/v1/issues", get(handlers::list_issues))
.route("/api/v1/scan-runs", get(handlers::list_scan_runs))
// MCP token management (per-tenant API tokens for the MCP server)
.route(
"/api/v1/mcp-tokens",
get(handlers::mcp_tokens::list_mcp_tokens).post(handlers::mcp_tokens::create_mcp_token),
)
.route(
"/api/v1/mcp-tokens/{id}",
delete(handlers::mcp_tokens::revoke_mcp_token),
)
// Graph API endpoints
.route("/api/v1/graph/{repo_id}", get(handlers::graph::get_graph))
.route(
@@ -183,10 +175,17 @@ pub fn build_router() -> Router {
"/api/v1/pentest/stats",
get(handlers::pentest::pentest_stats),
)
// Webhook routes live on the separate webhook server (port 3002,
// see crate::webhooks::server). The M7.2-C tenant-in-URL form is
// `/webhook/{tenant_id}/{platform}/{repo_id}` and the handlers
// expect a (tenant_id, repo_id) path tuple. Anything mounting
// them here on the API server would mismatch the handler
// signature, so the routes are not exported.
// Webhook endpoints (proxied through dashboard)
.route(
"/webhook/github/{repo_id}",
post(webhooks::github::handle_github_webhook),
)
.route(
"/webhook/gitlab/{repo_id}",
post(webhooks::gitlab::handle_gitlab_webhook),
)
.route(
"/webhook/gitea/{repo_id}",
post(webhooks::gitea::handle_gitea_webhook),
)
}
+1 -24
View File
@@ -4,8 +4,7 @@ use axum::extract::Request;
use axum::http::HeaderValue;
use axum::middleware::Next;
use axum::response::Response;
use axum::routing::{delete, get};
use axum::{middleware, Extension, Router};
use axum::{middleware, Extension};
use tokio::sync::RwLock;
use tower_http::cors::CorsLayer;
use tower_http::set_header::SetResponseHeaderLayer;
@@ -15,7 +14,6 @@ use compliance_core::auth::{require_jwt_auth, require_tenant_status, JwksState};
use compliance_core::{TenantContext, TenantStatus};
use crate::agent::ComplianceAgent;
use crate::api::handlers;
use crate::api::routes;
use crate::error::AgentError;
@@ -52,28 +50,7 @@ pub async fn inject_dev_tenant(mut request: Request, next: Next) -> Response {
}
pub async fn start_api_server(agent: ComplianceAgent, port: u16) -> Result<(), AgentError> {
// Admin sub-router. Routes are only mounted when ADMIN_API_TOKEN is
// configured — without it, the paths don't exist at all (404 rather
// than 401), so an operator who hasn't opted in can't fingerprint
// the surface area.
let admin_router: Router = if agent.config.admin_api_token.is_some() {
tracing::info!("Admin API enabled — /api/v1/admin/* mounted behind ADMIN_API_TOKEN bearer");
Router::new()
.route(
"/api/v1/admin/tenants",
get(handlers::admin::list_tenant_dbs),
)
.route(
"/api/v1/admin/tenants/{tenant_id}",
delete(handlers::admin::drop_tenant_db),
)
.layer(middleware::from_fn(handlers::admin::require_admin_token))
} else {
Router::new()
};
let mut app = routes::build_router()
.merge(admin_router)
.layer(Extension(Arc::new(agent.clone())))
.layer(CorsLayer::permissive())
.layer(TraceLayer::new_for_http())
-217
View File
@@ -1,217 +0,0 @@
//! Firmware classification via tramiton.
//!
//! tramiton is the company's firmware build/repro engine; we do not re-implement
//! its detection. We depend on `tramiton-core` directly (same-company IP) and run
//! its provider analysis in-process behind a [`FirmwareDetector`] port, mapping
//! tramiton's `BuildPlan` onto a [`TargetType`]. A deterministic
//! [`MockFirmwareDetector`] backs the tests so CI unit tests need neither the
//! tramiton sources nor a real firmware tree.
use std::path::Path;
use compliance_core::error::CoreError;
use compliance_core::models::{DetectedFact, TargetType};
use compliance_core::traits::ClassifierVerdict;
/// A minimal firmware-detection summary, mapped from tramiton's `BuildPlan`.
/// Kept small and tramiton-independent so the classifier and the test mock don't
/// need to construct a full tramiton plan.
#[derive(Debug, Clone, Default)]
pub struct FirmwareDetection {
/// The detecting provider (e.g. `zephyr`, `cmake`, `source-archaeology`).
pub provider: String,
/// Detection confidence: `low` | `medium` | `high`.
pub confidence: String,
/// Build-system label (e.g. `Zephyr`, `ESP-IDF`, `CMake`).
pub build_system: String,
/// Framework, when known (`zephyr`, `esp-idf`, `bare-metal`, ...).
pub framework: Option<String>,
/// Target board / MCU / arch.
pub target: FirmwareTarget,
/// Unresolved gaps in the plan.
pub gaps: Vec<String>,
}
/// The detected firmware target (board / MCU / arch).
#[derive(Debug, Clone, Default)]
pub struct FirmwareTarget {
/// Board name.
pub board: Option<String>,
/// MCU part.
pub mcu: Option<String>,
/// Architecture.
pub arch: Option<String>,
}
/// A source of tramiton firmware detection.
#[allow(async_fn_in_trait)]
pub trait FirmwareDetector: Send + Sync {
/// Run detection over a path, returning a firmware detection if tramiton
/// could form a build plan.
async fn detect(&self, path: &Path) -> Result<Option<FirmwareDetection>, CoreError>;
}
/// Uses `tramiton-core` in-process. The analysis is blocking (filesystem walk),
/// so it runs on a blocking thread to avoid stalling the async runtime. A path
/// with no recognizable build system yields `Ok(None)`.
pub struct TramitonNative;
impl FirmwareDetector for TramitonNative {
async fn detect(&self, path: &Path) -> Result<Option<FirmwareDetection>, CoreError> {
let path = path.to_path_buf();
let plan = tokio::task::spawn_blocking(move || {
let repo = tramiton_core::Repo::new(&path);
tramiton_core::provider::analyze(&repo)
})
.await
.map_err(|e| CoreError::Other(format!("tramiton detect task join error: {e}")))?
.map_err(|e| CoreError::Other(format!("tramiton analyze error: {e}")))?;
Ok(plan.map(|bp| detection_from_build_plan(&bp)))
}
}
/// Map tramiton's `BuildPlan` onto our minimal detection summary.
fn detection_from_build_plan(bp: &tramiton_core::BuildPlan) -> FirmwareDetection {
FirmwareDetection {
provider: bp.provider.clone(),
confidence: bp.confidence.to_string(),
build_system: bp.build_system.label().to_string(),
framework: bp.framework.clone(),
target: FirmwareTarget {
board: bp.target.board.clone(),
mcu: bp.target.mcu.clone(),
arch: bp.target.arch.clone(),
},
gaps: bp.gaps.clone(),
}
}
/// Map a firmware detection to a target type. Framework/build-system signals
/// distinguish RTOS from bare-metal from Yocto.
pub fn detection_to_target_type(det: &FirmwareDetection) -> TargetType {
let framework = det.framework.as_deref().unwrap_or("").to_lowercase();
let build_system = det.build_system.to_lowercase();
let signal = format!("{framework} {build_system} {}", det.provider.to_lowercase());
const RTOS: [&str; 6] = ["zephyr", "esp-idf", "freertos", "nuttx", "riot", "chibios"];
if signal.contains("bitbake") || signal.contains("yocto") || signal.contains("openembedded") {
TargetType::EmbeddedLinuxYocto
} else if RTOS.iter().any(|k| signal.contains(k)) {
TargetType::FirmwareRtos
} else {
TargetType::FirmwareBareMetal
}
}
/// Map tramiton's confidence label to a `[0,1]` score.
fn confidence_score(label: &str) -> f32 {
match label.to_lowercase().as_str() {
"high" => 0.9,
"medium" => 0.6,
"low" => 0.3,
_ => 0.4,
}
}
/// Turn a firmware detection into a classifier verdict, carrying the MCU / board
/// / build-system as facts.
pub fn detection_to_verdict(det: &FirmwareDetection) -> ClassifierVerdict {
let target_type = detection_to_target_type(det);
let mut facts = vec![DetectedFact::new(
"build_system",
det.build_system.clone(),
"tramiton",
)];
if let Some(fw) = &det.framework {
facts.push(DetectedFact::new("framework", fw.clone(), "tramiton"));
}
if let Some(mcu) = &det.target.mcu {
facts.push(DetectedFact::new("mcu", mcu.clone(), "tramiton"));
}
if let Some(board) = &det.target.board {
facts.push(DetectedFact::new("board", board.clone(), "tramiton"));
}
if let Some(arch) = &det.target.arch {
facts.push(DetectedFact::new("arch", arch.clone(), "tramiton"));
}
ClassifierVerdict {
target_type,
confidence: confidence_score(&det.confidence),
facts,
rationale: format!(
"tramiton detected build system '{}'{}",
det.build_system,
det.framework
.as_ref()
.map(|f| format!(" (framework {f})"))
.unwrap_or_default()
),
}
}
/// A deterministic [`FirmwareDetector`] for tests — returns a preset detection.
pub struct MockFirmwareDetector {
/// The detection to return (or `None` for "no detection").
pub detection: Option<FirmwareDetection>,
}
impl FirmwareDetector for MockFirmwareDetector {
async fn detect(&self, _path: &Path) -> Result<Option<FirmwareDetection>, CoreError> {
Ok(self.detection.clone())
}
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
fn detection(build_system: &str, framework: Option<&str>) -> FirmwareDetection {
FirmwareDetection {
provider: build_system.to_string(),
confidence: "high".to_string(),
build_system: build_system.to_string(),
framework: framework.map(|s| s.to_string()),
target: FirmwareTarget {
mcu: Some("stm32f429".to_string()),
..Default::default()
},
gaps: Vec::new(),
}
}
#[test]
fn zephyr_maps_to_rtos() {
assert_eq!(
detection_to_target_type(&detection("zephyr", Some("zephyr"))),
TargetType::FirmwareRtos
);
}
#[test]
fn bare_cmake_maps_to_bare_metal() {
assert_eq!(
detection_to_target_type(&detection("cmake", Some("bare-metal"))),
TargetType::FirmwareBareMetal
);
}
#[test]
fn bitbake_maps_to_yocto() {
assert_eq!(
detection_to_target_type(&detection("bitbake", None)),
TargetType::EmbeddedLinuxYocto
);
}
#[test]
fn verdict_carries_mcu_fact_and_confidence() {
let v = detection_to_verdict(&detection("esp-idf", Some("esp-idf")));
assert_eq!(v.target_type, TargetType::FirmwareRtos);
assert!((v.confidence - 0.9).abs() < f32::EPSILON);
assert!(v
.facts
.iter()
.any(|f| f.key == "mcu" && f.value == "stm32f429"));
}
}
-357
View File
@@ -1,357 +0,0 @@
//! Heuristic target-type classification from artifact kinds and source markers.
//!
//! Complements the tramiton firmware detector: this handles web / backend /
//! mobile / desktop / PLC by sniffing manifest files and file extensions in the
//! ingested code trees, plus strong priors from the artifact kinds themselves
//! (a PLC-project artifact is a PLC target; an `.ipa` is an iOS app).
use std::collections::HashSet;
use std::fs;
use std::path::Path;
use compliance_core::error::CoreError;
use compliance_core::models::{ArtifactKind, DetectedFact, TargetType};
use compliance_core::traits::{ClassificationInput, ClassifierVerdict, TargetClassifier};
/// Max directory depth scanned for marker files.
const SCAN_DEPTH: usize = 2;
/// Markers collected from a code tree.
#[derive(Default)]
struct Markers {
files: HashSet<String>,
dirs: HashSet<String>,
exts: HashSet<String>,
}
impl Markers {
fn has_file(&self, name: &str) -> bool {
self.files.contains(name)
}
fn has_ext(&self, ext: &str) -> bool {
self.exts.contains(ext)
}
fn any_dir_ends_with(&self, suffix: &str) -> bool {
self.dirs.iter().any(|d| d.ends_with(suffix))
}
}
/// Recursively collect marker file/dir/extension names up to [`SCAN_DEPTH`].
fn collect_markers(root: &Path) -> Markers {
let mut m = Markers::default();
scan_dir(root, 0, &mut m);
m
}
fn scan_dir(dir: &Path, depth: usize, m: &mut Markers) {
let Ok(entries) = fs::read_dir(dir) else {
return;
};
for entry in entries.flatten() {
let path = entry.path();
let name = entry.file_name().to_string_lossy().to_lowercase();
if path.is_dir() {
m.dirs.insert(name);
if depth < SCAN_DEPTH {
scan_dir(&path, depth + 1, m);
}
} else {
if let Some(ext) = path.extension() {
m.exts.insert(ext.to_string_lossy().to_lowercase());
}
m.files.insert(name);
}
}
}
/// Whether a `package.json` at `root` looks like a front-end app.
fn package_json_is_frontend(root: &Path) -> bool {
let Ok(content) = fs::read_to_string(root.join("package.json")) else {
return false;
};
let c = content.to_lowercase();
["react", "next", "vue", "@angular", "svelte", "vite"]
.iter()
.any(|f| c.contains(f))
}
/// The heuristic classifier: artifact-kind priors + source-tree markers.
pub struct HeuristicClassifier;
impl HeuristicClassifier {
/// Verdicts from the artifact kinds alone (no filesystem needed).
fn kind_priors(&self, input: &ClassificationInput<'_>) -> Vec<ClassifierVerdict> {
let mut out = Vec::new();
for a in input.artifacts {
let lower = a.source_ref.to_lowercase();
match a.kind {
ArtifactKind::PlcProject => out.push(verdict(
TargetType::PlcSps,
0.85,
"PLC project artifact",
vec![],
)),
ArtifactKind::MobilePackage => {
let (tt, why) = if lower.ends_with(".ipa") {
(TargetType::IosApp, "iOS package (.ipa)")
} else {
(TargetType::AndroidApp, "Android package (.apk/.aab)")
};
out.push(verdict(tt, 0.85, why, vec![]));
}
ArtifactKind::ContainerImage => out.push(verdict(
TargetType::BackendService,
0.4,
"container image",
vec![],
)),
ArtifactKind::FirmwareImage => out.push(verdict(
TargetType::FirmwareBareMetal,
0.35,
"firmware image (pending tramiton detection)",
vec![],
)),
ArtifactKind::LiveUrl if input.artifacts.len() == 1 => {
out.push(verdict(TargetType::WebApp, 0.3, "live URL only", vec![]))
}
_ => {}
}
}
out
}
/// Verdicts from scanning the ingested code trees for manifest markers.
fn source_verdicts(&self, input: &ClassificationInput<'_>) -> Vec<ClassifierVerdict> {
let mut out = Vec::new();
for a in input.artifacts {
if !matches!(a.kind, ArtifactKind::GitRepo | ArtifactKind::SourceArchive) {
continue;
}
let Some(path) = input.working_paths.get(&a.id) else {
continue;
};
let m = collect_markers(path);
// Mobile (checked first — strongest signal).
if m.has_file("androidmanifest.xml") || m.has_ext("apk") || m.has_ext("aab") {
out.push(verdict(
TargetType::AndroidApp,
0.8,
"Android manifest / gradle",
facts_lang("kotlin/java"),
));
}
if m.any_dir_ends_with(".xcodeproj")
|| m.has_file("info.plist")
|| m.has_file("podfile")
|| m.has_ext("ipa")
{
out.push(verdict(
TargetType::IosApp,
0.8,
"Xcode project / Info.plist",
facts_lang("swift/objc"),
));
}
// Desktop.
if m.has_ext("sln")
|| m.has_ext("csproj")
|| m.has_ext("vcxproj")
|| m.has_ext("desktop")
{
out.push(verdict(
TargetType::DesktopApp,
0.7,
"desktop project files",
facts_lang("dotnet/native"),
));
}
// PLC.
if m.has_ext("st") {
out.push(verdict(
TargetType::PlcSps,
0.8,
"Structured Text sources",
facts_lang("iec-61131-3"),
));
}
// Web vs backend from package.json.
if m.has_file("package.json") {
if package_json_is_frontend(path) {
out.push(verdict(
TargetType::WebApp,
0.65,
"package.json with a front-end framework",
facts_lang("javascript"),
));
} else {
out.push(verdict(
TargetType::BackendService,
0.55,
"package.json (no front-end framework)",
facts_lang("javascript"),
));
}
}
// Backend languages.
for (file, lang) in [
("cargo.toml", "rust"),
("go.mod", "go"),
("pom.xml", "java"),
("requirements.txt", "python"),
("pyproject.toml", "python"),
] {
if m.has_file(file) {
out.push(verdict(
TargetType::BackendService,
0.6,
"backend build manifest",
facts_lang(lang),
));
}
}
// Container-only.
if m.has_file("dockerfile") && out.is_empty() {
out.push(verdict(
TargetType::BackendService,
0.4,
"Dockerfile",
facts_lang("container"),
));
}
}
out
}
}
impl TargetClassifier for HeuristicClassifier {
fn name(&self) -> &str {
"heuristic"
}
async fn classify(
&self,
input: &ClassificationInput<'_>,
) -> Result<Vec<ClassifierVerdict>, CoreError> {
let mut out = self.kind_priors(input);
out.extend(self.source_verdicts(input));
Ok(out)
}
}
fn verdict(
target_type: TargetType,
confidence: f32,
rationale: &str,
facts: Vec<DetectedFact>,
) -> ClassifierVerdict {
ClassifierVerdict {
target_type,
confidence,
facts,
rationale: rationale.to_string(),
}
}
fn facts_lang(lang: &str) -> Vec<DetectedFact> {
vec![DetectedFact::new("language", lang, "heuristic")]
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
use compliance_core::models::Artifact;
use std::collections::HashMap;
use std::path::PathBuf;
struct Scratch(PathBuf);
impl Scratch {
fn new() -> Self {
let p = std::env::temp_dir().join(format!("cs-classify-{}", uuid::Uuid::new_v4()));
fs::create_dir_all(&p).expect("mkdir");
Self(p)
}
}
impl Drop for Scratch {
fn drop(&mut self) {
let _ = fs::remove_dir_all(&self.0);
}
}
async fn classify_tree(setup: impl FnOnce(&Path)) -> Vec<ClassifierVerdict> {
let scratch = Scratch::new();
setup(&scratch.0);
let artifact = Artifact::git_repo("https://git/x", "main");
let mut wp = HashMap::new();
wp.insert(artifact.id.clone(), scratch.0.clone());
let artifacts = vec![artifact];
let input = ClassificationInput {
artifacts: &artifacts,
working_paths: &wp,
description: None,
};
HeuristicClassifier
.classify(&input)
.await
.expect("classify")
}
#[tokio::test]
async fn frontend_package_json_is_webapp() {
let v = classify_tree(|root| {
fs::write(
root.join("package.json"),
r#"{"dependencies":{"react":"18"}}"#,
)
.unwrap();
})
.await;
assert!(v.iter().any(|x| x.target_type == TargetType::WebApp));
}
#[tokio::test]
async fn cargo_toml_is_backend() {
let v = classify_tree(|root| {
fs::write(root.join("Cargo.toml"), "[package]\nname='x'").unwrap();
})
.await;
assert!(v
.iter()
.any(|x| x.target_type == TargetType::BackendService));
}
#[tokio::test]
async fn android_manifest_is_android() {
let v = classify_tree(|root| {
fs::write(root.join("AndroidManifest.xml"), "<manifest/>").unwrap();
})
.await;
assert!(v.iter().any(|x| x.target_type == TargetType::AndroidApp));
}
#[tokio::test]
async fn structured_text_is_plc() {
let v = classify_tree(|root| {
fs::write(root.join("main.st"), "PROGRAM main END_PROGRAM").unwrap();
})
.await;
assert!(v.iter().any(|x| x.target_type == TargetType::PlcSps));
}
#[tokio::test]
async fn ipa_artifact_prior_is_ios() {
let artifacts = vec![Artifact::mobile_package("app.ipa")];
let wp = HashMap::new();
let input = ClassificationInput {
artifacts: &artifacts,
working_paths: &wp,
description: None,
};
let v = HeuristicClassifier
.classify(&input)
.await
.expect("classify");
assert!(v.iter().any(|x| x.target_type == TargetType::IosApp));
}
}
-226
View File
@@ -1,226 +0,0 @@
//! Target classification.
//!
//! Runs the classifier registry over a target's artifacts and their ingested
//! working paths, then merges and ranks the verdicts into a [`Classification`].
//! The registry is the heuristic classifier (artifact kinds + source markers)
//! plus the tramiton firmware detector (behind a [`FirmwareDetector`] port).
mod firmware;
mod language;
pub use firmware::{
FirmwareDetection, FirmwareDetector, FirmwareTarget, MockFirmwareDetector, TramitonNative,
};
pub use language::HeuristicClassifier;
use std::collections::HashMap;
use std::path::PathBuf;
use compliance_core::error::CoreError;
use compliance_core::models::{
ArtifactKind, Classification, DetectedFact, OnboardedTarget, TargetType, TargetTypeCandidate,
};
use compliance_core::traits::{ClassificationInput, ClassifierVerdict, TargetClassifier};
use firmware::detection_to_verdict;
/// Classify a target from its artifacts and their ingested working paths, using
/// the heuristic classifier plus the tramiton firmware detector. Verdicts are
/// merged (max confidence per target type) and ranked into a [`Classification`].
pub async fn classify_target<D: FirmwareDetector>(
target: &OnboardedTarget,
working_paths: &HashMap<String, PathBuf>,
firmware_detector: &D,
) -> Result<Classification, CoreError> {
let input = ClassificationInput {
artifacts: &target.artifacts,
working_paths,
description: target.description.as_deref(),
};
let mut verdicts = Vec::new();
let mut detected_by = Vec::new();
let heuristic = HeuristicClassifier.classify(&input).await?;
if !heuristic.is_empty() {
detected_by.push("heuristic".to_string());
}
verdicts.extend(heuristic);
// Tramiton firmware detection over firmware / code working paths.
let mut tramiton_used = false;
for artifact in &target.artifacts {
if !matches!(
artifact.kind,
ArtifactKind::FirmwareImage | ArtifactKind::GitRepo | ArtifactKind::SourceArchive
) {
continue;
}
let Some(path) = working_paths.get(&artifact.id) else {
continue;
};
if let Some(detection) = firmware_detector.detect(path).await? {
verdicts.push(detection_to_verdict(&detection));
tramiton_used = true;
}
}
if tramiton_used {
detected_by.push("tramiton".to_string());
}
Ok(rank(verdicts, detected_by, target.target_type))
}
/// Merge verdicts by target type (keeping the max confidence and its rationale),
/// dedupe facts, rank by descending confidence, and assemble a [`Classification`].
/// Falls back to the declared type when no verdict is produced.
fn rank(
verdicts: Vec<ClassifierVerdict>,
detected_by: Vec<String>,
fallback: TargetType,
) -> Classification {
let mut best: HashMap<TargetType, (f32, String)> = HashMap::new();
let mut facts: Vec<DetectedFact> = Vec::new();
for verdict in verdicts {
for fact in verdict.facts {
if !facts
.iter()
.any(|e| e.key == fact.key && e.value == fact.value)
{
facts.push(fact);
}
}
let entry = best
.entry(verdict.target_type)
.or_insert((0.0, String::new()));
if verdict.confidence > entry.0 {
*entry = (verdict.confidence, verdict.rationale);
}
}
let mut candidates: Vec<TargetTypeCandidate> = best
.into_iter()
.map(
|(target_type, (confidence, rationale))| TargetTypeCandidate {
target_type,
confidence,
rationale,
},
)
.collect();
// Descending confidence; ties broken by type name for deterministic ordering.
candidates.sort_by(|a, b| {
b.confidence
.partial_cmp(&a.confidence)
.unwrap_or(std::cmp::Ordering::Equal)
.then_with(|| a.target_type.to_string().cmp(&b.target_type.to_string()))
});
let suggested = candidates
.first()
.map(|c| c.target_type)
.unwrap_or(fallback);
Classification {
suggested,
candidates,
facts,
detected_by,
detected_at: chrono::Utc::now(),
confirmed: false,
}
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
use compliance_core::models::Artifact;
use std::fs;
use std::path::Path;
struct Scratch(PathBuf);
impl Scratch {
fn new() -> Self {
let p = std::env::temp_dir().join(format!("cs-classify-mod-{}", uuid::Uuid::new_v4()));
fs::create_dir_all(&p).expect("mkdir");
Self(p)
}
}
impl Drop for Scratch {
fn drop(&mut self) {
let _ = fs::remove_dir_all(&self.0);
}
}
fn no_firmware() -> MockFirmwareDetector {
MockFirmwareDetector { detection: None }
}
#[tokio::test]
async fn backend_repo_classifies_as_backend() {
let scratch = Scratch::new();
fs::write(scratch.0.join("go.mod"), "module x").unwrap();
let artifact = Artifact::git_repo("https://git/x", "main");
let mut wp = HashMap::new();
wp.insert(artifact.id.clone(), scratch.0.clone());
let mut target = OnboardedTarget::new("x".to_string(), TargetType::WebApp);
target.artifacts.push(artifact);
let c = classify_target(&target, &wp, &no_firmware())
.await
.expect("classify");
assert_eq!(c.suggested, TargetType::BackendService);
assert!(c.detected_by.contains(&"heuristic".to_string()));
assert!(!c.confirmed);
}
#[tokio::test]
async fn firmware_detector_verdict_ranks_top() {
let scratch = Scratch::new();
fs::write(scratch.0.join("fw.bin"), b"x").unwrap();
let artifact =
Artifact::firmware_image(scratch.0.join("fw.bin").to_string_lossy().to_string());
let mut wp = HashMap::new();
wp.insert(artifact.id.clone(), scratch.0.clone());
let mut target = OnboardedTarget::new("fw".to_string(), TargetType::FirmwareBareMetal);
target.artifacts.push(artifact);
let detector = MockFirmwareDetector {
detection: Some(FirmwareDetection {
provider: "zephyr".to_string(),
confidence: "high".to_string(),
build_system: "zephyr".to_string(),
framework: Some("zephyr".to_string()),
target: FirmwareTarget {
mcu: Some("nrf52840".to_string()),
..Default::default()
},
gaps: vec![],
}),
};
let c = classify_target(&target, &wp, &detector)
.await
.expect("classify");
// tramiton's high-confidence RTOS verdict beats the weak firmware prior.
assert_eq!(c.suggested, TargetType::FirmwareRtos);
assert!(c.detected_by.contains(&"tramiton".to_string()));
assert!(c.facts.iter().any(|f| f.key == "mcu"));
}
#[tokio::test]
async fn no_signal_falls_back_to_declared_type() {
let scratch = Scratch::new();
let _ = Path::new(&scratch.0);
let target = OnboardedTarget::new("empty".to_string(), TargetType::DesktopApp);
let wp = HashMap::new();
let c = classify_target(&target, &wp, &no_firmware())
.await
.expect("classify");
assert_eq!(c.suggested, TargetType::DesktopApp);
assert!(c.candidates.is_empty());
}
}
-4
View File
@@ -45,8 +45,6 @@ pub fn load_config() -> Result<AgentConfig, AgentError> {
.unwrap_or_else(|| "0 0 * * * *".to_string()),
git_clone_base_path: env_var_opt("GIT_CLONE_BASE_PATH")
.unwrap_or_else(|| "/tmp/compliance-scanner/repos".to_string()),
artifact_store_base_path: env_var_opt("ARTIFACT_STORE_BASE_PATH")
.unwrap_or_else(|| "/data/compliance-scanner/artifacts".to_string()),
ssh_key_path: env_var_opt("SSH_KEY_PATH")
.unwrap_or_else(|| "/data/compliance-scanner/ssh/id_ed25519".to_string()),
keycloak_url: env_var_opt("KEYCLOAK_URL"),
@@ -61,7 +59,5 @@ pub fn load_config() -> Result<AgentConfig, AgentError> {
.unwrap_or(true),
pentest_imap_username: env_var_opt("PENTEST_IMAP_USERNAME"),
pentest_imap_password: env_secret_opt("PENTEST_IMAP_PASSWORD"),
admin_api_token: env_secret_opt("ADMIN_API_TOKEN"),
tenant_registry_url: env_var_opt("TENANT_REGISTRY_URL"),
})
}
-56
View File
@@ -141,25 +141,6 @@ impl DatabasePool {
&self.client
}
/// Cross-tenant admin database used by features that intentionally
/// span tenants (today: MCP bearer tokens — each token row carries
/// a `tenant_id` and the MCP server reads them to route requests).
///
/// The name `<db_prefix>__admin` (double underscore) is reserved —
/// the sanitizer never produces it for a normal tenant DB because
/// the natural format is `<db_prefix>_<sanitized_tenant_id>` (one
/// underscore) and tenant_ids would have to start with `_admin` to
/// collide. New tenant provisioning should reject such ids.
pub fn admin_db(&self) -> mongodb::Database {
self.client.database(&self.admin_db_name())
}
/// Name of the admin database — public so tests / operators can
/// drop it via the raw client.
pub fn admin_db_name(&self) -> String {
format!("{}__admin", self.db_prefix)
}
/// List every Mongo database currently belonging to this pool,
/// identified by the `<db_prefix>_` prefix. The result is the raw
/// database names — opening one for offboarding/cleanup goes
@@ -428,36 +409,6 @@ impl Database {
)
.await?;
// onboarded_targets: multikey on artifact source ref (webhook + dedupe
// lookup). Non-unique — "one git URL per tenant" is enforced in the
// create handler, since a unique multikey index on an array field has
// null-collision caveats.
self.onboarded_targets()
.create_index(
IndexModel::builder()
.keys(doc! { "artifacts.source_ref": 1 })
.build(),
)
.await?;
// onboarded_targets: multikey on artifact kind
self.onboarded_targets()
.create_index(
IndexModel::builder()
.keys(doc! { "artifacts.kind": 1 })
.build(),
)
.await?;
// onboarded_targets: target_type filter
self.onboarded_targets()
.create_index(
IndexModel::builder()
.keys(doc! { "target_type": 1 })
.build(),
)
.await?;
tracing::info!("Database indexes ensured");
Ok(())
}
@@ -514,13 +465,6 @@ impl Database {
self.inner.collection("dast_targets")
}
/// The unified onboarding targets that replace `repositories` and
/// `dast_targets`. Ids are preserved from the legacy collections during
/// migration so downstream `repo_id` / `target_id` references keep resolving.
pub fn onboarded_targets(&self) -> Collection<OnboardedTarget> {
self.inner.collection("onboarded_targets")
}
pub fn dast_scan_runs(&self) -> Collection<DastScanRun> {
self.inner.collection("dast_scan_runs")
}
-154
View File
@@ -1,154 +0,0 @@
//! Content-addressed blob storage and archive extraction for ingest.
//!
//! Blobs are stored at `<base>/blobs/<sha[0:2]>/<sha>` and deduplicated by
//! digest; per-run working directories live under `<base>/work/`.
use std::fs::{self, File};
use std::io::{self, Read};
use std::path::{Path, PathBuf};
use sha2::{Digest, Sha256};
use crate::error::AgentError;
/// Read buffer size for streaming hashes/copies (64 KiB).
const BUF_LEN: usize = 64 * 1024;
/// Stream-hash a file with SHA-256, returning the lowercase-hex digest and the
/// byte length. Streams so large firmware images never load fully into memory.
pub fn hash_file(path: &Path) -> Result<(String, u64), AgentError> {
let mut file = File::open(path)?;
let mut hasher = Sha256::new();
let mut buf = [0u8; BUF_LEN];
let mut total: u64 = 0;
loop {
let n = file.read(&mut buf)?;
if n == 0 {
break;
}
hasher.update(&buf[..n]);
total += n as u64;
}
Ok((hex::encode(hasher.finalize()), total))
}
/// Copy `src` into the content-addressed blob store under `base`, returning the
/// stored path. Idempotent: an already-present blob is not rewritten.
pub fn store_file(base: &Path, src: &Path, sha: &str) -> Result<PathBuf, AgentError> {
if sha.len() < 2 {
return Err(AgentError::Other(format!("invalid content hash '{sha}'")));
}
let dir = base.join("blobs").join(&sha[0..2]);
fs::create_dir_all(&dir)?;
let dest = dir.join(sha);
if !dest.exists() {
fs::copy(src, &dest)?;
}
Ok(dest)
}
/// Extract a zip archive into `dest` (created if needed). `enclosed_name`
/// sanitizes each entry path, so this is safe against zip-slip traversal.
pub fn extract_zip(archive: &Path, dest: &Path) -> Result<(), AgentError> {
let file = File::open(archive)?;
let mut zip =
zip::ZipArchive::new(file).map_err(|e| AgentError::Other(format!("open zip: {e}")))?;
fs::create_dir_all(dest)?;
for i in 0..zip.len() {
let mut entry = zip
.by_index(i)
.map_err(|e| AgentError::Other(format!("read zip entry: {e}")))?;
// `enclosed_name` returns `None` for traversal-unsafe paths — skip them.
let Some(rel) = entry.enclosed_name() else {
continue;
};
let out = dest.join(rel);
if entry.is_dir() {
fs::create_dir_all(&out)?;
} else {
if let Some(parent) = out.parent() {
fs::create_dir_all(parent)?;
}
let mut outfile = File::create(&out)?;
io::copy(&mut entry, &mut outfile)?;
}
}
Ok(())
}
/// The working directory for one artifact of a target: `<base>/work/<target>/<artifact>`.
pub fn work_dir(base: &Path, target_id: &str, artifact_id: &str) -> PathBuf {
base.join("work").join(target_id).join(artifact_id)
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
/// A unique scratch directory, removed on drop.
struct Scratch(PathBuf);
impl Scratch {
fn new() -> Self {
let p = std::env::temp_dir().join(format!("cs-ingest-{}", uuid::Uuid::new_v4()));
fs::create_dir_all(&p).expect("mkdir scratch");
Self(p)
}
fn path(&self) -> &Path {
&self.0
}
}
impl Drop for Scratch {
fn drop(&mut self) {
let _ = fs::remove_dir_all(&self.0);
}
}
#[test]
fn hash_is_stable_and_reports_size() {
let dir = Scratch::new();
let f = dir.path().join("a.bin");
fs::write(&f, b"hello world").expect("write");
let (sha, size) = hash_file(&f).expect("hash");
assert_eq!(size, 11);
// Known SHA-256 of "hello world".
assert_eq!(
sha,
"b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9"
);
}
#[test]
fn store_is_content_addressed_and_idempotent() {
let base = Scratch::new();
let src = base.path().join("src.bin");
fs::write(&src, b"payload").expect("write");
let (sha, _) = hash_file(&src).expect("hash");
let p1 = store_file(base.path(), &src, &sha).expect("store");
let p2 = store_file(base.path(), &src, &sha).expect("store again");
assert_eq!(p1, p2);
assert!(p1.ends_with(&sha));
assert!(p1.starts_with(base.path().join("blobs").join(&sha[0..2])));
assert_eq!(fs::read(&p1).expect("read"), b"payload");
}
#[test]
fn extract_zip_writes_entries() {
let base = Scratch::new();
let archive = base.path().join("a.zip");
{
let file = File::create(&archive).expect("create");
let mut w = zip::ZipWriter::new(file);
let opts: zip::write::SimpleFileOptions = Default::default();
w.start_file("dir/hello.txt", opts).expect("start");
io::Write::write_all(&mut w, b"hi").expect("write");
w.finish().expect("finish");
}
let dest = base.path().join("out");
extract_zip(&archive, &dest).expect("extract");
assert_eq!(
fs::read_to_string(dest.join("dir/hello.txt")).expect("read"),
"hi"
);
}
}
-334
View File
@@ -1,334 +0,0 @@
//! Artifact ingest.
//!
//! Normalizes each [`Artifact`] on an [`OnboardedTarget`] into a local working
//! path plus recorded metadata (content hash, size, discovered facts) that the
//! classifier and scanners consume. Every blob is SHA-256 hashed — that digest
//! is also the reconciliation key against sibling products (a firmware sha256
//! matches tramiton's `Artifact.sha256`).
mod blob;
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use compliance_core::models::{Artifact, ArtifactKind, DetectedFact, OnboardedTarget};
use compliance_core::AgentConfig;
use crate::error::AgentError;
use crate::pipeline::git::{GitOps, RepoCredentials};
/// The paths and identifiers an ingest needs. Decoupled from the full
/// [`AgentConfig`] so ingest is testable without a complete config.
pub struct IngestContext<'a> {
/// Base directory for content-addressed blobs and working dirs.
pub artifact_store_base: &'a Path,
/// Base directory for git clones.
pub git_clone_base: &'a str,
/// Default SSH key path (used when an artifact provides none).
pub ssh_key_path: &'a str,
/// The id of the target these artifacts belong to (namespaces working dirs).
pub target_id: &'a str,
}
impl<'a> IngestContext<'a> {
/// Build an ingest context from the agent config for a given target.
pub fn from_config(config: &'a AgentConfig, target_id: &'a str) -> Self {
Self {
artifact_store_base: Path::new(&config.artifact_store_base_path),
git_clone_base: &config.git_clone_base_path,
ssh_key_path: &config.ssh_key_path,
target_id,
}
}
}
/// The result of ingesting one artifact.
pub struct IngestedArtifact {
/// The artifact this corresponds to ([`Artifact::id`]).
pub artifact_id: String,
/// The artifact kind.
pub kind: ArtifactKind,
/// Local working path (clone dir, extracted dir, or blob file). `None` for
/// artifacts with no on-disk form (live URL, plaintext, container ref).
pub working_path: Option<PathBuf>,
/// SHA-256 of the content (blobs) or git head SHA (git repos).
pub content_hash: Option<String>,
/// Stored blob size in bytes, when applicable.
pub size_bytes: Option<u64>,
/// Facts discovered during ingest.
pub facts: Vec<DetectedFact>,
}
/// All ingested artifacts for a target, keyed by artifact id.
pub struct IngestSet {
/// The ingested artifacts, keyed by [`Artifact::id`].
pub by_artifact: HashMap<String, IngestedArtifact>,
}
impl IngestSet {
/// The working paths of every ingested artifact that has one — the input the
/// classifier expects.
pub fn working_paths(&self) -> HashMap<String, PathBuf> {
self.by_artifact
.iter()
.filter_map(|(id, a)| a.working_path.clone().map(|p| (id.clone(), p)))
.collect()
}
/// The ingest result for a specific artifact.
pub fn get(&self, artifact_id: &str) -> Option<&IngestedArtifact> {
self.by_artifact.get(artifact_id)
}
}
/// Ingest every artifact on a target.
pub fn ingest_all(
target: &OnboardedTarget,
ctx: &IngestContext<'_>,
) -> Result<IngestSet, AgentError> {
let mut by_artifact = HashMap::new();
for artifact in &target.artifacts {
let ingested = ingest_artifact(artifact, ctx)?;
by_artifact.insert(artifact.id.clone(), ingested);
}
Ok(IngestSet { by_artifact })
}
/// Ingest a single artifact, dispatching on its kind.
pub fn ingest_artifact(
artifact: &Artifact,
ctx: &IngestContext<'_>,
) -> Result<IngestedArtifact, AgentError> {
match artifact.kind {
ArtifactKind::GitRepo => ingest_git(artifact, ctx),
ArtifactKind::SourceArchive | ArtifactKind::MobilePackage | ArtifactKind::PlcProject => {
ingest_blob(artifact, ctx, true)
}
ArtifactKind::FirmwareImage => ingest_blob(artifact, ctx, false),
ArtifactKind::ContainerImage => Ok(metadata_only(
artifact,
DetectedFact::new("container_ref", artifact.source_ref.as_str(), "ingest"),
)),
ArtifactKind::LiveUrl => Ok(metadata_only(
artifact,
DetectedFact::new("live_url", artifact.source_ref.as_str(), "ingest"),
)),
ArtifactKind::PlaintextDescription => Ok(metadata_only(
artifact,
DetectedFact::new(
"description_len",
artifact.source_ref.len().to_string(),
"ingest",
),
)),
}
}
/// Clone (or fetch) a git artifact, recording the head SHA as the content hash.
fn ingest_git(
artifact: &Artifact,
ctx: &IngestContext<'_>,
) -> Result<IngestedArtifact, AgentError> {
let creds = credentials_for(artifact, ctx.ssh_key_path);
let git_ops = GitOps::new(ctx.git_clone_base, creds);
let repo_path = git_ops.clone_or_fetch(&artifact.source_ref, &artifact.id)?;
let head = GitOps::get_head_sha(&repo_path).ok();
Ok(IngestedArtifact {
artifact_id: artifact.id.clone(),
kind: artifact.kind,
working_path: Some(repo_path),
content_hash: head,
size_bytes: None,
facts: Vec::new(),
})
}
/// Store a blob artifact content-addressed. When `extract` is set and the blob
/// is a zip container (source archive, APK/AAB/IPA), also unpack it into a
/// working directory; otherwise the working path is the stored blob.
fn ingest_blob(
artifact: &Artifact,
ctx: &IngestContext<'_>,
extract: bool,
) -> Result<IngestedArtifact, AgentError> {
let base = ctx.artifact_store_base;
let src = local_source(artifact)?;
let (sha, size) = blob::hash_file(&src)?;
let stored = blob::store_file(base, &src, &sha)?;
let mut facts = Vec::new();
let working_path = if extract {
let dest = blob::work_dir(base, ctx.target_id, &artifact.id);
match blob::extract_zip(&stored, &dest) {
Ok(()) => dest,
Err(e) => {
// Not a zip (e.g. a tar.gz source archive) — keep the blob and
// note it so later stages can decide what to do.
facts.push(DetectedFact::new(
"archive_unextracted",
e.to_string(),
"ingest",
));
stored.clone()
}
}
} else {
stored.clone()
};
Ok(IngestedArtifact {
artifact_id: artifact.id.clone(),
kind: artifact.kind,
working_path: Some(working_path),
content_hash: Some(sha),
size_bytes: Some(size),
facts,
})
}
/// An artifact with no on-disk form: record a single fact, no hash/path.
fn metadata_only(artifact: &Artifact, fact: DetectedFact) -> IngestedArtifact {
IngestedArtifact {
artifact_id: artifact.id.clone(),
kind: artifact.kind,
working_path: None,
content_hash: None,
size_bytes: None,
facts: vec![fact],
}
}
/// The local file backing a blob artifact: its `stored_path` if already
/// uploaded, else its `source_ref` interpreted as a filesystem path.
fn local_source(artifact: &Artifact) -> Result<PathBuf, AgentError> {
let path = artifact
.stored_path
.as_deref()
.unwrap_or(artifact.source_ref.as_str());
let path = PathBuf::from(path);
if !path.exists() {
return Err(AgentError::Other(format!(
"artifact {} source not found at {}",
artifact.id,
path.display()
)));
}
Ok(path)
}
/// Build git credentials from an artifact's auth plus a default SSH key path.
fn credentials_for(artifact: &Artifact, default_ssh_key_path: &str) -> RepoCredentials {
let auth = artifact.auth.as_ref();
RepoCredentials {
ssh_key_path: auth
.and_then(|a| a.ssh_key_path.clone())
.or_else(|| Some(default_ssh_key_path.to_string())),
auth_token: auth.and_then(|a| a.secret.clone()),
auth_username: auth.and_then(|a| a.username.clone()),
}
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
use compliance_core::models::{ArtifactAuth, TargetType};
/// A unique scratch directory, removed on drop.
struct Scratch(PathBuf);
impl Scratch {
fn new() -> Self {
let p = std::env::temp_dir().join(format!("cs-ingest-mod-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&p).expect("mkdir scratch");
Self(p)
}
}
impl Drop for Scratch {
fn drop(&mut self) {
let _ = std::fs::remove_dir_all(&self.0);
}
}
fn ctx_for<'a>(store: &'a Path, target_id: &'a str) -> IngestContext<'a> {
IngestContext {
artifact_store_base: store,
git_clone_base: "/tmp/cs-ingest-test-repos",
ssh_key_path: "/tmp/cs-ingest-test-ssh",
target_id,
}
}
#[test]
fn firmware_blob_is_hashed_and_stored() {
let scratch = Scratch::new();
let store = scratch.0.join("store");
let fw = scratch.0.join("fw.bin");
std::fs::write(&fw, b"firmware-bytes").expect("write");
let ctx = ctx_for(&store, "t1");
let artifact = Artifact::firmware_image(fw.to_string_lossy().to_string());
let out = ingest_artifact(&artifact, &ctx).expect("ingest");
assert_eq!(out.kind, ArtifactKind::FirmwareImage);
assert_eq!(out.size_bytes, Some(14));
let sha = out.content_hash.expect("hash");
assert_eq!(sha.len(), 64);
// working path is the content-addressed blob
let wp = out.working_path.expect("working path");
assert!(wp.starts_with(store.join("blobs")));
}
#[test]
fn live_url_has_no_blob() {
let scratch = Scratch::new();
let store = scratch.0.join("store");
let ctx = ctx_for(&store, "t1");
let artifact = Artifact::live_url("https://example.com");
let out = ingest_artifact(&artifact, &ctx).expect("ingest");
assert!(out.working_path.is_none());
assert!(out.content_hash.is_none());
assert!(out.facts.iter().any(|f| f.key == "live_url"));
}
#[test]
fn ingest_all_collects_working_paths() {
let scratch = Scratch::new();
let store = scratch.0.join("store");
let fw = scratch.0.join("fw.bin");
std::fs::write(&fw, b"abc").expect("write");
let ctx = ctx_for(&store, "t1");
let mut target = OnboardedTarget::new("t".to_string(), TargetType::FirmwareBareMetal);
target
.artifacts
.push(Artifact::firmware_image(fw.to_string_lossy().to_string()));
target.artifacts.push(Artifact::live_url("https://x"));
let set = ingest_all(&target, &ctx).expect("ingest all");
assert_eq!(set.by_artifact.len(), 2);
// Only the firmware artifact yields a working path.
assert_eq!(set.working_paths().len(), 1);
}
#[test]
fn credentials_prefer_artifact_auth() {
let mut artifact = Artifact::git_repo("https://git/x", "main");
artifact.auth = Some(ArtifactAuth {
method: "token".to_string(),
username: Some("bob".to_string()),
secret: Some("pat".to_string()),
..Default::default()
});
let creds = credentials_for(&artifact, "/default/ssh/key");
assert_eq!(creds.auth_token.as_deref(), Some("pat"));
assert_eq!(creds.auth_username.as_deref(), Some("bob"));
}
#[test]
fn credentials_fall_back_to_default_ssh_key() {
let artifact = Artifact::git_repo("git@host:x.git", "main");
let creds = credentials_for(&artifact, "/default/ssh/key");
assert_eq!(creds.ssh_key_path.as_deref(), Some("/default/ssh/key"));
assert!(creds.auth_token.is_none());
}
}
-2
View File
@@ -2,11 +2,9 @@
pub mod agent;
pub mod api;
pub mod classify;
pub mod config;
pub mod database;
pub mod error;
pub mod ingest;
pub mod llm;
pub mod pentest;
pub mod pipeline;
-3
View File
@@ -328,7 +328,6 @@ mod tests {
scan_schedule: String::new(),
cve_monitor_schedule: String::new(),
git_clone_base_path: String::new(),
artifact_store_base_path: String::new(),
ssh_key_path: String::new(),
keycloak_url: None,
keycloak_realm: None,
@@ -340,8 +339,6 @@ mod tests {
pentest_imap_tls: true,
pentest_imap_username: None,
pentest_imap_password: None,
admin_api_token: None,
tenant_registry_url: None,
}
}
+1 -1
View File
@@ -215,7 +215,7 @@ fn scan_with_patterns(
repo_id.to_string(),
fingerprint,
scanner_name.to_string(),
scan_type,
scan_type.clone(),
pattern.title.clone(),
pattern.description.clone(),
pattern.severity.clone(),
+11 -191
View File
@@ -7,18 +7,11 @@ use crate::agent::ComplianceAgent;
use crate::database::Database;
use crate::error::AgentError;
/// Default tenant the scheduler runs against when neither the tenant
/// registry nor `SCHEDULER_TENANT_IDS` are configured. Matches the
/// dev-injector default so a bare `cargo run` has the scheduler
/// scanning whatever lives in `<prefix>_dev`.
/// Default tenant the scheduler runs against when `SCHEDULER_TENANT_IDS`
/// isn't set. Matches the dev-injector default so a bare `cargo run` has
/// the scheduler scanning whatever lives in `<prefix>_dev`.
const DEFAULT_SCHEDULER_TENANT_ID: &str = "dev";
/// Request timeout when fetching the live tenant list from the
/// registry. Kept short — if the registry is slow we'd rather fall
/// back to env-configured ids and finish the tick than block the
/// scheduler loop.
const REGISTRY_FETCH_TIMEOUT_SECS: u64 = 5;
pub async fn start_scheduler(agent: &ComplianceAgent) -> Result<(), AgentError> {
let sched = JobScheduler::new()
.await
@@ -31,12 +24,7 @@ pub async fn start_scheduler(agent: &ComplianceAgent) -> Result<(), AgentError>
let agent = scan_agent.clone();
Box::pin(async move {
tracing::info!("Scheduled scan triggered");
let tenants = scheduler_tenants(&agent).await;
tracing::debug!(
tenant_count = tenants.len(),
"Scheduled scan: tenants resolved"
);
for tenant_id in tenants {
for tenant_id in scheduler_tenants() {
scan_all_repos(&agent, &tenant_id).await;
}
})
@@ -54,12 +42,7 @@ pub async fn start_scheduler(agent: &ComplianceAgent) -> Result<(), AgentError>
let agent = cve_agent.clone();
Box::pin(async move {
tracing::info!("CVE monitor triggered");
let tenants = scheduler_tenants(&agent).await;
tracing::debug!(
tenant_count = tenants.len(),
"CVE monitor: tenants resolved"
);
for tenant_id in tenants {
for tenant_id in scheduler_tenants() {
monitor_cves(&agent, &tenant_id).await;
}
})
@@ -75,14 +58,9 @@ pub async fn start_scheduler(agent: &ComplianceAgent) -> Result<(), AgentError>
.await
.map_err(|e| AgentError::Scheduler(format!("Failed to start scheduler: {e}")))?;
let tenants = scheduler_tenants(agent).await;
let source = if agent.config.tenant_registry_url.is_some() {
"tenant-registry (env fallback)"
} else {
"env (SCHEDULER_TENANT_IDS)"
};
let tenants = scheduler_tenants();
tracing::info!(
"Scheduler started: scans='{}', CVE monitor='{}', tenant source={source}, tenants={tenants:?}",
"Scheduler started: scans='{}', CVE monitor='{}', tenants={tenants:?}",
agent.config.scan_schedule,
agent.config.cve_monitor_schedule,
);
@@ -93,40 +71,10 @@ pub async fn start_scheduler(agent: &ComplianceAgent) -> Result<(), AgentError>
}
}
/// Tenants the scheduler iterates each tick.
///
/// Resolution order:
/// 1. **Tenant registry** at `agent.config.tenant_registry_url`
/// (`GET /v1/tenants`). Fresh on every tick — picks up newly
/// provisioned tenants without an agent restart.
/// 2. **`SCHEDULER_TENANT_IDS`** env (comma-separated) — fallback when
/// the registry is unreachable, the response is malformed, or no
/// registry URL is configured.
/// 3. **`DEFAULT_SCHEDULER_TENANT_ID`** (`"dev"`) — last-ditch fallback
/// so the scheduler keeps doing something useful in dev.
///
/// We never panic out of this function — the scheduler must keep
/// firing even if the registry is offline.
async fn scheduler_tenants(agent: &ComplianceAgent) -> Vec<String> {
if let Some(url) = agent.config.tenant_registry_url.as_deref() {
match fetch_tenants_from_registry(&agent.http, url).await {
Ok(v) if !v.is_empty() => return v,
Ok(_) => {
tracing::warn!("tenant-registry returned empty list; falling back to env");
}
Err(e) => {
tracing::warn!(
url = %url,
error = %e,
"tenant-registry fetch failed; falling back to env"
);
}
}
}
tenants_from_env()
}
fn tenants_from_env() -> Vec<String> {
/// Tenants the scheduler iterates each tick. From `SCHEDULER_TENANT_IDS`
/// (comma-separated), or `DEFAULT_SCHEDULER_TENANT_ID` if unset. M7.2-D
/// will replace this with a pull from the tenant-registry.
fn scheduler_tenants() -> Vec<String> {
std::env::var("SCHEDULER_TENANT_IDS")
.ok()
.map(|s| {
@@ -140,134 +88,6 @@ fn tenants_from_env() -> Vec<String> {
.unwrap_or_else(|| vec![DEFAULT_SCHEDULER_TENANT_ID.to_string()])
}
/// Shape we accept from the registry. Liberal in what we accept:
/// the registry can return any field shape as long as either `id` or
/// `tenant_id` is present. Other fields are ignored.
#[derive(serde::Deserialize)]
struct RegistryTenant {
#[serde(alias = "tenant_id")]
id: String,
/// Filter out non-running tenants if status is present. Missing
/// status defaults to "active" so older registry deployments keep
/// working.
#[serde(default = "default_status")]
status: String,
}
fn default_status() -> String {
"active".to_string()
}
#[derive(serde::Deserialize)]
struct RegistryListResponse {
data: Vec<RegistryTenant>,
}
async fn fetch_tenants_from_registry(
http: &reqwest::Client,
base_url: &str,
) -> Result<Vec<String>, String> {
let url = format!("{}/v1/tenants", base_url.trim_end_matches('/'));
let resp = http
.get(&url)
.timeout(std::time::Duration::from_secs(REGISTRY_FETCH_TIMEOUT_SECS))
.send()
.await
.map_err(|e| format!("request failed: {e}"))?;
if !resp.status().is_success() {
return Err(format!("registry returned {}", resp.status()));
}
let body: RegistryListResponse = resp
.json()
.await
.map_err(|e| format!("invalid JSON: {e}"))?;
Ok(filter_active(body.data))
}
/// Frozen/Archived tenants don't need scheduled scans; the M7.1
/// status gate would 402/410 anyway. Skip them so we don't waste
/// cycles. Active / trial / demo / anything-else-unknown all run.
fn filter_active(rows: Vec<RegistryTenant>) -> Vec<String> {
rows.into_iter()
.filter(|t| !matches!(t.status.as_str(), "frozen" | "archived"))
.map(|t| t.id)
.collect()
}
#[cfg(test)]
mod tests {
use super::*;
fn tenant(id: &str, status: &str) -> RegistryTenant {
RegistryTenant {
id: id.to_string(),
status: status.to_string(),
}
}
#[test]
fn filter_active_keeps_running_skips_frozen_archived() {
let rows = vec![
tenant("a", "active"),
tenant("b", "trial"),
tenant("c", "demo"),
tenant("d", "frozen"),
tenant("e", "archived"),
tenant("f", "weird-but-not-known-dead"),
];
let out = filter_active(rows);
assert_eq!(out, vec!["a", "b", "c", "f"]);
}
#[test]
fn deserialize_registry_response_accepts_id_or_tenant_id() {
let body = r#"{"data":[
{"id":"a","status":"active"},
{"tenant_id":"b","status":"trial"},
{"id":"c"}
]}"#;
let parsed: RegistryListResponse = serde_json::from_str(body).unwrap();
assert_eq!(parsed.data.len(), 3);
assert_eq!(parsed.data[0].id, "a");
assert_eq!(parsed.data[1].id, "b");
assert_eq!(parsed.data[2].id, "c");
// Default status for the third entry should be "active"
assert_eq!(parsed.data[2].status, "active");
}
/// Combined into a single test: cargo runs tests in parallel and
/// env vars are process-global, so two separate tests touching
/// `SCHEDULER_TENANT_IDS` race each other. Doing both checks in
/// one test keeps them in a deterministic order.
#[test]
fn tenants_from_env_resolution() {
std::env::remove_var("SCHEDULER_TENANT_IDS");
assert_eq!(
tenants_from_env(),
vec![DEFAULT_SCHEDULER_TENANT_ID.to_string()],
"unset → default"
);
std::env::set_var("SCHEDULER_TENANT_IDS", "acme, globex ,,hello");
let out = tenants_from_env();
std::env::remove_var("SCHEDULER_TENANT_IDS");
assert_eq!(
out,
vec!["acme", "globex", "hello"],
"splits + trims + drops empty"
);
std::env::set_var("SCHEDULER_TENANT_IDS", "");
let out = tenants_from_env();
std::env::remove_var("SCHEDULER_TENANT_IDS");
assert_eq!(
out,
vec![DEFAULT_SCHEDULER_TENANT_ID.to_string()],
"empty → default"
);
}
}
/// Resolve the per-tenant database. Logs and returns `None` on failure
/// so the loop in the caller can continue with other tenants.
async fn tenant_db(agent: &ComplianceAgent, tenant_id: &str) -> Option<Database> {
-3
View File
@@ -44,7 +44,6 @@ impl TestServer {
scan_schedule: String::new(),
cve_monitor_schedule: String::new(),
git_clone_base_path: "/tmp/compliance-scanner-tests/repos".into(),
artifact_store_base_path: "/tmp/compliance-scanner-tests/artifacts".into(),
ssh_key_path: "/tmp/compliance-scanner-tests/ssh/id_ed25519".into(),
github_token: None,
github_webhook_secret: None,
@@ -67,8 +66,6 @@ impl TestServer {
pentest_imap_tls: false,
pentest_imap_username: None,
pentest_imap_password: None,
admin_api_token: None,
tenant_registry_url: None,
};
let agent = ComplianceAgent::new(config, db_pool);
+4 -12
View File
@@ -63,24 +63,16 @@ struct Claims {
const PUBLIC_ENDPOINTS: &[&str] = &["/api/v1/health"];
/// Path prefixes that bypass JWT validation. The admin sub-router
/// (`/api/v1/admin/*`) has its own static-bearer middleware and must
/// not be routed through the customer-JWT path — a Keycloak token
/// always carries a single tenant_id and would semantically conflict
/// with cross-tenant admin operations.
const PUBLIC_PREFIXES: &[&str] = &["/api/v1/admin/"];
/// Middleware that validates Bearer JWT tokens against Keycloak's JWKS
/// and attaches a `TenantContext` extension on success.
///
/// Skips validation for the health endpoint and any path under one of
/// the [`PUBLIC_PREFIXES`]. If `JwksState` is not present (Keycloak
/// not configured), requests pass through and downstream code must
/// handle the missing context.
/// Skips validation for the health endpoint.
/// If `JwksState` is not present (Keycloak not configured), requests
/// pass through and downstream code must handle the missing context.
pub async fn require_jwt_auth(mut request: Request, next: Next) -> Response {
let path = request.uri().path();
if PUBLIC_ENDPOINTS.contains(&path) || PUBLIC_PREFIXES.iter().any(|p| path.starts_with(p)) {
if PUBLIC_ENDPOINTS.contains(&path) {
return next.run(request).await;
}
-12
View File
@@ -24,9 +24,6 @@ pub struct AgentConfig {
pub scan_schedule: String,
pub cve_monitor_schedule: String,
pub git_clone_base_path: String,
/// Base directory for content-addressed artifact blobs and per-run working
/// dirs (`<base>/blobs/<sha[0:2]>/<sha>`, `<base>/work/<target>/<artifact>/`).
pub artifact_store_base_path: String,
pub ssh_key_path: String,
pub keycloak_url: Option<String>,
pub keycloak_realm: Option<String>,
@@ -40,15 +37,6 @@ pub struct AgentConfig {
pub pentest_imap_tls: bool,
pub pentest_imap_username: Option<String>,
pub pentest_imap_password: Option<SecretString>,
/// Static bearer for the cross-tenant admin endpoints under
/// `/api/v1/admin/*`. When `None`, those endpoints are not
/// mounted at all (defense-in-depth: ops endpoints never reach
/// any auth path if no operator has explicitly opted in).
pub admin_api_token: Option<SecretString>,
/// Live tenant-registry URL the scheduler consults for the list
/// of tenants to iterate. When `None` or unreachable, scheduler
/// falls back to `SCHEDULER_TENANT_IDS` env (M7.2-C).
pub tenant_registry_url: Option<String>,
}
#[derive(Clone, Debug, Serialize, Deserialize)]
-1
View File
@@ -2,7 +2,6 @@ pub mod config;
pub mod db;
pub mod error;
pub mod models;
pub mod scan_matrix;
#[cfg(feature = "telemetry")]
pub mod telemetry;
pub mod tenant;
-69
View File
@@ -1,69 +0,0 @@
//! Per-tenant API tokens used by `compliance-mcp` to authenticate MCP
//! HTTP requests on behalf of LLM clients (Claude Desktop, Cursor,
//! ChatGPT, etc.) that can't run a Keycloak OIDC flow.
//!
//! Tokens are opaque strings of the form `mcpt_<44 url-safe random
//! chars>`. The raw value is shown to the user exactly once at
//! creation; the database only ever sees the SHA-256 hash. Lookups go
//! through the cross-tenant `<prefix>__admin.mcp_tokens` collection
//! and return the `tenant_id` the MCP server should route to.
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
/// Persisted token metadata. `token_hash` is the SHA-256 hex of the
/// raw token; the raw token itself is never stored.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct McpToken {
#[serde(rename = "_id", skip_serializing_if = "Option::is_none")]
pub id: Option<bson::oid::ObjectId>,
/// SHA-256 hex of the raw token. Unique index in the collection.
pub token_hash: String,
/// First 8 chars of the raw token — purely for UI display so users
/// can identify which token is which without re-issuing.
pub token_prefix: String,
/// Routes to `<db_prefix>_<tenant_id>` on MCP requests.
pub tenant_id: String,
/// User-given label, e.g. "Claude Desktop" or "Sharang's laptop".
pub name: String,
/// Keycloak `sub` of the user who created this token, for audit.
pub created_by: String,
#[serde(with = "super::serde_helpers::bson_datetime")]
pub created_at: DateTime<Utc>,
#[serde(default, with = "super::serde_helpers::opt_bson_datetime")]
pub last_used_at: Option<DateTime<Utc>>,
/// Soft-delete flag. A revoked token doc stays around for audit
/// but never authenticates.
#[serde(default)]
pub revoked: bool,
}
/// Public projection of a token — never includes the hash.
/// Returned by `GET /api/v1/mcp-tokens`.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct McpTokenView {
pub id: String,
pub name: String,
/// `mcpt_xxxx…` so the user can identify which row is which.
pub token_prefix: String,
pub created_by: String,
#[serde(with = "super::serde_helpers::bson_datetime")]
pub created_at: DateTime<Utc>,
#[serde(default, with = "super::serde_helpers::opt_bson_datetime")]
pub last_used_at: Option<DateTime<Utc>>,
pub revoked: bool,
}
impl From<&McpToken> for McpTokenView {
fn from(t: &McpToken) -> Self {
Self {
id: t.id.map(|o| o.to_hex()).unwrap_or_default(),
name: t.name.clone(),
token_prefix: t.token_prefix.clone(),
created_by: t.created_by.clone(),
created_at: t.created_at,
last_used_at: t.last_used_at,
revoked: t.revoked,
}
}
}
-9
View File
@@ -7,9 +7,7 @@ pub mod finding;
pub mod graph;
pub mod issue;
pub mod mcp;
pub mod mcp_token;
pub mod notification;
pub mod onboarding;
pub mod pentest;
pub mod repository;
pub mod sbom;
@@ -30,14 +28,7 @@ pub use graph::{
};
pub use issue::{IssueStatus, TrackerIssue, TrackerType};
pub use mcp::{McpServerConfig, McpServerStatus, McpTransport};
pub use mcp_token::{McpToken, McpTokenView};
pub use notification::{CveNotification, NotificationSeverity, NotificationStatus};
pub use onboarding::{
default_compliance_profile, Artifact, ArtifactAuth, ArtifactKind, Classification,
ComplianceFramework, ComplianceProfile, DetectedFact, ExternalRef, ExternalSystem,
GitArtifactConfig, IssueTrackerConfig, OnboardedTarget, PlcArtifactConfig, PlcFormat,
TargetScanConfig, TargetType, TargetTypeCandidate, WebArtifactConfig,
};
pub use pentest::{
AttackChainNode, AttackNodeStatus, AuthMode, CodeContextHint, Environment, IdentityProvider,
PentestAuthConfig, PentestConfig, PentestEvent, PentestMessage, PentestSession, PentestStats,
-753
View File
@@ -1,753 +0,0 @@
//! The unified onboarding model.
//!
//! An [`OnboardedTarget`] is the single source of truth for anything the scanner
//! can analyze. It records *what kind of software* the target is ([`TargetType`]),
//! the concrete [`Artifact`]s that were provided for it (a git repo, a firmware
//! image, a live URL, a PLC project, ...), the classifier's verdict, and the scan
//! configuration. It replaces the older git-only `TrackedRepository` and the
//! standalone `DastTarget`, both of which fold into this type as artifacts.
use std::collections::HashMap;
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
use super::dast::{DastAuthConfig, DastTargetType};
use super::issue::TrackerType;
use super::pentest::{Environment, PentestConfig, PentestStrategy};
use super::scan::ScanType;
/// The family of software a target belongs to.
///
/// Targets look endlessly varied but fall into a small enumerable set classified
/// by where the analyzable signal lives. This drives the scan-applicability
/// matrix and the onboarding wizard's type selection.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TargetType {
/// Browser-facing web application (front end + server).
WebApp,
/// Headless backend service / API (REST, GraphQL, gRPC).
BackendService,
/// Desktop application (Windows/macOS/Linux GUI or CLI binary).
DesktopApp,
/// Android application (APK / AAB).
AndroidApp,
/// iOS application (IPA).
IosApp,
/// Bare-metal embedded firmware (no operating system).
FirmwareBareMetal,
/// Embedded firmware running on an RTOS (Zephyr, FreeRTOS, ...).
FirmwareRtos,
/// Embedded Linux built with Yocto / OpenEmbedded (BSP + image).
EmbeddedLinuxYocto,
/// Programmable logic controller software (IEC 61131-3, PLCopen / SPS).
PlcSps,
}
impl std::fmt::Display for TargetType {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::WebApp => write!(f, "web_app"),
Self::BackendService => write!(f, "backend_service"),
Self::DesktopApp => write!(f, "desktop_app"),
Self::AndroidApp => write!(f, "android_app"),
Self::IosApp => write!(f, "ios_app"),
Self::FirmwareBareMetal => write!(f, "firmware_bare_metal"),
Self::FirmwareRtos => write!(f, "firmware_rtos"),
Self::EmbeddedLinuxYocto => write!(f, "embedded_linux_yocto"),
Self::PlcSps => write!(f, "plc_sps"),
}
}
}
/// The kind of artifact provided for a target.
///
/// Which scans are possible is a function of the target type *and* which of
/// these are present (SAST needs code, DAST needs a running URL, and so on).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ArtifactKind {
/// A git repository (cloned for static analysis).
GitRepo,
/// A source archive (zip / tarball) with no live git remote.
SourceArchive,
/// A firmware image or binary blob.
FirmwareImage,
/// A mobile package: Android APK/AAB or iOS IPA.
MobilePackage,
/// An OCI/Docker container image reference.
ContainerImage,
/// A reachable running instance (base URL / endpoint) for dynamic testing.
LiveUrl,
/// A PLC project: PLCopen XML or Structured Text source.
PlcProject,
/// Free-form plaintext describing the target (feeds classification only).
PlaintextDescription,
}
impl std::fmt::Display for ArtifactKind {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::GitRepo => write!(f, "git_repo"),
Self::SourceArchive => write!(f, "source_archive"),
Self::FirmwareImage => write!(f, "firmware_image"),
Self::MobilePackage => write!(f, "mobile_package"),
Self::ContainerImage => write!(f, "container_image"),
Self::LiveUrl => write!(f, "live_url"),
Self::PlcProject => write!(f, "plc_project"),
Self::PlaintextDescription => write!(f, "plaintext_description"),
}
}
}
/// Credentials attached to an artifact.
///
/// This folds both `TrackedRepository`'s git auth (`auth_token` / `auth_username`
/// / SSH key) and `DastAuthConfig`'s HTTP auth (form / bearer / cookie) into one
/// shape so a single artifact carries whatever it needs to be fetched or probed.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
pub struct ArtifactAuth {
/// Auth method: `none` | `token` | `basic` | `bearer` | `cookie` | `form` | `ssh`.
#[serde(default)]
pub method: String,
/// Username (git user, basic-auth user, or `x-access-token` for PATs).
pub username: Option<String>,
/// The secret credential: PAT, password, or bearer token. Encrypted at rest.
pub secret: Option<String>,
/// Path to an SSH private key for git-over-SSH.
pub ssh_key_path: Option<String>,
/// Login URL for form-based authentication.
pub login_url: Option<String>,
/// Extra headers to send when authenticating / probing.
pub headers: Option<HashMap<String, String>>,
}
impl From<DastAuthConfig> for ArtifactAuth {
fn from(c: DastAuthConfig) -> Self {
Self {
method: c.method,
username: c.username,
// Prefer a bearer token; otherwise fall back to the password.
secret: c.token.or(c.password),
ssh_key_path: None,
login_url: c.login_url,
headers: c.headers,
}
}
}
/// Git-specific configuration for a [`ArtifactKind::GitRepo`] artifact.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct GitArtifactConfig {
/// Branch to scan.
pub default_branch: String,
/// Commit SHA of the last completed scan (change-detection watermark).
pub last_scanned_commit: Option<String>,
/// Local clone path once the repo has been fetched.
pub local_path: Option<String>,
}
impl GitArtifactConfig {
/// Config for a fresh git artifact on the given branch.
pub fn on_branch(branch: impl Into<String>) -> Self {
Self {
default_branch: branch.into(),
last_scanned_commit: None,
local_path: None,
}
}
}
impl Default for GitArtifactConfig {
fn default() -> Self {
Self::on_branch("main")
}
}
/// Dynamic-analysis configuration for a [`ArtifactKind::LiveUrl`] artifact.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct WebArtifactConfig {
/// Whether the endpoint is a web app, REST API, or GraphQL API.
pub target_kind: DastTargetType,
/// URL paths to exclude from crawling / scanning.
#[serde(default)]
pub excluded_paths: Vec<String>,
/// Maximum crawl depth.
pub max_crawl_depth: u32,
/// Rate limit in requests per second.
pub rate_limit: u32,
/// Whether destructive methods (DELETE / PUT) are permitted.
#[serde(default)]
pub allow_destructive: bool,
}
impl Default for WebArtifactConfig {
fn default() -> Self {
Self {
target_kind: DastTargetType::WebApp,
excluded_paths: Vec::new(),
max_crawl_depth: 3,
rate_limit: 10,
allow_destructive: false,
}
}
}
/// The source format of a PLC project artifact.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum PlcFormat {
/// PLCopen XML project export.
PlcopenXml,
/// IEC 61131-3 Structured Text source.
StructuredText,
}
/// PLC-specific configuration for a [`ArtifactKind::PlcProject`] artifact.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct PlcArtifactConfig {
/// The project source format.
pub format: PlcFormat,
}
/// A single fact discovered about a target by ingest or classification
/// (e.g. `language=rust`, `build_system=cmake`, `mcu=stm32f429`).
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct DetectedFact {
/// The fact name.
pub key: String,
/// The fact value.
pub value: String,
/// What produced the fact (e.g. `tramiton`, `language-fingerprint`).
pub source: String,
}
impl DetectedFact {
/// Build a fact from its parts.
pub fn new(
key: impl Into<String>,
value: impl Into<String>,
source: impl Into<String>,
) -> Self {
Self {
key: key.into(),
value: value.into(),
source: source.into(),
}
}
}
/// One concrete thing provided for a target: code, a binary, a URL, etc.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Artifact {
/// Stable per-artifact id (UUID v4) — scan steps reference this.
pub id: String,
/// What kind of artifact this is.
pub kind: ArtifactKind,
/// The source reference: git URL, blob id, live URL, or image ref.
pub source_ref: String,
/// Optional human-friendly label.
pub display_name: Option<String>,
/// Content-addressed storage path once ingested (blobs only).
pub stored_path: Option<String>,
/// SHA-256 of the ingested content (git artifacts store the head SHA).
pub content_hash: Option<String>,
/// Size of the stored blob in bytes.
pub size_bytes: Option<u64>,
/// Credentials for fetching or probing this artifact.
pub auth: Option<ArtifactAuth>,
/// Git configuration (present for [`ArtifactKind::GitRepo`]).
pub git: Option<GitArtifactConfig>,
/// Dynamic-analysis configuration (present for [`ArtifactKind::LiveUrl`]).
pub web: Option<WebArtifactConfig>,
/// PLC configuration (present for [`ArtifactKind::PlcProject`]).
pub plc: Option<PlcArtifactConfig>,
/// Facts discovered about this artifact by ingest / classification.
#[serde(default)]
pub detected: Vec<DetectedFact>,
/// When this artifact was last ingested.
#[serde(default, with = "super::serde_helpers::opt_bson_datetime")]
pub ingested_at: Option<DateTime<Utc>>,
}
impl Artifact {
/// A bare artifact of the given kind and source reference.
fn bare(kind: ArtifactKind, source_ref: impl Into<String>) -> Self {
Self {
id: uuid::Uuid::new_v4().to_string(),
kind,
source_ref: source_ref.into(),
display_name: None,
stored_path: None,
content_hash: None,
size_bytes: None,
auth: None,
git: None,
web: None,
plc: None,
detected: Vec::new(),
ingested_at: None,
}
}
/// A git-repository artifact tracking the given branch.
pub fn git_repo(url: impl Into<String>, branch: impl Into<String>) -> Self {
let mut a = Self::bare(ArtifactKind::GitRepo, url);
a.git = Some(GitArtifactConfig::on_branch(branch));
a
}
/// A live-URL artifact with default crawl settings.
pub fn live_url(url: impl Into<String>) -> Self {
let mut a = Self::bare(ArtifactKind::LiveUrl, url);
a.web = Some(WebArtifactConfig::default());
a
}
/// A firmware-image artifact referenced by name (blob ingested later).
pub fn firmware_image(source_ref: impl Into<String>) -> Self {
Self::bare(ArtifactKind::FirmwareImage, source_ref)
}
/// A source-archive artifact referenced by name (blob ingested later).
pub fn source_archive(source_ref: impl Into<String>) -> Self {
Self::bare(ArtifactKind::SourceArchive, source_ref)
}
/// A mobile-package artifact (APK/AAB/IPA) referenced by name.
pub fn mobile_package(source_ref: impl Into<String>) -> Self {
Self::bare(ArtifactKind::MobilePackage, source_ref)
}
/// A container-image artifact referenced by OCI ref.
pub fn container_image(source_ref: impl Into<String>) -> Self {
Self::bare(ArtifactKind::ContainerImage, source_ref)
}
/// A PLC-project artifact in the given format.
pub fn plc_project(source_ref: impl Into<String>, format: PlcFormat) -> Self {
let mut a = Self::bare(ArtifactKind::PlcProject, source_ref);
a.plc = Some(PlcArtifactConfig { format });
a
}
/// A plaintext-description artifact (classification input only).
pub fn plaintext(text: impl Into<String>) -> Self {
Self::bare(ArtifactKind::PlaintextDescription, text)
}
}
/// One ranked candidate produced by the classifier.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TargetTypeCandidate {
/// The candidate target type.
pub target_type: TargetType,
/// Confidence in `[0.0, 1.0]`.
pub confidence: f32,
/// Why this candidate was proposed.
pub rationale: String,
}
/// The classifier's verdict for a target: a suggested type plus ranked
/// alternatives and the facts the decision rested on.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Classification {
/// The top-ranked target type.
pub suggested: TargetType,
/// All candidates, sorted by descending confidence.
#[serde(default)]
pub candidates: Vec<TargetTypeCandidate>,
/// Facts gathered during classification.
#[serde(default)]
pub facts: Vec<DetectedFact>,
/// Which classifiers contributed (e.g. `["tramiton", "language-fingerprint"]`).
#[serde(default)]
pub detected_by: Vec<String>,
/// When classification ran.
#[serde(with = "super::serde_helpers::bson_datetime")]
pub detected_at: DateTime<Utc>,
/// Whether a human confirmed the suggestion.
#[serde(default)]
pub confirmed: bool,
}
/// Issue-tracker linkage, migrated from `TrackedRepository`'s `tracker_*` fields.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
pub struct IssueTrackerConfig {
/// The tracker platform.
pub tracker_type: Option<TrackerType>,
/// Tracker owner / organization.
pub owner: Option<String>,
/// Tracker repository / project.
pub repo: Option<String>,
/// Per-target tracker access token.
pub token: Option<String>,
}
/// How a target should be scanned.
///
/// `enabled_scans` / `disabled_scans` override the scan-applicability matrix
/// defaults; the pentest and tracker blocks reuse the existing wizard config.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
pub struct TargetScanConfig {
/// Scans explicitly turned on (empty means "use matrix defaults").
#[serde(default)]
pub enabled_scans: Vec<ScanType>,
/// Scans explicitly turned off.
#[serde(default)]
pub disabled_scans: Vec<ScanType>,
/// Target environment (gates destructive / active testing).
#[serde(default)]
pub environment: Environment,
/// Whether destructive tests are permitted for this target.
#[serde(default)]
pub allow_destructive: bool,
/// Pentest strategy selector.
pub strategy: Option<PentestStrategy>,
/// Full pentest wizard configuration.
pub pentest: Option<PentestConfig>,
/// Issue-tracker linkage.
pub issue_tracker: Option<IssueTrackerConfig>,
}
/// A sibling product in the company suite that may already hold authoritative
/// data for a target. compliance-scanner reconciles with these rather than
/// recomputing what they already know.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ExternalSystem {
/// Reproducible-build & firmware compliance engine (build plan, SBOM, VEX,
/// attestation).
Tramiton,
/// Code assistant (downstream remediation consumer).
Werkpilot,
/// Compliance-controls RAG (atomic controls derived from laws).
BreakpilotCompliance,
}
impl std::fmt::Display for ExternalSystem {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::Tramiton => write!(f, "tramiton"),
Self::Werkpilot => write!(f, "werkpilot"),
Self::BreakpilotCompliance => write!(f, "breakpilot_compliance"),
}
}
}
/// A link from this target to a record in a sibling product, used to reconcile
/// existing evidence instead of recomputing it.
///
/// For tramiton, `project_id` is the shared cross-product key and
/// `subject_sha256` matches a firmware artifact's [`Artifact::content_hash`]
/// (which equals tramiton's `Artifact.sha256`).
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ExternalRef {
/// Which sibling product this reference points at.
pub system: ExternalSystem,
/// The sibling product's project identifier, if known.
pub project_id: Option<String>,
/// Content digest of the subject artifact (firmware sha256), if known.
pub subject_sha256: Option<String>,
/// Reconciliation status: `linked` | `reconciled` | `unavailable`.
#[serde(default)]
pub status: String,
/// Opaque, offline-verifiable entitlement grant (e.g. tramiton's signed
/// `LicenseGrant`), if the tenant provided one.
pub license_grant: Option<String>,
/// When evidence was last reconciled from this system.
#[serde(default, with = "super::serde_helpers::opt_bson_datetime")]
pub last_reconciled_at: Option<DateTime<Utc>>,
}
impl ExternalRef {
/// A freshly linked (not yet reconciled) reference to a sibling system.
pub fn linked(system: ExternalSystem) -> Self {
Self {
system,
project_id: None,
subject_sha256: None,
status: "linked".to_string(),
license_grant: None,
last_reconciled_at: None,
}
}
}
/// A regulatory / standards framework a target must comply with. Drives which
/// controls the mapping engine pulls from the [`crate::traits::ControlsProvider`].
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ComplianceFramework {
/// EU Cyber Resilience Act.
Cra,
/// IEC 62443 (industrial automation & control systems security).
Iec62443,
/// EU General Data Protection Regulation.
Gdpr,
/// SOC 2.
Soc2,
/// ISO/IEC 27001.
Iso27001,
/// EU Radio Equipment Directive (RED) cybersecurity articles.
RedDirective,
}
impl std::fmt::Display for ComplianceFramework {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::Cra => write!(f, "cra"),
Self::Iec62443 => write!(f, "iec_62443"),
Self::Gdpr => write!(f, "gdpr"),
Self::Soc2 => write!(f, "soc2"),
Self::Iso27001 => write!(f, "iso_27001"),
Self::RedDirective => write!(f, "red_directive"),
}
}
}
/// The compliance scope of a target: which frameworks apply and, optionally, the
/// jurisdiction. Captured at onboarding (with per-target-type defaults from
/// [`default_compliance_profile`]).
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
pub struct ComplianceProfile {
/// Applicable frameworks.
#[serde(default)]
pub frameworks: Vec<ComplianceFramework>,
/// Free-form jurisdiction (e.g. `eu`, `us`, `de`).
pub jurisdiction: Option<String>,
}
/// The sensible default compliance scope for a target type. Firmware / PLC /
/// embedded default to CRA + IEC 62443; software defaults to GDPR + SOC 2.
pub fn default_compliance_profile(target_type: TargetType) -> ComplianceProfile {
use ComplianceFramework::{Cra, Gdpr, Iec62443, Soc2};
let frameworks = match target_type {
TargetType::PlcSps
| TargetType::FirmwareBareMetal
| TargetType::FirmwareRtos
| TargetType::EmbeddedLinuxYocto => vec![Cra, Iec62443],
TargetType::WebApp | TargetType::BackendService => vec![Gdpr, Soc2],
TargetType::DesktopApp | TargetType::AndroidApp | TargetType::IosApp => {
vec![Gdpr, Cra]
}
};
ComplianceProfile {
frameworks,
jurisdiction: None,
}
}
/// A target onboarded for scanning: the unified replacement for the legacy
/// `TrackedRepository` (SAST) and `DastTarget` (DAST) records.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct OnboardedTarget {
/// Mongo id. Preserved from the legacy record during migration so every
/// downstream collection keyed by `repo_id` / `target_id` keeps resolving.
#[serde(rename = "_id", skip_serializing_if = "Option::is_none")]
pub id: Option<bson::oid::ObjectId>,
/// Human-friendly name.
#[serde(default)]
pub name: String,
/// The software family this target belongs to.
pub target_type: TargetType,
/// Optional free-form description (also a classification input).
pub description: Option<String>,
/// The artifacts provided for this target.
#[serde(default)]
pub artifacts: Vec<Artifact>,
/// The classifier's verdict, once run.
pub classification: Option<Classification>,
/// How this target should be scanned.
#[serde(default)]
pub scan_config: TargetScanConfig,
/// The compliance scope (applicable frameworks / jurisdiction).
#[serde(default)]
pub compliance_profile: ComplianceProfile,
/// Links to sibling products (tramiton, ...) holding reconcilable evidence.
#[serde(default)]
pub external_refs: Vec<ExternalRef>,
/// Cron schedule for recurring scans, if any.
pub scan_schedule: Option<String>,
/// Whether inbound webhooks are enabled for this target.
#[serde(default)]
pub webhook_enabled: bool,
/// HMAC secret for verifying inbound webhooks.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub webhook_secret: Option<String>,
/// Cached count of findings across this target's scans.
#[serde(default)]
pub findings_count: u32,
/// Creation timestamp.
#[serde(
default = "chrono::Utc::now",
with = "super::serde_helpers::bson_datetime"
)]
pub created_at: DateTime<Utc>,
/// Last-update timestamp.
#[serde(
default = "chrono::Utc::now",
with = "super::serde_helpers::bson_datetime"
)]
pub updated_at: DateTime<Utc>,
}
impl OnboardedTarget {
/// A new target of the given type with a freshly generated webhook secret.
pub fn new(name: String, target_type: TargetType) -> Self {
let now = Utc::now();
let webhook_secret = uuid::Uuid::new_v4().to_string().replace('-', "");
Self {
id: None,
name,
target_type,
description: None,
artifacts: Vec::new(),
classification: None,
scan_config: TargetScanConfig::default(),
compliance_profile: default_compliance_profile(target_type),
external_refs: Vec::new(),
scan_schedule: None,
webhook_enabled: false,
webhook_secret: Some(webhook_secret),
findings_count: 0,
created_at: now,
updated_at: now,
}
}
/// The first artifact of the given kind, if present.
pub fn first_of(&self, kind: ArtifactKind) -> Option<&Artifact> {
self.artifacts.iter().find(|a| a.kind == kind)
}
/// Whether the target has at least one artifact of the given kind.
pub fn has(&self, kind: ArtifactKind) -> bool {
self.artifacts.iter().any(|a| a.kind == kind)
}
/// The primary code artifact (git repo or source archive), if any.
pub fn code_artifact(&self) -> Option<&Artifact> {
self.artifacts
.iter()
.find(|a| matches!(a.kind, ArtifactKind::GitRepo | ArtifactKind::SourceArchive))
}
/// The live-URL artifact, if any.
pub fn live_url(&self) -> Option<&Artifact> {
self.first_of(ArtifactKind::LiveUrl)
}
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
fn sample_target() -> OnboardedTarget {
let mut t = OnboardedTarget::new("acme-web".to_string(), TargetType::WebApp);
t.artifacts.push(Artifact::git_repo(
"https://git.example.com/acme.git",
"main",
));
t.artifacts
.push(Artifact::live_url("https://acme.example.com"));
t
}
#[test]
fn onboarded_target_bson_round_trip() {
let t = sample_target();
let b = bson::to_bson(&t).expect("serialize");
let back: OnboardedTarget = bson::from_bson(b.clone()).expect("deserialize");
let b2 = bson::to_bson(&back).expect("re-serialize");
assert_eq!(b, b2);
}
#[test]
fn enum_display_is_snake_case() {
assert_eq!(
TargetType::FirmwareBareMetal.to_string(),
"firmware_bare_metal"
);
assert_eq!(TargetType::PlcSps.to_string(), "plc_sps");
assert_eq!(ArtifactKind::PlcProject.to_string(), "plc_project");
assert_eq!(ArtifactKind::MobilePackage.to_string(), "mobile_package");
}
#[test]
fn helpers_locate_artifacts() {
let t = sample_target();
assert!(t.has(ArtifactKind::GitRepo));
assert!(t.live_url().is_some());
assert!(t.code_artifact().is_some());
assert!(!t.has(ArtifactKind::FirmwareImage));
assert_eq!(
t.first_of(ArtifactKind::GitRepo).map(|a| a.kind),
Some(ArtifactKind::GitRepo)
);
}
#[test]
fn new_target_generates_webhook_secret() {
let t = OnboardedTarget::new("t".to_string(), TargetType::BackendService);
let secret = t.webhook_secret.expect("secret present");
assert_eq!(secret.len(), 32);
assert!(!secret.contains('-'));
}
#[test]
fn dast_auth_folds_into_artifact_auth() {
let dast = DastAuthConfig {
method: "bearer".to_string(),
login_url: Some("https://x/login".to_string()),
username: Some("user".to_string()),
password: Some("pw".to_string()),
token: Some("tok".to_string()),
headers: None,
};
let auth = ArtifactAuth::from(dast);
assert_eq!(auth.method, "bearer");
// Bearer token wins over password.
assert_eq!(auth.secret.as_deref(), Some("tok"));
assert_eq!(auth.login_url.as_deref(), Some("https://x/login"));
}
#[test]
fn each_artifact_gets_a_unique_id() {
let a = Artifact::firmware_image("fw.bin");
let b = Artifact::firmware_image("fw.bin");
assert_ne!(a.id, b.id);
}
#[test]
fn firmware_default_profile_is_cra_and_62443() {
let p = default_compliance_profile(TargetType::FirmwareBareMetal);
assert!(p.frameworks.contains(&ComplianceFramework::Cra));
assert!(p.frameworks.contains(&ComplianceFramework::Iec62443));
}
#[test]
fn webapp_default_profile_is_gdpr_and_soc2() {
let p = default_compliance_profile(TargetType::WebApp);
assert!(p.frameworks.contains(&ComplianceFramework::Gdpr));
assert!(p.frameworks.contains(&ComplianceFramework::Soc2));
}
#[test]
fn new_target_gets_default_profile_and_no_external_refs() {
let t = OnboardedTarget::new("fw".to_string(), TargetType::FirmwareRtos);
assert!(!t.compliance_profile.frameworks.is_empty());
assert!(t.external_refs.is_empty());
}
#[test]
fn external_ref_linked_defaults() {
let r = ExternalRef::linked(ExternalSystem::Tramiton);
assert_eq!(r.system, ExternalSystem::Tramiton);
assert_eq!(r.status, "linked");
assert!(r.project_id.is_none());
assert!(r.last_reconciled_at.is_none());
}
}
+1 -19
View File
@@ -3,7 +3,7 @@ use serde::{Deserialize, Serialize};
use super::repository::ScanTrigger;
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "lowercase")]
pub enum ScanType {
Sast,
@@ -16,14 +16,6 @@ pub enum ScanType {
SecretDetection,
Lint,
CodeReview,
/// Static analysis of a firmware image (unpack + component CVE).
FirmwareStatic,
/// Control-logic security analysis of PLC / SPS programs.
PlcControlLogic,
/// Static analysis of a mobile package (APK / AAB / IPA).
MobileStatic,
/// Static analysis of a container image.
ContainerScan,
}
impl std::fmt::Display for ScanType {
@@ -39,10 +31,6 @@ impl std::fmt::Display for ScanType {
Self::SecretDetection => write!(f, "secret_detection"),
Self::Lint => write!(f, "lint"),
Self::CodeReview => write!(f, "code_review"),
Self::FirmwareStatic => write!(f, "firmware_static"),
Self::PlcControlLogic => write!(f, "plc_control_logic"),
Self::MobileStatic => write!(f, "mobile_static"),
Self::ContainerScan => write!(f, "container_scan"),
}
}
}
@@ -59,8 +47,6 @@ pub enum ScanRunStatus {
#[serde(rename_all = "snake_case")]
pub enum ScanPhase {
ChangeDetection,
ArtifactIngest,
Classification,
Sast,
SbomGeneration,
CveScanning,
@@ -69,10 +55,6 @@ pub enum ScanPhase {
LintScanning,
CodeReview,
GraphBuilding,
FirmwareStatic,
PlcAnalysis,
MobileStatic,
ContainerScan,
LlmTriage,
IssueCreation,
DastScanning,
-379
View File
@@ -1,379 +0,0 @@
//! The scan-applicability matrix.
//!
//! Which scans are possible for a target is a function of its [`TargetType`] and
//! which [`ArtifactKind`]s are actually present: SAST needs code, DAST needs a
//! running URL, firmware-static analysis needs a firmware image, and so on. This
//! module encodes that as a table — one rule set per target type — and resolves
//! it against a concrete [`OnboardedTarget`] into a list of [`ScanOption`]s the
//! onboarding wizard and the scan pipeline both consume.
use crate::models::{ArtifactKind, OnboardedTarget, ScanType, TargetType};
/// What an artifact a scan needs in order to run.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ArtifactRequirement {
/// Source code — a git repo or a source archive.
Code,
/// A reachable running instance (live URL / endpoint).
RunningUrl,
/// A firmware image / binary blob.
Firmware,
/// A PLC project (PLCopen XML or Structured Text).
Plc,
/// A mobile package (APK / AAB / IPA).
Mobile,
/// A container image.
Container,
/// No specific artifact required.
Any,
}
/// A static rule: this scan applies to a target type, needs this artifact, and
/// defaults on/off. The rationale explains the entry to the user.
#[derive(Debug, Clone, Copy)]
pub struct ScanRule {
/// The scan this rule governs.
pub scan: ScanType,
/// Whether the scan is on by default (only when its artifact is present).
pub default_on: bool,
/// Human-readable explanation of what the scan does here.
pub rationale: &'static str,
/// The artifact the scan consumes.
pub requires: ArtifactRequirement,
}
impl ScanRule {
const fn new(
scan: ScanType,
default_on: bool,
rationale: &'static str,
requires: ArtifactRequirement,
) -> Self {
Self {
scan,
default_on,
rationale,
requires,
}
}
}
/// A resolved scan choice for a specific target: a rule intersected with the
/// artifacts actually present. `blocked_reason` is `Some` when the required
/// artifact is missing.
#[derive(Debug, Clone)]
pub struct ScanOption {
/// The scan.
pub scan: ScanType,
/// Whether to pre-select the scan (false when blocked).
pub default_on: bool,
/// Why the scan is offered.
pub rationale: String,
/// The artifact kind the scan needs, if any specific one.
pub required_artifact: Option<ArtifactKind>,
/// Set when the required artifact is absent, explaining the block.
pub blocked_reason: Option<String>,
}
/// The SAST umbrella: every static-analysis sub-scan that runs over source code.
fn sast_umbrella() -> Vec<ScanRule> {
use ArtifactRequirement::Code;
vec![
ScanRule::new(
ScanType::Sast,
true,
"Static analysis (Semgrep) over source",
Code,
),
ScanRule::new(
ScanType::Sbom,
true,
"Software bill of materials from source",
Code,
),
ScanRule::new(
ScanType::Cve,
true,
"Match dependencies against known CVEs",
Code,
),
ScanRule::new(
ScanType::SecretDetection,
true,
"Scan source for committed secrets",
Code,
),
ScanRule::new(ScanType::Lint, true, "Language linters over source", Code),
ScanRule::new(
ScanType::Gdpr,
true,
"GDPR data-handling pattern checks",
Code,
),
ScanRule::new(
ScanType::OAuth,
true,
"OAuth misconfiguration patterns",
Code,
),
ScanRule::new(
ScanType::Graph,
true,
"Build the code graph for impact analysis",
Code,
),
ScanRule::new(
ScanType::CodeReview,
false,
"LLM code review over changed source",
Code,
),
]
}
/// The rule set for a target type. Scans that are never applicable to a type are
/// simply absent (e.g. DAST is not listed for a PLC target).
pub fn rules_for(target_type: TargetType) -> Vec<ScanRule> {
use ArtifactRequirement::{Firmware, Mobile, Plc, RunningUrl};
match target_type {
TargetType::WebApp | TargetType::BackendService => {
let mut r = sast_umbrella();
r.push(ScanRule::new(
ScanType::Dast,
true,
"Dynamic scan of the running endpoint",
RunningUrl,
));
r
}
TargetType::DesktopApp => sast_umbrella(),
TargetType::AndroidApp | TargetType::IosApp => {
let mut r = sast_umbrella();
r.push(ScanRule::new(
ScanType::MobileStatic,
true,
"Static analysis of the mobile package (manifest, permissions, libs)",
Mobile,
));
r
}
TargetType::FirmwareBareMetal | TargetType::FirmwareRtos => {
let mut r = sast_umbrella();
r.push(ScanRule::new(
ScanType::FirmwareStatic,
true,
"Unpack and statically analyze the firmware image",
Firmware,
));
r.push(ScanRule::new(
ScanType::Sbom,
true,
"SBOM from the firmware image (binwalk / tramiton)",
Firmware,
));
r.push(ScanRule::new(
ScanType::Cve,
true,
"Match firmware components against known CVEs",
Firmware,
));
r
}
TargetType::EmbeddedLinuxYocto => {
let mut r = sast_umbrella();
r.push(ScanRule::new(
ScanType::FirmwareStatic,
true,
"EMBA / binwalk static analysis of the image",
Firmware,
));
r.push(ScanRule::new(
ScanType::Sbom,
true,
"SBOM from image layers / recipes",
Firmware,
));
r.push(ScanRule::new(
ScanType::Cve,
true,
"Match image components against known CVEs",
Firmware,
));
r.push(ScanRule::new(
ScanType::Dast,
false,
"Dynamic scan of exposed network services (if any)",
RunningUrl,
));
r
}
TargetType::PlcSps => vec![ScanRule::new(
ScanType::PlcControlLogic,
true,
"Control-logic security rules over the PLC program",
Plc,
)],
}
}
/// Whether an active penetration test is applicable to this target type.
///
/// Pentest runs as its own session (not a [`ScanType`] scan) and needs a
/// reachable running target, so it is offered only for the network-reachable
/// families.
pub fn supports_pentest(target_type: TargetType) -> bool {
matches!(
target_type,
TargetType::WebApp
| TargetType::BackendService
| TargetType::AndroidApp
| TargetType::IosApp
| TargetType::EmbeddedLinuxYocto
)
}
/// The representative artifact kind a requirement is satisfied by.
fn representative_kind(req: ArtifactRequirement) -> Option<ArtifactKind> {
match req {
ArtifactRequirement::Code => Some(ArtifactKind::GitRepo),
ArtifactRequirement::RunningUrl => Some(ArtifactKind::LiveUrl),
ArtifactRequirement::Firmware => Some(ArtifactKind::FirmwareImage),
ArtifactRequirement::Plc => Some(ArtifactKind::PlcProject),
ArtifactRequirement::Mobile => Some(ArtifactKind::MobilePackage),
ArtifactRequirement::Container => Some(ArtifactKind::ContainerImage),
ArtifactRequirement::Any => None,
}
}
/// Whether the target carries an artifact that satisfies the requirement.
fn requirement_satisfied(req: ArtifactRequirement, target: &OnboardedTarget) -> bool {
match req {
ArtifactRequirement::Code => target.code_artifact().is_some(),
ArtifactRequirement::RunningUrl => target.has(ArtifactKind::LiveUrl),
ArtifactRequirement::Firmware => target.has(ArtifactKind::FirmwareImage),
ArtifactRequirement::Plc => target.has(ArtifactKind::PlcProject),
ArtifactRequirement::Mobile => target.has(ArtifactKind::MobilePackage),
ArtifactRequirement::Container => target.has(ArtifactKind::ContainerImage),
ArtifactRequirement::Any => true,
}
}
/// Resolve the matrix for a concrete target into the scans it can run, marking
/// any whose required artifact is missing as blocked.
pub fn applicable_scans(target: &OnboardedTarget) -> Vec<ScanOption> {
rules_for(target.target_type)
.into_iter()
.map(|rule| {
let satisfied = requirement_satisfied(rule.requires, target);
let required_artifact = representative_kind(rule.requires);
let blocked_reason = if satisfied {
None
} else {
Some(match required_artifact {
Some(kind) => format!("no {kind} artifact provided"),
None => "required artifact missing".to_string(),
})
};
ScanOption {
scan: rule.scan,
default_on: rule.default_on && satisfied,
rationale: rule.rationale.to_string(),
required_artifact,
blocked_reason,
}
})
.collect()
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
use crate::models::{Artifact, PlcFormat};
fn target_with(target_type: TargetType, artifacts: Vec<Artifact>) -> OnboardedTarget {
let mut t = OnboardedTarget::new("t".to_string(), target_type);
t.artifacts = artifacts;
t
}
fn option<'a>(opts: &'a [ScanOption], scan: ScanType) -> Option<&'a ScanOption> {
opts.iter().find(|o| o.scan == scan)
}
#[test]
fn webapp_with_code_and_url_offers_sast_and_dast() {
let t = target_with(
TargetType::WebApp,
vec![
Artifact::git_repo("u", "main"),
Artifact::live_url("http://x"),
],
);
let opts = applicable_scans(&t);
let sast = option(&opts, ScanType::Sast).expect("sast offered");
assert!(sast.default_on && sast.blocked_reason.is_none());
let dast = option(&opts, ScanType::Dast).expect("dast offered");
assert!(dast.default_on && dast.blocked_reason.is_none());
}
#[test]
fn webapp_without_url_blocks_dast() {
let t = target_with(TargetType::WebApp, vec![Artifact::git_repo("u", "main")]);
let opts = applicable_scans(&t);
let dast = option(&opts, ScanType::Dast).expect("dast listed");
assert!(!dast.default_on);
assert!(dast.blocked_reason.is_some());
assert_eq!(dast.required_artifact, Some(ArtifactKind::LiveUrl));
}
#[test]
fn firmware_offers_firmware_static_and_not_dast() {
let t = target_with(
TargetType::FirmwareBareMetal,
vec![Artifact::firmware_image("fw.bin")],
);
let opts = applicable_scans(&t);
let fw = option(&opts, ScanType::FirmwareStatic).expect("firmware static offered");
assert!(fw.default_on && fw.blocked_reason.is_none());
assert!(option(&opts, ScanType::Dast).is_none());
}
#[test]
fn plc_offers_only_control_logic() {
let t = target_with(
TargetType::PlcSps,
vec![Artifact::plc_project("p.xml", PlcFormat::PlcopenXml)],
);
let opts = applicable_scans(&t);
assert_eq!(opts.len(), 1);
assert_eq!(opts[0].scan, ScanType::PlcControlLogic);
assert!(opts[0].default_on);
}
#[test]
fn pentest_support_matches_reachable_families() {
assert!(supports_pentest(TargetType::WebApp));
assert!(supports_pentest(TargetType::BackendService));
assert!(!supports_pentest(TargetType::PlcSps));
assert!(!supports_pentest(TargetType::FirmwareBareMetal));
assert!(!supports_pentest(TargetType::DesktopApp));
}
#[test]
fn every_target_type_has_at_least_one_rule() {
for tt in [
TargetType::WebApp,
TargetType::BackendService,
TargetType::DesktopApp,
TargetType::AndroidApp,
TargetType::IosApp,
TargetType::FirmwareBareMetal,
TargetType::FirmwareRtos,
TargetType::EmbeddedLinuxYocto,
TargetType::PlcSps,
] {
assert!(!rules_for(tt).is_empty(), "{tt} has no rules");
}
}
}
-51
View File
@@ -1,51 +0,0 @@
//! The target-classification port.
//!
//! A [`TargetClassifier`] inspects a target's artifacts (and optionally their
//! ingested working directories) and proposes one or more [`ClassifierVerdict`]s
//! — a target type, a confidence, and the facts the decision rested on. Concrete
//! classifiers live in the agent (language/build-system fingerprinting, a
//! firmware detector backed by tramiton, etc.); a registry merges and ranks
//! their verdicts. This mirrors the [`crate::traits::Scanner`] port so the two
//! read the same way.
use std::collections::HashMap;
use std::path::PathBuf;
use crate::error::CoreError;
use crate::models::{Artifact, DetectedFact, TargetType};
/// Everything a classifier needs to reason about a target.
pub struct ClassificationInput<'a> {
/// The artifacts declared for the target.
pub artifacts: &'a [Artifact],
/// Ingested working paths, keyed by [`Artifact::id`]. Absent for artifacts
/// with no on-disk form (e.g. a live URL).
pub working_paths: &'a HashMap<String, PathBuf>,
/// Free-form description of the target, if provided.
pub description: Option<&'a str>,
}
/// A single classifier's proposal for a target.
pub struct ClassifierVerdict {
/// The proposed target type.
pub target_type: TargetType,
/// Confidence in `[0.0, 1.0]`.
pub confidence: f32,
/// Facts that informed the proposal.
pub facts: Vec<DetectedFact>,
/// Human-readable explanation.
pub rationale: String,
}
/// A source of target-type classification.
#[allow(async_fn_in_trait)]
pub trait TargetClassifier: Send + Sync {
/// Stable identifier for this classifier (recorded in `detected_by`).
fn name(&self) -> &str;
/// Propose zero or more ranked verdicts for the given input.
async fn classify(
&self,
input: &ClassificationInput<'_>,
) -> Result<Vec<ClassifierVerdict>, CoreError>;
}
-45
View File
@@ -1,45 +0,0 @@
//! The compliance-controls provider port.
//!
//! The mapping engine turns findings into compliance status against a corpus of
//! controls. That corpus is pluggable: the built-in OSCAL catalog by default, or
//! a tenant-owned RAG of atomic controls derived from laws
//! (`breakpilot-compliance`) when available. A [`ControlsProvider`] abstracts the
//! source so the mapping engine does not hardcode a catalog.
use crate::error::CoreError;
use crate::models::ComplianceFramework;
/// A control retrieved from a controls corpus.
#[derive(Debug, Clone)]
pub struct Control {
/// Stable control identifier (e.g. an OSCAL control id or a RAG chunk id).
pub id: String,
/// The framework this control belongs to.
pub framework: ComplianceFramework,
/// Short human-readable title.
pub title: String,
/// The control text / requirement.
pub text: String,
/// Free-form source reference (catalog name, law citation, ...).
pub source: Option<String>,
}
/// A query for relevant controls.
pub struct ControlQuery<'a> {
/// Frameworks in scope for the target.
pub frameworks: &'a [ComplianceFramework],
/// Free-text describing what to map (a finding summary, a component, ...).
pub context: &'a str,
/// Maximum number of controls to return.
pub limit: usize,
}
/// A source of compliance controls (built-in OSCAL catalog, breakpilot RAG, ...).
#[allow(async_fn_in_trait)]
pub trait ControlsProvider: Send + Sync {
/// Stable identifier for this provider.
fn name(&self) -> &str;
/// Retrieve the controls most relevant to the query.
async fn controls(&self, query: &ControlQuery<'_>) -> Result<Vec<Control>, CoreError>;
}
-58
View File
@@ -1,58 +0,0 @@
//! The external-evidence provider port.
//!
//! A sibling product (tramiton, for firmware) may already hold authoritative
//! analysis for an artifact. An [`EvidenceProvider`] lets compliance-scanner
//! *reconcile* that evidence — a build plan, an SBOM, a VEX document, a
//! reproducible-build lock, an attestation — instead of recomputing it. The key
//! used to match is the artifact content digest (a firmware sha256, which equals
//! [`crate::models::Artifact::content_hash`]).
//!
//! Concrete providers live in the agent (a tramiton CLI shell-out today, a cloud
//! client later) plus a deterministic mock for tests, so nothing here depends on
//! an external binary.
use std::path::Path;
use crate::error::CoreError;
use crate::models::{Artifact, ExternalSystem};
/// A single reconcilable evidence document fetched from a sibling product.
#[derive(Debug, Clone)]
pub struct EvidenceDocument {
/// What the document is: `build_plan` | `sbom` | `vex` | `lock` | `attestation`.
pub kind: String,
/// The document's format (e.g. `cyclonedx-1.5`, `openvex-0.2.0`, `toml`, `json`).
pub format: String,
/// The raw document payload.
pub content: String,
}
/// The evidence a provider could return for a target's artifact.
#[derive(Debug, Clone, Default)]
pub struct ReconciledEvidence {
/// The sibling's project identifier, if resolved.
pub project_id: Option<String>,
/// The subject content digest the evidence pertains to.
pub subject_sha256: Option<String>,
/// The documents fetched (any of build plan / SBOM / VEX / lock / attestation).
pub documents: Vec<EvidenceDocument>,
}
/// A source of externally-held, reconcilable evidence for an artifact.
#[allow(async_fn_in_trait)]
pub trait EvidenceProvider: Send + Sync {
/// Which sibling product this provider integrates.
fn system(&self) -> ExternalSystem;
/// Whether this provider can handle the given artifact + working path
/// (e.g. tramiton handles firmware images / embedded source trees).
fn handles(&self, artifact: &Artifact, working_path: Option<&Path>) -> bool;
/// Reconcile existing evidence for the artifact, keyed by its content digest.
/// Returns `Ok(None)` when the provider has nothing for this artifact.
async fn reconcile(
&self,
artifact: &Artifact,
working_path: Option<&Path>,
) -> Result<Option<ReconciledEvidence>, CoreError>;
}
-6
View File
@@ -1,16 +1,10 @@
pub mod classifier;
pub mod controls;
pub mod dast_agent;
pub mod evidence;
pub mod graph_builder;
pub mod issue_tracker;
pub mod pentest_tool;
pub mod scanner;
pub use classifier::{ClassificationInput, ClassifierVerdict, TargetClassifier};
pub use controls::{Control, ControlQuery, ControlsProvider};
pub use dast_agent::{DastAgent, DastContext, DiscoveredEndpoint, EndpointParameter};
pub use evidence::{EvidenceDocument, EvidenceProvider, ReconciledEvidence};
pub use graph_builder::{LanguageParser, ParseOutput};
pub use issue_tracker::IssueTracker;
pub use pentest_tool::{PentestTool, PentestToolContext, PentestToolResult};
-2
View File
@@ -44,8 +44,6 @@ pub enum Route {
PentestSessionPage { session_id: String },
#[route("/mcp-servers")]
McpServersPage {},
#[route("/mcp-tokens")]
McpTokensPage {},
}
const FAVICON: Asset = asset!("/assets/favicon.svg");
@@ -1,90 +0,0 @@
//! Server-functions for the MCP-tokens management UI.
//!
//! These wrap the agent's `/api/v1/mcp-tokens` CRUD endpoints. The raw
//! token returned by `create_mcp_token` is only visible at creation
//! time — the agent's storage never holds the plaintext.
use dioxus::prelude::*;
use serde::{Deserialize, Serialize};
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
pub struct McpTokenView {
pub id: String,
pub name: String,
pub token_prefix: String,
pub created_by: String,
pub created_at: serde_json::Value,
#[serde(default)]
pub last_used_at: Option<serde_json::Value>,
#[serde(default)]
pub revoked: bool,
}
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
pub struct McpTokensListResponse {
pub data: Vec<McpTokenView>,
}
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
pub struct CreateMcpTokenResponse {
/// Raw token. Shown ONCE — the user must copy it now.
pub token: String,
pub view: McpTokenView,
}
#[server]
pub async fn fetch_mcp_tokens() -> Result<McpTokensListResponse, ServerFnError> {
let resp = super::agent_client::agent_get("/api/v1/mcp-tokens")
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
let body: McpTokensListResponse = resp
.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
Ok(body)
}
#[server]
pub async fn create_mcp_token(name: String) -> Result<CreateMcpTokenResponse, ServerFnError> {
if name.trim().is_empty() {
return Err(ServerFnError::new("Name is required"));
}
let resp = super::agent_client::agent_request(reqwest::Method::POST, "/api/v1/mcp-tokens")
.await?
.json(&serde_json::json!({ "name": name.trim() }))
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
if !resp.status().is_success() {
let body = resp.text().await.unwrap_or_default();
return Err(ServerFnError::new(format!(
"Failed to create token: {body}"
)));
}
let body: CreateMcpTokenResponse = resp
.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
Ok(body)
}
#[server]
pub async fn revoke_mcp_token(id: String) -> Result<(), ServerFnError> {
let resp = super::agent_client::agent_request(
reqwest::Method::DELETE,
&format!("/api/v1/mcp-tokens/{id}"),
)
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
if !resp.status().is_success() {
let body = resp.text().await.unwrap_or_default();
return Err(ServerFnError::new(format!(
"Failed to revoke token: {body}"
)));
}
Ok(())
}
@@ -8,7 +8,6 @@ pub mod graph;
pub mod help_chat;
pub mod issues;
pub mod mcp;
pub mod mcp_tokens;
pub mod notifications;
pub mod pentest;
#[allow(clippy::too_many_arguments)]
@@ -1,271 +0,0 @@
use dioxus::prelude::*;
use dioxus_free_icons::icons::bs_icons::*;
use dioxus_free_icons::Icon;
use crate::components::page_header::PageHeader;
use crate::components::toast::{ToastType, Toasts};
use crate::infrastructure::mcp_tokens::{
create_mcp_token, fetch_mcp_tokens, revoke_mcp_token, CreateMcpTokenResponse,
};
#[component]
pub fn McpTokensPage() -> Element {
let mut tokens = use_resource(|| async { fetch_mcp_tokens().await.ok() });
let mut toasts = use_context::<Toasts>();
// Create-form state
let mut show_form = use_signal(|| false);
let mut new_name = use_signal(String::new);
let mut submitting = use_signal(|| false);
// After creation, the raw token shows once in a banner
let mut just_created: Signal<Option<CreateMcpTokenResponse>> = use_signal(|| None);
// Revoke confirmation: (id, name)
let mut confirm_revoke: Signal<Option<(String, String)>> = use_signal(|| None);
rsx! {
PageHeader {
title: "MCP Tokens",
description: "Static bearer tokens for the MCP server. Use in your LLM client (Claude Desktop, Cursor, etc.) — one token per tool/device.",
}
// ── Just-created banner ────────────────────────────────────
if let Some(resp) = just_created() {
div { class: "card mb-4", style: "border: 1px solid var(--accent-warning); background: var(--bg-warning-subtle);",
div { class: "card-header", style: "color: var(--accent-warning);",
Icon { icon: BsExclamationTriangle, width: 14, height: 14 }
" Copy this token now — it won't be shown again"
}
div { style: "padding: 1rem;",
p { style: "margin-bottom: 0.5rem; color: var(--text-secondary);",
"Token for "
strong { "{resp.view.name}" }
}
div { class: "copyable", style: "background: var(--bg-secondary); padding: 0.75rem; border-radius: 4px;",
code { style: "font-family: var(--font-mono); word-break: break-all; flex: 1;", "{resp.token}" }
crate::components::copy_button::CopyButton { value: resp.token.clone(), small: false }
}
div { style: "margin-top: 0.75rem;",
button {
class: "btn btn-sm btn-ghost",
onclick: move |_| just_created.set(None),
"Dismiss"
}
}
}
}
}
// ── Create form ────────────────────────────────────────────
div { class: "mb-4",
button {
class: "btn btn-primary",
onclick: move |_| {
show_form.set(!show_form());
new_name.set(String::new());
},
if show_form() { "Cancel" } else {
Icon { icon: BsPlusLg, width: 14, height: 14 }
" Create Token"
}
}
}
if show_form() {
div { class: "card mb-4",
div { class: "card-header", "New MCP Token" }
div { style: "padding: 1rem;",
div { class: "form-group",
label { "Name" }
input {
r#type: "text",
placeholder: "Claude Desktop on my laptop",
value: "{new_name}",
oninput: move |e| new_name.set(e.value()),
}
small { style: "color: var(--text-secondary);", "A label so you can identify this token in the list. Not visible to LLM clients." }
}
div { style: "margin-top: 1rem;",
button {
class: "btn btn-primary",
disabled: submitting() || new_name().trim().is_empty(),
onclick: move |_| {
let name = new_name().trim().to_string();
if name.is_empty() {
return;
}
spawn(async move {
submitting.set(true);
match create_mcp_token(name).await {
Ok(resp) => {
toasts.push(ToastType::Success, "Token created. Copy it now — it won't be shown again.");
just_created.set(Some(resp));
show_form.set(false);
new_name.set(String::new());
tokens.restart();
}
Err(e) => {
toasts.push(ToastType::Error, format!("Failed to create token: {e}"));
}
}
submitting.set(false);
});
},
if submitting() { "Creating..." } else { "Create" }
}
}
}
}
}
// ── Tokens list ────────────────────────────────────────────
match &*tokens.read() {
Some(Some(resp)) => {
if resp.data.is_empty() {
rsx! {
div { class: "card",
p { style: "padding: 1rem; color: var(--text-secondary);", "No MCP tokens yet. Create one to start using the MCP server from an LLM client." }
}
}
} else {
rsx! {
div { class: "mcp-cards-grid",
for token in resp.data.iter() {
{
let id = token.id.clone();
let name = token.name.clone();
let prefix = token.token_prefix.clone();
let created_str = format_timestamp(&token.created_at);
let last_used_str = token
.last_used_at
.as_ref()
.map(format_timestamp)
.unwrap_or_else(|| "never".to_string());
let revoked = token.revoked;
rsx! {
div { class: "mcp-card", style: if revoked { "opacity: 0.55;" } else { "" },
div { class: "mcp-card-header",
div { class: "mcp-card-title",
Icon { icon: BsKey, width: 14, height: 14 }
h3 { "{name}" }
if revoked {
span { class: "mcp-card-status stopped", "revoked" }
}
}
if !revoked {
button {
class: "btn btn-sm btn-ghost btn-ghost-danger",
title: "Revoke token",
onclick: {
let id = id.clone();
let name = name.clone();
move |_| {
confirm_revoke.set(Some((id.clone(), name.clone())));
}
},
Icon { icon: BsTrash, width: 14, height: 14 }
}
}
}
div { class: "mcp-card-details",
div { class: "mcp-detail-row",
Icon { icon: BsKey, width: 13, height: 13 }
span { class: "mcp-detail-label", "Prefix" }
code { class: "mcp-detail-value", "{prefix}…" }
}
div { class: "mcp-detail-row",
Icon { icon: BsCalendar, width: 13, height: 13 }
span { class: "mcp-detail-label", "Created" }
span { class: "mcp-detail-value", "{created_str}" }
}
div { class: "mcp-detail-row",
Icon { icon: BsClockHistory, width: 13, height: 13 }
span { class: "mcp-detail-label", "Last used" }
span { class: "mcp-detail-value", "{last_used_str}" }
}
}
}
}
}
}
}
}
}
}
Some(None) => rsx! {
div { class: "card",
p { style: "padding: 1rem; color: var(--accent-danger);", "Failed to load MCP tokens." }
}
},
None => rsx! {
div { class: "card",
p { style: "padding: 1rem; color: var(--text-secondary);", "Loading..." }
}
},
}
// ── Revoke confirmation modal ──────────────────────────────
if let Some((id, name)) = confirm_revoke() {
div { class: "modal-overlay",
div { class: "modal",
h3 { "Revoke token?" }
p {
"The token "
strong { "{name}" }
" will stop working immediately. This cannot be undone. Any LLM client using it will start getting 401."
}
div { style: "display: flex; gap: 0.5rem; margin-top: 1rem; justify-content: flex-end;",
button {
class: "btn btn-ghost",
onclick: move |_| confirm_revoke.set(None),
"Cancel"
}
button {
class: "btn btn-danger",
onclick: {
let id = id.clone();
move |_| {
let id = id.clone();
spawn(async move {
match revoke_mcp_token(id).await {
Ok(()) => {
toasts.push(ToastType::Success, "Token revoked");
tokens.restart();
}
Err(e) => {
toasts.push(ToastType::Error, format!("Failed to revoke: {e}"));
}
}
confirm_revoke.set(None);
});
}
},
"Revoke"
}
}
}
}
}
}
}
/// Best-effort timestamp formatter. The agent serializes BSON DateTime
/// as `{"$date":{"$numberLong":"..."}}` in extended JSON. We accept
/// that shape, plain ISO strings, or anything else (best-effort).
fn format_timestamp(v: &serde_json::Value) -> String {
if let Some(s) = v.as_str() {
return s.to_string();
}
if let Some(ms) = v
.get("$date")
.and_then(|d| d.get("$numberLong"))
.and_then(|s| s.as_str())
.and_then(|s| s.parse::<i64>().ok())
{
return chrono::DateTime::<chrono::Utc>::from_timestamp_millis(ms)
.map(|d| d.format("%Y-%m-%d %H:%M").to_string())
.unwrap_or_else(|| ms.to_string());
}
"".to_string()
}
-2
View File
@@ -11,7 +11,6 @@ pub mod graph_index;
pub mod impact_analysis;
pub mod issues;
pub mod mcp_servers;
pub mod mcp_tokens;
pub mod overview;
pub mod pentest_dashboard;
pub mod pentest_session;
@@ -31,7 +30,6 @@ pub use graph_index::GraphIndexPage;
pub use impact_analysis::ImpactAnalysisPage;
pub use issues::IssuesPage;
pub use mcp_servers::McpServersPage;
pub use mcp_tokens::McpTokensPage;
pub use overview::OverviewPage;
pub use pentest_dashboard::PentestDashboardPage;
pub use pentest_session::PentestSessionPage;
+1 -4
View File
@@ -4,7 +4,7 @@ version = "0.1.0"
edition = "2021"
[dependencies]
compliance-core = { workspace = true, features = ["mongodb", "axum"] }
compliance-core = { workspace = true, features = ["mongodb"] }
rmcp = { version = "0.16", features = ["server", "macros", "transport-io", "transport-streamable-http-server"] }
tokio = { workspace = true }
serde = { workspace = true }
@@ -19,6 +19,3 @@ bson = { version = "2", features = ["chrono-0_4"] }
schemars = "1.0"
axum = "0.8"
tower-http = { version = "0.6", features = ["cors"] }
sha2 = { workspace = true }
hex = { workspace = true }
dashmap = { workspace = true }
-129
View File
@@ -1,129 +0,0 @@
//! Bearer-token authentication for incoming MCP HTTP requests.
//!
//! LLM clients (Claude Desktop / Cursor / ChatGPT / etc.) can't run
//! Keycloak OIDC, so the MCP server uses opaque static tokens minted
//! per-tenant via the agent's `POST /api/v1/mcp-tokens` endpoint.
//!
//! Flow per request:
//! 1. Extract `Authorization: Bearer <token>`. Missing → 401.
//! 2. SHA-256 hash the token.
//! 3. Look up the hash in `<prefix>__admin.mcp_tokens`. Missing or
//! revoked → 401.
//! 4. Fire-and-forget update of `last_used_at` so the dashboard can
//! show staleness without blocking the handler.
//! 5. Stash the tenant_id in [`TENANT_ID`] (a `tokio::task_local`) so
//! the MCP tool handlers can read it without modifying rmcp's
//! handler signatures.
//!
//! The `task_local` is scoped around the inner service call via
//! [`bearer_auth`], so every handler invoked downstream sees the
//! tenant_id without us having to thread it through the macro-
//! generated tool router.
use axum::body::Body;
use axum::extract::{Request, State};
use axum::http::StatusCode;
use axum::middleware::Next;
use axum::response::{IntoResponse, Response};
use mongodb::bson::doc;
use sha2::{Digest, Sha256};
use crate::database::DatabasePool;
tokio::task_local! {
/// Tenant id resolved from the bearer for this request. Set by
/// [`bearer_auth`] before the inner service runs; read by the
/// MCP tool handlers via [`current_tenant_id`].
pub static TENANT_ID: String;
}
/// Mongo collection name in `<prefix>__admin`.
const COLLECTION: &str = "mcp_tokens";
/// Returns the tenant_id set by the auth middleware. `None` outside a
/// request scope (e.g. unit tests that bypass the middleware).
pub fn current_tenant_id() -> Option<String> {
TENANT_ID.try_with(|s| s.clone()).ok()
}
/// Axum middleware: validate bearer → set [`TENANT_ID`] → call inner.
pub async fn bearer_auth(
State(pool): State<DatabasePool>,
request: Request,
next: Next,
) -> Response {
let Some(token) = extract_bearer(&request) else {
return (StatusCode::UNAUTHORIZED, "Missing bearer token").into_response();
};
if !token.starts_with("mcpt_") {
return (StatusCode::UNAUTHORIZED, "Invalid token format").into_response();
}
let token_hash = sha256_hex(&token);
let col = pool.admin_db().collection::<TokenLookup>(COLLECTION);
let found = match col
.find_one(doc! { "token_hash": &token_hash, "revoked": false })
.await
{
Ok(Some(t)) => t,
Ok(None) => {
return (StatusCode::UNAUTHORIZED, "Invalid or revoked token").into_response();
}
Err(e) => {
tracing::error!("MCP token lookup failed: {e}");
return (StatusCode::INTERNAL_SERVER_ERROR, "Token lookup error").into_response();
}
};
// Fire-and-forget last_used_at update — never block the handler.
let col2 = pool.admin_db().collection::<TokenLookup>(COLLECTION);
let hash_for_update = token_hash.clone();
tokio::spawn(async move {
let _ = col2
.update_one(
doc! { "token_hash": &hash_for_update },
doc! { "$set": { "last_used_at": mongodb::bson::DateTime::now() } },
)
.await;
});
let tenant_id = found.tenant_id;
let inner = next.run(request);
TENANT_ID.scope(tenant_id, inner).await
}
/// Bare-bones projection — we don't need the whole `McpToken` here,
/// just enough to route and confirm validity.
#[derive(serde::Deserialize)]
struct TokenLookup {
tenant_id: String,
}
fn extract_bearer(req: &Request<Body>) -> Option<String> {
req.headers()
.get(axum::http::header::AUTHORIZATION)
.and_then(|v| v.to_str().ok())
.and_then(|s| s.strip_prefix("Bearer "))
.map(|s| s.trim().to_string())
.filter(|s| !s.is_empty())
}
fn sha256_hex(s: &str) -> String {
let mut h = Sha256::new();
h.update(s.as_bytes());
hex::encode(h.finalize())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn sha256_known_value() {
// python -c 'import hashlib; print(hashlib.sha256(b"mcpt_known").hexdigest())'
assert_eq!(
sha256_hex("mcpt_known"),
"27cf6cf678a44244106863c1c031be8e57b84c2b3019d742f755f8e7afa75dfd"
);
}
}
+7 -115
View File
@@ -1,127 +1,19 @@
//! Per-tenant Mongo broker for the MCP server.
//!
//! Mirror of the agent's `compliance_agent::database::DatabasePool` —
//! duplicated here rather than lifted into `compliance-core` to keep
//! this PR focused. If a third consumer ever needs it, lift then.
//!
//! Bearer tokens (validated by the auth middleware) carry a tenant_id
//! and the handler resolves the per-tenant database via
//! [`DatabasePool::for_tenant_id`]. The admin database
//! (`<db_prefix>__admin`) holds the cross-tenant `mcp_tokens`
//! collection that the middleware queries on every request.
use std::sync::Arc;
use dashmap::DashMap;
use mongodb::{bson::doc, Client, Collection};
use sha2::{Digest, Sha256};
use mongodb::{Client, Collection};
use compliance_core::models::*;
/// 63-byte Mongo db-name cap; same invariant as the agent's pool.
const MAX_DB_NAME_LEN: usize = 63;
/// 16-byte SHA-256 truncation, hex-encoded → 32 chars.
const HASH_HEX_LEN: usize = 32;
const MAX_PREFIX_LEN: usize = MAX_DB_NAME_LEN - 1 - HASH_HEX_LEN;
#[derive(Clone, Debug)]
pub struct DatabasePool {
client: Client,
db_prefix: String,
/// Tenants we've handed out a [`Database`] for. The MCP server
/// doesn't ensure indexes (the agent owns that side of the
/// schema), so the marker exists only to satisfy the parallel
/// shape — current code never reads it.
#[allow(dead_code)]
seen: Arc<DashMap<String, ()>>,
}
#[derive(Debug, thiserror::Error)]
pub enum DbError {
#[error("db_prefix '{prefix}' is {len} chars; max is {max} so the hash-fallback DB name fits Mongo's 63-byte cap")]
PrefixTooLong {
prefix: String,
len: usize,
max: usize,
},
#[error(transparent)]
Mongo(#[from] mongodb::error::Error),
}
impl DatabasePool {
pub async fn connect(uri: &str, db_prefix: &str) -> Result<Self, DbError> {
if db_prefix.len() > MAX_PREFIX_LEN {
return Err(DbError::PrefixTooLong {
prefix: db_prefix.to_string(),
len: db_prefix.len(),
max: MAX_PREFIX_LEN,
});
}
let client = Client::with_uri_str(uri).await?;
client
.database("admin")
.run_command(doc! { "ping": 1 })
.await?;
tracing::info!(
"MCP MongoDB cluster reachable; per-tenant pool ready (db prefix '{db_prefix}')"
);
Ok(Self {
client,
db_prefix: db_prefix.to_string(),
seen: Arc::new(DashMap::new()),
})
}
/// Read-only handle to the tenant's database. No indexes are
/// ensured here — the agent owns writes, MCP only reads.
pub fn for_tenant_id(&self, tenant_id: &str) -> Database {
let db_name = self.tenant_db_name(tenant_id);
self.seen.insert(tenant_id.to_string(), ());
Database::new(self.client.database(&db_name))
}
/// Cross-tenant admin DB — holds the `mcp_tokens` collection that
/// the auth middleware queries to map bearer → tenant_id.
pub fn admin_db(&self) -> mongodb::Database {
self.client.database(&format!("{}__admin", self.db_prefix))
}
pub fn tenant_db_name(&self, tenant_id: &str) -> String {
let sanitized = sanitize_tenant_id(tenant_id);
let natural = format!("{}_{}", self.db_prefix, sanitized);
if natural.len() <= MAX_DB_NAME_LEN {
natural
} else {
let mut h = Sha256::new();
h.update(tenant_id.as_bytes());
let digest = h.finalize();
let suffix = hex::encode(&digest[..HASH_HEX_LEN / 2]);
format!("{}_{}", self.db_prefix, suffix)
}
}
}
fn sanitize_tenant_id(tenant_id: &str) -> String {
tenant_id
.chars()
.map(|c| match c {
'/' | '\\' | '.' | '"' | '$' | ' ' | '\0' => '_',
c => c,
})
.collect()
}
/// Typed accessors for the MCP-readable collections in a tenant DB.
/// Matches the agent's `Database` shape but only exposes what the MCP
/// tool handlers actually need.
#[derive(Clone, Debug)]
pub struct Database {
inner: mongodb::Database,
}
impl Database {
pub(crate) fn new(inner: mongodb::Database) -> Self {
Self { inner }
pub async fn connect(uri: &str, db_name: &str) -> Result<Self, mongodb::error::Error> {
let client = Client::with_uri_str(uri).await?;
let db = client.database(db_name);
db.run_command(mongodb::bson::doc! { "ping": 1 }).await?;
tracing::info!("MCP server connected to MongoDB '{db_name}'");
Ok(Self { inner: db })
}
pub fn findings(&self) -> Collection<Finding> {
+10 -35
View File
@@ -1,11 +1,10 @@
mod auth;
mod database;
mod server;
mod tools;
use std::sync::Arc;
use database::DatabasePool;
use database::Database;
use rmcp::transport::{
streamable_http_server::session::local::LocalSessionManager, StreamableHttpServerConfig,
StreamableHttpService,
@@ -25,60 +24,36 @@ async fn main() -> Result<(), Box<dyn std::error::Error>> {
let mongo_uri =
std::env::var("MONGODB_URI").unwrap_or_else(|_| "mongodb://localhost:27017".to_string());
// MONGODB_DATABASE is reused as the per-tenant DB-name prefix —
// same convention as the agent so `<prefix>__admin.mcp_tokens`
// and `<prefix>_<tenant_id>` line up across services.
let db_prefix =
let db_name =
std::env::var("MONGODB_DATABASE").unwrap_or_else(|_| "compliance_scanner".to_string());
let pool = DatabasePool::connect(&mongo_uri, &db_prefix).await?;
let db = Database::connect(&mongo_uri, &db_name).await?;
// HTTP transport: bind a small axum router with bearer-auth in
// front of the rmcp service. `/health` stays public for orca's
// container probe.
// If MCP_PORT is set, run as Streamable HTTP server; otherwise use stdio.
if let Ok(port_str) = std::env::var("MCP_PORT") {
let port: u16 = port_str.parse()?;
tracing::info!("Starting MCP server on HTTP port {port}");
let pool_for_factory = pool.clone();
let db_clone = db.clone();
let service = StreamableHttpService::new(
move || Ok(ComplianceMcpServer::new(pool_for_factory.clone())),
move || Ok(ComplianceMcpServer::new(db_clone.clone())),
Arc::new(LocalSessionManager::default()),
StreamableHttpServerConfig::default(),
);
let router = axum::Router::new()
.route("/health", axum::routing::get(|| async { "ok" }))
.nest_service(
"/mcp",
axum::Router::new().fallback_service(service).layer(
axum::middleware::from_fn_with_state(pool.clone(), auth::bearer_auth),
),
);
.nest_service("/mcp", service);
let listener = tokio::net::TcpListener::bind(("0.0.0.0", port)).await?;
tracing::info!("MCP HTTP server listening on 0.0.0.0:{port}");
axum::serve(listener, router).await?;
} else {
// stdio transport — used when run as a local MCP server next
// to the LLM client. There's no HTTP layer to do bearer auth,
// so we synthesize a tenant_id from STDIO_TENANT_ID for local
// development. NEVER use this in production.
tracing::info!("Starting MCP server on stdio");
let synth_tenant = std::env::var("STDIO_TENANT_ID").unwrap_or_else(|_| "dev".to_string());
tracing::warn!(
tenant_id = %synth_tenant,
"stdio transport — using synthetic tenant id; DO NOT use in production"
);
let server = ComplianceMcpServer::new(pool);
let server = ComplianceMcpServer::new(db);
let transport = rmcp::transport::stdio();
use rmcp::ServiceExt;
auth::TENANT_ID
.scope(synth_tenant, async {
let handle = server.serve(transport).await?;
handle.waiting().await?;
Ok::<_, Box<dyn std::error::Error>>(())
})
.await?;
let handle = server.serve(transport).await?;
handle.waiting().await?;
}
Ok(())
+17 -46
View File
@@ -2,37 +2,20 @@ use rmcp::{
handler::server::wrapper::Parameters, model::*, tool, tool_handler, tool_router, ServerHandler,
};
use crate::auth::current_tenant_id;
use crate::database::{Database, DatabasePool};
use crate::database::Database;
use crate::tools::{dast, findings, pentest, sbom};
pub struct ComplianceMcpServer {
pool: DatabasePool,
db: Database,
#[allow(dead_code)]
tool_router: rmcp::handler::server::router::tool::ToolRouter<Self>,
}
impl ComplianceMcpServer {
/// Resolve the per-tenant `Database` from the bearer-set
/// `task_local`. Every tool handler calls this; missing context
/// surfaces as `internal_error` because it means the auth
/// middleware was misconfigured (handler ran without scope).
fn tenant_db(&self) -> Result<Database, rmcp::ErrorData> {
let tenant_id = current_tenant_id().ok_or_else(|| {
rmcp::ErrorData::internal_error(
"no tenant context — bearer middleware not in chain".to_string(),
None,
)
})?;
Ok(self.pool.for_tenant_id(&tenant_id))
}
}
#[tool_router]
impl ComplianceMcpServer {
pub fn new(pool: DatabasePool) -> Self {
pub fn new(db: Database) -> Self {
Self {
pool,
db,
tool_router: Self::tool_router(),
}
}
@@ -46,8 +29,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<findings::ListFindingsParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
findings::list_findings(&db, params).await
findings::list_findings(&self.db, params).await
}
#[tool(description = "Get a single finding by its ID")]
@@ -55,8 +37,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<findings::GetFindingParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
findings::get_finding(&db, params).await
findings::get_finding(&self.db, params).await
}
#[tool(description = "Get a summary of findings counts grouped by severity and status")]
@@ -64,8 +45,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<findings::FindingsSummaryParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
findings::findings_summary(&db, params).await
findings::findings_summary(&self.db, params).await
}
// ── SBOM ──────────────────────────────────────────────
@@ -77,8 +57,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<sbom::ListSbomPackagesParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
sbom::list_sbom_packages(&db, params).await
sbom::list_sbom_packages(&self.db, params).await
}
#[tool(
@@ -88,8 +67,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<sbom::SbomVulnReportParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
sbom::sbom_vuln_report(&db, params).await
sbom::sbom_vuln_report(&self.db, params).await
}
// ── DAST ──────────────────────────────────────────────
@@ -101,8 +79,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<dast::ListDastFindingsParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
dast::list_dast_findings(&db, params).await
dast::list_dast_findings(&self.db, params).await
}
#[tool(description = "Get a summary of recent DAST scan runs and finding counts")]
@@ -110,8 +87,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<dast::DastScanSummaryParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
dast::dast_scan_summary(&db, params).await
dast::dast_scan_summary(&self.db, params).await
}
// ── Pentest ─────────────────────────────────────────────
@@ -123,8 +99,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<pentest::ListPentestSessionsParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
pentest::list_pentest_sessions(&db, params).await
pentest::list_pentest_sessions(&self.db, params).await
}
#[tool(description = "Get a single AI pentest session by its ID")]
@@ -132,8 +107,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<pentest::GetPentestSessionParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
pentest::get_pentest_session(&db, params).await
pentest::get_pentest_session(&self.db, params).await
}
#[tool(
@@ -143,8 +117,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<pentest::GetAttackChainParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
pentest::get_attack_chain(&db, params).await
pentest::get_attack_chain(&self.db, params).await
}
#[tool(description = "Get chat messages from a pentest session")]
@@ -152,8 +125,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<pentest::GetPentestMessagesParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
pentest::get_pentest_messages(&db, params).await
pentest::get_pentest_messages(&self.db, params).await
}
#[tool(
@@ -163,8 +135,7 @@ impl ComplianceMcpServer {
&self,
Parameters(params): Parameters<pentest::PentestStatsParams>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let db = self.tenant_db()?;
pentest::pentest_stats(&db, params).await
pentest::pentest_stats(&self.db, params).await
}
}
@@ -178,7 +149,7 @@ impl ServerHandler for ComplianceMcpServer {
.build(),
server_info: Implementation::from_build_env(),
instructions: Some(
"Compliance Scanner MCP server. Query security findings, SBOM data, DAST results, and AI pentest sessions for your tenant."
"Compliance Scanner MCP server. Query security findings, SBOM data, DAST results, and AI pentest sessions."
.to_string(),
),
}