Compare commits

..
Author SHA1 Message Date
Sharang ParnerkarandClaude Fable 5 43a1900850 feat(onboarding): use tramiton-core natively for firmware detection
CI / Check (pull_request) Has been cancelled
CI / Detect Changes (pull_request) Has been cancelled
CI / Deploy Agent (pull_request) Has been cancelled
CI / Deploy Dashboard (pull_request) Has been cancelled
CI / Deploy Docs (pull_request) Has been cancelled
CI / Deploy MCP (pull_request) Has been cancelled
Replace the `tramiton detect --json` CLI shell-out with a direct dependency on
tramiton-core (same-company IP), so firmware bare-metal/RTOS classification runs
in-process and the whole tramiton suite is available to onboarding.

- compliance-agent depends on tramiton-core (git, tag v0.4.0).
- classify/firmware.rs: TramitonNative runs tramiton_core::provider::analyze on a
  blocking thread and maps its BuildPlan → a minimal FirmwareDetection. Drops the
  mirrored JSON structs and the CLI wrapper. FirmwareDetector port + a
  deterministic MockFirmwareDetector are kept so unit tests need neither the
  tramiton sources nor a firmware tree.
- CI: enable CARGO_NET_GIT_FETCH_WITH_CLI and add a git-auth step so the runner
  can fetch the private tramiton repo. Requires a repo secret TRAMITON_FETCH_TOKEN
  (Gitea PAT with read access to sharang/tramiton).

Refs #118, #121, #135.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-10 15:47:55 +02:00
Sharang ParnerkarandClaude Fable 5 c6e82bc331 feat(onboarding): artifact ingest + classifier + suite-integration seams
Steps 3-4 of the onboarding plan, plus the sibling-product reconciliation seams.

Ingest (compliance-agent/src/ingest, #120):
- ingest_all / ingest_artifact normalize each artifact to a working path +
  metadata. Every blob is SHA-256 hashed (content-addressed store, dedup) —
  that digest is also the tramiton reconciliation key.
- git via GitOps reuse; zip archives + mobile packages extracted; firmware
  stored as blob; live URL / plaintext / container = metadata only.
- IngestContext decoupled from the full AgentConfig (testable in isolation).

Classify (compliance-agent/src/classify, #121):
- FirmwareDetector port + TramitonCli (shell out `tramiton detect --json`,
  parse a mirrored BuildPlan subset — no dependency on the proprietary crate)
  + a deterministic MockFirmwareDetector so CI never needs the binary.
- HeuristicClassifier: artifact-kind priors + source-marker fingerprinting
  (web/backend/mobile/desktop/PLC).
- classify_target merges + ranks verdicts into a Classification.

Suite-integration seams (compliance-core, #135/#136/#137):
- Model: ExternalRef/ExternalSystem (reconcile with tramiton/werkpilot/breakpilot),
  ComplianceProfile/ComplianceFramework + default_compliance_profile per type.
- Ports: EvidenceProvider (fetch external SBOM/VEX/lock/attestation) and
  ControlsProvider (built-in OSCAL vs breakpilot RAG).
- TargetType now derives Hash; AgentConfig gains artifact_store_base_path.

44 unit tests (23 core + 8 ingest + 13 classify). Passes fmt + clippy -D warnings
across agent, dashboard (server + web), and mcp. Additive; legacy paths untouched.

Refs #118, #120, #121, #135, #136, #137.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-10 15:41:46 +02:00
15 changed files with 13 additions and 1195 deletions
+9 -42
View File
@@ -9,25 +9,13 @@ on:
env:
CARGO_TERM_COLOR: always
RUSTFLAGS: "-D warnings"
# Compile cache: sccache -> Hetzner S3 (breakpilot-sccache), runner-independent
# and persistent across CI runs (own key prefix). Reuses the shared cluster S3
# creds (same bucket as werkpilot). Requires repo secrets HETZNER_S3_ACCESS_KEY
# and HETZNER_S3_SECRET_KEY.
# sccache caches compilation artifacts within a job so that compiling
# both --features server and --features web shares common crate work.
RUSTC_WRAPPER: /usr/local/bin/sccache
SCCACHE_BUCKET: breakpilot-sccache
SCCACHE_ENDPOINT: https://nbg1.your-objectstorage.com
SCCACHE_REGION: auto
SCCACHE_S3_USE_SSL: "true"
SCCACHE_S3_KEY_PREFIX: compliance-scanner
AWS_ACCESS_KEY_ID: ${{ secrets.HETZNER_S3_ACCESS_KEY }}
AWS_SECRET_ACCESS_KEY: ${{ secrets.HETZNER_S3_SECRET_KEY }}
SCCACHE_DIR: /tmp/sccache
# compliance-agent depends on tramiton-core via git; use the system git so the
# credential rewrite below (see "Configure git auth ...") is honored on fetch.
CARGO_NET_GIT_FETCH_WITH_CLI: "true"
# Throttle cargo so a ~670-crate concurrent download burst doesn't 429 the
# Kellnr mirror: fewer concurrent connections (HTTP/1.1) + more retries.
CARGO_NET_RETRY: "10"
CARGO_HTTP_MULTIPLEXING: "false"
# Cancel in-progress runs for the same branch/PR
concurrency:
@@ -51,36 +39,20 @@ jobs:
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
git fetch --depth=1 origin "${GITHUB_SHA}"
git checkout FETCH_HEAD
# Resolve crates.io deps through the self-hosted Kellnr mirror (cached,
# crates.io-independent). Git deps (tramiton-core) are unaffected — source
# replacement only applies to crates.io-sourced crates.
- name: Use Kellnr crates.io mirror
run: |
: "${CARGO_HOME:=/usr/local/cargo}"
mkdir -p "$CARGO_HOME"
{
echo '[source.crates-io]'
echo 'replace-with = "kellnr"'
echo '[registries.kellnr]'
echo 'index = "sparse+https://crates.meghsakha.com/api/v1/cratesio/"'
} >> "$CARGO_HOME/config.toml"
env:
RUSTC_WRAPPER: ""
- name: Install tools
run: |
rustup component add rustfmt clippy
curl -fsSL https://github.com/mozilla/sccache/releases/download/v0.10.0/sccache-v0.10.0-x86_64-unknown-linux-musl.tar.gz \
| tar xz --strip-components=1 -C /usr/local/bin/ sccache-v0.10.0-x86_64-unknown-linux-musl/sccache
curl -fsSL https://github.com/mozilla/sccache/releases/download/v0.9.1/sccache-v0.9.1-x86_64-unknown-linux-musl.tar.gz \
| tar xz --strip-components=1 -C /usr/local/bin/ sccache-v0.9.1-x86_64-unknown-linux-musl/sccache
chmod +x /usr/local/bin/sccache
cargo install cargo-audit --locked
env:
RUSTC_WRAPPER: ""
# compliance-agent has a git dependency on tramiton-core (a private repo on
# this Gitea instance). Rewrite its SSH URL to HTTPS + a PAT so the runner
# can fetch it. Requires the repo secret TRAMITON_FETCH_TOKEN (a Gitea PAT
# with read:repository, owned by a user with access to sharang/tramiton).
# (Honored on fetch because CARGO_NET_GIT_FETCH_WITH_CLI=true uses system git.)
# this Gitea instance). Rewrite its SSH URL to HTTPS + a read token so the
# runner can fetch it. Requires a repo secret TRAMITON_FETCH_TOKEN — a
# Gitea PAT for a user with read access to sharang/tramiton.
- name: Configure git auth for private tramiton dependency
run: |
git config --global \
@@ -191,18 +163,13 @@ jobs:
image: docker:27-cli
steps:
- name: Build, push and trigger orca redeploy
env:
# PAT for fetching the private tramiton-core git dependency during the
# image build (injected as a BuildKit secret, never baked into a layer).
TRAMITON_FETCH_TOKEN: ${{ secrets.TRAMITON_FETCH_TOKEN }}
run: |
apk add --no-cache git curl openssl
git init && git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
git fetch --depth=1 origin "${GITHUB_SHA}" && git checkout FETCH_HEAD
IMAGE=registry.meghsakha.com/compliance-agent
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login registry.meghsakha.com -u "${{ secrets.REGISTRY_USERNAME }}" --password-stdin
DOCKER_BUILDKIT=1 docker build --secret id=tramiton_token,env=TRAMITON_FETCH_TOKEN \
-f Dockerfile.agent -t "$IMAGE:latest" -t "$IMAGE:${GITHUB_SHA}" .
docker build -f Dockerfile.agent -t "$IMAGE:latest" -t "$IMAGE:${GITHUB_SHA}" .
docker push "$IMAGE:latest" && docker push "$IMAGE:${GITHUB_SHA}"
PAYLOAD=$(printf '{"ref":"refs/heads/main","repository":{"full_name":"sharang/compliance-scanner-agent"},"head_commit":{"id":"%s","message":"deploy agent"}}' "${GITHUB_SHA}")
SIG=$(printf '%s' "$PAYLOAD" | openssl dgst -sha256 -hmac "${{ secrets.ORCA_WEBHOOK_SECRET }}" | awk '{print $2}')
+1 -10
View File
@@ -2,16 +2,7 @@ FROM rust:1.94-bookworm AS builder
WORKDIR /app
COPY . .
# compliance-agent depends on the private tramiton-core git repo. Authenticate
# the fetch with a PAT passed as a BuildKit secret (never baked into a layer).
# Build with: DOCKER_BUILDKIT=1 docker build --secret id=tramiton_token,env=TRAMITON_FETCH_TOKEN ...
RUN --mount=type=secret,id=tramiton_token \
if [ -s /run/secrets/tramiton_token ]; then \
git config --global \
url."https://sharang:$(cat /run/secrets/tramiton_token)@gitea.meghsakha.com/".insteadOf \
"ssh://git@gitea.meghsakha.com:22222/"; \
fi && \
CARGO_NET_GIT_FETCH_WITH_CLI=true cargo build --release -p compliance-agent
RUN cargo build --release -p compliance-agent
FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y ca-certificates libssl3 git curl python3 python3-pip npm golang-go php-cli && rm -rf /var/lib/apt/lists/*
-1
View File
@@ -9,7 +9,6 @@ pub mod help_chat;
pub mod issues;
pub mod mcp_tokens;
pub mod notifications;
pub mod onboarding;
pub mod pentest_handlers;
pub use pentest_handlers as pentest;
pub mod repos;
@@ -1,350 +0,0 @@
//! Onboarding API — CRUD for unified targets, artifact add, classification, and
//! the scan-applicability matrix. The wizard (and future integrations) drive
//! onboarding through these endpoints.
use std::collections::HashMap;
use std::sync::Arc;
use axum::extract::{Extension, Path, Query};
use axum::http::StatusCode;
use axum::Json;
use mongodb::bson::{doc, oid::ObjectId, to_bson};
use serde::{Deserialize, Serialize};
use compliance_core::models::{
Artifact, ArtifactKind, ComplianceProfile, OnboardedTarget, PlcFormat, TargetScanConfig,
TargetType,
};
use compliance_core::scan_matrix::{applicable_scans, supports_pentest};
use compliance_core::tenant_ctx::TenantCtx;
use crate::agent::ComplianceAgent;
use crate::classify::{classify_target, MockFirmwareDetector};
use super::dto::tenant_db;
use super::{collect_cursor_async, ApiResponse, PaginationParams};
type AgentExt = Extension<Arc<ComplianceAgent>>;
/// A client-supplied artifact spec. The server builds the [`Artifact`] (and its
/// id) from it, so clients never set internal fields.
#[derive(Deserialize)]
pub struct ArtifactInput {
pub kind: ArtifactKind,
pub source_ref: String,
#[serde(default)]
pub branch: Option<String>,
#[serde(default)]
pub plc_format: Option<PlcFormat>,
}
impl ArtifactInput {
fn build(&self) -> Artifact {
let s = self.source_ref.clone();
match self.kind {
ArtifactKind::GitRepo => {
Artifact::git_repo(s, self.branch.clone().unwrap_or_else(|| "main".to_string()))
}
ArtifactKind::LiveUrl => Artifact::live_url(s),
ArtifactKind::FirmwareImage => Artifact::firmware_image(s),
ArtifactKind::SourceArchive => Artifact::source_archive(s),
ArtifactKind::MobilePackage => Artifact::mobile_package(s),
ArtifactKind::ContainerImage => Artifact::container_image(s),
ArtifactKind::PlcProject => {
Artifact::plc_project(s, self.plc_format.unwrap_or(PlcFormat::PlcopenXml))
}
ArtifactKind::PlaintextDescription => Artifact::plaintext(s),
}
}
}
#[derive(Deserialize)]
pub struct CreateTargetRequest {
pub name: String,
pub target_type: TargetType,
#[serde(default)]
pub description: Option<String>,
#[serde(default)]
pub artifacts: Vec<ArtifactInput>,
}
#[derive(Deserialize)]
pub struct UpdateTargetRequest {
pub name: Option<String>,
pub target_type: Option<TargetType>,
pub scan_config: Option<TargetScanConfig>,
pub compliance_profile: Option<ComplianceProfile>,
pub scan_schedule: Option<String>,
}
/// One applicable-scan option, serialized for the wizard.
#[derive(Serialize)]
pub struct ScanOptionDto {
pub scan: String,
pub default_on: bool,
pub rationale: String,
pub required_artifact: Option<String>,
pub blocked_reason: Option<String>,
}
#[derive(Serialize)]
pub struct ApplicableScansResponse {
pub scans: Vec<ScanOptionDto>,
pub pentest_supported: bool,
}
fn parse_oid(id: &str) -> Result<ObjectId, StatusCode> {
ObjectId::parse_str(id).map_err(|_| StatusCode::BAD_REQUEST)
}
/// GET /api/v1/targets — list onboarded targets (paginated).
#[tracing::instrument(skip_all)]
pub async fn list_targets(
Extension(agent): AgentExt,
tenant: TenantCtx,
Query(params): Query<PaginationParams>,
) -> Result<Json<ApiResponse<Vec<OnboardedTarget>>>, StatusCode> {
let db = tenant_db(&agent, &tenant).await?;
let skip = (params.page.saturating_sub(1)) * params.limit as u64;
let total = db
.onboarded_targets()
.count_documents(doc! {})
.await
.unwrap_or(0);
let targets = match db
.onboarded_targets()
.find(doc! {})
.skip(skip)
.limit(params.limit)
.await
{
Ok(cursor) => collect_cursor_async(cursor).await,
Err(e) => {
tracing::warn!("Failed to fetch onboarded targets: {e}");
Vec::new()
}
};
Ok(Json(ApiResponse {
data: targets,
total: Some(total),
page: Some(params.page),
}))
}
/// POST /api/v1/targets — create an onboarded target.
#[tracing::instrument(skip_all)]
pub async fn create_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Json(req): Json<CreateTargetRequest>,
) -> Result<Json<ApiResponse<OnboardedTarget>>, StatusCode> {
let mut target = OnboardedTarget::new(req.name, req.target_type);
target.description = req.description;
target.artifacts = req.artifacts.iter().map(ArtifactInput::build).collect();
let db = tenant_db(&agent, &tenant).await?;
let res = db
.onboarded_targets()
.insert_one(&target)
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
target.id = res.inserted_id.as_object_id();
Ok(Json(ApiResponse {
data: target,
total: None,
page: None,
}))
}
/// GET /api/v1/targets/{id} — fetch one target.
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn get_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<ApiResponse<OnboardedTarget>>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
let target = db
.onboarded_targets()
.find_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?
.ok_or(StatusCode::NOT_FOUND)?;
Ok(Json(ApiResponse {
data: target,
total: None,
page: None,
}))
}
/// PATCH /api/v1/targets/{id} — update mutable fields.
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn update_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
Json(req): Json<UpdateTargetRequest>,
) -> Result<Json<ApiResponse<OnboardedTarget>>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
let mut set = doc! { "updated_at": mongodb::bson::DateTime::now() };
if let Some(name) = req.name {
set.insert("name", name);
}
if let Some(tt) = req.target_type {
set.insert(
"target_type",
to_bson(&tt).map_err(|_| StatusCode::BAD_REQUEST)?,
);
}
if let Some(sc) = req.scan_config {
set.insert(
"scan_config",
to_bson(&sc).map_err(|_| StatusCode::BAD_REQUEST)?,
);
}
if let Some(cp) = req.compliance_profile {
set.insert(
"compliance_profile",
to_bson(&cp).map_err(|_| StatusCode::BAD_REQUEST)?,
);
}
if let Some(ss) = req.scan_schedule {
set.insert("scan_schedule", ss);
}
db.onboarded_targets()
.update_one(doc! { "_id": oid }, doc! { "$set": set })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
get_target(Extension(agent), tenant, Path(id)).await
}
/// DELETE /api/v1/targets/{id} — remove the target and its findings/scans.
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn delete_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<serde_json::Value>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
db.onboarded_targets()
.delete_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
// Cascade the collections keyed by repo_id == target id (best-effort).
let by_repo = doc! { "repo_id": &id };
let _ = db.findings().delete_many(by_repo.clone()).await;
let _ = db.scan_runs().delete_many(by_repo.clone()).await;
let _ = db.sbom_entries().delete_many(by_repo.clone()).await;
let _ = db.cve_alerts().delete_many(by_repo).await;
Ok(Json(serde_json::json!({ "status": "deleted" })))
}
/// POST /api/v1/targets/{id}/artifacts — attach an artifact (by reference).
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn add_artifact(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
Json(input): Json<ArtifactInput>,
) -> Result<Json<ApiResponse<OnboardedTarget>>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
let artifact = to_bson(&input.build()).map_err(|_| StatusCode::BAD_REQUEST)?;
db.onboarded_targets()
.update_one(
doc! { "_id": oid },
doc! { "$push": { "artifacts": artifact }, "$set": { "updated_at": mongodb::bson::DateTime::now() } },
)
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
get_target(Extension(agent), tenant, Path(id)).await
}
/// GET /api/v1/targets/{id}/applicable-scans — the scan-applicability matrix.
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn applicable_scans_for_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<ApiResponse<ApplicableScansResponse>>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
let target = db
.onboarded_targets()
.find_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?
.ok_or(StatusCode::NOT_FOUND)?;
let scans = applicable_scans(&target)
.into_iter()
.map(|o| ScanOptionDto {
scan: o.scan.to_string(),
default_on: o.default_on,
rationale: o.rationale,
required_artifact: o.required_artifact.map(|k| k.to_string()),
blocked_reason: o.blocked_reason,
})
.collect();
Ok(Json(ApiResponse {
data: ApplicableScansResponse {
scans,
pentest_supported: supports_pentest(target.target_type),
},
total: None,
page: None,
}))
}
/// POST /api/v1/targets/{id}/detect — classify the target from its artifacts.
///
/// This is the lightweight pass: it classifies from artifact kinds without
/// ingesting (cloning) sources, so it returns immediately. Deep detection (after
/// ingest, with tramiton firmware analysis) is a follow-up background step.
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn detect_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<ApiResponse<OnboardedTarget>>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
let mut target = db
.onboarded_targets()
.find_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?
.ok_or(StatusCode::NOT_FOUND)?;
// No ingested working paths here → kind-based classification only; the mock
// firmware detector is never invoked (no firmware working path present).
let empty = HashMap::new();
let detector = MockFirmwareDetector { detection: None };
let classification = classify_target(&target, &empty, &detector)
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
let classification_bson =
to_bson(&classification).map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
db.onboarded_targets()
.update_one(
doc! { "_id": oid },
doc! { "$set": { "classification": classification_bson, "updated_at": mongodb::bson::DateTime::now() } },
)
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
target.classification = Some(classification);
Ok(Json(ApiResponse {
data: target,
total: None,
page: None,
}))
}
-23
View File
@@ -25,29 +25,6 @@ pub fn build_router() -> Router {
"/api/v1/repositories/{id}/webhook-config",
get(handlers::get_webhook_config),
)
// Unified onboarding targets (#131).
.route(
"/api/v1/targets",
get(handlers::onboarding::list_targets).post(handlers::onboarding::create_target),
)
.route(
"/api/v1/targets/{id}",
get(handlers::onboarding::get_target)
.patch(handlers::onboarding::update_target)
.delete(handlers::onboarding::delete_target),
)
.route(
"/api/v1/targets/{id}/artifacts",
post(handlers::onboarding::add_artifact),
)
.route(
"/api/v1/targets/{id}/applicable-scans",
get(handlers::onboarding::applicable_scans_for_target),
)
.route(
"/api/v1/targets/{id}/detect",
post(handlers::onboarding::detect_target),
)
.route("/api/v1/findings", get(handlers::list_findings))
.route("/api/v1/findings/{id}", get(handlers::get_finding))
.route(
-24
View File
@@ -179,23 +179,6 @@ impl DatabasePool {
.collect())
}
/// Tenant ids for every provisioned tenant database, derived by stripping
/// the `<prefix>_` from the database names. Skips the admin database
/// (`<prefix>__admin`). Hash-fallback names (very long tenant_ids) are lost
/// at the cluster level and cannot be recovered here — in practice tenant
/// ids are UUIDs and never hit that path. Used by the migration CLI's
/// `--all` mode.
pub async fn list_tenant_ids(&self) -> Result<Vec<String>, AgentError> {
let prefix = format!("{}_", self.db_prefix);
Ok(self
.list_tenant_db_names()
.await?
.into_iter()
.filter_map(|n| n.strip_prefix(&prefix).map(str::to_string))
.filter(|id| !id.starts_with('_'))
.collect())
}
/// Drop the database for a specific tenant. Used by GDPR delete
/// and tenant offboarding. Idempotent — dropping a non-existent
/// database is a no-op at the driver level.
@@ -538,13 +521,6 @@ impl Database {
self.inner.collection("onboarded_targets")
}
/// A typed handle to an arbitrary collection by name. For bookkeeping
/// collections without a dedicated model (e.g. `schema_migrations`,
/// `onboarding_migration_log`).
pub fn collection_named<T: Send + Sync>(&self, name: &str) -> Collection<T> {
self.inner.collection(name)
}
pub fn dast_scan_runs(&self) -> Collection<DastScanRun> {
self.inner.collection("dast_scan_runs")
}
-1
View File
@@ -8,7 +8,6 @@ pub mod database;
pub mod error;
pub mod ingest;
pub mod llm;
pub mod migrate;
pub mod pentest;
pub mod pipeline;
pub mod rag;
+1 -54
View File
@@ -1,50 +1,4 @@
use compliance_agent::{agent, api, config, database, migrate, scheduler, ssh, webhooks};
/// Run the `migrate onboarding` subcommand and exit. Backfills (or reverts) the
/// unified `onboarded_targets` collection per tenant.
///
/// Usage: `compliance-agent migrate onboarding [--all | --tenant <id>] [--dry-run] [--revert]`
async fn run_migration(
args: &[String],
pool: &database::DatabasePool,
) -> Result<(), compliance_agent::error::AgentError> {
if args.get(2).map(String::as_str) != Some("onboarding") {
eprintln!(
"usage: compliance-agent migrate onboarding [--all | --tenant <id>] [--dry-run] [--revert]"
);
std::process::exit(2);
}
let has = |flag: &str| args.iter().any(|a| a == flag);
let dry_run = has("--dry-run");
let revert = has("--revert");
let tenant = args
.iter()
.position(|a| a == "--tenant")
.and_then(|i| args.get(i + 1))
.cloned();
let tenants: Vec<String> = if has("--all") {
pool.list_tenant_ids().await?
} else if let Some(t) = tenant {
vec![t]
} else {
eprintln!("specify --all or --tenant <id>");
std::process::exit(2);
};
for tenant_id in tenants {
let db = pool.for_tenant_id(&tenant_id).await?;
if revert {
migrate::onboarding::revert(&db).await?;
println!("[{tenant_id}] reverted onboarding backfill");
} else {
let report = migrate::onboarding::backfill_onboarded_targets(&db, dry_run).await?;
let prefix = if dry_run { "(dry-run) " } else { "" };
println!("[{tenant_id}] {prefix}{report:?}");
}
}
Ok(())
}
use compliance_agent::{agent, api, config, database, scheduler, ssh, webhooks};
#[tokio::main]
async fn main() -> Result<(), Box<dyn std::error::Error>> {
@@ -77,13 +31,6 @@ async fn main() -> Result<(), Box<dyn std::error::Error>> {
let db_pool =
database::DatabasePool::connect(&config.mongodb_uri, &config.mongodb_database).await?;
// One-shot subcommands run and exit without starting the servers.
let args: Vec<String> = std::env::args().collect();
if args.get(1).map(String::as_str) == Some("migrate") {
run_migration(&args, &db_pool).await?;
return Ok(());
}
let agent = agent::ComplianceAgent::new(config.clone(), db_pool);
tracing::info!("Starting scheduler...");
-8
View File
@@ -1,8 +0,0 @@
//! One-time data migrations.
//!
//! Currently just the onboarding backfill ([`onboarding`]), which folds the
//! legacy `repositories` and `dast_targets` collections into the unified
//! `onboarded_targets` collection, preserving `_id` so every downstream record
//! keyed by `repo_id` / `target_id` keeps resolving.
pub mod onboarding;
-406
View File
@@ -1,406 +0,0 @@
//! Backfill: legacy `repositories` + `dast_targets` → `onboarded_targets`.
//!
//! The transforms here are **id-preserving**: an [`OnboardedTarget`] keeps the
//! same `_id` as the `TrackedRepository` / `DastTarget` it came from, so every
//! downstream collection keyed by that hex id (findings, sbom, scan_runs,
//! graph, dast_*, pentest_*) keeps resolving with zero row rewrites, and
//! existing webhook URLs keep working. The mapping functions are pure and unit
//! tested; the DB orchestration (idempotent per-tenant backfill + revert) is a
//! thin driver over them.
use compliance_core::models::{
Artifact, ArtifactKind, DastTarget, DastTargetType, GitArtifactConfig, IssueTrackerConfig,
OnboardedTarget, TargetType, TrackedRepository, WebArtifactConfig,
};
use futures_util::TryStreamExt;
use mongodb::bson::{doc, Document};
use crate::database::Database;
use crate::error::AgentError;
/// Marker id in `schema_migrations` recording that the backfill has run.
const MIGRATION_MARKER: &str = "onboarding_backfill_v1";
/// Summary of a backfill run.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct MigrationReport {
/// Repositories turned into onboarded targets.
pub repos_migrated: u64,
/// DAST targets folded into an existing (repo-linked) target as a LiveUrl.
pub dast_targets_folded: u64,
/// DAST targets with no repo link, migrated as standalone targets.
pub dast_targets_standalone: u64,
/// Records skipped because a target with that `_id` already existed.
pub skipped_existing: u64,
}
/// Map a legacy `DastTargetType` to a unified [`TargetType`]. REST/GraphQL APIs
/// are backend services; a browser app is a web app.
fn target_type_for_dast(kind: &DastTargetType) -> TargetType {
match kind {
DastTargetType::WebApp => TargetType::WebApp,
DastTargetType::RestApi | DastTargetType::GraphQl => TargetType::BackendService,
}
}
/// Build the LiveUrl artifact for a DAST target (its base URL + crawl config +
/// auth). Shared by fold-in and standalone migration.
pub fn dast_to_artifact(dast: &DastTarget) -> Artifact {
let mut artifact = Artifact::live_url(dast.base_url.clone());
artifact.web = Some(WebArtifactConfig {
target_kind: dast.target_type.clone(),
excluded_paths: dast.excluded_paths.clone(),
max_crawl_depth: dast.max_crawl_depth,
rate_limit: dast.rate_limit,
allow_destructive: dast.allow_destructive,
});
artifact.auth = dast.auth_config.clone().map(Into::into);
artifact
}
/// Map a `TrackedRepository` to an onboarded target, preserving `_id`. The git
/// remote becomes a `GitRepo` artifact carrying the repo's branch, watermark,
/// and auth; tracker config folds into `scan_config`.
///
/// `target_type` is a safe default (`BackendService`) — the classifier can
/// refine it later; `classification` is left `None` (unconfirmed).
pub fn repo_to_target(repo: &TrackedRepository) -> OnboardedTarget {
let mut target = OnboardedTarget::new(repo.name.clone(), TargetType::BackendService);
target.id = repo.id;
let mut artifact = Artifact::git_repo(repo.git_url.clone(), repo.default_branch.clone());
artifact.git = Some(GitArtifactConfig {
default_branch: repo.default_branch.clone(),
last_scanned_commit: repo.last_scanned_commit.clone(),
local_path: repo.local_path.clone(),
});
if repo.auth_token.is_some() || repo.auth_username.is_some() {
artifact.auth = Some(compliance_core::models::ArtifactAuth {
method: "token".to_string(),
username: repo.auth_username.clone(),
secret: repo.auth_token.clone(),
..Default::default()
});
}
target.artifacts.push(artifact);
if repo.tracker_type.is_some() {
target.scan_config.issue_tracker = Some(IssueTrackerConfig {
tracker_type: repo.tracker_type.clone(),
owner: repo.tracker_owner.clone(),
repo: repo.tracker_repo.clone(),
token: repo.tracker_token.clone(),
});
}
target.scan_schedule = repo.scan_schedule.clone();
target.webhook_enabled = repo.webhook_enabled;
target.webhook_secret = repo.webhook_secret.clone();
target.findings_count = repo.findings_count;
target.created_at = repo.created_at;
target.updated_at = repo.updated_at;
target
}
/// Append a DAST target's LiveUrl artifact onto an existing (repo-derived)
/// target. If the repo default was `BackendService` but the DAST target is a
/// browser web app, promote the type to `WebApp`.
pub fn fold_dast_into_target(target: &mut OnboardedTarget, dast: &DastTarget) {
if matches!(dast.target_type, DastTargetType::WebApp)
&& target.target_type == TargetType::BackendService
{
target.target_type = TargetType::WebApp;
}
if !target.has(ArtifactKind::LiveUrl) {
target.artifacts.push(dast_to_artifact(dast));
}
}
/// Map a repo-less DAST target to a standalone onboarded target, preserving `_id`.
pub fn dast_to_standalone_target(dast: &DastTarget) -> OnboardedTarget {
let mut target =
OnboardedTarget::new(dast.name.clone(), target_type_for_dast(&dast.target_type));
target.id = dast.id;
target.artifacts.push(dast_to_artifact(dast));
target.created_at = dast.created_at;
target.updated_at = dast.updated_at;
target
}
/// Whether the onboarding backfill has already been applied to this database.
pub async fn already_applied(db: &Database) -> Result<bool, AgentError> {
let found = db
.collection_named::<Document>("schema_migrations")
.find_one(doc! { "_id": MIGRATION_MARKER })
.await?;
Ok(found.is_some())
}
/// Backfill `onboarded_targets` from `repositories` + `dast_targets` for one
/// tenant database.
///
/// Id-preserving and **idempotent**: targets that already exist (by `_id`) are
/// skipped, so re-running is safe. With `dry_run`, computes the report without
/// writing. The legacy collections are never deleted; the only mutation outside
/// `onboarded_targets` is the history relink of folded DAST targets, which is
/// logged so [`revert`] can undo it.
pub async fn backfill_onboarded_targets(
db: &Database,
dry_run: bool,
) -> Result<MigrationReport, AgentError> {
let mut report = MigrationReport::default();
// 1. repositories -> onboarded_targets (preserve _id, skip existing).
let mut repos = db.repositories().find(doc! {}).await?;
while let Some(repo) = repos.try_next().await? {
let Some(id) = repo.id else { continue };
if db
.onboarded_targets()
.find_one(doc! { "_id": id })
.await?
.is_some()
{
report.skipped_existing += 1;
continue;
}
if !dry_run {
db.onboarded_targets()
.insert_one(repo_to_target(&repo))
.await?;
}
report.repos_migrated += 1;
}
// 2. dast_targets -> fold into the linked repo target, or migrate standalone.
let mut dasts = db.dast_targets().find(doc! {}).await?;
while let Some(dast) = dasts.try_next().await? {
let Some(dast_id) = dast.id else { continue };
let repo_oid = dast
.repo_id
.as_deref()
.and_then(|r| mongodb::bson::oid::ObjectId::parse_str(r).ok());
let linked = match repo_oid {
Some(oid) => db.onboarded_targets().find_one(doc! { "_id": oid }).await?,
None => None,
};
match (linked, repo_oid) {
// Fold into an existing repo-derived target.
(Some(mut target), Some(oid)) => {
if target.has(ArtifactKind::LiveUrl) {
report.skipped_existing += 1; // already folded on a prior run
continue;
}
fold_dast_into_target(&mut target, &dast);
if !dry_run {
db.onboarded_targets()
.replace_one(doc! { "_id": oid }, &target)
.await?;
relink_history(db, &dast_id.to_hex(), &oid.to_hex()).await?;
}
report.dast_targets_folded += 1;
}
// No linked repo target: migrate as a standalone target (keeps _id).
_ => {
if db
.onboarded_targets()
.find_one(doc! { "_id": dast_id })
.await?
.is_some()
{
report.skipped_existing += 1;
continue;
}
if !dry_run {
db.onboarded_targets()
.insert_one(dast_to_standalone_target(&dast))
.await?;
}
report.dast_targets_standalone += 1;
}
}
}
if !dry_run {
db.collection_named::<Document>("schema_migrations")
.update_one(
doc! { "_id": MIGRATION_MARKER },
doc! { "$set": { "applied_at": mongodb::bson::DateTime::now() } },
)
.upsert(true)
.await?;
}
Ok(report)
}
/// Relink DAST scan runs and pentest sessions from the old DAST target id to the
/// unified target id, logging each move so [`revert`] can undo it.
///
/// Note: if multiple DAST targets fold into the same repo target, revert
/// restores only the last-logged mapping — a rare edge. The source collections
/// (`repositories`, `dast_targets`) are never deleted, so no data is lost.
async fn relink_history(db: &Database, old_id: &str, new_id: &str) -> Result<(), AgentError> {
db.dast_scan_runs()
.update_many(
doc! { "target_id": old_id },
doc! { "$set": { "target_id": new_id } },
)
.await?;
db.pentest_sessions()
.update_many(
doc! { "target_id": old_id },
doc! { "$set": { "target_id": new_id } },
)
.await?;
db.collection_named::<Document>("onboarding_migration_log")
.insert_one(doc! { "old_target_id": old_id, "new_target_id": new_id })
.await?;
Ok(())
}
/// Undo the backfill: replay the relink log in reverse, drop `onboarded_targets`
/// and the log, and clear the marker. The legacy collections are untouched, so
/// this restores the pre-migration state.
pub async fn revert(db: &Database) -> Result<(), AgentError> {
let log = db.collection_named::<Document>("onboarding_migration_log");
let mut cursor = log.find(doc! {}).await?;
while let Some(entry) = cursor.try_next().await? {
if let (Ok(old), Ok(new)) = (
entry.get_str("old_target_id"),
entry.get_str("new_target_id"),
) {
db.dast_scan_runs()
.update_many(
doc! { "target_id": new },
doc! { "$set": { "target_id": old } },
)
.await?;
db.pentest_sessions()
.update_many(
doc! { "target_id": new },
doc! { "$set": { "target_id": old } },
)
.await?;
}
}
db.onboarded_targets().drop().await?;
log.drop().await?;
db.collection_named::<Document>("schema_migrations")
.delete_one(doc! { "_id": MIGRATION_MARKER })
.await?;
Ok(())
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
use compliance_core::models::{DastAuthConfig, TrackerType};
fn repo() -> TrackedRepository {
let mut r = TrackedRepository::new("acme".to_string(), "https://git/acme.git".to_string());
r.id = Some(mongodb::bson::oid::ObjectId::new());
r.default_branch = "develop".to_string();
r.last_scanned_commit = Some("abc123".to_string());
r.auth_token = Some("pat".to_string());
r.auth_username = Some("bob".to_string());
r.tracker_type = Some(TrackerType::Gitea);
r.tracker_owner = Some("acme".to_string());
r.findings_count = 7;
r
}
fn dast(repo_id: Option<String>, kind: DastTargetType) -> DastTarget {
let mut d = DastTarget::new(
"acme-web".to_string(),
"https://acme.example.com".to_string(),
kind,
);
d.id = Some(mongodb::bson::oid::ObjectId::new());
d.repo_id = repo_id;
d.max_crawl_depth = 5;
d.auth_config = Some(DastAuthConfig {
method: "bearer".to_string(),
login_url: None,
username: None,
password: None,
token: Some("tok".to_string()),
headers: None,
});
d
}
#[test]
fn repo_maps_preserving_id_and_git_artifact() {
let r = repo();
let t = repo_to_target(&r);
assert_eq!(t.id, r.id); // id preserved
assert_eq!(t.findings_count, 7);
assert_eq!(t.scan_schedule, r.scan_schedule);
let git = t.code_artifact().expect("git artifact");
assert_eq!(git.kind, ArtifactKind::GitRepo);
assert_eq!(git.source_ref, "https://git/acme.git");
let gc = git.git.as_ref().expect("git config");
assert_eq!(gc.default_branch, "develop");
assert_eq!(gc.last_scanned_commit.as_deref(), Some("abc123"));
let auth = git.auth.as_ref().expect("auth");
assert_eq!(auth.secret.as_deref(), Some("pat"));
assert_eq!(auth.username.as_deref(), Some("bob"));
assert_eq!(
t.scan_config
.issue_tracker
.as_ref()
.and_then(|it| it.tracker_type.clone()),
Some(TrackerType::Gitea)
);
}
#[test]
fn standalone_dast_maps_preserving_id_and_live_url() {
let d = dast(None, DastTargetType::WebApp);
let t = dast_to_standalone_target(&d);
assert_eq!(t.id, d.id);
assert_eq!(t.target_type, TargetType::WebApp);
let url = t.live_url().expect("live url");
assert_eq!(url.source_ref, "https://acme.example.com");
let web = url.web.as_ref().expect("web config");
assert_eq!(web.max_crawl_depth, 5);
assert_eq!(
url.auth.as_ref().and_then(|a| a.secret.clone()),
Some("tok".to_string())
);
}
#[test]
fn rest_api_dast_maps_to_backend_service() {
let d = dast(None, DastTargetType::RestApi);
assert_eq!(
dast_to_standalone_target(&d).target_type,
TargetType::BackendService
);
}
#[test]
fn fold_adds_live_url_and_promotes_webapp() {
let mut t = repo_to_target(&repo());
assert_eq!(t.target_type, TargetType::BackendService);
fold_dast_into_target(&mut t, &dast(Some("x".to_string()), DastTargetType::WebApp));
assert_eq!(t.target_type, TargetType::WebApp); // promoted
assert!(t.has(ArtifactKind::LiveUrl));
assert!(t.has(ArtifactKind::GitRepo));
}
#[test]
fn fold_is_idempotent_on_live_url() {
let mut t = repo_to_target(&repo());
let d = dast(Some("x".to_string()), DastTargetType::WebApp);
fold_dast_into_target(&mut t, &d);
fold_dast_into_target(&mut t, &d);
let live_urls = t
.artifacts
.iter()
.filter(|a| a.kind == ArtifactKind::LiveUrl)
.count();
assert_eq!(live_urls, 1);
}
}
+2 -3
View File
@@ -25,9 +25,8 @@ impl TestServer {
let mongodb_uri = std::env::var("TEST_MONGODB_URI")
.unwrap_or_else(|_| "mongodb://root:example@localhost:27017/?authSource=admin".into());
// Unique db-name prefix per run. Must fit the pool's 30-char cap
// (`<prefix>_<32 hex>` <= 63), so use a 16-hex-char suffix.
let db_name = format!("t_{}", &uuid::Uuid::new_v4().simple().to_string()[..16]);
// Unique database name per test run to avoid collisions
let db_name = format!("test_{}", uuid::Uuid::new_v4().simple());
let db_pool = DatabasePool::connect(&mongodb_uri, &db_name)
.await
@@ -2,6 +2,5 @@ mod cascade_delete;
mod dast;
mod findings;
mod health;
mod onboarding;
mod repositories;
mod stats;
@@ -1,115 +0,0 @@
use crate::common::TestServer;
use serde_json::json;
#[tokio::test]
async fn create_list_and_applicable_scans() {
let server = TestServer::start().await;
// Initially empty.
let resp = server.get("/api/v1/targets").await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
assert_eq!(body["data"].as_array().unwrap().len(), 0);
// Create a web-app target with a git repo + a live URL.
let resp = server
.post(
"/api/v1/targets",
&json!({
"name": "acme-web",
"target_type": "web_app",
"artifacts": [
{ "kind": "git_repo", "source_ref": "https://git/acme.git", "branch": "main" },
{ "kind": "live_url", "source_ref": "https://acme.example.com" }
]
}),
)
.await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
let id = body["data"]["_id"]["$oid"].as_str().unwrap().to_string();
assert!(!id.is_empty());
assert_eq!(body["data"]["artifacts"].as_array().unwrap().len(), 2);
// List returns it.
let resp = server.get("/api/v1/targets").await;
let body: serde_json::Value = resp.json().await.unwrap();
assert_eq!(body["data"].as_array().unwrap().len(), 1);
// Applicable scans: SAST present + DAST offered (live URL present), pentest supported.
let resp = server
.get(&format!("/api/v1/targets/{id}/applicable-scans"))
.await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
let scans = body["data"]["scans"].as_array().unwrap();
let names: Vec<&str> = scans.iter().filter_map(|s| s["scan"].as_str()).collect();
assert!(names.contains(&"sast"));
assert!(names.contains(&"dast"));
assert_eq!(body["data"]["pentest_supported"], true);
server.cleanup().await;
}
#[tokio::test]
async fn detect_classifies_a_plc_target() {
let server = TestServer::start().await;
// A PLC project artifact is a strong kind-based signal.
let resp = server
.post(
"/api/v1/targets",
&json!({
"name": "line-controller",
"target_type": "backend_service", // deliberately wrong; detect should suggest PLC
"artifacts": [
{ "kind": "plc_project", "source_ref": "line.xml", "plc_format": "plcopen_xml" }
]
}),
)
.await;
let body: serde_json::Value = resp.json().await.unwrap();
let id = body["data"]["_id"]["$oid"].as_str().unwrap().to_string();
let resp = server
.post(&format!("/api/v1/targets/{id}/detect"), &json!({}))
.await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
assert_eq!(body["data"]["classification"]["suggested"], "plc_sps");
server.cleanup().await;
}
#[tokio::test]
async fn add_artifact_and_delete_target() {
let server = TestServer::start().await;
let resp = server
.post(
"/api/v1/targets",
&json!({ "name": "svc", "target_type": "backend_service" }),
)
.await;
let body: serde_json::Value = resp.json().await.unwrap();
let id = body["data"]["_id"]["$oid"].as_str().unwrap().to_string();
// Attach a git repo.
let resp = server
.post(
&format!("/api/v1/targets/{id}/artifacts"),
&json!({ "kind": "git_repo", "source_ref": "https://git/svc.git" }),
)
.await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
assert_eq!(body["data"]["artifacts"].as_array().unwrap().len(), 1);
// Delete it.
let resp = server.delete(&format!("/api/v1/targets/{id}")).await;
assert_eq!(resp.status(), 200);
let resp = server.get(&format!("/api/v1/targets/{id}")).await;
assert_eq!(resp.status(), 404);
server.cleanup().await;
}
@@ -1,156 +0,0 @@
// Integration tests for the onboarding backfill migration.
//
// Requires MongoDB (set TEST_MONGODB_URI if not at the default).
// Not run in CI (which is `--lib` only) — run locally:
// cargo test -p compliance-agent --test e2e migration
use compliance_agent::database::{Database, DatabasePool};
use compliance_agent::migrate::onboarding;
use compliance_core::models::{
ArtifactKind, DastTarget, DastTargetType, TargetType, TrackedRepository,
};
use mongodb::bson::{doc, Document};
async fn fresh_db() -> (DatabasePool, String, Database) {
let uri = std::env::var("TEST_MONGODB_URI")
.unwrap_or_else(|_| "mongodb://root:example@localhost:27017/?authSource=admin".into());
// Prefix must fit the pool's 30-char cap (`<prefix>_<32 hex>` <= 63).
let prefix = format!("t_{}", &uuid::Uuid::new_v4().simple().to_string()[..16]);
let pool = DatabasePool::connect(&uri, &prefix)
.await
.expect("connect mongo");
let db = pool.for_tenant_id("t1").await.expect("tenant db");
(pool, prefix, db)
}
async fn cleanup(pool: &DatabasePool, prefix: &str) {
if let Ok(names) = pool.client().list_database_names().await {
for n in names {
if n.starts_with(prefix) {
pool.client().database(&n).drop().await.ok();
}
}
}
}
#[tokio::test]
async fn backfill_folds_relinks_is_idempotent_and_reversible() {
let (pool, prefix, db) = fresh_db().await;
// Seed a repo.
let repo = TrackedRepository::new("acme".into(), "https://git/acme.git".into());
let repo_id = db
.repositories()
.insert_one(repo)
.await
.expect("insert repo")
.inserted_id
.as_object_id()
.expect("repo oid");
// A DAST target linked to the repo (folds + promotes to WebApp + relinks).
let mut linked = DastTarget::new(
"acme-web".into(),
"https://acme.example.com".into(),
DastTargetType::WebApp,
);
linked.repo_id = Some(repo_id.to_hex());
let linked_id = db
.dast_targets()
.insert_one(linked)
.await
.expect("insert linked dast")
.inserted_id
.as_object_id()
.expect("linked oid");
// A repo-less DAST target (standalone).
let standalone = DastTarget::new(
"acme-api".into(),
"https://api.acme.com".into(),
DastTargetType::RestApi,
);
let standalone_id = db
.dast_targets()
.insert_one(standalone)
.await
.expect("insert standalone dast")
.inserted_id
.as_object_id()
.expect("standalone oid");
// A DAST scan run pointing at the linked target — should be relinked to the repo.
db.collection_named::<Document>("dast_scan_runs")
.insert_one(doc! { "target_id": linked_id.to_hex(), "status": "completed" })
.await
.expect("insert dast run");
// --- Backfill ---
assert!(!onboarding::already_applied(&db).await.unwrap());
let report = onboarding::backfill_onboarded_targets(&db, false)
.await
.expect("backfill");
assert_eq!(report.repos_migrated, 1);
assert_eq!(report.dast_targets_folded, 1);
assert_eq!(report.dast_targets_standalone, 1);
assert!(onboarding::already_applied(&db).await.unwrap());
// Repo target: preserved _id, has git + folded live-url, promoted to WebApp.
let repo_target = db
.onboarded_targets()
.find_one(doc! { "_id": repo_id })
.await
.unwrap()
.expect("repo target");
assert!(repo_target.has(ArtifactKind::GitRepo));
assert!(repo_target.has(ArtifactKind::LiveUrl));
assert_eq!(repo_target.target_type, TargetType::WebApp);
// Standalone target: preserved _id, live-url, backend service.
let standalone_target = db
.onboarded_targets()
.find_one(doc! { "_id": standalone_id })
.await
.unwrap()
.expect("standalone target");
assert!(standalone_target.has(ArtifactKind::LiveUrl));
assert_eq!(standalone_target.target_type, TargetType::BackendService);
// The DAST run was relinked from the old dast id to the repo (unified) id.
let run = db
.collection_named::<Document>("dast_scan_runs")
.find_one(doc! {})
.await
.unwrap()
.expect("run");
assert_eq!(run.get_str("target_id").unwrap(), repo_id.to_hex());
// --- Idempotent: re-run migrates nothing new ---
let again = onboarding::backfill_onboarded_targets(&db, false)
.await
.expect("backfill again");
assert_eq!(again.repos_migrated, 0);
assert_eq!(again.dast_targets_folded, 0);
assert_eq!(again.dast_targets_standalone, 0);
assert!(again.skipped_existing >= 2);
// --- Revert: onboarded targets gone, relink undone, marker cleared ---
onboarding::revert(&db).await.expect("revert");
assert_eq!(
db.onboarded_targets()
.count_documents(doc! {})
.await
.unwrap(),
0
);
let run_after = db
.collection_named::<Document>("dast_scan_runs")
.find_one(doc! {})
.await
.unwrap()
.expect("run");
assert_eq!(run_after.get_str("target_id").unwrap(), linked_id.to_hex());
assert!(!onboarding::already_applied(&db).await.unwrap());
cleanup(&pool, &prefix).await;
}
@@ -7,4 +7,3 @@
// Or nightly: (via CI with MongoDB service container)
mod api;
mod migration;