Compare commits

..
Author SHA1 Message Date
Sharang ParnerkarandClaude Opus 4.8 e3b918b365 fix(dashboard): Findings/SBOM filter by onboarded targets; accurate target findings_count
CI / Check (pull_request) Successful in 7m4s
CI / Detect Changes (pull_request) Has been skipped
CI / Deploy Agent (pull_request) Has been skipped
CI / Deploy Dashboard (pull_request) Has been skipped
CI / Deploy Docs (pull_request) Has been skipped
CI / Deploy MCP (pull_request) Has been skipped
The Findings and SBOM pages populated their target dropdown from the legacy
`repositories` collection, so onboarded targets never appeared and their
findings/SBOM couldn't be filtered by name (the data was there, keyed by the
target id). Point both dropdowns at `onboarded_targets` via `fetch_targets`.

Also refresh `OnboardedTarget.findings_count` at the end of `run_target`: the
shared pipeline (Stage 7) increments the legacy `repositories` doc, which the
unified path has none of, so the Targets page always showed 0. Set the accurate
total (count of findings keyed by the target id) on the target itself.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-13 09:52:09 +02:00
134 changed files with 2379 additions and 17117 deletions
-18
View File
@@ -34,24 +34,6 @@ SCAN_SCHEDULE=0 0 */6 * * *
CVE_MONITOR_SCHEDULE=0 0 0 * * *
GIT_CLONE_BASE_PATH=/tmp/compliance-scanner/repos
# Dynamic PLC testing — ephemeral soft-PLC provisioning (#183). Off unless
# enabled; requires the agent container to have Docker access (socket mount).
# When on, a PLC/SPS target with control logic but no reachable device gets its
# logic instantiated on a throwaway OpenPLC, probed, then torn down.
PLC_RUNTIME_ENABLED=0
PLC_RUNTIME_IMAGE=registry.meghsakha.com/openplc:latest
PLC_RUNTIME_NETWORK=certifai
PLC_RUNTIME_MEMORY=512m
PLC_RUNTIME_CPUS=0.5
PLC_RUNTIME_MAX_LIFETIME_SECS=180
PLC_RUNTIME_OPENPLC_USER=openplc
PLC_RUNTIME_OPENPLC_PASSWORD=openplc
# Werkbank runner API (/api/v1/werkbank/jobs/*, /api/v1/werkbank/artifacts/*).
# When set, mounts the runner-facing queue + artifact endpoints behind this
# bearer token; runners present the same token. Unset = endpoints not mounted.
WERKBANK_RUNNER_TOKEN=
# Dashboard
DASHBOARD_PORT=8080
AGENT_API_URL=http://localhost:3001
+13 -36
View File
@@ -7,13 +7,6 @@ on:
pull_request:
env:
# registry + cosign creds via env, NOT inline ${{ }}: the Harbor robot
# username contains '$', which sh expands when interpolated into the
# script (robot$ci-push -> robot-push) => docker login unauthorized.
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
COSIGN_KEY: ${{ secrets.COSIGN_KEY }}
COSIGN_PASSWORD: ${{ secrets.COSIGN_PASSWORD }}
CARGO_TERM_COLOR: always
RUSTFLAGS: "-D warnings"
# Compile cache: sccache -> Hetzner S3 (breakpilot-sccache), runner-independent
@@ -72,7 +65,7 @@ jobs:
echo '[source.crates-io]'
echo 'replace-with = "kellnr"'
echo '[registries.kellnr]'
echo 'index = "sparse+https://crates.breakpilot.com/api/v1/cratesio/"'
echo 'index = "sparse+https://crates.meghsakha.com/api/v1/cratesio/"'
} >> "$CARGO_HOME/config.toml"
env:
RUSTC_WRAPPER: ""
@@ -94,8 +87,8 @@ jobs:
- name: Configure git auth for private tramiton dependency
run: |
git config --global \
url."https://sharang:${{ secrets.TRAMITON_FETCH_TOKEN }}@git.breakpilot.com/".insteadOf \
"ssh://git@git.breakpilot.com:22222/"
url."https://sharang:${{ secrets.TRAMITON_FETCH_TOKEN }}@gitea.meghsakha.com/".insteadOf \
"ssh://git@gitea.meghsakha.com:22222/"
env:
RUSTC_WRAPPER: ""
@@ -114,10 +107,6 @@ jobs:
run: cargo clippy -p compliance-dashboard --features web --no-default-features -- -D warnings
- name: Clippy (mcp)
run: cargo clippy -p compliance-mcp -- -D warnings
- name: Clippy (werkbank-exec)
run: cargo clippy -p werkbank-exec -- -D warnings
- name: Clippy (control-map)
run: cargo clippy -p control-map -- -D warnings
# Security audit
- name: Security Audit
@@ -126,8 +115,8 @@ jobs:
RUSTC_WRAPPER: ""
# Tests (reuses compilation artifacts from clippy)
- name: Tests (core + agent + werkbank-exec + control-map)
run: cargo test -p compliance-core -p compliance-agent -p werkbank-exec -p control-map --lib
- name: Tests (core + agent)
run: cargo test -p compliance-core -p compliance-agent --lib
- name: Tests (dashboard server)
run: cargo test -p compliance-dashboard --features server --no-default-features
- name: Tests (dashboard web)
@@ -213,14 +202,11 @@ jobs:
apk add --no-cache git curl openssl
git init && git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
git fetch --depth=1 origin "${GITHUB_SHA}" && git checkout FETCH_HEAD
IMAGE=repo.breakpilot.com/certifai/compliance-agent
echo "$REGISTRY_PASSWORD" | docker login repo.breakpilot.com -u "$REGISTRY_USERNAME" --password-stdin
IMAGE=registry.meghsakha.com/compliance-agent
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login registry.meghsakha.com -u "${{ secrets.REGISTRY_USERNAME }}" --password-stdin
DOCKER_BUILDKIT=1 docker build --secret id=tramiton_token,env=TRAMITON_FETCH_TOKEN \
-f Dockerfile.agent -t "$IMAGE:latest" -t "$IMAGE:${GITHUB_SHA}" .
docker push "$IMAGE:latest" && docker push "$IMAGE:${GITHUB_SHA}"
{ command -v cosign >/dev/null 2>&1 || curl -sSfLo /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64 || wget -qO /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64; } || echo "::warning::cosign fetch failed"
chmod +x /usr/local/bin/cosign 2>/dev/null || true
cosign sign --yes --key env://COSIGN_KEY "$IMAGE:latest" || echo "::warning::cosign failed"
PAYLOAD=$(printf '{"ref":"refs/heads/main","repository":{"full_name":"sharang/compliance-scanner-agent"},"head_commit":{"id":"%s","message":"deploy agent"}}' "${GITHUB_SHA}")
SIG=$(printf '%s' "$PAYLOAD" | openssl dgst -sha256 -hmac "${{ secrets.ORCA_WEBHOOK_SECRET }}" | awk '{print $2}')
RESP=$(curl -fsS -w "\nHTTP %{http_code}" -X POST "http://46.225.100.82:6880/api/v1/webhooks/github" -H "Content-Type: application/json" -H "X-Hub-Signature-256: sha256=$SIG" -d "$PAYLOAD"); echo "$RESP"
@@ -240,14 +226,11 @@ jobs:
apk add --no-cache git curl openssl
git init && git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
git fetch --depth=1 origin "${GITHUB_SHA}" && git checkout FETCH_HEAD
IMAGE=repo.breakpilot.com/certifai/compliance-dashboard
echo "$REGISTRY_PASSWORD" | docker login repo.breakpilot.com -u "$REGISTRY_USERNAME" --password-stdin
IMAGE=registry.meghsakha.com/compliance-dashboard
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login registry.meghsakha.com -u "${{ secrets.REGISTRY_USERNAME }}" --password-stdin
DOCKER_BUILDKIT=1 docker build --secret id=tramiton_token,env=TRAMITON_FETCH_TOKEN \
-f Dockerfile.dashboard -t "$IMAGE:latest" -t "$IMAGE:${GITHUB_SHA}" .
docker push "$IMAGE:latest" && docker push "$IMAGE:${GITHUB_SHA}"
{ command -v cosign >/dev/null 2>&1 || curl -sSfLo /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64 || wget -qO /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64; } || echo "::warning::cosign fetch failed"
chmod +x /usr/local/bin/cosign 2>/dev/null || true
cosign sign --yes --key env://COSIGN_KEY "$IMAGE:latest" || echo "::warning::cosign failed"
PAYLOAD=$(printf '{"ref":"refs/heads/main","repository":{"full_name":"sharang/compliance-scanner-agent"},"head_commit":{"id":"%s","message":"deploy dashboard"}}' "${GITHUB_SHA}")
SIG=$(printf '%s' "$PAYLOAD" | openssl dgst -sha256 -hmac "${{ secrets.ORCA_WEBHOOK_SECRET }}" | awk '{print $2}')
RESP=$(curl -fsS -w "\nHTTP %{http_code}" -X POST "http://46.225.100.82:6880/api/v1/webhooks/github" -H "Content-Type: application/json" -H "X-Hub-Signature-256: sha256=$SIG" -d "$PAYLOAD"); echo "$RESP"
@@ -265,13 +248,10 @@ jobs:
apk add --no-cache git curl openssl
git init && git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
git fetch --depth=1 origin "${GITHUB_SHA}" && git checkout FETCH_HEAD
IMAGE=repo.breakpilot.com/certifai/compliance-docs
echo "$REGISTRY_PASSWORD" | docker login repo.breakpilot.com -u "$REGISTRY_USERNAME" --password-stdin
IMAGE=registry.meghsakha.com/compliance-docs
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login registry.meghsakha.com -u "${{ secrets.REGISTRY_USERNAME }}" --password-stdin
docker build -f Dockerfile.docs -t "$IMAGE:latest" -t "$IMAGE:${GITHUB_SHA}" .
docker push "$IMAGE:latest" && docker push "$IMAGE:${GITHUB_SHA}"
{ command -v cosign >/dev/null 2>&1 || curl -sSfLo /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64 || wget -qO /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64; } || echo "::warning::cosign fetch failed"
chmod +x /usr/local/bin/cosign 2>/dev/null || true
cosign sign --yes --key env://COSIGN_KEY "$IMAGE:latest" || echo "::warning::cosign failed"
PAYLOAD=$(printf '{"ref":"refs/heads/main","repository":{"full_name":"sharang/compliance-scanner-agent"},"head_commit":{"id":"%s","message":"deploy docs"}}' "${GITHUB_SHA}")
SIG=$(printf '%s' "$PAYLOAD" | openssl dgst -sha256 -hmac "${{ secrets.ORCA_WEBHOOK_SECRET }}" | awk '{print $2}')
RESP=$(curl -fsS -w "\nHTTP %{http_code}" -X POST "http://46.225.100.82:6880/api/v1/webhooks/github" -H "Content-Type: application/json" -H "X-Hub-Signature-256: sha256=$SIG" -d "$PAYLOAD"); echo "$RESP"
@@ -291,14 +271,11 @@ jobs:
apk add --no-cache git curl openssl
git init && git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
git fetch --depth=1 origin "${GITHUB_SHA}" && git checkout FETCH_HEAD
IMAGE=repo.breakpilot.com/certifai/compliance-mcp
echo "$REGISTRY_PASSWORD" | docker login repo.breakpilot.com -u "$REGISTRY_USERNAME" --password-stdin
IMAGE=registry.meghsakha.com/compliance-mcp
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login registry.meghsakha.com -u "${{ secrets.REGISTRY_USERNAME }}" --password-stdin
DOCKER_BUILDKIT=1 docker build --secret id=tramiton_token,env=TRAMITON_FETCH_TOKEN \
-f Dockerfile.mcp -t "$IMAGE:latest" -t "$IMAGE:${GITHUB_SHA}" .
docker push "$IMAGE:latest" && docker push "$IMAGE:${GITHUB_SHA}"
{ command -v cosign >/dev/null 2>&1 || curl -sSfLo /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64 || wget -qO /usr/local/bin/cosign https://github.com/sigstore/cosign/releases/download/v2.4.3/cosign-linux-amd64; } || echo "::warning::cosign fetch failed"
chmod +x /usr/local/bin/cosign 2>/dev/null || true
cosign sign --yes --key env://COSIGN_KEY "$IMAGE:latest" || echo "::warning::cosign failed"
PAYLOAD=$(printf '{"ref":"refs/heads/main","repository":{"full_name":"sharang/compliance-scanner-agent"},"head_commit":{"id":"%s","message":"deploy mcp"}}' "${GITHUB_SHA}")
SIG=$(printf '%s' "$PAYLOAD" | openssl dgst -sha256 -hmac "${{ secrets.ORCA_WEBHOOK_SECRET }}" | awk '{print $2}')
RESP=$(curl -fsS -w "\nHTTP %{http_code}" -X POST "http://46.225.100.82:6880/api/v1/webhooks/github" -H "Content-Type: application/json" -H "X-Hub-Signature-256: sha256=$SIG" -d "$PAYLOAD"); echo "$RESP"
Generated
+3 -87
View File
@@ -666,7 +666,6 @@ dependencies = [
"compliance-core",
"compliance-dast",
"compliance-graph",
"control-map",
"dashmap",
"dotenvy",
"futures-core",
@@ -680,7 +679,6 @@ dependencies = [
"rand 0.9.2",
"regex",
"reqwest",
"roxmltree",
"secrecy",
"serde",
"serde_json",
@@ -695,12 +693,9 @@ dependencies = [
"tracing",
"tracing-subscriber",
"tramiton-core",
"tramiton-repro",
"tramiton-sbom",
"urlencoding",
"uuid",
"walkdir",
"werkbank-exec",
"zip",
]
@@ -725,7 +720,6 @@ dependencies = [
"sha2",
"thiserror 2.0.18",
"tokio",
"toml",
"tracing",
"tracing-opentelemetry",
"tracing-subscriber",
@@ -969,15 +963,6 @@ dependencies = [
"charset",
]
[[package]]
name = "control-map"
version = "0.1.0"
dependencies = [
"serde",
"serde_json",
"thiserror 2.0.18",
]
[[package]]
name = "convert_case"
version = "0.8.0"
@@ -3782,15 +3767,6 @@ dependencies = [
"syn",
]
[[package]]
name = "object"
version = "0.36.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "62948e14d923ea95ea2c7c86c71013138b66525b86bdc08d2dcc262bdb497b87"
dependencies = [
"memchr",
]
[[package]]
name = "octocrab"
version = "0.44.1"
@@ -4641,12 +4617,6 @@ dependencies = [
"syn",
]
[[package]]
name = "roxmltree"
version = "0.20.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6c20b6793b5c2fa6553b250154b78d6d0db37e72700ae35fad9387a46f487c97"
[[package]]
name = "rust-stemmers"
version = "1.2.0"
@@ -5099,12 +5069,6 @@ dependencies = [
"digest",
]
[[package]]
name = "sha1_smol"
version = "1.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bbfa15b3dddfee50a0fff136974b3e1bde555604ba463834a7eb7deb6417705d"
[[package]]
name = "sha2"
version = "0.10.9"
@@ -5240,7 +5204,7 @@ version = "0.8.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c1c97747dbf44bb1ca44a561ece23508e99cb592e862f22222dcf42f51d1e451"
dependencies = [
"heck 0.5.0",
"heck 0.4.1",
"proc-macro2",
"quote",
"syn",
@@ -6175,8 +6139,8 @@ dependencies = [
[[package]]
name = "tramiton-core"
version = "0.4.1"
source = "git+ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git?tag=v0.4.1#ae4fc1376279f9edb9882605b20877335e7ba8ba"
version = "0.4.0"
source = "git+ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git?tag=v0.4.0#e3dc1bf7027a2f6d7b1fe43043d6dfa887ce4af3"
dependencies = [
"serde",
"tempfile",
@@ -6185,34 +6149,6 @@ dependencies = [
"walkdir",
]
[[package]]
name = "tramiton-repro"
version = "0.4.1"
source = "git+ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git?tag=v0.4.1#ae4fc1376279f9edb9882605b20877335e7ba8ba"
dependencies = [
"serde",
"serde_json",
"sha2",
"tempfile",
"thiserror 1.0.69",
"toml",
"tramiton-core",
"walkdir",
]
[[package]]
name = "tramiton-sbom"
version = "0.4.1"
source = "git+ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git?tag=v0.4.1#ae4fc1376279f9edb9882605b20877335e7ba8ba"
dependencies = [
"object",
"serde",
"serde_json",
"sha2",
"tramiton-core",
"tramiton-repro",
]
[[package]]
name = "tree-sitter"
version = "0.24.7"
@@ -6488,7 +6424,6 @@ dependencies = [
"getrandom 0.4.1",
"js-sys",
"serde_core",
"sha1_smol",
"wasm-bindgen",
]
@@ -6732,25 +6667,6 @@ dependencies = [
"rustls-pki-types",
]
[[package]]
name = "werkbank-exec"
version = "0.1.0"
dependencies = [
"compliance-core",
"compliance-dast",
"futures-util",
"hex",
"regex",
"reqwest",
"secrecy",
"sha2",
"thiserror 2.0.18",
"tokio",
"tracing",
"uuid",
"walkdir",
]
[[package]]
name = "which"
version = "6.0.3"
+2 -5
View File
@@ -7,8 +7,6 @@ members = [
"compliance-dast",
"compliance-mcp",
"compliance-smoke",
"werkbank-exec",
"control-map",
]
resolver = "2"
@@ -18,7 +16,6 @@ expect_used = "deny"
[workspace.dependencies]
compliance-core = { path = "compliance-core", default-features = false }
control-map = { path = "control-map" }
serde = { version = "1", features = ["derive"] }
serde_json = "1"
tokio = { version = "1", features = ["full"] }
@@ -26,11 +23,11 @@ tracing = "0.1"
tracing-subscriber = { version = "0.3", features = ["env-filter"] }
chrono = { version = "0.4", features = ["serde"] }
mongodb = { version = "3", features = ["rustls-tls", "compat-3-0-0"] }
reqwest = { version = "0.12", features = ["json", "rustls-tls", "multipart", "cookies"], default-features = false }
reqwest = { version = "0.12", features = ["json", "rustls-tls"], default-features = false }
thiserror = "2"
sha2 = "0.10"
hex = "0.4"
uuid = { version = "1", features = ["v4", "v5", "serde"] }
uuid = { version = "1", features = ["v4", "serde"] }
secrecy = { version = "0.10", features = ["serde"] }
regex = "1"
zip = { version = "2", features = ["aes-crypto", "deflate"] }
+3 -33
View File
@@ -8,17 +8,11 @@ COPY . .
RUN --mount=type=secret,id=tramiton_token \
if [ -s /run/secrets/tramiton_token ]; then \
git config --global \
url."https://sharang:$(cat /run/secrets/tramiton_token)@git.breakpilot.com/".insteadOf \
"ssh://git@git.breakpilot.com:22222/"; \
url."https://sharang:$(cat /run/secrets/tramiton_token)@gitea.meghsakha.com/".insteadOf \
"ssh://git@gitea.meghsakha.com:22222/"; \
fi && \
CARGO_NET_GIT_FETCH_WITH_CLI=true cargo build --release -p compliance-agent
# A throwaway stage that packs a real nix store (store paths + the validity DB)
# into a compressed bootstrap tarball. Only the tarball is copied into the final
# image, so we don't carry a raw /nix copy layer.
FROM nixos/nix:latest AS nixseed
RUN tar -C / -czf /nix-bootstrap.tar.gz nix
FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y ca-certificates libssl3 git curl python3 python3-pip npm golang-go php-cli && rm -rf /var/lib/apt/lists/*
@@ -46,30 +40,7 @@ RUN pip3 install --break-system-packages semgrep
# Install ruff for Python linting
RUN pip3 install --break-system-packages ruff
# Real nix for the tramiton reproducible-build firmware SBOM.
#
# nix-portable's proot fallback can't run here: user namespaces are blocked by
# the container's default seccomp/apparmor profile, and orca exposes no way to
# relax it. So ship a *real* nix and disable its build sandbox
# (`sandbox = false`) — a plain gcc/make firmware build needs no user namespace,
# so it runs fine under the locked-down profile with no proot involved.
#
# The store is shipped as a bootstrap tarball and seeded onto /nix at first
# start (see docker/agent-entrypoint.sh), so a persistent /nix volume survives
# redeploys. A missing/broken nix just falls back to the analysis-only SBOM.
COPY --from=nixseed /nix-bootstrap.tar.gz /opt/nix-bootstrap.tar.gz
ENV PATH="/nix/var/nix/profiles/default/bin:${PATH}"
RUN mkdir -p /etc/nix && printf '%s\n' \
'experimental-features = nix-command flakes' \
'sandbox = false' \
'build-users-group =' \
'substituters = https://cache.nixos.org' \
'trusted-public-keys = cache.nixos.org-1:6NCHdD59X431o0gWypbMrAURkbJ16ZPMQFGspcDShjY=' \
> /etc/nix/nix.conf
COPY --from=builder /app/target/release/compliance-agent /usr/local/bin/compliance-agent
COPY docker/agent-entrypoint.sh /usr/local/bin/agent-entrypoint.sh
RUN chmod +x /usr/local/bin/agent-entrypoint.sh
# Copy documentation for the help chat assistant
COPY --from=builder /app/README.md /app/README.md
@@ -81,6 +52,5 @@ RUN mkdir -p /data/compliance-scanner/ssh
EXPOSE 3001 3002
# Seeds /nix (fresh volume) from the bootstrap tarball, then runs the agent.
ENTRYPOINT ["/usr/local/bin/agent-entrypoint.sh"]
ENTRYPOINT ["compliance-agent"]
+2 -2
View File
@@ -13,8 +13,8 @@ ENV DOCS_URL=${DOCS_URL}
RUN --mount=type=secret,id=tramiton_token \
if [ -s /run/secrets/tramiton_token ]; then \
git config --global \
url."https://sharang:$(cat /run/secrets/tramiton_token)@git.breakpilot.com/".insteadOf \
"ssh://git@git.breakpilot.com:22222/"; \
url."https://sharang:$(cat /run/secrets/tramiton_token)@gitea.meghsakha.com/".insteadOf \
"ssh://git@gitea.meghsakha.com:22222/"; \
fi && \
CARGO_NET_GIT_FETCH_WITH_CLI=true dx build --release --package compliance-dashboard
+2 -2
View File
@@ -8,8 +8,8 @@ COPY . .
RUN --mount=type=secret,id=tramiton_token \
if [ -s /run/secrets/tramiton_token ]; then \
git config --global \
url."https://sharang:$(cat /run/secrets/tramiton_token)@git.breakpilot.com/".insteadOf \
"ssh://git@git.breakpilot.com:22222/"; \
url."https://sharang:$(cat /run/secrets/tramiton_token)@gitea.meghsakha.com/".insteadOf \
"ssh://git@gitea.meghsakha.com:22222/"; \
fi && \
CARGO_NET_GIT_FETCH_WITH_CLI=true cargo build --release -p compliance-mcp
+3 -14
View File
@@ -8,22 +8,13 @@ workspace = true
[dependencies]
compliance-core = { workspace = true, features = ["mongodb", "telemetry", "axum"] }
control-map = { workspace = true }
compliance-graph = { path = "../compliance-graph" }
compliance-dast = { path = "../compliance-dast" }
# Shared dynamic-execution logic (soft-PLC provisioning + ICS probing), also
# used by the Werkbank runner.
werkbank-exec = { path = "../werkbank-exec" }
# Native firmware build/target detection for bare-metal & RTOS artifacts.
# Same-company IP, used directly (not via CLI) so the whole tramiton suite is
# available to the onboarding classifier. NOTE: CI must be able to fetch this
# private repo (see the git-auth step in .gitea/workflows/ci.yml).
tramiton-core = { git = "ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git", tag = "v0.4.1" }
# tramiton-repro drives the reproducible build (NixBackend seal_and_build) that
# yields a sealed lock; `libraries_from_inputs` is the analysis-only fallback.
tramiton-repro = { git = "ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git", tag = "v0.4.1" }
# tramiton-sbom renders the bill of materials from a sealed lock (+ binary SCA).
tramiton-sbom = { git = "ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git", tag = "v0.4.1" }
tramiton-core = { git = "ssh://git@gitea.meghsakha.com:22222/sharang/tramiton.git", tag = "v0.4.0" }
serde = { workspace = true }
serde_json = { workspace = true }
tokio = { workspace = true }
@@ -38,7 +29,7 @@ hex = { workspace = true }
uuid = { workspace = true }
secrecy = { workspace = true }
regex = { workspace = true }
axum = { version = "0.8", features = ["multipart"] }
axum = "0.8"
tower-http = { version = "0.6", features = ["cors", "trace", "set-header"] }
git2 = "0.20"
octocrab = "0.44"
@@ -46,8 +37,6 @@ tokio-cron-scheduler = "0.13"
dotenvy = "0.15"
hmac = "0.12"
walkdir = "2"
# Read-only XML tree parsing for PLCopen project files (POU extraction).
roxmltree = "0.20"
base64 = "0.22"
urlencoding = "2"
futures-util = "0.3"
@@ -69,5 +58,5 @@ tokio = { workspace = true }
mongodb = { workspace = true }
uuid = { workspace = true }
secrecy = { workspace = true }
axum = { version = "0.8", features = ["multipart"] }
axum = "0.8"
tower-http = { version = "0.6", features = ["cors"] }
-117
View File
@@ -1,117 +0,0 @@
# Custom semgrep rules for CRA controls that no off-the-shelf ruleset digs out.
# Each rule id is `cra-ai-<n>-<slug>` and is keyed back to its control via the
# `control-map` LUT (by rule-id suffix, so semgrep's path prefix on check_id does
# not matter). Detection here is deterministic; the grounded LLM judge downstream
# only confirms/refutes — it never detects. Keep patterns tight: a false positive
# that the judge refutes marks the whole finding a false positive.
rules:
# --- cra-ai-1: Secure-by-Default-Konfiguration -------------------------------
- id: cra-ai-1-flask-debug-enabled
languages: [python]
severity: WARNING
message: Flask app started with debug=True — ships an interactive debugger / code execution in production (secure-by-default violation).
metadata:
cwe: ["CWE-489: Active Debug Code"]
control: cra-ai-1
patterns:
- pattern: '$APP.run(..., debug=True, ...)'
- id: cra-ai-1-django-debug-true
languages: [python]
severity: WARNING
message: Django DEBUG = True — leaks stack traces / settings in production (secure-by-default violation).
metadata:
cwe: ["CWE-489: Active Debug Code"]
control: cra-ai-1
patterns:
- pattern: 'DEBUG = True'
- id: cra-ai-1-tls-verify-disabled
languages: [python]
severity: ERROR
message: TLS certificate verification disabled (verify=False) — defeats transport security by default.
metadata:
cwe: ["CWE-295: Improper Certificate Validation"]
control: cra-ai-1
patterns:
- pattern: 'requests.$M(..., verify=False, ...)'
- id: cra-ai-1-cors-wildcard
languages: [javascript, typescript]
severity: WARNING
message: CORS Access-Control-Allow-Origin set to "*" — opens the API to any origin by default.
metadata:
cwe: ["CWE-942: Permissive Cross-domain Policy with Untrusted Domains"]
control: cra-ai-1
patterns:
- pattern-either:
- pattern: '$RES.header("Access-Control-Allow-Origin", "*")'
- pattern: '$RES.setHeader("Access-Control-Allow-Origin", "*")'
# --- cra-ai-7: Starke Authentifizierung (weak password hashing) --------------
- id: cra-ai-7-weak-password-hash
languages: [python]
severity: ERROR
message: Password/secret hashed with a fast, broken digest (md5/sha1) — use a password KDF (bcrypt/scrypt/argon2).
metadata:
cwe: ["CWE-916: Use of Password Hash With Insufficient Computational Effort"]
control: cra-ai-7
patterns:
- pattern-either:
- pattern: 'hashlib.md5($PW)'
- pattern: 'hashlib.sha1($PW)'
- metavariable-regex:
metavariable: $PW
regex: '(?i).*(pass|pwd|secret|cred|token).*'
# --- cra-ai-10: Sitzungsmanagement (insecure session cookies) ----------------
- id: cra-ai-10-session-cookie-insecure
languages: [python]
severity: ERROR
message: Session cookie hardened flag explicitly disabled (Secure/HttpOnly = False) — session token exposed to theft.
metadata:
cwe: ["CWE-614: Sensitive Cookie in HTTPS Session Without 'Secure' Attribute"]
control: cra-ai-10
patterns:
- pattern-either:
- pattern: 'SESSION_COOKIE_SECURE = False'
- pattern: 'SESSION_COOKIE_HTTPONLY = False'
- id: cra-ai-10-express-cookie-insecure
languages: [javascript, typescript]
severity: ERROR
message: Express cookie set with secure/httpOnly = false — session token exposed to interception / XSS theft.
metadata:
cwe: ["CWE-614: Sensitive Cookie in HTTPS Session Without 'Secure' Attribute"]
control: cra-ai-10
patterns:
- pattern-either:
- pattern: '$RES.cookie($NAME, $VAL, {..., secure: false, ...})'
- pattern: '$RES.cookie($NAME, $VAL, {..., httpOnly: false, ...})'
# --- cra-ai-14: Speicher-Schutz / Data at Rest (weak cipher) -----------------
- id: cra-ai-14-python-weak-cipher
languages: [python]
severity: ERROR
message: Data-at-rest encrypted with a broken cipher/mode (ECB, DES, 3DES) — provides no real confidentiality.
metadata:
cwe: ["CWE-327: Use of a Broken or Risky Cryptographic Algorithm"]
control: cra-ai-14
patterns:
- pattern-either:
- pattern: 'AES.new($K, AES.MODE_ECB, ...)'
- pattern: 'DES.new(...)'
- pattern: 'DES3.new(...)'
- id: cra-ai-14-node-weak-cipher
languages: [javascript, typescript]
severity: ERROR
message: Data-at-rest encrypted with a broken cipher (DES / deprecated createCipher) — provides no real confidentiality.
metadata:
cwe: ["CWE-327: Use of a Broken or Risky Cryptographic Algorithm"]
control: cra-ai-14
patterns:
- pattern-either:
- pattern: 'crypto.createCipheriv("des-ecb", ...)'
- pattern: 'crypto.createCipheriv("des", ...)'
- pattern: 'crypto.createCipher(...)'
+20 -14
View File
@@ -63,12 +63,21 @@ impl ComplianceAgent {
let db = self.db_pool.for_tenant_id(tenant_id).await?;
let orchestrator =
PipelineOrchestrator::new(self.config.clone(), db, self.llm.clone(), self.http.clone());
orchestrator.run_target(repo_id, trigger).await
if self.config.unified_pipeline {
orchestrator.run_target(repo_id, trigger).await
} else {
orchestrator.run(repo_id, trigger).await
}
}
/// Alias for [`Self::run_scan`] — every scan runs the unified onboarded-target
/// pipeline. Kept as a distinct name for the `/targets/{id}/scan` endpoint's
/// intent.
/// Run a scan for an onboarded target through the unified pipeline,
/// unconditionally.
///
/// Unlike [`Self::run_scan`], this does *not* consult the
/// `unified_pipeline` transition flag: the caller (the `/targets/{id}/scan`
/// endpoint) operates on `onboarded_targets` by construction, so it must
/// always dispatch to `run_target` regardless of how the legacy paths
/// (scheduler, webhooks, `/repositories/{id}/scan`) are configured.
pub async fn run_target_scan(
&self,
tenant_id: &str,
@@ -91,19 +100,16 @@ impl ComplianceAgent {
head_sha: &str,
) -> Result<(), crate::error::AgentError> {
let db = self.db_pool.for_tenant_id(tenant_id).await?;
let oid = mongodb::bson::oid::ObjectId::parse_str(repo_id)
.map_err(|e| crate::error::AgentError::Other(e.to_string()))?;
let target = db
.onboarded_targets()
.find_one(mongodb::bson::doc! { "_id": oid })
let repo = db
.repositories()
.find_one(mongodb::bson::doc! {
"_id": mongodb::bson::oid::ObjectId::parse_str(repo_id)
.map_err(|e| crate::error::AgentError::Other(e.to_string()))?
})
.await?
.ok_or_else(|| {
crate::error::AgentError::Other(format!("Target {repo_id} not found"))
crate::error::AgentError::Other(format!("Repository {repo_id} not found"))
})?;
let code = target.code_artifact().ok_or_else(|| {
crate::error::AgentError::Other(format!("Target {repo_id} has no code artifact"))
})?;
let repo = crate::pipeline::repo_view::RepoView::from_target(&target, code);
let orchestrator =
PipelineOrchestrator::new(self.config.clone(), db, self.llm.clone(), self.http.clone());
+4 -12
View File
@@ -146,7 +146,7 @@ pub async fn build_embeddings(
let agent_clone = (*agent).clone();
tokio::spawn(async move {
let repo = match db
.onboarded_targets()
.repositories()
.find_one(doc! { "_id": mongodb::bson::oid::ObjectId::parse_str(&repo_id).ok() })
.await
{
@@ -194,22 +194,14 @@ pub async fn build_embeddings(
}
};
let code = match repo.code_artifact() {
Some(c) => c,
None => {
tracing::error!("Target {repo_id} has no code artifact for embedding build");
return;
}
};
let view = crate::pipeline::repo_view::RepoView::from_target(&repo, code);
let creds = crate::pipeline::git::RepoCredentials {
ssh_key_path: Some(agent_clone.config.ssh_key_path.clone()),
auth_token: view.auth_token.clone(),
auth_username: view.auth_username.clone(),
auth_token: repo.auth_token.clone(),
auth_username: repo.auth_username.clone(),
};
let git_ops =
crate::pipeline::git::GitOps::new(&agent_clone.config.git_clone_base_path, creds);
let repo_path = match git_ops.clone_or_fetch(&view.git_url, &view.name) {
let repo_path = match git_ops.clone_or_fetch(&repo.git_url, &repo.name) {
Ok(p) => p,
Err(e) => {
tracing::error!("Failed to clone repo for embedding build: {e}");
+5 -13
View File
@@ -255,7 +255,7 @@ pub async fn get_file_content(
// Look up the repository to get repo name
let repo = db
.onboarded_targets()
.repositories()
.find_one(doc! { "_id": mongodb::bson::oid::ObjectId::parse_str(&repo_id).ok() })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?
@@ -317,7 +317,7 @@ pub async fn trigger_build(
let agent_clone = (*agent).clone();
tokio::spawn(async move {
let repo = match db
.onboarded_targets()
.repositories()
.find_one(doc! { "_id": mongodb::bson::oid::ObjectId::parse_str(&repo_id).ok() })
.await
{
@@ -328,22 +328,14 @@ pub async fn trigger_build(
}
};
let code = match repo.code_artifact() {
Some(c) => c,
None => {
tracing::error!("Target {repo_id} has no code artifact for graph build");
return;
}
};
let view = crate::pipeline::repo_view::RepoView::from_target(&repo, code);
let creds = crate::pipeline::git::RepoCredentials {
ssh_key_path: Some(agent_clone.config.ssh_key_path.clone()),
auth_token: view.auth_token.clone(),
auth_username: view.auth_username.clone(),
auth_token: repo.auth_token.clone(),
auth_username: repo.auth_username.clone(),
};
let git_ops =
crate::pipeline::git::GitOps::new(&agent_clone.config.git_clone_base_path, creds);
let repo_path = match git_ops.clone_or_fetch(&view.git_url, &view.name) {
let repo_path = match git_ops.clone_or_fetch(&repo.git_url, &repo.name) {
Ok(p) => p,
Err(e) => {
tracing::error!("Failed to clone repo for graph build: {e}");
+1 -13
View File
@@ -10,18 +10,6 @@ pub async fn health() -> Json<serde_json::Value> {
Json(serde_json::json!({ "status": "ok" }))
}
/// GET /api/v1/settings/ssh-public-key — the agent's SSH deploy public key,
/// for adding as a read-only deploy key on private git targets.
#[tracing::instrument(skip_all)]
pub async fn get_ssh_public_key(
axum::extract::Extension(agent): AgentExt,
) -> Result<Json<serde_json::Value>, axum::http::StatusCode> {
let public_path = format!("{}.pub", agent.config.ssh_key_path);
let public_key =
std::fs::read_to_string(&public_path).map_err(|_| axum::http::StatusCode::NOT_FOUND)?;
Ok(Json(serde_json::json!({ "public_key": public_key.trim() })))
}
#[tracing::instrument(skip_all)]
pub async fn stats_overview(
axum::extract::Extension(agent): AgentExt,
@@ -31,7 +19,7 @@ pub async fn stats_overview(
let db = &db;
let total_repositories = db
.onboarded_targets()
.repositories()
.count_documents(doc! {})
.await
.unwrap_or(0);
+2 -2
View File
@@ -10,17 +10,17 @@ pub mod issues;
pub mod mcp_tokens;
pub mod notifications;
pub mod onboarding;
pub mod oscal;
pub mod pentest_handlers;
pub use pentest_handlers as pentest;
pub mod repos;
pub mod sbom;
pub mod scans;
pub mod werkbank_jobs;
// Re-export all handler functions so routes.rs can use `handlers::function_name`
pub use dto::*;
pub use findings::*;
pub use health::*;
pub use issues::*;
pub use repos::*;
pub use sbom::*;
pub use scans::*;
+7 -228
View File
@@ -5,7 +5,7 @@
use std::collections::HashMap;
use std::sync::Arc;
use axum::extract::{Extension, Multipart, Path, Query};
use axum::extract::{Extension, Path, Query};
use axum::http::StatusCode;
use axum::Json;
use mongodb::bson::{doc, oid::ObjectId, to_bson};
@@ -75,9 +75,6 @@ pub struct UpdateTargetRequest {
pub scan_config: Option<TargetScanConfig>,
pub compliance_profile: Option<ComplianceProfile>,
pub scan_schedule: Option<String>,
/// Replace the target's artifacts wholesale (used by the dashboard editor).
#[serde(default)]
pub artifacts: Option<Vec<ArtifactInput>>,
}
/// One applicable-scan option, serialized for the wizard.
@@ -217,13 +214,6 @@ pub async fn update_target(
if let Some(ss) = req.scan_schedule {
set.insert("scan_schedule", ss);
}
if let Some(arts) = req.artifacts {
let built: Vec<Artifact> = arts.iter().map(ArtifactInput::build).collect();
set.insert(
"artifacts",
to_bson(&built).map_err(|_| StatusCode::BAD_REQUEST)?,
);
}
db.onboarded_targets()
.update_one(doc! { "_id": oid }, doc! { "$set": set })
@@ -246,116 +236,15 @@ pub async fn delete_target(
.delete_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
// Cascade all data keyed by repo_id == target id (best-effort).
let db = &db;
let _ = db.findings().delete_many(doc! { "repo_id": &id }).await;
let _ = db.sbom_entries().delete_many(doc! { "repo_id": &id }).await;
let _ = db.scan_runs().delete_many(doc! { "repo_id": &id }).await;
let _ = db.cve_alerts().delete_many(doc! { "repo_id": &id }).await;
let _ = db
.tracker_issues()
.delete_many(doc! { "repo_id": &id })
.await;
let _ = db.graph_nodes().delete_many(doc! { "repo_id": &id }).await;
let _ = db.graph_edges().delete_many(doc! { "repo_id": &id }).await;
let _ = db.graph_builds().delete_many(doc! { "repo_id": &id }).await;
let _ = db
.impact_analyses()
.delete_many(doc! { "repo_id": &id })
.await;
let _ = db
.code_embeddings()
.delete_many(doc! { "repo_id": &id })
.await;
let _ = db
.embedding_builds()
.delete_many(doc! { "repo_id": &id })
.await;
// DAST targets linked to this target, and all their downstream data.
if let Ok(mut cursor) = db.dast_targets().find(doc! { "repo_id": &id }).await {
use futures_util::StreamExt;
while let Some(Ok(dt)) = cursor.next().await {
let dast_target_id = dt.id.map(|oid| oid.to_hex()).unwrap_or_default();
if !dast_target_id.is_empty() {
cascade_delete_dast_target(db, &dast_target_id).await;
}
}
}
// Pentest sessions linked directly to this target (not via a DAST target).
if let Ok(mut cursor) = db.pentest_sessions().find(doc! { "repo_id": &id }).await {
use futures_util::StreamExt;
while let Some(Ok(session)) = cursor.next().await {
let session_id = session.id.map(|oid| oid.to_hex()).unwrap_or_default();
if !session_id.is_empty() {
let _ = db
.attack_chain_nodes()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.pentest_messages()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.dast_findings()
.delete_many(doc! { "session_id": &session_id })
.await;
}
}
}
let _ = db
.pentest_sessions()
.delete_many(doc! { "repo_id": &id })
.await;
// Cascade the collections keyed by repo_id == target id (best-effort).
let by_repo = doc! { "repo_id": &id };
let _ = db.findings().delete_many(by_repo.clone()).await;
let _ = db.scan_runs().delete_many(by_repo.clone()).await;
let _ = db.sbom_entries().delete_many(by_repo.clone()).await;
let _ = db.cve_alerts().delete_many(by_repo).await;
Ok(Json(serde_json::json!({ "status": "deleted" })))
}
/// Delete a DAST target and everything downstream of it (pentest sessions +
/// their attack chains / messages / findings, DAST scan runs + findings).
async fn cascade_delete_dast_target(db: &crate::database::Database, target_id: &str) {
use futures_util::StreamExt;
if let Ok(mut cursor) = db
.pentest_sessions()
.find(doc! { "target_id": target_id })
.await
{
while let Some(Ok(session)) = cursor.next().await {
let session_id = session.id.map(|oid| oid.to_hex()).unwrap_or_default();
if !session_id.is_empty() {
let _ = db
.attack_chain_nodes()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.pentest_messages()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.dast_findings()
.delete_many(doc! { "session_id": &session_id })
.await;
}
}
}
let _ = db
.pentest_sessions()
.delete_many(doc! { "target_id": target_id })
.await;
let _ = db
.dast_findings()
.delete_many(doc! { "target_id": target_id })
.await;
let _ = db
.dast_scan_runs()
.delete_many(doc! { "target_id": target_id })
.await;
if let Ok(oid) = mongodb::bson::oid::ObjectId::parse_str(target_id) {
let _ = db.dast_targets().delete_one(doc! { "_id": oid }).await;
}
}
/// POST /api/v1/targets/{id}/artifacts — attach an artifact (by reference).
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn add_artifact(
@@ -377,116 +266,6 @@ pub async fn add_artifact(
get_target(Extension(agent), tenant, Path(id)).await
}
/// POST /api/v1/targets/{id}/artifacts/upload — attach an artifact by uploading
/// its file (PLC project, firmware image, source archive, mobile package). The
/// bytes are written to the artifact blob store and referenced by `stored_path`,
/// so ingest resolves them locally (no URL fetch).
///
/// Multipart fields: `file` (required), `kind` (required, snake_case
/// `ArtifactKind`), `plc_format` (optional, for PLC projects).
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn upload_artifact(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
mut multipart: Multipart,
) -> Result<Json<ApiResponse<OnboardedTarget>>, StatusCode> {
let oid = parse_oid(&id)?;
let db = tenant_db(&agent, &tenant).await?;
if db
.onboarded_targets()
.find_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?
.is_none()
{
return Err(StatusCode::NOT_FOUND);
}
let mut kind: Option<ArtifactKind> = None;
let mut plc_format: Option<PlcFormat> = None;
let mut filename = String::from("upload.bin");
let mut bytes: Option<axum::body::Bytes> = None;
while let Some(field) = multipart
.next_field()
.await
.map_err(|_| StatusCode::BAD_REQUEST)?
{
match field.name().unwrap_or("") {
"kind" => {
let v = field.text().await.map_err(|_| StatusCode::BAD_REQUEST)?;
kind = parse_enum(&v);
}
"plc_format" => {
let v = field.text().await.map_err(|_| StatusCode::BAD_REQUEST)?;
plc_format = parse_enum(&v);
}
"file" => {
if let Some(fname) = field.file_name() {
filename = fname.to_string();
}
bytes = Some(field.bytes().await.map_err(|_| StatusCode::BAD_REQUEST)?);
}
_ => {}
}
}
let (Some(kind), Some(bytes)) = (kind, bytes) else {
return Err(StatusCode::BAD_REQUEST);
};
// Store the uploaded bytes under the artifact blob store.
let safe_name: String = filename
.chars()
.map(|c| {
if c.is_ascii_alphanumeric() || matches!(c, '.' | '-' | '_') {
c
} else {
'_'
}
})
.collect();
let dir = std::path::Path::new(&agent.config.artifact_store_base_path)
.join("uploads")
.join(&id);
std::fs::create_dir_all(&dir).map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
let dest = dir.join(format!("{}_{safe_name}", uuid::Uuid::new_v4()));
std::fs::write(&dest, bytes.as_ref()).map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
// Build the artifact for this kind, referencing the stored file.
let mut artifact = match kind {
ArtifactKind::PlcProject => Artifact::plc_project(
filename.clone(),
plc_format.unwrap_or(PlcFormat::PlcopenXml),
),
ArtifactKind::FirmwareImage => Artifact::firmware_image(filename.clone()),
ArtifactKind::SourceArchive => Artifact::source_archive(filename.clone()),
ArtifactKind::MobilePackage => Artifact::mobile_package(filename.clone()),
// Non-file kinds (git repo, live URL, container ref, text) use the JSON
// add-artifact endpoint, not upload.
_ => return Err(StatusCode::BAD_REQUEST),
};
artifact.stored_path = Some(dest.to_string_lossy().to_string());
artifact.size_bytes = Some(bytes.len() as u64);
let artifact_bson = to_bson(&artifact).map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
db.onboarded_targets()
.update_one(
doc! { "_id": oid },
doc! { "$push": { "artifacts": artifact_bson }, "$set": { "updated_at": mongodb::bson::DateTime::now() } },
)
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
get_target(Extension(agent), tenant, Path(id)).await
}
/// Deserialize a snake_case enum value from a plain string.
fn parse_enum<T: for<'de> Deserialize<'de>>(s: &str) -> Option<T> {
serde_json::from_value(serde_json::Value::String(s.to_string())).ok()
}
/// GET /api/v1/targets/{id}/applicable-scans — the scan-applicability matrix.
#[tracing::instrument(skip_all, fields(target_id = %id))]
pub async fn applicable_scans_for_target(
@@ -1,48 +0,0 @@
//! OSCAL assessment endpoint.
//!
//! Returns a standard OSCAL assessment-results document for a target's findings,
//! driven by each finding's stamped `control_refs` (from the scan's control-triage
//! stage): mapped findings target their controls, unmapped findings are reported
//! as-is. See `compliance_core::models::oscal_assessment`.
use axum::extract::Extension;
use axum::http::StatusCode;
use axum::response::{IntoResponse, Response};
use axum::Json;
use mongodb::bson::doc;
use serde::Deserialize;
use compliance_core::models::oscal_assessment::assess;
use compliance_core::models::Finding;
use compliance_core::tenant_ctx::TenantCtx;
use super::dto::{collect_cursor_async, tenant_db, AgentExt};
#[derive(Debug, Deserialize)]
pub struct AssessRequest {
/// The target / repo id whose findings are assessed.
pub target_id: String,
}
/// `POST /api/v1/oscal/assess` — OSCAL assessment-results for a target's findings.
pub async fn assess_target(
Extension(agent): AgentExt,
tenant: TenantCtx,
Json(req): Json<AssessRequest>,
) -> Response {
let db = match tenant_db(&agent, &tenant).await {
Ok(db) => db,
Err(code) => return code.into_response(),
};
let findings: Vec<Finding> = match db.findings().find(doc! { "repo_id": &req.target_id }).await
{
Ok(cursor) => collect_cursor_async(cursor).await,
Err(e) => {
tracing::warn!(error = %e, "failed to load findings for OSCAL assessment");
return StatusCode::INTERNAL_SERVER_ERROR.into_response();
}
};
Json(assess(&findings, chrono::Utc::now())).into_response()
}
@@ -113,14 +113,14 @@ pub async fn create_session(
session.config = Some(config.clone());
session.repo_id = target.repo_id.clone();
// Resolve repo_id (target id) from git_repo_url if provided
// Resolve repo_id from git_repo_url if provided
if let Some(ref git_url) = config.git_repo_url {
if let Ok(Some(target)) = db
.onboarded_targets()
.find_one(doc! { "artifacts.source_ref": git_url })
if let Ok(Some(repo)) = db
.repositories()
.find_one(doc! { "git_url": git_url })
.await
{
session.repo_id = target.id.map(|oid| oid.to_hex());
session.repo_id = repo.id.map(|oid| oid.to_hex());
}
}
@@ -380,20 +380,17 @@ pub async fn lookup_repo(
) -> Result<Json<ApiResponse<serde_json::Value>>, StatusCode> {
let db = tenant_db(&agent, &tenant).await?;
let repo = db
.onboarded_targets()
.find_one(doc! { "artifacts.source_ref": &params.url })
.repositories()
.find_one(doc! { "git_url": &params.url })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
let data = match repo {
Some(r) => {
let git = r.code_artifact().and_then(|c| c.git.as_ref());
serde_json::json!({
"name": r.name,
"default_branch": git.map(|g| g.default_branch.clone()),
"last_scanned_commit": git.and_then(|g| g.last_scanned_commit.clone()),
})
}
Some(r) => serde_json::json!({
"name": r.name,
"default_branch": r.default_branch,
"last_scanned_commit": r.last_scanned_commit,
}),
None => serde_json::Value::Null,
};
+339
View File
@@ -0,0 +1,339 @@
use axum::extract::{Extension, Path, Query};
use axum::http::StatusCode;
use axum::Json;
use mongodb::bson::doc;
use super::dto::*;
use compliance_core::models::*;
use compliance_core::tenant_ctx::TenantCtx;
#[tracing::instrument(skip_all)]
pub async fn list_repositories(
Extension(agent): AgentExt,
tenant: TenantCtx,
Query(params): Query<PaginationParams>,
) -> ApiResult<Vec<TrackedRepository>> {
let db = tenant_db(&agent, &tenant).await?;
let db = &db;
let skip = (params.page.saturating_sub(1)) * params.limit as u64;
let total = db
.repositories()
.count_documents(doc! {})
.await
.unwrap_or(0);
let repos = match db
.repositories()
.find(doc! {})
.skip(skip)
.limit(params.limit)
.await
{
Ok(cursor) => collect_cursor_async(cursor).await,
Err(e) => {
tracing::warn!("Failed to fetch repositories: {e}");
Vec::new()
}
};
Ok(Json(ApiResponse {
data: repos,
total: Some(total),
page: Some(params.page),
}))
}
#[tracing::instrument(skip_all)]
pub async fn add_repository(
Extension(agent): AgentExt,
tenant: TenantCtx,
Json(req): Json<AddRepositoryRequest>,
) -> Result<Json<ApiResponse<TrackedRepository>>, (StatusCode, String)> {
// Validate repository access before saving
let creds = crate::pipeline::git::RepoCredentials {
ssh_key_path: Some(agent.config.ssh_key_path.clone()),
auth_token: req.auth_token.clone(),
auth_username: req.auth_username.clone(),
};
if let Err(e) = crate::pipeline::git::GitOps::test_access(&req.git_url, &creds) {
return Err((
StatusCode::BAD_REQUEST,
format!("Cannot access repository: {e}"),
));
}
let mut repo = TrackedRepository::new(req.name, req.git_url);
repo.default_branch = req.default_branch;
repo.auth_token = req.auth_token;
repo.auth_username = req.auth_username;
repo.tracker_type = req.tracker_type;
repo.tracker_owner = req.tracker_owner;
repo.tracker_repo = req.tracker_repo;
repo.tracker_token = req.tracker_token;
repo.scan_schedule = req.scan_schedule;
let db = tenant_db(&agent, &tenant)
.await
.map_err(|s| (s, "failed to acquire tenant database".to_string()))?;
db.repositories().insert_one(&repo).await.map_err(|_| {
(
StatusCode::CONFLICT,
"Repository already exists".to_string(),
)
})?;
Ok(Json(ApiResponse {
data: repo,
total: None,
page: None,
}))
}
#[tracing::instrument(skip_all, fields(repo_id = %id))]
pub async fn update_repository(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
Json(req): Json<UpdateRepositoryRequest>,
) -> Result<Json<serde_json::Value>, StatusCode> {
let oid = mongodb::bson::oid::ObjectId::parse_str(&id).map_err(|_| StatusCode::BAD_REQUEST)?;
let db = tenant_db(&agent, &tenant).await?;
let mut set_doc = doc! { "updated_at": mongodb::bson::DateTime::now() };
if let Some(name) = &req.name {
set_doc.insert("name", name);
}
if let Some(branch) = &req.default_branch {
set_doc.insert("default_branch", branch);
}
if let Some(token) = &req.auth_token {
set_doc.insert("auth_token", token);
}
if let Some(username) = &req.auth_username {
set_doc.insert("auth_username", username);
}
if let Some(tracker_type) = &req.tracker_type {
set_doc.insert("tracker_type", tracker_type.to_string());
}
if let Some(owner) = &req.tracker_owner {
set_doc.insert("tracker_owner", owner);
}
if let Some(repo) = &req.tracker_repo {
set_doc.insert("tracker_repo", repo);
}
if let Some(token) = &req.tracker_token {
set_doc.insert("tracker_token", token);
}
if let Some(schedule) = &req.scan_schedule {
set_doc.insert("scan_schedule", schedule);
}
let result = db
.repositories()
.update_one(doc! { "_id": oid }, doc! { "$set": set_doc })
.await
.map_err(|e| {
tracing::warn!("Failed to update repository: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
if result.matched_count == 0 {
return Err(StatusCode::NOT_FOUND);
}
Ok(Json(serde_json::json!({ "status": "updated" })))
}
#[tracing::instrument(skip_all)]
pub async fn get_ssh_public_key(
Extension(agent): AgentExt,
) -> Result<Json<serde_json::Value>, StatusCode> {
let public_path = format!("{}.pub", agent.config.ssh_key_path);
let public_key = std::fs::read_to_string(&public_path).map_err(|_| StatusCode::NOT_FOUND)?;
Ok(Json(serde_json::json!({ "public_key": public_key.trim() })))
}
#[tracing::instrument(skip_all, fields(repo_id = %id))]
pub async fn trigger_scan(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<serde_json::Value>, StatusCode> {
let agent_clone = (*agent).clone();
let tenant_id = tenant.0.tenant_id.clone();
tokio::spawn(async move {
if let Err(e) = agent_clone
.run_scan(&tenant_id, &id, ScanTrigger::Manual)
.await
{
tracing::error!("Manual scan failed for {id}: {e}");
}
});
Ok(Json(serde_json::json!({ "status": "scan_triggered" })))
}
/// Return the webhook secret for a repository (used by dashboard to display it)
pub async fn get_webhook_config(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<serde_json::Value>, StatusCode> {
let oid = mongodb::bson::oid::ObjectId::parse_str(&id).map_err(|_| StatusCode::BAD_REQUEST)?;
let db = tenant_db(&agent, &tenant).await?;
let repo = db
.repositories()
.find_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?
.ok_or(StatusCode::NOT_FOUND)?;
let tracker_type = repo
.tracker_type
.as_ref()
.map(|t| t.to_string())
.unwrap_or_else(|| "gitea".to_string());
Ok(Json(serde_json::json!({
"webhook_secret": repo.webhook_secret,
"tracker_type": tracker_type,
})))
}
#[tracing::instrument(skip_all, fields(repo_id = %id))]
pub async fn delete_repository(
Extension(agent): AgentExt,
tenant: TenantCtx,
Path(id): Path<String>,
) -> Result<Json<serde_json::Value>, StatusCode> {
let oid = mongodb::bson::oid::ObjectId::parse_str(&id).map_err(|_| StatusCode::BAD_REQUEST)?;
let db = tenant_db(&agent, &tenant).await?;
let db = &db;
// Delete the repository
let result = db
.repositories()
.delete_one(doc! { "_id": oid })
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
if result.deleted_count == 0 {
return Err(StatusCode::NOT_FOUND);
}
// Cascade delete all related data
let _ = db.findings().delete_many(doc! { "repo_id": &id }).await;
let _ = db.sbom_entries().delete_many(doc! { "repo_id": &id }).await;
let _ = db.scan_runs().delete_many(doc! { "repo_id": &id }).await;
let _ = db.cve_alerts().delete_many(doc! { "repo_id": &id }).await;
let _ = db
.tracker_issues()
.delete_many(doc! { "repo_id": &id })
.await;
let _ = db.graph_nodes().delete_many(doc! { "repo_id": &id }).await;
let _ = db.graph_edges().delete_many(doc! { "repo_id": &id }).await;
let _ = db.graph_builds().delete_many(doc! { "repo_id": &id }).await;
let _ = db
.impact_analyses()
.delete_many(doc! { "repo_id": &id })
.await;
let _ = db
.code_embeddings()
.delete_many(doc! { "repo_id": &id })
.await;
let _ = db
.embedding_builds()
.delete_many(doc! { "repo_id": &id })
.await;
// Cascade delete DAST targets linked to this repo, and all their downstream data
// (scan runs, findings, pentest sessions, attack chains, messages)
if let Ok(mut cursor) = db.dast_targets().find(doc! { "repo_id": &id }).await {
use futures_util::StreamExt;
while let Some(Ok(target)) = cursor.next().await {
let target_id = target.id.map(|oid| oid.to_hex()).unwrap_or_default();
if !target_id.is_empty() {
cascade_delete_dast_target(db, &target_id).await;
}
}
}
// Also delete pentest sessions linked directly to this repo (not via target)
if let Ok(mut cursor) = db.pentest_sessions().find(doc! { "repo_id": &id }).await {
use futures_util::StreamExt;
while let Some(Ok(session)) = cursor.next().await {
let session_id = session.id.map(|oid| oid.to_hex()).unwrap_or_default();
if !session_id.is_empty() {
let _ = db
.attack_chain_nodes()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.pentest_messages()
.delete_many(doc! { "session_id": &session_id })
.await;
// Delete DAST findings produced by this session
let _ = db
.dast_findings()
.delete_many(doc! { "session_id": &session_id })
.await;
}
}
}
let _ = db
.pentest_sessions()
.delete_many(doc! { "repo_id": &id })
.await;
Ok(Json(serde_json::json!({ "status": "deleted" })))
}
/// Cascade-delete a DAST target and all its downstream data.
async fn cascade_delete_dast_target(db: &crate::database::Database, target_id: &str) {
// Delete pentest sessions for this target (and their attack chains + messages)
if let Ok(mut cursor) = db
.pentest_sessions()
.find(doc! { "target_id": target_id })
.await
{
use futures_util::StreamExt;
while let Some(Ok(session)) = cursor.next().await {
let session_id = session.id.map(|oid| oid.to_hex()).unwrap_or_default();
if !session_id.is_empty() {
let _ = db
.attack_chain_nodes()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.pentest_messages()
.delete_many(doc! { "session_id": &session_id })
.await;
let _ = db
.dast_findings()
.delete_many(doc! { "session_id": &session_id })
.await;
}
}
}
let _ = db
.pentest_sessions()
.delete_many(doc! { "target_id": target_id })
.await;
// Delete DAST scan runs and their findings
let _ = db
.dast_findings()
.delete_many(doc! { "target_id": target_id })
.await;
let _ = db
.dast_scan_runs()
.delete_many(doc! { "target_id": target_id })
.await;
// Delete the target itself
if let Ok(oid) = mongodb::bson::oid::ObjectId::parse_str(target_id) {
let _ = db.dast_targets().delete_one(doc! { "_id": oid }).await;
}
}
+1 -1
View File
@@ -282,7 +282,7 @@ pub async fn license_summary(
}
})
.collect();
summaries.sort_by_key(|s| std::cmp::Reverse(s.count));
summaries.sort_by(|a, b| b.count.cmp(&a.count));
Ok(Json(ApiResponse {
data: summaries,
@@ -1,289 +0,0 @@
//! Werkbank runner endpoints (`/api/v1/werkbank/jobs/*`).
//!
//! The pull API a Werkbank runner talks to: lease a job, heartbeat while it runs,
//! and post the result back. Machine auth is a **static bearer token**
//! (`WERKBANK_RUNNER_TOKEN`) — not a Keycloak JWT, because a runner acts across
//! tenants (each request names its `tenant`). Routes are only mounted when the
//! token is configured; with none set they don't exist (404).
//!
//! On completion the runner's findings are persisted against the job's target,
//! so a job run by a remote runner lands the same findings an in-process run
//! would (WB-05, the control-plane cut-over).
use axum::extract::{Extension, Path, Request};
use axum::http::{header, StatusCode};
use axum::middleware::Next;
use axum::response::{IntoResponse, Response};
use axum::Json;
use mongodb::bson::{doc, oid::ObjectId};
use secrecy::ExposeSecret;
use serde::{Deserialize, Serialize};
use std::time::Duration;
use compliance_core::models::werkbank::{
CompleteRequest, CompleteResponse, HeartbeatRequest, InputRef, Job, JobResult, LeaseRequest,
};
use compliance_core::models::ArtifactKind;
use super::dto::AgentExt;
use crate::database::Database;
use crate::werkbank::JobQueue;
/// Gate the runner endpoints behind the static runner bearer token.
pub async fn require_runner_token(
Extension(agent): AgentExt,
request: Request,
next: Next,
) -> Response {
let Some(expected) = agent.config.werkbank_runner_token.as_ref() else {
return (StatusCode::NOT_FOUND, "werkbank runner API disabled").into_response();
};
let presented = request
.headers()
.get(header::AUTHORIZATION)
.and_then(|v| v.to_str().ok())
.and_then(|s| s.strip_prefix("Bearer "))
.map(str::trim)
.filter(|s| !s.is_empty());
let Some(presented) = presented else {
return (StatusCode::UNAUTHORIZED, "Missing bearer token").into_response();
};
if !constant_time_eq(presented, expected.expose_secret()) {
return (StatusCode::UNAUTHORIZED, "Invalid runner token").into_response();
}
next.run(request).await
}
/// `POST /api/v1/werkbank/jobs/lease` — lease the oldest runnable job, or `204`.
#[tracing::instrument(skip_all, fields(tenant = %req.tenant, runner = %req.runner_id))]
pub async fn lease(
Extension(agent): AgentExt,
Json(req): Json<LeaseRequest>,
) -> Result<Response, StatusCode> {
let queue = JobQueue::new(&tenant_db(&agent, &req.tenant).await?);
let leased = queue
.lease(
&req.runner_id,
req.executor,
&req.labels,
Duration::from_secs(req.lease_ttl_secs),
chrono::Utc::now(),
)
.await
.map_err(internal)?;
Ok(match leased {
Some(job) => Json(job).into_response(),
None => StatusCode::NO_CONTENT.into_response(),
})
}
/// `POST /api/v1/werkbank/jobs/heartbeat` — extend the lease; `409` if it's lost.
#[tracing::instrument(skip_all, fields(tenant = %req.tenant, job = %req.job_id))]
pub async fn heartbeat(
Extension(agent): AgentExt,
Json(req): Json<HeartbeatRequest>,
) -> Result<Response, StatusCode> {
let queue = JobQueue::new(&tenant_db(&agent, &req.tenant).await?);
let ack = queue
.heartbeat(
&req.job_id,
&req.lease_token,
Duration::from_secs(req.lease_ttl_secs),
chrono::Utc::now(),
)
.await
.map_err(internal)?;
Ok(match ack {
Some(ack) => Json(ack).into_response(),
// Lease lost — the runner should abandon the job.
None => StatusCode::CONFLICT.into_response(),
})
}
/// `POST /api/v1/werkbank/jobs/complete` — record the result and persist findings.
#[tracing::instrument(skip_all, fields(tenant = %req.tenant, job = %req.job_id))]
pub async fn complete(
Extension(agent): AgentExt,
Json(req): Json<CompleteRequest>,
) -> Result<Json<CompleteResponse>, StatusCode> {
let db = tenant_db(&agent, &req.tenant).await?;
let queue = JobQueue::new(&db);
let now = chrono::Utc::now();
let recorded = queue
.complete(&req.job_id, &req.lease_token, &req.result, now)
.await
.map_err(internal)?;
// Only persist findings for the run that actually recorded the result, so a
// duplicate/late completion can't double-insert.
if recorded {
if let Some(record) = queue.get(&req.job_id).await.map_err(internal)? {
persist_findings(&db, &record.job.target_id, &req.result).await;
}
}
Ok(Json(CompleteResponse { recorded }))
}
/// `GET /api/v1/werkbank/artifacts/{hash}` — serve a content-addressed blob (the
/// program a runner needs to load). The hash is validated against traversal by
/// [`crate::ingest::blob::read_blob`]; a runner fetches this for a job's `blob`
/// input.
#[tracing::instrument(skip_all, fields(hash = %hash))]
pub async fn serve_artifact(
Extension(agent): AgentExt,
Path(hash): Path<String>,
) -> Result<Response, StatusCode> {
let base = std::path::Path::new(&agent.config.artifact_store_base_path);
match crate::ingest::blob::read_blob(base, &hash) {
Ok(bytes) => {
Ok(([(header::CONTENT_TYPE, "application/octet-stream")], bytes).into_response())
}
Err(_) => Err(StatusCode::NOT_FOUND),
}
}
/// Enqueue a `plc-provision` job for a target: extract its control-logic program,
/// stash it as a content-addressed blob (which the runner fetches via
/// [`serve_artifact`]), and queue the job. This is the control-plane "enqueue"
/// half of the loop — a runner then leases it, provisions, and posts results.
#[derive(Debug, Deserialize)]
pub struct EnqueueRequest {
/// The tenant whose queue to enqueue into.
pub tenant: String,
/// The onboarded target to test.
pub target_id: String,
}
/// The enqueued job's id.
#[derive(Debug, Serialize)]
pub struct EnqueueResponse {
/// The new job id.
pub job_id: String,
/// Whether this call inserted it (false = already queued).
pub enqueued: bool,
}
#[tracing::instrument(skip_all, fields(tenant = %req.tenant, target = %req.target_id))]
pub async fn enqueue(
Extension(agent): AgentExt,
Json(req): Json<EnqueueRequest>,
) -> Result<Json<EnqueueResponse>, StatusCode> {
let db = tenant_db(&agent, &req.tenant).await?;
let oid = ObjectId::parse_str(&req.target_id).map_err(|_| StatusCode::BAD_REQUEST)?;
let target = db
.onboarded_targets()
.find_one(doc! { "_id": oid })
.await
.map_err(internal)?
.ok_or(StatusCode::NOT_FOUND)?;
// Extract the control-logic program from the target's PLC-source artifacts
// (same selection as the in-process PLC scan).
let ctx = crate::ingest::IngestContext::from_config(&agent.config, &req.target_id);
let ingest_set = crate::ingest::ingest_all(&target, &ctx).map_err(internal)?;
let program = target
.artifacts
.iter()
.filter(|a| {
matches!(
a.kind,
ArtifactKind::PlcProject | ArtifactKind::GitRepo | ArtifactKind::SourceArchive
)
})
.find_map(|a| {
let path = ingest_set
.get(&a.id)
.and_then(|ia| ia.working_path.clone())?;
werkbank_exec::plc::extract_program(&path)
})
.ok_or(StatusCode::UNPROCESSABLE_ENTITY)?;
// Stash the program source so the runner can fetch it by hash.
let base = std::path::Path::new(&agent.config.artifact_store_base_path);
let hash =
crate::ingest::blob::store_bytes(base, program.source.as_bytes()).map_err(internal)?;
let job_id = format!("job_{}", uuid::Uuid::new_v4().simple());
let job = Job::plc_provision(
&job_id,
&req.tenant,
&req.target_id,
InputRef::blob(hash),
agent.config.plc_runtime.max_lifetime_secs,
);
let enqueued = JobQueue::new(&db)
.enqueue(job, chrono::Utc::now())
.await
.map_err(internal)?;
Ok(Json(EnqueueResponse { job_id, enqueued }))
}
/// Persist a job result's findings against its target: general findings
/// (dedup'd by fingerprint) and DAST findings. Best-effort — a persistence hiccup
/// is logged, not surfaced to the runner (its result is already recorded).
async fn persist_findings(db: &Database, target_id: &str, result: &JobResult) {
for finding in &result.findings {
let exists = db
.findings()
.find_one(doc! { "fingerprint": &finding.fingerprint })
.await
.ok()
.flatten()
.is_some();
if !exists {
if let Err(e) = db.findings().insert_one(finding).await {
tracing::warn!(target_id, error = %e, "werkbank: persist finding failed");
}
}
}
for finding in &result.dast_findings {
if let Err(e) = db.dast_findings().insert_one(finding).await {
tracing::warn!(target_id, error = %e, "werkbank: persist DAST finding failed");
}
}
tracing::info!(
target_id,
findings = result.findings.len(),
dast = result.dast_findings.len(),
"werkbank: persisted runner results"
);
}
/// Resolve the tenant-scoped database for a request.
async fn tenant_db(
agent: &crate::agent::ComplianceAgent,
tenant: &str,
) -> Result<Database, StatusCode> {
agent.db_pool.for_tenant_id(tenant).await.map_err(internal)
}
/// Map any internal error to a 500.
fn internal<E: std::fmt::Display>(e: E) -> StatusCode {
tracing::error!("werkbank endpoint error: {e}");
StatusCode::INTERNAL_SERVER_ERROR
}
/// Length-checked, constant-time-ish token comparison.
fn constant_time_eq(a: &str, b: &str) -> bool {
if a.len() != b.len() {
return false;
}
let mut diff = 0u8;
for (x, y) in a.bytes().zip(b.bytes()) {
diff |= x ^ y;
}
diff == 0
}
#[cfg(test)]
mod tests {
use super::constant_time_eq;
#[test]
fn token_compare() {
assert!(constant_time_eq("secret", "secret"));
assert!(!constant_time_eq("secret", "secrex"));
assert!(!constant_time_eq("secret", "secretx"));
assert!(!constant_time_eq("", "x"));
}
}
+14 -5
View File
@@ -6,12 +6,25 @@ use crate::api::handlers;
pub fn build_router() -> Router {
Router::new()
.route("/api/v1/health", get(handlers::health))
.route("/api/v1/oscal/assess", post(handlers::oscal::assess_target))
.route("/api/v1/stats/overview", get(handlers::stats_overview))
.route(
"/api/v1/settings/ssh-public-key",
get(handlers::get_ssh_public_key),
)
.route("/api/v1/repositories", get(handlers::list_repositories))
.route("/api/v1/repositories", post(handlers::add_repository))
.route(
"/api/v1/repositories/{id}/scan",
post(handlers::trigger_scan),
)
.route(
"/api/v1/repositories/{id}",
delete(handlers::delete_repository).patch(handlers::update_repository),
)
.route(
"/api/v1/repositories/{id}/webhook-config",
get(handlers::get_webhook_config),
)
// Unified onboarding targets (#131).
.route(
"/api/v1/targets",
@@ -27,10 +40,6 @@ pub fn build_router() -> Router {
"/api/v1/targets/{id}/artifacts",
post(handlers::onboarding::add_artifact),
)
.route(
"/api/v1/targets/{id}/artifacts/upload",
post(handlers::onboarding::upload_artifact),
)
.route(
"/api/v1/targets/{id}/applicable-scans",
get(handlers::onboarding::applicable_scans_for_target),
+2 -40
View File
@@ -1,10 +1,10 @@
use std::sync::Arc;
use axum::extract::{DefaultBodyLimit, Request};
use axum::extract::Request;
use axum::http::HeaderValue;
use axum::middleware::Next;
use axum::response::Response;
use axum::routing::{delete, get, post};
use axum::routing::{delete, get};
use axum::{middleware, Extension, Router};
use tokio::sync::RwLock;
use tower_http::cors::CorsLayer;
@@ -72,46 +72,8 @@ pub async fn start_api_server(agent: ComplianceAgent, port: u16) -> Result<(), A
Router::new()
};
// Werkbank runner API. Like admin, only mounted when its bearer token is
// configured; runners authenticate with WERKBANK_RUNNER_TOKEN (not a JWT).
let werkbank_router: Router = if agent.config.werkbank_runner_token.is_some() {
tracing::info!(
"Werkbank runner API enabled — /api/v1/werkbank/jobs/* behind WERKBANK_RUNNER_TOKEN"
);
Router::new()
.route(
"/api/v1/werkbank/jobs/lease",
post(handlers::werkbank_jobs::lease),
)
.route(
"/api/v1/werkbank/jobs/heartbeat",
post(handlers::werkbank_jobs::heartbeat),
)
.route(
"/api/v1/werkbank/jobs/complete",
post(handlers::werkbank_jobs::complete),
)
.route(
"/api/v1/werkbank/jobs/enqueue",
post(handlers::werkbank_jobs::enqueue),
)
.route(
"/api/v1/werkbank/artifacts/{hash}",
get(handlers::werkbank_jobs::serve_artifact),
)
.layer(middleware::from_fn(
handlers::werkbank_jobs::require_runner_token,
))
} else {
Router::new()
};
let mut app = routes::build_router()
.merge(admin_router)
.merge(werkbank_router)
// Allow large artifact uploads (PLC .projectarchive, firmware images,
// mobile packages) — axum's default request-body limit is only 2 MiB.
.layer(DefaultBodyLimit::max(512 * 1024 * 1024))
.layer(Extension(Arc::new(agent.clone())))
.layer(CorsLayer::permissive())
.layer(TraceLayer::new_for_http())
+6 -43
View File
@@ -1,4 +1,3 @@
use compliance_core::config::{BreakpilotConfig, PlcRuntimeConfig};
use compliance_core::AgentConfig;
use secrecy::SecretString;
@@ -48,6 +47,12 @@ pub fn load_config() -> Result<AgentConfig, AgentError> {
.unwrap_or_else(|| "/tmp/compliance-scanner/repos".to_string()),
artifact_store_base_path: env_var_opt("ARTIFACT_STORE_BASE_PATH")
.unwrap_or_else(|| "/data/compliance-scanner/artifacts".to_string()),
// Defaults ON: the unified onboarded-target pipeline is now the primary
// path (no legacy `repositories` data in production). Set
// `UNIFIED_PIPELINE=0` to fall back to the legacy repository pipeline.
unified_pipeline: env_var_opt("UNIFIED_PIPELINE")
.map(|v| v == "1" || v.eq_ignore_ascii_case("true"))
.unwrap_or(true),
ssh_key_path: env_var_opt("SSH_KEY_PATH")
.unwrap_or_else(|| "/data/compliance-scanner/ssh/id_ed25519".to_string()),
keycloak_url: env_var_opt("KEYCLOAK_URL"),
@@ -64,47 +69,5 @@ pub fn load_config() -> Result<AgentConfig, AgentError> {
pentest_imap_password: env_secret_opt("PENTEST_IMAP_PASSWORD"),
admin_api_token: env_secret_opt("ADMIN_API_TOKEN"),
tenant_registry_url: env_var_opt("TENANT_REGISTRY_URL"),
plc_runtime: load_plc_runtime_config(),
werkbank_runner_token: env_secret_opt("WERKBANK_RUNNER_TOKEN"),
breakpilot: load_breakpilot_config(),
})
}
/// Build the ephemeral soft-PLC provisioning config from the environment,
/// falling back to [`PlcRuntimeConfig::default`] for any unset knob. Disabled
/// unless `PLC_RUNTIME_ENABLED` is truthy — it requires Docker access.
fn load_plc_runtime_config() -> PlcRuntimeConfig {
let d = PlcRuntimeConfig::default();
PlcRuntimeConfig {
enabled: env_var_opt("PLC_RUNTIME_ENABLED")
.map(|v| v == "1" || v.eq_ignore_ascii_case("true"))
.unwrap_or(d.enabled),
image: env_var_opt("PLC_RUNTIME_IMAGE").unwrap_or(d.image),
network: env_var_opt("PLC_RUNTIME_NETWORK").unwrap_or(d.network),
memory: env_var_opt("PLC_RUNTIME_MEMORY").unwrap_or(d.memory),
cpus: env_var_opt("PLC_RUNTIME_CPUS").unwrap_or(d.cpus),
max_lifetime_secs: env_var_opt("PLC_RUNTIME_MAX_LIFETIME_SECS")
.and_then(|v| v.parse().ok())
.unwrap_or(d.max_lifetime_secs),
openplc_user: env_var_opt("PLC_RUNTIME_OPENPLC_USER").unwrap_or(d.openplc_user),
openplc_password: env_secret_opt("PLC_RUNTIME_OPENPLC_PASSWORD")
.unwrap_or(d.openplc_password),
}
}
/// Assemble the breakpilot OSCAL-catalog source from env, defaulting the snapshot
/// directory. A missing `BREAKPILOT_BASE_URL` leaves the controls provider off.
fn load_breakpilot_config() -> BreakpilotConfig {
let d = BreakpilotConfig::default();
BreakpilotConfig {
base_url: env_var_opt("BREAKPILOT_BASE_URL"),
token: env_secret_opt("BREAKPILOT_TOKEN"),
snapshot_dir: env_var_opt("BREAKPILOT_SNAPSHOT_DIR").unwrap_or(d.snapshot_dir),
semantic_mapping: env_var_opt("BREAKPILOT_SEMANTIC_MAPPING")
.map(|v| v == "1" || v.eq_ignore_ascii_case("true"))
.unwrap_or(d.semantic_mapping),
grounded_control_checks: env_var_opt("BREAKPILOT_GROUNDED_CHECKS")
.map(|v| v == "1" || v.eq_ignore_ascii_case("true"))
.unwrap_or(d.grounded_control_checks),
}
}
-114
View File
@@ -1,114 +0,0 @@
//! The grounded control checker: judge each candidate region for a control, then
//! keep only the verdicts that survive the grounding gate.
//!
//! Generic over [`ControlJudge`] so tests drive it with a deterministic stub —
//! the whole recognize → ground path is then exercised without an LLM. With the
//! real judge, determinism comes from temperature 0 plus the gate.
use compliance_core::control_check::{ground, CandidateRegion, ControlCheckSpec};
use compliance_core::models::Finding;
use super::judge::ControlJudge;
/// Runs a [`ControlJudge`] over candidate regions and grounds the results.
pub struct GroundedControlChecker<J> {
judge: J,
}
impl<J: ControlJudge> GroundedControlChecker<J> {
pub fn new(judge: J) -> Self {
Self { judge }
}
/// Judge every candidate region for `spec` and return the grounded findings.
/// A verdict that doesn't quote real code in its region is dropped by
/// [`ground`], so nothing fabricated reaches the caller.
pub async fn check(
&self,
spec: &ControlCheckSpec,
regions: &[CandidateRegion],
repo_id: &str,
) -> Vec<Finding> {
let mut findings = Vec::new();
for region in regions {
let verdict = self.judge.judge(spec, region).await;
if let Some(finding) = ground(spec, region, &verdict, repo_id) {
findings.push(finding);
}
}
findings
}
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::control_check::LlmVerdict;
use compliance_core::models::finding::Severity;
/// Deterministic stub: returns a fixed verdict for every region, so the
/// recognize → ground composition is tested without an LLM.
struct StubJudge {
verdict: LlmVerdict,
}
impl ControlJudge for StubJudge {
async fn judge(&self, _spec: &ControlCheckSpec, _region: &CandidateRegion) -> LlmVerdict {
self.verdict.clone()
}
}
fn spec() -> ControlCheckSpec {
ControlCheckSpec {
control_id: "cra-ai-8".into(),
title: "No default passwords".into(),
requirement: "No default credentials".into(),
default_cwe: Some("CWE-798".into()),
severity: Severity::High,
}
}
fn region(content: &str) -> CandidateRegion {
CandidateRegion {
file: "src/auth.py".into(),
start_line: 1,
content: content.into(),
}
}
#[tokio::test]
async fn keeps_grounded_and_drops_ungrounded() {
let checker = GroundedControlChecker::new(StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "PASSWORD = \"admin\"".into(),
cwe: None,
confidence: 0.9,
},
});
let regions = vec![
region("x = 1\nPASSWORD = \"admin\"\n"), // quotes real code → grounded
region("totally unrelated code\n"), // snippet absent → dropped
];
let findings = checker.check(&spec(), &regions, "repo").await;
assert_eq!(findings.len(), 1);
assert_eq!(findings[0].control_refs, vec!["cra-ai-8".to_string()]);
assert_eq!(findings[0].line_number, Some(2));
}
#[tokio::test]
async fn non_violation_yields_nothing() {
let checker = GroundedControlChecker::new(StubJudge {
verdict: LlmVerdict {
violates: false,
snippet: String::new(),
cwe: None,
confidence: 0.0,
},
});
let findings = checker
.check(&spec(), &[region("PASSWORD = \"admin\"\n")], "repo")
.await;
assert!(findings.is_empty());
}
}
-245
View File
@@ -1,245 +0,0 @@
//! In-memory embedding index over the control corpus, for region → control
//! retrieval.
//!
//! At master-control scale (~13.6k) findings can't be mapped by CWE (the master
//! controls carry none), so we map by *similarity*: embed each control's
//! requirement text once, then for a code region pull the top-K nearest controls
//! to hand to the grounded judge. This is the retrieval half of the semantic path.
use std::path::Path;
use serde::{Deserialize, Serialize};
use sha2::{Digest, Sha256};
use compliance_core::control_check::ControlCheckSpec;
use compliance_core::error::CoreError;
use crate::llm::LlmClient;
/// A control spec paired with its requirement-text embedding.
pub struct ControlIndex {
entries: Vec<(ControlCheckSpec, Vec<f64>)>,
}
/// On-disk form of the index: the corpus identity hash plus every spec+embedding.
/// The hash lets a later scan reuse the embeddings only if the corpus is unchanged.
#[derive(Serialize, Deserialize)]
struct PersistedIndex {
corpus_hash: String,
entries: Vec<PersistedEntry>,
}
#[derive(Serialize, Deserialize)]
struct PersistedEntry {
spec: ControlCheckSpec,
embedding: Vec<f64>,
}
/// Stable hash of the corpus identity (each control's id + requirement text, in
/// order). Same catalog → same hash → the cached embeddings are reused instead of
/// re-embedding the whole corpus.
fn corpus_hash(specs: &[ControlCheckSpec]) -> String {
let mut hasher = Sha256::new();
for s in specs {
hasher.update(s.control_id.as_bytes());
hasher.update([0u8]);
hasher.update(s.requirement.as_bytes());
hasher.update([0u8]);
}
format!("{:x}", hasher.finalize())
}
impl ControlIndex {
/// Build directly from precomputed embeddings (used by tests + callers that
/// already embedded the corpus).
pub fn from_embeddings(entries: Vec<(ControlCheckSpec, Vec<f64>)>) -> Self {
Self { entries }
}
/// Load the index from `cache_path` if it still matches the current corpus,
/// otherwise embed the corpus and persist it there. This turns the per-scan
/// re-embed of the whole (~13.6k) master-control corpus into a one-time cost
/// that survives across scans; the cache self-invalidates when the catalog
/// changes (its [`corpus_hash`] no longer matches).
pub async fn load_or_build(
llm: &LlmClient,
specs: Vec<ControlCheckSpec>,
cache_path: &Path,
) -> Result<Self, CoreError> {
let hash = corpus_hash(&specs);
if let Some(index) = Self::load_cache(cache_path, &hash).await {
tracing::debug!(
controls = index.len(),
"reusing cached control embedding index"
);
return Ok(index);
}
let index = Self::build(llm, specs).await?;
if let Err(e) = index.write_cache(cache_path, &hash).await {
tracing::warn!(error = %e, "failed to persist control embedding index");
}
Ok(index)
}
/// Read a persisted index, returning it only if its corpus hash matches.
async fn load_cache(path: &Path, hash: &str) -> Option<Self> {
let raw = tokio::fs::read(path).await.ok()?;
let persisted: PersistedIndex = serde_json::from_slice(&raw).ok()?;
if persisted.corpus_hash != hash {
return None;
}
Some(Self {
entries: persisted
.entries
.into_iter()
.map(|e| (e.spec, e.embedding))
.collect(),
})
}
/// Persist the index atomically (temp file + rename) keyed by corpus hash.
async fn write_cache(&self, path: &Path, hash: &str) -> Result<(), CoreError> {
if let Some(parent) = path.parent() {
tokio::fs::create_dir_all(parent).await?;
}
let persisted = PersistedIndex {
corpus_hash: hash.to_string(),
entries: self
.entries
.iter()
.map(|(spec, emb)| PersistedEntry {
spec: spec.clone(),
embedding: emb.clone(),
})
.collect(),
};
let raw = serde_json::to_vec(&persisted)?;
let tmp = path.with_extension("json.tmp");
tokio::fs::write(&tmp, &raw).await?;
tokio::fs::rename(&tmp, path).await?;
Ok(())
}
/// Build by embedding each control's requirement text.
pub async fn build(llm: &LlmClient, specs: Vec<ControlCheckSpec>) -> Result<Self, CoreError> {
if specs.is_empty() {
return Ok(Self {
entries: Vec::new(),
});
}
let texts: Vec<String> = specs.iter().map(|s| s.requirement.clone()).collect();
let embeddings = llm
.embed(texts)
.await
.map_err(|e| CoreError::Llm(e.to_string()))?;
Ok(Self {
entries: specs.into_iter().zip(embeddings).collect(),
})
}
pub fn len(&self) -> usize {
self.entries.len()
}
pub fn is_empty(&self) -> bool {
self.entries.is_empty()
}
/// The top-`k` control specs whose embedding is nearest (cosine) to `query`.
pub fn nearest(&self, query: &[f64], k: usize) -> Vec<ControlCheckSpec> {
let mut scored: Vec<(f64, &ControlCheckSpec)> = self
.entries
.iter()
.map(|(spec, emb)| (cosine(query, emb), spec))
.collect();
scored.sort_by(|a, b| b.0.total_cmp(&a.0));
scored.into_iter().take(k).map(|(_, s)| s.clone()).collect()
}
}
/// Cosine similarity; 0.0 for length-mismatched, empty, or zero vectors.
fn cosine(a: &[f64], b: &[f64]) -> f64 {
if a.len() != b.len() || a.is_empty() {
return 0.0;
}
let dot: f64 = a.iter().zip(b).map(|(x, y)| x * y).sum();
let na: f64 = a.iter().map(|x| x * x).sum();
let nb: f64 = b.iter().map(|x| x * x).sum();
if na == 0.0 || nb == 0.0 {
return 0.0;
}
dot / (na.sqrt() * nb.sqrt())
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::models::finding::Severity;
fn spec(id: &str) -> ControlCheckSpec {
ControlCheckSpec {
control_id: id.into(),
title: id.into(),
requirement: id.into(),
default_cwe: None,
severity: Severity::Medium,
}
}
#[test]
fn nearest_ranks_by_cosine() {
let index = ControlIndex::from_embeddings(vec![
(spec("a"), vec![1.0, 0.0]),
(spec("b"), vec![0.0, 1.0]),
(spec("c"), vec![0.7, 0.7]),
]);
let hits = index.nearest(&[0.9, 0.1], 2);
assert_eq!(hits.len(), 2);
assert_eq!(hits[0].control_id, "a"); // closest to [0.9,0.1]
}
#[test]
fn cosine_edges_are_zero() {
assert_eq!(cosine(&[1.0], &[1.0, 2.0]), 0.0); // length mismatch
assert_eq!(cosine(&[0.0, 0.0], &[1.0, 1.0]), 0.0); // zero vector
assert!((cosine(&[1.0, 0.0], &[1.0, 0.0]) - 1.0).abs() < 1e-9); // identical
}
#[test]
fn corpus_hash_is_stable_and_identity_sensitive() {
let a = corpus_hash(&[spec("x"), spec("y")]);
assert_eq!(a, corpus_hash(&[spec("x"), spec("y")])); // same corpus → same hash
assert_ne!(a, corpus_hash(&[spec("y"), spec("x")])); // reorder → different
assert_ne!(a, corpus_hash(&[spec("x")])); // fewer controls → different
}
#[tokio::test]
#[allow(clippy::unwrap_used)]
async fn cache_round_trips_and_misses_on_corpus_change() {
let dir = std::env::temp_dir().join(format!("cidx-{}", uuid::Uuid::new_v4()));
let path = dir.join("control-index.json");
let specs = [spec("a"), spec("b")];
let hash = corpus_hash(&specs);
let index = ControlIndex::from_embeddings(vec![
(spec("a"), vec![1.0, 0.0]),
(spec("b"), vec![0.0, 1.0]),
]);
index.write_cache(&path, &hash).await.unwrap();
// matching corpus hash → hit
let loaded = ControlIndex::load_cache(&path, &hash).await.unwrap();
assert_eq!(loaded.len(), 2);
assert_eq!(loaded.nearest(&[0.9, 0.1], 1)[0].control_id, "a");
// corpus changed → miss (forces a rebuild)
assert!(ControlIndex::load_cache(&path, "differenthash")
.await
.is_none());
// absent file → miss, not an error
assert!(
ControlIndex::load_cache(dir.join("nope.json").as_path(), &hash)
.await
.is_none()
);
let _ = std::fs::remove_dir_all(&dir);
}
}
-167
View File
@@ -1,167 +0,0 @@
//! The "recognize" stage: judge whether a code region violates a control.
//!
//! Behind the [`ControlJudge`] trait so the grounded checker can be driven by a
//! deterministic stub in tests. The real [`LlmControlJudge`] runs the model at
//! temperature 0 with a closed prompt — it must quote the offending code VERBATIM,
//! and everything it returns is then re-checked by the grounding gate
//! ([`compliance_core::control_check::ground`]). The judge is allowed to be
//! smart; it is never trusted.
use std::sync::Arc;
use serde::Deserialize;
use compliance_core::control_check::{CandidateRegion, ControlCheckSpec, LlmVerdict};
use crate::llm::LlmClient;
/// Prompt/logic version — part of the verdict cache key, bump on any change here.
pub const PROMPT_VERSION: &str = "control-judge-v1";
const SYSTEM_PROMPT: &str = "You are a precise security & compliance code auditor. \
You are given ONE compliance control (a requirement) and ONE code region. Decide \
ONLY whether the code region VIOLATES the control. Rules: (1) Judge only the code \
shown — never assume code that is not present. (2) If and only if it violates, copy \
the EXACT offending code VERBATIM into `snippet`, character-for-character from the \
region — do not paraphrase, reformat, or reconstruct it. (3) If it does not clearly \
violate, set violates=false and leave snippet empty. (4) Prefer false over guessing. \
Respond with STRICT JSON only, no prose: \
{\"violates\": bool, \"snippet\": \"<verbatim code or empty>\", \"cwe\": \"CWE-NNN or null\", \"confidence\": 0.0-1.0}";
/// Judges one (control, region). Async-in-trait so a stub can drive tests.
#[allow(async_fn_in_trait)]
pub trait ControlJudge: Send + Sync {
async fn judge(&self, spec: &ControlCheckSpec, region: &CandidateRegion) -> LlmVerdict;
}
/// The real judge: the LLM at temperature 0 with the closed, verbatim-snippet prompt.
pub struct LlmControlJudge {
llm: Arc<LlmClient>,
}
impl LlmControlJudge {
pub fn new(llm: Arc<LlmClient>) -> Self {
Self { llm }
}
}
impl ControlJudge for LlmControlJudge {
async fn judge(&self, spec: &ControlCheckSpec, region: &CandidateRegion) -> LlmVerdict {
let user = build_user_prompt(spec, region);
match self.llm.chat(SYSTEM_PROMPT, &user, Some(0.0)).await {
Ok(response) => parse_verdict(&response),
Err(e) => {
// Fail closed: a transient model error yields no finding, never a
// fabricated one.
tracing::warn!(control = %spec.control_id, error = %e, "control judge call failed");
no_violation()
}
}
}
}
fn build_user_prompt(spec: &ControlCheckSpec, region: &CandidateRegion) -> String {
format!(
"CONTROL {id}{title}\nRequirement: {req}\n\nCODE ({file}, first line = {line}):\n```\n{code}\n```\n\nReturn the JSON verdict.",
id = spec.control_id,
title = spec.title,
req = spec.requirement,
file = region.file,
line = region.start_line,
code = region.content,
)
}
#[derive(Debug, Default, Deserialize)]
struct RawVerdict {
#[serde(default)]
violates: bool,
#[serde(default)]
snippet: String,
#[serde(default)]
cwe: Option<String>,
#[serde(default)]
confidence: f64,
}
/// Parse the model's JSON verdict, tolerant of ```json fencing. Any parse failure
/// degrades to a non-violation (never a fabricated finding).
fn parse_verdict(response: &str) -> LlmVerdict {
let cleaned = response
.trim()
.trim_start_matches("```json")
.trim_start_matches("```")
.trim_end_matches("```")
.trim();
match serde_json::from_str::<RawVerdict>(cleaned) {
Ok(raw) => LlmVerdict {
violates: raw.violates,
snippet: raw.snippet,
cwe: raw.cwe.filter(|c| !c.trim().is_empty()),
confidence: raw.confidence,
},
Err(e) => {
tracing::debug!(error = %e, "failed to parse control verdict; treating as non-violation");
no_violation()
}
}
}
fn no_violation() -> LlmVerdict {
LlmVerdict {
violates: false,
snippet: String::new(),
cwe: None,
confidence: 0.0,
}
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::models::finding::Severity;
fn spec() -> ControlCheckSpec {
ControlCheckSpec {
control_id: "cra-ai-8".into(),
title: "No default passwords".into(),
requirement: "Products must not ship default credentials".into(),
default_cwe: Some("CWE-798".into()),
severity: Severity::High,
}
}
#[test]
fn parses_plain_and_fenced_json() {
let plain = r#"{"violates": true, "snippet": "PASSWORD = \"x\"", "cwe": "CWE-798", "confidence": 0.9}"#;
let v = parse_verdict(plain);
assert!(v.violates);
assert_eq!(v.snippet, "PASSWORD = \"x\"");
assert_eq!(v.cwe.as_deref(), Some("CWE-798"));
let fenced = "```json\n{\"violates\": false, \"snippet\": \"\", \"cwe\": null, \"confidence\": 0.1}\n```";
assert!(!parse_verdict(fenced).violates);
}
#[test]
fn garbage_and_empty_cwe_are_safe() {
assert!(!parse_verdict("not json at all").violates); // fail closed
let no_cwe =
parse_verdict(r#"{"violates": true, "snippet": "x", "cwe": " ", "confidence": 0.5}"#);
assert!(no_cwe.cwe.is_none()); // blank CWE normalised away
}
#[test]
fn user_prompt_carries_control_and_code() {
let region = CandidateRegion {
file: "src/auth.py".into(),
start_line: 10,
content: "PASSWORD = \"admin\"".into(),
};
let p = build_user_prompt(&spec(), &region);
assert!(p.contains("cra-ai-8"));
assert!(p.contains("Products must not ship default credentials"));
assert!(p.contains("PASSWORD = \"admin\""));
assert!(p.contains("src/auth.py"));
}
}
-23
View File
@@ -1,23 +0,0 @@
//! Controls corpus providers.
//!
//! Implementations of [`compliance_core::traits::ControlsProvider`] that supply
//! the control corpus the mapping engine assesses findings against. Currently:
//! [`OscalControlsProvider`], which pulls breakpilot-compliance's OSCAL catalog
//! and snapshots it locally.
mod checker;
mod index;
mod judge;
mod oscal_provider;
mod scan_triage;
mod semantic;
mod surface;
mod triage;
pub use checker::GroundedControlChecker;
pub use index::ControlIndex;
pub use judge::{ControlJudge, LlmControlJudge, PROMPT_VERSION};
pub use oscal_provider::OscalControlsProvider;
pub use scan_triage::{grounded_surface_findings, semantic_stamp_findings, triage_repo_findings};
pub use semantic::SemanticControlChecker;
pub use triage::{ControlTriage, TriageOutcome};
@@ -1,229 +0,0 @@
//! Pull + snapshot [`ControlsProvider`] backed by breakpilot-compliance's OSCAL
//! catalog export.
//!
//! Fetches `GET {base}/api/compliance/v1/oscal/catalog?framework=<fw>`, snapshots
//! the exact bytes to disk (so scans are deterministic and keep working offline /
//! on-prem), and maps the catalog into the corpus controls the mapping engine
//! consumes. The producer owns the catalog; we own the assessment — this is the
//! ingest half of the loop.
use std::path::PathBuf;
use secrecy::{ExposeSecret, SecretString};
use compliance_core::error::CoreError;
use compliance_core::models::onboarding::ComplianceFramework;
use compliance_core::models::oscal::OscalDocument;
use compliance_core::traits::{Control, ControlQuery, ControlsProvider};
/// A [`ControlsProvider`] that pulls the OSCAL catalog from breakpilot-compliance
/// and snapshots it locally for deterministic / offline reuse.
pub struct OscalControlsProvider {
http: reqwest::Client,
base_url: String,
token: Option<SecretString>,
snapshot_dir: PathBuf,
}
impl OscalControlsProvider {
/// Create a provider. `base_url` is the breakpilot-compliance root (e.g.
/// `http://backend-compliance:8002`); `snapshot_dir` is where catalog
/// snapshots are written so a later scan can reuse them without the network.
pub fn new(
http: reqwest::Client,
base_url: impl Into<String>,
token: Option<SecretString>,
snapshot_dir: impl Into<PathBuf>,
) -> Self {
Self {
http,
base_url: base_url.into(),
token,
snapshot_dir: snapshot_dir.into(),
}
}
fn catalog_url(&self, framework: &str) -> String {
format!(
"{}/api/compliance/v1/oscal/catalog?framework={framework}",
self.base_url.trim_end_matches('/')
)
}
fn snapshot_path(&self, framework: &str) -> PathBuf {
self.snapshot_dir
.join(format!("oscal-catalog-{framework}.json"))
}
/// Fetch the raw catalog bytes for a framework token over HTTP.
async fn fetch_raw(&self, framework: &str) -> Result<Vec<u8>, CoreError> {
let mut req = self.http.get(self.catalog_url(framework));
if let Some(token) = &self.token {
req = req.bearer_auth(token.expose_secret());
}
let resp = req
.send()
.await
.map_err(|e| CoreError::Http(e.to_string()))?;
if !resp.status().is_success() {
return Err(CoreError::Http(format!(
"catalog fetch for {framework} returned HTTP {}",
resp.status()
)));
}
resp.bytes()
.await
.map(|b| b.to_vec())
.map_err(|e| CoreError::Http(e.to_string()))
}
/// Write a catalog snapshot atomically (temp file + rename).
async fn write_snapshot(&self, framework: &str, raw: &[u8]) -> Result<(), CoreError> {
tokio::fs::create_dir_all(&self.snapshot_dir).await?;
let path = self.snapshot_path(framework);
let tmp = path.with_extension("json.tmp");
tokio::fs::write(&tmp, raw).await?;
tokio::fs::rename(&tmp, &path).await?;
Ok(())
}
/// Read a previously written snapshot, if one exists.
async fn read_snapshot(&self, framework: &str) -> Result<Option<OscalDocument>, CoreError> {
match tokio::fs::read(self.snapshot_path(framework)).await {
Ok(raw) => Ok(Some(serde_json::from_slice(&raw)?)),
Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(None),
Err(e) => Err(e.into()),
}
}
/// Load the catalog for a framework token: fetch fresh + snapshot the exact
/// bytes; on network failure, fall back to the last snapshot so scans run.
async fn load_token(&self, framework: &str) -> Result<OscalDocument, CoreError> {
match self.fetch_raw(framework).await {
Ok(raw) => {
let doc: OscalDocument = serde_json::from_slice(&raw)?;
if let Err(e) = self.write_snapshot(framework, &raw).await {
tracing::warn!(framework, error = %e, "failed to write OSCAL snapshot");
}
Ok(doc)
}
Err(fetch_err) => match self.read_snapshot(framework).await? {
Some(doc) => {
tracing::warn!(
framework, error = %fetch_err,
"OSCAL catalog fetch failed; falling back to snapshot"
);
Ok(doc)
}
None => Err(fetch_err),
},
}
}
/// Load the OSCAL catalog for a compliance framework.
pub async fn load(&self, framework: ComplianceFramework) -> Result<OscalDocument, CoreError> {
self.load_token(&framework.to_string()).await
}
/// Load the code-checkable master-controls catalog
/// (`?framework=master-controls`).
pub async fn load_master_controls(&self) -> Result<OscalDocument, CoreError> {
self.load_token("master-controls").await
}
}
/// Order controls whose title/text mention the query context first (stable), then
/// truncate to the requested limit. Naive relevance — refined when the assessment
/// layer lands.
fn rank_and_truncate(mut controls: Vec<Control>, context: &str, limit: usize) -> Vec<Control> {
if !context.is_empty() {
let needle = context.to_lowercase();
controls.sort_by_key(|c| {
let hit =
c.title.to_lowercase().contains(&needle) || c.text.to_lowercase().contains(&needle);
u8::from(!hit)
});
}
controls.truncate(limit);
controls
}
impl ControlsProvider for OscalControlsProvider {
fn name(&self) -> &str {
"breakpilot-oscal"
}
async fn controls(&self, query: &ControlQuery<'_>) -> Result<Vec<Control>, CoreError> {
let mut out: Vec<Control> = Vec::new();
for &framework in query.frameworks {
match self.load(framework).await {
Ok(doc) => out.extend(doc.to_controls()),
Err(e) => {
tracing::warn!(%framework, error = %e, "skipping framework: catalog unavailable")
}
}
}
Ok(rank_and_truncate(out, query.context, query.limit))
}
}
#[cfg(test)]
#[allow(clippy::unwrap_used)]
mod tests {
use super::*;
const MINI_CATALOG: &str = r#"{"catalog":{"uuid":"u","metadata":{"title":"T",
"version":"1.0.0","oscal-version":"1.1.2","props":[{"name":"framework","value":"cra"}]},
"groups":[{"id":"g","title":"G","controls":[{"id":"cra-ai-1","title":"MFA",
"props":[],"parts":[{"name":"statement","prose":"require mfa"}]}]}]}}"#;
fn provider(dir: &std::path::Path) -> OscalControlsProvider {
OscalControlsProvider::new(reqwest::Client::new(), "http://unused/", None, dir)
}
#[test]
fn builds_catalog_url_and_snapshot_path() {
let p = provider(std::path::Path::new("/snap"));
assert_eq!(
p.catalog_url("cra"),
"http://unused/api/compliance/v1/oscal/catalog?framework=cra"
);
assert_eq!(
p.snapshot_path("cra"),
std::path::Path::new("/snap/oscal-catalog-cra.json")
);
}
#[test]
fn ranks_context_hits_first_then_truncates() {
let mk = |id: &str, title: &str| Control {
id: id.into(),
framework: ComplianceFramework::Cra,
title: title.into(),
text: String::new(),
source: None,
};
let controls = vec![
mk("a", "logging policy"),
mk("b", "multi-factor auth"),
mk("c", "backup"),
];
let ranked = rank_and_truncate(controls, "auth", 2);
assert_eq!(ranked.len(), 2);
assert_eq!(ranked[0].id, "b"); // the "auth" hit floats to the top
}
#[tokio::test]
async fn snapshot_round_trip_and_offline_fallback() {
let dir = std::env::temp_dir().join(format!("oscal-test-{}", uuid::Uuid::new_v4()));
let p = provider(&dir);
assert!(p.read_snapshot("cra").await.unwrap().is_none());
p.write_snapshot("cra", MINI_CATALOG.as_bytes())
.await
.unwrap();
let doc = p.read_snapshot("cra").await.unwrap().unwrap();
assert_eq!(doc.to_controls().len(), 1);
assert_eq!(doc.framework(), Some(ComplianceFramework::Cra));
let _ = std::fs::remove_dir_all(&dir);
}
}
@@ -1,299 +0,0 @@
//! Scan-pipeline integration for control triage.
//!
//! After the deterministic tools have produced findings, this stamps each finding
//! with the compliance control(s) it's evidence for and marks control-level false
//! positives — using the ingested OSCAL catalog for control text, the
//! `control-map` LUT for the finding→control link, and the grounded LLM judge to
//! confirm. Skipped entirely unless breakpilot is configured.
use std::collections::HashMap;
use std::path::Path;
use std::sync::Arc;
use compliance_core::control_check::{CandidateRegion, ControlCheckSpec};
use compliance_core::models::finding::{Finding, FindingStatus, Severity};
use compliance_core::models::onboarding::ComplianceFramework;
use compliance_core::AgentConfig;
use control_map::ControlMap;
use super::surface;
use super::{
ControlIndex, ControlTriage, GroundedControlChecker, LlmControlJudge, OscalControlsProvider,
SemanticControlChecker, TriageOutcome,
};
use crate::llm::LlmClient;
/// Nearest master controls judged per code region in the semantic pass.
const SEMANTIC_TOP_K: usize = 5;
/// Lines of context to read on each side of a finding's line.
const REGION_WINDOW: usize = 6;
/// Triage every finding in `findings` against the CRA control map: stamp
/// `control_refs` on confirmed findings and flag control false positives. Returns
/// the number of findings tagged with at least one control.
pub async fn triage_repo_findings(
config: &AgentConfig,
llm: Arc<LlmClient>,
repo_path: &Path,
findings: &mut [Finding],
) -> usize {
let Some(base_url) = config.breakpilot.base_url.clone() else {
return 0; // control triage is opt-in via BREAKPILOT_BASE_URL
};
let provider = OscalControlsProvider::new(
reqwest::Client::new(),
base_url,
config.breakpilot.token.clone(),
&config.breakpilot.snapshot_dir,
);
let specs = build_specs(&provider).await;
if specs.is_empty() {
return 0;
}
let map = match ControlMap::cra() {
Ok(m) => m,
Err(e) => {
tracing::warn!(error = %e, "control map failed to load; skipping control triage");
return 0;
}
};
let triage = ControlTriage::new(LlmControlJudge::new(llm), map, specs);
let mut tagged = 0;
for finding in findings.iter_mut() {
let (Some(file), Some(line)) = (finding.file_path.clone(), finding.line_number) else {
continue;
};
let Some(region) = fetch_region(repo_path, &file, line) else {
continue;
};
match triage.triage(finding, &region).await {
TriageOutcome::Confirmed(controls) => {
finding.control_refs = controls;
tagged += 1;
}
TriageOutcome::FalsePositive => {
finding.status = FindingStatus::FalsePositive;
finding.triage_action = Some("control_false_positive".to_string());
}
TriageOutcome::Unmapped => {}
}
}
tagged
}
/// Build the control requirement specs (by id) from the ingested OSCAL catalog.
async fn build_specs(provider: &OscalControlsProvider) -> HashMap<String, ControlCheckSpec> {
let mut specs = HashMap::new();
match provider.load(ComplianceFramework::Cra).await {
Ok(doc) => {
for control in doc.to_controls() {
specs.insert(
control.id.clone(),
ControlCheckSpec {
control_id: control.id,
title: control.title,
requirement: control.text,
default_cwe: None,
severity: Severity::Medium,
},
);
}
}
Err(e) => tracing::warn!(error = %e, "could not load control catalog for triage"),
}
specs
}
/// Absence-based control pass (the grounded half of the hybrid coverage): for each
/// control with a [`surface`] definition, deterministically retrieve the code
/// surfaces it governs (login routes, logging setup, update/download code) and have
/// the grounded judge decide whether the control holds there. Returns net-new
/// findings, each already tagged with its control and grounded to a real snippet.
///
/// The orchestrator runs this when `breakpilot.grounded_control_checks` is set
/// (on by default). Validated live; it covers the 8 absence-based CRA controls
/// (the judge decides presence/absence, grounded to a real snippet).
pub async fn grounded_surface_findings(
config: &AgentConfig,
llm: Arc<LlmClient>,
repo_path: &Path,
repo_id: &str,
) -> Vec<Finding> {
let Some(base_url) = config.breakpilot.base_url.clone() else {
return Vec::new();
};
let provider = OscalControlsProvider::new(
reqwest::Client::new(),
base_url,
config.breakpilot.token.clone(),
&config.breakpilot.snapshot_dir,
);
let specs = build_specs(&provider).await;
if specs.is_empty() {
return Vec::new();
}
let checker = GroundedControlChecker::new(LlmControlJudge::new(llm));
let mut out = Vec::new();
for surf in surface::SURFACES {
let Some(spec) = specs.get(surf.control_id) else {
continue; // catalog doesn't carry this control
};
let regions = surface::retrieve(repo_path, surf.terms);
if regions.is_empty() {
continue;
}
out.extend(checker.check(spec, &regions, repo_id).await);
}
out
}
/// Read a window of lines around `line` (1-based) from `repo_path/file`.
fn fetch_region(repo_path: &Path, file: &str, line: u32) -> Option<CandidateRegion> {
let content = std::fs::read_to_string(repo_path.join(file)).ok()?;
let lines: Vec<&str> = content.lines().collect();
if lines.is_empty() {
return None;
}
let center = (line.saturating_sub(1) as usize).min(lines.len() - 1);
let start = center.saturating_sub(REGION_WINDOW);
let end = (center + REGION_WINDOW + 1).min(lines.len());
Some(CandidateRegion {
file: file.to_string(),
start_line: (start as u32) + 1,
content: lines[start..end].join("\n"),
})
}
/// Master-controls **semantic** pass: for each finding's code region, retrieve the
/// top-K nearest master controls by embedding, have the grounded judge confirm,
/// and stamp the confirmed control ids onto the finding — the scale path for the
/// ~13.6k master-control corpus (which has no CWE to LUT on). Returns the number
/// of findings that gained a master-control ref.
///
/// The orchestrator runs this when `breakpilot.semantic_mapping` is set (on by
/// default). The control embedding index is built once and cached to
/// `snapshot_dir` keyed by corpus hash ([`ControlIndex::load_or_build`]), so only
/// the first scan after a catalog change pays the embedding cost.
pub async fn semantic_stamp_findings(
config: &AgentConfig,
llm: Arc<LlmClient>,
repo_path: &Path,
findings: &mut [Finding],
) -> usize {
let Some(base_url) = config.breakpilot.base_url.clone() else {
return 0;
};
let provider = OscalControlsProvider::new(
reqwest::Client::new(),
base_url,
config.breakpilot.token.clone(),
&config.breakpilot.snapshot_dir,
);
let doc = match provider.load_master_controls().await {
Ok(d) => d,
Err(e) => {
tracing::warn!(error = %e, "master-controls catalog unavailable; skipping semantic pass");
return 0;
}
};
let specs: Vec<ControlCheckSpec> = doc
.to_controls()
.into_iter()
.map(|c| ControlCheckSpec {
control_id: c.id,
title: c.title,
requirement: c.text,
default_cwe: None,
severity: Severity::Medium,
})
.collect();
let cache_path =
Path::new(&config.breakpilot.snapshot_dir).join("control-index-master-controls.json");
let index = match ControlIndex::load_or_build(&llm, specs, &cache_path).await {
Ok(i) if !i.is_empty() => i,
Ok(_) => return 0,
Err(e) => {
tracing::warn!(error = %e, "failed to embed master-controls corpus");
return 0;
}
};
let checker = SemanticControlChecker::new(LlmControlJudge::new(llm.clone()));
let mut tagged = 0;
for finding in findings.iter_mut() {
if finding.status == FindingStatus::FalsePositive {
continue;
}
let (Some(file), Some(line)) = (finding.file_path.clone(), finding.line_number) else {
continue;
};
let Some(region) = fetch_region(repo_path, &file, line) else {
continue;
};
// Retrieve on the finding's intent + the code, not the region alone: two
// findings in one file share overlapping windows and otherwise embed alike,
// collapsing onto the same controls. The finding's title/description carry
// the discriminating signal (e.g. "brute-force protection" vs "weak hash").
// The raw `region` still goes to the judge for snippet grounding.
let query = format!(
"{}\n{}\n\n{}",
finding.title, finding.description, region.content
);
let query_emb = match llm.embed(vec![query]).await {
Ok(mut embs) => match embs.pop() {
Some(v) => v,
None => continue,
},
Err(e) => {
tracing::warn!(error = %e, "query embed failed; skipping finding");
continue;
}
};
let confirmed = checker
.check(
&index,
&region,
&query_emb,
SEMANTIC_TOP_K,
&finding.repo_id,
)
.await;
let before = finding.control_refs.len();
for f in confirmed {
for cref in f.control_refs {
if !finding.control_refs.contains(&cref) {
finding.control_refs.push(cref);
}
}
}
if finding.control_refs.len() > before {
tagged += 1;
}
}
tagged
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn fetch_region_windows_around_the_line() {
let dir = std::env::temp_dir().join(format!("triage-region-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&dir).unwrap();
let file = "a.py";
std::fs::write(dir.join(file), "l1\nl2\nl3\nSECRET=1\nl5\nl6\n").unwrap();
let r = fetch_region(&dir, file, 4).unwrap();
assert!(r.content.contains("SECRET=1"));
assert_eq!(r.start_line, 1); // window clamps to file start
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn fetch_region_missing_file_is_none() {
assert!(fetch_region(Path::new("/nonexistent"), "nope.py", 1).is_none());
}
}
-121
View File
@@ -1,121 +0,0 @@
//! Semantic control mapping: retrieve the top-K controls nearest a code region,
//! then confirm each with the grounded judge.
//!
//! The `region → controls` direction (vs. the CWE-LUT's `finding → control`) is
//! what scales to the full master-control corpus: the LLM only ever judges a
//! handful of retrieved candidates, and every surviving verdict is still anchored
//! to real code by the grounding gate.
use compliance_core::control_check::{ground, CandidateRegion};
use compliance_core::models::Finding;
use super::index::ControlIndex;
use super::judge::ControlJudge;
/// Retrieve → judge → ground, generic over the judge so tests use a stub.
pub struct SemanticControlChecker<J> {
judge: J,
}
impl<J: ControlJudge> SemanticControlChecker<J> {
pub fn new(judge: J) -> Self {
Self { judge }
}
/// Map a code region to the controls it violates. `query_embedding` is the
/// caller-supplied retrieval embedding — typically the finding's intent
/// (title/description) plus the region, so retrieval keys on what the finding
/// is *about*, not just the ambient code. The top-`k` nearest controls in
/// `index` are then judged against the raw `region` and grounded.
pub async fn check(
&self,
index: &ControlIndex,
region: &CandidateRegion,
query_embedding: &[f64],
k: usize,
repo_id: &str,
) -> Vec<Finding> {
let candidates = index.nearest(query_embedding, k);
let mut findings = Vec::new();
for spec in &candidates {
let verdict = self.judge.judge(spec, region).await;
if let Some(finding) = ground(spec, region, &verdict, repo_id) {
findings.push(finding);
}
}
findings
}
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::control_check::{ControlCheckSpec, LlmVerdict};
use compliance_core::models::finding::Severity;
struct StubJudge {
verdict: LlmVerdict,
}
impl ControlJudge for StubJudge {
async fn judge(&self, _s: &ControlCheckSpec, _r: &CandidateRegion) -> LlmVerdict {
self.verdict.clone()
}
}
fn spec(id: &str) -> ControlCheckSpec {
ControlCheckSpec {
control_id: id.into(),
title: id.into(),
requirement: id.into(),
default_cwe: None,
severity: Severity::Medium,
}
}
#[tokio::test]
async fn retrieves_then_grounds_the_nearest_control() {
let index = ControlIndex::from_embeddings(vec![
(spec("mc-near"), vec![1.0, 0.0]),
(spec("mc-far"), vec![0.0, 1.0]),
]);
let checker = SemanticControlChecker::new(StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "PASSWORD = \"admin\"".into(),
cwe: None,
confidence: 0.9,
},
});
let region = CandidateRegion {
file: "src/auth.py".into(),
start_line: 1,
content: "PASSWORD = \"admin\"\n".into(),
};
// Query embedding nearest to mc-near; k=1 → only mc-near is judged.
let findings = checker
.check(&index, &region, &[0.95, 0.05], 1, "repo")
.await;
assert_eq!(findings.len(), 1);
assert_eq!(findings[0].control_refs, vec!["mc-near".to_string()]);
}
#[tokio::test]
async fn ungrounded_verdict_is_dropped() {
let index = ControlIndex::from_embeddings(vec![(spec("mc-near"), vec![1.0, 0.0])]);
let checker = SemanticControlChecker::new(StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "not in the region".into(),
cwe: None,
confidence: 0.9,
},
});
let region = CandidateRegion {
file: "f".into(),
start_line: 1,
content: "real code\n".into(),
};
let findings = checker.check(&index, &region, &[1.0, 0.0], 1, "repo").await;
assert!(findings.is_empty());
}
}
-258
View File
@@ -1,258 +0,0 @@
//! Surface retrieval for absence-based controls.
//!
//! Some CRA controls are violated by an *absence* — no rate limiting on login, no
//! security logging, no signature check on an update — so there's no offending
//! pattern for semgrep to match. Instead we deterministically locate the code
//! *surface* the control governs (a login route, a logging setup, update/download
//! code) by identifier/route terms, then hand each surface region to the grounded
//! judge, which decides whether the control is satisfied there. The resulting
//! finding grounds to the surface snippet, so nothing fabricated survives.
//!
//! Retrieval is intentionally cheap and bounded: keyword match + a fixed window,
//! capped per control to keep the downstream LLM cost predictable.
use std::path::Path;
use compliance_core::control_check::CandidateRegion;
/// An absence-based control and the case-insensitive terms that mark the code
/// surface it governs.
pub struct Surface {
pub control_id: &'static str,
pub terms: &'static [&'static str],
}
/// The absence-based CRA controls we retrieve surfaces for — the grounded half of
/// the hybrid coverage (the pattern-expressible half is custom semgrep rules).
pub const SURFACES: &[Surface] = &[
Surface {
control_id: "cra-ai-6", // Integritaetspruefung
terms: &[
"checksum",
"sha256",
"signature",
"hmac",
"integrity",
"verify",
],
},
Surface {
control_id: "cra-ai-11", // Brute-Force-Schutz
terms: &[
"login",
"signin",
"authenticate",
"/auth",
"password",
"ratelimit",
],
},
Surface {
control_id: "cra-ai-12", // Rollenbasierte Autorisierung (RBAC)
terms: &[
"authorize",
"permission",
"role",
"rbac",
"require_role",
"has_role",
],
},
Surface {
control_id: "cra-ai-24", // Security-Logging
terms: &["login", "authorize", "permission", "role", "admin", "audit"],
},
Surface {
control_id: "cra-ai-27", // Log-Integritaet und -Aufbewahrung
terms: &["logging", "logger", "getlogger", "audit_log"],
},
Surface {
control_id: "cra-ai-28", // Sichere Update-Mechanismen
terms: &["update", "upgrade", "download", "firmware"],
},
Surface {
control_id: "cra-ai-29", // Update-Authentizitaet
terms: &["update", "signature", "verify", "pubkey", "certificate"],
},
Surface {
control_id: "cra-ai-30", // Update-Integritaet
terms: &["update", "checksum", "digest", "integrity", "verify"],
},
];
/// Source file extensions worth reading (skip binaries/assets/lockfiles).
const CODE_EXTS: &[&str] = &[
"py", "js", "ts", "tsx", "jsx", "go", "java", "rb", "php", "rs", "cs", "kt",
];
/// Directories never worth walking.
const SKIP_DIRS: &[&str] = &[
".git",
"node_modules",
"target",
"vendor",
".venv",
"__pycache__",
"dist",
"build",
];
/// Lines of context on each side of a hit.
const WINDOW: usize = 6;
/// Cap on regions per control, to bound downstream LLM calls.
const MAX_REGIONS_PER_CONTROL: usize = 8;
/// Skip files larger than this (generated/minified).
const MAX_FILE_BYTES: u64 = 512 * 1024;
/// Deterministically retrieve up to [`MAX_REGIONS_PER_CONTROL`] code regions in
/// `repo_path` whose lines mention any of `terms`. Hits close together within a
/// file are merged into one region; results are capped to bound LLM cost.
pub fn retrieve(repo_path: &Path, terms: &[&str]) -> Vec<CandidateRegion> {
let lowered: Vec<String> = terms.iter().map(|t| t.to_lowercase()).collect();
let mut regions = Vec::new();
for entry in walk(repo_path) {
if regions.len() >= MAX_REGIONS_PER_CONTROL {
break;
}
let path = entry.path();
if !has_code_ext(path) {
continue;
}
let Ok(meta) = entry.metadata() else { continue };
if !meta.is_file() || meta.len() > MAX_FILE_BYTES {
continue;
}
let Ok(content) = std::fs::read_to_string(path) else {
continue;
};
let rel = path
.strip_prefix(repo_path)
.unwrap_or(path)
.to_string_lossy()
.to_string();
let lines: Vec<&str> = content.lines().collect();
let hits: Vec<usize> = lines
.iter()
.enumerate()
.filter(|(_, line)| {
let ll = line.to_lowercase();
lowered.iter().any(|t| ll.contains(t.as_str()))
})
.map(|(i, _)| i)
.collect();
for center in merge_centers(&hits) {
if regions.len() >= MAX_REGIONS_PER_CONTROL {
break;
}
let start = center.saturating_sub(WINDOW);
let end = (center + WINDOW + 1).min(lines.len());
regions.push(CandidateRegion {
file: rel.clone(),
start_line: (start as u32) + 1,
content: lines[start..end].join("\n"),
});
}
}
regions
}
/// Collapse ascending hit indices that fall within one window into a single
/// representative center, so overlapping regions aren't judged repeatedly.
fn merge_centers(hits: &[usize]) -> Vec<usize> {
let mut out: Vec<usize> = Vec::new();
for &h in hits {
match out.last() {
Some(&last) if h.saturating_sub(last) <= WINDOW => {}
_ => out.push(h),
}
}
out
}
fn has_code_ext(path: &Path) -> bool {
path.extension()
.and_then(|e| e.to_str())
.is_some_and(|e| CODE_EXTS.contains(&e))
}
fn walk(root: &Path) -> Vec<walkdir::DirEntry> {
walkdir::WalkDir::new(root)
.into_iter()
.filter_entry(|e| {
let name = e.file_name().to_string_lossy();
!SKIP_DIRS.contains(&name.as_ref())
})
.filter_map(|e| e.ok())
.collect()
}
#[cfg(test)]
#[allow(clippy::unwrap_used)]
mod tests {
use super::*;
fn write(dir: &Path, rel: &str, body: &str) {
let p = dir.join(rel);
if let Some(parent) = p.parent() {
std::fs::create_dir_all(parent).unwrap();
}
std::fs::write(p, body).unwrap();
}
fn terms_for(control_id: &str) -> &'static [&'static str] {
SURFACES
.iter()
.find(|s| s.control_id == control_id)
.unwrap()
.terms
}
#[test]
fn surfaces_cover_the_absence_based_controls() {
assert_eq!(SURFACES.len(), 8);
for id in [
"cra-ai-6",
"cra-ai-11",
"cra-ai-12",
"cra-ai-24",
"cra-ai-27",
"cra-ai-28",
"cra-ai-29",
"cra-ai-30",
] {
assert!(SURFACES.iter().any(|s| s.control_id == id), "{id} missing");
}
}
#[test]
fn retrieves_matching_region_with_context() {
let dir = std::env::temp_dir().join(format!("surface-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&dir).unwrap();
write(
&dir,
"app/auth.py",
"import x\n\n\n\n\n\n\ndef login(u, p):\n return check(u, p)\n",
);
let regions = retrieve(&dir, terms_for("cra-ai-11"));
assert_eq!(regions.len(), 1);
assert!(regions[0].content.contains("def login"));
assert_eq!(regions[0].file, "app/auth.py");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn skips_non_code_and_vendored() {
let dir = std::env::temp_dir().join(format!("surface-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&dir).unwrap();
write(&dir, "README.md", "login and password and audit\n"); // not code ext
write(&dir, "node_modules/pkg/index.js", "function login() {}\n"); // vendored
assert!(retrieve(&dir, terms_for("cra-ai-11")).is_empty());
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn merges_adjacent_hits_into_one_region() {
// Two hits one line apart collapse to a single center/region.
assert_eq!(merge_centers(&[10, 11, 30]), vec![10, 30]);
assert_eq!(merge_centers(&[]), Vec::<usize>::new());
assert_eq!(merge_centers(&[5]), vec![5]);
}
}
-234
View File
@@ -1,234 +0,0 @@
//! Triage step: confirm/refute a deterministic tool finding against the controls
//! it maps to (via the `control-map` LUT), grounding the judgment.
//!
//! This is where the LLM finally enters — as a **false-positive filter over tool
//! output**, never as the detector (the ZeroFalse / IRIS pattern). A tool
//! (semgrep, gitleaks, syft/osv) detects deterministically; `controls_for(tool,
//! cwe)` attaches the finding to the control(s) it's evidence for; the grounded
//! judge then confirms or refutes each, and only judgments anchored to real code
//! survive.
use std::collections::HashMap;
use compliance_core::control_check::{ground, CandidateRegion, ControlCheckSpec};
use compliance_core::models::Finding;
use control_map::ControlMap;
use super::judge::ControlJudge;
/// What triage decided for one tool finding.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum TriageOutcome {
/// The finding maps to no control in the LUT — keep it, untagged.
Unmapped,
/// Maps to controls and the grounded judge confirmed at least one — keep the
/// finding and tag it with these control ids.
Confirmed(Vec<String>),
/// Maps to controls but the judge grounded none — treat as a false positive.
FalsePositive,
}
/// Triages tool findings against the control map, confirming with a grounded judge.
pub struct ControlTriage<J> {
judge: J,
map: ControlMap,
/// Control requirement specs (by control id), built from the ingested catalog.
specs: HashMap<String, ControlCheckSpec>,
}
impl<J: ControlJudge> ControlTriage<J> {
pub fn new(judge: J, map: ControlMap, specs: HashMap<String, ControlCheckSpec>) -> Self {
Self { judge, map, specs }
}
/// Triage one tool finding. `region` is the code around the finding, used as
/// the grounding evidence for the judge.
pub async fn triage(&self, finding: &Finding, region: &CandidateRegion) -> TriageOutcome {
// Match by CWE (off-the-shelf findings) and/or rule id (our custom
// detectors, which carry no LUT-bound CWE). A finding with neither is
// simply unmapped.
let mapped = self.map.controls_for_finding(
&finding.scanner,
finding.cwe.as_deref(),
finding.rule_id.as_deref(),
);
if mapped.is_empty() {
return TriageOutcome::Unmapped;
}
let mut confirmed = Vec::new();
for entry in mapped {
let Some(spec) = self.specs.get(&entry.control) else {
continue;
};
let verdict = self.judge.judge(spec, region).await;
// The verdict only counts if it grounds to real code in the region.
if ground(spec, region, &verdict, &finding.repo_id).is_some() {
confirmed.push(entry.control.clone());
}
}
if confirmed.is_empty() {
TriageOutcome::FalsePositive
} else {
TriageOutcome::Confirmed(confirmed)
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use compliance_core::control_check::LlmVerdict;
use compliance_core::models::finding::Severity;
use compliance_core::models::scan::ScanType;
struct StubJudge {
verdict: LlmVerdict,
}
impl ControlJudge for StubJudge {
async fn judge(&self, _s: &ControlCheckSpec, _r: &CandidateRegion) -> LlmVerdict {
self.verdict.clone()
}
}
fn specs() -> HashMap<String, ControlCheckSpec> {
let mut m = HashMap::new();
m.insert(
"cra-ai-8".to_string(),
ControlCheckSpec {
control_id: "cra-ai-8".into(),
title: "No default passwords".into(),
requirement: "No default credentials".into(),
default_cwe: Some("CWE-798".into()),
severity: Severity::High,
},
);
m
}
fn semgrep_finding(cwe: &str) -> Finding {
let mut f = Finding::new(
"repo".into(),
"fp1".into(),
"semgrep".into(),
ScanType::Sast,
"hardcoded credential".into(),
"desc".into(),
Severity::High,
);
f.cwe = Some(cwe.into());
f
}
fn region() -> CandidateRegion {
CandidateRegion {
file: "src/auth.py".into(),
start_line: 1,
content: "PASSWORD = \"admin123\"\n".into(),
}
}
#[tokio::test]
async fn confirmed_finding_is_tagged_with_control() {
let triage = ControlTriage::new(
StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "PASSWORD = \"admin123\"".into(),
cwe: None,
confidence: 0.9,
},
},
ControlMap::cra().unwrap(),
specs(),
);
let out = triage.triage(&semgrep_finding("CWE-798"), &region()).await;
assert_eq!(out, TriageOutcome::Confirmed(vec!["cra-ai-8".to_string()]));
}
#[tokio::test]
async fn refuted_mapped_finding_is_false_positive() {
// Maps to cra-ai-8, but the judge doesn't confirm (no violation) → FP.
let triage = ControlTriage::new(
StubJudge {
verdict: LlmVerdict {
violates: false,
snippet: String::new(),
cwe: None,
confidence: 0.1,
},
},
ControlMap::cra().unwrap(),
specs(),
);
let out = triage.triage(&semgrep_finding("CWE-798"), &region()).await;
assert_eq!(out, TriageOutcome::FalsePositive);
}
#[tokio::test]
async fn custom_rule_finding_without_cwe_is_confirmed() {
// A custom detector finding carries a rule id but no LUT-bound CWE; it must
// still map (by rule id) and confirm.
let mut specs = specs();
specs.insert(
"cra-ai-1".to_string(),
ControlCheckSpec {
control_id: "cra-ai-1".into(),
title: "Secure-by-Default".into(),
requirement: "Ship secure defaults".into(),
default_cwe: None,
severity: Severity::Medium,
},
);
let triage = ControlTriage::new(
StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "app.run(debug=True)".into(),
cwe: None,
confidence: 0.9,
},
},
ControlMap::cra().unwrap(),
specs,
);
let mut f = Finding::new(
"repo".into(),
"fp".into(),
"semgrep".into(),
ScanType::Sast,
"flask debug".into(),
"desc".into(),
Severity::Medium,
);
f.rule_id = Some("tmp.x.cra-ai-1-flask-debug-enabled".into()); // no cwe
let region = CandidateRegion {
file: "app.py".into(),
start_line: 1,
content: "app.run(debug=True)\n".into(),
};
let out = triage.triage(&f, &region).await;
assert_eq!(out, TriageOutcome::Confirmed(vec!["cra-ai-1".to_string()]));
}
#[tokio::test]
async fn unmapped_cwe_is_left_untagged() {
let triage = ControlTriage::new(
StubJudge {
verdict: LlmVerdict {
violates: true,
snippet: "PASSWORD = \"admin123\"".into(),
cwe: None,
confidence: 0.9,
},
},
ControlMap::cra().unwrap(),
specs(),
);
let out = triage
.triage(&semgrep_finding("CWE-99999"), &region())
.await;
assert_eq!(out, TriageOutcome::Unmapped);
}
}
+14 -34
View File
@@ -249,6 +249,16 @@ impl Database {
}
pub async fn ensure_indexes(&self) -> Result<(), AgentError> {
// repositories: unique git_url
self.repositories()
.create_index(
IndexModel::builder()
.keys(doc! { "git_url": 1 })
.options(IndexOptions::builder().unique(true).build())
.build(),
)
.await?;
// findings: unique fingerprint
self.findings()
.create_index(
@@ -465,38 +475,14 @@ impl Database {
)
.await?;
// werkbank_jobs: unique job id (idempotent enqueue by job id)
self.werkbank_jobs()
.create_index(
IndexModel::builder()
.keys(doc! { "job.id": 1 })
.options(IndexOptions::builder().unique(true).build())
.build(),
)
.await?;
// werkbank_jobs: lease query — oldest queued job for an executor
self.werkbank_jobs()
.create_index(
IndexModel::builder()
.keys(doc! { "status": 1, "job.executor": 1, "created_at": 1 })
.build(),
)
.await?;
// werkbank_jobs: visibility-timeout sweep of expired leases
self.werkbank_jobs()
.create_index(
IndexModel::builder()
.keys(doc! { "status": 1, "lease_expires_at": 1 })
.build(),
)
.await?;
tracing::info!("Database indexes ensured");
Ok(())
}
pub fn repositories(&self) -> Collection<TrackedRepository> {
self.inner.collection("repositories")
}
pub fn findings(&self) -> Collection<Finding> {
self.inner.collection("findings")
}
@@ -591,12 +577,6 @@ impl Database {
self.inner.collection("pentest_messages")
}
/// The Werkbank job queue (WB-02): declarative dynamic-execution jobs the
/// control plane enqueues and runners lease.
pub fn werkbank_jobs(&self) -> Collection<compliance_core::models::werkbank::JobRecord> {
self.inner.collection("werkbank_jobs")
}
#[allow(dead_code)]
pub fn raw_collection(&self, name: &str) -> Collection<mongodb::bson::Document> {
self.inner.collection(name)
-3
View File
@@ -27,9 +27,6 @@ pub enum AgentError {
#[error("Configuration error: {0}")]
Config(String),
#[error("Dynamic-execution error: {0}")]
Exec(#[from] werkbank_exec::ExecError),
#[error("{0}")]
Other(String),
}
-24
View File
@@ -32,30 +32,6 @@ pub fn hash_file(path: &Path) -> Result<(String, u64), AgentError> {
Ok((hex::encode(hasher.finalize()), total))
}
/// Store raw bytes in the content-addressed blob store under `base`, returning
/// the SHA-256 digest. Used to stash a small derived artifact (e.g. the extracted
/// PLC program source) so a Werkbank runner can fetch it by hash. Idempotent.
pub fn store_bytes(base: &Path, bytes: &[u8]) -> Result<String, AgentError> {
let sha = hex::encode(Sha256::digest(bytes));
let dir = base.join("blobs").join(&sha[0..2]);
fs::create_dir_all(&dir)?;
let dest = dir.join(&sha);
if !dest.exists() {
fs::write(&dest, bytes)?;
}
Ok(sha)
}
/// Read a blob's bytes by its SHA-256 digest. Rejects a non-hex/wrong-length hash
/// so a request can't traverse outside the blob store.
pub fn read_blob(base: &Path, sha: &str) -> Result<Vec<u8>, AgentError> {
if sha.len() != 64 || !sha.bytes().all(|b| b.is_ascii_hexdigit()) {
return Err(AgentError::Other(format!("invalid content hash '{sha}'")));
}
let path = base.join("blobs").join(&sha[0..2]).join(sha);
Ok(fs::read(path)?)
}
/// Copy `src` into the content-addressed blob store under `base`, returning the
/// stored path. Idempotent: an already-present blob is not rewritten.
pub fn store_file(base: &Path, src: &Path, sha: &str) -> Result<PathBuf, AgentError> {
+4 -86
View File
@@ -6,7 +6,7 @@
//! is also the reconciliation key against sibling products (a firmware sha256
//! matches tramiton's `Artifact.sha256`).
pub(crate) mod blob;
mod blob;
use std::collections::HashMap;
use std::path::{Path, PathBuf};
@@ -162,27 +162,14 @@ fn ingest_blob(
match blob::extract_zip(&stored, &dest) {
Ok(()) => dest,
Err(e) => {
// Not a zip container — this is a single uploaded file (e.g. a
// `.st`/`.xml` PLC project or a `.tar.gz`). The content-addressed
// blob has no extension, so materialize it into a working dir
// under its original name; extension-based scanners (PLC) can then
// discover it and report a readable path.
// Not a zip (e.g. a tar.gz source archive) — keep the blob and
// note it so later stages can decide what to do.
facts.push(DetectedFact::new(
"archive_unextracted",
e.to_string(),
"ingest",
));
match materialize_single(&stored, &dest, &blob_file_name(artifact)) {
Ok(dir) => dir,
Err(copy_err) => {
facts.push(DetectedFact::new(
"materialize_failed",
copy_err.to_string(),
"ingest",
));
stored.clone()
}
}
stored.clone()
}
}
} else {
@@ -199,27 +186,6 @@ fn ingest_blob(
})
}
/// Copy a stored blob into `dest`/`name`, returning `dest`. Used when an
/// "extractable" artifact turns out to be a single file rather than an archive.
fn materialize_single(stored: &Path, dest: &Path, name: &str) -> Result<PathBuf, AgentError> {
std::fs::create_dir_all(dest)?;
std::fs::copy(stored, dest.join(name))?;
Ok(dest.to_path_buf())
}
/// A safe, single-segment file name for an artifact, preserving the original
/// extension so scanners can identify it. Derives from `source_ref` (the
/// uploaded/original file name); `file_name` strips any directory components,
/// so this is traversal-safe. Falls back to the artifact id.
fn blob_file_name(artifact: &Artifact) -> String {
Path::new(&artifact.source_ref)
.file_name()
.and_then(|n| n.to_str())
.map(str::to_string)
.filter(|s| !s.is_empty())
.unwrap_or_else(|| format!("artifact-{}", artifact.id))
}
/// An artifact with no on-disk form: record a single fact, no hash/path.
fn metadata_only(artifact: &Artifact, fact: DetectedFact) -> IngestedArtifact {
IngestedArtifact {
@@ -365,52 +331,4 @@ mod tests {
assert_eq!(creds.ssh_key_path.as_deref(), Some("/default/ssh/key"));
assert!(creds.auth_token.is_none());
}
/// A single uploaded PLC file (not an archive) must land in a working dir
/// under its original name so the PLC scanner can discover it by extension
/// and report a readable path — the demo's upload → scan path.
#[test]
fn single_uploaded_plc_file_is_materialized_and_scannable() {
use compliance_core::models::PlcFormat;
let scratch = Scratch::new();
let store = scratch.0.join("store");
// Simulate the upload handler: bytes written to an `uploads/` path,
// `source_ref` carrying the original (clean) file name.
let uploads = scratch.0.join("uploads");
std::fs::create_dir_all(&uploads).expect("mkdir uploads");
let uploaded = uploads.join("a1b2c3_pump_station.st");
std::fs::write(
&uploaded,
"PROGRAM P\nVAR\n ApiKey : STRING := 'sk-live-1234';\nEND_VAR\nEND_PROGRAM\n",
)
.expect("write st");
let mut artifact = Artifact::plc_project("pump_station.st", PlcFormat::StructuredText);
artifact.stored_path = Some(uploaded.to_string_lossy().to_string());
let ctx = ctx_for(&store, "t-plc");
let out = ingest_artifact(&artifact, &ctx).expect("ingest");
// Working path is a directory (not the extensionless blob) holding the
// file under its original name.
let wp = out.working_path.expect("working path");
assert!(wp.is_dir(), "expected a working dir, got {wp:?}");
assert!(wp.join("pump_station.st").is_file());
// The PLC scanner finds the hardcoded credential and reports a clean path.
let findings = crate::pipeline::plc::analyze_tree(&wp, "t-plc");
assert!(
!findings.is_empty(),
"scanner should flag the uploaded file"
);
assert!(findings
.iter()
.any(|f| f.rule_id.as_deref() == Some("plc-hardcoded-credential")));
assert_eq!(
findings[0].file_path.as_deref(),
Some("pump_station.st"),
"finding should reference the original file name"
);
}
}
+1 -2
View File
@@ -4,11 +4,11 @@ pub mod agent;
pub mod api;
pub mod classify;
pub mod config;
pub mod controls;
pub mod database;
pub mod error;
pub mod ingest;
pub mod llm;
pub mod migrate;
pub mod pentest;
pub mod pipeline;
pub mod rag;
@@ -17,4 +17,3 @@ pub mod ssh;
#[allow(dead_code)]
pub mod trackers;
pub mod webhooks;
pub mod werkbank;
+1 -49
View File
@@ -22,11 +22,6 @@ struct EmbeddingData {
index: usize,
}
/// Max inputs per embedding request. The bge/OpenAI-like backends cap the input
/// array (bge-multilingual-gemma2 rejects >25 with "batch size overflow"), so we
/// chunk larger corpora — a whole control catalog (~1.8k) would otherwise 500.
const EMBED_BATCH_SIZE: usize = 16;
// ── Embedding implementation ───────────────────────────────────
impl LlmClient {
@@ -34,21 +29,8 @@ impl LlmClient {
&self.embed_model
}
/// Generate embeddings for a batch of texts, chunking into backend-sized
/// requests and preserving input order across chunks.
/// Generate embeddings for a batch of texts
pub async fn embed(&self, texts: Vec<String>) -> Result<Vec<Vec<f64>>, AgentError> {
if texts.is_empty() {
return Ok(Vec::new());
}
let mut out = Vec::with_capacity(texts.len());
for chunk in texts.chunks(EMBED_BATCH_SIZE) {
out.extend(self.embed_batch(chunk.to_vec()).await?);
}
Ok(out)
}
/// Embed one backend-sized batch (≤ [`EMBED_BATCH_SIZE`]) in a single request.
async fn embed_batch(&self, texts: Vec<String>) -> Result<Vec<Vec<f64>>, AgentError> {
let url = format!("{}/v1/embeddings", self.base_url.trim_end_matches('/'));
let request_body = EmbeddingRequest {
@@ -90,33 +72,3 @@ impl LlmClient {
Ok(data.into_iter().map(|d| d.embedding).collect())
}
}
#[cfg(test)]
mod tests {
use super::*;
use secrecy::SecretString;
fn client() -> LlmClient {
LlmClient::new(
"http://unused".into(),
SecretString::from(String::new()),
"m".into(),
"e".into(),
)
}
#[tokio::test]
async fn empty_input_makes_no_request() {
// Must short-circuit before any HTTP call (base_url is unroutable).
let out = client().embed(Vec::new()).await.unwrap();
assert!(out.is_empty());
}
#[test]
fn batch_size_is_within_backend_cap() {
assert!(
EMBED_BATCH_SIZE <= 25,
"must stay under the bge 25-input cap"
);
}
}
+54 -1
View File
@@ -1,4 +1,50 @@
use compliance_agent::{agent, api, config, database, scheduler, ssh, webhooks};
use compliance_agent::{agent, api, config, database, migrate, scheduler, ssh, webhooks};
/// Run the `migrate onboarding` subcommand and exit. Backfills (or reverts) the
/// unified `onboarded_targets` collection per tenant.
///
/// Usage: `compliance-agent migrate onboarding [--all | --tenant <id>] [--dry-run] [--revert]`
async fn run_migration(
args: &[String],
pool: &database::DatabasePool,
) -> Result<(), compliance_agent::error::AgentError> {
if args.get(2).map(String::as_str) != Some("onboarding") {
eprintln!(
"usage: compliance-agent migrate onboarding [--all | --tenant <id>] [--dry-run] [--revert]"
);
std::process::exit(2);
}
let has = |flag: &str| args.iter().any(|a| a == flag);
let dry_run = has("--dry-run");
let revert = has("--revert");
let tenant = args
.iter()
.position(|a| a == "--tenant")
.and_then(|i| args.get(i + 1))
.cloned();
let tenants: Vec<String> = if has("--all") {
pool.list_tenant_ids().await?
} else if let Some(t) = tenant {
vec![t]
} else {
eprintln!("specify --all or --tenant <id>");
std::process::exit(2);
};
for tenant_id in tenants {
let db = pool.for_tenant_id(&tenant_id).await?;
if revert {
migrate::onboarding::revert(&db).await?;
println!("[{tenant_id}] reverted onboarding backfill");
} else {
let report = migrate::onboarding::backfill_onboarded_targets(&db, dry_run).await?;
let prefix = if dry_run { "(dry-run) " } else { "" };
println!("[{tenant_id}] {prefix}{report:?}");
}
}
Ok(())
}
#[tokio::main]
async fn main() -> Result<(), Box<dyn std::error::Error>> {
@@ -31,6 +77,13 @@ async fn main() -> Result<(), Box<dyn std::error::Error>> {
let db_pool =
database::DatabasePool::connect(&config.mongodb_uri, &config.mongodb_database).await?;
// One-shot subcommands run and exit without starting the servers.
let args: Vec<String> = std::env::args().collect();
if args.get(1).map(String::as_str) == Some("migrate") {
run_migration(&args, &db_pool).await?;
return Ok(());
}
let agent = agent::ComplianceAgent::new(config.clone(), db_pool);
tracing::info!("Starting scheduler...");
+8
View File
@@ -0,0 +1,8 @@
//! One-time data migrations.
//!
//! Currently just the onboarding backfill ([`onboarding`]), which folds the
//! legacy `repositories` and `dast_targets` collections into the unified
//! `onboarded_targets` collection, preserving `_id` so every downstream record
//! keyed by `repo_id` / `target_id` keeps resolving.
pub mod onboarding;
+406
View File
@@ -0,0 +1,406 @@
//! Backfill: legacy `repositories` + `dast_targets` → `onboarded_targets`.
//!
//! The transforms here are **id-preserving**: an [`OnboardedTarget`] keeps the
//! same `_id` as the `TrackedRepository` / `DastTarget` it came from, so every
//! downstream collection keyed by that hex id (findings, sbom, scan_runs,
//! graph, dast_*, pentest_*) keeps resolving with zero row rewrites, and
//! existing webhook URLs keep working. The mapping functions are pure and unit
//! tested; the DB orchestration (idempotent per-tenant backfill + revert) is a
//! thin driver over them.
use compliance_core::models::{
Artifact, ArtifactKind, DastTarget, DastTargetType, GitArtifactConfig, IssueTrackerConfig,
OnboardedTarget, TargetType, TrackedRepository, WebArtifactConfig,
};
use futures_util::TryStreamExt;
use mongodb::bson::{doc, Document};
use crate::database::Database;
use crate::error::AgentError;
/// Marker id in `schema_migrations` recording that the backfill has run.
const MIGRATION_MARKER: &str = "onboarding_backfill_v1";
/// Summary of a backfill run.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct MigrationReport {
/// Repositories turned into onboarded targets.
pub repos_migrated: u64,
/// DAST targets folded into an existing (repo-linked) target as a LiveUrl.
pub dast_targets_folded: u64,
/// DAST targets with no repo link, migrated as standalone targets.
pub dast_targets_standalone: u64,
/// Records skipped because a target with that `_id` already existed.
pub skipped_existing: u64,
}
/// Map a legacy `DastTargetType` to a unified [`TargetType`]. REST/GraphQL APIs
/// are backend services; a browser app is a web app.
fn target_type_for_dast(kind: &DastTargetType) -> TargetType {
match kind {
DastTargetType::WebApp => TargetType::WebApp,
DastTargetType::RestApi | DastTargetType::GraphQl => TargetType::BackendService,
}
}
/// Build the LiveUrl artifact for a DAST target (its base URL + crawl config +
/// auth). Shared by fold-in and standalone migration.
pub fn dast_to_artifact(dast: &DastTarget) -> Artifact {
let mut artifact = Artifact::live_url(dast.base_url.clone());
artifact.web = Some(WebArtifactConfig {
target_kind: dast.target_type.clone(),
excluded_paths: dast.excluded_paths.clone(),
max_crawl_depth: dast.max_crawl_depth,
rate_limit: dast.rate_limit,
allow_destructive: dast.allow_destructive,
});
artifact.auth = dast.auth_config.clone().map(Into::into);
artifact
}
/// Map a `TrackedRepository` to an onboarded target, preserving `_id`. The git
/// remote becomes a `GitRepo` artifact carrying the repo's branch, watermark,
/// and auth; tracker config folds into `scan_config`.
///
/// `target_type` is a safe default (`BackendService`) — the classifier can
/// refine it later; `classification` is left `None` (unconfirmed).
pub fn repo_to_target(repo: &TrackedRepository) -> OnboardedTarget {
let mut target = OnboardedTarget::new(repo.name.clone(), TargetType::BackendService);
target.id = repo.id;
let mut artifact = Artifact::git_repo(repo.git_url.clone(), repo.default_branch.clone());
artifact.git = Some(GitArtifactConfig {
default_branch: repo.default_branch.clone(),
last_scanned_commit: repo.last_scanned_commit.clone(),
local_path: repo.local_path.clone(),
});
if repo.auth_token.is_some() || repo.auth_username.is_some() {
artifact.auth = Some(compliance_core::models::ArtifactAuth {
method: "token".to_string(),
username: repo.auth_username.clone(),
secret: repo.auth_token.clone(),
..Default::default()
});
}
target.artifacts.push(artifact);
if repo.tracker_type.is_some() {
target.scan_config.issue_tracker = Some(IssueTrackerConfig {
tracker_type: repo.tracker_type.clone(),
owner: repo.tracker_owner.clone(),
repo: repo.tracker_repo.clone(),
token: repo.tracker_token.clone(),
});
}
target.scan_schedule = repo.scan_schedule.clone();
target.webhook_enabled = repo.webhook_enabled;
target.webhook_secret = repo.webhook_secret.clone();
target.findings_count = repo.findings_count;
target.created_at = repo.created_at;
target.updated_at = repo.updated_at;
target
}
/// Append a DAST target's LiveUrl artifact onto an existing (repo-derived)
/// target. If the repo default was `BackendService` but the DAST target is a
/// browser web app, promote the type to `WebApp`.
pub fn fold_dast_into_target(target: &mut OnboardedTarget, dast: &DastTarget) {
if matches!(dast.target_type, DastTargetType::WebApp)
&& target.target_type == TargetType::BackendService
{
target.target_type = TargetType::WebApp;
}
if !target.has(ArtifactKind::LiveUrl) {
target.artifacts.push(dast_to_artifact(dast));
}
}
/// Map a repo-less DAST target to a standalone onboarded target, preserving `_id`.
pub fn dast_to_standalone_target(dast: &DastTarget) -> OnboardedTarget {
let mut target =
OnboardedTarget::new(dast.name.clone(), target_type_for_dast(&dast.target_type));
target.id = dast.id;
target.artifacts.push(dast_to_artifact(dast));
target.created_at = dast.created_at;
target.updated_at = dast.updated_at;
target
}
/// Whether the onboarding backfill has already been applied to this database.
pub async fn already_applied(db: &Database) -> Result<bool, AgentError> {
let found = db
.collection_named::<Document>("schema_migrations")
.find_one(doc! { "_id": MIGRATION_MARKER })
.await?;
Ok(found.is_some())
}
/// Backfill `onboarded_targets` from `repositories` + `dast_targets` for one
/// tenant database.
///
/// Id-preserving and **idempotent**: targets that already exist (by `_id`) are
/// skipped, so re-running is safe. With `dry_run`, computes the report without
/// writing. The legacy collections are never deleted; the only mutation outside
/// `onboarded_targets` is the history relink of folded DAST targets, which is
/// logged so [`revert`] can undo it.
pub async fn backfill_onboarded_targets(
db: &Database,
dry_run: bool,
) -> Result<MigrationReport, AgentError> {
let mut report = MigrationReport::default();
// 1. repositories -> onboarded_targets (preserve _id, skip existing).
let mut repos = db.repositories().find(doc! {}).await?;
while let Some(repo) = repos.try_next().await? {
let Some(id) = repo.id else { continue };
if db
.onboarded_targets()
.find_one(doc! { "_id": id })
.await?
.is_some()
{
report.skipped_existing += 1;
continue;
}
if !dry_run {
db.onboarded_targets()
.insert_one(repo_to_target(&repo))
.await?;
}
report.repos_migrated += 1;
}
// 2. dast_targets -> fold into the linked repo target, or migrate standalone.
let mut dasts = db.dast_targets().find(doc! {}).await?;
while let Some(dast) = dasts.try_next().await? {
let Some(dast_id) = dast.id else { continue };
let repo_oid = dast
.repo_id
.as_deref()
.and_then(|r| mongodb::bson::oid::ObjectId::parse_str(r).ok());
let linked = match repo_oid {
Some(oid) => db.onboarded_targets().find_one(doc! { "_id": oid }).await?,
None => None,
};
match (linked, repo_oid) {
// Fold into an existing repo-derived target.
(Some(mut target), Some(oid)) => {
if target.has(ArtifactKind::LiveUrl) {
report.skipped_existing += 1; // already folded on a prior run
continue;
}
fold_dast_into_target(&mut target, &dast);
if !dry_run {
db.onboarded_targets()
.replace_one(doc! { "_id": oid }, &target)
.await?;
relink_history(db, &dast_id.to_hex(), &oid.to_hex()).await?;
}
report.dast_targets_folded += 1;
}
// No linked repo target: migrate as a standalone target (keeps _id).
_ => {
if db
.onboarded_targets()
.find_one(doc! { "_id": dast_id })
.await?
.is_some()
{
report.skipped_existing += 1;
continue;
}
if !dry_run {
db.onboarded_targets()
.insert_one(dast_to_standalone_target(&dast))
.await?;
}
report.dast_targets_standalone += 1;
}
}
}
if !dry_run {
db.collection_named::<Document>("schema_migrations")
.update_one(
doc! { "_id": MIGRATION_MARKER },
doc! { "$set": { "applied_at": mongodb::bson::DateTime::now() } },
)
.upsert(true)
.await?;
}
Ok(report)
}
/// Relink DAST scan runs and pentest sessions from the old DAST target id to the
/// unified target id, logging each move so [`revert`] can undo it.
///
/// Note: if multiple DAST targets fold into the same repo target, revert
/// restores only the last-logged mapping — a rare edge. The source collections
/// (`repositories`, `dast_targets`) are never deleted, so no data is lost.
async fn relink_history(db: &Database, old_id: &str, new_id: &str) -> Result<(), AgentError> {
db.dast_scan_runs()
.update_many(
doc! { "target_id": old_id },
doc! { "$set": { "target_id": new_id } },
)
.await?;
db.pentest_sessions()
.update_many(
doc! { "target_id": old_id },
doc! { "$set": { "target_id": new_id } },
)
.await?;
db.collection_named::<Document>("onboarding_migration_log")
.insert_one(doc! { "old_target_id": old_id, "new_target_id": new_id })
.await?;
Ok(())
}
/// Undo the backfill: replay the relink log in reverse, drop `onboarded_targets`
/// and the log, and clear the marker. The legacy collections are untouched, so
/// this restores the pre-migration state.
pub async fn revert(db: &Database) -> Result<(), AgentError> {
let log = db.collection_named::<Document>("onboarding_migration_log");
let mut cursor = log.find(doc! {}).await?;
while let Some(entry) = cursor.try_next().await? {
if let (Ok(old), Ok(new)) = (
entry.get_str("old_target_id"),
entry.get_str("new_target_id"),
) {
db.dast_scan_runs()
.update_many(
doc! { "target_id": new },
doc! { "$set": { "target_id": old } },
)
.await?;
db.pentest_sessions()
.update_many(
doc! { "target_id": new },
doc! { "$set": { "target_id": old } },
)
.await?;
}
}
db.onboarded_targets().drop().await?;
log.drop().await?;
db.collection_named::<Document>("schema_migrations")
.delete_one(doc! { "_id": MIGRATION_MARKER })
.await?;
Ok(())
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
use compliance_core::models::{DastAuthConfig, TrackerType};
fn repo() -> TrackedRepository {
let mut r = TrackedRepository::new("acme".to_string(), "https://git/acme.git".to_string());
r.id = Some(mongodb::bson::oid::ObjectId::new());
r.default_branch = "develop".to_string();
r.last_scanned_commit = Some("abc123".to_string());
r.auth_token = Some("pat".to_string());
r.auth_username = Some("bob".to_string());
r.tracker_type = Some(TrackerType::Gitea);
r.tracker_owner = Some("acme".to_string());
r.findings_count = 7;
r
}
fn dast(repo_id: Option<String>, kind: DastTargetType) -> DastTarget {
let mut d = DastTarget::new(
"acme-web".to_string(),
"https://acme.example.com".to_string(),
kind,
);
d.id = Some(mongodb::bson::oid::ObjectId::new());
d.repo_id = repo_id;
d.max_crawl_depth = 5;
d.auth_config = Some(DastAuthConfig {
method: "bearer".to_string(),
login_url: None,
username: None,
password: None,
token: Some("tok".to_string()),
headers: None,
});
d
}
#[test]
fn repo_maps_preserving_id_and_git_artifact() {
let r = repo();
let t = repo_to_target(&r);
assert_eq!(t.id, r.id); // id preserved
assert_eq!(t.findings_count, 7);
assert_eq!(t.scan_schedule, r.scan_schedule);
let git = t.code_artifact().expect("git artifact");
assert_eq!(git.kind, ArtifactKind::GitRepo);
assert_eq!(git.source_ref, "https://git/acme.git");
let gc = git.git.as_ref().expect("git config");
assert_eq!(gc.default_branch, "develop");
assert_eq!(gc.last_scanned_commit.as_deref(), Some("abc123"));
let auth = git.auth.as_ref().expect("auth");
assert_eq!(auth.secret.as_deref(), Some("pat"));
assert_eq!(auth.username.as_deref(), Some("bob"));
assert_eq!(
t.scan_config
.issue_tracker
.as_ref()
.and_then(|it| it.tracker_type.clone()),
Some(TrackerType::Gitea)
);
}
#[test]
fn standalone_dast_maps_preserving_id_and_live_url() {
let d = dast(None, DastTargetType::WebApp);
let t = dast_to_standalone_target(&d);
assert_eq!(t.id, d.id);
assert_eq!(t.target_type, TargetType::WebApp);
let url = t.live_url().expect("live url");
assert_eq!(url.source_ref, "https://acme.example.com");
let web = url.web.as_ref().expect("web config");
assert_eq!(web.max_crawl_depth, 5);
assert_eq!(
url.auth.as_ref().and_then(|a| a.secret.clone()),
Some("tok".to_string())
);
}
#[test]
fn rest_api_dast_maps_to_backend_service() {
let d = dast(None, DastTargetType::RestApi);
assert_eq!(
dast_to_standalone_target(&d).target_type,
TargetType::BackendService
);
}
#[test]
fn fold_adds_live_url_and_promotes_webapp() {
let mut t = repo_to_target(&repo());
assert_eq!(t.target_type, TargetType::BackendService);
fold_dast_into_target(&mut t, &dast(Some("x".to_string()), DastTargetType::WebApp));
assert_eq!(t.target_type, TargetType::WebApp); // promoted
assert!(t.has(ArtifactKind::LiveUrl));
assert!(t.has(ArtifactKind::GitRepo));
}
#[test]
fn fold_is_idempotent_on_live_url() {
let mut t = repo_to_target(&repo());
let d = dast(Some("x".to_string()), DastTargetType::WebApp);
fold_dast_into_target(&mut t, &d);
fold_dast_into_target(&mut t, &d);
let live_urls = t
.artifacts
.iter()
.filter(|a| a.kind == ArtifactKind::LiveUrl)
.count();
assert_eq!(live_urls, 1);
}
}
+1 -3
View File
@@ -342,9 +342,7 @@ mod tests {
pentest_imap_password: None,
admin_api_token: None,
tenant_registry_url: None,
plc_runtime: compliance_core::PlcRuntimeConfig::default(),
werkbank_runner_token: None,
breakpilot: compliance_core::config::BreakpilotConfig::default(),
unified_pipeline: false,
}
}
-283
View File
@@ -204,202 +204,6 @@ impl CveScanner {
Ok(results)
}
/// Match the CODESYS **runtime** component against NVD by CPE.
///
/// CODESYS advisories (the CoDe16 cluster and friends) are indexed in NVD by
/// CPE (`cpe:2.3:a:codesys:control*`) keyed off the *runtime* version — not by
/// the internal `Cmp*`/`Sys*` library names OSV-by-purl would look up. So we
/// find the runtime SBOM entry, pull every `cpe:2.3:a:codesys:*` CVE from NVD,
/// and keep the ones whose affected-version range covers our runtime version.
/// Best-effort: returns empty without an NVD key, on a network error, or when
/// no CODESYS runtime component is present.
pub async fn scan_codesys(&self, repo_id: &str, entries: &mut [SbomEntry]) -> Vec<CveAlert> {
let Some((name, version)) = codesys_runtime(entries) else {
return Vec::new();
};
let url = "https://services.nvd.nist.gov/rest/json/cves/2.0\
?virtualMatchString=cpe:2.3:a:codesys";
let mut req = self.http.get(url);
if let Some(key) = &self.nvd_api_key {
req = req.header("apiKey", key.as_str());
}
let body: serde_json::Value = match req.send().await {
Ok(r) if r.status().is_success() => match r.json().await {
Ok(b) => b,
Err(e) => {
tracing::warn!("CODESYS NVD parse failed: {e}");
return Vec::new();
}
},
Ok(r) => {
tracing::warn!("CODESYS NVD returned {}", r.status());
return Vec::new();
}
Err(e) => {
tracing::warn!("CODESYS NVD request failed: {e}");
return Vec::new();
}
};
let matched = parse_codesys_nvd(&body, &version);
let mut alerts = Vec::new();
for cve in matched {
if let Some(e) = entries
.iter_mut()
.find(|e| e.name == name && e.version == version)
{
e.known_vulnerabilities.push(VulnRef {
id: cve.id.clone(),
source: "nvd".to_string(),
severity: None,
url: Some(format!("https://nvd.nist.gov/vuln/detail/{}", cve.id)),
});
}
let mut alert = CveAlert::new(
cve.id,
repo_id.to_string(),
name.clone(),
version.clone(),
CveSource::Nvd,
);
alert.summary = cve.summary;
alert.cvss_score = cve.cvss;
alerts.push(alert);
}
tracing::info!(runtime = %name, version = %version, cves = alerts.len(), "CODESYS CVE match");
alerts
}
}
/// The CODESYS runtime component (name + version) from an SBOM, if present. The
/// runtime carries the version CODESYS advisories key off; the internal library
/// components do not.
fn codesys_runtime(entries: &[SbomEntry]) -> Option<(String, String)> {
entries
.iter()
.find(|e| e.package_manager == "codesys" && e.name.starts_with("CODESYS Control"))
.map(|e| (e.name.clone(), e.version.clone()))
}
/// A parsed NVD CVE that affects the CODESYS runtime.
struct CodesysCve {
id: String,
summary: Option<String>,
cvss: Option<f64>,
}
/// Version constraints from an NVD `cpeMatch` node.
#[derive(Default)]
struct CpeRange {
exact: Option<String>,
start_incl: Option<String>,
start_excl: Option<String>,
end_incl: Option<String>,
end_excl: Option<String>,
}
/// Parse an NVD CVE-list response and keep the CVEs whose CODESYS CPE match covers
/// `runtime_version`.
fn parse_codesys_nvd(body: &serde_json::Value, runtime_version: &str) -> Vec<CodesysCve> {
let mut out = Vec::new();
let Some(vulns) = body["vulnerabilities"].as_array() else {
return out;
};
for v in vulns {
let cve = &v["cve"];
let Some(id) = cve["id"].as_str() else {
continue;
};
let covered = cve["configurations"]
.as_array()
.into_iter()
.flatten()
.flat_map(|c| c["nodes"].as_array().into_iter().flatten())
.flat_map(|n| n["cpeMatch"].as_array().into_iter().flatten())
.any(|cm| {
cm["vulnerable"].as_bool() == Some(true)
&& cm["criteria"]
.as_str()
.is_some_and(|c| c.contains(":codesys:"))
&& version_matches(runtime_version, &cpe_range(cm))
});
if covered {
let summary = cve["descriptions"]
.as_array()
.and_then(|d| d.iter().find(|x| x["lang"].as_str() == Some("en")))
.and_then(|x| x["value"].as_str())
.map(String::from);
let cvss = cve["metrics"]["cvssMetricV31"]
.as_array()
.and_then(|m| m.first())
.and_then(|m| m["cvssData"]["baseScore"].as_f64());
out.push(CodesysCve {
id: id.to_string(),
summary,
cvss,
});
}
}
out
}
/// Build a [`CpeRange`] from an NVD `cpeMatch` object.
fn cpe_range(cm: &serde_json::Value) -> CpeRange {
let exact = cm["criteria"]
.as_str()
.and_then(cpe_version)
.filter(|v| v != "*" && v != "-" && !v.is_empty());
CpeRange {
exact,
start_incl: cm["versionStartIncluding"].as_str().map(String::from),
start_excl: cm["versionStartExcluding"].as_str().map(String::from),
end_incl: cm["versionEndIncluding"].as_str().map(String::from),
end_excl: cm["versionEndExcluding"].as_str().map(String::from),
}
}
/// The version field (6th component) of a CPE 2.3 string.
fn cpe_version(criteria: &str) -> Option<String> {
criteria.split(':').nth(5).map(String::from)
}
/// Whether `v` satisfies a CPE version range.
fn version_matches(v: &str, r: &CpeRange) -> bool {
use std::cmp::Ordering::{Equal, Greater, Less};
if let Some(exact) = &r.exact {
return cmp_dotted(v, exact) == Equal;
}
let mut ok = true;
if let Some(s) = &r.start_incl {
ok &= cmp_dotted(v, s) != Less;
}
if let Some(s) = &r.start_excl {
ok &= cmp_dotted(v, s) == Greater;
}
if let Some(e) = &r.end_incl {
ok &= cmp_dotted(v, e) != Greater;
}
if let Some(e) = &r.end_excl {
ok &= cmp_dotted(v, e) == Less;
}
ok
}
/// Compare two dotted numeric versions (`4.17.0.0` vs `4.9.0.0`); missing
/// components count as 0, non-numeric components as 0.
fn cmp_dotted(a: &str, b: &str) -> std::cmp::Ordering {
let pa: Vec<u64> = a.split('.').map(|x| x.parse().unwrap_or(0)).collect();
let pb: Vec<u64> = b.split('.').map(|x| x.parse().unwrap_or(0)).collect();
for i in 0..pa.len().max(pb.len()) {
let x = pa.get(i).copied().unwrap_or(0);
let y = pb.get(i).copied().unwrap_or(0);
match x.cmp(&y) {
std::cmp::Ordering::Equal => continue,
other => return other,
}
}
std::cmp::Ordering::Equal
}
#[derive(serde::Deserialize)]
@@ -424,90 +228,3 @@ struct OsvVuln {
summary: Option<String>,
severity: Option<String>,
}
#[cfg(test)]
mod tests {
use super::*;
use std::cmp::Ordering::{Equal, Greater, Less};
fn entry(name: &str, ver: &str, pm: &str) -> SbomEntry {
SbomEntry::new("t".into(), name.into(), ver.into(), pm.into())
}
#[test]
fn finds_the_codesys_runtime_component() {
let entries = vec![
entry("Standard", "3.5.18.0", "codesys"),
entry("CODESYS Control for Linux ARM SL", "4.17.0.0", "codesys"),
];
assert_eq!(
codesys_runtime(&entries),
Some(("CODESYS Control for Linux ARM SL".into(), "4.17.0.0".into()))
);
// Internal library components are not the runtime.
assert!(codesys_runtime(&[entry("Util", "3.5.21.0", "codesys")]).is_none());
}
#[test]
fn dotted_version_comparison() {
assert_eq!(cmp_dotted("4.17.0.0", "4.9.0.0"), Greater);
assert_eq!(cmp_dotted("4.9.0.0", "4.17.0.0"), Less);
assert_eq!(cmp_dotted("3.5.18.0", "3.5.18.0"), Equal);
assert_eq!(cmp_dotted("4.2", "4.2.0.0"), Equal); // missing components = 0
}
#[test]
fn version_range_matching() {
let end_excl = CpeRange {
end_excl: Some("4.9.0.0".into()),
..Default::default()
};
assert!(!version_matches("4.17.0.0", &end_excl)); // patched
assert!(version_matches("4.5.0.0", &end_excl)); // affected
let exact = CpeRange {
exact: Some("3.5.16.0".into()),
..Default::default()
};
assert!(version_matches("3.5.16.0", &exact));
assert!(!version_matches("3.5.17.0", &exact));
let span = CpeRange {
start_incl: Some("3.0.0.0".into()),
end_incl: Some("3.5.16.0".into()),
..Default::default()
};
assert!(version_matches("3.5.16.0", &span));
assert!(!version_matches("3.5.17.0", &span));
}
#[test]
fn parses_nvd_and_matches_by_runtime_version() {
// Two CODESYS CVEs: one affects < 4.9 (our 4.17 is patched), one affects
// <= 4.20 (our 4.17 is affected). Only the latter should match.
let body = serde_json::json!({
"vulnerabilities": [
{"cve": {"id":"CVE-2023-0001",
"descriptions":[{"lang":"en","value":"old CmpBlkDrvTcp bug"}],
"metrics":{"cvssMetricV31":[{"cvssData":{"baseScore":7.5}}]},
"configurations":[{"nodes":[{"cpeMatch":[
{"vulnerable":true,
"criteria":"cpe:2.3:a:codesys:control_for_linux_sl:*:*:*:*:*:*:*:*",
"versionEndExcluding":"4.9.0.0"}
]}]}]}},
{"cve": {"id":"CVE-2024-0002",
"descriptions":[{"lang":"en","value":"recent runtime bug"}],
"metrics":{"cvssMetricV31":[{"cvssData":{"baseScore":9.8}}]},
"configurations":[{"nodes":[{"cpeMatch":[
{"vulnerable":true,
"criteria":"cpe:2.3:a:codesys:control_for_linux_sl:*:*:*:*:*:*:*:*",
"versionEndIncluding":"4.20.0.0"}
]}]}]}}
]
});
let matched = parse_codesys_nvd(&body, "4.17.0.0");
let ids: Vec<&str> = matched.iter().map(|c| c.id.as_str()).collect();
assert_eq!(ids, vec!["CVE-2024-0002"]);
assert_eq!(matched[0].cvss, Some(9.8));
}
}
@@ -1,152 +0,0 @@
//! Firmware SBOM via tramiton.
//!
//! Phase 2 (full, the default): drive a **reproducible build** with tramiton's
//! `NixBackend` — `analyze` → `seal_and_build` → a sealed lock whose libraries
//! are pinned and whose firmware artifact carries a content hash — then render
//! the SBOM from the lock plus deep binary SCA of pre-compiled inputs. This is
//! the complete bill of materials (toolchain + every fetched library + the
//! firmware image), the same one `tramiton sbom` produces.
//!
//! Phase 1 fallback (analysis-only): when no nix backend is available or the
//! build fails, fall back to the resolvable libraries + toolchain from the build
//! plan alone (no build). A scan therefore always yields *something*, and a nix
//! that can't run in the deployment never breaks a scan.
use std::path::Path;
use compliance_core::models::{SbomEntry, TargetType};
use tramiton_repro::ReproBackend;
use tramiton_sbom::ComponentKind;
/// Whether firmware SBOM applies to this target family.
pub fn is_firmware_target(target_type: TargetType) -> bool {
matches!(
target_type,
TargetType::FirmwareBareMetal | TargetType::FirmwareRtos | TargetType::EmbeddedLinuxYocto
)
}
/// Build SBOM entries for a firmware target from its source tree. Prefers a full
/// reproducible build (sealed lock); falls back to analysis-only. Returns an
/// empty vector when tramiton cannot even form a build plan.
pub async fn firmware_sbom_entries(path: &Path, repo_id: &str) -> Vec<SbomEntry> {
let p = path.to_path_buf();
let repo = repo_id.to_string();
// The whole analyze → seal → build → render sequence is blocking (it shells
// out to nix), so keep it off the async runtime. Bound it: a firmware build
// that hangs must not wedge the scan (the orphaned task is abandoned).
let handle = tokio::task::spawn_blocking(move || build_sbom_blocking(&p, &repo));
match tokio::time::timeout(std::time::Duration::from_secs(900), handle).await {
Ok(Ok(entries)) => entries,
Ok(Err(e)) => {
tracing::warn!(repo_id, error = %e, "Firmware SBOM: task join error");
Vec::new()
}
Err(_) => {
tracing::warn!(repo_id, "Firmware SBOM: build exceeded 15m; skipping");
Vec::new()
}
}
}
fn build_sbom_blocking(path: &Path, repo_id: &str) -> Vec<SbomEntry> {
let repo = tramiton_core::Repo::new(path);
let plan = match tramiton_core::provider::analyze(&repo) {
Ok(Some(bp)) => bp,
Ok(None) => return Vec::new(),
Err(e) => {
tracing::warn!(repo_id, error = %e, "Firmware SBOM: tramiton analyze failed");
return Vec::new();
}
};
// Phase 2: reproducible build → sealed lock → complete SBOM.
if let Some(backend) = tramiton_repro::NixBackend::detect() {
match tramiton_repro::seal_and_build(&backend, &plan, path) {
Ok(lock) => {
let mut sbom = tramiton_sbom::Sbom::from_lock(&lock, repo_id);
// Deep binary SCA of any pre-compiled inputs in the tree.
sbom.components.extend(tramiton_sbom::binary::scan(path));
let entries = sbom_to_entries(&sbom, repo_id);
tracing::info!(
repo_id,
backend = backend.name(),
count = entries.len(),
"Firmware SBOM: sealed reproducible build"
);
return entries;
}
Err(e) => {
tracing::warn!(repo_id, error = %e, "Firmware SBOM: reproducible build failed; falling back to analysis-only")
}
}
} else {
tracing::info!(
repo_id,
"Firmware SBOM: no nix backend available; analysis-only SBOM"
);
}
// Phase 1 fallback: analysis-only (toolchain + resolvable libraries).
analysis_entries(&plan, repo_id)
}
/// Map a rendered [`tramiton_sbom::Sbom`] (primary firmware + components) into
/// our [`SbomEntry`] rows. Source-file (`File`) components are dropped — they are
/// build inputs, not a dependency inventory.
fn sbom_to_entries(sbom: &tramiton_sbom::Sbom, repo_id: &str) -> Vec<SbomEntry> {
let mut entries = Vec::new();
if let Some(primary) = &sbom.primary {
entries.push(component_to_entry(primary, repo_id));
}
for c in &sbom.components {
if matches!(c.kind, ComponentKind::File) {
continue;
}
entries.push(component_to_entry(c, repo_id));
}
entries
}
fn component_to_entry(c: &tramiton_sbom::Component, repo_id: &str) -> SbomEntry {
let manager = match c.kind {
ComponentKind::Firmware => "firmware",
ComponentKind::Library => "library",
ComponentKind::Toolchain => "toolchain",
ComponentKind::File => "file",
};
let mut entry = SbomEntry::new(
repo_id.to_string(),
c.name.clone(),
c.version.clone().unwrap_or_default(),
manager.to_string(),
);
entry.purl = c.source.clone();
entry
}
/// Analysis-only components from the build plan: the cross-toolchain plus the
/// resolvable fetched libraries, without a build.
fn analysis_entries(bp: &tramiton_core::BuildPlan, repo_id: &str) -> Vec<SbomEntry> {
let mut entries = Vec::new();
if let Some(id) = bp.toolchain.id.clone() {
let version = bp.toolchain.version.clone().unwrap_or_default();
entries.push(SbomEntry::new(
repo_id.to_string(),
id,
version,
"toolchain".to_string(),
));
}
for lib in tramiton_repro::lock::libraries_from_inputs(&bp.inputs) {
let mut entry = SbomEntry::new(
repo_id.to_string(),
lib.name,
lib.revision,
"library".to_string(),
);
entry.purl = lib.source;
entries.push(entry);
}
entries
}
+2 -48
View File
@@ -80,10 +80,7 @@ impl GitOps {
#[tracing::instrument(skip_all, fields(repo_name = %repo_name))]
pub fn clone_or_fetch(&self, git_url: &str, repo_name: &str) -> Result<PathBuf, AgentError> {
// Names can contain slashes or other path-hostile characters (a target
// named after a repo path, say); collapse to one safe directory segment
// so the clone path never nests or breaks.
let repo_path = self.base_path.join(sanitize_repo_dir(repo_name));
let repo_path = self.base_path.join(repo_name);
if repo_path.exists() {
tracing::info!("fetching updates for existing repo");
@@ -138,7 +135,7 @@ impl GitOps {
/// Build credentials from agent config + per-repo overrides
pub fn make_repo_credentials(
config: &compliance_core::AgentConfig,
repo: &crate::pipeline::repo_view::RepoView,
repo: &compliance_core::models::TrackedRepository,
) -> RepoCredentials {
RepoCredentials {
ssh_key_path: Some(config.ssh_key_path.clone()),
@@ -256,46 +253,3 @@ pub struct DiffFile {
pub path: String,
pub hunks: String,
}
/// Collapse a repository name into a single filesystem-safe directory segment.
/// Names may carry slashes or other path-hostile characters (a target named
/// after a repo path, for instance); those would otherwise nest or break the
/// clone path, so map anything outside `[A-Za-z0-9._-]` to `_`.
fn sanitize_repo_dir(name: &str) -> String {
let mapped: String = name
.chars()
.map(|c| {
if c.is_ascii_alphanumeric() || c == '-' || c == '_' || c == '.' {
c
} else {
'_'
}
})
.collect();
let trimmed = mapped.trim_matches(|c| c == '.' || c == '_');
if trimmed.is_empty() {
"repo".to_string()
} else {
trimmed.to_string()
}
}
#[cfg(test)]
mod tests {
use super::sanitize_repo_dir;
#[test]
fn sanitizes_path_hostile_names() {
assert_eq!(
sanitize_repo_dir("zephyr-example-app"),
"zephyr-example-app"
);
assert_eq!(
sanitize_repo_dir("ChristianRinn/bare_metal_stm32f411xe"),
"ChristianRinn_bare_metal_stm32f411xe"
);
assert_eq!(sanitize_repo_dir("../../etc/passwd"), "etc_passwd");
assert_eq!(sanitize_repo_dir("a b:c"), "a_b_c");
assert_eq!(sanitize_repo_dir("///"), "repo");
}
}
@@ -1,6 +1,5 @@
use mongodb::bson::doc;
use crate::pipeline::repo_view::RepoView;
use compliance_core::models::*;
use super::orchestrator::{extract_base_url, PipelineOrchestrator};
@@ -11,7 +10,7 @@ use crate::trackers;
impl PipelineOrchestrator {
/// Build an issue tracker client from a repository's tracker configuration.
/// Returns `None` if the repo has no tracker configured.
pub(super) fn build_tracker(&self, repo: &RepoView) -> Option<TrackerDispatch> {
pub(super) fn build_tracker(&self, repo: &TrackedRepository) -> Option<TrackerDispatch> {
let tracker_type = repo.tracker_type.as_ref()?;
// Per-repo token takes precedence, fall back to global config
match tracker_type {
@@ -82,7 +81,7 @@ impl PipelineOrchestrator {
#[tracing::instrument(skip_all, fields(repo_id = %repo_id))]
pub(super) async fn create_tracker_issues(
&self,
repo: &RepoView,
repo: &TrackedRepository,
repo_id: &str,
new_findings: &[Finding],
) -> Result<(), AgentError> {
-3
View File
@@ -1,7 +1,6 @@
pub mod code_review;
pub mod cve;
pub mod dedup;
pub mod firmware_sbom;
pub mod git;
pub mod gitleaks;
mod graph_build;
@@ -10,9 +9,7 @@ pub mod lint;
pub mod orchestrator;
pub mod patterns;
pub mod plan;
pub mod plc;
mod pr_review;
pub mod repo_view;
pub mod sbom;
pub mod semgrep;
mod tracker_dispatch;
+185 -587
View File
@@ -16,7 +16,6 @@ use crate::pipeline::gitleaks::GitleaksScanner;
use crate::pipeline::lint::LintScanner;
use crate::pipeline::patterns::{GdprPatternScanner, OAuthPatternScanner};
use crate::pipeline::plan::build_scan_plan;
use crate::pipeline::repo_view::RepoView;
use crate::pipeline::sbom::SbomScanner;
use crate::pipeline::semgrep::SemgrepScanner;
@@ -52,8 +51,72 @@ impl PipelineOrchestrator {
}
}
#[tracing::instrument(skip_all, fields(repo_id = %repo_id, trigger = ?trigger))]
pub async fn run(&self, repo_id: &str, trigger: ScanTrigger) -> Result<(), AgentError> {
// Look up the repository
let repo = self
.db
.repositories()
.find_one(doc! { "_id": mongodb::bson::oid::ObjectId::parse_str(repo_id).map_err(|e| AgentError::Other(e.to_string()))? })
.await?
.ok_or_else(|| AgentError::Other(format!("Repository {repo_id} not found")))?;
// Create scan run
let scan_run = ScanRun::new(repo_id.to_string(), trigger);
let insert = self.db.scan_runs().insert_one(&scan_run).await?;
let scan_run_id = insert
.inserted_id
.as_object_id()
.map(|id| id.to_hex())
.unwrap_or_default();
let result = self.run_pipeline(&repo, &scan_run_id).await;
// Update scan run status
match &result {
Ok(count) => {
self.db
.scan_runs()
.update_one(
doc! { "_id": &insert.inserted_id },
doc! {
"$set": {
"status": "completed",
"current_phase": "completed",
"new_findings_count": *count as i64,
"completed_at": mongodb::bson::DateTime::now(),
}
},
)
.await?;
}
Err(e) => {
tracing::error!(repo_id, error = %e, "Scan pipeline failed");
self.db
.scan_runs()
.update_one(
doc! { "_id": &insert.inserted_id },
doc! {
"$set": {
"status": "failed",
"error_message": e.to_string(),
"completed_at": mongodb::bson::DateTime::now(),
}
},
)
.await?;
}
}
result.map(|_| ())
}
#[tracing::instrument(skip_all, fields(repo_id = repo.name.as_str()))]
async fn run_pipeline(&self, repo: &RepoView, scan_run_id: &str) -> Result<u32, AgentError> {
async fn run_pipeline(
&self,
repo: &TrackedRepository,
scan_run_id: &str,
) -> Result<u32, AgentError> {
let repo_id = repo.id.as_ref().map(|id| id.to_hex()).unwrap_or_default();
// Stage 0: Change detection
@@ -67,6 +130,7 @@ impl PipelineOrchestrator {
return Ok(0);
}
let current_sha = GitOps::get_head_sha(&repo_path)?;
let mut all_findings: Vec<Finding> = Vec::new();
// Stage 1: Semgrep SAST
@@ -215,72 +279,8 @@ impl PipelineOrchestrator {
.await;
tracing::info!("[{repo_id}] Triaged: {triaged} findings passed confidence threshold");
// Stage 5b: control triage — stamp findings with the compliance control(s)
// they're evidence for and flag control false positives (grounded LLM over
// deterministic tool output). No-op unless breakpilot is configured.
self.update_phase(scan_run_id, "control_triage").await;
let tagged = crate::controls::triage_repo_findings(
&self.config,
self.llm.clone(),
&repo_path,
&mut all_findings,
)
.await;
if tagged > 0 {
tracing::info!("[{repo_id}] Control triage tagged {tagged} findings with control refs");
}
// Stage 5c: semantic control mapping — scale path for the master-controls
// corpus (no CWE to LUT on): embed each finding's region, retrieve the
// nearest master controls, grounded-judge, and stamp confirmed refs. On by
// default (validated live); the corpus embedding is cached so only the
// first scan after a catalog change pays it.
if self.config.breakpilot.semantic_mapping {
self.update_phase(scan_run_id, "semantic_control_mapping")
.await;
let sem = crate::controls::semantic_stamp_findings(
&self.config,
self.llm.clone(),
&repo_path,
&mut all_findings,
)
.await;
if sem > 0 {
tracing::info!(
"[{repo_id}] Semantic mapping tagged {sem} findings with master-control refs"
);
}
}
// Stage 5d: grounded surface checks — the absence-based controls (no
// rate limiting, no security logging, no update-signature check) have no
// syntactic pattern to match, so we retrieve the code surface each governs
// and let the grounded judge decide whether it holds, producing net-new
// findings already tagged + grounded. On by default (validated live); it
// covers the 8 absence-based CRA controls.
if self.config.breakpilot.grounded_control_checks {
self.update_phase(scan_run_id, "grounded_control_checks")
.await;
let grounded = crate::controls::grounded_surface_findings(
&self.config,
self.llm.clone(),
&repo_path,
&repo_id,
)
.await;
if !grounded.is_empty() {
tracing::info!(
"[{repo_id}] Grounded surface checks raised {} control findings",
grounded.len()
);
all_findings.extend(grounded);
}
}
// Dedup against existing findings: insert first-seen ones, and refresh the
// control mappings on ones we've seen before.
// Dedup against existing findings and insert new ones
let mut new_count = 0u32;
let mut refreshed_count = 0u32;
let mut new_findings: Vec<Finding> = Vec::new();
for mut finding in all_findings {
finding.scan_run_id = Some(scan_run_id.to_string());
@@ -295,25 +295,8 @@ impl PipelineOrchestrator {
finding.id = result.inserted_id.as_object_id();
new_findings.push(finding);
new_count += 1;
} else if !finding.control_refs.is_empty() {
// Re-scan refresh: a mapping pass (newly enabled or tuned) computed
// control_refs for a finding first seen before mapping ran. Persist
// them onto the existing row — the insert path alone never would.
self.db
.findings()
.update_one(
doc! { "fingerprint": &finding.fingerprint },
doc! { "$set": { "control_refs": finding.control_refs.clone() } },
)
.await?;
refreshed_count += 1;
}
}
if refreshed_count > 0 {
tracing::info!(
"[{repo_id}] Refreshed control_refs on {refreshed_count} existing findings"
);
}
// Remove stale SBOM entries for this repo before reinserting
if !sbom_entries.is_empty() {
@@ -340,12 +323,67 @@ impl PipelineOrchestrator {
.await?;
}
// Persist CVE alerts and create notifications (shared with the PLC path).
let new_notif_count = self
.persist_cve_alerts(&repo_id, &repo.name, &cve_alerts)
.await?;
if new_notif_count > 0 {
tracing::info!("[{repo_id}] Created {new_notif_count} CVE notification(s)");
// Persist CVE alerts and create notifications
{
use compliance_core::models::notification::{parse_severity, CveNotification};
let repo_name = repo.name.clone();
let mut new_notif_count = 0u32;
for alert in &cve_alerts {
// Upsert the alert
let filter = doc! {
"cve_id": &alert.cve_id,
"repo_id": &alert.repo_id,
};
let update = mongodb::bson::to_document(alert)
.map(|d| doc! { "$set": d })
.unwrap_or_else(|_| doc! {});
self.db
.cve_alerts()
.update_one(filter, update)
.upsert(true)
.await?;
// Create notification (dedup by cve_id + repo + package + version)
let notif_filter = doc! {
"cve_id": &alert.cve_id,
"repo_id": &alert.repo_id,
"package_name": &alert.affected_package,
"package_version": &alert.affected_version,
};
let severity = parse_severity(alert.severity.as_deref(), alert.cvss_score);
let mut notification = CveNotification::new(
alert.cve_id.clone(),
repo_id.clone(),
repo_name.clone(),
alert.affected_package.clone(),
alert.affected_version.clone(),
severity,
);
notification.cvss_score = alert.cvss_score;
notification.summary = alert.summary.clone();
notification.url = Some(format!("https://osv.dev/vulnerability/{}", alert.cve_id));
let notif_update = doc! {
"$setOnInsert": mongodb::bson::to_bson(&notification).unwrap_or_default()
};
if let Ok(result) = self
.db
.cve_notifications()
.update_one(notif_filter, notif_update)
.upsert(true)
.await
{
if result.upserted_id.is_some() {
new_notif_count += 1;
}
}
}
if new_notif_count > 0 {
tracing::info!("[{repo_id}] Created {new_notif_count} CVE notification(s)");
}
}
// Stage 6: Issue Creation
@@ -358,9 +396,20 @@ impl PipelineOrchestrator {
tracing::warn!("[{repo_id}] Issue creation failed: {e}");
}
// The onboarded target's findings_count and the git artifact's
// last_scanned_commit watermark are persisted by `finalize_target` after
// `run_pipeline` returns.
// Stage 7: Update repository
self.db
.repositories()
.update_one(
doc! { "_id": repo.id },
doc! {
"$set": {
"last_scanned_commit": &current_sha,
"updated_at": mongodb::bson::DateTime::now(),
},
"$inc": { "findings_count": new_count as i64 },
},
)
.await?;
// Stage 8: DAST (async, optional — only if a DastTarget is configured)
tracing::info!("[{repo_id}] Stage 8: Checking for DAST targets");
@@ -475,474 +524,33 @@ impl PipelineOrchestrator {
// wizard-created targets, not just migrated ones.
self.ensure_dast_target(target, &plan).await;
// PLC/SPS targets: the control-logic scan consumes the PLC source (an
// uploaded PlcProject *or* a git repo / source archive of PLCopen XML / ST
// exports), so it takes over the code artifact — we don't also run the
// SAST pipeline over it. A PLC device is reachable, so DAST still runs
// against a WebVisu / exposed endpoint when one is provisioned.
let mut new_count = 0u32;
let plc = plan.has(ScanType::PlcControlLogic);
let ics = plan.has(ScanType::IcsProbe);
if plc {
new_count += self.run_plc_scan(target, &target_id, scan_run_id).await?;
// Provision-and-test (#183): with the control logic but no reachable
// device, instantiate it on an ephemeral soft-PLC and probe that
// instead of the customer's OT network. Opt-in (needs Docker) and only
// when there is no live URL to probe directly. Never fails the scan.
if self.config.plc_runtime.enabled && target.live_url().is_none() {
match self
.run_provisioned_plc_test(target, &target_id, scan_run_id)
.await
{
Ok(n) => new_count += n,
Err(e) => {
tracing::warn!(target_id = %target_id, error = %e, "provision-and-test failed")
}
}
}
}
if ics {
new_count += self.run_ics_probe(target, &target_id, scan_run_id).await?;
}
if plc || ics {
// PLC/SPS device: also DAST against a WebVisu / exposed endpoint, but
// only when DAST is actually planned — a device reachable only over an
// industrial protocol (e.g. modbus://) has no web surface to crawl, and
// running DAST there just fails at reconnaissance. Gating here (not only
// at provisioning) also stops a DAST target left over from an earlier
// run from re-triggering. The control-logic scan already consumed the
// code artifact, so the SAST pipeline is not re-run.
if plan.has(ScanType::Dast) {
self.update_phase(scan_run_id, "dast_scanning").await;
self.maybe_trigger_dast(&target_id, scan_run_id).await;
}
return Ok(new_count);
}
match target.code_artifact() {
Some(code) if code.kind == ArtifactKind::GitRepo => {
let repo = RepoView::from_target(target, code);
let n = self.run_pipeline(&repo, scan_run_id).await?;
self.finalize_target(target, &repo, n).await?;
new_count += n;
let repo = repo_view_from_target(target, code);
let new_count = self.run_pipeline(&repo, scan_run_id).await?;
self.finalize_target(target, &repo, new_count).await?;
Ok(new_count)
}
Some(_) => {
tracing::warn!(
target_id = %target_id,
"Unified pipeline: source-archive scanning not yet wired; skipping"
);
Ok(0)
}
None => {
// No code to scan (a migrated DAST target). Firmware/mobile static
// scanners land in #128/#129; DAST for a running URL works when a
// DastTarget row exists (provisioned above from a LiveUrl, or from
// a migrated target).
// No code to scan. Firmware/PLC/mobile static scanners land in
// #128/#129/#130; DAST for a running URL still works when a
// DastTarget row exists (migrated targets).
tracing::info!(
target_id = %target_id,
"Unified pipeline: no code artifact; attempting DAST"
"Unified pipeline: no code artifact; attempting DAST only"
);
self.update_phase(scan_run_id, "dast_scanning").await;
self.maybe_trigger_dast(&target_id, scan_run_id).await;
Ok(0)
}
}
Ok(new_count)
}
/// Analyze a PLC/SPS project (Structured Text / PLCopen XML) for
/// control-logic security issues and persist the new findings.
async fn run_plc_scan(
&self,
target: &OnboardedTarget,
target_id: &str,
scan_run_id: &str,
) -> Result<u32, AgentError> {
tracing::info!(target_id, "[{target_id}] PLC control-logic analysis");
self.update_phase(scan_run_id, "plc_analysis").await;
let ctx = crate::ingest::IngestContext::from_config(&self.config, target_id);
let ingest_set = crate::ingest::ingest_all(target, &ctx)?;
// Every PLC-source artifact on the target: dedicated PLC projects plus any
// code artifacts (git repo / source archive) holding PLCopen XML / ST
// exports. A target can carry several (e.g. one POU export per file).
let sources: Vec<&Artifact> = target
.artifacts
.iter()
.filter(|a| {
matches!(
a.kind,
ArtifactKind::PlcProject | ArtifactKind::GitRepo | ArtifactKind::SourceArchive
)
})
.collect();
if sources.is_empty() {
tracing::warn!(target_id, "PLC scan: no PLC source artifact");
return Ok(0);
}
let mut all_findings = Vec::new();
let mut all_sbom: Vec<SbomEntry> = Vec::new();
let mut sbom_seen = std::collections::BTreeSet::new();
for a in &sources {
let Some(path) = ingest_set.get(&a.id).and_then(|ia| ia.working_path.clone()) else {
continue;
};
let mut source_findings = crate::pipeline::plc::analyze_tree(&path, target_id);
// Control mapping for the PLC path (run_plc_scan is separate from
// run_pipeline, which does its own mapping). PLC findings carry
// file_path/line/cwe, so the semantic pass reads each region under this
// source's `path` and stamps master-control refs. The LUT + grounded
// surface passes are code-pattern / CRA-specific and don't apply to
// IEC 61131-3 control logic, so only the semantic pass runs here.
crate::controls::semantic_stamp_findings(
&self.config,
self.llm.clone(),
&path,
&mut source_findings,
)
.await;
all_findings.extend(source_findings);
// Control-application SBOM: CODESYS libraries + runtime from a
// `.projectarchive` (uploaded, or committed in the working tree).
let archive = a
.stored_path
.clone()
.unwrap_or_else(|| a.source_ref.clone());
for e in crate::pipeline::plc::sbom::collect_sbom(
std::path::Path::new(&archive),
&path,
target_id,
) {
if sbom_seen.insert((e.name.clone(), e.version.clone())) {
all_sbom.push(e);
}
}
}
tracing::info!(
target_id,
artifacts = sources.len(),
found = all_findings.len(),
"PLC control-logic analysis complete"
);
let mut new_count = 0u32;
let mut refreshed_count = 0u32;
for mut finding in all_findings {
finding.scan_run_id = Some(scan_run_id.to_string());
if self
.db
.findings()
.find_one(doc! { "fingerprint": &finding.fingerprint })
.await?
.is_none()
{
self.db.findings().insert_one(&finding).await?;
new_count += 1;
} else if !finding.control_refs.is_empty() {
// Re-scan refresh: mirror run_pipeline — persist newly-computed
// control_refs onto a PLC finding first seen before the semantic
// pass ran. The insert path alone never would, so without this a
// PLC re-scan can only pick up mappings via a delete + re-add.
self.db
.findings()
.update_one(
doc! { "fingerprint": &finding.fingerprint },
doc! { "$set": { "control_refs": finding.control_refs.clone() } },
)
.await?;
refreshed_count += 1;
}
}
if refreshed_count > 0 {
tracing::info!(
target_id,
"Refreshed control_refs on {refreshed_count} existing PLC findings"
);
}
if !all_sbom.is_empty() {
if let Err(e) = self
.persist_control_app_sbom(target_id, &target.name, all_sbom)
.await
{
tracing::warn!(target_id, error = %e, "control-app SBOM persist failed");
}
}
Ok(new_count)
}
/// Provision-and-test (#183): instantiate the target's control logic on an
/// ephemeral soft-PLC (OpenPLC), start it, probe the provisioned Modbus
/// endpoint, and tear the instance down. Used when a PLC/SPS target has the
/// control logic but no reachable live device to probe directly. Guarded by
/// `plc_runtime.enabled` (needs Docker); persists the same [`ScanType::IcsProbe`]
/// findings as a live probe.
async fn run_provisioned_plc_test(
&self,
target: &OnboardedTarget,
target_id: &str,
scan_run_id: &str,
) -> Result<u32, AgentError> {
self.update_phase(scan_run_id, "plc_provision").await;
// Locate a loadable control-logic program among the PLC-source artifacts
// (same selection as the static PLC scan: dedicated PLC projects plus code
// artifacts holding PLCopen XML / ST exports).
let ctx = crate::ingest::IngestContext::from_config(&self.config, target_id);
let ingest_set = crate::ingest::ingest_all(target, &ctx)?;
let program = target
.artifacts
.iter()
.filter(|a| {
matches!(
a.kind,
ArtifactKind::PlcProject | ArtifactKind::GitRepo | ArtifactKind::SourceArchive
)
})
.find_map(|a| {
let path = ingest_set
.get(&a.id)
.and_then(|ia| ia.working_path.clone())?;
werkbank_exec::plc::extract_program(&path)
});
let Some(program) = program else {
tracing::info!(
target_id,
"provision-and-test: no loadable control-logic program"
);
return Ok(0);
};
let http = werkbank_exec::plc::http_client()?;
let provisioner = werkbank_exec::plc::DockerSoftPlc::new(self.config.plc_runtime.clone());
let outcome = werkbank_exec::plc::provision_and_test(
&provisioner,
&http,
&self.config.plc_runtime,
&program,
target_id,
)
.await?;
tracing::info!(
target_id,
found = outcome.findings.len(),
dast = outcome.dast.is_some(),
"provision-and-test complete"
);
let mut new_count = 0u32;
for mut finding in outcome.findings {
finding.scan_run_id = Some(scan_run_id.to_string());
if self
.db
.findings()
.find_one(doc! { "fingerprint": &finding.fingerprint })
.await?
.is_none()
{
self.db.findings().insert_one(&finding).await?;
new_count += 1;
}
}
// Persist the DAST scan of the provisioned web endpoint, linked to this
// scan run (mirrors `maybe_trigger_dast`).
if let Some(dast) = outcome.dast {
let mut scan_run = dast.scan_run;
scan_run.sast_scan_run_id = Some(scan_run_id.to_string());
if let Err(e) = self.db.dast_scan_runs().insert_one(&scan_run).await {
tracing::warn!(target_id, error = %e, "failed to store provisioned DAST scan run");
}
for finding in &dast.findings {
if let Err(e) = self.db.dast_findings().insert_one(finding).await {
tracing::warn!(target_id, error = %e, "failed to store provisioned DAST finding");
}
}
}
Ok(new_count)
}
/// Probe a running PLC/SPS device over industrial protocols (Modbus/TCP, …)
/// and persist findings for exposed / unauthenticated control access. The
/// probe is read-only; it targets the Modbus port of the target's live URL.
async fn run_ics_probe(
&self,
target: &OnboardedTarget,
target_id: &str,
scan_run_id: &str,
) -> Result<u32, AgentError> {
self.update_phase(scan_run_id, "ics_probe").await;
let Some(endpoint) = target.live_url().map(|a| a.source_ref.clone()) else {
tracing::warn!(target_id, "ICS probe: no live URL");
return Ok(0);
};
// Short per-request budget so an unreachable device doesn't stall the scan.
let budget = std::time::Duration::from_secs(5);
let findings = werkbank_exec::ics::probe_target(&endpoint, target_id, budget).await;
tracing::info!(
target_id,
endpoint = %endpoint,
found = findings.len(),
"ICS probe complete"
);
let mut new_count = 0u32;
for mut finding in findings {
finding.scan_run_id = Some(scan_run_id.to_string());
if self
.db
.findings()
.find_one(doc! { "fingerprint": &finding.fingerprint })
.await?
.is_none()
{
self.db.findings().insert_one(&finding).await?;
new_count += 1;
}
}
Ok(new_count)
}
/// Store a control-application SBOM (CODESYS libraries + runtime) for a target
/// and match it against known CVEs. Scoped to `package_manager = "codesys"` so
/// it refreshes on re-scan and coexists with any firmware/source SBOM. The
/// runtime `Cmp*` / `3SLicense` components carry real CODESYS advisories, so
/// this is where PLC-device CVE coverage comes from.
async fn persist_control_app_sbom(
&self,
target_id: &str,
target_name: &str,
mut entries: Vec<SbomEntry>,
) -> Result<(), AgentError> {
if entries.is_empty() {
return Ok(());
}
self.db
.sbom_entries()
.delete_many(doc! { "repo_id": target_id, "package_manager": "codesys" })
.await?;
let cve_scanner = CveScanner::new(
self.http.clone(),
self.config.searxng_url.clone(),
self.config.nvd_api_key.as_ref().map(|k| {
use secrecy::ExposeSecret;
k.expose_secret().to_string()
}),
);
let mut alerts = match tokio::time::timeout(
std::time::Duration::from_secs(600),
cve_scanner.scan_dependencies(target_id, &mut entries),
)
.await
{
Ok(Ok(a)) => a,
Ok(Err(e)) => {
tracing::warn!(target_id, error = %e, "control-app CVE scan failed");
Vec::new()
}
Err(_) => {
tracing::warn!(target_id, "control-app CVE scan timed out");
Vec::new()
}
};
// OSV can't match `pkg:codesys/*` (no such ecosystem); CODESYS advisories
// live in NVD keyed by CPE + runtime version. Add those (best-effort).
if let Ok(codesys) = tokio::time::timeout(
std::time::Duration::from_secs(120),
cve_scanner.scan_codesys(target_id, &mut entries),
)
.await
{
alerts.extend(codesys);
} else {
tracing::warn!(target_id, "CODESYS CVE match timed out");
}
for entry in &entries {
let filter = doc! {
"repo_id": &entry.repo_id,
"name": &entry.name,
"version": &entry.version,
};
if let Ok(d) = mongodb::bson::to_document(entry) {
self.db
.sbom_entries()
.update_one(filter, doc! { "$set": d })
.upsert(true)
.await?;
}
}
let new_notifs = self
.persist_cve_alerts(target_id, target_name, &alerts)
.await?;
tracing::info!(
target_id,
components = entries.len(),
alerts = alerts.len(),
notifications = new_notifs,
"control-app SBOM stored"
);
Ok(())
}
/// Upsert CVE alerts for a target and create dedup'd CVE notifications;
/// returns the number of newly-created notifications. Shared by the SAST
/// pipeline and the PLC control-app SBOM path, so every SBOM source (source,
/// firmware, CODESYS libraries/runtime) raises the same notifications.
async fn persist_cve_alerts(
&self,
repo_id: &str,
repo_name: &str,
alerts: &[CveAlert],
) -> Result<u32, AgentError> {
use compliance_core::models::notification::{parse_severity, CveNotification};
let mut new_notif = 0u32;
for alert in alerts {
let filter = doc! { "cve_id": &alert.cve_id, "repo_id": &alert.repo_id };
let update = mongodb::bson::to_document(alert)
.map(|d| doc! { "$set": d })
.unwrap_or_else(|_| doc! {});
self.db
.cve_alerts()
.update_one(filter, update)
.upsert(true)
.await?;
// Dedup notifications by cve + repo + package + version.
let notif_filter = doc! {
"cve_id": &alert.cve_id,
"repo_id": &alert.repo_id,
"package_name": &alert.affected_package,
"package_version": &alert.affected_version,
};
let severity = parse_severity(alert.severity.as_deref(), alert.cvss_score);
let mut notification = CveNotification::new(
alert.cve_id.clone(),
repo_id.to_string(),
repo_name.to_string(),
alert.affected_package.clone(),
alert.affected_version.clone(),
severity,
);
notification.cvss_score = alert.cvss_score;
notification.summary = alert.summary.clone();
notification.url = Some(format!("https://osv.dev/vulnerability/{}", alert.cve_id));
let notif_update = doc! {
"$setOnInsert": mongodb::bson::to_bson(&notification).unwrap_or_default()
};
if let Ok(result) = self
.db
.cve_notifications()
.update_one(notif_filter, notif_update)
.upsert(true)
.await
{
if result.upserted_id.is_some() {
new_notif += 1;
}
}
}
Ok(new_notif)
}
/// Ingest the target's artifacts, classify (tramiton for firmware/RTOS/Yocto,
@@ -993,47 +601,6 @@ impl PipelineOrchestrator {
tracing::warn!(target_id, error = %e, "Unified pipeline: classification failed")
}
}
// Analysis-based firmware SBOM: for embedded targets, derive components
// (resolved libraries + cross-toolchain) from tramiton's build-plan
// analysis over the already-ingested source — no build, no binary
// upload. Best-effort; empty when no build plan forms.
if crate::pipeline::firmware_sbom::is_firmware_target(target.target_type) {
if let Some(code) = target.code_artifact() {
if let Some(path) = working_paths.get(&code.id) {
let entries =
crate::pipeline::firmware_sbom::firmware_sbom_entries(path, target_id)
.await;
if !entries.is_empty() {
let _ = self
.db
.sbom_entries()
.delete_many(doc! { "repo_id": target_id })
.await;
for entry in &entries {
let filter = doc! {
"repo_id": &entry.repo_id,
"name": &entry.name,
"version": &entry.version,
};
if let Ok(d) = mongodb::bson::to_document(entry) {
let _ = self
.db
.sbom_entries()
.update_one(filter, doc! { "$set": d })
.upsert(true)
.await;
}
}
tracing::info!(
target_id,
count = entries.len(),
"Firmware SBOM: stored components from tramiton analysis"
);
}
}
}
}
}
/// If the target has a `LiveUrl` artifact and DAST is planned, provision a
@@ -1095,7 +662,7 @@ impl PipelineOrchestrator {
async fn finalize_target(
&self,
target: &OnboardedTarget,
repo: &RepoView,
repo: &TrackedRepository,
new_count: u32,
) -> Result<(), AgentError> {
let oid = match target.id {
@@ -1143,6 +710,37 @@ impl PipelineOrchestrator {
}
}
/// Build a legacy `TrackedRepository` view from an onboarded target's code
/// artifact, so the unified pipeline can reuse the existing repo pipeline. The
/// inverse of the migration's `repo_to_target`. `_id` is preserved so findings
/// and DAST lookups resolve against the same key.
fn repo_view_from_target(target: &OnboardedTarget, code: &Artifact) -> TrackedRepository {
let mut repo = TrackedRepository::new(target.name.clone(), code.source_ref.clone());
repo.id = target.id;
if let Some(git) = &code.git {
repo.default_branch = git.default_branch.clone();
repo.last_scanned_commit = git.last_scanned_commit.clone();
repo.local_path = git.local_path.clone();
}
if let Some(auth) = &code.auth {
repo.auth_token = auth.secret.clone();
repo.auth_username = auth.username.clone();
}
if let Some(it) = &target.scan_config.issue_tracker {
repo.tracker_type = it.tracker_type.clone();
repo.tracker_owner = it.owner.clone();
repo.tracker_repo = it.repo.clone();
repo.tracker_token = it.token.clone();
}
repo.scan_schedule = target.scan_schedule.clone();
repo.webhook_enabled = target.webhook_enabled;
repo.webhook_secret = target.webhook_secret.clone();
repo.findings_count = target.findings_count;
repo.created_at = target.created_at;
repo.updated_at = target.updated_at;
repo
}
/// Extract the scheme + host from a git URL.
/// e.g. "https://gitea.example.com/owner/repo.git" -> "https://gitea.example.com"
/// e.g. "ssh://git@gitea.example.com:22/owner/repo.git" -> "https://gitea.example.com"
@@ -1201,7 +799,7 @@ mod tests {
target.artifacts.push(artifact);
let code = target.code_artifact().expect("code artifact");
let repo = RepoView::from_target(&target, code);
let repo = repo_view_from_target(&target, code);
assert_eq!(repo.id, target.id); // preserved
assert_eq!(repo.git_url, "https://git/acme.git");
+1 -20
View File
@@ -75,16 +75,10 @@ pub fn build_scan_plan(target: &OnboardedTarget) -> ScanPlan {
}
/// Resolve the artifact a scan consumes. A "code" requirement (represented by
/// `GitRepo`) is satisfied by a git repo *or* a source archive. The PLC
/// control-logic requirement (represented by `PlcProject`) prefers an uploaded
/// PLC project but also accepts a code artifact — a git repo / source archive
/// holding PLCopen XML / ST exports.
/// `GitRepo`) is satisfied by a git repo *or* a source archive.
fn resolve_artifact(target: &OnboardedTarget, required: Option<ArtifactKind>) -> Option<&Artifact> {
match required {
Some(ArtifactKind::GitRepo) => target.code_artifact(),
Some(ArtifactKind::PlcProject) => target
.first_of(ArtifactKind::PlcProject)
.or_else(|| target.code_artifact()),
Some(kind) => target.first_of(kind),
None => target.code_artifact().or_else(|| target.artifacts.first()),
}
@@ -106,7 +100,6 @@ fn phase_for(scan: ScanType) -> ScanPhase {
ScanType::PlcControlLogic => ScanPhase::PlcAnalysis,
ScanType::MobileStatic => ScanPhase::MobileStatic,
ScanType::ContainerScan => ScanPhase::ContainerScan,
ScanType::IcsProbe => ScanPhase::IcsProbe,
}
}
@@ -181,18 +174,6 @@ mod tests {
assert_eq!(plan.steps[0].phase, ScanPhase::PlcAnalysis);
}
#[test]
fn plc_control_logic_binds_to_a_git_repo() {
// A CODESYS project in git (PLCopen XML / ST exports) with no uploaded
// PlcProject: control-logic still plans, bound to the git artifact.
let git = Artifact::git_repo("https://git/plc", "main");
let git_id = git.id.clone();
let t = target(TargetType::PlcSps, vec![git]);
let plan = build_scan_plan(&t);
let step = step_for(&plan, ScanType::PlcControlLogic).expect("control-logic planned");
assert_eq!(step.artifact_id, git_id, "PLC scan binds to the git repo");
}
#[test]
fn disabled_scan_is_dropped_and_off_by_default_can_be_enabled() {
let mut t = target(TargetType::WebApp, vec![Artifact::git_repo("u", "main")]);
-226
View File
@@ -1,226 +0,0 @@
//! Abstract syntax tree for IEC 61131-3 Structured Text (ST).
//!
//! This is the security-relevant subset: POUs with their variable declarations
//! and statement bodies, enough to run semantic control-logic rules over. It is
//! deliberately not a full language model — declarations we don't reason about
//! (e.g. exotic type definitions) are parsed loosely and kept as raw text.
/// A Program Organization Unit: a PROGRAM, FUNCTION, or FUNCTION_BLOCK.
#[derive(Debug, Clone)]
pub struct Pou {
pub name: String,
pub kind: PouKind,
/// The declared variables, across all VAR_* sections.
pub vars: Vec<VarDecl>,
/// The statement body.
pub body: Vec<Stmt>,
/// 1-based line where the POU header appears (in the source that was parsed).
pub line: u32,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum PouKind {
Program,
Function,
FunctionBlock,
}
impl PouKind {
pub fn label(self) -> &'static str {
match self {
PouKind::Program => "PROGRAM",
PouKind::Function => "FUNCTION",
PouKind::FunctionBlock => "FUNCTION_BLOCK",
}
}
}
/// A single declared variable.
#[derive(Debug, Clone)]
pub struct VarDecl {
pub name: String,
pub section: VarSection,
/// The declared type as written (e.g. `BOOL`, `INT`, `ARRAY[0..9] OF INT`).
pub type_name: String,
/// Whether the type is an ARRAY, and its declared bounds `(lo, hi)` when
/// they are literal integers — used by the array-bounds rule.
pub array_bounds: Option<(i64, i64)>,
/// The initializer expression, if any (`:= <expr>`).
pub init: Option<Expr>,
pub line: u32,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum VarSection {
Var,
Input,
Output,
InOut,
Global,
Temp,
External,
}
/// A statement.
#[derive(Debug, Clone)]
pub enum Stmt {
Assign {
target: Expr,
value: Expr,
line: u32,
},
If {
/// (condition, body) for IF and each ELSIF, in order.
branches: Vec<(Expr, Vec<Stmt>)>,
else_body: Option<Vec<Stmt>>,
line: u32,
},
Case {
selector: Expr,
/// (label expressions, body) per CASE arm.
arms: Vec<(Vec<Expr>, Vec<Stmt>)>,
else_body: Option<Vec<Stmt>>,
line: u32,
},
For {
var: String,
from: Expr,
to: Expr,
by: Option<Expr>,
body: Vec<Stmt>,
line: u32,
},
While {
cond: Expr,
body: Vec<Stmt>,
line: u32,
},
Repeat {
body: Vec<Stmt>,
until: Expr,
line: u32,
},
/// A bare call statement, e.g. `TON1(IN := x, PT := T#5s);`.
Call {
callee: String,
args: Vec<CallArg>,
line: u32,
},
Return {
line: u32,
},
Exit {
line: u32,
},
/// `JMP label;` — an unstructured jump.
Jump {
label: String,
line: u32,
},
/// `label:` — a jump target.
Label {
name: String,
line: u32,
},
}
/// One argument in a call: positional (`name: None`) or named (`X := expr`).
#[derive(Debug, Clone)]
pub struct CallArg {
pub name: Option<String>,
pub value: Expr,
}
/// An expression.
#[derive(Debug, Clone)]
pub enum Expr {
Int(i64, u32),
Real(f64, u32),
Bool(bool, u32),
/// A string literal, with the unquoted contents.
Str(String, u32),
/// A duration / date / time literal, kept as raw text (`T#5s`, `DT#...`).
Time(String, u32),
Ident(String, u32),
/// `base[index]`.
Index {
base: Box<Expr>,
index: Box<Expr>,
line: u32,
},
/// `base.field`.
Member {
base: Box<Expr>,
field: String,
line: u32,
},
Unary {
op: UnOp,
expr: Box<Expr>,
line: u32,
},
Binary {
op: BinOp,
lhs: Box<Expr>,
rhs: Box<Expr>,
line: u32,
},
/// A function call used as an expression, e.g. `LIMIT(a, b, c)`.
Call {
callee: String,
args: Vec<CallArg>,
line: u32,
},
}
impl Expr {
/// The 1-based source line this expression starts on.
pub fn line(&self) -> u32 {
match self {
Expr::Int(_, l)
| Expr::Real(_, l)
| Expr::Bool(_, l)
| Expr::Str(_, l)
| Expr::Time(_, l)
| Expr::Ident(_, l)
| Expr::Index { line: l, .. }
| Expr::Member { line: l, .. }
| Expr::Unary { line: l, .. }
| Expr::Binary { line: l, .. }
| Expr::Call { line: l, .. } => *l,
}
}
/// If this expression is a plain identifier, its name.
pub fn as_ident(&self) -> Option<&str> {
match self {
Expr::Ident(name, _) => Some(name.as_str()),
_ => None,
}
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum UnOp {
Not,
Neg,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum BinOp {
Add,
Sub,
Mul,
Div,
Mod,
Pow,
Eq,
Ne,
Lt,
Le,
Gt,
Ge,
And,
Or,
Xor,
}
-372
View File
@@ -1,372 +0,0 @@
//! Lexer for IEC 61131-3 Structured Text.
//!
//! Tokenizes ST source into a flat token stream with 1-based line numbers.
//! Keywords are case-insensitive. Handles `(* *)` and `//` comments, `'..'` and
//! `".."` strings (with `''`/`""` escapes), based integers (`16#FF`, `2#1010`),
//! and duration/date literals (`T#5s`, `DT#...`) kept as raw text.
/// A lexed token with its source line.
#[derive(Debug, Clone)]
pub struct Token {
pub kind: Tok,
pub line: u32,
}
#[derive(Debug, Clone, PartialEq)]
pub enum Tok {
Int(i64),
Real(f64),
Str(String),
Time(String),
Bool(bool),
Ident(String),
Kw(Keyword),
Assign, // :=
Plus, // +
Minus, // -
Star, // *
Slash, // /
Power, // **
LParen, // (
RParen, // )
LBrack, // [
RBrack, // ]
Dot, // .
DotDot, // ..
Comma, // ,
Semi, // ;
Colon, // :
Lt, // <
Le, // <=
Gt, // >
Ge, // >=
Eq, // =
Ne, // <>
Amp, // &
Eof,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Keyword {
Program,
EndProgram,
Function,
EndFunction,
FunctionBlock,
EndFunctionBlock,
Var,
VarInput,
VarOutput,
VarInOut,
VarGlobal,
VarTemp,
VarExternal,
Constant,
EndVar,
Array,
Of,
If,
Then,
Elsif,
Else,
EndIf,
Case,
EndCase,
For,
To,
By,
Do,
EndFor,
While,
EndWhile,
Repeat,
Until,
EndRepeat,
Return,
Exit,
Jmp,
Not,
And,
Or,
Xor,
Mod,
Type,
EndType,
Struct,
EndStruct,
}
fn keyword_from(word: &str) -> Option<Keyword> {
use Keyword::*;
Some(match word.to_ascii_uppercase().as_str() {
"PROGRAM" => Program,
"END_PROGRAM" => EndProgram,
"FUNCTION" => Function,
"END_FUNCTION" => EndFunction,
"FUNCTION_BLOCK" => FunctionBlock,
"END_FUNCTION_BLOCK" => EndFunctionBlock,
"VAR" => Var,
"VAR_INPUT" => VarInput,
"VAR_OUTPUT" => VarOutput,
"VAR_IN_OUT" => VarInOut,
"VAR_GLOBAL" => VarGlobal,
"VAR_TEMP" => VarTemp,
"VAR_EXTERNAL" => VarExternal,
"CONSTANT" => Constant,
"END_VAR" => EndVar,
"ARRAY" => Array,
"OF" => Of,
"IF" => If,
"THEN" => Then,
"ELSIF" => Elsif,
"ELSE" => Else,
"END_IF" => EndIf,
"CASE" => Case,
"END_CASE" => EndCase,
"FOR" => For,
"TO" => To,
"BY" => By,
"DO" => Do,
"END_FOR" => EndFor,
"WHILE" => While,
"END_WHILE" => EndWhile,
"REPEAT" => Repeat,
"UNTIL" => Until,
"END_REPEAT" => EndRepeat,
"RETURN" => Return,
"EXIT" => Exit,
"JMP" => Jmp,
"NOT" => Not,
"AND" => And,
"OR" => Or,
"XOR" => Xor,
"MOD" => Mod,
"TYPE" => Type,
"END_TYPE" => EndType,
"STRUCT" => Struct,
"END_STRUCT" => EndStruct,
_ => return None,
})
}
/// Tokenize `src`. Unknown characters are skipped (best-effort — a scanner must
/// not die on odd input).
pub fn lex(src: &str) -> Vec<Token> {
let chars: Vec<char> = src.chars().collect();
let mut i = 0usize;
let mut line = 1u32;
let mut out = Vec::new();
let bump_line = |c: char, line: &mut u32| {
if c == '\n' {
*line += 1;
}
};
while i < chars.len() {
let c = chars[i];
// Whitespace.
if c.is_whitespace() {
bump_line(c, &mut line);
i += 1;
continue;
}
// Line comment: //
if c == '/' && i + 1 < chars.len() && chars[i + 1] == '/' {
while i < chars.len() && chars[i] != '\n' {
i += 1;
}
continue;
}
// Block comment: (* ... *)
if c == '(' && i + 1 < chars.len() && chars[i + 1] == '*' {
i += 2;
while i + 1 < chars.len() && !(chars[i] == '*' && chars[i + 1] == ')') {
bump_line(chars[i], &mut line);
i += 1;
}
i = (i + 2).min(chars.len());
continue;
}
let tok_line = line;
// String literal: '...' or "..."
if c == '\'' || c == '"' {
let quote = c;
i += 1;
let mut s = String::new();
while i < chars.len() {
let ch = chars[i];
if ch == quote {
// Doubled quote is an escaped quote.
if i + 1 < chars.len() && chars[i + 1] == quote {
s.push(quote);
i += 2;
continue;
}
i += 1;
break;
}
bump_line(ch, &mut line);
s.push(ch);
i += 1;
}
out.push(Token {
kind: Tok::Str(s),
line: tok_line,
});
continue;
}
// Identifier / keyword / time literal / boolean.
if c.is_ascii_alphabetic() || c == '_' {
let start = i;
while i < chars.len() && (chars[i].is_ascii_alphanumeric() || chars[i] == '_') {
i += 1;
}
let word: String = chars[start..i].iter().collect();
// Duration/date/time literal prefix: T#, TIME#, DT#, D#, TOD#, LT# ...
if i < chars.len() && chars[i] == '#' {
let up = word.to_ascii_uppercase();
if matches!(
up.as_str(),
"T" | "TIME" | "DT" | "D" | "TOD" | "LT" | "DATE"
) {
let lit_start = start;
i += 1; // consume '#'
while i < chars.len()
&& (chars[i].is_ascii_alphanumeric()
|| chars[i] == '.'
|| chars[i] == '_'
|| chars[i] == ':')
{
i += 1;
}
let lit: String = chars[lit_start..i].iter().collect();
out.push(Token {
kind: Tok::Time(lit),
line: tok_line,
});
continue;
}
}
let kind = match word.to_ascii_uppercase().as_str() {
"TRUE" => Tok::Bool(true),
"FALSE" => Tok::Bool(false),
_ => match keyword_from(&word) {
Some(kw) => Tok::Kw(kw),
None => Tok::Ident(word),
},
};
out.push(Token {
kind,
line: tok_line,
});
continue;
}
// Number: decimal, real, or based (16#..., 2#...).
if c.is_ascii_digit() {
let start = i;
while i < chars.len() && (chars[i].is_ascii_digit() || chars[i] == '_') {
i += 1;
}
// Based literal: <base>#<digits>
if i < chars.len() && chars[i] == '#' {
let base_str: String = chars[start..i].iter().filter(|c| **c != '_').collect();
i += 1;
let dstart = i;
while i < chars.len() && (chars[i].is_ascii_alphanumeric() || chars[i] == '_') {
i += 1;
}
let digits: String = chars[dstart..i].iter().filter(|c| **c != '_').collect();
let radix = base_str.parse::<u32>().unwrap_or(10);
let val = i64::from_str_radix(&digits, radix.clamp(2, 36)).unwrap_or(0);
out.push(Token {
kind: Tok::Int(val),
line: tok_line,
});
continue;
}
// Real: has a '.' (not '..') or exponent.
let is_real =
i < chars.len() && chars[i] == '.' && !(i + 1 < chars.len() && chars[i + 1] == '.');
if is_real {
i += 1;
while i < chars.len() && (chars[i].is_ascii_digit() || chars[i] == '_') {
i += 1;
}
let raw: String = chars[start..i].iter().filter(|c| **c != '_').collect();
out.push(Token {
kind: Tok::Real(raw.parse().unwrap_or(0.0)),
line: tok_line,
});
continue;
}
let raw: String = chars[start..i].iter().filter(|c| **c != '_').collect();
out.push(Token {
kind: Tok::Int(raw.parse().unwrap_or(0)),
line: tok_line,
});
continue;
}
// Operators / punctuation (longest match first).
let two: String = chars[i..(i + 2).min(chars.len())].iter().collect();
let kind = match two.as_str() {
":=" => Some(Tok::Assign),
"<=" => Some(Tok::Le),
">=" => Some(Tok::Ge),
"<>" => Some(Tok::Ne),
".." => Some(Tok::DotDot),
"**" => Some(Tok::Power),
_ => None,
};
if let Some(k) = kind {
out.push(Token {
kind: k,
line: tok_line,
});
i += 2;
continue;
}
let one = match c {
'+' => Some(Tok::Plus),
'-' => Some(Tok::Minus),
'*' => Some(Tok::Star),
'/' => Some(Tok::Slash),
'(' => Some(Tok::LParen),
')' => Some(Tok::RParen),
'[' => Some(Tok::LBrack),
']' => Some(Tok::RBrack),
'.' => Some(Tok::Dot),
',' => Some(Tok::Comma),
';' => Some(Tok::Semi),
':' => Some(Tok::Colon),
'<' => Some(Tok::Lt),
'>' => Some(Tok::Gt),
'=' => Some(Tok::Eq),
'&' => Some(Tok::Amp),
_ => None,
};
if let Some(k) = one {
out.push(Token {
kind: k,
line: tok_line,
});
}
i += 1;
}
out.push(Token {
kind: Tok::Eof,
line,
});
out
}
-234
View File
@@ -1,234 +0,0 @@
//! PLC control-logic security scanner for IEC 61131-3 targets.
//!
//! Parses Structured Text (raw `.st`/`.scl`/`.exp` files and PLCopen-XML
//! projects) into an AST and runs semantic control-logic security rules over it.
//! Implements [`ScanType::PlcControlLogic`].
pub mod ast;
pub mod lexer;
pub mod parser;
pub mod plcopen;
pub mod rules;
pub mod sbom;
use std::path::Path;
use compliance_core::error::CoreError;
use compliance_core::models::{Finding, ScanType};
use compliance_core::traits::{ScanOutput, Scanner};
use crate::pipeline::dedup;
/// Scanner for `ScanType::PlcControlLogic`.
pub struct PlcControlLogicScanner;
impl Scanner for PlcControlLogicScanner {
fn name(&self) -> &str {
"plc-control-logic"
}
fn scan_type(&self) -> ScanType {
ScanType::PlcControlLogic
}
#[tracing::instrument(skip_all)]
async fn scan(&self, repo_path: &Path, repo_id: &str) -> Result<ScanOutput, CoreError> {
let findings = analyze_tree(repo_path, repo_id);
Ok(ScanOutput {
findings,
sbom_entries: Vec::new(),
})
}
}
/// Walk a PLC project tree and produce findings.
pub(crate) fn analyze_tree(root: &Path, repo_id: &str) -> Vec<Finding> {
let mut findings = Vec::new();
for entry in walkdir::WalkDir::new(root)
.into_iter()
.filter_map(|e| e.ok())
{
if !entry.file_type().is_file() {
continue;
}
let path = entry.path();
let ext = path
.extension()
.and_then(|e| e.to_str())
.unwrap_or("")
.to_ascii_lowercase();
let is_st = matches!(ext.as_str(), "st" | "iecst" | "scl" | "exp" | "il");
let is_xml = matches!(ext.as_str(), "xml" | "plcopen" | "project");
if !is_st && !is_xml {
continue;
}
let Ok(content) = std::fs::read_to_string(path) else {
continue;
};
let pous = if is_xml {
plcopen::parse_plcopen(&content)
} else {
parser::parse(&content)
};
if pous.is_empty() {
continue;
}
let rel = path
.strip_prefix(root)
.unwrap_or(path)
.to_string_lossy()
.to_string();
for pou in &pous {
for hit in rules::analyze(pou) {
let line_s = hit.line.to_string();
let fingerprint =
dedup::compute_fingerprint(&[repo_id, &rel, hit.rule_id, &pou.name, &line_s]);
let mut f = Finding::new(
repo_id.to_string(),
fingerprint,
"plc-control-logic".to_string(),
ScanType::PlcControlLogic,
hit.title,
hit.description,
hit.severity,
);
f.file_path = Some(rel.clone());
f.line_number = Some(hit.line);
f.rule_id = Some(hit.rule_id.to_string());
f.cwe = hit.cwe.map(String::from);
f.remediation = Some(hit.remediation.to_string());
findings.push(f);
}
}
}
findings
}
#[cfg(test)]
mod tests {
use super::*;
use std::collections::HashSet;
use std::path::PathBuf;
fn demo_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
.parent()
.expect("workspace root")
.join("examples/plc-demo")
}
#[test]
fn scans_demo_project_end_to_end() {
let findings = analyze_tree(&demo_dir(), "demo-target");
assert!(!findings.is_empty(), "demo project should produce findings");
let rules: HashSet<&str> = findings
.iter()
.filter_map(|f| f.rule_id.as_deref())
.collect();
for r in [
"plc-hardcoded-credential",
"plc-default-password",
"plc-safety-bypass",
"plc-array-unchecked-index",
"plc-insecure-comm",
"plc-insecure-protocol-port",
"plc-unstructured-jump",
"plc-division-by-zero",
] {
assert!(rules.contains(r), "expected rule {r}; got {rules:?}");
}
// Every finding is well-formed for storage.
for f in &findings {
assert_eq!(f.repo_id, "demo-target");
assert!(f.file_path.is_some(), "finding needs a file");
assert!(f.line_number.is_some(), "finding needs a line");
}
// The guarded division (IF ScaleFactor <> 0.0) must not be double-counted:
// exactly one division-by-zero (the unguarded MeasuredFlow divide).
let div0 = findings
.iter()
.filter(|f| f.rule_id.as_deref() == Some("plc-division-by-zero"))
.count();
assert_eq!(div0, 1, "only the unguarded division should be flagged");
}
/// The realistic OpenPLC-style traffic-light sample is mostly sound control
/// logic: the scanner must surface its few genuine defects and stay quiet on
/// the timed state machine and the guarded duty-cycle division.
#[test]
fn realistic_sample_flags_only_real_issues() {
let all = analyze_tree(&demo_dir(), "demo-target");
let tl: Vec<_> = all
.iter()
.filter(|f| {
f.file_path
.as_deref()
.is_some_and(|p| p.ends_with("traffic_light.st"))
})
.collect();
assert!(!tl.is_empty(), "traffic_light.st should produce findings");
let rules: HashSet<&str> = tl.iter().filter_map(|f| f.rule_id.as_deref()).collect();
// The three planted defects: hardcoded SCADA password, cleartext Modbus
// master (no auth), and a maintenance mode that drops the PedPermit.
for r in [
"plc-hardcoded-credential",
"plc-insecure-comm",
"plc-safety-bypass",
] {
assert!(rules.contains(r), "expected rule {r}; got {rules:?}");
}
// Modbus/TCP on 502 is also an insecure-protocol port.
assert!(rules.contains("plc-insecure-protocol-port"));
// Low false positives: the guarded `IF LampCount <> 0` division and the
// JMP-free state machine must not trip anything.
assert_eq!(
tl.iter()
.filter(|f| f.rule_id.as_deref() == Some("plc-division-by-zero"))
.count(),
0,
"the guarded duty-cycle division must not be flagged"
);
assert!(
!rules.contains("plc-unstructured-jump"),
"the CASE state machine uses no JMP"
);
}
/// Graphical logic must be analysed too: an FBD POU (blocks + in/out
/// variables) is translated to synthetic ST, so the same rules fire on the
/// cleartext Modbus block, the hardcoded HMI password and the safety write.
#[test]
fn fbd_graphical_body_is_analysed() {
let all = analyze_tree(&demo_dir(), "demo-target");
let fbd: Vec<_> = all
.iter()
.filter(|f| {
f.file_path
.as_deref()
.is_some_and(|p| p.ends_with("pump_fbd.xml"))
})
.collect();
assert!(
!fbd.is_empty(),
"pump_fbd.xml (FBD) should produce findings"
);
let rules: HashSet<&str> = fbd.iter().filter_map(|f| f.rule_id.as_deref()).collect();
for r in [
"plc-insecure-comm", // Modbus_TCP_Master(AUTH := FALSE)
"plc-insecure-protocol-port", // PORT := 502
"plc-hardcoded-credential", // HmiPassword := 'admin123'
"plc-safety-bypass", // Safety_Enable := FALSE
] {
assert!(
rules.contains(r),
"expected rule {r} from FBD; got {rules:?}"
);
}
}
}
-766
View File
@@ -1,766 +0,0 @@
//! Recursive-descent parser for the security-relevant subset of Structured Text.
//!
//! Tolerant by design: it parses the POUs, variable sections, and statement
//! bodies it understands, and skips (with statement/POU-level recovery) anything
//! it does not, so a single odd construct never sinks the whole file.
use super::ast::*;
use super::lexer::{Keyword as K, Tok, Token};
pub struct Parser {
toks: Vec<Token>,
pos: usize,
}
impl Parser {
pub fn new(toks: Vec<Token>) -> Self {
Self { toks, pos: 0 }
}
// ── token helpers ──────────────────────────────────────────────
fn peek(&self) -> &Tok {
&self.toks[self.pos.min(self.toks.len() - 1)].kind
}
fn line(&self) -> u32 {
self.toks[self.pos.min(self.toks.len() - 1)].line
}
fn at_end(&self) -> bool {
matches!(self.peek(), Tok::Eof)
}
fn advance(&mut self) -> Tok {
let t = self.toks[self.pos.min(self.toks.len() - 1)].kind.clone();
if self.pos < self.toks.len() - 1 {
self.pos += 1;
}
t
}
fn eat(&mut self, t: &Tok) -> bool {
if self.peek() == t {
self.advance();
true
} else {
false
}
}
fn eat_kw(&mut self, k: K) -> bool {
if matches!(self.peek(), Tok::Kw(x) if *x == k) {
self.advance();
true
} else {
false
}
}
fn at_kw(&self, k: K) -> bool {
matches!(self.peek(), Tok::Kw(x) if *x == k)
}
fn ident(&mut self) -> Option<String> {
if let Tok::Ident(s) = self.peek() {
let s = s.clone();
self.advance();
Some(s)
} else {
None
}
}
// ── top level ──────────────────────────────────────────────────
/// Parse every POU in the token stream.
pub fn parse_units(&mut self) -> Vec<Pou> {
let mut pous = Vec::new();
while !self.at_end() {
match self.peek() {
Tok::Kw(K::Program) => {
self.advance();
if let Some(p) = self.parse_pou(PouKind::Program, K::EndProgram) {
pous.push(p);
}
}
Tok::Kw(K::Function) => {
self.advance();
if let Some(p) = self.parse_pou(PouKind::Function, K::EndFunction) {
pous.push(p);
}
}
Tok::Kw(K::FunctionBlock) => {
self.advance();
if let Some(p) = self.parse_pou(PouKind::FunctionBlock, K::EndFunctionBlock) {
pous.push(p);
}
}
// Skip TYPE...END_TYPE and anything else at top level.
_ => {
self.advance();
}
}
}
pous
}
fn parse_pou(&mut self, kind: PouKind, end: K) -> Option<Pou> {
let line = self.line();
let name = self.ident().unwrap_or_else(|| "<anonymous>".to_string());
// Optional `: return_type` for functions.
if self.eat(&Tok::Colon) {
let _ = self.advance(); // return type token
}
let mut vars = Vec::new();
// Variable sections precede the body.
while let Some(section) = self.var_section_kw() {
self.advance();
let _ = self.eat_kw(K::Constant); // CONSTANT is informational for our rules
self.parse_var_decls(section, &mut vars);
}
// Body statements until END_<kind>.
let mut body = Vec::new();
while !self.at_end() && !self.at_kw(end) {
if let Some(s) = self.parse_stmt() {
body.push(s);
}
}
self.eat_kw(end);
Some(Pou {
name,
kind,
vars,
body,
line,
})
}
fn var_section_kw(&self) -> Option<VarSection> {
match self.peek() {
Tok::Kw(K::Var) => Some(VarSection::Var),
Tok::Kw(K::VarInput) => Some(VarSection::Input),
Tok::Kw(K::VarOutput) => Some(VarSection::Output),
Tok::Kw(K::VarInOut) => Some(VarSection::InOut),
Tok::Kw(K::VarGlobal) => Some(VarSection::Global),
Tok::Kw(K::VarTemp) => Some(VarSection::Temp),
Tok::Kw(K::VarExternal) => Some(VarSection::External),
_ => None,
}
}
fn parse_var_decls(&mut self, section: VarSection, out: &mut Vec<VarDecl>) {
while !self.at_end() && !self.at_kw(K::EndVar) {
let line = self.line();
// names: a, b, c
let mut names = Vec::new();
match self.ident() {
Some(n) => names.push(n),
None => {
// Not a declaration we understand — skip to next ; or END_VAR.
self.sync_decl();
continue;
}
}
while self.eat(&Tok::Comma) {
if let Some(n) = self.ident() {
names.push(n);
}
}
if !self.eat(&Tok::Colon) {
self.sync_decl();
continue;
}
let (type_name, array_bounds) = self.parse_type();
let init = if self.eat(&Tok::Assign) {
Some(self.parse_expr())
} else {
None
};
self.eat(&Tok::Semi);
for n in names {
out.push(VarDecl {
name: n,
section,
type_name: type_name.clone(),
array_bounds,
init: init.clone(),
line,
});
}
}
self.eat_kw(K::EndVar);
}
/// Parse a (possibly ARRAY) type, returning its rendered name and literal
/// bounds when present.
fn parse_type(&mut self) -> (String, Option<(i64, i64)>) {
if self.eat_kw(K::Array) {
let mut bounds = None;
if self.eat(&Tok::LBrack) {
let lo = self.int_lit();
self.eat(&Tok::DotDot);
let hi = self.int_lit();
if let (Some(lo), Some(hi)) = (lo, hi) {
bounds = Some((lo, hi));
}
// Skip any further dimensions / tokens to the closing bracket.
while !self.at_end() && !self.eat(&Tok::RBrack) {
self.advance();
}
}
self.eat_kw(K::Of);
let elem = self.type_ident();
(format!("ARRAY OF {elem}"), bounds)
} else {
(self.type_ident(), None)
}
}
fn type_ident(&mut self) -> String {
// Types can be qualified idents; keep it simple: one token, plus any
// string-length suffix like STRING[80].
let base = match self.advance() {
Tok::Ident(s) => s,
Tok::Kw(_) => "TYPE".to_string(),
other => format!("{other:?}"),
};
if self.eat(&Tok::LBrack) {
while !self.at_end() && !self.eat(&Tok::RBrack) {
self.advance();
}
}
base
}
fn int_lit(&mut self) -> Option<i64> {
match self.peek() {
Tok::Int(n) => {
let n = *n;
self.advance();
Some(n)
}
Tok::Minus => {
self.advance();
if let Tok::Int(n) = self.peek() {
let n = -*n;
self.advance();
Some(n)
} else {
None
}
}
_ => None,
}
}
fn sync_decl(&mut self) {
while !self.at_end() && !self.eat(&Tok::Semi) && !self.at_kw(K::EndVar) {
self.advance();
}
}
fn sync_stmt(&mut self) {
while !self.at_end() && !self.eat(&Tok::Semi) {
// Stop at block terminators so recovery doesn't swallow structure.
if matches!(
self.peek(),
Tok::Kw(
K::EndIf
| K::EndFor
| K::EndWhile
| K::EndCase
| K::EndRepeat
| K::EndProgram
| K::EndFunction
| K::EndFunctionBlock
| K::Else
| K::Elsif
)
) {
return;
}
self.advance();
}
}
// ── statements ─────────────────────────────────────────────────
fn parse_stmt(&mut self) -> Option<Stmt> {
let line = self.line();
match self.peek().clone() {
Tok::Semi => {
self.advance();
None
}
Tok::Kw(K::If) => self.parse_if(),
Tok::Kw(K::Case) => self.parse_case(),
Tok::Kw(K::For) => self.parse_for(),
Tok::Kw(K::While) => self.parse_while(),
Tok::Kw(K::Repeat) => self.parse_repeat(),
Tok::Kw(K::Return) => {
self.advance();
self.eat(&Tok::Semi);
Some(Stmt::Return { line })
}
Tok::Kw(K::Exit) => {
self.advance();
self.eat(&Tok::Semi);
Some(Stmt::Exit { line })
}
Tok::Kw(K::Jmp) => {
self.advance();
let label = self.ident().unwrap_or_default();
self.eat(&Tok::Semi);
Some(Stmt::Jump { label, line })
}
Tok::Ident(name) => {
// Could be `label:`, `call(...)`, or an assignment.
// Lookahead: ident ':' (not ':=') → label.
if matches!(
self.toks.get(self.pos + 1).map(|t| &t.kind),
Some(Tok::Colon)
) && !matches!(self.toks.get(self.pos + 2).map(|t| &t.kind), Some(Tok::Eq))
{
self.advance(); // ident
self.advance(); // ':'
return Some(Stmt::Label { name, line });
}
let lhs = self.parse_expr();
if self.eat(&Tok::Assign) {
let value = self.parse_expr();
self.eat(&Tok::Semi);
Some(Stmt::Assign {
target: lhs,
value,
line,
})
} else if let Expr::Call { callee, args, .. } = lhs {
self.eat(&Tok::Semi);
Some(Stmt::Call { callee, args, line })
} else {
// Bare expression / FB invocation without args recognized —
// skip to the terminator.
self.sync_stmt();
None
}
}
_ => {
self.sync_stmt();
None
}
}
}
fn parse_block_until(&mut self, terms: &[K]) -> Vec<Stmt> {
let mut body = Vec::new();
while !self.at_end() && !terms.iter().any(|k| self.at_kw(*k)) {
if let Some(s) = self.parse_stmt() {
body.push(s);
}
}
body
}
fn parse_if(&mut self) -> Option<Stmt> {
let line = self.line();
self.eat_kw(K::If);
let mut branches = Vec::new();
let cond = self.parse_expr();
self.eat_kw(K::Then);
let body = self.parse_block_until(&[K::Elsif, K::Else, K::EndIf]);
branches.push((cond, body));
while self.eat_kw(K::Elsif) {
let c = self.parse_expr();
self.eat_kw(K::Then);
let b = self.parse_block_until(&[K::Elsif, K::Else, K::EndIf]);
branches.push((c, b));
}
let else_body = if self.eat_kw(K::Else) {
Some(self.parse_block_until(&[K::EndIf]))
} else {
None
};
self.eat_kw(K::EndIf);
self.eat(&Tok::Semi);
Some(Stmt::If {
branches,
else_body,
line,
})
}
fn parse_case(&mut self) -> Option<Stmt> {
let line = self.line();
self.eat_kw(K::Case);
let selector = self.parse_expr();
self.eat_kw(K::Of);
let mut arms = Vec::new();
let mut else_body = None;
while !self.at_end() && !self.at_kw(K::EndCase) {
if self.eat_kw(K::Else) {
else_body = Some(self.parse_block_until(&[K::EndCase]));
break;
}
// labels: expr {, expr} :
let mut labels = vec![self.parse_expr()];
while self.eat(&Tok::Comma) {
labels.push(self.parse_expr());
}
self.eat(&Tok::Colon);
let body = self.parse_block_until(&[K::EndCase, K::Else]);
arms.push((labels, body));
}
self.eat_kw(K::EndCase);
self.eat(&Tok::Semi);
Some(Stmt::Case {
selector,
arms,
else_body,
line,
})
}
fn parse_for(&mut self) -> Option<Stmt> {
let line = self.line();
self.eat_kw(K::For);
let var = self.ident().unwrap_or_default();
self.eat(&Tok::Assign);
let from = self.parse_expr();
self.eat_kw(K::To);
let to = self.parse_expr();
let by = if self.eat_kw(K::By) {
Some(self.parse_expr())
} else {
None
};
self.eat_kw(K::Do);
let body = self.parse_block_until(&[K::EndFor]);
self.eat_kw(K::EndFor);
self.eat(&Tok::Semi);
Some(Stmt::For {
var,
from,
to,
by,
body,
line,
})
}
fn parse_while(&mut self) -> Option<Stmt> {
let line = self.line();
self.eat_kw(K::While);
let cond = self.parse_expr();
self.eat_kw(K::Do);
let body = self.parse_block_until(&[K::EndWhile]);
self.eat_kw(K::EndWhile);
self.eat(&Tok::Semi);
Some(Stmt::While { cond, body, line })
}
fn parse_repeat(&mut self) -> Option<Stmt> {
let line = self.line();
self.eat_kw(K::Repeat);
let body = self.parse_block_until(&[K::Until, K::EndRepeat]);
self.eat_kw(K::Until);
let until = self.parse_expr();
self.eat_kw(K::EndRepeat);
self.eat(&Tok::Semi);
Some(Stmt::Repeat { body, until, line })
}
// ── expressions (precedence climbing) ──────────────────────────
pub fn parse_expr(&mut self) -> Expr {
self.parse_or()
}
fn parse_or(&mut self) -> Expr {
let mut lhs = self.parse_and();
loop {
let op = match self.peek() {
Tok::Kw(K::Or) => BinOp::Or,
Tok::Kw(K::Xor) => BinOp::Xor,
_ => break,
};
let line = self.line();
self.advance();
let rhs = self.parse_and();
lhs = Expr::Binary {
op,
lhs: Box::new(lhs),
rhs: Box::new(rhs),
line,
};
}
lhs
}
fn parse_and(&mut self) -> Expr {
let mut lhs = self.parse_cmp();
while matches!(self.peek(), Tok::Kw(K::And) | Tok::Amp) {
let op = BinOp::And;
let line = self.line();
self.advance();
let rhs = self.parse_cmp();
lhs = Expr::Binary {
op,
lhs: Box::new(lhs),
rhs: Box::new(rhs),
line,
};
}
lhs
}
fn parse_cmp(&mut self) -> Expr {
let mut lhs = self.parse_add();
loop {
let op = match self.peek() {
Tok::Eq => BinOp::Eq,
Tok::Ne => BinOp::Ne,
Tok::Lt => BinOp::Lt,
Tok::Le => BinOp::Le,
Tok::Gt => BinOp::Gt,
Tok::Ge => BinOp::Ge,
_ => break,
};
let line = self.line();
self.advance();
let rhs = self.parse_add();
lhs = Expr::Binary {
op,
lhs: Box::new(lhs),
rhs: Box::new(rhs),
line,
};
}
lhs
}
fn parse_add(&mut self) -> Expr {
let mut lhs = self.parse_mul();
loop {
let op = match self.peek() {
Tok::Plus => BinOp::Add,
Tok::Minus => BinOp::Sub,
_ => break,
};
let line = self.line();
self.advance();
let rhs = self.parse_mul();
lhs = Expr::Binary {
op,
lhs: Box::new(lhs),
rhs: Box::new(rhs),
line,
};
}
lhs
}
fn parse_mul(&mut self) -> Expr {
let mut lhs = self.parse_unary();
loop {
let op = match self.peek() {
Tok::Star => BinOp::Mul,
Tok::Slash => BinOp::Div,
Tok::Kw(K::Mod) => BinOp::Mod,
Tok::Power => BinOp::Pow,
_ => break,
};
let line = self.line();
self.advance();
let rhs = self.parse_unary();
lhs = Expr::Binary {
op,
lhs: Box::new(lhs),
rhs: Box::new(rhs),
line,
};
}
lhs
}
fn parse_unary(&mut self) -> Expr {
let line = self.line();
match self.peek() {
Tok::Kw(K::Not) => {
self.advance();
Expr::Unary {
op: UnOp::Not,
expr: Box::new(self.parse_unary()),
line,
}
}
Tok::Minus => {
self.advance();
Expr::Unary {
op: UnOp::Neg,
expr: Box::new(self.parse_unary()),
line,
}
}
_ => self.parse_postfix(),
}
}
fn parse_postfix(&mut self) -> Expr {
let mut e = self.parse_primary();
loop {
let line = self.line();
match self.peek() {
Tok::LBrack => {
self.advance();
let index = self.parse_expr();
self.eat(&Tok::RBrack);
e = Expr::Index {
base: Box::new(e),
index: Box::new(index),
line,
};
}
Tok::Dot => {
self.advance();
let field = self.ident().unwrap_or_default();
e = Expr::Member {
base: Box::new(e),
field,
line,
};
}
_ => break,
}
}
e
}
fn parse_primary(&mut self) -> Expr {
let line = self.line();
match self.advance() {
Tok::Int(n) => Expr::Int(n, line),
Tok::Real(r) => Expr::Real(r, line),
Tok::Bool(b) => Expr::Bool(b, line),
Tok::Str(s) => Expr::Str(s, line),
Tok::Time(t) => Expr::Time(t, line),
Tok::LParen => {
let e = self.parse_expr();
self.eat(&Tok::RParen);
e
}
Tok::Ident(name) => {
if self.eat(&Tok::LParen) {
let args = self.parse_call_args();
Expr::Call {
callee: name,
args,
line,
}
} else {
Expr::Ident(name, line)
}
}
// Unrecognized start of expression — yield a placeholder identifier.
_ => Expr::Ident(String::new(), line),
}
}
fn parse_call_args(&mut self) -> Vec<CallArg> {
let mut args = Vec::new();
if self.eat(&Tok::RParen) {
return args;
}
loop {
// Named arg: ident := expr (peek two tokens).
if let Tok::Ident(name) = self.peek().clone() {
if matches!(
self.toks.get(self.pos + 1).map(|t| &t.kind),
Some(Tok::Assign)
) {
self.advance(); // ident
self.advance(); // :=
let value = self.parse_expr();
args.push(CallArg {
name: Some(name),
value,
});
if self.eat(&Tok::Comma) {
continue;
}
break;
}
}
let value = self.parse_expr();
args.push(CallArg { name: None, value });
if self.eat(&Tok::Comma) {
continue;
}
break;
}
self.eat(&Tok::RParen);
args
}
}
/// Parse ST source into its POUs.
pub fn parse(src: &str) -> Vec<Pou> {
let toks = super::lexer::lex(src);
Parser::new(toks).parse_units()
}
#[cfg(test)]
mod tests {
use super::*;
const SAMPLE: &str = r#"
PROGRAM Main
VAR
idx : INT;
pw : STRING := 'admin123';
buf : ARRAY[0..9] OF INT;
ok : BOOL := FALSE;
END_VAR
// a comment
IF idx > 0 THEN
buf[idx] := idx * 2;
ELSE
JMP done;
END_IF;
Comm(IP := '10.0.0.1', PORT := 502);
done:
ok := TRUE;
END_PROGRAM
"#;
#[test]
fn parses_program_vars_and_body() {
let pous = parse(SAMPLE);
assert_eq!(pous.len(), 1, "one POU");
let p = &pous[0];
assert_eq!(p.name, "Main");
assert_eq!(p.kind, PouKind::Program);
// vars: idx, pw, buf, ok
assert_eq!(p.vars.len(), 4);
let pw = p.vars.iter().find(|v| v.name == "pw").expect("pw");
assert!(matches!(&pw.init, Some(Expr::Str(s, _)) if s == "admin123"));
let buf = p.vars.iter().find(|v| v.name == "buf").expect("buf");
assert_eq!(buf.array_bounds, Some((0, 9)));
// body has an IF, a Call, a Label, and an Assign
assert!(p.body.iter().any(|s| matches!(s, Stmt::If { .. })));
assert!(p
.body
.iter()
.any(|s| matches!(s, Stmt::Call { callee, .. } if callee == "Comm")));
assert!(p
.body
.iter()
.any(|s| matches!(s, Stmt::Label { name, .. } if name == "done")));
}
#[test]
fn jmp_inside_if_is_captured() {
let pous = parse(SAMPLE);
let p = &pous[0];
// find the IF, check its else branch has a JMP
let has_jmp = p.body.iter().any(|s| match s {
Stmt::If { else_body, .. } => else_body
.as_ref()
.map(|b| b.iter().any(|s| matches!(s, Stmt::Jump { .. })))
.unwrap_or(false),
_ => false,
});
assert!(has_jmp, "JMP should be parsed inside the ELSE branch");
}
}
@@ -1,418 +0,0 @@
//! PLCopen XML → Structured Text POUs.
//!
//! A PLCopen project stores each POU as `<pou name=".." pouType="..">` with an
//! `<interface>` (typed variable sections) and a `<body>` in one of the IEC
//! 61131-3 languages. We reconstruct an equivalent Structured-Text source for
//! each POU (a `VAR` block from the interface + statements from the body) and run
//! it through the ST parser, so raw `.st` files and PLCopen projects — textual or
//! graphical — flow through one analysis path.
//!
//! Body languages:
//! - **ST** — taken verbatim.
//! - **FBD / LD** — the graphical network is translated to synthetic ST: blocks
//! become calls (`TypeName(pin := arg, …)`), out-variables / coils become
//! assignments, with input pins resolved by tracing connections. This lets the
//! semantic rules see comm calls, hardcoded arguments and safety writes that
//! live in graphical logic, not just in text.
//! - **SFC** — the step/transition graph itself is skipped; the ST/FBD/LD bodies
//! embedded in its actions and transitions are still translated.
use std::collections::HashMap;
use roxmltree::Node;
use super::ast::Pou;
use super::parser;
/// Parse every POU out of a PLCopen XML document (ST, FBD or LD bodies).
pub fn parse_plcopen(xml: &str) -> Vec<Pou> {
let doc = match roxmltree::Document::parse(xml) {
Ok(d) => d,
Err(_) => return Vec::new(),
};
let mut pous = Vec::new();
for pou in doc.descendants().filter(|n| n.has_tag_name("pou")) {
let name = pou.attribute("name").unwrap_or("pou").to_string();
let pou_type = pou.attribute("pouType").unwrap_or("program");
let Some(body) = reconstruct_body(pou) else {
continue;
};
if body.trim().is_empty() {
continue;
}
let var_block = build_var_block(pou);
let kw = match pou_type.to_ascii_lowercase().as_str() {
"function" => "FUNCTION",
"functionblock" | "functionblocktype" => "FUNCTION_BLOCK",
_ => "PROGRAM",
};
let synthetic = format!("{kw} {name}\n{var_block}{body}\nEND_{kw}\n");
pous.extend(parser::parse(&synthetic));
}
pous
}
/// Case-insensitive tag match (PLCopen uses `FBD`/`LD`/`ST`, CODESYS may vary).
fn tag_is(n: &Node, name: &str) -> bool {
n.tag_name().name().eq_ignore_ascii_case(name)
}
/// Reconstruct a POU's body as Structured Text, whatever language it is written
/// in. Concatenates every language body found under `<body>` (SFC actions and
/// transitions carry their own ST/FBD/LD sub-bodies).
fn reconstruct_body(pou: Node) -> Option<String> {
let mut out = String::new();
for body in pou.descendants().filter(|n| tag_is(n, "body")) {
for lang in body.children().filter(|n| n.is_element()) {
let piece = match lang.tag_name().name().to_ascii_uppercase().as_str() {
"ST" | "IL" => collect_text(lang),
"FBD" | "LD" => translate_network(lang),
_ => continue,
};
if !piece.trim().is_empty() {
out.push_str(&piece);
if !piece.ends_with('\n') {
out.push('\n');
}
}
}
}
if out.trim().is_empty() {
None
} else {
Some(out)
}
}
// ── graphical (FBD / LD) → synthetic ST ────────────────────────────────
/// Translate one FBD/LD network into ST statements: blocks → calls,
/// out-variables and coils → assignments.
fn translate_network(net: Node) -> String {
let by_id = index_local_ids(net);
let mut out = String::new();
for el in net.children().filter(|n| n.is_element()) {
let stmt = match el.tag_name().name().to_ascii_lowercase().as_str() {
"block" => block_call(el, &by_id).map(|c| format!("{c};")),
"outvariable" => out_assignment(el, &by_id),
"coil" => coil_assignment(el, &by_id),
_ => None,
};
if let Some(s) = stmt {
out.push_str(&s);
out.push('\n');
}
}
out
}
/// Index every element in a network by its `localId` so connections resolve.
fn index_local_ids<'a, 'input>(net: Node<'a, 'input>) -> HashMap<String, Node<'a, 'input>> {
net.descendants()
.filter(|n| n.is_element())
.filter_map(|n| n.attribute("localId").map(|id| (id.to_string(), n)))
.collect()
}
/// Build a call expression for a block: `TypeName(pin := arg, …)`.
fn block_call(block: Node, by_id: &HashMap<String, Node>) -> Option<String> {
let ty = block.attribute("typeName")?;
let mut args = Vec::new();
if let Some(inputs) = block.children().find(|n| tag_is(n, "inputVariables")) {
for v in inputs.children().filter(|n| tag_is(n, "variable")) {
let Some(expr) = input_expr(v, by_id, 0) else {
continue;
};
match v.attribute("formalParameter") {
Some(pin) if !pin.is_empty() => args.push(format!("{pin} := {expr}")),
_ => args.push(expr),
}
}
}
Some(format!("{ty}({})", args.join(", ")))
}
/// `target := <traced expression>;` for an FBD out-variable.
fn out_assignment(outvar: Node, by_id: &HashMap<String, Node>) -> Option<String> {
let target = expression_text(outvar)?;
let value = input_expr(outvar, by_id, 0).unwrap_or_else(|| "0".to_string());
Some(format!("{target} := {value};"))
}
/// `coil := <traced rung expression>;` for an LD coil (negated → `NOT (…)`).
fn coil_assignment(coil: Node, by_id: &HashMap<String, Node>) -> Option<String> {
let target = child_text(coil, "variable")?;
let rung = input_expr(coil, by_id, 0).unwrap_or_else(|| "TRUE".to_string());
let negated = matches!(coil.attribute("negated"), Some(v) if v.eq_ignore_ascii_case("true"));
let rhs = if negated {
format!("NOT ({rung})")
} else {
rung
};
Some(format!("{target} := {rhs};"))
}
/// Resolve the expression feeding `node`'s single input connection.
fn input_expr(node: Node, by_id: &HashMap<String, Node>, depth: u8) -> Option<String> {
let refid = ref_local_id(node)?;
Some(expr_for(&refid, by_id, depth))
}
/// Build the ST expression produced by the element with this `localId`.
fn expr_for(local_id: &str, by_id: &HashMap<String, Node>, depth: u8) -> String {
if depth > 24 {
return "0".to_string();
}
let Some(node) = by_id.get(local_id) else {
return format!("__net{local_id}");
};
match node.tag_name().name().to_ascii_lowercase().as_str() {
"invariable" | "inoutvariable" => {
expression_text(*node).unwrap_or_else(|| format!("__net{local_id}"))
}
// A block feeding another element: reference it by a synthetic result
// name; the block is emitted as its own call statement, so we neither
// duplicate the call nor lose it.
"block" => format!("__blk{local_id}"),
"contact" => {
let var = child_text(*node, "variable").unwrap_or_else(|| "TRUE".to_string());
let negated =
matches!(node.attribute("negated"), Some(v) if v.eq_ignore_ascii_case("true"));
let term = if negated { format!("NOT {var}") } else { var };
match ref_local_id(*node) {
Some(up) => {
let upstream = expr_for(&up, by_id, depth + 1);
if upstream == "TRUE" {
term
} else {
format!("({upstream} AND {term})")
}
}
None => term,
}
}
"leftpowerrail" => "TRUE".to_string(),
_ => format!("__net{local_id}"),
}
}
/// The `refLocalId` of `node`'s first input connection, if any.
fn ref_local_id(node: Node) -> Option<String> {
node.descendants()
.find(|n| tag_is(n, "connectionPointIn"))
.and_then(|cpi| cpi.descendants().find(|n| tag_is(n, "connection")))
.and_then(|c| c.attribute("refLocalId"))
.map(|s| s.to_string())
}
/// Text of a node's `<expression>` child (variable name or literal).
fn expression_text(node: Node) -> Option<String> {
let e = node.children().find(|n| tag_is(n, "expression"))?;
let t = collect_text(e).trim().to_string();
if t.is_empty() {
None
} else {
Some(t)
}
}
/// Text of a named child element (e.g. `<variable>` of a contact/coil).
fn child_text(node: Node, name: &str) -> Option<String> {
let c = node.children().find(|n| tag_is(n, name))?;
let t = collect_text(c).trim().to_string();
if t.is_empty() {
None
} else {
Some(t)
}
}
/// Concatenate the text of a node's descendant text nodes (bodies are often
/// wrapped in `<xhtml>` and may contain multiple text runs). Only text nodes are
/// gathered: an element's `.text()` would re-yield its first child's text, which
/// (with the text node itself) would duplicate every value.
fn collect_text(node: Node) -> String {
node.descendants()
.filter(|n| n.is_text())
.filter_map(|n| n.text())
.collect::<String>()
}
/// Build an ST `VAR … END_VAR` block from a POU's `<interface>` variable
/// sections, so declarations (types, initial values) reach the rules.
fn build_var_block(pou: Node) -> String {
let Some(interface) = pou.children().find(|n| n.has_tag_name("interface")) else {
return String::new();
};
let mut out = String::from("VAR\n");
let mut any = false;
for container in interface.children().filter(|n| n.is_element()) {
// localVars / inputVars / outputVars / inOutVars / tempVars / globalVars / externalVars
if !container.tag_name().name().ends_with("Vars") {
continue;
}
for var in container.children().filter(|n| n.has_tag_name("variable")) {
let Some(vname) = var.attribute("name") else {
continue;
};
let ty = var
.children()
.find(|n| n.has_tag_name("type"))
.map(type_name)
.unwrap_or_else(|| "BOOL".to_string());
let init = var
.children()
.find(|n| n.has_tag_name("initialValue"))
.and_then(initial_value);
match init {
Some(v) => out.push_str(&format!(" {vname} : {ty} := {v};\n")),
None => out.push_str(&format!(" {vname} : {ty};\n")),
}
any = true;
}
}
out.push_str("END_VAR\n");
if any {
out
} else {
String::new()
}
}
/// Render a PLCopen `<type>` element as an ST type string.
fn type_name(type_node: Node) -> String {
let Some(inner) = type_node.children().find(|n| n.is_element()) else {
return "BOOL".to_string();
};
let tag = inner.tag_name().name();
match tag {
"derived" => inner.attribute("name").unwrap_or("DERIVED").to_string(),
"array" => {
let dim = inner.children().find(|n| n.has_tag_name("dimension"));
let (lo, hi) = dim
.map(|d| {
(
d.attribute("lower").unwrap_or("0").to_string(),
d.attribute("upper").unwrap_or("0").to_string(),
)
})
.unwrap_or_else(|| ("0".to_string(), "0".to_string()));
let base = inner
.children()
.find(|n| n.has_tag_name("baseType"))
.map(type_name)
.unwrap_or_else(|| "INT".to_string());
format!("ARRAY[{lo}..{hi}] OF {base}")
}
"string" | "wstring" => "STRING".to_string(),
// BOOL, INT, DINT, REAL, TIME, ... — the tag name is the ST type.
other => other.to_ascii_uppercase(),
}
}
/// Extract an initial value as an ST literal (quoting strings).
fn initial_value(iv: Node) -> Option<String> {
let simple = iv.descendants().find(|n| n.has_tag_name("simpleValue"))?;
let raw = simple.attribute("value")?.trim().to_string();
if raw.is_empty() {
return None;
}
// Numbers / booleans / time literals pass through; everything else is a
// string literal.
let is_scalar = raw.eq_ignore_ascii_case("true")
|| raw.eq_ignore_ascii_case("false")
|| raw.starts_with(['T', 't', 'D', 'd']) && raw.contains('#')
|| raw
.chars()
.all(|c| c.is_ascii_digit() || c == '.' || c == '-' || c == '+');
if is_scalar || raw.starts_with('\'') || raw.starts_with('"') {
Some(raw)
} else {
Some(format!("'{}'", raw.replace('\'', "''")))
}
}
#[cfg(test)]
mod tests {
use super::parse_plcopen;
use crate::pipeline::plc::rules;
use std::collections::HashSet;
fn rule_ids(xml: &str) -> HashSet<&'static str> {
parse_plcopen(xml)
.iter()
.flat_map(rules::analyze)
.map(|h| h.rule_id)
.collect()
}
/// A Ladder Diagram network: a rung (power rail → contact → coil) plus an
/// insecure comm block. Coils/contacts translate to assignments; the block
/// translates to a call so the port rule fires.
#[test]
fn ld_coil_and_block_translate_and_are_analysed() {
let xml = r#"<?xml version="1.0"?>
<project xmlns="http://www.plcopen.org/xml/tc6_0201">
<types><pous>
<pou name="Rung" pouType="program">
<interface><localVars>
<variable name="Motor"><type><BOOL/></type></variable>
</localVars></interface>
<body><LD>
<leftPowerRail localId="0"/>
<contact localId="1"><variable>Start</variable>
<connectionPointIn><connection refLocalId="0"/></connectionPointIn></contact>
<coil localId="2"><variable>Motor</variable>
<connectionPointIn><connection refLocalId="1"/></connectionPointIn></coil>
<inVariable localId="3"><expression>21</expression></inVariable>
<inVariable localId="4"><expression>FALSE</expression></inVariable>
<block localId="10" typeName="Ftp_Send">
<inputVariables>
<variable formalParameter="PORT">
<connectionPointIn><connection refLocalId="3"/></connectionPointIn></variable>
<variable formalParameter="ENCRYPT">
<connectionPointIn><connection refLocalId="4"/></connectionPointIn></variable>
</inputVariables>
</block>
</LD></body>
</pou>
</pous></types>
</project>"#;
let ids = rule_ids(xml);
// Ftp_Send(PORT := 21, ENCRYPT := FALSE) — port 21 is an insecure protocol.
assert!(
ids.contains("plc-insecure-protocol-port"),
"LD block should flag port 21; got {ids:?}"
);
}
/// Doubled-text regression: a graphical expression must be extracted once,
/// so literals like `502` and `FALSE` stay intact (not `502502`/`FALSEFALSE`).
#[test]
fn graphical_expression_text_is_not_duplicated() {
let xml = r#"<?xml version="1.0"?>
<project xmlns="http://www.plcopen.org/xml/tc6_0201">
<types><pous>
<pou name="Comm" pouType="program">
<body><FBD>
<inVariable localId="1"><expression>502</expression></inVariable>
<inVariable localId="2"><expression>FALSE</expression></inVariable>
<block localId="10" typeName="Modbus_TCP_Master">
<inputVariables>
<variable formalParameter="PORT">
<connectionPointIn><connection refLocalId="1"/></connectionPointIn></variable>
<variable formalParameter="AUTH">
<connectionPointIn><connection refLocalId="2"/></connectionPointIn></variable>
</inputVariables>
</block>
</FBD></body>
</pou>
</pous></types>
</project>"#;
let ids = rule_ids(xml);
assert!(ids.contains("plc-insecure-protocol-port")); // PORT := 502 (not 502502)
assert!(ids.contains("plc-insecure-comm")); // AUTH := FALSE (not FALSEFALSE)
}
}
-632
View File
@@ -1,632 +0,0 @@
//! Semantic control-logic security rules over the Structured Text AST.
//!
//! Each rule walks the parsed [`Pou`] and yields [`RuleHit`]s the scanner turns
//! into findings. Rules reason over structure (declarations, assignments, calls,
//! array accesses, division, jumps) rather than raw text, so they see through
//! formatting and comments.
use std::collections::{HashMap, HashSet};
use compliance_core::models::Severity;
use super::ast::*;
/// One rule match within a POU.
pub struct RuleHit {
pub line: u32,
pub severity: Severity,
pub rule_id: &'static str,
pub title: String,
pub description: String,
pub cwe: Option<&'static str>,
pub remediation: &'static str,
}
/// Run every rule over a POU.
pub fn analyze(pou: &Pou) -> Vec<RuleHit> {
let mut hits = Vec::new();
let ctx = Ctx::build(pou);
// Declaration-level rules.
for v in &pou.vars {
if let Some(init) = &v.init {
check_credential_binding(&v.name, init, &pou.name, &mut hits);
check_default_password(init, &v.name, &pou.name, &mut hits);
}
}
// Body walk.
walk(&pou.body, pou, &ctx, &GuardSet::default(), &mut hits);
hits
}
/// Per-POU context precomputed once.
struct Ctx {
/// Names declared in VAR_INPUT (untrusted / externally driven).
input_vars: HashSet<String>,
/// Array variable name → declared (lo, hi) bounds.
arrays: HashMap<String, (i64, i64)>,
}
impl Ctx {
fn build(pou: &Pou) -> Self {
let mut input_vars = HashSet::new();
let mut arrays = HashMap::new();
for v in &pou.vars {
if v.section == VarSection::Input {
input_vars.insert(v.name.to_ascii_lowercase());
}
if let Some(b) = v.array_bounds {
arrays.insert(v.name.to_ascii_lowercase(), b);
}
}
Self { input_vars, arrays }
}
}
/// Variables proven non-zero on the current control-flow path (from enclosing
/// `IF`/`WHILE` conditions), so guarded divisions aren't false-flagged.
#[derive(Default, Clone)]
struct GuardSet {
nonzero: HashSet<String>,
}
impl GuardSet {
fn with(&self, names: Vec<String>) -> Self {
let mut g = self.clone();
g.nonzero.extend(names);
g
}
fn is_nonzero(&self, name: &str) -> bool {
self.nonzero.contains(name)
}
}
/// Variable names a condition proves non-zero (`v <> 0`, `v > 0`, `v >= 1`,
/// `v < 0`, and conjunctions thereof).
fn guards_from_cond(cond: &Expr) -> Vec<String> {
let mut out = Vec::new();
collect_nonzero(cond, &mut out);
out
}
fn collect_nonzero(e: &Expr, out: &mut Vec<String>) {
let Expr::Binary { op, lhs, rhs, .. } = e else {
return;
};
let is_zero = |x: &Expr| {
matches!(x, Expr::Int(0, _)) || matches!(x, Expr::Real(r, _) if r.abs() < f64::EPSILON)
};
let int_of = |x: &Expr| match x {
Expr::Int(n, _) => Some(*n),
_ => None,
};
match op {
BinOp::And => {
collect_nonzero(lhs, out);
collect_nonzero(rhs, out);
}
BinOp::Ne => {
if let (Some(v), true) = (lhs.as_ident(), is_zero(rhs)) {
out.push(v.to_ascii_lowercase());
}
if let (true, Some(v)) = (is_zero(lhs), rhs.as_ident()) {
out.push(v.to_ascii_lowercase());
}
}
BinOp::Gt | BinOp::Lt => {
// v > 0 or v < 0
if let (Some(v), true) = (lhs.as_ident(), is_zero(rhs)) {
out.push(v.to_ascii_lowercase());
}
}
BinOp::Ge => {
// v >= n, n >= 1
if let (Some(v), Some(n)) = (lhs.as_ident(), int_of(rhs)) {
if n >= 1 {
out.push(v.to_ascii_lowercase());
}
}
}
_ => {}
}
}
// ── the walker ─────────────────────────────────────────────────────
fn walk(stmts: &[Stmt], pou: &Pou, ctx: &Ctx, guards: &GuardSet, hits: &mut Vec<RuleHit>) {
for s in stmts {
match s {
Stmt::Assign {
target,
value,
line,
} => {
check_safety_bypass(target, value, *line, &pou.name, hits);
// A string bound to a secret-looking target is a credential.
if let Some(name) = flatten_ident(target) {
check_credential_binding(&name, value, &pou.name, hits);
check_default_password(value, &name, &pou.name, hits);
}
walk_expr(target, pou, ctx, guards, hits);
walk_expr(value, pou, ctx, guards, hits);
}
Stmt::Call { callee, args, line } => {
check_insecure_comm(callee, args, *line, &pou.name, hits);
check_credentials_in_call(callee, args, *line, &pou.name, hits);
for a in args {
walk_expr(&a.value, pou, ctx, guards, hits);
}
}
Stmt::Jump { label, line } => hits.push(RuleHit {
line: *line,
severity: Severity::Medium,
rule_id: "plc-unstructured-jump",
title: "Unstructured jump (JMP) in control logic".to_string(),
description: format!(
"POU `{}` uses `JMP {label}`. Unstructured jumps make control flow hard to \
verify and can bypass safety interlocks or leave outputs in an undefined \
state on unexpected paths.",
pou.name
),
cwe: Some("CWE-691"),
remediation: "Replace JMP with structured constructs (IF/CASE/loops); reserve \
jumps for well-reviewed state machines only.",
}),
Stmt::If {
branches,
else_body,
..
} => {
for (cond, body) in branches {
walk_expr(cond, pou, ctx, guards, hits);
let child = guards.with(guards_from_cond(cond));
walk(body, pou, ctx, &child, hits);
}
if let Some(b) = else_body {
walk(b, pou, ctx, guards, hits);
}
}
Stmt::Case {
selector,
arms,
else_body,
..
} => {
walk_expr(selector, pou, ctx, guards, hits);
for (labels, body) in arms {
for l in labels {
walk_expr(l, pou, ctx, guards, hits);
}
walk(body, pou, ctx, guards, hits);
}
if let Some(b) = else_body {
walk(b, pou, ctx, guards, hits);
}
}
Stmt::For {
from, to, by, body, ..
} => {
walk_expr(from, pou, ctx, guards, hits);
walk_expr(to, pou, ctx, guards, hits);
if let Some(b) = by {
walk_expr(b, pou, ctx, guards, hits);
}
walk(body, pou, ctx, guards, hits);
}
Stmt::While { cond, body, .. } => {
walk_expr(cond, pou, ctx, guards, hits);
let child = guards.with(guards_from_cond(cond));
walk(body, pou, ctx, &child, hits);
}
Stmt::Repeat { body, until, .. } => {
walk(body, pou, ctx, guards, hits);
walk_expr(until, pou, ctx, guards, hits);
}
Stmt::Return { .. } | Stmt::Exit { .. } | Stmt::Label { .. } => {}
}
}
}
fn walk_expr(e: &Expr, pou: &Pou, ctx: &Ctx, guards: &GuardSet, hits: &mut Vec<RuleHit>) {
match e {
Expr::Index { base, index, line } => {
check_array_bounds(base, index, *line, ctx, &pou.name, hits);
walk_expr(base, pou, ctx, guards, hits);
walk_expr(index, pou, ctx, guards, hits);
}
Expr::Binary { op, lhs, rhs, line } => {
if matches!(op, BinOp::Div | BinOp::Mod) {
check_division(rhs, *line, &pou.name, guards, hits);
}
walk_expr(lhs, pou, ctx, guards, hits);
walk_expr(rhs, pou, ctx, guards, hits);
}
Expr::Unary { expr, .. } => walk_expr(expr, pou, ctx, guards, hits),
Expr::Member { base, .. } => walk_expr(base, pou, ctx, guards, hits),
Expr::Call { args, .. } => {
for a in args {
walk_expr(&a.value, pou, ctx, guards, hits);
}
}
_ => {}
}
}
// ── individual rules ───────────────────────────────────────────────
const SECRET_HINTS: &[&str] = &[
"password",
"passwd",
"pwd",
"secret",
"apikey",
"api_key",
"token",
"credential",
"privkey",
"private_key",
"passphrase",
];
const DEFAULT_PASSWORDS: &[&str] = &[
"admin",
"administrator",
"password",
"passwd",
"1234",
"12345",
"123456",
"0000",
"1111",
"root",
"default",
"admin123",
"changeme",
"letmein",
"guest",
"user",
"system",
"plc",
"codesys",
];
const COMM_FB_HINTS: &[&str] = &[
"modbus", "tcp", "udp", "socket", "mqtt", "opcua", "opc_ua", "ethernet", "ethip", "enip",
"dnp3", "ftp", "telnet", "http", "send", "connect", "sock", "comm", "profinet", "s7",
];
/// Insecure cleartext service ports.
const INSECURE_PORTS: &[i64] = &[21, 23, 80, 502, 20000, 44818, 102];
fn check_credential_binding(var_name: &str, value: &Expr, pou: &str, hits: &mut Vec<RuleHit>) {
let name = var_name.to_ascii_lowercase();
let looks_secret = SECRET_HINTS.iter().any(|h| name.contains(h));
if looks_secret {
if let Expr::Str(s, line) = value {
if !s.is_empty() {
hits.push(RuleHit {
line: *line,
severity: Severity::High,
rule_id: "plc-hardcoded-credential",
title: "Hardcoded credential in PLC program".to_string(),
description: format!(
"POU `{pou}` binds a hardcoded secret to `{var_name}`. Credentials \
embedded in control logic are extracted trivially from a project export \
or a firmware dump and cannot be rotated without a redeploy."
),
cwe: Some("CWE-798"),
remediation: "Store secrets outside the program (secure parameter store / \
operator-entered, retained-but-protected memory); never commit \
them to the POU.",
});
}
}
}
}
fn check_default_password(value: &Expr, var_name: &str, pou: &str, hits: &mut Vec<RuleHit>) {
if let Expr::Str(s, line) = value {
let lower = s.to_ascii_lowercase();
if DEFAULT_PASSWORDS.contains(&lower.as_str()) {
hits.push(RuleHit {
line: *line,
severity: Severity::Critical,
rule_id: "plc-default-password",
title: "Default/weak password in PLC program".to_string(),
description: format!(
"POU `{pou}` uses the well-known default/weak password `{s}` (bound to \
`{var_name}`). Default PLC credentials are the first thing an attacker tries."
),
cwe: Some("CWE-1393"),
remediation:
"Require a strong, unique, operator-set password; block commissioning \
until the default is changed.",
});
}
}
}
fn check_credentials_in_call(
callee: &str,
args: &[CallArg],
line: u32,
pou: &str,
hits: &mut Vec<RuleHit>,
) {
for a in args {
if let Some(name) = &a.name {
let n = name.to_ascii_lowercase();
if SECRET_HINTS.iter().any(|h| n.contains(h)) {
if let Expr::Str(s, l) = &a.value {
if !s.is_empty() {
hits.push(RuleHit {
line: *l,
severity: Severity::High,
rule_id: "plc-hardcoded-credential",
title: "Hardcoded credential passed to a function block".to_string(),
description: format!(
"POU `{pou}` passes a hardcoded secret as `{name}` to `{callee}`."
),
cwe: Some("CWE-798"),
remediation: "Supply credentials from protected configuration at \
runtime, not as a literal argument.",
});
}
}
}
}
}
let _ = line;
}
fn check_safety_bypass(target: &Expr, value: &Expr, line: u32, pou: &str, hits: &mut Vec<RuleHit>) {
let Some(name) = flatten_ident(target) else {
return;
};
let n = name.to_ascii_lowercase();
let safety = [
"safety",
"estop",
"e_stop",
"emergency",
"interlock",
"guard",
"permit",
]
.iter()
.any(|h| n.contains(h));
let watchdog = n.contains("watchdog") || n.contains("wdt");
// A safety enable / interlock / watchdog signal driven to FALSE or 0 in
// application logic is a bypass (e.g. `Safety_Enable := FALSE`, `Watchdog_Kick := 0`).
let disabling = matches!(value, Expr::Bool(false, _)) || matches!(value, Expr::Int(0, _));
if (safety || watchdog) && disabling {
hits.push(RuleHit {
line,
severity: Severity::Critical,
rule_id: "plc-safety-bypass",
title: "Safety interlock / watchdog disabled in logic".to_string(),
description: format!(
"POU `{pou}` disables a safety-related signal (`{name}`) in program logic. \
Bypassing interlocks or watchdogs in code defeats the plant's protective \
functions and is a direct hazard."
),
cwe: Some("CWE-1384"),
remediation: "Never disable safety functions from application logic; safety must be \
handled by a certified safety controller / hard-wired circuit.",
});
}
}
fn check_array_bounds(
base: &Expr,
index: &Expr,
line: u32,
ctx: &Ctx,
pou: &str,
hits: &mut Vec<RuleHit>,
) {
// Only reason about arrays we know the bounds of.
let Some(arr_name) = base.as_ident() else {
return;
};
if !ctx.arrays.contains_key(&arr_name.to_ascii_lowercase()) {
return;
}
// Index by an untrusted input variable → potential out-of-bounds access.
if let Some(idx_name) = index.as_ident() {
if ctx.input_vars.contains(&idx_name.to_ascii_lowercase()) {
hits.push(RuleHit {
line,
severity: Severity::High,
rule_id: "plc-array-unchecked-index",
title: "Array indexed by unvalidated input".to_string(),
description: format!(
"POU `{pou}` indexes array `{arr_name}` with the input variable `{idx_name}` \
without a validated bounds check. An out-of-range index corrupts adjacent \
memory or faults the PLC (loss of control)."
),
cwe: Some("CWE-129"),
remediation: "Clamp or validate the index against the array bounds (e.g. \
`LIMIT`/explicit `IF idx >= lo AND idx <= hi`) before the access.",
});
}
}
}
fn check_division(
divisor: &Expr,
line: u32,
pou: &str,
guards: &GuardSet,
hits: &mut Vec<RuleHit>,
) {
// A divisor proven non-zero by an enclosing guard is safe.
if let Expr::Ident(name, _) = divisor {
if guards.is_nonzero(&name.to_ascii_lowercase()) {
return;
}
}
// Flag division by a variable (could be zero); nonzero literals are fine.
let risky = matches!(
divisor,
Expr::Ident(_, _) | Expr::Member { .. } | Expr::Index { .. } | Expr::Int(0, _)
);
if risky {
hits.push(RuleHit {
line,
severity: Severity::Medium,
rule_id: "plc-division-by-zero",
title: "Division by a variable without a zero-guard".to_string(),
description: format!(
"POU `{pou}` divides by a variable that is not proven non-zero. A zero divisor \
raises a PLC exception and can halt the scan cycle (denial of control)."
),
cwe: Some("CWE-369"),
remediation: "Guard the divisor (`IF d <> 0 THEN …`) or use a safe-divide helper that \
returns a defined value for a zero denominator.",
});
}
}
fn check_insecure_comm(
callee: &str,
args: &[CallArg],
line: u32,
pou: &str,
hits: &mut Vec<RuleHit>,
) {
let c = callee.to_ascii_lowercase();
let is_comm = COMM_FB_HINTS.iter().any(|h| c.contains(h));
if !is_comm {
return;
}
// Auth/encryption explicitly disabled.
for a in args {
if let Some(name) = &a.name {
let n = name.to_ascii_lowercase();
let security_flag = ["auth", "secure", "encrypt", "tls", "ssl", "authentication"]
.iter()
.any(|h| n.contains(h));
if security_flag && matches!(a.value, Expr::Bool(false, _)) {
hits.push(RuleHit {
line,
severity: Severity::High,
rule_id: "plc-insecure-comm",
title: "Network communication with security disabled".to_string(),
description: format!(
"POU `{pou}` calls `{callee}` with `{name} := FALSE`, disabling \
authentication/encryption on an industrial network link."
),
cwe: Some("CWE-319"),
remediation: "Enable authentication + transport encryption; segment OT \
networks and restrict the endpoint to trusted peers.",
});
}
}
// Well-known cleartext port literal.
if let Expr::Int(p, _) = &a.value {
if INSECURE_PORTS.contains(p) {
hits.push(RuleHit {
line,
severity: Severity::Medium,
rule_id: "plc-insecure-protocol-port",
title: "Cleartext industrial protocol port".to_string(),
description: format!(
"POU `{pou}` opens `{callee}` on port {p}, a well-known cleartext OT \
protocol port with no built-in authentication or encryption."
),
cwe: Some("CWE-319"),
remediation: "Front the protocol with a secure gateway/VPN, or use the \
authenticated/encrypted variant; never expose it to untrusted \
networks.",
});
}
}
}
let _ = line;
}
/// The dotted/base identifier of an lvalue expression (`a`, `a.b` → `a.b`,
/// `a[i]` → `a`), for name-based rules.
fn flatten_ident(e: &Expr) -> Option<String> {
match e {
Expr::Ident(n, _) => Some(n.clone()),
Expr::Member { base, field, .. } => flatten_ident(base).map(|b| format!("{b}.{field}")),
Expr::Index { base, .. } => flatten_ident(base),
_ => None,
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::pipeline::plc::parser;
const VULN: &str = r#"
FUNCTION_BLOCK CommCtrl
VAR_INPUT
cmdIndex : INT;
END_VAR
VAR
Password : STRING := 'admin123';
buffer : ARRAY[0..15] OF INT;
Safety_Enable : BOOL := TRUE;
divisor : INT;
result : INT;
END_VAR
Safety_Enable := FALSE;
result := 100 / divisor;
buffer[cmdIndex] := 1;
Modbus_Connect(IP := '192.168.0.10', PORT := 502, AUTH := FALSE);
IF cmdIndex > 100 THEN
JMP fault;
END_IF;
fault:
result := 0;
END_FUNCTION_BLOCK
"#;
fn rule_ids(src: &str) -> Vec<&'static str> {
parser::parse(src)
.iter()
.flat_map(analyze)
.map(|h| h.rule_id)
.collect()
}
#[test]
fn vulnerable_program_triggers_every_rule() {
let ids = rule_ids(VULN);
for expected in [
"plc-hardcoded-credential",
"plc-default-password",
"plc-safety-bypass",
"plc-division-by-zero",
"plc-array-unchecked-index",
"plc-insecure-comm",
"plc-insecure-protocol-port",
"plc-unstructured-jump",
] {
assert!(
ids.contains(&expected),
"expected rule {expected}, got {ids:?}"
);
}
}
#[test]
fn clean_program_has_no_findings() {
let clean = r#"
PROGRAM Clean
VAR
a : INT := 5;
b : INT := 3;
total : INT;
END_VAR
IF b <> 0 THEN
total := a / b;
END_IF;
END_PROGRAM
"#;
assert!(rule_ids(clean).is_empty(), "clean program should be quiet");
}
}
-253
View File
@@ -1,253 +0,0 @@
//! Control-application dependency SBOM from a CODESYS `.projectarchive`.
//!
//! A `.projectarchive` is a ZIP that bundles the project plus its referenced
//! libraries and the target runtime. Each referenced library is an entry whose
//! path segment follows the CODESYS convention
//! `Name, Major.Minor.Patch.Build (Company)` (e.g. `Standard, 3.5.18.0 (System)`,
//! `CSV Utility SL, 1.9.0.0 (CODESYS)`); the runtime appears as a device-descriptor
//! entry `CODESYS Control … <version> …`. We enumerate those entries — no binary
//! parsing — and emit SBOM components tagged `pkg:codesys/…`, so the CVE pipeline
//! can match them (the runtime `Cmp*` / `3SLicense` components carry real CODESYS
//! CVEs).
use std::collections::BTreeSet;
use std::path::{Path, PathBuf};
use compliance_core::models::SbomEntry;
/// Collect the control-application SBOM from every `.projectarchive` reachable for
/// a target: the ingested artifact file itself (an uploaded archive), plus any
/// `*.projectarchive` committed inside the working tree — e.g. a git repo or an
/// extracted source archive that ships the archive alongside its PLCopen XML / ST
/// exports. Deduplicated by (name, version).
pub fn collect_sbom(artifact_file: &Path, working_path: &Path, repo_id: &str) -> Vec<SbomEntry> {
let mut archives: Vec<PathBuf> = Vec::new();
if artifact_file.is_file() {
archives.push(artifact_file.to_path_buf());
}
for entry in walkdir::WalkDir::new(working_path)
.max_depth(8)
.into_iter()
.filter_map(|e| e.ok())
{
let p = entry.path();
if entry.file_type().is_file()
&& p.extension()
.and_then(|x| x.to_str())
.is_some_and(|x| x.eq_ignore_ascii_case("projectarchive"))
{
archives.push(p.to_path_buf());
}
}
let mut seen: BTreeSet<(String, String)> = BTreeSet::new();
let mut out = Vec::new();
for a in archives {
for e in projectarchive_sbom(&a, repo_id) {
if seen.insert((e.name.clone(), e.version.clone())) {
out.push(e);
}
}
}
out
}
/// Extract CODESYS library + runtime components from a `.projectarchive` (a zip).
/// Best-effort: returns empty if the file is not a readable zip (e.g. a bare
/// `.st`/`.xml` project, which carries no library manifest).
pub fn projectarchive_sbom(archive: &Path, repo_id: &str) -> Vec<SbomEntry> {
let Ok(file) = std::fs::File::open(archive) else {
return Vec::new();
};
let Ok(mut zip) = zip::ZipArchive::new(file) else {
return Vec::new();
};
let mut seen: BTreeSet<(String, String)> = BTreeSet::new();
let mut entries = Vec::new();
for i in 0..zip.len() {
let Ok(entry) = zip.by_index(i) else {
continue;
};
// Entry paths use `\` (Windows-authored) and/or `/` separators; the
// component id is one path segment.
for seg in entry.name().split(['/', '\\']) {
if let Some((name, version)) = parse_library(seg).or_else(|| parse_runtime(seg)) {
if seen.insert((name.clone(), version.clone())) {
let mut e = SbomEntry::new(
repo_id.to_string(),
name.clone(),
version.clone(),
"codesys".to_string(),
);
e.purl = Some(format!(
"pkg:codesys/{}@{version}",
name.replace(' ', "%20")
));
entries.push(e);
}
}
}
}
entries
}
/// `Name, X.Y.Z.W (Company)` → (name, version).
fn parse_library(seg: &str) -> Option<(String, String)> {
let seg = seg.trim();
// Company is the trailing "(…)".
let open = seg.rfind(" (")?;
let rest = &seg[open + 2..];
let close = rest.find(')')?;
if rest[..close].trim().is_empty() {
return None;
}
let head = seg[..open].trim(); // "Name, X.Y.Z.W"
let comma = head.rfind(", ")?;
let name = head[..comma].trim().to_string();
let version = head[comma + 2..].trim().to_string();
if name.is_empty() || !is_dotted_version(&version) {
return None;
}
Some((name, version))
}
/// Device-descriptor entry `CODESYS Control … X.Y.Z.W …` → (runtime name, version).
fn parse_runtime(seg: &str) -> Option<(String, String)> {
let seg = seg.trim();
if !seg.starts_with("CODESYS Control") {
return None;
}
let version = seg
.split_whitespace()
.find(|t| is_dotted_version(t))?
.to_string();
// The runtime name is the first field, before the run of padding spaces that
// precede the descriptor's numeric columns.
let name = seg.split(" ").next().unwrap_or(seg).trim().to_string();
if name.is_empty() {
return None;
}
Some((name, version))
}
/// A dotted numeric version with at least 3 components (`3.5.18.0`, `4.17.0.0`).
fn is_dotted_version(s: &str) -> bool {
let parts: Vec<&str> = s.split('.').collect();
parts.len() >= 3
&& parts
.iter()
.all(|p| !p.is_empty() && p.chars().all(|c| c.is_ascii_digit()))
}
#[cfg(test)]
mod tests {
use super::*;
use std::collections::HashMap;
use std::io::Write;
/// Build a synthetic `.projectarchive` (zip) mirroring the real CODESYS entry
/// naming (verified against Proemion/codesys-examples): a native `.project`,
/// referenced libraries as `Name, Version (Company)` segments, and a runtime
/// device descriptor.
fn synthetic_archive(dir: &Path) -> std::path::PathBuf {
let path = dir.join("App.projectarchive");
write_synthetic_archive(&path);
path
}
fn write_synthetic_archive(path: &Path) {
let file = std::fs::File::create(path).expect("create");
let mut zip = zip::ZipWriter::new(file);
let opts: zip::write::SimpleFileOptions = Default::default();
let names = [
"App.project",
r"{b0b5}\App.Device.Plc.compileinfo",
r"{e179}\Standard, 3.5.18.0 (System) standard.compiled-library-v3",
r"{e179}\Util, 3.5.21.0 (System) util.compiled-library-v3",
r"{e179}\CSV Utility SL, 1.9.0.0 (CODESYS) csv utility sl.compiled-library-v3",
r"{e179}\3SLicense, 3.5.20.0 (CODESYS) 3slicense.compiled-library-v3",
r"{0c63}\CODESYS Control for Linux ARM SL 0000 0006 4.17.0.0 4096 .zip",
];
for n in names {
zip.start_file(n, opts).expect("start");
zip.write_all(b"x").expect("write");
}
zip.finish().expect("finish");
}
#[test]
fn extracts_libraries_and_runtime_from_projectarchive() {
let tmp = std::env::temp_dir().join(format!("cs-plc-sbom-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&tmp).expect("mkdir");
let archive = synthetic_archive(&tmp);
let entries = projectarchive_sbom(&archive, "plc-target");
let by_name: HashMap<&str, &SbomEntry> =
entries.iter().map(|e| (e.name.as_str(), e)).collect();
// Libraries with their versions.
assert_eq!(
by_name.get("Standard").map(|e| e.version.as_str()),
Some("3.5.18.0")
);
assert_eq!(
by_name.get("Util").map(|e| e.version.as_str()),
Some("3.5.21.0")
);
assert_eq!(
by_name.get("CSV Utility SL").map(|e| e.version.as_str()),
Some("1.9.0.0"),
"multi-word library names must parse"
);
assert!(by_name.contains_key("3SLicense"));
// The runtime, from the device descriptor.
assert_eq!(
by_name
.get("CODESYS Control for Linux ARM SL")
.map(|e| e.version.as_str()),
Some("4.17.0.0")
);
// Every component is CODESYS-tagged with a purl the CVE pipeline can match,
// and the native `.project` / compileinfo are not mistaken for components.
for e in &entries {
assert_eq!(e.package_manager, "codesys");
assert!(e.purl.as_deref().unwrap_or("").starts_with("pkg:codesys/"));
}
assert!(!by_name.contains_key("App"));
let _ = std::fs::remove_dir_all(&tmp);
}
#[test]
fn collect_sbom_finds_a_projectarchive_committed_in_a_git_tree() {
let tmp = std::env::temp_dir().join(format!("cs-plc-collect-{}", uuid::Uuid::new_v4()));
let src = tmp.join("clone/src");
std::fs::create_dir_all(&src).expect("mkdir");
// Simulate a git clone that commits the archive alongside its exports.
write_synthetic_archive(&src.join("PumpStation.projectarchive"));
// The artifact "file" is a git URL (not a real file), so the SBOM must
// come from walking the cloned tree.
let entries = collect_sbom(Path::new("https://git.example/plc.git"), &tmp, "t");
let names: std::collections::HashSet<&str> =
entries.iter().map(|e| e.name.as_str()).collect();
assert!(
names.contains("Standard"),
"found libs in the committed archive"
);
assert!(names.contains("CODESYS Control for Linux ARM SL"));
let _ = std::fs::remove_dir_all(&tmp);
}
#[test]
fn non_zip_file_yields_no_sbom() {
let tmp = std::env::temp_dir().join(format!("cs-plc-sbom-st-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&tmp).expect("mkdir");
let st = tmp.join("prog.st");
std::fs::write(&st, "PROGRAM P\nVAR x : INT; END_VAR\nEND_PROGRAM\n").expect("write");
assert!(projectarchive_sbom(&st, "t").is_empty());
let _ = std::fs::remove_dir_all(&tmp);
}
}
+1 -2
View File
@@ -1,4 +1,3 @@
use crate::pipeline::repo_view::RepoView;
use compliance_core::models::*;
use super::dedup::compute_fingerprint;
@@ -15,7 +14,7 @@ impl PipelineOrchestrator {
#[tracing::instrument(skip_all, fields(repo_id = %repo_id, pr_number))]
pub async fn run_pr_review(
&self,
repo: &RepoView,
repo: &TrackedRepository,
repo_id: &str,
pr_number: u64,
base_sha: &str,
@@ -1,74 +0,0 @@
//! `RepoView` — an internal, non-persisted view of a code target for the scan
//! pipeline.
//!
//! It replaces the old persisted `TrackedRepository` model. The pipeline
//! (SAST → SBOM → CVE → triage → issues → DAST, and PR review) only ever needs a
//! flat bundle of git + issue-tracker + auth fields; those are projected from an
//! [`OnboardedTarget`] and its code [`Artifact`] by [`RepoView::from_target`].
//! Nothing here is written to Mongo — onboarded targets are the sole persisted
//! entity.
use compliance_core::models::{Artifact, OnboardedTarget, TrackerType};
/// A flat, pipeline-facing view of a code target. Built from an onboarded
/// target; never persisted.
#[derive(Debug, Clone)]
pub struct RepoView {
/// The onboarded target's id (used as `repo_id` across findings/sbom/etc.).
pub id: Option<mongodb::bson::oid::ObjectId>,
pub name: String,
pub git_url: String,
pub default_branch: String,
pub local_path: Option<String>,
pub scan_schedule: Option<String>,
pub webhook_enabled: bool,
pub webhook_secret: Option<String>,
pub tracker_type: Option<TrackerType>,
pub tracker_owner: Option<String>,
pub tracker_repo: Option<String>,
pub tracker_token: Option<String>,
pub auth_token: Option<String>,
pub auth_username: Option<String>,
pub last_scanned_commit: Option<String>,
pub findings_count: u32,
}
impl RepoView {
/// Project an onboarded target + its code artifact into a pipeline view.
pub fn from_target(target: &OnboardedTarget, code: &Artifact) -> Self {
let mut view = Self {
id: target.id,
name: target.name.clone(),
git_url: code.source_ref.clone(),
default_branch: "main".to_string(),
local_path: None,
scan_schedule: target.scan_schedule.clone(),
webhook_enabled: target.webhook_enabled,
webhook_secret: target.webhook_secret.clone(),
tracker_type: None,
tracker_owner: None,
tracker_repo: None,
tracker_token: None,
auth_token: None,
auth_username: None,
last_scanned_commit: None,
findings_count: target.findings_count,
};
if let Some(git) = &code.git {
view.default_branch = git.default_branch.clone();
view.last_scanned_commit = git.last_scanned_commit.clone();
view.local_path = git.local_path.clone();
}
if let Some(auth) = &code.auth {
view.auth_token = auth.secret.clone();
view.auth_username = auth.username.clone();
}
if let Some(it) = &target.scan_config.issue_tracker {
view.tracker_type = it.tracker_type.clone();
view.tracker_owner = it.owner.clone();
view.tracker_repo = it.repo.clone();
view.tracker_token = it.token.clone();
}
view
}
}
+29 -70
View File
@@ -1,4 +1,4 @@
use std::path::{Path, PathBuf};
use std::path::Path;
use compliance_core::models::{Finding, ScanType, Severity};
use compliance_core::traits::{ScanOutput, Scanner};
@@ -6,30 +6,6 @@ use compliance_core::CoreError;
use crate::pipeline::dedup;
/// Custom CRA-control detectors bundled into the binary and staged to a temp file
/// at scan time so semgrep can `--config` them alongside the auto ruleset. These
/// cover controls no off-the-shelf rule digs out (secure defaults, weak password
/// hashing, insecure session cookies, weak data-at-rest ciphers); each rule id is
/// keyed back to its control by the `control-map` LUT.
const CRA_RULES: &str = include_str!("../../rules/cra_semgrep.yaml");
/// Write the bundled CRA rules to a stable temp path (atomic: unique tmp +
/// rename). Returns `None` on failure — the scan then runs with auto rules only.
async fn stage_cra_rules() -> Option<PathBuf> {
let dir = std::env::temp_dir();
let path = dir.join("compliance-cra-semgrep.yaml");
let tmp = dir.join(format!("compliance-cra-semgrep.{}.tmp", std::process::id()));
if let Err(e) = tokio::fs::write(&tmp, CRA_RULES).await {
tracing::warn!(error = %e, "failed to stage custom CRA semgrep rules; using auto rules only");
return None;
}
if let Err(e) = tokio::fs::rename(&tmp, &path).await {
tracing::warn!(error = %e, "failed to stage custom CRA semgrep rules; using auto rules only");
return None;
}
Some(path)
}
pub struct SemgrepScanner;
impl Scanner for SemgrepScanner {
@@ -43,26 +19,30 @@ impl Scanner for SemgrepScanner {
#[tracing::instrument(skip_all)]
async fn scan(&self, repo_path: &Path, repo_id: &str) -> Result<ScanOutput, CoreError> {
let cra_rules = stage_cra_rules().await;
let mut command = tokio::process::Command::new("semgrep");
command.arg("--config=auto");
if let Some(path) = &cra_rules {
command.arg(format!("--config={}", path.display()));
}
command
.args(["--json", "--quiet", "--max-memory", "500", "--jobs", "1"])
.arg(repo_path);
let output = tokio::time::timeout(std::time::Duration::from_secs(600), command.output())
.await
.map_err(|_| CoreError::Scanner {
scanner: "semgrep".to_string(),
source: "timed out after 10 minutes".into(),
})?
.map_err(|e| CoreError::Scanner {
scanner: "semgrep".to_string(),
source: Box::new(e),
})?;
let output = tokio::time::timeout(
std::time::Duration::from_secs(600),
tokio::process::Command::new("semgrep")
.args([
"--config=auto",
"--json",
"--quiet",
"--max-memory",
"500",
"--jobs",
"1",
])
.arg(repo_path)
.output(),
)
.await
.map_err(|_| CoreError::Scanner {
scanner: "semgrep".to_string(),
source: "timed out after 10 minutes".into(),
})?
.map_err(|e| CoreError::Scanner {
scanner: "semgrep".to_string(),
source: Box::new(e),
})?;
if !output.status.success() && output.stdout.is_empty() {
let stderr = String::from_utf8_lossy(&output.stderr);
@@ -102,7 +82,10 @@ impl Scanner for SemgrepScanner {
finding.file_path = Some(r.path);
finding.line_number = Some(r.start.line);
finding.code_snippet = Some(r.extra.lines);
finding.cwe = r.extra.metadata.as_ref().and_then(extract_cwe);
finding.cwe = r
.extra
.metadata
.and_then(|m| m.get("cwe").and_then(|v| v.as_str()).map(|s| s.to_string()));
finding
})
.collect();
@@ -141,34 +124,10 @@ struct SemgrepExtra {
metadata: Option<serde_json::Value>,
}
/// semgrep emits `metadata.cwe` as a list of strings like
/// `"CWE-798: Use of Hard-coded Credentials"` (occasionally a bare string). Take
/// the first entry and normalise it to just the `CWE-NNN` id.
fn extract_cwe(metadata: &serde_json::Value) -> Option<String> {
let raw = metadata.get("cwe")?;
let text = match raw {
serde_json::Value::Array(items) => items.first()?.as_str()?,
serde_json::Value::String(s) => s.as_str(),
_ => return None,
};
let id = text.split(':').next().unwrap_or(text).trim();
(!id.is_empty()).then(|| id.to_string())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn extract_cwe_handles_list_and_normalises() {
let md = serde_json::json!({"cwe": ["CWE-798: Use of Hard-coded Credentials"]});
assert_eq!(extract_cwe(&md).as_deref(), Some("CWE-798"));
let bare = serde_json::json!({"cwe": "CWE-89"});
assert_eq!(extract_cwe(&bare).as_deref(), Some("CWE-89"));
let none = serde_json::json!({"severity": "ERROR"});
assert_eq!(extract_cwe(&none), None);
}
#[test]
fn deserialize_semgrep_output() {
let json = r#"{
+1 -1
View File
@@ -348,7 +348,7 @@ async fn monitor_cves(agent: &ComplianceAgent, tenant_id: &str) {
std::collections::HashMap::new();
for rid in &repo_ids {
if let Ok(oid) = mongodb::bson::oid::ObjectId::parse_str(rid) {
if let Ok(Some(repo)) = db.onboarded_targets().find_one(doc! { "_id": oid }).await {
if let Ok(Some(repo)) = db.repositories().find_one(doc! { "_id": oid }).await {
repo_names.insert(rid.clone(), repo.name.clone());
}
}
+1 -1
View File
@@ -31,7 +31,7 @@ pub async fn handle_gitea_webhook(
}
};
let repo = match db
.onboarded_targets()
.repositories()
.find_one(mongodb::bson::doc! { "_id": oid })
.await
{
+1 -1
View File
@@ -31,7 +31,7 @@ pub async fn handle_github_webhook(
}
};
let repo = match db
.onboarded_targets()
.repositories()
.find_one(mongodb::bson::doc! { "_id": oid })
.await
{
+1 -1
View File
@@ -27,7 +27,7 @@ pub async fn handle_gitlab_webhook(
}
};
let repo = match db
.onboarded_targets()
.repositories()
.find_one(mongodb::bson::doc! { "_id": oid })
.await
{
-10
View File
@@ -1,10 +0,0 @@
//! Werkbank control-plane: the dynamic-execution job queue.
//!
//! The control plane enqueues declarative [`Job`](compliance_core::models::werkbank::Job)s
//! and Werkbank runners lease, run, and complete them. [`queue::JobQueue`] is the
//! Mongo-backed queue behind that flow (WB-02); the runner-facing HTTP transport
//! and the runner itself land in later stories.
pub mod queue;
pub use queue::{JobQueue, SweepOutcome};
-309
View File
@@ -1,309 +0,0 @@
//! The Mongo-backed Werkbank job queue (WB-02).
//!
//! A pull queue: the control plane [`enqueue`](JobQueue::enqueue)s jobs; a runner
//! [`lease`](JobQueue::lease)s the oldest queued job it can run (matched by
//! executor + labels), [`heartbeat`](JobQueue::heartbeat)s while it works, and
//! [`complete`](JobQueue::complete)s it. Leases carry a visibility timeout: if a
//! runner dies mid-job its heartbeats stop, the lease expires, and
//! [`sweep_expired`](JobQueue::sweep_expired) returns the job to `queued` (or
//! `expired` once it has been retried too many times).
//!
//! All state transitions are single atomic Mongo updates guarded by the lease
//! token, so two runners can never both own a job. Every operation takes an
//! explicit `now` so the queue's time-dependent behaviour is deterministically
//! testable.
use std::time::Duration;
use chrono::{DateTime, Utc};
use mongodb::bson::{doc, Bson, DateTime as BsonDateTime};
use mongodb::error::{ErrorKind, WriteFailure};
use mongodb::options::ReturnDocument;
use mongodb::Collection;
use compliance_core::models::werkbank::{
Executor, HeartbeatAck, Job, JobRecord, JobResult, JobStatus, LeasedJob,
};
use crate::database::Database;
use crate::error::AgentError;
/// The non-terminal states a job can be swept or cancelled from.
const ACTIVE_STATES: [&str; 2] = ["leased", "running"];
/// Every terminal state (no further transitions).
const TERMINAL_STATES: [&str; 4] = ["succeeded", "failed", "expired", "cancelled"];
/// What a visibility-timeout sweep did.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub struct SweepOutcome {
/// Expired-lease jobs returned to `queued` for another runner.
pub requeued: u64,
/// Jobs that had exhausted their attempts and were marked `expired`.
pub expired: u64,
}
/// The Mongo-backed job queue.
pub struct JobQueue {
coll: Collection<JobRecord>,
}
impl JobQueue {
/// Build a queue over a tenant database's `werkbank_jobs` collection.
pub fn new(db: &Database) -> Self {
Self {
coll: db.werkbank_jobs(),
}
}
/// Enqueue a job. Idempotent by job id: a job that is already present is a
/// no-op. Returns `true` if this call inserted it, `false` if it existed.
pub async fn enqueue(&self, job: Job, now: DateTime<Utc>) -> Result<bool, AgentError> {
let record = JobRecord::queued(job, now);
match self.coll.insert_one(&record).await {
Ok(_) => Ok(true),
Err(e) if is_duplicate_key(&e) => Ok(false),
Err(e) => Err(e.into()),
}
}
/// Atomically lease the oldest `queued` job this runner can run — matched by
/// executor and by labels (every label the job requires must be one the
/// runner advertises). Returns the job plus a lease token, or `None` if
/// nothing is runnable.
pub async fn lease(
&self,
runner_id: &str,
executor: Executor,
runner_labels: &[String],
lease_ttl: Duration,
now: DateTime<Utc>,
) -> Result<Option<LeasedJob>, AgentError> {
let token = uuid::Uuid::new_v4().to_string();
let expires = bson_dt(now + ttl(lease_ttl));
let executor_bson = mongodb::bson::to_bson(&executor).unwrap_or(Bson::Null);
let filter = doc! {
"status": "queued",
"cancel_requested": { "$ne": true },
"job.executor": executor_bson,
// Every label the job requires must be in the runner's set — i.e. the
// job has no label that is not offered by the runner. Absent/empty
// job labels match any runner.
"job.labels": { "$not": { "$elemMatch": { "$nin": runner_labels.to_vec() } } },
};
let update = doc! {
"$set": {
"status": "leased",
"lease_token": &token,
"leased_by": runner_id,
"lease_expires_at": expires,
"heartbeat_at": bson_dt(now),
"updated_at": bson_dt(now),
},
"$inc": { "attempts": 1 },
};
let record = self
.coll
.find_one_and_update(filter, update)
.sort(doc! { "created_at": 1 }) // FIFO
.return_document(ReturnDocument::After)
.await?;
Ok(record.map(|r| LeasedJob {
job: r.job,
lease_token: token,
}))
}
/// Extend a lease and report whether the job has been asked to cancel.
/// Transitions the job to `running` on the first heartbeat. Returns `None`
/// when the lease is no longer valid (token mismatch, or the job is already
/// terminal) — the runner should then abandon the work.
pub async fn heartbeat(
&self,
job_id: &str,
lease_token: &str,
lease_ttl: Duration,
now: DateTime<Utc>,
) -> Result<Option<HeartbeatAck>, AgentError> {
let filter = doc! {
"job.id": job_id,
"lease_token": lease_token,
"status": { "$in": ACTIVE_STATES.to_vec() },
};
let update = doc! {
"$set": {
"status": "running",
"lease_expires_at": bson_dt(now + ttl(lease_ttl)),
"heartbeat_at": bson_dt(now),
"updated_at": bson_dt(now),
},
};
let record = self
.coll
.find_one_and_update(filter, update)
.return_document(ReturnDocument::After)
.await?;
Ok(record.map(|r| HeartbeatAck {
cancelled: r.cancel_requested,
}))
}
/// Record a job's terminal result. Guarded by the lease token and only from
/// an active (`leased`/`running`) state, so it is idempotent — a duplicate or
/// late submission after the job already finished matches nothing. Returns
/// `true` if this call recorded the result.
pub async fn complete(
&self,
job_id: &str,
lease_token: &str,
result: &JobResult,
now: DateTime<Utc>,
) -> Result<bool, AgentError> {
let status = result.status.unwrap_or(JobStatus::Failed);
let status_bson = mongodb::bson::to_bson(&status).unwrap_or(Bson::String("failed".into()));
let result_bson =
mongodb::bson::to_bson(result).map_err(|e| AgentError::Other(e.to_string()))?;
let filter = doc! {
"job.id": job_id,
"lease_token": lease_token,
"status": { "$in": ACTIVE_STATES.to_vec() },
};
let update = doc! {
"$set": {
"status": status_bson,
"result": result_bson,
"lease_token": Bson::Null,
"lease_expires_at": Bson::Null,
"updated_at": bson_dt(now),
},
};
let res = self.coll.update_one(filter, update).await?;
Ok(res.modified_count == 1)
}
/// Request cancellation of a job. A still-`queued` job is cancelled outright;
/// an in-flight one is flagged so the runner sees it on its next heartbeat and
/// tears down. Returns `true` if a non-terminal job matched.
pub async fn cancel(&self, job_id: &str, now: DateTime<Utc>) -> Result<bool, AgentError> {
let filter = doc! {
"job.id": job_id,
"status": { "$nin": TERMINAL_STATES.to_vec() },
};
// Pipeline update: flag cancellation, and if still queued flip straight to
// cancelled (nothing is running it).
let pipeline = vec![doc! {
"$set": {
"cancel_requested": true,
"status": {
"$cond": [ { "$eq": ["$status", "queued"] }, "cancelled", "$status" ]
},
"updated_at": bson_dt(now),
}
}];
let res = self.coll.update_one(filter, pipeline).await?;
Ok(res.matched_count == 1)
}
/// Sweep leases whose visibility timeout has elapsed: return them to `queued`
/// for another runner, or mark them `expired` once they have been leased
/// `max_attempts` times. This is what makes a crashed runner's job recover.
pub async fn sweep_expired(
&self,
now: DateTime<Utc>,
max_attempts: u32,
// (kept explicit rather than a const so callers can tune retry policy)
) -> Result<SweepOutcome, AgentError> {
let now_bson = bson_dt(now);
let max = i64::from(max_attempts);
let requeue = self
.coll
.update_many(
doc! {
"status": { "$in": ACTIVE_STATES.to_vec() },
"lease_expires_at": { "$lt": &now_bson },
"attempts": { "$lt": max },
},
doc! { "$set": {
"status": "queued",
"lease_token": Bson::Null,
"leased_by": Bson::Null,
"lease_expires_at": Bson::Null,
"updated_at": &now_bson,
} },
)
.await?;
let expire = self
.coll
.update_many(
doc! {
"status": { "$in": ACTIVE_STATES.to_vec() },
"lease_expires_at": { "$lt": &now_bson },
"attempts": { "$gte": max },
},
doc! { "$set": {
"status": "expired",
"lease_token": Bson::Null,
"lease_expires_at": Bson::Null,
"updated_at": &now_bson,
} },
)
.await?;
Ok(SweepOutcome {
requeued: requeue.modified_count,
expired: expire.modified_count,
})
}
/// Fetch a job record by job id (inspection / control-plane reads).
pub async fn get(&self, job_id: &str) -> Result<Option<JobRecord>, AgentError> {
Ok(self.coll.find_one(doc! { "job.id": job_id }).await?)
}
}
/// A `chrono::Duration` for a lease TTL, saturating rather than panicking on an
/// absurd input (`chrono::Duration::seconds` panics past its internal bound).
fn ttl(d: Duration) -> chrono::Duration {
let secs = i64::try_from(d.as_secs()).unwrap_or(i64::MAX);
chrono::Duration::try_seconds(secs).unwrap_or(chrono::Duration::MAX)
}
/// A chrono instant as a BSON date (so Mongo stores/compares it as a real date).
fn bson_dt(dt: DateTime<Utc>) -> BsonDateTime {
BsonDateTime::from_chrono(dt)
}
/// Whether a Mongo error is a duplicate-key (E11000) violation — a job with this
/// id is already enqueued.
fn is_duplicate_key(e: &mongodb::error::Error) -> bool {
match &*e.kind {
ErrorKind::Write(WriteFailure::WriteError(we)) => we.code == 11000,
_ => false,
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn ttl_saturates_and_converts() {
assert_eq!(ttl(Duration::from_secs(30)), chrono::Duration::seconds(30));
// An absurd TTL saturates instead of panicking.
assert_eq!(ttl(Duration::from_secs(u64::MAX)), chrono::Duration::MAX);
}
#[test]
fn state_constants_are_disjoint() {
for s in ACTIVE_STATES {
assert!(
!TERMINAL_STATES.contains(&s),
"{s} cannot be both active and terminal"
);
}
}
}
-125
View File
@@ -1,125 +0,0 @@
//! C5 example 2 — exploratory (not a committed regression test). Four topically
//! distinct findings, to see whether tuned semantic retrieval maps each to the
//! right master-control family. Run:
//! export ... (LITELLM_* + BREAKPILOT_BASE_URL)
//! cargo test -p compliance-agent --test c5_example2 -- --ignored --nocapture
mod common;
use std::sync::Arc;
use compliance_agent::llm::LlmClient;
use compliance_core::config::BreakpilotConfig;
use compliance_core::models::finding::{Finding, Severity};
use compliance_core::models::scan::ScanType;
use secrecy::SecretString;
fn env(k: &str) -> String {
std::env::var(k).unwrap_or_else(|_| panic!("env {k} must be set"))
}
fn mk(file: &str, line: u32, title: &str, desc: &str) -> Finding {
let mut f = Finding::new(
"repo-c5b".into(),
format!("{file}:{line}"),
"semgrep".into(),
ScanType::Sast,
title.into(),
desc.into(),
Severity::High,
);
f.file_path = Some(file.into());
f.line_number = Some(line);
f
}
fn write(repo: &std::path::Path, rel: &str, body: &str) {
let p = repo.join(rel);
if let Some(parent) = p.parent() {
std::fs::create_dir_all(parent).unwrap();
}
std::fs::write(p, body).unwrap();
}
#[tokio::test]
#[ignore = "live: api-dev + LiteLLM"]
async fn c5b_varied_findings() {
let llm = Arc::new(LlmClient::new(
env("LITELLM_URL"),
SecretString::from(env("LITELLM_API_KEY")),
env("LITELLM_MODEL"),
env("LITELLM_EMBED_MODEL"),
));
let mut config = common::dev_config("mongodb://unused".into(), "c5b".into());
config.breakpilot = BreakpilotConfig {
base_url: Some(env("BREAKPILOT_BASE_URL")),
token: None,
snapshot_dir: std::env::temp_dir()
.join("c5-oscal-snap")
.to_string_lossy()
.into_owned(),
semantic_mapping: true,
grounded_control_checks: false,
};
let repo = std::env::temp_dir().join("c5b-fixture-repo");
let _ = std::fs::remove_dir_all(&repo);
write(
&repo,
"app/db.py",
"import sqlite3\n\ndef get_user(username):\n q = \"SELECT * FROM users WHERE name = '\" + username + \"'\"\n return conn.execute(q)\n",
);
write(
&repo,
"app/config.py",
"# service config\nAPI_KEY = \"sk_live_51H8xYz3kQ9v2bNmR7wT4uSpQ\"\nDB_HOST = \"db.internal\"\n",
);
write(
&repo,
"app/net.py",
"import requests\n\ndef fetch(url):\n return requests.get(url, verify=False, timeout=5)\n",
);
write(
&repo,
"app/ser.py",
"import pickle\n\ndef load_state(blob):\n return pickle.loads(blob)\n",
);
let mut findings = vec![
mk(
"app/db.py",
4,
"SQL injection via string-concatenated query",
"User input is concatenated directly into a SQL statement, allowing SQL injection.",
),
mk(
"app/config.py",
2,
"Hardcoded API credential in source",
"A live API key is hardcoded in source code instead of a secret store.",
),
mk(
"app/net.py",
4,
"TLS certificate verification disabled",
"requests is called with verify=False, disabling TLS certificate validation.",
),
mk(
"app/ser.py",
3,
"Insecure deserialization with pickle.loads",
"Untrusted data is deserialized with pickle.loads, allowing remote code execution.",
),
];
let tagged =
compliance_agent::controls::semantic_stamp_findings(&config, llm, &repo, &mut findings)
.await;
println!("\n=== C5 example 2: varied findings ===");
for f in &findings {
println!(" {:52} -> {:?}", f.title, f.control_refs);
}
println!("tagged: {tagged}/4");
let _ = std::fs::remove_dir_all(&repo);
assert!(tagged >= 1);
}
-145
View File
@@ -1,145 +0,0 @@
//! C5 live verification — the semantic master-controls path end to end against the
//! deployed api-dev catalog. Ignored (hits api-dev + LiteLLM). Run explicitly:
//!
//! set -a; . ./.env; set +a
//! BREAKPILOT_BASE_URL=https://api-dev.breakpilot.ai \
//! cargo test -p compliance-agent --test c5_semantic_live -- --ignored --nocapture
//!
//! Pulls the live master-controls catalog, embeds the corpus (chunked), then for a
//! couple of real vulnerable findings retrieves the nearest master controls and
//! grounded-judges them, stamping master-control refs.
mod common;
use std::sync::Arc;
use compliance_agent::llm::LlmClient;
use compliance_core::config::BreakpilotConfig;
use compliance_core::models::finding::{Finding, Severity};
use compliance_core::models::scan::ScanType;
use secrecy::SecretString;
fn env(k: &str) -> String {
std::env::var(k).unwrap_or_else(|_| panic!("env {k} must be set for the live C5 test"))
}
fn mk_finding(file: &str, line: u32, title: &str) -> Finding {
let mut f = Finding::new(
"repo-c5".into(),
format!("{file}:{line}"),
"semgrep".into(),
ScanType::Sast,
title.into(),
title.into(),
Severity::High,
);
f.file_path = Some(file.into());
f.line_number = Some(line);
f
}
#[tokio::test]
#[ignore = "live: requires deployed api-dev master-controls (fetch+parse only, no LLM)"]
async fn c5_ingest_master_controls_catalog() {
use compliance_agent::controls::OscalControlsProvider;
let provider = OscalControlsProvider::new(
reqwest::Client::new(),
env("BREAKPILOT_BASE_URL"),
None,
std::env::temp_dir().join("c5-ingest-snap"),
);
let doc = provider
.load_master_controls()
.await
.expect("pull + parse master-controls catalog");
let controls = doc.to_controls();
println!(
"\n=== C5 ingest: {} master controls parsed ===",
controls.len()
);
for c in controls.iter().take(4) {
let text: String = c.text.chars().take(90).collect();
println!(" {} | {} | {}", c.id, c.title, text);
}
assert!(
!controls.is_empty(),
"expected a non-empty master-control corpus"
);
}
#[tokio::test]
#[ignore = "live: requires deployed api-dev master-controls + LiteLLM"]
async fn c5_semantic_stamps_master_control_refs() {
let llm = Arc::new(LlmClient::new(
env("LITELLM_URL"),
SecretString::from(env("LITELLM_API_KEY")),
env("LITELLM_MODEL"),
env("LITELLM_EMBED_MODEL"),
));
let mut config = common::dev_config("mongodb://unused".into(), "c5".into());
let snapshot = std::env::temp_dir().join("c5-oscal-snap");
config.breakpilot = BreakpilotConfig {
base_url: Some(env("BREAKPILOT_BASE_URL")),
token: None,
snapshot_dir: snapshot.to_string_lossy().into_owned(),
semantic_mapping: true,
grounded_control_checks: false,
};
// Fixture repo with recognizable code-checkable surfaces.
let repo = std::env::temp_dir().join("c5-fixture-repo");
let _ = std::fs::remove_dir_all(&repo);
std::fs::create_dir_all(repo.join("app")).expect("mkdir");
std::fs::write(
repo.join("app/auth.py"),
concat!(
"import hashlib\n",
"\n",
"def store_password(user, password):\n",
" # weak, unsalted password hashing\n",
" digest = hashlib.md5(password.encode()).hexdigest()\n",
" db.save(user, digest)\n",
"\n",
"@app.route('/login', methods=['POST'])\n",
"def login():\n",
" u = request.form['username']\n",
" p = request.form['password']\n",
" return 'ok' if check(u, p) else ('bad', 401)\n",
),
)
.expect("write fixture");
let mut findings = vec![
mk_finding("app/auth.py", 5, "Weak password hash (md5, unsalted)"),
mk_finding(
"app/auth.py",
9,
"Login endpoint without brute-force protection",
),
];
let tagged =
compliance_agent::controls::semantic_stamp_findings(&config, llm, &repo, &mut findings)
.await;
println!("\n=== C5 semantic master-controls stamping ===");
for f in &findings {
println!(
" {:50} {}:{:?} -> {:?}",
f.title,
f.file_path.as_deref().unwrap_or(""),
f.line_number,
f.control_refs
);
}
println!("findings that gained >=1 master-control ref: {tagged}");
let _ = std::fs::remove_dir_all(&repo);
// Live corpus — assert only that the path runs and stamps at least one ref.
assert!(
tagged >= 1,
"expected at least one finding to gain a master-control ref"
);
}
+39 -54
View File
@@ -2,10 +2,6 @@
//
// Spins up the agent API server on a random port with an isolated test
// database. Each test gets a fresh database that is dropped on cleanup.
//
// Included via `mod common;` in several test binaries; not every binary uses
// every helper, so allow dead code here.
#![allow(dead_code)]
use std::sync::Arc;
@@ -15,55 +11,6 @@ use compliance_agent::database::DatabasePool;
use compliance_core::AgentConfig;
use secrecy::SecretString;
/// The runner bearer token wired into the test config.
pub const TEST_RUNNER_TOKEN: &str = "test-runner-token";
/// A minimal dev [`AgentConfig`] for tests: unauthenticated (no Keycloak), the
/// Werkbank runner API enabled with [`TEST_RUNNER_TOKEN`].
pub fn dev_config(mongodb_uri: String, db_name: String) -> AgentConfig {
AgentConfig {
mongodb_uri,
mongodb_database: db_name,
litellm_url: std::env::var("TEST_LITELLM_URL")
.unwrap_or_else(|_| "http://localhost:4000".into()),
litellm_api_key: SecretString::from(String::new()),
litellm_model: "gpt-4o".into(),
litellm_embed_model: "text-embedding-3-small".into(),
agent_port: 0, // not used — we bind ourselves
scan_schedule: String::new(),
cve_monitor_schedule: String::new(),
git_clone_base_path: "/tmp/compliance-scanner-tests/repos".into(),
artifact_store_base_path: "/tmp/compliance-scanner-tests/artifacts".into(),
ssh_key_path: "/tmp/compliance-scanner-tests/ssh/id_ed25519".into(),
github_token: None,
github_webhook_secret: None,
gitlab_url: None,
gitlab_token: None,
gitlab_webhook_secret: None,
jira_url: None,
jira_email: None,
jira_api_token: None,
jira_project_key: None,
searxng_url: None,
nvd_api_key: None,
keycloak_url: None,
keycloak_realm: None,
keycloak_admin_username: None,
keycloak_admin_password: None,
pentest_verification_email: None,
pentest_imap_host: None,
pentest_imap_port: None,
pentest_imap_tls: false,
pentest_imap_username: None,
pentest_imap_password: None,
admin_api_token: None,
tenant_registry_url: None,
plc_runtime: compliance_core::PlcRuntimeConfig::default(),
werkbank_runner_token: Some(SecretString::from(TEST_RUNNER_TOKEN.to_string())),
breakpilot: compliance_core::config::BreakpilotConfig::default(),
}
}
/// A running test server with a unique database.
pub struct TestServer {
pub base_url: String,
@@ -86,7 +33,45 @@ impl TestServer {
.await
.expect("Failed to build DatabasePool");
let config = dev_config(mongodb_uri.clone(), db_name.clone());
let config = AgentConfig {
mongodb_uri: mongodb_uri.clone(),
mongodb_database: db_name.clone(),
litellm_url: std::env::var("TEST_LITELLM_URL")
.unwrap_or_else(|_| "http://localhost:4000".into()),
litellm_api_key: SecretString::from(String::new()),
litellm_model: "gpt-4o".into(),
litellm_embed_model: "text-embedding-3-small".into(),
agent_port: 0, // not used — we bind ourselves
scan_schedule: String::new(),
cve_monitor_schedule: String::new(),
git_clone_base_path: "/tmp/compliance-scanner-tests/repos".into(),
artifact_store_base_path: "/tmp/compliance-scanner-tests/artifacts".into(),
ssh_key_path: "/tmp/compliance-scanner-tests/ssh/id_ed25519".into(),
github_token: None,
github_webhook_secret: None,
gitlab_url: None,
gitlab_token: None,
gitlab_webhook_secret: None,
jira_url: None,
jira_email: None,
jira_api_token: None,
jira_project_key: None,
searxng_url: None,
nvd_api_key: None,
keycloak_url: None,
keycloak_realm: None,
keycloak_admin_username: None,
keycloak_admin_password: None,
pentest_verification_email: None,
pentest_imap_host: None,
pentest_imap_port: None,
pentest_imap_tls: false,
pentest_imap_username: None,
pentest_imap_password: None,
admin_api_token: None,
tenant_registry_url: None,
unified_pipeline: false,
};
let agent = ComplianceAgent::new(config, db_pool);
@@ -1,92 +0,0 @@
//! Live validation of the grounded surface path (Stage 5d) for absence-based CRA
//! controls. Ignored (hits api-dev CRA catalog + LiteLLM). Run:
//! export ... (LITELLM_* + BREAKPILOT_BASE_URL)
//! cargo test -p compliance-agent --test grounded_surface_live -- --ignored --nocapture
//!
//! Builds a fixture whose code surfaces trigger several absence-based controls
//! (no rate limiting, no security logging, unverified update) and checks that the
//! grounded checker produces control-tagged findings.
mod common;
use std::sync::Arc;
use compliance_agent::llm::LlmClient;
use compliance_core::config::BreakpilotConfig;
use secrecy::SecretString;
fn env(k: &str) -> String {
std::env::var(k).unwrap_or_else(|_| panic!("env {k} must be set"))
}
fn write(repo: &std::path::Path, rel: &str, body: &str) {
let p = repo.join(rel);
if let Some(parent) = p.parent() {
std::fs::create_dir_all(parent).unwrap();
}
std::fs::write(p, body).unwrap();
}
#[tokio::test]
#[ignore = "live: api-dev CRA catalog + LiteLLM"]
async fn grounded_surface_flags_absence_controls() {
let llm = Arc::new(LlmClient::new(
env("LITELLM_URL"),
SecretString::from(env("LITELLM_API_KEY")),
env("LITELLM_MODEL"),
env("LITELLM_EMBED_MODEL"),
));
let mut config = common::dev_config("mongodb://unused".into(), "grounded".into());
config.breakpilot = BreakpilotConfig {
base_url: Some(env("BREAKPILOT_BASE_URL")),
token: None,
snapshot_dir: std::env::temp_dir()
.join("grounded-snap")
.to_string_lossy()
.into_owned(),
semantic_mapping: false,
grounded_control_checks: true,
};
let repo = std::env::temp_dir().join("grounded-fixture-repo");
let _ = std::fs::remove_dir_all(&repo);
// cra-ai-11: login endpoint with no rate limiting / lockout
write(
&repo,
"app/auth.py",
"@app.route('/login', methods=['POST'])\ndef login():\n u = request.form['username']\n p = request.form['password']\n if authenticate(u, p):\n return redirect('/')\n return 'bad credentials', 401\n",
);
// cra-ai-24: privileged admin action with no security/audit logging
write(
&repo,
"app/admin.py",
"@app.route('/admin/delete_user', methods=['POST'])\ndef admin_delete_user():\n uid = request.form['uid']\n db.users.delete_one({'_id': uid})\n return 'ok', 200\n",
);
// cra-ai-28/29/30: firmware update applied without signature / checksum verification
write(
&repo,
"app/updater.py",
"def apply_firmware_update(url):\n blob = download(url)\n install_firmware(blob)\n reboot_device()\n",
);
let findings =
compliance_agent::controls::grounded_surface_findings(&config, llm, &repo, "repo-grounded")
.await;
println!("\n=== Grounded surface findings ({}) ===", findings.len());
for f in &findings {
println!(
" {:24} {}:{:?} {}",
f.control_refs.join(","),
f.file_path.as_deref().unwrap_or(""),
f.line_number,
f.title
);
}
let _ = std::fs::remove_dir_all(&repo);
assert!(
!findings.is_empty(),
"expected the grounded pass to flag at least one absence-based control"
);
}
@@ -113,16 +113,15 @@ async fn delete_repo_cascades_to_dast_and_pentest_data() {
// Create a repo
let resp = server
.post(
"/api/v1/targets",
"/api/v1/repositories",
&json!({
"name": "cascade-test",
"target_type": "web_app",
"artifacts": [{ "kind": "git_repo", "source_ref": "https://github.com/example/cascade-test.git", "branch": "main" }],
"git_url": "https://github.com/example/cascade-test.git",
}),
)
.await;
let body: serde_json::Value = resp.json().await.unwrap();
let repo_id = body["data"]["_id"]["$oid"].as_str().unwrap().to_string();
let repo_id = body["data"]["id"].as_str().unwrap().to_string();
// Insert DAST target linked to repo
let target_id = insert_dast_target(&server, &repo_id, "cascade-target").await;
@@ -141,7 +140,9 @@ async fn delete_repo_cascades_to_dast_and_pentest_data() {
assert_eq!(count_docs(&server, "dast_findings").await, 1);
// Delete the repo
let resp = server.delete(&format!("/api/v1/targets/{repo_id}")).await;
let resp = server
.delete(&format!("/api/v1/repositories/{repo_id}"))
.await;
assert_eq!(resp.status(), 200);
// All downstream data should be gone
@@ -160,16 +161,15 @@ async fn delete_repo_cascades_sast_findings_and_sbom() {
// Create a repo
let resp = server
.post(
"/api/v1/targets",
"/api/v1/repositories",
&json!({
"name": "sast-cascade",
"target_type": "web_app",
"artifacts": [{ "kind": "git_repo", "source_ref": "https://github.com/example/sast-cascade.git", "branch": "main" }],
"git_url": "https://github.com/example/sast-cascade.git",
}),
)
.await;
let body: serde_json::Value = resp.json().await.unwrap();
let repo_id = body["data"]["_id"]["$oid"].as_str().unwrap().to_string();
let repo_id = body["data"]["id"].as_str().unwrap().to_string();
// Insert SAST finding and SBOM entry
let mongodb_uri = std::env::var("TEST_MONGODB_URI")
@@ -209,7 +209,9 @@ async fn delete_repo_cascades_sast_findings_and_sbom() {
assert_eq!(count_docs(&server, "sbom_entries").await, 1);
// Delete repo
server.delete(&format!("/api/v1/targets/{repo_id}")).await;
server
.delete(&format!("/api/v1/repositories/{repo_id}"))
.await;
// Both should be gone
assert_eq!(count_docs(&server, "findings").await, 0);
@@ -3,4 +3,5 @@ mod dast;
mod findings;
mod health;
mod onboarding;
mod repositories;
mod stats;
@@ -0,0 +1,110 @@
use crate::common::TestServer;
use serde_json::json;
#[tokio::test]
async fn add_and_list_repository() {
let server = TestServer::start().await;
// Initially empty
let resp = server.get("/api/v1/repositories").await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
assert_eq!(body["data"].as_array().unwrap().len(), 0);
// Add a repository
let resp = server
.post(
"/api/v1/repositories",
&json!({
"name": "test-repo",
"git_url": "https://github.com/example/test-repo.git",
}),
)
.await;
assert_eq!(resp.status(), 200);
let body: serde_json::Value = resp.json().await.unwrap();
let repo_id = body["data"]["id"].as_str().unwrap().to_string();
assert!(!repo_id.is_empty());
// List should now return 1
let resp = server.get("/api/v1/repositories").await;
let body: serde_json::Value = resp.json().await.unwrap();
let repos = body["data"].as_array().unwrap();
assert_eq!(repos.len(), 1);
assert_eq!(repos[0]["name"], "test-repo");
server.cleanup().await;
}
#[tokio::test]
async fn add_duplicate_repository_fails() {
let server = TestServer::start().await;
let payload = json!({
"name": "dup-repo",
"git_url": "https://github.com/example/dup-repo.git",
});
// First add succeeds
let resp = server.post("/api/v1/repositories", &payload).await;
assert_eq!(resp.status(), 200);
// Second add with same git_url should fail (unique index)
let resp = server.post("/api/v1/repositories", &payload).await;
assert_ne!(resp.status(), 200);
server.cleanup().await;
}
#[tokio::test]
async fn delete_repository() {
let server = TestServer::start().await;
// Add a repo
let resp = server
.post(
"/api/v1/repositories",
&json!({
"name": "to-delete",
"git_url": "https://github.com/example/to-delete.git",
}),
)
.await;
let body: serde_json::Value = resp.json().await.unwrap();
let repo_id = body["data"]["id"].as_str().unwrap();
// Delete it
let resp = server
.delete(&format!("/api/v1/repositories/{repo_id}"))
.await;
assert_eq!(resp.status(), 200);
// List should be empty again
let resp = server.get("/api/v1/repositories").await;
let body: serde_json::Value = resp.json().await.unwrap();
assert_eq!(body["data"].as_array().unwrap().len(), 0);
server.cleanup().await;
}
#[tokio::test]
async fn delete_nonexistent_repository_returns_404() {
let server = TestServer::start().await;
let resp = server
.delete("/api/v1/repositories/000000000000000000000000")
.await;
assert_eq!(resp.status(), 404);
server.cleanup().await;
}
#[tokio::test]
async fn delete_invalid_id_returns_400() {
let server = TestServer::start().await;
let resp = server.delete("/api/v1/repositories/not-a-valid-id").await;
assert_eq!(resp.status(), 400);
server.cleanup().await;
}
@@ -5,14 +5,13 @@ use serde_json::json;
async fn stats_overview_reflects_inserted_data() {
let server = TestServer::start().await;
// Add a target
// Add a repo
server
.post(
"/api/v1/targets",
"/api/v1/repositories",
&json!({
"name": "stats-repo",
"target_type": "web_app",
"artifacts": [{ "kind": "git_repo", "source_ref": "https://github.com/example/stats-repo.git", "branch": "main" }],
"git_url": "https://github.com/example/stats-repo.git",
}),
)
.await;
@@ -0,0 +1,156 @@
// Integration tests for the onboarding backfill migration.
//
// Requires MongoDB (set TEST_MONGODB_URI if not at the default).
// Not run in CI (which is `--lib` only) — run locally:
// cargo test -p compliance-agent --test e2e migration
use compliance_agent::database::{Database, DatabasePool};
use compliance_agent::migrate::onboarding;
use compliance_core::models::{
ArtifactKind, DastTarget, DastTargetType, TargetType, TrackedRepository,
};
use mongodb::bson::{doc, Document};
async fn fresh_db() -> (DatabasePool, String, Database) {
let uri = std::env::var("TEST_MONGODB_URI")
.unwrap_or_else(|_| "mongodb://root:example@localhost:27017/?authSource=admin".into());
// Prefix must fit the pool's 30-char cap (`<prefix>_<32 hex>` <= 63).
let prefix = format!("t_{}", &uuid::Uuid::new_v4().simple().to_string()[..16]);
let pool = DatabasePool::connect(&uri, &prefix)
.await
.expect("connect mongo");
let db = pool.for_tenant_id("t1").await.expect("tenant db");
(pool, prefix, db)
}
async fn cleanup(pool: &DatabasePool, prefix: &str) {
if let Ok(names) = pool.client().list_database_names().await {
for n in names {
if n.starts_with(prefix) {
pool.client().database(&n).drop().await.ok();
}
}
}
}
#[tokio::test]
async fn backfill_folds_relinks_is_idempotent_and_reversible() {
let (pool, prefix, db) = fresh_db().await;
// Seed a repo.
let repo = TrackedRepository::new("acme".into(), "https://git/acme.git".into());
let repo_id = db
.repositories()
.insert_one(repo)
.await
.expect("insert repo")
.inserted_id
.as_object_id()
.expect("repo oid");
// A DAST target linked to the repo (folds + promotes to WebApp + relinks).
let mut linked = DastTarget::new(
"acme-web".into(),
"https://acme.example.com".into(),
DastTargetType::WebApp,
);
linked.repo_id = Some(repo_id.to_hex());
let linked_id = db
.dast_targets()
.insert_one(linked)
.await
.expect("insert linked dast")
.inserted_id
.as_object_id()
.expect("linked oid");
// A repo-less DAST target (standalone).
let standalone = DastTarget::new(
"acme-api".into(),
"https://api.acme.com".into(),
DastTargetType::RestApi,
);
let standalone_id = db
.dast_targets()
.insert_one(standalone)
.await
.expect("insert standalone dast")
.inserted_id
.as_object_id()
.expect("standalone oid");
// A DAST scan run pointing at the linked target — should be relinked to the repo.
db.collection_named::<Document>("dast_scan_runs")
.insert_one(doc! { "target_id": linked_id.to_hex(), "status": "completed" })
.await
.expect("insert dast run");
// --- Backfill ---
assert!(!onboarding::already_applied(&db).await.unwrap());
let report = onboarding::backfill_onboarded_targets(&db, false)
.await
.expect("backfill");
assert_eq!(report.repos_migrated, 1);
assert_eq!(report.dast_targets_folded, 1);
assert_eq!(report.dast_targets_standalone, 1);
assert!(onboarding::already_applied(&db).await.unwrap());
// Repo target: preserved _id, has git + folded live-url, promoted to WebApp.
let repo_target = db
.onboarded_targets()
.find_one(doc! { "_id": repo_id })
.await
.unwrap()
.expect("repo target");
assert!(repo_target.has(ArtifactKind::GitRepo));
assert!(repo_target.has(ArtifactKind::LiveUrl));
assert_eq!(repo_target.target_type, TargetType::WebApp);
// Standalone target: preserved _id, live-url, backend service.
let standalone_target = db
.onboarded_targets()
.find_one(doc! { "_id": standalone_id })
.await
.unwrap()
.expect("standalone target");
assert!(standalone_target.has(ArtifactKind::LiveUrl));
assert_eq!(standalone_target.target_type, TargetType::BackendService);
// The DAST run was relinked from the old dast id to the repo (unified) id.
let run = db
.collection_named::<Document>("dast_scan_runs")
.find_one(doc! {})
.await
.unwrap()
.expect("run");
assert_eq!(run.get_str("target_id").unwrap(), repo_id.to_hex());
// --- Idempotent: re-run migrates nothing new ---
let again = onboarding::backfill_onboarded_targets(&db, false)
.await
.expect("backfill again");
assert_eq!(again.repos_migrated, 0);
assert_eq!(again.dast_targets_folded, 0);
assert_eq!(again.dast_targets_standalone, 0);
assert!(again.skipped_existing >= 2);
// --- Revert: onboarded targets gone, relink undone, marker cleared ---
onboarding::revert(&db).await.expect("revert");
assert_eq!(
db.onboarded_targets()
.count_documents(doc! {})
.await
.unwrap(),
0
);
let run_after = db
.collection_named::<Document>("dast_scan_runs")
.find_one(doc! {})
.await
.unwrap()
.expect("run");
assert_eq!(run_after.get_str("target_id").unwrap(), linked_id.to_hex());
assert!(!onboarding::already_applied(&db).await.unwrap());
cleanup(&pool, &prefix).await;
}
@@ -7,3 +7,4 @@
// Or nightly: (via CI with MongoDB service container)
mod api;
mod migration;
+28 -13
View File
@@ -11,7 +11,7 @@
#![allow(clippy::expect_used, clippy::unwrap_used)]
use compliance_agent::database::DatabasePool;
use compliance_core::models::{Artifact, OnboardedTarget, TargetType};
use compliance_core::models::TrackedRepository;
use compliance_core::{OrgRole, TenantContext, TenantStatus};
use mongodb::bson::doc;
@@ -28,12 +28,27 @@ fn ctx(tenant_id: &str, slug: &str) -> TenantContext {
}
}
fn fixture_repo(name: &str, git_url: &str) -> OnboardedTarget {
let mut target = OnboardedTarget::new(name.to_string(), TargetType::WebApp);
target
.artifacts
.push(Artifact::git_repo(git_url.to_string(), "main".to_string()));
target
fn fixture_repo(name: &str, git_url: &str) -> TrackedRepository {
TrackedRepository {
id: None,
name: name.to_string(),
git_url: git_url.to_string(),
default_branch: "main".to_string(),
local_path: None,
scan_schedule: None,
webhook_enabled: false,
webhook_secret: None,
tracker_type: None,
tracker_owner: None,
tracker_repo: None,
tracker_token: None,
auth_token: None,
auth_username: None,
last_scanned_commit: None,
findings_count: 0,
created_at: chrono::Utc::now(),
updated_at: chrono::Utc::now(),
}
}
#[tokio::test]
@@ -56,12 +71,12 @@ async fn pool_isolates_tenants_at_driver_level() {
// Write distinct repos into each tenant's database.
acme_db
.onboarded_targets()
.repositories()
.insert_one(fixture_repo("acme-app", "git@example.com:acme/app.git"))
.await
.expect("insert acme");
globex_db
.onboarded_targets()
.repositories()
.insert_one(fixture_repo(
"globex-platform",
"git@example.com:globex/platform.git",
@@ -158,12 +173,12 @@ async fn admin_helpers_list_and_drop_tenant_dbs() {
let acme_db = pool.for_tenant(&acme).await.expect("acme db");
let globex_db = pool.for_tenant(&globex).await.expect("globex db");
acme_db
.onboarded_targets()
.repositories()
.insert_one(fixture_repo("acme-app", "git@example.com:acme/app.git"))
.await
.expect("insert acme");
globex_db
.onboarded_targets()
.repositories()
.insert_one(fixture_repo("globex-app", "git@example.com:globex/app.git"))
.await
.expect("insert globex");
@@ -269,9 +284,9 @@ fn short_id() -> String {
}
/// Drain a `repositories` find cursor on the given tenant database.
async fn collect(db: &compliance_agent::database::Database) -> Vec<OnboardedTarget> {
async fn collect(db: &compliance_agent::database::Database) -> Vec<TrackedRepository> {
let mut cursor = db
.onboarded_targets()
.repositories()
.find(doc! {})
.await
.expect("find repositories");
-291
View File
@@ -1,291 +0,0 @@
//! Integration tests for the Werkbank runner endpoints (WB-05).
//!
//! Drives the real HTTP handlers (lease/heartbeat/complete) against a live Mongo:
//! a runner leases a seeded job, completes it, and the result's findings are
//! persisted against the job's target. Also checks the bearer-token gate. Skips
//! cleanly when no Mongo is reachable.
#![allow(clippy::expect_used, clippy::unwrap_used)]
mod common;
use std::sync::Arc;
use axum::routing::{get, post};
use axum::{middleware, Extension, Router};
use compliance_agent::agent::ComplianceAgent;
use compliance_agent::api::handlers::werkbank_jobs;
use compliance_agent::database::DatabasePool;
use compliance_agent::werkbank::JobQueue;
use compliance_core::models::werkbank::{InputRef, Job, JobResult, JobStatus, LeasedJob};
use compliance_core::models::{
Artifact, Finding, OnboardedTarget, PlcFormat, ScanType, Severity, TargetType,
};
use common::{dev_config, TEST_RUNNER_TOKEN};
const TENANT: &str = "dev";
/// A running werkbank API on a random port, or `None` if no Mongo.
struct Harness {
base_url: String,
client: reqwest::Client,
pool: DatabasePool,
db_name: String,
}
async fn start() -> Option<Harness> {
let uri = std::env::var("TEST_MONGODB_URI")
.unwrap_or_else(|_| "mongodb://root:example@localhost:27017/?authSource=admin".into());
let db_name = format!("wba_{}", &uuid::Uuid::new_v4().simple().to_string()[..12]);
let pool = match DatabasePool::connect(&uri, &db_name).await {
Ok(p) => p,
Err(_) => {
eprintln!("SKIP werkbank_api: no MongoDB reachable at {uri}");
return None;
}
};
// Touch the tenant DB so indexes are ensured before the queue is used.
pool.for_tenant_id(TENANT).await.expect("tenant db");
let agent = ComplianceAgent::new(dev_config(uri, db_name.clone()), pool.clone());
let app = Router::new()
.route("/api/v1/werkbank/jobs/lease", post(werkbank_jobs::lease))
.route(
"/api/v1/werkbank/jobs/heartbeat",
post(werkbank_jobs::heartbeat),
)
.route(
"/api/v1/werkbank/jobs/complete",
post(werkbank_jobs::complete),
)
.route(
"/api/v1/werkbank/jobs/enqueue",
post(werkbank_jobs::enqueue),
)
.route(
"/api/v1/werkbank/artifacts/{hash}",
get(werkbank_jobs::serve_artifact),
)
.layer(middleware::from_fn(werkbank_jobs::require_runner_token))
.layer(Extension(Arc::new(agent)));
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
let port = listener.local_addr().unwrap().port();
tokio::spawn(async move {
axum::serve(listener, app).await.ok();
});
Some(Harness {
base_url: format!("http://127.0.0.1:{port}"),
client: reqwest::Client::new(),
pool,
db_name,
})
}
impl Harness {
fn post(
&self,
path: &str,
token: Option<&str>,
body: serde_json::Value,
) -> reqwest::RequestBuilder {
let mut r = self
.client
.post(format!("{}{path}", self.base_url))
.json(&body);
if let Some(t) = token {
r = r.bearer_auth(t);
}
r
}
async fn cleanup(&self) {
let _ = self
.pool
.client()
.database(&format!("{}_{TENANT}", self.db_name))
.drop()
.await;
}
}
fn finding_for(target: &str, fp: &str) -> Finding {
let mut f = Finding::new(
target.to_string(),
fp.to_string(),
"ics-probe".to_string(),
ScanType::IcsProbe,
"Modbus exposed".to_string(),
"unauthenticated".to_string(),
Severity::Critical,
);
f.rule_id = Some("ics-modbus-exposed".to_string());
f
}
#[tokio::test]
async fn lease_complete_persists_findings_against_the_target() {
let Some(h) = start().await else { return };
let db = h.pool.for_tenant_id(TENANT).await.unwrap();
let queue = JobQueue::new(&db);
// Seed a queued job.
let job = Job::plc_provision("job-1", TENANT, "target-1", InputRef::blob("sha256:x"), 180);
assert!(queue.enqueue(job, chrono::Utc::now()).await.unwrap());
// Lease it over HTTP.
let resp = h
.post(
"/api/v1/werkbank/jobs/lease",
Some(TEST_RUNNER_TOKEN),
serde_json::json!({
"tenant": TENANT, "runner_id": "r1", "executor": "docker",
"labels": [], "lease_ttl_secs": 60
}),
)
.send()
.await
.unwrap();
assert_eq!(resp.status(), 200, "lease should return a job");
let leased: LeasedJob = resp.json().await.unwrap();
assert_eq!(leased.job.id, "job-1");
// Complete it with a finding.
let mut result = JobResult::succeeded("job-1");
result.findings = vec![finding_for("target-1", "fp-abc")];
let resp = h
.post(
"/api/v1/werkbank/jobs/complete",
Some(TEST_RUNNER_TOKEN),
serde_json::json!({
"tenant": TENANT, "job_id": "job-1",
"lease_token": leased.lease_token, "result": result
}),
)
.send()
.await
.unwrap();
assert_eq!(resp.status(), 200);
assert!(resp.json::<serde_json::Value>().await.unwrap()["recorded"]
.as_bool()
.unwrap());
// The job is now succeeded, and the finding was persisted to the target.
assert_eq!(
queue.get("job-1").await.unwrap().unwrap().status,
JobStatus::Succeeded
);
let stored = db
.findings()
.find_one(mongodb::bson::doc! { "fingerprint": "fp-abc" })
.await
.unwrap();
assert!(stored.is_some(), "finding should be persisted");
h.cleanup().await;
}
#[tokio::test]
async fn enqueue_extracts_program_stores_a_blob_and_serves_it() {
let Some(h) = start().await else { return };
let db = h.pool.for_tenant_id(TENANT).await.unwrap();
// A PlcSps target with a single complete ST program uploaded.
let dir = std::env::temp_dir().join(format!("wbq-prog-{}", uuid::Uuid::new_v4()));
std::fs::create_dir_all(&dir).unwrap();
let st = dir.join("main.st");
std::fs::write(
&st,
"PROGRAM Main\nEND_PROGRAM\nCONFIGURATION C\n RESOURCE R\nEND_CONFIGURATION\n",
)
.unwrap();
let mut target = OnboardedTarget::new("plc".into(), TargetType::PlcSps);
let mut art = Artifact::plc_project("main.st", PlcFormat::StructuredText);
art.stored_path = Some(st.to_string_lossy().to_string());
target.artifacts.push(art);
let ins = db.onboarded_targets().insert_one(&target).await.unwrap();
let target_id = ins.inserted_id.as_object_id().unwrap().to_hex();
// Enqueue → a plc-provision job whose program is a content-addressed blob.
let resp = h
.post(
"/api/v1/werkbank/jobs/enqueue",
Some(TEST_RUNNER_TOKEN),
serde_json::json!({ "tenant": TENANT, "target_id": target_id }),
)
.send()
.await
.unwrap();
assert_eq!(resp.status(), 200, "enqueue should succeed");
let body: serde_json::Value = resp.json().await.unwrap();
let job_id = body["job_id"].as_str().unwrap().to_string();
let rec = JobQueue::new(&db).get(&job_id).await.unwrap().unwrap();
let hash = rec
.job
.inputs
.get("program")
.and_then(|i| i.blob.clone())
.expect("program blob");
// Serve the blob back and confirm it's the program source (what the runner
// would fetch).
let served = h
.client
.get(format!("{}/api/v1/werkbank/artifacts/{hash}", h.base_url))
.bearer_auth(TEST_RUNNER_TOKEN)
.send()
.await
.unwrap();
assert_eq!(served.status(), 200);
assert!(served.text().await.unwrap().contains("CONFIGURATION"));
h.cleanup().await;
let _ = std::fs::remove_dir_all(&dir);
}
#[tokio::test]
async fn empty_queue_leases_nothing() {
let Some(h) = start().await else { return };
let resp = h
.post(
"/api/v1/werkbank/jobs/lease",
Some(TEST_RUNNER_TOKEN),
serde_json::json!({
"tenant": TENANT, "runner_id": "r1", "executor": "docker",
"labels": [], "lease_ttl_secs": 60
}),
)
.send()
.await
.unwrap();
assert_eq!(resp.status(), 204, "no job → 204");
h.cleanup().await;
}
#[tokio::test]
async fn runner_endpoints_require_the_bearer_token() {
let Some(h) = start().await else { return };
let body = serde_json::json!({
"tenant": TENANT, "runner_id": "r1", "executor": "docker",
"labels": [], "lease_ttl_secs": 60
});
let no_token = h
.post("/api/v1/werkbank/jobs/lease", None, body.clone())
.send()
.await
.unwrap();
assert_eq!(no_token.status(), 401, "missing token → 401");
let bad_token = h
.post("/api/v1/werkbank/jobs/lease", Some("wrong"), body)
.send()
.await
.unwrap();
assert_eq!(bad_token.status(), 401, "wrong token → 401");
h.cleanup().await;
}
-258
View File
@@ -1,258 +0,0 @@
//! Integration tests for the Werkbank job queue (WB-02).
//!
//! Exercises the atomic lease/heartbeat/complete/sweep flow against a real
//! MongoDB — the guarantees (idempotent enqueue, single-owner lease, visibility
//! timeout) are Mongo-semantics-dependent and can't be unit-tested in isolation.
//! Skips cleanly when no Mongo is reachable (set `TEST_MONGODB_URI` to point at
//! one; defaults to the local dev cluster).
#![allow(clippy::expect_used, clippy::unwrap_used)]
use std::time::Duration;
use chrono::{DateTime, TimeZone, Utc};
use compliance_agent::database::Database;
use compliance_agent::werkbank::JobQueue;
use compliance_core::models::werkbank::{Executor, InputRef, Job, JobResult};
/// Connect + ensure indexes on a throwaway database, or `None` if no Mongo.
async fn setup() -> Option<(JobQueue, mongodb::Database)> {
let uri = std::env::var("TEST_MONGODB_URI")
.unwrap_or_else(|_| "mongodb://root:example@localhost:27017/?authSource=admin".into());
let db_name = format!("wbq_{}", &uuid::Uuid::new_v4().simple().to_string()[..12]);
let db = match Database::connect(&uri, &db_name).await {
Ok(d) => d,
Err(_) => {
eprintln!("SKIP werkbank_queue: no MongoDB reachable at {uri}");
return None;
}
};
db.ensure_indexes().await.expect("ensure indexes");
let queue = JobQueue::new(&db);
Some((queue, db.inner().clone()))
}
fn base_time() -> DateTime<Utc> {
Utc.timestamp_opt(1_700_000_000, 0).unwrap()
}
fn job(id: &str) -> Job {
Job::plc_provision(id, "acme", "target-1", InputRef::blob("sha256:abc"), 180)
}
fn job_with_labels(id: &str, labels: &[&str]) -> Job {
let mut j = job(id);
j.labels = labels.iter().map(|s| s.to_string()).collect();
j
}
macro_rules! skip_if_no_mongo {
() => {
match setup().await {
Some(v) => v,
None => return,
}
};
}
#[tokio::test]
async fn enqueue_is_idempotent() {
let (q, db) = skip_if_no_mongo!();
let now = base_time();
assert!(q.enqueue(job("j1"), now).await.expect("enqueue"));
// Same id again — no duplicate row, reports "already present".
assert!(!q.enqueue(job("j1"), now).await.expect("enqueue2"));
let rec = q.get("j1").await.expect("get").expect("exists");
assert_eq!(
rec.status,
compliance_core::models::werkbank::JobStatus::Queued
);
assert_eq!(rec.attempts, 0);
db.drop().await.ok();
}
#[tokio::test]
async fn lease_matches_executor_and_labels_and_is_fifo() {
let (q, db) = skip_if_no_mongo!();
let t0 = base_time();
// Two docker jobs (j_old older than j_new) + one requiring a kvm label.
q.enqueue(job("j_old"), t0).await.unwrap();
q.enqueue(job("j_new"), t0 + chrono::Duration::seconds(5))
.await
.unwrap();
q.enqueue(job_with_labels("j_kvm", &["kvm=true"]), t0)
.await
.unwrap();
// Wrong executor: a shell runner leases nothing.
assert!(q
.lease("r-shell", Executor::Shell, &[], Duration::from_secs(30), t0)
.await
.unwrap()
.is_none());
// A docker runner without the kvm label gets the oldest label-free job (FIFO).
let leased = q
.lease("r1", Executor::Docker, &[], Duration::from_secs(30), t0)
.await
.unwrap()
.expect("leased");
assert_eq!(leased.job.id, "j_old", "oldest matching job first");
assert!(!leased.lease_token.is_empty());
// The kvm job stays unleased for that runner (missing label)...
let none = q
.lease("r1", Executor::Docker, &[], Duration::from_secs(30), t0)
.await
.unwrap()
.expect("next");
assert_eq!(none.job.id, "j_new", "label-free job, not the kvm one");
// ...but a runner advertising kvm can take it.
let kvm = q
.lease(
"r2",
Executor::Docker,
&["kvm=true".to_string(), "arch=amd64".to_string()],
Duration::from_secs(30),
t0,
)
.await
.unwrap()
.expect("kvm leased");
assert_eq!(kvm.job.id, "j_kvm");
// A leased job increments attempts and is no longer queued.
let rec = q.get("j_old").await.unwrap().unwrap();
assert_eq!(rec.attempts, 1);
assert_eq!(rec.leased_by.as_deref(), Some("r1"));
db.drop().await.ok();
}
#[tokio::test]
async fn heartbeat_extends_lease_and_surfaces_cancel() {
let (q, db) = skip_if_no_mongo!();
let now = base_time();
q.enqueue(job("j1"), now).await.unwrap();
let leased = q
.lease("r1", Executor::Docker, &[], Duration::from_secs(30), now)
.await
.unwrap()
.unwrap();
// A valid heartbeat moves it to running and reports not-cancelled.
let ack = q
.heartbeat("j1", &leased.lease_token, Duration::from_secs(30), now)
.await
.unwrap()
.expect("valid lease");
assert!(!ack.cancelled);
assert_eq!(
q.get("j1").await.unwrap().unwrap().status,
compliance_core::models::werkbank::JobStatus::Running
);
// A wrong token is a lost lease.
assert!(q
.heartbeat("j1", "wrong-token", Duration::from_secs(30), now)
.await
.unwrap()
.is_none());
// Cancelling an in-flight job flags it; the next heartbeat reports cancelled.
assert!(q.cancel("j1", now).await.unwrap());
let ack = q
.heartbeat("j1", &leased.lease_token, Duration::from_secs(30), now)
.await
.unwrap()
.expect("still leased");
assert!(ack.cancelled);
db.drop().await.ok();
}
#[tokio::test]
async fn complete_is_idempotent_and_token_guarded() {
let (q, db) = skip_if_no_mongo!();
let now = base_time();
q.enqueue(job("j1"), now).await.unwrap();
let leased = q
.lease("r1", Executor::Docker, &[], Duration::from_secs(30), now)
.await
.unwrap()
.unwrap();
// Wrong token cannot complete.
let mut result = JobResult::succeeded("j1");
result.findings = Vec::new();
assert!(!q.complete("j1", "nope", &result, now).await.unwrap());
// The lease holder completes it once...
assert!(q
.complete("j1", &leased.lease_token, &result, now)
.await
.unwrap());
let rec = q.get("j1").await.unwrap().unwrap();
assert_eq!(
rec.status,
compliance_core::models::werkbank::JobStatus::Succeeded
);
assert!(rec.result.is_some());
assert!(rec.lease_token.is_none(), "lease cleared on completion");
// ...and a second (duplicate) completion is a no-op.
assert!(!q
.complete("j1", &leased.lease_token, &result, now)
.await
.unwrap());
db.drop().await.ok();
}
#[tokio::test]
async fn sweep_requeues_expired_then_expires_after_max_attempts() {
let (q, db) = skip_if_no_mongo!();
let t0 = base_time();
q.enqueue(job("j1"), t0).await.unwrap();
// Lease #1 with a 10s TTL; then time jumps past expiry.
q.lease("r1", Executor::Docker, &[], Duration::from_secs(10), t0)
.await
.unwrap()
.unwrap();
let past = t0 + chrono::Duration::seconds(60);
// attempts=1 < max=2 → requeued.
let swept = q.sweep_expired(past, 2).await.unwrap();
assert_eq!(swept.requeued, 1);
assert_eq!(swept.expired, 0);
assert_eq!(
q.get("j1").await.unwrap().unwrap().status,
compliance_core::models::werkbank::JobStatus::Queued
);
// Lease #2 (attempts=2), let it expire again → now expired (>= max).
q.lease("r2", Executor::Docker, &[], Duration::from_secs(10), past)
.await
.unwrap()
.unwrap();
let later = past + chrono::Duration::seconds(60);
let swept = q.sweep_expired(later, 2).await.unwrap();
assert_eq!(swept.requeued, 0);
assert_eq!(swept.expired, 1);
assert_eq!(
q.get("j1").await.unwrap().unwrap().status,
compliance_core::models::werkbank::JobStatus::Expired
);
db.drop().await.ok();
}
-4
View File
@@ -50,7 +50,3 @@ axum = { version = "0.8", optional = true }
jsonwebtoken = { version = "9", optional = true }
reqwest = { workspace = true, optional = true }
tokio = { workspace = true, optional = true }
[dev-dependencies]
# Parse the declarative TOML job specs in the Werkbank contract tests.
toml = "0.8"
+5 -5
View File
@@ -64,11 +64,11 @@ struct Claims {
const PUBLIC_ENDPOINTS: &[&str] = &["/api/v1/health"];
/// Path prefixes that bypass JWT validation. The admin sub-router
/// (`/api/v1/admin/*`) and the Werkbank runner API (`/api/v1/werkbank/*`)
/// have their own static-bearer middleware and must not be routed through the
/// customer-JWT path — a Keycloak token always carries a single tenant_id and
/// would semantically conflict with these cross-tenant / machine operations.
const PUBLIC_PREFIXES: &[&str] = &["/api/v1/admin/", "/api/v1/werkbank/"];
/// (`/api/v1/admin/*`) has its own static-bearer middleware and must
/// not be routed through the customer-JWT path — a Keycloak token
/// always carries a single tenant_id and would semantically conflict
/// with cross-tenant admin operations.
const PUBLIC_PREFIXES: &[&str] = &["/api/v1/admin/"];
/// Middleware that validates Bearer JWT tokens against Keycloak's JWKS
/// and attaches a `TenantContext` extension on success.
+5 -96
View File
@@ -49,102 +49,11 @@ pub struct AgentConfig {
/// of tenants to iterate. When `None` or unreachable, scheduler
/// falls back to `SCHEDULER_TENANT_IDS` env (M7.2-C).
pub tenant_registry_url: Option<String>,
/// Ephemeral soft-PLC provisioning for dynamic PLC testing (#183). Off by
/// default: it needs Docker access in the agent's runtime, which is a
/// deployment opt-in.
pub plc_runtime: PlcRuntimeConfig,
/// Static bearer for the Werkbank runner endpoints
/// (`/api/v1/werkbank/jobs/*`). Machine auth for runners leasing/completing
/// jobs — NOT a Keycloak JWT, since a runner acts across tenants. When
/// `None`, those endpoints are not mounted at all.
pub werkbank_runner_token: Option<SecretString>,
/// Source for the OSCAL control catalog pulled from breakpilot-compliance
/// (drives the [`crate::traits::ControlsProvider`]). Disabled when
/// `base_url` is `None`.
pub breakpilot: BreakpilotConfig,
}
/// Where to pull the OSCAL control catalog from breakpilot-compliance, and where
/// to snapshot it for deterministic / offline reuse.
#[derive(Debug, Clone)]
pub struct BreakpilotConfig {
/// Backend base URL (e.g. `http://backend-compliance:8002`). `None` disables
/// the OSCAL controls provider.
pub base_url: Option<String>,
/// Optional bearer token for the catalog endpoint.
pub token: Option<SecretString>,
/// Directory for catalog snapshots.
pub snapshot_dir: String,
/// Enable the master-controls **semantic** mapping pass (embed regions,
/// Enable the master-controls **semantic** mapping pass (embed regions,
/// retrieve nearest controls, grounded-judge). On by default — validated live
/// against the deployed master-controls catalog. Still a no-op unless
/// `base_url` is set and the catalog is reachable.
pub semantic_mapping: bool,
/// Enable the **grounded surface** pass for absence-based controls (retrieve
/// the code surface a control governs, judge whether it holds). On by default
/// — validated live; it covers the 8 absence-based CRA controls that no
/// syntactic rule can.
pub grounded_control_checks: bool,
}
impl Default for BreakpilotConfig {
fn default() -> Self {
Self {
base_url: None,
token: None,
snapshot_dir: "/data/compliance-scanner/oscal".to_string(),
semantic_mapping: true,
grounded_control_checks: true,
}
}
}
/// Configuration for the ephemeral soft-PLC "provision-and-test" path (#183).
///
/// When a PLC/SPS target ships control logic but no reachable live device, the
/// agent can instantiate that logic itself: spin up a throwaway soft-PLC
/// (OpenPLC) container in-cluster, load the program, start the runtime, probe it
/// over industrial protocols, then tear it down. This struct carries the knobs
/// for that container's lifecycle and the OpenPLC web-UI credentials used to
/// upload the program.
#[derive(Clone, Debug)]
pub struct PlcRuntimeConfig {
/// Master switch. Provision-and-test does nothing unless this is set — it
/// shells out to `docker`, which requires the agent container to have Docker
/// access (socket mount), an explicit deployment decision.
pub enabled: bool,
/// Container image for the ephemeral soft-PLC (OpenPLC).
pub image: String,
/// Docker network the instance joins. Must be the agent's own network so it
/// is reachable in-cluster by container name and never published to the host.
pub network: String,
/// Memory cap passed to `docker run --memory` (e.g. `512m`).
pub memory: String,
/// CPU cap passed to `docker run --cpus` (e.g. `0.5`).
pub cpus: String,
/// Hard ceiling on a provisioned instance's lifetime. Teardown is guaranteed
/// no later than this even if a load/probe step hangs.
pub max_lifetime_secs: u64,
/// OpenPLC web-UI username for the program upload (image default `openplc`).
pub openplc_user: String,
/// OpenPLC web-UI password (image default `openplc`).
pub openplc_password: SecretString,
}
impl Default for PlcRuntimeConfig {
fn default() -> Self {
Self {
enabled: false,
image: "registry.meghsakha.com/openplc:latest".to_string(),
network: "certifai".to_string(),
memory: "512m".to_string(),
cpus: "0.5".to_string(),
max_lifetime_secs: 180,
openplc_user: "openplc".to_string(),
openplc_password: SecretString::from("openplc".to_string()),
}
}
/// When true, `run_scan` dispatches to the unified `run_target` pipeline
/// (reads `onboarded_targets`) instead of the legacy repository pipeline.
/// Env `UNIFIED_PIPELINE`. Defaults on; set `UNIFIED_PIPELINE=0` to use the
/// legacy repository pipeline.
pub unified_pipeline: bool,
}
#[derive(Clone, Debug, Serialize, Deserialize)]
-205
View File
@@ -1,205 +0,0 @@
//! Grounded control-driven checking.
//!
//! Turns a *text* control into findings via an LLM used as a **pattern-recognizer**
//! whose output is grounded to real code — so a hallucinated finding cannot
//! survive. Determinism is structural, not a prompt plea:
//!
//! 1. the LLM only ever judges *retrieved* regions — it can't invent findings in
//! code it never saw;
//! 2. a verdict becomes a finding only if its quoted snippet appears **verbatim**
//! in the region, and the line is recomputed from that match — the model's own
//! line number is never trusted ([`ground`]);
//! 3. verdicts are cached by content hash ([`cache_key`]) so re-scans reproduce.
//!
//! The LLM supplies cross-language / cross-stack pattern recognition; this module
//! supplies the determinism.
use serde::{Deserialize, Serialize};
use sha2::{Digest, Sha256};
use crate::models::finding::{Finding, Severity};
use crate::models::scan::ScanType;
/// A control rendered as a check the LLM judges code against.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ControlCheckSpec {
/// Stable control id, e.g. `"cra-ai-8"`.
pub control_id: String,
/// Short control title (used in the finding title).
pub title: String,
/// The requirement text the LLM judges against (control objective/statement).
pub requirement: String,
/// CWE to fall back to when the model doesn't supply one.
pub default_cwe: Option<String>,
/// Severity for findings raised from this control.
pub severity: Severity,
}
/// A retrieved code region the LLM judges — never the whole repo.
#[derive(Debug, Clone)]
pub struct CandidateRegion {
/// Repo-relative path.
pub file: String,
/// 1-based line number of the region's first line in `file`.
pub start_line: u32,
/// The region's source text.
pub content: String,
}
/// The LLM's structured verdict for one (control, region). `snippet` is the
/// verbatim code the model claims proves the violation — it is the anchor the
/// grounding gate checks.
#[derive(Debug, Clone)]
pub struct LlmVerdict {
pub violates: bool,
pub snippet: String,
pub cwe: Option<String>,
pub confidence: f64,
}
/// The grounding gate. A verdict becomes a [`Finding`] only if it claims a
/// violation AND its quoted `snippet` appears verbatim in `region.content`; the
/// finding's line is computed from the match, so a fabricated or mis-located
/// snippet is dropped. Pure — no LLM, no I/O.
pub fn ground(
spec: &ControlCheckSpec,
region: &CandidateRegion,
verdict: &LlmVerdict,
repo_id: &str,
) -> Option<Finding> {
if !verdict.violates {
return None;
}
let snippet = verdict.snippet.trim();
if snippet.is_empty() {
return None;
}
// Grounding: the quoted snippet must literally exist in the retrieved region.
let pos = region.content.find(snippet)?;
// Recompute the real line from the match — never trust the model's number.
let newlines_before = region.content[..pos].matches('\n').count();
let line = region.start_line + newlines_before as u32;
let mut finding = Finding::new(
repo_id.to_string(),
control_finding_fingerprint(&spec.control_id, &region.file, snippet),
"control-check".to_string(),
ScanType::CodeReview,
format!("{}: {}", spec.control_id, spec.title),
format!(
"Control {} appears violated ({}) at {}:{line}",
spec.control_id, spec.requirement, region.file
),
spec.severity.clone(),
);
finding.cwe = verdict.cwe.clone().or_else(|| spec.default_cwe.clone());
finding.file_path = Some(region.file.clone());
finding.line_number = Some(line);
finding.code_snippet = Some(snippet.to_string());
finding.confidence = Some(verdict.confidence);
// Carry the control reference on the finding.
finding.control_refs = vec![spec.control_id.clone()];
Some(finding)
}
/// Deterministic cache key for a (control, region, model, prompt-version) verdict
/// so identical inputs reproduce the same verdict without another LLM call.
pub fn cache_key(
control_id: &str,
region_content: &str,
model: &str,
prompt_version: &str,
) -> String {
hash_parts(&[control_id, region_content, model, prompt_version])
}
fn control_finding_fingerprint(control_id: &str, file: &str, snippet: &str) -> String {
hash_parts(&[control_id, file, snippet])
}
fn hash_parts(parts: &[&str]) -> String {
let mut hasher = Sha256::new();
for part in parts {
hasher.update(part.as_bytes());
hasher.update([0u8]); // domain separator between parts
}
hex::encode(hasher.finalize())
}
#[cfg(test)]
mod tests {
use super::*;
fn spec() -> ControlCheckSpec {
ControlCheckSpec {
control_id: "cra-ai-8".into(),
title: "No default passwords".into(),
requirement: "Products must not ship default credentials".into(),
default_cwe: Some("CWE-798".into()),
severity: Severity::High,
}
}
fn region() -> CandidateRegion {
CandidateRegion {
file: "src/auth.py".into(),
start_line: 10,
content: "def login():\n PASSWORD = \"admin123\"\n return PASSWORD\n".into(),
}
}
#[test]
fn grounds_real_snippet_with_recomputed_line() {
let v = LlmVerdict {
violates: true,
snippet: "PASSWORD = \"admin123\"".into(),
cwe: None,
confidence: 0.9,
};
let f = ground(&spec(), &region(), &v, "repo").expect("should ground");
assert_eq!(f.line_number, Some(11)); // 2nd line of a region starting at 10
assert_eq!(f.cwe.as_deref(), Some("CWE-798")); // fell back to the spec default
assert_eq!(f.control_refs, vec!["cra-ai-8".to_string()]); // control ref carried
assert_eq!(f.file_path.as_deref(), Some("src/auth.py"));
assert_eq!(f.code_snippet.as_deref(), Some("PASSWORD = \"admin123\""));
}
#[test]
fn drops_fabricated_snippet_not_in_region() {
let v = LlmVerdict {
violates: true,
snippet: "SECRET = \"totally-made-up\"".into(),
cwe: None,
confidence: 0.99,
};
assert!(ground(&spec(), &region(), &v, "repo").is_none());
}
#[test]
fn drops_non_violation_and_empty_snippet() {
let no = LlmVerdict {
violates: false,
snippet: "PASSWORD = \"admin123\"".into(),
cwe: None,
confidence: 0.9,
};
assert!(ground(&spec(), &region(), &no, "repo").is_none());
let empty = LlmVerdict {
violates: true,
snippet: " ".into(),
cwe: None,
confidence: 0.9,
};
assert!(ground(&spec(), &region(), &empty, "repo").is_none());
}
#[test]
fn cache_key_and_fingerprint_are_deterministic() {
assert_eq!(cache_key("c", "x", "m", "v"), cache_key("c", "x", "m", "v"));
assert_ne!(cache_key("c", "x", "m", "v"), cache_key("c", "y", "m", "v"));
assert_eq!(
control_finding_fingerprint("c", "f", "s"),
control_finding_fingerprint("c", "f", "s")
);
}
}
+1 -2
View File
@@ -1,5 +1,4 @@
pub mod config;
pub mod control_check;
pub mod db;
pub mod error;
pub mod models;
@@ -14,6 +13,6 @@ pub mod auth;
#[cfg(feature = "axum")]
pub mod tenant_ctx;
pub use config::{AgentConfig, DashboardConfig, PlcRuntimeConfig};
pub use config::{AgentConfig, DashboardConfig};
pub use error::CoreError;
pub use tenant::{OrgRole, TenantContext, TenantStatus};
-5
View File
@@ -76,10 +76,6 @@ pub struct Finding {
pub triage_rationale: Option<String>,
/// Developer feedback on finding quality
pub developer_feedback: Option<String>,
/// Compliance control ids this finding is evidence for (stamped by control
/// triage against the `control-map` LUT). Empty when unmapped.
#[serde(default)]
pub control_refs: Vec<String>,
#[serde(with = "super::serde_helpers::bson_datetime")]
pub created_at: DateTime<Utc>,
#[serde(with = "super::serde_helpers::bson_datetime")]
@@ -122,7 +118,6 @@ impl Finding {
triage_action: None,
triage_rationale: None,
developer_feedback: None,
control_refs: Vec::new(),
created_at: now,
updated_at: now,
}
+1 -11
View File
@@ -10,14 +10,11 @@ pub mod mcp;
pub mod mcp_token;
pub mod notification;
pub mod onboarding;
pub mod oscal;
pub mod oscal_assessment;
pub mod pentest;
pub mod repository;
pub mod sbom;
pub mod scan;
pub(crate) mod serde_helpers;
pub mod werkbank;
pub use auth::AuthInfo;
pub use chat::{ChatMessage, ChatRequest, ChatResponse, SourceReference};
@@ -41,19 +38,12 @@ pub use onboarding::{
GitArtifactConfig, IssueTrackerConfig, OnboardedTarget, PlcArtifactConfig, PlcFormat,
TargetScanConfig, TargetType, TargetTypeCandidate, WebArtifactConfig,
};
pub use oscal::OscalDocument;
pub use oscal_assessment::{assess, AssessmentResultsDoc, ControlLinker};
pub use pentest::{
AttackChainNode, AttackNodeStatus, AuthMode, CodeContextHint, Environment, IdentityProvider,
PentestAuthConfig, PentestConfig, PentestEvent, PentestMessage, PentestSession, PentestStats,
PentestStatus, PentestStrategy, SeverityDistribution, TestUserRecord, TesterInfo,
ToolCallRecord,
};
pub use repository::ScanTrigger;
pub use repository::{ScanTrigger, TrackedRepository};
pub use sbom::{SbomEntry, VulnRef};
pub use scan::{ScanPhase, ScanRun, ScanRunStatus, ScanType};
pub use werkbank::{
CompleteRequest, CompleteResponse, DastCollect, Executor, HeartbeatAck, HeartbeatRequest,
InputRef, Job, JobCollect, JobRecord, JobResult, JobRuntime, JobStatus, JobType, LeaseRequest,
LeasedJob,
};
-3
View File
@@ -202,9 +202,6 @@ pub enum PlcFormat {
PlcopenXml,
/// IEC 61131-3 Structured Text source.
StructuredText,
/// A CODESYS project archive (`.projectarchive` — a zip bundling the project
/// plus its referenced libraries and runtime; the source of the control-app SBOM).
ProjectArchive,
}
/// PLC-specific configuration for a [`ArtifactKind::PlcProject`] artifact.
-250
View File
@@ -1,250 +0,0 @@
//! OSCAL 1.1 catalog types + mapping into the controls corpus.
//!
//! Deserialises the OSCAL catalog served by breakpilot-compliance
//! (`GET /api/compliance/v1/oscal/catalog`) and maps its controls into the
//! framework-agnostic [`crate::traits::Control`] that the mapping engine consumes.
//! Only the fields we use are modelled; unknown OSCAL fields are ignored so the
//! producer can add detail without breaking us.
//!
//! Scope boundary: this is the *catalog* (domain content). Assessment objectives
//! and scanner routing live in our assessment layer, not here — see
//! [`crate::traits::ControlsProvider`].
use serde::Deserialize;
use crate::models::onboarding::ComplianceFramework;
use crate::traits::Control as CorpusControl;
/// A parsed OSCAL catalog document (`{"catalog": {...}}`).
#[derive(Debug, Clone, Deserialize)]
pub struct OscalDocument {
pub catalog: Catalog,
}
/// An OSCAL catalog: metadata + a tree of control groups.
#[derive(Debug, Clone, Deserialize)]
pub struct Catalog {
pub uuid: String,
pub metadata: Metadata,
#[serde(default)]
pub groups: Vec<Group>,
#[serde(rename = "back-matter", default)]
pub back_matter: Option<BackMatter>,
}
/// Catalog metadata (title/version + provenance props).
#[derive(Debug, Clone, Deserialize)]
pub struct Metadata {
pub title: String,
pub version: String,
#[serde(rename = "oscal-version")]
pub oscal_version: String,
#[serde(default)]
pub props: Vec<Prop>,
}
/// A name/value property, optionally namespaced.
#[derive(Debug, Clone, Deserialize)]
pub struct Prop {
pub name: String,
pub value: String,
#[serde(default)]
pub ns: Option<String>,
}
/// A control group (may nest sub-groups and controls).
#[derive(Debug, Clone, Deserialize)]
pub struct Group {
#[serde(default)]
pub id: String,
#[serde(default)]
pub title: String,
#[serde(default)]
pub controls: Vec<Control>,
#[serde(default)]
pub groups: Vec<Group>,
}
/// An OSCAL control (may nest enhancement controls).
#[derive(Debug, Clone, Deserialize)]
pub struct Control {
pub id: String,
#[serde(default)]
pub title: String,
#[serde(default)]
pub props: Vec<Prop>,
#[serde(default)]
pub parts: Vec<Part>,
#[serde(default)]
pub links: Vec<Link>,
#[serde(default)]
pub controls: Vec<Control>,
}
/// A control part (e.g. the `statement`), may nest sub-parts.
#[derive(Debug, Clone, Deserialize)]
pub struct Part {
#[serde(default)]
pub name: String,
#[serde(default)]
pub prose: Option<String>,
#[serde(default)]
pub parts: Vec<Part>,
}
/// A link, e.g. a `reference` to a back-matter resource.
#[derive(Debug, Clone, Deserialize)]
pub struct Link {
pub href: String,
#[serde(default)]
pub rel: Option<String>,
}
/// Back-matter holding referenced resources (e.g. the CRA measures).
#[derive(Debug, Clone, Deserialize)]
pub struct BackMatter {
#[serde(default)]
pub resources: Vec<Resource>,
}
/// A back-matter resource referenced by control links.
#[derive(Debug, Clone, Deserialize)]
pub struct Resource {
pub uuid: String,
#[serde(default)]
pub title: Option<String>,
#[serde(default)]
pub description: Option<String>,
}
impl Metadata {
/// First prop value with the given name.
pub fn prop(&self, name: &str) -> Option<&str> {
self.props
.iter()
.find(|p| p.name == name)
.map(|p| p.value.as_str())
}
}
impl Control {
/// First prop value with the given name.
pub fn prop(&self, name: &str) -> Option<&str> {
self.props
.iter()
.find(|p| p.name == name)
.map(|p| p.value.as_str())
}
/// The control's `statement` prose, if present.
pub fn statement(&self) -> Option<&str> {
self.parts
.iter()
.find(|p| p.name == "statement")
.and_then(|p| p.prose.as_deref())
}
}
impl OscalDocument {
/// The framework this catalog declares (`metadata.props[name="framework"]`).
pub fn framework(&self) -> Option<ComplianceFramework> {
framework_from_str(self.catalog.metadata.prop("framework")?)
}
/// The catalog `content-hash` prop — consumers pin this to snapshot/detect drift.
pub fn content_hash(&self) -> Option<&str> {
self.catalog.metadata.prop("content-hash")
}
/// Flatten the catalog into the corpus controls the mapping engine consumes.
pub fn to_controls(&self) -> Vec<CorpusControl> {
let framework = self.framework().unwrap_or(ComplianceFramework::Cra);
let source_label = self.catalog.metadata.title.as_str();
let mut out = Vec::new();
for group in &self.catalog.groups {
collect_group(group, framework, source_label, &mut out);
}
out
}
}
/// Map an OSCAL framework token (e.g. `"cra"`) to [`ComplianceFramework`] via its
/// serde snake_case representation.
fn framework_from_str(raw: &str) -> Option<ComplianceFramework> {
serde_json::from_value(serde_json::Value::String(raw.to_string())).ok()
}
fn collect_group(
group: &Group,
framework: ComplianceFramework,
source_label: &str,
out: &mut Vec<CorpusControl>,
) {
for control in &group.controls {
collect_control(control, framework, source_label, out);
}
for sub in &group.groups {
collect_group(sub, framework, source_label, out);
}
}
fn collect_control(
control: &Control,
framework: ComplianceFramework,
source_label: &str,
out: &mut Vec<CorpusControl>,
) {
let source = match control.prop("annex-anchor") {
Some(anchor) => Some(format!("{source_label} · {anchor}")),
None => Some(source_label.to_string()),
};
out.push(CorpusControl {
id: control.id.clone(),
framework,
title: control.title.clone(),
text: control.statement().unwrap_or_default().to_string(),
source,
});
for enhancement in &control.controls {
collect_control(enhancement, framework, source_label, out);
}
}
#[cfg(test)]
#[allow(clippy::unwrap_used)]
mod tests {
use super::*;
const CATALOG: &str = include_str!("../../tests/data/cra_catalog.json");
fn parse() -> OscalDocument {
serde_json::from_str(CATALOG).unwrap()
}
#[test]
fn parses_full_catalog() {
let doc = parse();
assert_eq!(doc.catalog.metadata.oscal_version, "1.1.2");
assert!(!doc.catalog.groups.is_empty());
assert!(doc.catalog.back_matter.is_some());
}
#[test]
fn maps_all_controls_to_corpus() {
let doc = parse();
let controls = doc.to_controls();
assert_eq!(controls.len(), 40);
assert_eq!(doc.framework(), Some(ComplianceFramework::Cra));
let c8 = controls.iter().find(|c| c.id == "cra-ai-8").unwrap();
assert_eq!(c8.framework, ComplianceFramework::Cra);
assert!(!c8.title.is_empty());
assert!(!c8.text.is_empty(), "statement prose should map into text");
assert!(c8.source.as_deref().unwrap_or_default().contains("Annex I"));
}
#[test]
fn exposes_content_hash_for_snapshotting() {
assert_eq!(parse().content_hash().map(str::len), Some(64));
}
}
@@ -1,424 +0,0 @@
//! OSCAL 1.1 assessment-results — assess our findings against catalog controls.
//!
//! The catalog (domain content) comes from the producer; the **assessment** is
//! ours. This links compliance [`Finding`]s to catalog control-ids and emits a
//! standard OSCAL assessment-results document: an observation per linked finding,
//! and a per-control finding with a `not-satisfied` status. `reviewed-controls`
//! records the full catalog set we considered.
//!
//! Deterministic: stable `uuid5` ids; the caller supplies the assessment
//! timestamp. Pure — no DB, no network.
use std::collections::HashMap;
use chrono::{DateTime, Utc};
use serde::Serialize;
use uuid::Uuid;
use crate::models::finding::{Finding, FindingStatus};
const OSCAL_VERSION: &str = "1.1.2";
/// Same namespace as the catalog exporter, so ids are stable and correlatable.
const NAMESPACE: Uuid = Uuid::from_bytes([
0x6f, 0x1e, 0x7c, 0x2a, 0x3b, 0x4d, 0x5e, 0x6f, 0x8a, 0x9b, 0x0c, 0x1d, 0x2e, 0x3f, 0x4a, 0x5b,
]);
fn det_uuid(name: &str) -> String {
Uuid::new_v5(&NAMESPACE, name.as_bytes()).to_string()
}
/// Links findings to the catalog control-ids they provide evidence for.
pub struct ControlLinker {
cwe_to_controls: HashMap<u32, Vec<String>>,
}
impl ControlLinker {
/// Build a linker from an explicit CWE → control-id map.
pub fn new(cwe_to_controls: HashMap<u32, Vec<String>>) -> Self {
Self { cwe_to_controls }
}
/// Seed of CWE → CRA Annex I control mappings (mirrors breakpilot's
/// `_CWE_TO_REQ`; extend as scanner coverage grows).
pub fn cra_seed() -> Self {
let pairs: &[(u32, &str)] = &[
(798, "cra-ai-8"),
(259, "cra-ai-8"),
(1392, "cra-ai-8"),
(327, "cra-ai-13"),
(326, "cra-ai-13"),
(319, "cra-ai-15"),
(311, "cra-ai-15"),
(89, "cra-ai-20"),
(79, "cra-ai-20"),
(78, "cra-ai-20"),
(22, "cra-ai-20"),
];
let mut map: HashMap<u32, Vec<String>> = HashMap::new();
for (cwe, id) in pairs {
map.entry(*cwe).or_default().push((*id).to_string());
}
Self::new(map)
}
/// Parse a CWE token such as `"CWE-798"` or `"798"` into its number.
fn parse_cwe(raw: &str) -> Option<u32> {
raw.trim_start_matches(|c: char| !c.is_ascii_digit())
.split(|c: char| !c.is_ascii_digit())
.next()
.filter(|s| !s.is_empty())
.and_then(|s| s.parse().ok())
}
/// The control-ids a finding provides evidence for (via its CWE).
pub fn controls_for(&self, finding: &Finding) -> Vec<String> {
finding
.cwe
.as_deref()
.and_then(Self::parse_cwe)
.and_then(|cwe| self.cwe_to_controls.get(&cwe))
.cloned()
.unwrap_or_default()
}
}
/// Build a standard OSCAL assessment-results document from `findings`, using each
/// finding's stamped `control_refs` for control linkage. EVERY non-false-positive
/// finding is emitted as an observation — mapped findings additionally produce a
/// per-control `not-satisfied` finding; **unmapped findings are reported as-is**
/// (an observation carrying their CWE/tool/severity, with no control target) so
/// nothing is lost. `at` is the assessment timestamp.
pub fn assess(findings: &[Finding], at: DateTime<Utc>) -> AssessmentResultsDoc {
let ts = at.to_rfc3339();
let mut observations = Vec::new();
let mut obs_by_control: HashMap<String, Vec<String>> = HashMap::new();
let mut mapped = 0usize;
let mut unmapped = 0usize;
for finding in findings {
if finding.status == FindingStatus::FalsePositive {
continue; // flagged tool false positive — excluded from the report
}
let obs_uuid = det_uuid(&format!("obs:{}", finding.fingerprint));
let location = match (&finding.file_path, finding.line_number) {
(Some(f), Some(l)) => Some(format!("{f}:{l}")),
(Some(f), None) => Some(f.clone()),
_ => None,
};
let is_mapped = !finding.control_refs.is_empty();
if is_mapped {
mapped += 1;
} else {
unmapped += 1;
}
let mut props = vec![
ObsProp::new("tool", &finding.scanner),
ObsProp::new("severity", &finding.severity.to_string()),
ObsProp::new("mapping", if is_mapped { "mapped" } else { "unmapped" }),
];
if let Some(cwe) = &finding.cwe {
props.push(ObsProp::new("cwe", cwe));
}
observations.push(Observation {
uuid: obs_uuid.clone(),
title: finding.title.clone(),
description: finding.description.clone(),
methods: vec!["TEST".to_string()],
collected: ts.clone(),
props,
relevant_evidence: vec![RelevantEvidence {
href: location.map(|l| format!("file://{l}")),
description: format!("[{}] {}", finding.scanner, finding.title),
}],
});
for control_id in &finding.control_refs {
obs_by_control
.entry(control_id.clone())
.or_default()
.push(obs_uuid.clone());
}
}
let mut hit_controls: Vec<&String> = obs_by_control.keys().collect();
hit_controls.sort();
let ar_findings: Vec<ArFinding> = hit_controls
.iter()
.map(|control_id| ArFinding {
uuid: det_uuid(&format!("finding:{control_id}")),
title: format!("Findings affect {control_id}"),
target: FindingTarget {
target_type: "statement-id".to_string(),
target_id: format!("{control_id}_smt"),
status: TargetStatus {
state: "not-satisfied".to_string(),
},
},
related_observations: obs_by_control[*control_id]
.iter()
.map(|u| RelatedObservation {
observation_uuid: u.clone(),
})
.collect(),
})
.collect();
let include_controls: Vec<SelectControlById> = hit_controls
.iter()
.map(|c| SelectControlById {
control_id: (*c).clone(),
})
.collect();
let result = ArResult {
uuid: det_uuid("result:cra"),
title: "Automated code-compliance assessment".to_string(),
description: format!(
"{} observation(s): {mapped} control-linked, {unmapped} unmapped (as-is); {} control(s) affected",
observations.len(),
include_controls.len()
),
start: ts.clone(),
reviewed_controls: ReviewedControls {
control_selections: vec![ControlSelection { include_controls }],
},
observations,
findings: ar_findings,
};
AssessmentResultsDoc {
assessment_results: AssessmentResults {
uuid: det_uuid("assessment-results:cra"),
metadata: ArMetadata {
title: "Compliance scanner — OSCAL assessment results".to_string(),
last_modified: ts,
version: "1.0.0".to_string(),
oscal_version: OSCAL_VERSION.to_string(),
},
import_ap: ImportAp {
href: "#cra-annex-i".to_string(),
},
results: vec![result],
},
}
}
// ── OSCAL assessment-results document (serialise) ────────────────────────────
/// The root OSCAL assessment-results document.
#[derive(Debug, Clone, Serialize)]
pub struct AssessmentResultsDoc {
#[serde(rename = "assessment-results")]
pub assessment_results: AssessmentResults,
}
#[derive(Debug, Clone, Serialize)]
pub struct AssessmentResults {
pub uuid: String,
pub metadata: ArMetadata,
#[serde(rename = "import-ap")]
pub import_ap: ImportAp,
pub results: Vec<ArResult>,
}
#[derive(Debug, Clone, Serialize)]
pub struct ArMetadata {
pub title: String,
#[serde(rename = "last-modified")]
pub last_modified: String,
pub version: String,
#[serde(rename = "oscal-version")]
pub oscal_version: String,
}
#[derive(Debug, Clone, Serialize)]
pub struct ImportAp {
pub href: String,
}
#[derive(Debug, Clone, Serialize)]
pub struct ArResult {
pub uuid: String,
pub title: String,
pub description: String,
pub start: String,
#[serde(rename = "reviewed-controls")]
pub reviewed_controls: ReviewedControls,
#[serde(skip_serializing_if = "Vec::is_empty")]
pub observations: Vec<Observation>,
#[serde(skip_serializing_if = "Vec::is_empty")]
pub findings: Vec<ArFinding>,
}
#[derive(Debug, Clone, Serialize)]
pub struct ReviewedControls {
#[serde(rename = "control-selections")]
pub control_selections: Vec<ControlSelection>,
}
#[derive(Debug, Clone, Serialize)]
pub struct ControlSelection {
#[serde(rename = "include-controls", skip_serializing_if = "Vec::is_empty")]
pub include_controls: Vec<SelectControlById>,
}
#[derive(Debug, Clone, Serialize)]
pub struct SelectControlById {
#[serde(rename = "control-id")]
pub control_id: String,
}
#[derive(Debug, Clone, Serialize)]
pub struct Observation {
pub uuid: String,
pub title: String,
pub description: String,
pub methods: Vec<String>,
pub collected: String,
#[serde(skip_serializing_if = "Vec::is_empty")]
pub props: Vec<ObsProp>,
#[serde(rename = "relevant-evidence", skip_serializing_if = "Vec::is_empty")]
pub relevant_evidence: Vec<RelevantEvidence>,
}
/// A name/value observation property (cwe, tool, severity, mapping status). Lets an
/// unmapped finding be reported fully as-is.
#[derive(Debug, Clone, Serialize)]
pub struct ObsProp {
pub name: String,
pub value: String,
}
impl ObsProp {
fn new(name: &str, value: &str) -> Self {
Self {
name: name.to_string(),
value: value.to_string(),
}
}
}
#[derive(Debug, Clone, Serialize)]
pub struct RelevantEvidence {
#[serde(skip_serializing_if = "Option::is_none")]
pub href: Option<String>,
pub description: String,
}
#[derive(Debug, Clone, Serialize)]
pub struct ArFinding {
pub uuid: String,
pub title: String,
pub target: FindingTarget,
#[serde(rename = "related-observations", skip_serializing_if = "Vec::is_empty")]
pub related_observations: Vec<RelatedObservation>,
}
#[derive(Debug, Clone, Serialize)]
pub struct FindingTarget {
#[serde(rename = "type")]
pub target_type: String,
#[serde(rename = "target-id")]
pub target_id: String,
pub status: TargetStatus,
}
#[derive(Debug, Clone, Serialize)]
pub struct TargetStatus {
pub state: String,
}
#[derive(Debug, Clone, Serialize)]
pub struct RelatedObservation {
#[serde(rename = "observation-uuid")]
pub observation_uuid: String,
}
#[cfg(test)]
#[allow(clippy::unwrap_used)]
mod tests {
use super::*;
use crate::models::finding::Severity;
use crate::models::scan::ScanType;
fn finding(fp: &str, cwe: Option<&str>, refs: &[&str]) -> Finding {
let mut f = Finding::new(
"repo".into(),
fp.into(),
"semgrep".into(),
ScanType::Sast,
"hardcoded credential".into(),
"desc".into(),
Severity::High,
);
f.cwe = cwe.map(Into::into);
f.file_path = Some("src/auth.rs".into());
f.line_number = Some(42);
f.control_refs = refs.iter().map(|s| s.to_string()).collect();
f
}
fn at() -> DateTime<Utc> {
DateTime::parse_from_rfc3339("2026-07-20T00:00:00Z")
.unwrap()
.with_timezone(&Utc)
}
#[test]
fn mapped_finding_becomes_control_finding() {
let doc = assess(&[finding("f1", Some("CWE-798"), &["cra-ai-8"])], at());
let r = &doc.assessment_results.results[0];
assert_eq!(r.observations.len(), 1);
assert_eq!(r.findings.len(), 1);
assert_eq!(r.findings[0].target.target_id, "cra-ai-8_smt");
assert_eq!(r.findings[0].target.status.state, "not-satisfied");
assert_eq!(
r.reviewed_controls.control_selections[0]
.include_controls
.len(),
1
);
}
#[test]
fn unmapped_finding_is_reported_as_is() {
let doc = assess(&[finding("f1", Some("CWE-319"), &[])], at());
let r = &doc.assessment_results.results[0];
assert_eq!(r.observations.len(), 1); // still emitted...
assert!(r.findings.is_empty()); // ...but no control finding
assert!(r.reviewed_controls.control_selections[0]
.include_controls
.is_empty());
let props: Vec<(&str, &str)> = r.observations[0]
.props
.iter()
.map(|p| (p.name.as_str(), p.value.as_str()))
.collect();
assert!(props.contains(&("mapping", "unmapped")));
assert!(props.contains(&("cwe", "CWE-319")));
}
#[test]
fn false_positive_is_excluded() {
let mut f = finding("f1", Some("CWE-798"), &["cra-ai-8"]);
f.status = FindingStatus::FalsePositive;
let doc = assess(&[f], at());
assert!(doc.assessment_results.results[0].observations.is_empty());
}
#[test]
fn deterministic_and_valid_oscal() {
let mk = || {
vec![
finding("f1", Some("CWE-798"), &["cra-ai-8"]),
finding("f2", Some("CWE-319"), &[]),
]
};
let a = serde_json::to_string(&assess(&mk(), at())).unwrap();
let b = serde_json::to_string(&assess(&mk(), at())).unwrap();
assert_eq!(a, b);
assert!(a.contains("\"oscal-version\":\"1.1.2\""));
assert!(a.contains("\"not-satisfied\""));
assert!(a.contains("\"mapping\""));
}
}
+93 -2
View File
@@ -1,6 +1,8 @@
use serde::{Deserialize, Serialize};
use chrono::{DateTime, Utc};
use serde::{Deserialize, Deserializer, Serialize};
use super::issue::TrackerType;
/// What initiated a scan.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub enum ScanTrigger {
@@ -8,3 +10,92 @@ pub enum ScanTrigger {
Webhook,
Manual,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TrackedRepository {
#[serde(rename = "_id", skip_serializing_if = "Option::is_none")]
pub id: Option<bson::oid::ObjectId>,
#[serde(default)]
pub name: String,
#[serde(default)]
pub git_url: String,
#[serde(default = "default_branch")]
pub default_branch: String,
pub local_path: Option<String>,
pub scan_schedule: Option<String>,
#[serde(default)]
pub webhook_enabled: bool,
/// Auto-generated HMAC secret for verifying incoming webhooks
#[serde(default, skip_serializing_if = "Option::is_none")]
pub webhook_secret: Option<String>,
pub tracker_type: Option<TrackerType>,
pub tracker_owner: Option<String>,
pub tracker_repo: Option<String>,
/// Optional per-repo PAT for the issue tracker (GitHub/GitLab/Jira)
#[serde(default, skip_serializing_if = "Option::is_none")]
pub tracker_token: Option<String>,
/// Optional auth token for HTTPS private repos (PAT or password)
#[serde(default, skip_serializing_if = "Option::is_none")]
pub auth_token: Option<String>,
/// Optional username for HTTPS auth (defaults to "x-access-token" for PATs)
#[serde(default, skip_serializing_if = "Option::is_none")]
pub auth_username: Option<String>,
pub last_scanned_commit: Option<String>,
#[serde(default, deserialize_with = "deserialize_findings_count")]
pub findings_count: u32,
#[serde(
default = "chrono::Utc::now",
with = "super::serde_helpers::bson_datetime"
)]
pub created_at: DateTime<Utc>,
#[serde(
default = "chrono::Utc::now",
with = "super::serde_helpers::bson_datetime"
)]
pub updated_at: DateTime<Utc>,
}
fn default_branch() -> String {
"main".to_string()
}
fn deserialize_findings_count<'de, D>(deserializer: D) -> Result<u32, D::Error>
where
D: Deserializer<'de>,
{
let bson = bson::Bson::deserialize(deserializer)?;
match &bson {
bson::Bson::Int32(n) => Ok(*n as u32),
bson::Bson::Int64(n) => Ok(*n as u32),
bson::Bson::Double(n) => Ok(*n as u32),
_ => Ok(0),
}
}
impl TrackedRepository {
pub fn new(name: String, git_url: String) -> Self {
let now = Utc::now();
// Generate a random webhook secret (hex-encoded UUID v4, no dashes)
let webhook_secret = uuid::Uuid::new_v4().to_string().replace('-', "");
Self {
id: None,
name,
git_url,
default_branch: "main".to_string(),
local_path: None,
scan_schedule: None,
auth_token: None,
auth_username: None,
webhook_enabled: false,
webhook_secret: Some(webhook_secret),
tracker_type: None,
tracker_owner: None,
tracker_repo: None,
tracker_token: None,
last_scanned_commit: None,
findings_count: 0,
created_at: now,
updated_at: now,
}
}
}
-5
View File
@@ -24,9 +24,6 @@ pub enum ScanType {
MobileStatic,
/// Static analysis of a container image.
ContainerScan,
/// Dynamic probing of a running PLC/SPS device over industrial protocols
/// (Modbus/TCP, OPC UA, …) for exposed/unauthenticated control access.
IcsProbe,
}
impl std::fmt::Display for ScanType {
@@ -46,7 +43,6 @@ impl std::fmt::Display for ScanType {
Self::PlcControlLogic => write!(f, "plc_control_logic"),
Self::MobileStatic => write!(f, "mobile_static"),
Self::ContainerScan => write!(f, "container_scan"),
Self::IcsProbe => write!(f, "ics_probe"),
}
}
}
@@ -80,7 +76,6 @@ pub enum ScanPhase {
LlmTriage,
IssueCreation,
DastScanning,
IcsProbe,
Completed,
}
-514
View File
@@ -1,514 +0,0 @@
//! The Werkbank job/result contract (WB-01).
//!
//! The shared, dependency-free vocabulary the control plane and the Werkbank
//! execution runner agree on: what a [`Job`] is, which [`Executor`] runs it, how
//! it moves through the queue ([`JobStatus`]), and what a [`JobResult`] carries
//! back. Jobs are declarative — TOML on disk, JSON on the wire — and results
//! reuse the existing scanner result types ([`Finding`], [`DastFinding`],
//! [`SbomEntry`]) so the runner produces exactly what the control plane persists.
//!
//! This module is intentionally free of the `mongodb`/`axum` features so the
//! runner can depend on `compliance-core` without pulling the server stack.
use std::collections::BTreeMap;
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
use super::dast::DastFinding;
use super::finding::Finding;
use super::sbom::SbomEntry;
/// The kind of dynamic-execution job.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "kebab-case")]
pub enum JobType {
/// Instantiate control logic on an ephemeral soft-PLC and probe it.
PlcProvision,
/// Boot a firmware image under QEMU and run dynamic checks.
QemuBoot,
/// Crawl and dynamically test a running web endpoint.
Dast,
/// Run an active penetration test against a running target.
Pentest,
}
/// How a runner executes a job — the CI-runner-style classification. A runner
/// advertises exactly one; a job requires one.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum Executor {
/// A subprocess on the runner host (dev / trusted single-node).
Shell,
/// One or more containers on the runner's Docker (default; QEMU runs here).
Docker,
/// A Pod/Job in a Kubernetes cluster (scale-out / multi-tenant).
K8s,
}
/// Lifecycle state of a job in the queue.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum JobStatus {
/// Waiting to be leased.
Queued,
/// Leased by a runner but not yet started.
Leased,
/// Executing on a runner.
Running,
/// Completed successfully.
Succeeded,
/// Completed with an error.
Failed,
/// The lease/lifetime deadline elapsed before completion.
Expired,
/// Cancelled by the control plane.
Cancelled,
}
impl JobStatus {
/// Whether the job has reached a terminal state (no further transitions).
pub fn is_terminal(self) -> bool {
matches!(
self,
JobStatus::Succeeded | JobStatus::Failed | JobStatus::Expired | JobStatus::Cancelled
)
}
}
/// A reference to an input artifact. Resolved by the runner from a source it can
/// reach; the blob itself never flows through the control plane (so an on-prem
/// runner keeps customer data local). Exactly one of `blob`/`url` should be set.
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct InputRef {
/// Content-addressed blob (e.g. `sha256:…`) the runner fetches from its store.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub blob: Option<String>,
/// A URL the runner can reach (git repo, internal artifact store, …).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub url: Option<String>,
}
impl InputRef {
/// A content-addressed blob reference.
pub fn blob(id: impl Into<String>) -> Self {
Self {
blob: Some(id.into()),
url: None,
}
}
}
/// Sandbox runtime knobs. Fields are executor/job-type specific and all optional;
/// `extra` carries anything not modelled explicitly.
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct JobRuntime {
/// Container image (Docker executor).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub image: Option<String>,
/// Memory cap (e.g. `512m`).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub memory: Option<String>,
/// CPU cap (e.g. `0.5`).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub cpus: Option<String>,
/// Network to join (e.g. `isolated`).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub network: Option<String>,
/// QEMU machine type (qemu-boot).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub machine: Option<String>,
/// QEMU target architecture (qemu-boot).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub arch: Option<String>,
/// Executor-specific extras not modelled above.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub extra: BTreeMap<String, String>,
}
/// DAST collection settings for jobs that scan a web endpoint.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct DastCollect {
/// Maximum crawl depth (kept shallow for ephemeral instances).
pub max_crawl_depth: u32,
}
/// What to collect from a run.
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct JobCollect {
/// Run the industrial-protocol probe (Modbus/OPC-UA/EtherNet-IP).
#[serde(default)]
pub ics_probe: bool,
/// Run DAST against the provisioned/booted web endpoint.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub dast: Option<DastCollect>,
/// Run an active pentest.
#[serde(default)]
pub pentest: bool,
/// Collect an SBOM.
#[serde(default)]
pub sbom: bool,
}
/// A declarative dynamic-execution job the control plane enqueues and a Werkbank
/// runner leases and executes.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct Job {
/// Unique job id (assigned by the control plane on enqueue).
pub id: String,
/// What kind of job this is.
#[serde(rename = "type")]
pub job_type: JobType,
/// Owning tenant.
pub tenant: String,
/// The onboarded target this job tests.
pub target_id: String,
/// The executor a runner must provide to run this job.
pub executor: Executor,
/// Runner capabilities this job requires (e.g. `arch=amd64`, `kvm=true`).
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub labels: Vec<String>,
/// Hard lifetime deadline for the whole job.
pub timeout_secs: u64,
/// Named input artifacts (e.g. `program`, `firmware`), by reference.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub inputs: BTreeMap<String, InputRef>,
/// Sandbox runtime knobs.
#[serde(default)]
pub runtime: JobRuntime,
/// What to collect from the run.
#[serde(default)]
pub collect: JobCollect,
}
impl Job {
/// A `plc-provision` job: instantiate the control logic named `program` on an
/// ephemeral soft-PLC (Docker executor) and collect the ICS probe + DAST.
pub fn plc_provision(
id: impl Into<String>,
tenant: impl Into<String>,
target_id: impl Into<String>,
program: InputRef,
timeout_secs: u64,
) -> Self {
let mut inputs = BTreeMap::new();
inputs.insert("program".to_string(), program);
Self {
id: id.into(),
job_type: JobType::PlcProvision,
tenant: tenant.into(),
target_id: target_id.into(),
executor: Executor::Docker,
labels: Vec::new(),
timeout_secs,
inputs,
runtime: JobRuntime::default(),
collect: JobCollect {
ics_probe: true,
dast: Some(DastCollect { max_crawl_depth: 2 }),
pentest: false,
sbom: false,
},
}
}
}
/// The outcome of running a job, posted back to the control plane. Findings and
/// SBOM reuse the shared scanner types, so the control plane persists them
/// unchanged. Submission is idempotent — keyed by [`JobResult::job_id`].
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
pub struct JobResult {
/// The job this result is for.
pub job_id: String,
/// Terminal status of the job.
pub status: Option<JobStatus>,
/// General scanner findings (e.g. ICS-probe findings).
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub findings: Vec<Finding>,
/// DAST findings from a web-endpoint scan.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub dast_findings: Vec<DastFinding>,
/// SBOM components collected from the run.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub sbom: Vec<SbomEntry>,
/// Error message when the job failed.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub error: Option<String>,
/// Captured execution log (truncated by the runner).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub logs: Option<String>,
/// When execution started on the runner.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub started_at: Option<DateTime<Utc>>,
/// When execution finished.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub finished_at: Option<DateTime<Utc>>,
}
impl JobResult {
/// A successful result for a job.
pub fn succeeded(job_id: impl Into<String>) -> Self {
Self {
job_id: job_id.into(),
status: Some(JobStatus::Succeeded),
..Default::default()
}
}
/// A failed result carrying an error message.
pub fn failed(job_id: impl Into<String>, error: impl Into<String>) -> Self {
Self {
job_id: job_id.into(),
status: Some(JobStatus::Failed),
error: Some(error.into()),
..Default::default()
}
}
}
/// A queued job as persisted by the control plane (WB-02): the [`Job`] contract
/// plus the queue bookkeeping — status, lease ownership, attempt count, and the
/// eventual result. The runner never sees this record; on lease it receives a
/// [`LeasedJob`] (the job plus a token it presents to heartbeat/complete).
///
/// Timestamps persist as native BSON dates so the queue's range queries (lease
/// FIFO by `created_at`, visibility-timeout sweep by `lease_expires_at`) compare
/// correctly.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct JobRecord {
/// The job to run.
pub job: Job,
/// Current queue state.
pub status: JobStatus,
/// The lease token held by the current runner (proves lease ownership).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub lease_token: Option<String>,
/// Id of the runner holding the lease.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub leased_by: Option<String>,
/// When the current lease expires — the visibility timeout after which a
/// crashed runner's job is swept back to `queued`.
#[serde(default, with = "super::serde_helpers::opt_bson_datetime")]
pub lease_expires_at: Option<DateTime<Utc>>,
/// Last heartbeat from the runner.
#[serde(default, with = "super::serde_helpers::opt_bson_datetime")]
pub heartbeat_at: Option<DateTime<Utc>>,
/// How many times the job has been leased (incremented on each lease).
#[serde(default)]
pub attempts: u32,
/// Set when the control plane requests cancellation; the runner sees it on
/// its next heartbeat and aborts.
#[serde(default)]
pub cancel_requested: bool,
/// The result, once the job reaches a terminal state.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub result: Option<JobResult>,
/// When the job was enqueued.
#[serde(with = "super::serde_helpers::bson_datetime")]
pub created_at: DateTime<Utc>,
/// Last modification.
#[serde(with = "super::serde_helpers::bson_datetime")]
pub updated_at: DateTime<Utc>,
}
impl JobRecord {
/// A freshly-enqueued (`queued`) record for a job.
pub fn queued(job: Job, now: DateTime<Utc>) -> Self {
Self {
job,
status: JobStatus::Queued,
lease_token: None,
leased_by: None,
lease_expires_at: None,
heartbeat_at: None,
attempts: 0,
cancel_requested: false,
result: None,
created_at: now,
updated_at: now,
}
}
}
/// A job handed to a runner on lease: what to run plus the token the runner must
/// present to heartbeat and complete it.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct LeasedJob {
/// The job to execute.
pub job: Job,
/// The lease token proving ownership (opaque to the runner).
pub lease_token: String,
}
/// The runner's view of a heartbeat: whether the control plane has asked the job
/// to stop. `None` from the queue means the lease was lost (token mismatch or the
/// job already terminal) and the runner should abandon the work.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
pub struct HeartbeatAck {
/// The control plane requested cancellation — the runner should tear down.
pub cancelled: bool,
}
// --- Runner ↔ control-plane transport (the pull API wire types) ---------------
// Shared so the runner (client) and the control plane (server) agree on shapes.
/// Runner → control plane: lease the oldest runnable job for this runner.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LeaseRequest {
/// The tenant queue to lease from.
pub tenant: String,
/// The runner id (advertised for attribution).
pub runner_id: String,
/// The executor this runner provides.
pub executor: Executor,
/// The capability labels this runner advertises.
#[serde(default)]
pub labels: Vec<String>,
/// Requested lease lifetime (the visibility timeout), in seconds.
pub lease_ttl_secs: u64,
}
/// Runner → control plane: prove lease ownership and extend it.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct HeartbeatRequest {
/// The tenant queue.
pub tenant: String,
/// The job being worked.
pub job_id: String,
/// The lease token from the [`LeasedJob`].
pub lease_token: String,
/// Lease lifetime to extend to, in seconds.
pub lease_ttl_secs: u64,
}
/// Runner → control plane: record a job's terminal result.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct CompleteRequest {
/// The tenant queue.
pub tenant: String,
/// The job being completed.
pub job_id: String,
/// The lease token proving ownership.
pub lease_token: String,
/// The result to record.
pub result: JobResult,
}
/// Control plane → runner: whether the completion was recorded (false if the
/// lease was already lost — token mismatch or the job had become terminal).
#[derive(Debug, Clone, Copy, Serialize, Deserialize)]
pub struct CompleteResponse {
/// Whether the result was recorded.
pub recorded: bool,
}
#[cfg(test)]
#[allow(clippy::expect_used, clippy::unwrap_used)]
mod tests {
use super::*;
#[test]
fn job_round_trips_through_json() {
let job = Job::plc_provision("job_1", "acme", "64f0aa", InputRef::blob("sha256:abc"), 180);
let json = serde_json::to_string(&job).expect("serialize");
let back: Job = serde_json::from_str(&json).expect("deserialize");
assert_eq!(job, back);
// Enum wire forms are the kebab/lowercase the contract documents.
assert!(json.contains("\"type\":\"plc-provision\""));
assert!(json.contains("\"executor\":\"docker\""));
}
#[test]
fn parses_the_design_doc_plc_provision_toml() {
// The exact shape from docs/DESIGN.md §5 (wrapped in a [job] table).
#[derive(Deserialize)]
struct JobFile {
job: Job,
}
let src = r#"
[job]
id = "job_01H"
type = "plc-provision"
tenant = "acme"
target_id = "64f0"
executor = "docker"
labels = ["arch=amd64"]
timeout_secs = 180
[job.inputs]
program = { blob = "sha256:deadbeef" }
[job.runtime]
image = "openplc:latest"
memory = "512m"
cpus = "0.5"
network = "isolated"
[job.collect]
ics_probe = true
dast = { max_crawl_depth = 2 }
"#;
let file: JobFile = toml::from_str(src).expect("parse job toml");
let job = file.job;
assert_eq!(job.job_type, JobType::PlcProvision);
assert_eq!(job.executor, Executor::Docker);
assert_eq!(job.labels, vec!["arch=amd64".to_string()]);
assert_eq!(
job.inputs.get("program").and_then(|i| i.blob.as_deref()),
Some("sha256:deadbeef")
);
assert_eq!(job.runtime.image.as_deref(), Some("openplc:latest"));
assert!(job.collect.ics_probe);
assert_eq!(job.collect.dast.map(|d| d.max_crawl_depth), Some(2));
}
#[test]
fn qemu_boot_runtime_fields_parse() {
#[derive(Deserialize)]
struct JobFile {
job: Job,
}
let src = r#"
[job]
id = "j2"
type = "qemu-boot"
tenant = "acme"
target_id = "t"
executor = "docker"
labels = ["kvm=true"]
timeout_secs = 600
[job.inputs]
firmware = { blob = "sha256:cafe" }
[job.runtime]
machine = "virt"
arch = "arm"
memory = "1g"
"#;
let file: JobFile = toml::from_str(src).expect("parse");
assert_eq!(file.job.job_type, JobType::QemuBoot);
assert_eq!(file.job.runtime.arch.as_deref(), Some("arm"));
assert_eq!(
file.job
.inputs
.get("firmware")
.and_then(|i| i.blob.as_deref()),
Some("sha256:cafe")
);
}
#[test]
fn status_terminality() {
assert!(JobStatus::Succeeded.is_terminal());
assert!(JobStatus::Expired.is_terminal());
assert!(!JobStatus::Queued.is_terminal());
assert!(!JobStatus::Running.is_terminal());
}
#[test]
fn result_constructors() {
assert_eq!(JobResult::succeeded("j").status, Some(JobStatus::Succeeded));
let f = JobResult::failed("j", "boom");
assert_eq!(f.status, Some(JobStatus::Failed));
assert_eq!(f.error.as_deref(), Some("boom"));
}
}
+17 -194
View File
@@ -14,12 +14,8 @@ use crate::models::{ArtifactKind, OnboardedTarget, ScanType, TargetType};
pub enum ArtifactRequirement {
/// Source code — a git repo or a source archive.
Code,
/// A reachable running instance (any live URL / endpoint, scheme-agnostic —
/// e.g. the ICS probe works off the host:port of a modbus:// or http:// ref).
/// A reachable running instance (live URL / endpoint).
RunningUrl,
/// A reachable **web** endpoint — a live URL with an http(s) scheme. DAST is
/// an HTTP crawler, so a modbus:// / opc.tcp:// endpoint does not satisfy it.
HttpUrl,
/// A firmware image / binary blob.
Firmware,
/// A PLC project (PLCopen XML or Structured Text).
@@ -138,7 +134,7 @@ fn sast_umbrella() -> Vec<ScanRule> {
/// The rule set for a target type. Scans that are never applicable to a type are
/// simply absent (e.g. DAST is not listed for a PLC target).
pub fn rules_for(target_type: TargetType) -> Vec<ScanRule> {
use ArtifactRequirement::{Firmware, HttpUrl, Mobile, Plc, RunningUrl};
use ArtifactRequirement::{Firmware, Mobile, Plc, RunningUrl};
match target_type {
TargetType::WebApp | TargetType::BackendService => {
let mut r = sast_umbrella();
@@ -146,7 +142,7 @@ pub fn rules_for(target_type: TargetType) -> Vec<ScanRule> {
ScanType::Dast,
true,
"Dynamic scan of the running endpoint",
HttpUrl,
RunningUrl,
));
r
}
@@ -207,62 +203,16 @@ pub fn rules_for(target_type: TargetType) -> Vec<ScanRule> {
ScanType::Dast,
false,
"Dynamic scan of exposed network services (if any)",
HttpUrl,
RunningUrl,
));
r
}
TargetType::PlcSps => {
// A PLC/SPS device is a composite: the control application *and* the
// device it runs on (firmware/OS + reachable runtime services). The
// control-logic scan runs on the PLC project; the firmware and DAST
// scans light up only when a firmware image / running endpoint is
// attached (e.g. a CODESYS runtime on a Yocto image with WebVisu).
// Firmware-image SBOM/CVE *execution* is shared with the firmware
// families and tracked in #151/#128; DAST over a WebVisu/OPC-UA
// endpoint uses the existing DAST path.
vec![
ScanRule::new(
ScanType::PlcControlLogic,
true,
"Control-logic security rules over the PLC program",
Plc,
),
// Device-level scans are offered but opt-in (default-off): they
// apply only when a firmware image is attached, and firmware-image
// SBOM/CVE *execution* is shared with the firmware families and
// still landing (#151/#128), so they must not silently auto-run.
ScanRule::new(
ScanType::FirmwareStatic,
false,
"Static analysis of the device firmware image (OS + runtime)",
Firmware,
),
ScanRule::new(
ScanType::Sbom,
false,
"SBOM from the device firmware image (OS packages + CODESYS runtime)",
Firmware,
),
ScanRule::new(
ScanType::Cve,
false,
"Match device firmware components against known CVEs",
Firmware,
),
ScanRule::new(
ScanType::Dast,
false,
"Dynamic scan of the running device (WebVisu / exposed services)",
HttpUrl,
),
ScanRule::new(
ScanType::IcsProbe,
false,
"Probe the running device over industrial protocols (Modbus/TCP, …)",
RunningUrl,
),
]
}
TargetType::PlcSps => vec![ScanRule::new(
ScanType::PlcControlLogic,
true,
"Control-logic security rules over the PLC program",
Plc,
)],
}
}
@@ -279,9 +229,6 @@ pub fn supports_pentest(target_type: TargetType) -> bool {
| TargetType::AndroidApp
| TargetType::IosApp
| TargetType::EmbeddedLinuxYocto
// A PLC/SPS device exposes reachable runtime services (WebVisu, OPC UA,
// the CODESYS programming protocol), so an active pentest applies.
| TargetType::PlcSps
)
}
@@ -289,9 +236,7 @@ pub fn supports_pentest(target_type: TargetType) -> bool {
fn representative_kind(req: ArtifactRequirement) -> Option<ArtifactKind> {
match req {
ArtifactRequirement::Code => Some(ArtifactKind::GitRepo),
ArtifactRequirement::RunningUrl | ArtifactRequirement::HttpUrl => {
Some(ArtifactKind::LiveUrl)
}
ArtifactRequirement::RunningUrl => Some(ArtifactKind::LiveUrl),
ArtifactRequirement::Firmware => Some(ArtifactKind::FirmwareImage),
ArtifactRequirement::Plc => Some(ArtifactKind::PlcProject),
ArtifactRequirement::Mobile => Some(ArtifactKind::MobilePackage),
@@ -300,29 +245,13 @@ fn representative_kind(req: ArtifactRequirement) -> Option<ArtifactKind> {
}
}
/// Whether a live-URL reference is an http(s) web endpoint (vs. an industrial
/// endpoint like `modbus://` / `opc.tcp://`, which DAST cannot crawl).
fn is_http_url(source_ref: &str) -> bool {
let s = source_ref.trim();
s.starts_with("http://") || s.starts_with("https://")
}
/// Whether the target carries an artifact that satisfies the requirement.
fn requirement_satisfied(req: ArtifactRequirement, target: &OnboardedTarget) -> bool {
match req {
ArtifactRequirement::Code => target.code_artifact().is_some(),
ArtifactRequirement::RunningUrl => target.has(ArtifactKind::LiveUrl),
ArtifactRequirement::HttpUrl => target
.artifacts
.iter()
.any(|a| a.kind == ArtifactKind::LiveUrl && is_http_url(&a.source_ref)),
ArtifactRequirement::Firmware => target.has(ArtifactKind::FirmwareImage),
// A PLC project artifact, or a code artifact (git repo / source archive)
// holding the control logic as PLCopen XML / ST exports — the common way
// CODESYS projects are version-controlled.
ArtifactRequirement::Plc => {
target.has(ArtifactKind::PlcProject) || target.code_artifact().is_some()
}
ArtifactRequirement::Plc => target.has(ArtifactKind::PlcProject),
ArtifactRequirement::Mobile => target.has(ArtifactKind::MobilePackage),
ArtifactRequirement::Container => target.has(ArtifactKind::ContainerImage),
ArtifactRequirement::Any => true,
@@ -339,10 +268,6 @@ pub fn applicable_scans(target: &OnboardedTarget) -> Vec<ScanOption> {
let required_artifact = representative_kind(rule.requires);
let blocked_reason = if satisfied {
None
} else if rule.requires == ArtifactRequirement::HttpUrl {
// A live URL may be present but non-HTTP (e.g. modbus://): be
// specific so the user knows DAST needs a web endpoint.
Some("no http(s) live URL — DAST needs a web endpoint".to_string())
} else {
Some(match required_artifact {
Some(kind) => format!("no {kind} artifact provided"),
@@ -415,124 +340,22 @@ mod tests {
}
#[test]
fn plc_control_logic_is_default_on_and_device_scans_block_without_artifacts() {
// A PLC project alone: control-logic runs; the device-level scans are
// offered but blocked until a firmware image / running endpoint is added.
fn plc_offers_only_control_logic() {
let t = target_with(
TargetType::PlcSps,
vec![Artifact::plc_project("p.xml", PlcFormat::PlcopenXml)],
);
let opts = applicable_scans(&t);
let plc = option(&opts, ScanType::PlcControlLogic).expect("control-logic offered");
assert!(plc.default_on && plc.blocked_reason.is_none());
for scan in [ScanType::FirmwareStatic, ScanType::Sbom, ScanType::Cve] {
let o = option(&opts, scan).expect("device scan offered");
assert!(
!o.default_on,
"{scan} must not pre-select without a firmware image"
);
assert!(o.blocked_reason.is_some());
}
let dast = option(&opts, ScanType::Dast).expect("dast offered");
assert!(!dast.default_on);
assert!(dast.blocked_reason.is_some());
}
#[test]
fn plc_control_logic_is_satisfied_by_a_git_repo() {
// A CODESYS project version-controlled in git (PLCopen XML / ST exports),
// no uploaded PlcProject artifact.
let t = target_with(TargetType::PlcSps, vec![Artifact::git_repo("u", "main")]);
let opts = applicable_scans(&t);
let plc = option(&opts, ScanType::PlcControlLogic).expect("control-logic offered");
assert!(
plc.default_on && plc.blocked_reason.is_none(),
"a git repo should satisfy PLC control-logic"
);
}
#[test]
fn plc_composite_lights_up_device_scans_with_firmware_and_url() {
// A CODESYS-on-Yocto device: PLC project + firmware image + WebVisu URL.
let t = target_with(
TargetType::PlcSps,
vec![
Artifact::plc_project("p.xml", PlcFormat::PlcopenXml),
Artifact::firmware_image("device.img"),
Artifact::live_url("http://plc.local/webvisu"),
],
);
let opts = applicable_scans(&t);
for scan in [
ScanType::PlcControlLogic,
ScanType::FirmwareStatic,
ScanType::Sbom,
ScanType::Cve,
] {
let o = option(&opts, scan).expect("scan offered");
assert!(o.blocked_reason.is_none(), "{scan} should be unblocked");
}
// Control-logic auto-runs; the device-level scans are unblocked but opt-in
// (default-off) until firmware-image execution lands (#151/#128).
assert!(option(&opts, ScanType::PlcControlLogic).unwrap().default_on);
assert!(!option(&opts, ScanType::Sbom).unwrap().default_on);
assert!(!option(&opts, ScanType::Dast).unwrap().default_on);
assert!(option(&opts, ScanType::Dast)
.unwrap()
.blocked_reason
.is_none());
}
#[test]
fn plc_with_modbus_url_offers_ics_probe_but_blocks_dast() {
// A soft-PLC reachable only over Modbus/TCP (no WebVisu). The ICS probe
// is applicable (it works off host:port), but DAST — an HTTP crawler —
// must be blocked so it isn't offered/run against a non-web endpoint.
let t = target_with(
TargetType::PlcSps,
vec![Artifact::live_url("modbus://plc-sim:502")],
);
let opts = applicable_scans(&t);
let ics = option(&opts, ScanType::IcsProbe).expect("ics probe offered");
assert!(
ics.blocked_reason.is_none(),
"ICS probe should be unblocked for a modbus:// endpoint"
);
assert!(!ics.default_on, "ICS probe stays opt-in (default-off)");
let dast = option(&opts, ScanType::Dast).expect("dast listed");
assert!(
dast.blocked_reason.is_some(),
"DAST must be blocked without an http(s) endpoint"
);
assert!(!dast.default_on);
}
#[test]
fn plc_with_http_webvisu_offers_both_dast_and_ics_probe() {
// A PLC exposing a WebVisu over HTTP: both DAST (web) and the ICS probe
// (OT ports on the same host) are applicable.
let t = target_with(
TargetType::PlcSps,
vec![Artifact::live_url("http://plc.local/webvisu")],
);
let opts = applicable_scans(&t);
assert!(option(&opts, ScanType::Dast)
.expect("dast offered")
.blocked_reason
.is_none());
assert!(option(&opts, ScanType::IcsProbe)
.expect("ics probe offered")
.blocked_reason
.is_none());
assert_eq!(opts.len(), 1);
assert_eq!(opts[0].scan, ScanType::PlcControlLogic);
assert!(opts[0].default_on);
}
#[test]
fn pentest_support_matches_reachable_families() {
assert!(supports_pentest(TargetType::WebApp));
assert!(supports_pentest(TargetType::BackendService));
assert!(supports_pentest(TargetType::EmbeddedLinuxYocto));
// A PLC/SPS device is network-reachable (WebVisu / OPC UA / 11740).
assert!(supports_pentest(TargetType::PlcSps));
assert!(!supports_pentest(TargetType::PlcSps));
assert!(!supports_pentest(TargetType::FirmwareBareMetal));
assert!(!supports_pentest(TargetType::DesktopApp));
}
File diff suppressed because it is too large Load Diff
+2
View File
@@ -10,6 +10,8 @@ pub enum Route {
#[layout(AppShell)]
#[route("/")]
OverviewPage {},
#[route("/repositories")]
RepositoriesPage {},
#[route("/targets")]
TargetsPage {},
#[route("/onboard")]
@@ -4,9 +4,8 @@ use dioxus_free_icons::Icon;
use crate::app::Route;
use crate::infrastructure::dast::fetch_dast_targets;
use crate::infrastructure::onboarding::fetch_targets;
use crate::infrastructure::pentest::{create_pentest_session_wizard, lookup_repo_by_url};
use crate::infrastructure::repositories::fetch_ssh_public_key;
use crate::infrastructure::repositories::{fetch_repositories, fetch_ssh_public_key};
const DISCLAIMER_TEXT: &str = "I confirm that I have authorization to perform security testing \
against the specified target. I understand that penetration testing may cause disruption to the \
@@ -40,7 +39,7 @@ pub fn PentestWizard(show: Signal<bool>) -> Element {
let mut show_target_dropdown = use_signal(|| false);
let mut show_repo_dropdown = use_signal(|| false);
let existing_targets = use_resource(|| async { fetch_dast_targets().await.ok() });
let existing_repos = use_resource(|| async { fetch_targets().await.ok() });
let existing_repos = use_resource(|| async { fetch_repositories(1).await.ok() });
// SSH key state for private repos
let mut ssh_public_key = use_signal(String::new);
@@ -212,25 +211,7 @@ pub fn PentestWizard(show: Signal<bool>) -> Element {
Some(Some(data)) => data
.data
.iter()
.filter_map(|t| {
let name = t
.get("name")
.and_then(|v| v.as_str())
.unwrap_or_default()
.to_string();
let git_url = t
.get("artifacts")
.and_then(|a| a.as_array())
.and_then(|arr| {
arr.iter().find(|a| {
a.get("kind").and_then(|k| k.as_str()) == Some("git_repo")
})
})
.and_then(|a| a.get("source_ref"))
.and_then(|s| s.as_str())?
.to_string();
Some((git_url, name))
})
.map(|r| (r.git_url.clone(), r.name.clone()))
.collect(),
_ => Vec::new(),
}
@@ -19,6 +19,10 @@ impl Database {
Ok(Self { inner: db })
}
pub fn repositories(&self) -> Collection<TrackedRepository> {
self.inner.collection("repositories")
}
pub fn findings(&self) -> Collection<Finding> {
self.inner.collection("findings")
}
@@ -39,67 +39,6 @@ pub struct ApplicableScansResponse {
pub data: ApplicableScansData,
}
/// Validate a target name. The name is used as the clone directory downstream,
/// so it must be a single safe segment (no slashes) and free of stray spaces.
pub fn validate_target_name(name: &str) -> Option<String> {
let n = name.trim();
if n.is_empty() {
return Some("Enter a name".to_string());
}
if name != n {
return Some("Remove the leading/trailing spaces".to_string());
}
if n.contains('/') || n.contains('\\') {
return Some("No slashes — the name becomes a folder (e.g. stm32f411-blinky)".to_string());
}
None
}
/// Client-side validation of an artifact reference for its kind. Returns an
/// error message when the value is obviously wrong for its category, so the
/// wizard / editor can flag it up front instead of the scan discovering it.
pub fn validate_artifact_ref(kind: &str, source_ref: &str) -> Option<String> {
let s = source_ref;
if s.trim().is_empty() {
return Some("Cannot be empty".to_string());
}
if s != s.trim() {
return Some("Remove the leading/trailing spaces".to_string());
}
let no_space = !s.contains(char::is_whitespace);
match kind {
"git_repo" => {
let looks_git = s.starts_with("https://")
|| s.starts_with("http://")
|| s.starts_with("ssh://")
|| s.starts_with("git://")
|| (s.contains('@') && s.contains(':'));
(!(looks_git && no_space))
.then(|| "Enter a git URL — https://…, ssh://…, or git@host:path".to_string())
}
"live_url" => {
// http(s) for web/DAST targets; modbus:// and opc.tcp:// for ICS
// devices probed by the ICS probe (e.g. modbus://plc:502).
let ok = (s.starts_with("https://")
|| s.starts_with("http://")
|| s.starts_with("modbus://")
|| s.starts_with("opc.tcp://"))
&& no_space;
(!ok).then(|| {
"Enter a URL — https://app.example.com, or modbus://host:502 for a PLC".to_string()
})
}
"container_image" => {
(!no_space).then(|| "Enter an image ref, e.g. registry/name:tag".to_string())
}
"source_archive" | "firmware_image" | "mobile_package" | "plc_project" => {
(!no_space).then(|| "Enter a path or URL (no spaces)".to_string())
}
// plaintext_description (and anything unknown): accept free-form text.
_ => None,
}
}
/// List onboarded targets.
#[server]
pub async fn fetch_targets() -> Result<TargetsResponse, ServerFnError> {
@@ -138,94 +77,6 @@ pub async fn create_target(
.map_err(|e| ServerFnError::new(e.to_string()))
}
/// Upload a file artifact (PLC project, firmware image, source archive, mobile
/// package) to a target — proxied to the agent as multipart.
#[server]
pub async fn upload_target_artifact(
id: String,
kind: String,
plc_format: Option<String>,
filename: String,
bytes: Vec<u8>,
) -> Result<TargetResponse, ServerFnError> {
let mut form = reqwest::multipart::Form::new().text("kind", kind).part(
"file",
reqwest::multipart::Part::bytes(bytes).file_name(filename),
);
if let Some(pf) = plc_format {
form = form.text("plc_format", pf);
}
let resp = super::agent_client::agent_request(
reqwest::Method::POST,
&format!("/api/v1/targets/{id}/artifacts/upload"),
)
.await?
.multipart(form)
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
resp.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))
}
/// Update a target's name / type / artifacts (dashboard editor).
#[server]
pub async fn update_target(
id: String,
name: Option<String>,
target_type: Option<String>,
artifacts: Option<Vec<ArtifactInputDto>>,
) -> Result<TargetResponse, ServerFnError> {
let mut body = serde_json::Map::new();
if let Some(n) = name {
body.insert("name".to_string(), serde_json::json!(n));
}
if let Some(t) = target_type {
body.insert("target_type".to_string(), serde_json::json!(t));
}
if let Some(a) = artifacts {
body.insert(
"artifacts".to_string(),
serde_json::to_value(a).map_err(|e| ServerFnError::new(e.to_string()))?,
);
}
let resp = super::agent_client::agent_request(
reqwest::Method::PATCH,
&format!("/api/v1/targets/{id}"),
)
.await?
.json(&serde_json::Value::Object(body))
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
resp.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))
}
/// Enable specific opt-in scans on a target by setting `scan_config.enabled_scans`.
/// `scans` are serde scan-type names (lowercase, no underscores — e.g. `icsprobe`).
#[server]
pub async fn enable_target_scans(
id: String,
scans: Vec<String>,
) -> Result<TargetResponse, ServerFnError> {
let body = serde_json::json!({ "scan_config": { "enabled_scans": scans } });
let resp = super::agent_client::agent_request(
reqwest::Method::PATCH,
&format!("/api/v1/targets/{id}"),
)
.await?
.json(&body)
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
resp.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))
}
/// Run kind-based classification on a target.
#[server]
pub async fn detect_target(id: String) -> Result<TargetResponse, ServerFnError> {
@@ -1,10 +1,145 @@
//! The agent's SSH deploy public key — shown so a read-only deploy key can be
//! added to private git targets. (The legacy repositories CRUD moved to the
//! unified onboarding/targets API.)
use dioxus::prelude::*;
use serde::{Deserialize, Serialize};
use compliance_core::models::TrackedRepository;
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
pub struct RepositoryListResponse {
pub data: Vec<TrackedRepository>,
pub total: Option<u64>,
pub page: Option<u64>,
}
#[server]
pub async fn fetch_repositories(page: u64) -> Result<RepositoryListResponse, ServerFnError> {
let path = format!("/api/v1/repositories?page={page}&limit=20");
let resp = super::agent_client::agent_get(&path)
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
let body: RepositoryListResponse = resp
.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
Ok(body)
}
#[server]
pub async fn add_repository(
name: String,
git_url: String,
default_branch: String,
auth_token: Option<String>,
auth_username: Option<String>,
tracker_type: Option<String>,
tracker_owner: Option<String>,
tracker_repo: Option<String>,
tracker_token: Option<String>,
) -> Result<(), ServerFnError> {
let mut body = serde_json::json!({
"name": name,
"git_url": git_url,
"default_branch": default_branch,
});
if let Some(token) = auth_token.filter(|t| !t.is_empty()) {
body["auth_token"] = serde_json::Value::String(token);
}
if let Some(username) = auth_username.filter(|u| !u.is_empty()) {
body["auth_username"] = serde_json::Value::String(username);
}
if let Some(tt) = tracker_type.filter(|t| !t.is_empty()) {
body["tracker_type"] = serde_json::Value::String(tt);
}
if let Some(to) = tracker_owner.filter(|t| !t.is_empty()) {
body["tracker_owner"] = serde_json::Value::String(to);
}
if let Some(tr) = tracker_repo.filter(|t| !t.is_empty()) {
body["tracker_repo"] = serde_json::Value::String(tr);
}
if let Some(tk) = tracker_token.filter(|t| !t.is_empty()) {
body["tracker_token"] = serde_json::Value::String(tk);
}
let resp = super::agent_client::agent_request(reqwest::Method::POST, "/api/v1/repositories")
.await?
.json(&body)
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
if !resp.status().is_success() {
let body = resp.text().await.unwrap_or_default();
return Err(ServerFnError::new(format!(
"Failed to add repository: {body}"
)));
}
Ok(())
}
#[server]
pub async fn update_repository(
repo_id: String,
name: Option<String>,
default_branch: Option<String>,
auth_token: Option<String>,
auth_username: Option<String>,
tracker_type: Option<String>,
tracker_owner: Option<String>,
tracker_repo: Option<String>,
tracker_token: Option<String>,
scan_schedule: Option<String>,
) -> Result<(), ServerFnError> {
let mut body = serde_json::Map::new();
if let Some(v) = name.filter(|s| !s.is_empty()) {
body.insert("name".into(), serde_json::Value::String(v));
}
if let Some(v) = default_branch.filter(|s| !s.is_empty()) {
body.insert("default_branch".into(), serde_json::Value::String(v));
}
if let Some(v) = auth_token {
body.insert("auth_token".into(), serde_json::Value::String(v));
}
if let Some(v) = auth_username {
body.insert("auth_username".into(), serde_json::Value::String(v));
}
if let Some(v) = tracker_type {
body.insert("tracker_type".into(), serde_json::Value::String(v));
}
if let Some(v) = tracker_owner {
body.insert("tracker_owner".into(), serde_json::Value::String(v));
}
if let Some(v) = tracker_repo {
body.insert("tracker_repo".into(), serde_json::Value::String(v));
}
if let Some(v) = tracker_token {
body.insert("tracker_token".into(), serde_json::Value::String(v));
}
if let Some(v) = scan_schedule {
body.insert("scan_schedule".into(), serde_json::Value::String(v));
}
let resp = super::agent_client::agent_request(
reqwest::Method::PATCH,
&format!("/api/v1/repositories/{repo_id}"),
)
.await?
.json(&body)
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
if !resp.status().is_success() {
let text = resp.text().await.unwrap_or_default();
return Err(ServerFnError::new(format!(
"Failed to update repository: {text}"
)));
}
Ok(())
}
/// Fetch the agent's SSH deploy public key.
#[server]
pub async fn fetch_ssh_public_key() -> Result<String, ServerFnError> {
let resp = super::agent_client::agent_get("/api/v1/settings/ssh-public-key")
@@ -28,3 +163,86 @@ pub async fn fetch_ssh_public_key() -> Result<String, ServerFnError> {
.unwrap_or("")
.to_string())
}
#[server]
pub async fn delete_repository(repo_id: String) -> Result<(), ServerFnError> {
let resp = super::agent_client::agent_request(
reqwest::Method::DELETE,
&format!("/api/v1/repositories/{repo_id}"),
)
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
if !resp.status().is_success() {
let body = resp.text().await.unwrap_or_default();
return Err(ServerFnError::new(format!(
"Failed to delete repository: {body}"
)));
}
Ok(())
}
#[server]
pub async fn trigger_repo_scan(repo_id: String) -> Result<(), ServerFnError> {
super::agent_client::agent_request(
reqwest::Method::POST,
&format!("/api/v1/repositories/{repo_id}/scan"),
)
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
Ok(())
}
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
pub struct WebhookConfigResponse {
pub webhook_secret: Option<String>,
pub tracker_type: String,
}
#[server]
pub async fn fetch_webhook_config(repo_id: String) -> Result<WebhookConfigResponse, ServerFnError> {
let resp =
super::agent_client::agent_get(&format!("/api/v1/repositories/{repo_id}/webhook-config"))
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
let body: WebhookConfigResponse = resp
.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
Ok(body)
}
/// Check if a repository has any running scans
#[server]
pub async fn check_repo_scanning(repo_id: String) -> Result<bool, ServerFnError> {
let resp = super::agent_client::agent_get("/api/v1/scan-runs?page=1&limit=1")
.await?
.send()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
let body: serde_json::Value = resp
.json()
.await
.map_err(|e| ServerFnError::new(e.to_string()))?;
// Check if the most recent scan for this repo is still running
if let Some(scans) = body.get("data").and_then(|d| d.as_array()) {
for scan in scans {
let scan_repo = scan.get("repo_id").and_then(|v| v.as_str()).unwrap_or("");
let status = scan.get("status").and_then(|v| v.as_str()).unwrap_or("");
if scan_repo == repo_id && status == "running" {
return Ok(true);
}
}
}
Ok(false)
}
@@ -1,4 +1,3 @@
use axum::extract::DefaultBodyLimit;
use axum::routing::{get, post};
use axum::{middleware, Extension};
use dioxus::prelude::*;
@@ -67,9 +66,6 @@ pub fn server_start(app: fn() -> Element) -> Result<(), DashboardError> {
// Webhook proxy: forward to agent (no auth required)
.route("/webhook/{platform}/{repo_id}", post(webhook_proxy))
.serve_dioxus_application(ServeConfig::new(), app)
// Allow large artifact uploads through the upload server function
// (PLC .projectarchive, firmware, mobile) — default is 2 MiB.
.layer(DefaultBodyLimit::max(512 * 1024 * 1024))
.layer(Extension(PendingOAuthStore::default()))
.layer(middleware::from_fn(require_auth))
.layer(Extension(server_state))
+6 -28
View File
@@ -2,11 +2,11 @@ use dioxus::prelude::*;
use crate::app::Route;
use crate::components::page_header::PageHeader;
use crate::infrastructure::onboarding::fetch_targets;
use crate::infrastructure::repositories::fetch_repositories;
#[component]
pub fn ChatIndexPage() -> Element {
let repos = use_resource(|| async { fetch_targets().await.ok() });
let repos = use_resource(|| async { fetch_repositories(1).await.ok() });
rsx! {
PageHeader {
@@ -28,32 +28,10 @@ pub fn ChatIndexPage() -> Element {
div { class: "graph-index-grid",
for repo in repo_list {
{
let repo_id = repo.get("_id").and_then(|o| o.get("$oid")).and_then(|s| s.as_str()).unwrap_or_default().to_string();
let name = repo.get("name").and_then(|n| n.as_str()).unwrap_or_default().to_string();
let url = repo
.get("artifacts")
.and_then(|a| a.as_array())
.and_then(|arr| {
arr.iter().find(|a| {
a.get("kind").and_then(|k| k.as_str()) == Some("git_repo")
})
})
.and_then(|a| a.get("source_ref"))
.and_then(|s| s.as_str())
.unwrap_or_default()
.to_string();
let branch = repo
.get("artifacts")
.and_then(|a| a.as_array())
.and_then(|arr| {
arr.iter().find_map(|a| {
a.get("git")
.and_then(|g| g.get("default_branch"))
.and_then(|b| b.as_str())
})
})
.unwrap_or("main")
.to_string();
let repo_id = repo.id.map(|id| id.to_hex()).unwrap_or_default();
let name = repo.name.clone();
let url = repo.git_url.clone();
let branch = repo.default_branch.clone();
rsx! {
Link {
to: Route::ChatPage { repo_id },

Some files were not shown because too many files have changed in this diff Show More