Commit f3c12544e8
f3c12544e8bef2bf45f4aa68ce0dec2c44e531ca
parent: 250db17024
Verified · cmc
cmc <hello@cleberg.net> · 2026-08-07 02:13 UTC
Stamp tool and ruleset provenance into reports
- Ruleset carries name, version, and a SHA-256 of its file
- JSON, Markdown, and HTML reports record the tool version and the ruleset
identity + hash so findings are traceable and re-performable
- Version the aws/github/gitlab rulesets and add MAPPING.md (authorship,
review status, framework revisions, disclaimers)
- Add ruff + pytest CI
Layout: unified · split
.github/workflows/ci.yml
added
+26
| @@ -0,0 +1,26 @@ |
| 1 | name: CI |
| 2 | |
| 3 | on: |
| 4 | push: |
| 5 | pull_request: |
| 6 | |
| 7 | jobs: |
| 8 | test: |
| 9 | runs-on: ubuntu-latest |
| 10 | strategy: |
| 11 | matrix: |
| 12 | python-version: ["3.10", "3.12"] |
| 13 | steps: |
| 14 | - uses: actions/checkout@v5 |
| 15 | - name: Set up Python ${{ matrix.python-version }} |
| 16 | uses: actions/setup-python@v6 |
| 17 | with: |
| 18 | python-version: ${{ matrix.python-version }} |
| 19 | - name: Install |
| 20 | run: | |
| 21 | python -m pip install --upgrade pip |
| 22 | pip install -e ".[dev]" |
| 23 | - name: Ruff |
| 24 | run: ruff check . |
| 25 | - name: Tests |
| 26 | run: pytest -q |
MAPPING.md
added
+40
| @@ -0,0 +1,40 @@ |
| 1 | # Rulesets — provenance and control-mapping rationale |
| 2 | |
| 3 | An `audit-report` **ruleset** (`audit_report/rulesets/*.yaml`) turns collected |
| 4 | evidence into control-relevant findings: each rule names a table, a check, and the |
| 5 | control identifiers the signal is offered as evidence *for*. This document records |
| 6 | where those mappings come from and their limits. |
| 7 | |
| 8 | ## What a mapping claims — and does not |
| 9 | |
| 10 | - A rule's `controls` list says: "this signal is relevant to these controls." |
| 11 | - A **fail** means a setting is in a state that does **not** support the control. |
| 12 | It is not a compliance verdict — the auditor still owns the conclusion. |
| 13 | - The control identifiers (`SOC2:CC6.1`, `ISO:A.5.17`, `NIST:IA-2`, …) are |
| 14 | reproduced; the frameworks' normative control text is not. See the catalog |
| 15 | provenance in [control-coverage `MAPPING.md`](https://github.com/audit-labs/control-coverage/blob/main/MAPPING.md). |
| 16 | - The mappings are the **maintainers' interpretation**, not reviewed or endorsed |
| 17 | by the AICPA, ISO/IEC, or NIST. |
| 18 | |
| 19 | ## Framework revisions referenced |
| 20 | |
| 21 | - **SOC 2** — Trust Services Criteria 2017 (2022 revised points of focus). |
| 22 | - **ISO/IEC 27001** — 27001:2022 Annex A. |
| 23 | - **NIST SP 800-53** — Rev. 5. |
| 24 | |
| 25 | ## Versioning and traceability |
| 26 | |
| 27 | - Each ruleset carries `name` and `version` fields. |
| 28 | - Every report stamps the tool name + version and the ruleset `name`, `version`, |
| 29 | and a **SHA-256 of the ruleset file** into its output (`tool` / `ruleset` in |
| 30 | JSON; the header line in Markdown/HTML). |
| 31 | - An auditor can therefore tie any finding back to the exact ruleset that produced |
| 32 | it, and re-perform against it. Bump `version` on any change to a rule's |
| 33 | controls, checks, or thresholds. |
| 34 | |
| 35 | ## Authorship and review |
| 36 | |
| 37 | - **Author:** the audit-labs maintainer. |
| 38 | - **Review status:** maintainer self-review; no independent professional review. |
| 39 | Validate a ruleset against your own control set before relying on it. |
| 40 | - **Effective date:** 2026-08. |
audit_report/cli.py
+1 −1
| @@ -181,7 +181,7 @@ def main(argv: list[str] | None = None) -> int: |
| 181 | 181 | return _run_diff(args, package, ruleset, formats) |
| 182 | 182 | |
| 183 | 183 | findings = evaluate(package, ruleset) |
| 184 | | report = reporters.build_report(package, findings) |
| 184 | report = reporters.build_report(package, findings, ruleset) |
| 185 | 185 | _emit(lambda fmt: reporters.render(report, fmt), formats, args.out, "report") |
| 186 | 186 | |
| 187 | 187 | counts = report.counts |
audit_report/reporters/__init__.py
+23 −2
| @@ -5,8 +5,10 @@ from __future__ import annotations |
| 5 | 5 | from dataclasses import dataclass |
| 6 | 6 | from datetime import datetime, timezone |
| 7 | 7 | |
| 8 | from .. import __version__ |
| 8 | 9 | from ..engine import Finding, control_coverage, summarize |
| 9 | 10 | from ..loader import Package |
| 11 | from ..rules import Ruleset |
| 10 | 12 | from . import html as _html |
| 11 | 13 | from . import json as _json |
| 12 | 14 | from . import markdown as _markdown |
| @@ -19,6 +21,7 @@ class Report: |
| 19 | 21 | package: Package |
| 20 | 22 | findings: list[Finding] |
| 21 | 23 | generated_at: str |
| 24 | ruleset: Ruleset | None = None |
| 22 | 25 | |
| 23 | 26 | @property |
| 24 | 27 | def counts(self) -> dict[str, int]: |
| @@ -28,11 +31,29 @@ class Report: |
| 28 | 31 | def coverage(self) -> dict[str, dict]: |
| 29 | 32 | return control_coverage(self.findings) |
| 30 | 33 | |
| 34 | @property |
| 35 | def provenance(self) -> dict[str, dict]: |
| 36 | """Tool + ruleset identity, so a report can be tied to what produced it.""" |
| 37 | rs = self.ruleset |
| 38 | return { |
| 39 | "tool": {"name": "audit-report", "version": __version__}, |
| 40 | "ruleset": { |
| 41 | "name": rs.name if rs else "", |
| 42 | "platform": rs.platform if rs else "", |
| 43 | "version": rs.version if rs else "", |
| 44 | "sha256": rs.sha256 if rs else "", |
| 45 | }, |
| 46 | } |
| 47 | |
| 31 | 48 | |
| 32 | | def build_report(package: Package, findings: list[Finding]) -> Report: |
| 49 | def build_report( |
| 50 | package: Package, findings: list[Finding], ruleset: Ruleset | None = None |
| 51 | ) -> Report: |
| 33 | 52 | """Assemble a :class:`Report` with a UTC generation timestamp.""" |
| 34 | 53 | stamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC") |
| 35 | | return Report(package=package, findings=findings, generated_at=stamp) |
| 54 | return Report( |
| 55 | package=package, findings=findings, generated_at=stamp, ruleset=ruleset |
| 56 | ) |
| 36 | 57 | |
| 37 | 58 | |
| 38 | 59 | RENDERERS = { |
audit_report/reporters/html.py
+12 −1
| @@ -90,10 +90,21 @@ def render(report: Report) -> str: |
| 90 | 90 | parts: list[str] = [] |
| 91 | 91 | |
| 92 | 92 | parts.append(f"<h1>Evidence Report — {escape(pkg.subject)}</h1>") |
| 93 | prov = report.provenance |
| 94 | tool, rs = prov["tool"], prov["ruleset"] |
| 95 | ruleset_meta = "" |
| 96 | if rs["sha256"]: |
| 97 | rs_ver = f" {escape(rs['version'])}" if rs["version"] else "" |
| 98 | ruleset_meta = ( |
| 99 | f" · Ruleset <code>{escape(rs['name'])}{rs_ver}</code> " |
| 100 | f"<code>sha256:{escape(rs['sha256'][:12])}</code>" |
| 101 | ) |
| 93 | 102 | parts.append( |
| 94 | 103 | f"<p class='meta'>Platform <code>{escape(pkg.platform)}</code> · " |
| 95 | 104 | f"Source <code>{escape(pkg.path.name)}</code> · " |
| 96 | | f"Generated {escape(report.generated_at)}</p>" |
| 105 | f"Generated {escape(report.generated_at)} · " |
| 106 | f"Tool <code>{escape(tool['name'])} {escape(tool['version'])}</code>" |
| 107 | f"{ruleset_meta}</p>" |
| 97 | 108 | ) |
| 98 | 109 | parts.append( |
| 99 | 110 | "<p class='summary-pills'>" |
audit_report/reporters/json.py
+2
| @@ -21,6 +21,8 @@ def to_dict(report: Report) -> dict: |
| 21 | 21 | "platform": pkg.platform, |
| 22 | 22 | "source_package": pkg.path.name, |
| 23 | 23 | "generated_at": report.generated_at, |
| 24 | "tool": report.provenance["tool"], |
| 25 | "ruleset": report.provenance["ruleset"], |
| 24 | 26 | "summary": report.counts, |
| 25 | 27 | "coverage": report.coverage, |
| 26 | 28 | "findings": [ |
audit_report/reporters/markdown.py
+9
| @@ -42,6 +42,15 @@ def render(report: Report) -> str: |
| 42 | 42 | out.append(f"- **Subject:** {pkg.subject}") |
| 43 | 43 | out.append(f"- **Source package:** `{pkg.path.name}`") |
| 44 | 44 | out.append(f"- **Generated:** {report.generated_at}") |
| 45 | prov = report.provenance |
| 46 | tool, rs = prov["tool"], prov["ruleset"] |
| 47 | out.append(f"- **Tool:** {tool['name']} {tool['version']}") |
| 48 | if rs["sha256"]: |
| 49 | rs_ver = f" {rs['version']}" if rs["version"] else "" |
| 50 | out.append( |
| 51 | f"- **Ruleset:** {rs['name']}{rs_ver} " |
| 52 | f"(`sha256:{rs['sha256'][:12]}`)" |
| 53 | ) |
| 45 | 54 | out.append( |
| 46 | 55 | f"- **Result:** {counts[FAIL]} failing · {counts[PASS]} passing · " |
| 47 | 56 | f"{counts[NOT_APPLICABLE]} not applicable" |
audit_report/rules.py
+19 −3
| @@ -17,6 +17,7 @@ Operators (``op``): ``equals``, ``not_equals``, ``is_true``, ``is_false``, |
| 17 | 17 | |
| 18 | 18 | from __future__ import annotations |
| 19 | 19 | |
| 20 | import hashlib |
| 20 | 21 | from dataclasses import dataclass, field |
| 21 | 22 | from pathlib import Path |
| 22 | 23 | |
| @@ -49,10 +50,18 @@ class Rule: |
| 49 | 50 | |
| 50 | 51 | @dataclass |
| 51 | 52 | class Ruleset: |
| 52 | | """A named collection of rules for one platform.""" |
| 53 | """A named collection of rules for one platform. |
| 54 | |
| 55 | ``version`` and ``sha256`` identify *which* ruleset produced a report, so an |
| 56 | auditor can re-perform against the exact mapping used. ``sha256`` is the |
| 57 | digest of the ruleset file's bytes as loaded. |
| 58 | """ |
| 53 | 59 | |
| 54 | 60 | platform: str |
| 55 | 61 | rules: list[Rule] |
| 62 | name: str = "" |
| 63 | version: str = "" |
| 64 | sha256: str = "" |
| 56 | 65 | |
| 57 | 66 | |
| 58 | 67 | def _as_number(value: str) -> float | None: |
| @@ -115,7 +124,8 @@ def match(condition: dict, row: dict[str, str]) -> bool: |
| 115 | 124 | |
| 116 | 125 | def load_ruleset(path: str | Path) -> Ruleset: |
| 117 | 126 | """Parse a ruleset YAML file into a :class:`Ruleset`, validating each rule.""" |
| 118 | | data = yaml.safe_load(Path(path).read_text(encoding="utf-8")) or {} |
| 127 | text = Path(path).read_text(encoding="utf-8") |
| 128 | data = yaml.safe_load(text) or {} |
| 119 | 129 | platform = data.get("platform") |
| 120 | 130 | if not platform: |
| 121 | 131 | raise ValueError(f"{path}: ruleset is missing a 'platform'") |
| @@ -137,4 +147,10 @@ def load_ruleset(path: str | Path) -> Ruleset: |
| 137 | 147 | raise ValueError(f"{rule.id}: unknown check type {check_type!r}") |
| 138 | 148 | rules.append(rule) |
| 139 | 149 | |
| 140 | | return Ruleset(platform=platform, rules=rules) |
| 150 | return Ruleset( |
| 151 | platform=platform, |
| 152 | rules=rules, |
| 153 | name=data.get("name", platform), |
| 154 | version=str(data.get("version", "")), |
| 155 | sha256=hashlib.sha256(text.encode("utf-8")).hexdigest(), |
| 156 | ) |
audit_report/rulesets/aws.yaml
+5
| @@ -4,6 +4,11 @@ |
| 4 | 4 | # (aws_audit_<profile>_<date>/). Each rule names a CSV table, a check, and the |
| 5 | 5 | # controls the signal is offered as evidence for. A "fail" means a setting is in |
| 6 | 6 | # a state that does NOT support the control — an auditor still owns the verdict. |
| 7 | # |
| 8 | # Provenance and control-mapping rationale: see MAPPING.md. Bump `version` on any |
| 9 | # change to a rule's controls, checks, or thresholds so reports stay traceable. |
| 10 | name: AWS ITGC ruleset |
| 11 | version: "2026.08.0" |
| 7 | 12 | platform: aws |
| 8 | 13 | |
| 9 | 14 | rules: |
audit_report/rulesets/github.yaml
+5
| @@ -3,6 +3,11 @@ |
| 3 | 3 | # Evaluated against an audit-tools GitHub evidence package |
| 4 | 4 | # (github_audit_<org>_<date>/). Complements gh-attest: same spirit of mapping |
| 5 | 5 | # GitHub signals to controls, applied offline to a captured CSV package. |
| 6 | # |
| 7 | # Provenance and control-mapping rationale: see MAPPING.md. Bump `version` on any |
| 8 | # change to a rule's controls, checks, or thresholds so reports stay traceable. |
| 9 | name: GitHub ITGC ruleset |
| 10 | version: "2026.08.0" |
| 6 | 11 | platform: github |
| 7 | 12 | |
| 8 | 13 | rules: |
audit_report/rulesets/gitlab.yaml
+5
| @@ -10,6 +10,11 @@ |
| 10 | 10 | # * audit-tools omits a CSV entirely when a collector returns no rows, so an |
| 11 | 11 | # absent table reports as "not applicable", not "pass". A rule can only |
| 12 | 12 | # speak to data that was actually collected. |
| 13 | # |
| 14 | # Provenance and control-mapping rationale: see MAPPING.md. Bump `version` on any |
| 15 | # change to a rule's controls, checks, or thresholds so reports stay traceable. |
| 16 | name: GitLab ITGC ruleset |
| 17 | version: "2026.08.0" |
| 13 | 18 | platform: gitlab |
| 14 | 19 | |
| 15 | 20 | rules: |
tests/test_reporters.py
+20 −3
| @@ -5,7 +5,7 @@ from pathlib import Path |
| 5 | 5 | |
| 6 | 6 | import pytest |
| 7 | 7 | |
| 8 | | from audit_report import reporters |
| 8 | from audit_report import __version__, reporters |
| 9 | 9 | from audit_report.cli import main |
| 10 | 10 | from audit_report.engine import evaluate |
| 11 | 11 | from audit_report.loader import load_package |
| @@ -17,8 +17,9 @@ RULESETS = Path("audit_report/rulesets") |
| 17 | 17 | |
| 18 | 18 | def _report(pkg_name="aws_audit_acme_2026-01-01", ruleset="aws.yaml"): |
| 19 | 19 | pkg = load_package(FIXTURES / pkg_name) |
| 20 | | findings = evaluate(pkg, load_ruleset(RULESETS / ruleset)) |
| 21 | | return reporters.build_report(pkg, findings) |
| 20 | rs = load_ruleset(RULESETS / ruleset) |
| 21 | findings = evaluate(pkg, rs) |
| 22 | return reporters.build_report(pkg, findings, rs) |
| 22 | 23 | |
| 23 | 24 | |
| 24 | 25 | def test_markdown_render_contains_sections(): |
| @@ -46,6 +47,22 @@ def test_json_render_roundtrips(): |
| 46 | 47 | assert ids["aws.iam.console-mfa"] == "fail" |
| 47 | 48 | |
| 48 | 49 | |
| 50 | def test_json_stamps_tool_and_ruleset_provenance(): |
| 51 | data = json.loads(reporters.render(_report(), "json")) |
| 52 | assert data["tool"] == {"name": "audit-report", "version": __version__} |
| 53 | rs = data["ruleset"] |
| 54 | assert rs["name"] == "AWS ITGC ruleset" |
| 55 | assert rs["platform"] == "aws" |
| 56 | assert rs["version"] == "2026.08.0" |
| 57 | assert len(rs["sha256"]) == 64 # full SHA-256 hex digest of the ruleset file |
| 58 | |
| 59 | |
| 60 | def test_markdown_shows_ruleset_provenance(): |
| 61 | md = reporters.render(_report(), "md") |
| 62 | assert "**Tool:** audit-report" in md |
| 63 | assert "sha256:" in md |
| 64 | |
| 65 | |
| 49 | 66 | def test_html_escapes_evidence(tmp_path): |
| 50 | 67 | pkg_dir = tmp_path / "aws_audit_x_2026-01-01" |
| 51 | 68 | pkg_dir.mkdir() |