audit-labs/audit-tools

A collection of scripts, queries, and other goodies you can use in an audit. audit automation compliance evidence scripts

Commit cb8640beec

cb8640beeced7c564c367d6694c578b6acdeafc7

parent: 4867af4f3d

Unsigned

cmc <hello@cleberg.net> · 2026-08-07 03:04 UTC

Resolve SonarCloud maintainability findings

Mechanical clean-ups across the shipped scripts and tools:
- Shell: use [[ ]] tests, redirect error messages to stderr, add explicit
  returns, and assign positional params to locals (shelldre S7688/S7677/S7682/S7679)
- Python: silence interface-mandated collector params with a _ prefix, drop
  genuinely unused params in the sampling writers, and extract duplicated
  string literals into constants (S1172, S1192)
- Tests: split a composite assertion and move a non-throwing call out of a
  pytest.raises block (S9073, S5778)
- sample.html: prefer Number.parseInt / Number.isNaN over the globals (S7773)

Layout: unified · split

applications/aws/aws_iam_users.sh +12 −12
@@ -9,7 +9,7 @@ ACCOUNT_NAME=""
99
1010# --- Prerequisite check ---
1111if ! command -v aws &> /dev/null || ! command -v jq &> /dev/null; then
12 echo "Error: Both AWS CLI and jq are required. Please install them and ensure they are in your PATH."
12 echo "Error: Both AWS CLI and jq are required. Please install them and ensure they are in your PATH." >&2
1313 exit 1
1414fi
1515
@@ -19,15 +19,15 @@ echo "Fetching IAM Identity Center and Account details..."
1919INSTANCE_ARN=$(aws sso-admin list-instances --query "Instances[0].InstanceArn" --output text)
2020IDENTITY_STORE_ID=$(aws sso-admin list-instances --query "Instances[0].IdentityStoreId" --output text)
2121
22if [ -z "$INSTANCE_ARN" ] || [ -z "$IDENTITY_STORE_ID" ]; then
23 echo "Error: Could not find IAM Identity Center instance ARN or Identity Store ID."
22if [[ -z "$INSTANCE_ARN" ]] || [[ -z "$IDENTITY_STORE_ID" ]]; then
23 echo "Error: Could not find IAM Identity Center instance ARN or Identity Store ID." >&2
2424 exit 1
2525fi
2626
2727ACCOUNT_ID=$(aws organizations list-accounts --query "Accounts[?Name=='$ACCOUNT_NAME' && Status=='ACTIVE'].Id" --output text)
2828
29if [ -z "$ACCOUNT_ID" ]; then
30 echo "Error: Could not find an active AWS account with the name '$ACCOUNT_NAME'."
29if [[ -z "$ACCOUNT_ID" ]]; then
30 echo "Error: Could not find an active AWS account with the name '$ACCOUNT_NAME'." >&2
3131 exit 1
3232fi
3333
@@ -41,7 +41,7 @@ PROVISIONED_SETS_ARN=$(aws sso-admin list-permission-sets-provisioned-to-account
4141 --account-id "$ACCOUNT_ID" \
4242 --query "PermissionSets[]" --output text)
4343
44if [ -z "$PROVISIONED_SETS_ARN" ]; then
44if [[ -z "$PROVISIONED_SETS_ARN" ]]; then
4545 echo "No permission sets are provisioned for account '$ACCOUNT_NAME'."
4646 exit 0
4747fi
@@ -64,14 +64,14 @@ for PS_ARN in $PROVISIONED_SETS_ARN; do
6464 --permission-set-arn "$PS_ARN" \
6565 --query "AccountAssignments[]" --output json)
6666
67 if [ "$(echo "$ACCOUNT_ASSIGNMENTS" | jq 'length')" -eq 0 ]; then
67 if [[ "$(echo "$ACCOUNT_ASSIGNMENTS" | jq 'length')" -eq 0 ]]; then
6868 echo " -> Permission Set ARN $PS_ARN is provisioned but has no active assignments."
6969 continue
7070 fi
7171
7272 # Since there are assignments, let's get the permission set's details (policies, name)
7373 # Using a cache to avoid redundant calls if a PS is somehow listed twice
74 if [ -z "${PERMISSION_SET_CACHE[$PS_ARN]}" ]; then
74 if [[ -z "${PERMISSION_SET_CACHE[$PS_ARN]}" ]]; then
7575 echo " -> Fetching policies for Permission Set: $PS_ARN"
7676 PS_NAME=$(aws sso-admin describe-permission-set --instance-arn "$INSTANCE_ARN" --permission-set-arn "$PS_ARN" --query "PermissionSet.Name" --output text)
7777 MANAGED_POLICIES=$(aws sso-admin list-managed-policies-in-permission-set --instance-arn "$INSTANCE_ARN" --permission-set-arn "$PS_ARN" --query "AttachedManagedPolicies[].Arn" --output json)
@@ -87,16 +87,16 @@ for PS_ARN in $PROVISIONED_SETS_ARN; do
8787
8888 # Now process each assignment found for this permission set
8989 for row in $(echo "${ACCOUNT_ASSIGNMENTS}" | jq -r '.[] | @base64'); do
90 _jq() { echo ${row} | base64 --decode | jq -r ${1}; }
90 _jq() { echo ${row} | base64 --decode | jq -r ${1}; return 0; }
9191 PRINCIPAL_TYPE=$(_jq '.PrincipalType')
9292 PRINCIPAL_ID=$(_jq '.PrincipalId')
9393
9494 # Get Principal (User/Group) Name, using a cache
95 if [ -z "${PRINCIPAL_NAME_CACHE[$PRINCIPAL_ID]}" ]; then
95 if [[ -z "${PRINCIPAL_NAME_CACHE[$PRINCIPAL_ID]}" ]]; then
9696 PRINCIPAL_NAME=""
97 if [ "$PRINCIPAL_TYPE" == "USER" ]; then
97 if [[ "$PRINCIPAL_TYPE" == "USER" ]]; then
9898 PRINCIPAL_NAME=$(aws identitystore describe-user --identity-store-id "$IDENTITY_STORE_ID" --user-id "$PRINCIPAL_ID" --query "UserName" --output text 2>/dev/null)
99 elif [ "$PRINCIPAL_TYPE" == "GROUP" ]; then
99 elif [[ "$PRINCIPAL_TYPE" == "GROUP" ]]; then
100100 PRINCIPAL_NAME=$(aws identitystore describe-group --identity-store-id "$IDENTITY_STORE_ID" --group-id "$PRINCIPAL_ID" --query "DisplayName" --output text 2>/dev/null)
101101 fi
102102 PRINCIPAL_NAME_CACHE[$PRINCIPAL_ID]=${PRINCIPAL_NAME:-"ID: $PRINCIPAL_ID"}
applications/aws/aws_s3_buckets.sh +12 −12
@@ -18,7 +18,7 @@ echo "---"
1818echo "1. Retrieving all bucket names..."
1919BUCKET_LIST=$(aws s3api list-buckets --region "$MASTER_REGION" --query 'Buckets[].Name' --output text)
2020
21if [ -z "$BUCKET_LIST" ]; then
21if [[ -z "$BUCKET_LIST" ]]; then
2222 echo "✅ No S3 buckets found in this account."
2323 exit 0
2424fi
@@ -32,14 +32,14 @@ for BUCKET_NAME in $BUCKET_LIST; do
3232 # 2. Find the bucket region
3333 for REGION in $AWS_REGIONS; do
3434 BUCKET_LOCATION_RESPONSE=$(aws s3api get-bucket-location --bucket "$BUCKET_NAME" --region "$REGION" 2>/dev/null)
35 if [ $? -eq 0 ]; then
35 if [[ $? -eq 0 ]]; then
3636 LOCATION_CONSTRAINT=$(echo "$BUCKET_LOCATION_RESPONSE" | jq -r '.LocationConstraint')
3737 BUCKET_REGION=${LOCATION_CONSTRAINT:-"us-east-1"}
3838 break
3939 fi
4040 done
4141
42 if [ -z "$BUCKET_REGION" ]; then
42 if [[ -z "$BUCKET_REGION" ]]; then
4343 echo " ⚠️ WARNING: Could not determine region for $BUCKET_NAME. Skipping all checks."
4444 echo "$BUCKET_NAME,UNKNOWN,N/A,N/A,N/A,N/A,UNKNOWN" >> "$REPORT_FILE"
4545 continue
@@ -57,14 +57,14 @@ for BUCKET_NAME in $BUCKET_LIST; do
5757 # --- CHECK A: Public Access Block (PAB) ---
5858 PAB_STATUS=$(aws s3api get-public-access-block --bucket "$BUCKET_NAME" --region "$BUCKET_REGION" 2>/dev/null)
5959
60 if [ $? -ne 0 ]; then
60 if [[ $? -ne 0 ]]; then
6161 # PAB Missing is the highest risk state.
6262 PAB_FULLY_RESTRICTED="CRITICAL-MISSING"
6363 OVERALL_PUBLIC_STATUS="TRUE - PAB Missing"
6464 else
6565 # Check if ALL four PAB flags are true
6666 PAB_CONFIG=$(echo "$PAB_STATUS" | jq -r '.PublicAccessBlockConfiguration')
67 if [ "$(echo "$PAB_CONFIG" | jq -r '.BlockPublicAcls and .IgnorePublicAcls and .BlockPublicPolicy and .RestrictPublicBuckets')" = "true" ]; then
67 if [[ "$(echo "$PAB_CONFIG" | jq -r '.BlockPublicAcls and .IgnorePublicAcls and .BlockPublicPolicy and .RestrictPublicBuckets')" = "true" ]]; then
6868 PAB_FULLY_RESTRICTED="TRUE"
6969 else
7070 PAB_FULLY_RESTRICTED="FALSE-VULNERABLE"
@@ -74,9 +74,9 @@ for BUCKET_NAME in $BUCKET_LIST; do
7474 # --- CHECK B: Bucket Policy Status (If S3 service thinks it's public) ---
7575 POLICY_STATUS=$(aws s3api get-bucket-policy-status --bucket "$BUCKET_NAME" --region "$BUCKET_REGION" 2>/dev/null)
7676
77 if [ $? -eq 0 ]; then
77 if [[ $? -eq 0 ]]; then
7878 POLICY_IS_PUBLIC=$(echo "$POLICY_STATUS" | jq -r '.PolicyStatus.IsPublic')
79 if [ "$POLICY_IS_PUBLIC" = "true" ]; then
79 if [[ "$POLICY_IS_PUBLIC" = "true" ]]; then
8080 OVERALL_PUBLIC_STATUS="TRUE - Policy"
8181 fi
8282 else
@@ -87,13 +87,13 @@ for BUCKET_NAME in $BUCKET_LIST; do
8787 # --- CHECK C: Bucket ACLs (for AllUsers group) ---
8888 ACL_RESPONSE=$(aws s3api get-bucket-acl --bucket "$BUCKET_NAME" --region "$BUCKET_REGION" 2>/dev/null)
8989
90 if [ $? -eq 0 ]; then
90 if [[ $? -eq 0 ]]; then
9191 # Find if any grant to 'http://acs.amazonaws.com/groups/global/AllUsers' exists
9292
9393 # Check for READ access
9494 if echo "$ACL_RESPONSE" | jq -e '.Grants[] | select(.Grantee.URI=="http://acs.amazonaws.com/groups/global/AllUsers") | select(.Permission | test("READ|FULL_CONTROL"))' >/dev/null; then
9595 ACL_ALL_USERS_READ="TRUE"
96 if [ "$OVERALL_PUBLIC_STATUS" = "FALSE" ]; then
96 if [[ "$OVERALL_PUBLIC_STATUS" = "FALSE" ]]; then
9797 OVERALL_PUBLIC_STATUS="TRUE - ACL Read"
9898 fi
9999 fi
@@ -101,7 +101,7 @@ for BUCKET_NAME in $BUCKET_LIST; do
101101 # Check for WRITE access (often less common for public, but still public exposure)
102102 if echo "$ACL_RESPONSE" | jq -e '.Grants[] | select(.Grantee.URI=="http://acs.amazonaws.com/groups/global/AllUsers") | select(.Permission | test("WRITE|FULL_CONTROL"))' >/dev/null; then
103103 ACL_ALL_USERS_WRITE="TRUE"
104 if [ "$OVERALL_PUBLIC_STATUS" = "FALSE" ]; then
104 if [[ "$OVERALL_PUBLIC_STATUS" = "FALSE" ]]; then
105105 OVERALL_PUBLIC_STATUS="TRUE - ACL Write"
106106 fi
107107 fi
@@ -111,9 +111,9 @@ for BUCKET_NAME in $BUCKET_LIST; do
111111 fi
112112
113113 # Final check for PAB failure (PAB is the highest authority)
114 if [ "$PAB_FULLY_RESTRICTED" = "CRITICAL-MISSING" ]; then
114 if [[ "$PAB_FULLY_RESTRICTED" = "CRITICAL-MISSING" ]]; then
115115 OVERALL_PUBLIC_STATUS="TRUE - PAB Missing (CRITICAL)"
116 elif [ "$OVERALL_PUBLIC_STATUS" != "FALSE" ] && [ "$PAB_FULLY_RESTRICTED" != "TRUE" ]; then
116 elif [[ "$OVERALL_PUBLIC_STATUS" != "FALSE" ]] && [[ "$PAB_FULLY_RESTRICTED" != "TRUE" ]]; then
117117 # If the bucket is found public by Policy or ACL AND PAB isn't fully set, confirm it's public
118118 : # Status already set by Policy or ACL check above
119119 fi
applications/github/collectors/members.py +2 −2
@@ -132,7 +132,7 @@ def outside_collaborators(org, cfg, repo_collabs):
132132 return rows
133133
134134
135def privileged_access(org, cfg, repo_collabs):
135def privileged_access(_org, _cfg, repo_collabs):
136136 """
137137 All users with admin permission on any repo.
138138 Accepts pre-fetched repo_collabs from fetch_repo_collaborators().
@@ -196,7 +196,7 @@ def team_permissions(org, cfg):
196196 return rows
197197
198198
199def permission_matrix(org, cfg, repo_collabs):
199def permission_matrix(_org, _cfg, repo_collabs):
200200 """
201201 Full per-repo/per-user permission cross-reference.
202202 Accepts pre-fetched repo_collabs from fetch_repo_collaborators().
applications/gitlab/collectors/approvals.py +1 −1
@@ -12,7 +12,7 @@ import requests
1212from .api import paginate
1313
1414
15def approval_rules(group, cfg, projects):
15def approval_rules(_group, cfg, projects):
1616 rows = []
1717 for p in projects:
1818 try:
applications/gitlab/collectors/branch_protections.py +1 −1
@@ -12,7 +12,7 @@ def _levels(entries):
1212 return ", ".join(e.get("access_level_description", "") for e in entries) or "(none)"
1313
1414
15def branch_protections(group, cfg, projects):
15def branch_protections(_group, cfg, projects):
1616 rows = []
1717 for p in projects:
1818 try:
applications/gitlab/collectors/members.py +1 −1
@@ -44,7 +44,7 @@ def group_members(group, cfg):
4444 ]
4545
4646
47def project_members(group, cfg, projects):
47def project_members(_group, cfg, projects):
4848 """Direct and inherited members of every project in the group."""
4949 rows = []
5050 for p in projects:
applications/gitlab/collectors/pipelines.py +1 −1
@@ -7,7 +7,7 @@ import requests
77from .api import paginate
88
99
10def pipelines(group, cfg, projects):
10def pipelines(_group, cfg, projects):
1111 rows = []
1212 for p in projects:
1313 try:
applications/gitlab/collectors/projects.py +1 −1
@@ -17,7 +17,7 @@ def fetch_projects(group, cfg):
1717 )
1818
1919
20def project_list(group, cfg, projects):
20def project_list(_group, _cfg, projects):
2121 """Format the project cache into audit rows."""
2222 return [
2323 {
databases/sql/passwords/passwords.py +15 −10
@@ -5,6 +5,11 @@ Checks SQL Server user data for compliance with Windows policies.
55# Import packages
66import pandas as pd
77
8# Report column labels (defined once to avoid duplicated string literals).
9TYPE_CHECK = "Type Check"
10POLICY_CHECK = "Policy Check"
11EXPIRATION_CHECK = "Expiration Check"
12
813# Load the data into a pandas DataFrame
914df_input = pd.read_csv("./data.csv")
1015
@@ -24,38 +29,38 @@ def apply_rules_and_report(df):
2429 for _, row in df.iterrows():
2530 result = {
2631 "Name": row["name"],
27 "Type Check": "",
28 "Policy Check": "",
29 "Expiration Check": "",
32 TYPE_CHECK: "",
33 POLICY_CHECK: "",
34 EXPIRATION_CHECK: "",
3035 "Reason": "",
3136 }
3237
3338 # Check the type_desc
3439 if row["type_desc"] == "SQL_LOGIN":
35 result["Type Check"] = "SQL_LOGIN"
40 result[TYPE_CHECK] = "SQL_LOGIN"
3641 elif row["type_desc"] == "WINDOWS_LOGIN":
37 result["Type Check"] = "N/A"
42 result[TYPE_CHECK] = "N/A"
3843 result["Reason"] = "Refer to Windows password policy."
3944 else:
40 result["Type Check"] = "Manual Review"
45 result[TYPE_CHECK] = "Manual Review"
4146 result["Reason"] = "Reviewer to manually review."
4247
4348 # Check if password policy is enforced
4449 if row["is_policy_checked"] == 1:
45 result["Policy Check"] = "PASS"
50 result[POLICY_CHECK] = "PASS"
4651 result["Reason"] += """Password policy is enforced. Reviewer to
4752 check the assigned policy."""
4853 else:
49 result["Policy Check"] = "FAIL"
54 result[POLICY_CHECK] = "FAIL"
5055 result["Reason"] += "Password policy is not enforced."
5156
5257 # Check if password expiration is enforced
5358 if row["is_expiration_checked"] == 1:
54 result["Expiration Check"] = "PASS"
59 result[EXPIRATION_CHECK] = "PASS"
5560 result["Reason"] += """Password expiration is enforced. Reviewer to
5661 check the expiration policy."""
5762 else:
58 result["Expiration Check"] = "FAIL"
63 result[EXPIRATION_CHECK] = "FAIL"
5964 result["Reason"] += "Password expiration is not enforced."
6065
6166 report.append(result)
os/linux/passwords.sh +5 −3
@@ -4,11 +4,11 @@
44extract_password_params() {
55 echo "Checking /etc/pam.d/system-auth for password parameters..."
66
7 if [ -f /etc/pam.d/system-auth ]; then
7 if [[ -f /etc/pam.d/system-auth ]]; then
88 # Extract the line containing the password complexity parameters
99 param_line=$(grep -E 'difok=.* minlen=.* dcredit=.* ocredit=.* ucredit=.* lcredit=.* minclass=.* maxsequence=.*' /etc/pam.d/system-auth)
1010
11 if [ -n "$param_line" ]; then
11 if [[ -n "$param_line" ]]; then
1212 echo "Password complexity parameters found:"
1313 echo "$param_line"
1414 echo ""
@@ -44,12 +44,13 @@ extract_password_params() {
4444 else
4545 echo "/etc/pam.d/system-auth file not found."
4646 fi
47 return 0
4748}
4849
4950# Function to analyze /etc/login.defs
5051analyze_login_defs() {
5152 echo "Analyzing /etc/login.defs..."
52 if [ -f /etc/login.defs ]; then
53 if [[ -f /etc/login.defs ]]; then
5354 echo "Contents of /etc/login.defs:"
5455 cat /etc/login.defs
5556 echo ""
@@ -61,6 +62,7 @@ analyze_login_defs() {
6162 else
6263 echo "/etc/login.defs file not found."
6364 fi
65 return 0
6466}
6567
6668# Main script execution
os/linux/report/linux.sh +10 −3
@@ -6,10 +6,13 @@ TRIM_COMMENTS=false
66
77# Function to log section header
88log_section() {
9 local section_num="$1"
10 local section_title="$2"
911 echo -e "\n\n" >> "$REPORT_FILE"
1012 echo "==========================================" >> "$REPORT_FILE"
11 echo "# SECTION $1: $2" >> "$REPORT_FILE"
13 echo "# SECTION $section_num: $section_title" >> "$REPORT_FILE"
1214 echo "==========================================" >> "$REPORT_FILE"
15 return 0
1316}
1417
1518# Function to log file content
@@ -27,12 +30,16 @@ log_file_content() {
2730 else
2831 echo "File $FILE_PATH not found!" >> "$REPORT_FILE"
2932 fi
33 return 0
3034}
3135
3236# Function to log command output
3337log_command_output() {
34 echo "## $1" >> "$REPORT_FILE"
35 $2 >> "$REPORT_FILE" 2>&1
38 local label="$1"
39 local command="$2"
40 echo "## $label" >> "$REPORT_FILE"
41 $command >> "$REPORT_FILE" 2>&1
42 return 0
3643}
3744
3845# Check for sudo privileges
os/linux/ssh_root_login.sh +6 −6
@@ -1,14 +1,14 @@
11#!/bin/bash
22
33# Check if the script is being run as root
4if [ "$EUID" -ne 0 ]; then
5 echo "Error: This script must be run as root or with sudo."
4if [[ "$EUID" -ne 0 ]]; then
5 echo "Error: This script must be run as root or with sudo." >&2
66 exit 1
77fi
88
99# Check if the sshd_config file exists
10if [ ! -f /etc/ssh/sshd_config ]; then
11 echo "Error: /etc/ssh/sshd_config not found."
10if [[ ! -f /etc/ssh/sshd_config ]]; then
11 echo "Error: /etc/ssh/sshd_config not found." >&2
1212 exit 1
1313fi
1414
@@ -28,7 +28,7 @@ if ! echo "$permit_root_login" | grep -q "no"; then
2828 # Look for an explicitly set AuthorizedKeysFile path
2929 auth_keys_path_line=$(grep -E "^[[:space:]]*AuthorizedKeysFile" /etc/ssh/sshd_config)
3030
31 if [ -n "$auth_keys_path_line" ]; then
31 if [[ -n "$auth_keys_path_line" ]]; then
3232 # An explicit path is set. Extract the path.
3333 # This removes the 'AuthorizedKeysFile' keyword and leading/trailing whitespace.
3434 auth_keys_path=$(echo "$auth_keys_path_line" | awk '{print $2}')
@@ -45,7 +45,7 @@ if ! echo "$permit_root_login" | grep -q "no"; then
4545 fi
4646
4747 echo "Checking for file at: $actual_path"
48 if [ -f "$actual_path" ]; then
48 if [[ -f "$actual_path" ]]; then
4949 echo "[CRITICAL] Found authorized keys file for root at $actual_path"
5050 echo "Contents:"
5151 echo "----------------------------------------"
sampling/sample.html +5 −5
@@ -80,7 +80,7 @@ function handleFormSubmit(event) {
8080 // Use the custom seed if provided; otherwise draw a strong random seed and
8181 // write it back so the (reproducible) sample can always be tied to a seed.
8282 const seed = customSeedInput
83 ? parseInt(customSeedInput)
83 ? Number.parseInt(customSeedInput)
8484 : crypto.getRandomValues(new Uint32Array(1))[0] % 1000000;
8585 if (!customSeedInput) {
8686 document.getElementById('customSeed').value = seed;
@@ -89,16 +89,16 @@ function handleFormSubmit(event) {
8989}
9090
9191function generateSamples(seed) {
92 const populationSize = parseInt(document.getElementById('populationSize').value);
93 const sampleSize = parseInt(document.getElementById('sampleSize').value);
94 const replacementSize = parseInt(document.getElementById('replacementSize').value || 0);
92 const populationSize = Number.parseInt(document.getElementById('populationSize').value);
93 const sampleSize = Number.parseInt(document.getElementById('sampleSize').value);
94 const replacementSize = Number.parseInt(document.getElementById('replacementSize').value || 0);
9595 const resultsDiv = document.getElementById('results');
9696
9797 // Clear previous results
9898 resultsDiv.innerHTML = '';
9999
100100 // Validate inputs
101 if (isNaN(populationSize) || isNaN(sampleSize) || populationSize <= 0 || sampleSize <= 0) {
101 if (Number.isNaN(populationSize) || Number.isNaN(sampleSize) || populationSize <= 0 || sampleSize <= 0) {
102102 alert("Please enter valid numbers for required fields.");
103103 return;
104104 }
sampling/sampling_tool/cli.py +17 −18
@@ -23,6 +23,11 @@ from .reconciliation import build_reconciliation, build_strata_summary
2323from .reporting import RunLogger, build_methodology
2424from .validation import validate_and_prepare
2525
26# Output filenames, defined once so writer and tracker never drift.
27_POPULATION_VALIDATED_CSV = "population_validated.csv"
28_EXCLUDED_ROWS_CSV = "excluded_rows.csv"
29_DUPLICATE_IDS_CSV = "duplicate_ids.csv"
30
2631
2732def build_parser() -> ArgumentParser:
2833 parser = ArgumentParser(description="Generate documented audit samples.")
@@ -211,7 +216,6 @@ def run(options) -> Path:
211216 _write_outputs(
212217 run_dir,
213218 options,
214 source,
215219 validated,
216220 excluded_rows,
217221 duplicate_rows,
@@ -227,8 +231,6 @@ def run(options) -> Path:
227231 print(f"ERROR: {exc}", file=sys.stderr)
228232 _write_failure_outputs(
229233 run_dir,
230 options,
231 source,
232234 filtered,
233235 excluded_rows,
234236 duplicate_rows,
@@ -307,7 +309,6 @@ def add_sample_metadata(
307309def _write_outputs(
308310 run_dir: Path,
309311 options,
310 source: pd.DataFrame,
311312 validated: pd.DataFrame,
312313 excluded_rows: pd.DataFrame,
313314 duplicate_rows: pd.DataFrame,
@@ -315,17 +316,17 @@ def _write_outputs(
315316 strata_rows: list[dict[str, object]],
316317 output_files: list[str],
317318) -> None:
318 write_csv(validated, run_dir / "population_validated.csv")
319 _track(output_files, "population_validated.csv")
319 write_csv(validated, run_dir / _POPULATION_VALIDATED_CSV)
320 _track(output_files, _POPULATION_VALIDATED_CSV)
320321 if options.method in {"random", "stratified"}:
321322 write_csv(sample, run_dir / "sample.csv")
322323 _track(output_files, "sample.csv")
323324 if not excluded_rows.empty:
324 write_csv(excluded_rows, run_dir / "excluded_rows.csv")
325 _track(output_files, "excluded_rows.csv")
325 write_csv(excluded_rows, run_dir / _EXCLUDED_ROWS_CSV)
326 _track(output_files, _EXCLUDED_ROWS_CSV)
326327 if not duplicate_rows.empty:
327 write_csv(duplicate_rows, run_dir / "duplicate_ids.csv")
328 _track(output_files, "duplicate_ids.csv")
328 write_csv(duplicate_rows, run_dir / _DUPLICATE_IDS_CSV)
329 _track(output_files, _DUPLICATE_IDS_CSV)
329330 if options.method == "stratified":
330331 write_csv(build_strata_summary(strata_rows), run_dir / "strata_summary.csv")
331332 _track(output_files, "strata_summary.csv")
@@ -333,22 +334,20 @@ def _write_outputs(
333334
334335def _write_failure_outputs(
335336 run_dir: Path,
336 options,
337 source: pd.DataFrame,
338337 filtered: pd.DataFrame,
339338 excluded_rows: pd.DataFrame,
340339 duplicate_rows: pd.DataFrame,
341340 output_files: list[str],
342341) -> None:
343342 if not filtered.empty:
344 write_csv(filtered, run_dir / "population_validated.csv")
345 _track(output_files, "population_validated.csv")
343 write_csv(filtered, run_dir / _POPULATION_VALIDATED_CSV)
344 _track(output_files, _POPULATION_VALIDATED_CSV)
346345 if not excluded_rows.empty:
347 write_csv(excluded_rows, run_dir / "excluded_rows.csv")
348 _track(output_files, "excluded_rows.csv")
346 write_csv(excluded_rows, run_dir / _EXCLUDED_ROWS_CSV)
347 _track(output_files, _EXCLUDED_ROWS_CSV)
349348 if not duplicate_rows.empty:
350 write_csv(duplicate_rows, run_dir / "duplicate_ids.csv")
351 _track(output_files, "duplicate_ids.csv")
349 write_csv(duplicate_rows, run_dir / _DUPLICATE_IDS_CSV)
350 _track(output_files, _DUPLICATE_IDS_CSV)
352351
353352
354353def _concat_nonempty(frames: list[pd.DataFrame]) -> pd.DataFrame:
sampling/tests/test_validation.py +2 −1
@@ -34,8 +34,9 @@ def test_duplicate_ids_fail_by_default_and_write_duplicate_file(tmp_path):
3434 source, index=False
3535 )
3636
37 options = _options(source, tmp_path / "out")
3738 with pytest.raises(AuditSamplingError):
38 run(_options(source, tmp_path / "out"))
39 run(options)
3940
4041 run_dir = next((tmp_path / "out").glob("sample_*"))
4142 duplicates = pd.read_csv(run_dir / "duplicate_ids.csv")
tui/tests/test_aws_runner.py +2 −1
@@ -101,7 +101,8 @@ def test_session_build_failure_is_reported(tmp_path, fake_checks, monkeypatch):
101101 assert ("error", "AWS session") in kinds
102102 # Run still ends with a summary and writes the (empty) package.
103103 summary = [e for e in events if e.kind == "summary"]
104 assert summary and summary[0].count == 0
104 assert summary
105 assert summary[0].count == 0
105106 assert sections == []
106107 assert os.path.exists(tmp_path / "summary.txt")
107108