audit-labs/audit-tools

A collection of scripts, queries, and other goodies you can use in an audit. audit automation compliance evidence scripts

Commit f484a105f0

f484a105f060a6b6a97eda3a5b466dd575df76d4

parent: f54b04a292

Unsigned

cmc <hello@cleberg.net> · 2025-05-29 16:41 UTC
committer: <noreply@github.com>

feat: add stratified sampling script (#12)

* feat: add stratified sampling script

* Commit from GitHub Actions (Ruff)

---------

Co-authored-by: github-actions <41898282+github-actions[bot]@users.noreply.github.com>

Layout: unified · split

.gitignore +2 −1
@@ -1,2 +1,3 @@
1venv/ 1.venv
2venv
2readme.html 3readme.html
sampling/stratified_sample.py added +62
@@ -0,0 +1,62 @@
1# Import packages
2import pandas as pd
3import math
4
5# Load data
6df = pd.read_csv("FILENAME_GOES_HERE.csv")
7
8# ALTERNATIVE: If you use Excel, use this instead. Supports xls, xlsx, xlsm,
9# xlsb, odf, ods and odt file extensions.
10# df = pd.read_excel("FILENAME_GOES_HERE.xlsx")
11
12# Print totals prior to sampling
13print("Dataframe size (rows, columns):", df.shape)
14
15# User-defined parameters
16SAMPLE_SIZE = 25
17STRATIFY_COLUMN = "Category" # <- Change this to your column name
18
19# Define stratum proportions (as fractions)
20# Example: if you have categories A, B, and C
21stratum_proportions = {"A": 0.4, "B": 0.4, "C": 0.2}
22
23# Validate proportions sum to 1
24if not math.isclose(sum(stratum_proportions.values()), 1.0):
25 raise ValueError("Stratum proportions must sum to 1.")
26
27# Check that all strata exist in the data
28missing_strata = set(stratum_proportions.keys()) - set(df[STRATIFY_COLUMN].unique())
29if missing_strata:
30 raise ValueError(
31 f"Strata {missing_strata} not found in column '{STRATIFY_COLUMN}'."
32 )
33
34# Perform stratified sampling
35samples = []
36for stratum, proportion in stratum_proportions.items():
37 stratum_df = df[df[STRATIFY_COLUMN] == stratum]
38 n_samples = math.floor(SAMPLE_SIZE * proportion)
39 if n_samples > len(stratum_df):
40 raise ValueError(
41 f"Not enough data in stratum '{stratum}' to sample {n_samples} rows."
42 )
43 stratum_sample = stratum_df.sample(n=n_samples, random_state=42)
44 samples.append(stratum_sample)
45
46# Combine all stratum samples into one DataFrame
47final_sample = pd.concat(samples).reset_index()
48
49# If needed, randomly sample extra rows to fill any rounding gap
50current_sample_size = len(final_sample)
51if current_sample_size < SAMPLE_SIZE:
52 remaining = SAMPLE_SIZE - current_sample_size
53 remaining_sample = df.sample(n=remaining, random_state=42)
54 final_sample = pd.concat([final_sample, remaining_sample])
55
56# Print sample results
57print("Final sample size:", final_sample.shape[0])
58print("Sample breakdown by stratum:\n", final_sample[STRATIFY_COLUMN].value_counts())
59print("\nSample:\n", final_sample)
60
61# Optionally, save the sample to a new CSV
62# final_sample.to_csv("sample_output.csv", index=False)