Commit 64d53350b0
Unsigned
Layout: unified · split
.gitignore +2
| @@ -4,8 +4,10 @@ readme.html | |||
| 4 | 4 | ||
| 5 | # Python | 5 | # Python |
| 6 | __pycache__/ | 6 | __pycache__/ |
| 7 | **/__pycache__/ | ||
| 7 | *.py[cod] | 8 | *.py[cod] |
| 8 | 9 | ||
| 9 | # Audit output | 10 | # Audit output |
| 10 | applications/github/output/ | 11 | applications/github/output/ |
| 11 | output/ | 12 | output/ |
| 13 | **/output/ | ||
README.org +1 −1
| @@ -28,7 +28,7 @@ specific audit environments. | |||
| 28 | *Clone the Repository* | 28 | *Clone the Repository* |
| 29 | 29 | ||
| 30 | #+begin_src bash | 30 | #+begin_src bash |
| 31 | git clone https://github.com/audit-lab/audit-toolss | 31 | git clone https://github.com/audit-labs/audit-tools |
| 32 | cd audit-tools | 32 | cd audit-tools |
| 33 | #+end_src | 33 | #+end_src |
| 34 | 34 | ||
requirements.txt +4
| @@ -1,4 +1,8 @@ | |||
| 1 | pandas | 1 | pandas |
| 2 | openpyxl | ||
| 3 | xlrd | ||
| 4 | PyYAML | ||
| 5 | pytest | ||
| 2 | requests | 6 | requests |
| 3 | dash | 7 | dash |
| 4 | plotly | 8 | plotly |
sampling/README.md +102 −23
| @@ -1,31 +1,110 @@ | |||
| 1 | # `sample.py` | 1 | # Audit Sampling Tools |
| 2 | 2 | ||
| 3 | ``` bash | 3 | This directory contains simple audit sampling utilities. The production-ready |
| 4 | python ./sample.py | 4 | CLI is `audit_sample.py`; the older `sample.py`, `sample.html`, and |
| 5 | `stratified_sample.py` examples remain for compatibility and learning. | ||
| 6 | |||
| 7 | ## Purpose | ||
| 8 | |||
| 9 | `audit_sample.py` generates reproducible, documented samples from CSV and Excel | ||
| 10 | populations. Each run writes the selected sample, validated population, | ||
| 11 | reconciliation files, methodology notes, a manifest, and a log suitable for an | ||
| 12 | audit workpaper package. | ||
| 13 | |||
| 14 | ## Installation | ||
| 15 | |||
| 16 | Install the repository requirements: | ||
| 17 | |||
| 18 | ```bash | ||
| 19 | pip install -r requirements.txt | ||
| 20 | ``` | ||
| 21 | |||
| 22 | Supported input formats are `.csv`, `.xlsx`, `.xls`, and `.xlsm`. | ||
| 23 | |||
| 24 | ## Random Sampling Example | ||
| 25 | |||
| 26 | ```bash | ||
| 27 | python sampling/audit_sample.py \ | ||
| 28 | --input sampling/examples/users_population.csv \ | ||
| 29 | --id-column "User ID" \ | ||
| 30 | --method random \ | ||
| 31 | --sample-size 5 \ | ||
| 32 | --seed 20260707 \ | ||
| 33 | --out ./output | ||
| 5 | ``` | 34 | ``` |
| 6 | 35 | ||
| 7 | ``` text | 36 | ## Stratified Counts Example |
| 8 | Dataframe size (rows, columns): (100, 9) | 37 | |
| 9 | Sample size: 5 | 38 | ```bash |
| 10 | Sample: | 39 | python sampling/audit_sample.py \ |
| 11 | Index Organization Id ... Industry Number of employees | 40 | --input sampling/examples/changes_population.csv \ |
| 12 | 79 80 cBa7EFe5D05Adaf ... Online Publishing 7805 | 41 | --id-column "Change ID" \ |
| 13 | 97 98 E7df80C60Abd7f9 ... Broadcast Media 236 | 42 | --method stratified \ |
| 14 | 3 4 2bFC1Be8a4ce42f ... Automotive 921 | 43 | --stratify-column "Change Type" \ |
| 15 | 42 43 A2D89Ab9bCcAd4e ... Capital Markets / Hedge Fund / Private Equity 3816 | 44 | --strata-counts "Normal=3,Emergency=2,Standard=2" \ |
| 16 | 70 71 32BB9Ff4d939788 ... Wireless 6146 | 45 | --seed 20260707 \ |
| 17 | 46 | --out ./output | |
| 18 | [5 rows x 9 columns] | ||
| 19 | ``` | 47 | ``` |
| 20 | 48 | ||
| 21 | # `sample.html` | 49 | ## Stratified Proportions Example |
| 50 | |||
| 51 | ```bash | ||
| 52 | python sampling/audit_sample.py \ | ||
| 53 | --input sampling/examples/changes_population.csv \ | ||
| 54 | --id-column "Change ID" \ | ||
| 55 | --method stratified \ | ||
| 56 | --stratify-column "Change Type" \ | ||
| 57 | --strata-proportions "Normal=0.50,Emergency=0.25,Standard=0.25" \ | ||
| 58 | --sample-size 8 \ | ||
| 59 | --seed 20260707 \ | ||
| 60 | --out ./output | ||
| 61 | ``` | ||
| 62 | |||
| 63 | ## Validate-Only Example | ||
| 64 | |||
| 65 | ```bash | ||
| 66 | python sampling/audit_sample.py \ | ||
| 67 | --input sampling/examples/changes_population.csv \ | ||
| 68 | --id-column "Change ID" \ | ||
| 69 | --method validate-only \ | ||
| 70 | --filter "Status=Closed" \ | ||
| 71 | --out ./output | ||
| 72 | ``` | ||
| 73 | |||
| 74 | YAML config files are also supported. CLI arguments override config values: | ||
| 75 | |||
| 76 | ```bash | ||
| 77 | python sampling/audit_sample.py --config sampling/examples/stratified_config.yml | ||
| 78 | ``` | ||
| 79 | |||
| 80 | ## Output Files | ||
| 81 | |||
| 82 | Each run creates `sample_<YYYY-MM-DD>_<HHMMSS>` under the selected output | ||
| 83 | directory. | ||
| 84 | |||
| 85 | - `sample.csv`: selected sample rows for random and stratified runs. | ||
| 86 | - `population_validated.csv`: population after filters, blank-ID handling, and | ||
| 87 | dedupe handling. | ||
| 88 | - `population_reconciliation.csv`: row-count tie-out metrics. | ||
| 89 | - `excluded_rows.csv`: rows removed by filters, blank-ID exclusion, or dedupe. | ||
| 90 | - `duplicate_ids.csv`: duplicate ID rows when duplicates are identified. | ||
| 91 | - `strata_summary.csv`: requested and actual counts for stratified runs. | ||
| 92 | - `methodology.txt`: audit workpaper narrative. | ||
| 93 | - `manifest.json`: machine-readable run metadata, input hash, options, and | ||
| 94 | output list. | ||
| 95 | - `run.log`: start time, warnings, errors, output folder, and status. | ||
| 96 | |||
| 97 | ## Reproducibility | ||
| 98 | |||
| 99 | Provide `--seed` to make row selection reproducible for the same input and | ||
| 100 | options. If no seed is provided for sampling, the tool generates one, prints a | ||
| 101 | warning, and records the generated seed in `manifest.json` and | ||
| 102 | `methodology.txt`. | ||
| 22 | 103 | ||
| 23 | This is an interactive web page that allows users to submit their | 104 | The tool samples without replacement. Stratified sampling derives each stratum |
| 24 | population size, sample size(s), and generate a psuedo-random sample | 105 | seed from the base seed by adding the stratum index. |
| 25 | list of numbers to use when sampling against their population. | ||
| 26 | 106 | ||
| 27 | Samples can be re-generated and validated using the seed numbers | 107 | ## Limitations |
| 28 | provided during the original generation. | ||
| 29 | 108 | ||
| 30 | <span class="spurious-link" | 109 | Filters are exact matches in `Column=Value` form only. The tool does not yet |
| 31 | target="sample-html.png">*sample-html.png*</span> | 110 | perform monetary-unit sampling or statistical sample-size calculation. |
sampling/audit_sample.py added +15
| @@ -0,0 +1,15 @@ | |||
| 1 | #!/usr/bin/env python3 | ||
| 2 | """Command-line entrypoint for the audit sampling tool.""" | ||
| 3 | |||
| 4 | from pathlib import Path | ||
| 5 | import sys | ||
| 6 | |||
| 7 | |||
| 8 | if __package__ is None or __package__ == "": | ||
| 9 | sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) | ||
| 10 | |||
| 11 | from sampling.sampling_tool.cli import main | ||
| 12 | |||
| 13 | |||
| 14 | if __name__ == "__main__": | ||
| 15 | raise SystemExit(main()) | ||
sampling/examples/changes_population.csv added +15
| @@ -0,0 +1,15 @@ | |||
| 1 | Change ID,Change Type,Risk Rating,Status,In Scope | ||
| 2 | CHG001,Normal,High,Closed,Yes | ||
| 3 | CHG002,Normal,Medium,Closed,Yes | ||
| 4 | CHG003,Normal,Low,Open,No | ||
| 5 | CHG004,Normal,Medium,Closed,Yes | ||
| 6 | CHG005,Normal,High,Closed,Yes | ||
| 7 | CHG006,Emergency,High,Closed,Yes | ||
| 8 | CHG007,Emergency,High,Closed,Yes | ||
| 9 | CHG008,Emergency,Medium,Open,Yes | ||
| 10 | CHG009,Emergency,Low,Closed,No | ||
| 11 | CHG010,Standard,Low,Closed,Yes | ||
| 12 | CHG011,Standard,Medium,Closed,Yes | ||
| 13 | CHG012,Standard,Low,Open,No | ||
| 14 | CHG013,Standard,High,Closed,Yes | ||
| 15 | CHG014,Standard,Medium,Closed,Yes | ||
sampling/examples/stratified_config.yml added +7
| @@ -0,0 +1,7 @@ | |||
| 1 | input: sampling/examples/changes_population.csv | ||
| 2 | id_column: Change ID | ||
| 3 | method: stratified | ||
| 4 | stratify_column: Change Type | ||
| 5 | strata_counts: Normal=3,Emergency=2,Standard=2 | ||
| 6 | seed: 20260707 | ||
| 7 | out: ./output | ||
sampling/examples/users_population.csv added +13
| @@ -0,0 +1,13 @@ | |||
| 1 | User ID,Name,Role,Status,In Scope | ||
| 2 | U001,Avery Chen,Administrator,Active,Yes | ||
| 3 | U002,Blake Rivera,Developer,Active,Yes | ||
| 4 | U003,Casey Morgan,Analyst,Inactive,No | ||
| 5 | U004,Drew Patel,Reviewer,Active,Yes | ||
| 6 | U005,Emerson Lee,Developer,Active,Yes | ||
| 7 | U006,Finley Brooks,Analyst,Inactive,No | ||
| 8 | U007,Gray Taylor,Administrator,Active,Yes | ||
| 9 | U008,Harper Smith,Developer,Active,Yes | ||
| 10 | U009,Indigo Clark,Reviewer,Active,No | ||
| 11 | U010,Jordan Kim,Analyst,Active,Yes | ||
| 12 | U011,Kai Johnson,Developer,Inactive,No | ||
| 13 | U012,Logan Davis,Reviewer,Active,Yes | ||
sampling/sampling_tool/__init__.py added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | """Reusable audit sampling helpers.""" | ||
| 2 | |||
| 3 | __version__ = "1.0.0" | ||
sampling/sampling_tool/cli.py added +367
| @@ -0,0 +1,367 @@ | |||
| 1 | """CLI orchestration for the audit sampling tool.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | from argparse import ArgumentParser, Namespace | ||
| 6 | from datetime import datetime, timezone | ||
| 7 | from pathlib import Path | ||
| 8 | from types import SimpleNamespace | ||
| 9 | import sys | ||
| 10 | |||
| 11 | import pandas as pd | ||
| 12 | |||
| 13 | from .filters import apply_filters, parse_filters | ||
| 14 | from .io import AuditSamplingError, load_population, sha256_file, write_csv | ||
| 15 | from .manifest import build_manifest, write_manifest | ||
| 16 | from .methods import ( | ||
| 17 | ensure_seed, | ||
| 18 | largest_remainder_allocation, | ||
| 19 | random_sample, | ||
| 20 | stratified_sample, | ||
| 21 | ) | ||
| 22 | from .reconciliation import build_reconciliation, build_strata_summary | ||
| 23 | from .reporting import RunLogger, build_methodology | ||
| 24 | from .validation import validate_and_prepare | ||
| 25 | |||
| 26 | |||
| 27 | def build_parser() -> ArgumentParser: | ||
| 28 | parser = ArgumentParser(description="Generate documented audit samples.") | ||
| 29 | parser.add_argument("--input") | ||
| 30 | parser.add_argument("--sheet") | ||
| 31 | parser.add_argument("--id-column") | ||
| 32 | parser.add_argument("--method", choices=["random", "stratified", "validate-only"]) | ||
| 33 | parser.add_argument("--sample-size", type=int) | ||
| 34 | parser.add_argument("--stratify-column") | ||
| 35 | parser.add_argument("--strata-counts") | ||
| 36 | parser.add_argument("--strata-proportions") | ||
| 37 | parser.add_argument("--seed", type=int) | ||
| 38 | parser.add_argument("--out") | ||
| 39 | parser.add_argument("--exclude-blank-id", action="store_true", default=None) | ||
| 40 | parser.add_argument("--dedupe-id", choices=["fail", "first", "last"]) | ||
| 41 | parser.add_argument("--filter", action="append", dest="filter_values") | ||
| 42 | parser.add_argument("--config") | ||
| 43 | parser.add_argument("--allow-shortfall", action="store_true", default=None) | ||
| 44 | return parser | ||
| 45 | |||
| 46 | |||
| 47 | def load_config(path: str | None) -> dict[str, object]: | ||
| 48 | if not path: | ||
| 49 | return {} | ||
| 50 | try: | ||
| 51 | import yaml | ||
| 52 | except ImportError as exc: | ||
| 53 | raise AuditSamplingError( | ||
| 54 | "YAML config support requires PyYAML. Install requirements.txt." | ||
| 55 | ) from exc | ||
| 56 | |||
| 57 | config_path = Path(path) | ||
| 58 | with config_path.open("r", encoding="utf-8") as handle: | ||
| 59 | data = yaml.safe_load(handle) or {} | ||
| 60 | if not isinstance(data, dict): | ||
| 61 | raise AuditSamplingError("Config file must contain a YAML mapping.") | ||
| 62 | return data | ||
| 63 | |||
| 64 | |||
| 65 | def merge_options(args: Namespace, config: dict[str, object]) -> SimpleNamespace: | ||
| 66 | mapping = { | ||
| 67 | "input": "input", | ||
| 68 | "sheet": "sheet", | ||
| 69 | "id_column": "id_column", | ||
| 70 | "method": "method", | ||
| 71 | "sample_size": "sample_size", | ||
| 72 | "stratify_column": "stratify_column", | ||
| 73 | "strata_counts": "strata_counts", | ||
| 74 | "strata_proportions": "strata_proportions", | ||
| 75 | "seed": "seed", | ||
| 76 | "out": "out", | ||
| 77 | "exclude_blank_id": "exclude_blank_id", | ||
| 78 | "dedupe_id": "dedupe_id", | ||
| 79 | "allow_shortfall": "allow_shortfall", | ||
| 80 | } | ||
| 81 | merged: dict[str, object] = {} | ||
| 82 | for attr, key in mapping.items(): | ||
| 83 | cli_value = getattr(args, attr) | ||
| 84 | merged[attr] = cli_value if cli_value is not None else config.get(key) | ||
| 85 | |||
| 86 | config_filters = config.get("filters", {}) | ||
| 87 | if isinstance(config_filters, list): | ||
| 88 | config_filters = parse_filters(config_filters) | ||
| 89 | if not isinstance(config_filters, dict): | ||
| 90 | raise AuditSamplingError("Config filters must be a mapping or list.") | ||
| 91 | cli_filters = parse_filters(args.filter_values) | ||
| 92 | merged["filters"] = {**config_filters, **cli_filters} | ||
| 93 | |||
| 94 | merged["out"] = merged["out"] or "./output" | ||
| 95 | merged["dedupe_id"] = merged["dedupe_id"] or "fail" | ||
| 96 | merged["exclude_blank_id"] = bool(merged["exclude_blank_id"]) | ||
| 97 | merged["allow_shortfall"] = bool(merged["allow_shortfall"]) | ||
| 98 | if merged["input"] is None: | ||
| 99 | raise AuditSamplingError("--input is required unless provided by --config.") | ||
| 100 | if merged["method"] is None: | ||
| 101 | raise AuditSamplingError("--method is required unless provided by --config.") | ||
| 102 | if merged["method"] not in {"random", "stratified", "validate-only"}: | ||
| 103 | raise AuditSamplingError("--method must be random, stratified, or validate-only.") | ||
| 104 | if merged["sample_size"] is not None: | ||
| 105 | merged["sample_size"] = int(merged["sample_size"]) | ||
| 106 | if merged["seed"] is not None: | ||
| 107 | merged["seed"] = int(merged["seed"]) | ||
| 108 | return SimpleNamespace(**merged) | ||
| 109 | |||
| 110 | |||
| 111 | def create_run_dir(out_dir: Path, now: datetime) -> Path: | ||
| 112 | run_dir = out_dir / f"sample_{now.strftime('%Y-%m-%d_%H%M%S')}" | ||
| 113 | suffix = 1 | ||
| 114 | while True: | ||
| 115 | candidate = run_dir if suffix == 1 else out_dir / f"{run_dir.name}_{suffix}" | ||
| 116 | try: | ||
| 117 | candidate.mkdir(parents=True, exist_ok=False) | ||
| 118 | return candidate | ||
| 119 | except FileExistsError: | ||
| 120 | suffix += 1 | ||
| 121 | |||
| 122 | |||
| 123 | def main(argv: list[str] | None = None) -> int: | ||
| 124 | parser = build_parser() | ||
| 125 | args = parser.parse_args(argv) | ||
| 126 | try: | ||
| 127 | options = merge_options(args, load_config(args.config)) | ||
| 128 | run(options) | ||
| 129 | return 0 | ||
| 130 | except AuditSamplingError as exc: | ||
| 131 | print(f"ERROR: {exc}", file=sys.stderr) | ||
| 132 | return 1 | ||
| 133 | |||
| 134 | |||
| 135 | def run(options) -> Path: | ||
| 136 | now = datetime.now(timezone.utc) | ||
| 137 | timestamp = now.replace(microsecond=0).isoformat().replace("+00:00", "Z") | ||
| 138 | run_dir = create_run_dir(Path(options.out), now) | ||
| 139 | logger = RunLogger() | ||
| 140 | logger.log(f"start timestamp: {timestamp}") | ||
| 141 | logger.log(f"input path: {options.input}") | ||
| 142 | logger.log(f"method: {options.method}") | ||
| 143 | logger.log(f"output folder: {run_dir}") | ||
| 144 | print(f"Writing audit sample package to {run_dir}") | ||
| 145 | |||
| 146 | output_files: list[str] = [] | ||
| 147 | input_path = Path(options.input) | ||
| 148 | input_hash = "" | ||
| 149 | source = pd.DataFrame() | ||
| 150 | filtered = pd.DataFrame() | ||
| 151 | validated = pd.DataFrame() | ||
| 152 | excluded_rows = pd.DataFrame() | ||
| 153 | duplicate_rows = pd.DataFrame() | ||
| 154 | sample = pd.DataFrame() | ||
| 155 | strata_rows: list[dict[str, object]] = [] | ||
| 156 | random_seed: int | None = options.seed | ||
| 157 | effective_id_column = None | ||
| 158 | id_column_omitted = False | ||
| 159 | blank_id_count = 0 | ||
| 160 | duplicate_id_count = 0 | ||
| 161 | |||
| 162 | try: | ||
| 163 | input_hash = sha256_file(input_path) | ||
| 164 | source = load_population(input_path, options.sheet) | ||
| 165 | filtered, filter_excluded = apply_filters(source, options.filters) | ||
| 166 | validated = filtered.copy() | ||
| 167 | excluded_rows = filter_excluded.copy() | ||
| 168 | validation = validate_and_prepare(filtered, options) | ||
| 169 | validated = validation.population | ||
| 170 | excluded_rows = _concat_nonempty([filter_excluded, validation.excluded_rows]) | ||
| 171 | duplicate_rows = validation.duplicate_rows | ||
| 172 | effective_id_column = validation.effective_id_column | ||
| 173 | id_column_omitted = validation.id_column_omitted | ||
| 174 | blank_id_count = validation.blank_id_count | ||
| 175 | duplicate_id_count = validation.duplicate_id_count | ||
| 176 | for warning in validation.warnings: | ||
| 177 | print(f"WARNING: {warning}") | ||
| 178 | logger.warning(warning) | ||
| 179 | |||
| 180 | if options.method in {"random", "stratified"}: | ||
| 181 | random_seed, generated = ensure_seed(options.seed) | ||
| 182 | if generated: | ||
| 183 | warning = f"No seed provided; generated seed {random_seed}." | ||
| 184 | print(f"WARNING: {warning}") | ||
| 185 | logger.warning(warning) | ||
| 186 | |||
| 187 | if options.method == "random": | ||
| 188 | sample = random_sample(validated, options.sample_size, random_seed) | ||
| 189 | sample = add_sample_metadata(sample, options, random_seed, timestamp, None) | ||
| 190 | elif options.method == "stratified": | ||
| 191 | counts = validation.strata_counts | ||
| 192 | if validation.strata_proportions: | ||
| 193 | counts = largest_remainder_allocation( | ||
| 194 | options.sample_size, validation.strata_proportions | ||
| 195 | ) | ||
| 196 | sampled, strata_rows = stratified_sample( | ||
| 197 | validated, | ||
| 198 | options.stratify_column, | ||
| 199 | counts, | ||
| 200 | random_seed, | ||
| 201 | options.allow_shortfall, | ||
| 202 | ) | ||
| 203 | sample = add_sample_metadata( | ||
| 204 | sampled, options, random_seed, timestamp, options.stratify_column | ||
| 205 | ) | ||
| 206 | |||
| 207 | _write_outputs( | ||
| 208 | run_dir, | ||
| 209 | options, | ||
| 210 | source, | ||
| 211 | validated, | ||
| 212 | excluded_rows, | ||
| 213 | duplicate_rows, | ||
| 214 | sample, | ||
| 215 | strata_rows, | ||
| 216 | output_files, | ||
| 217 | ) | ||
| 218 | logger.log("final status: success") | ||
| 219 | print("Audit sampling run completed.") | ||
| 220 | except AuditSamplingError as exc: | ||
| 221 | duplicate_rows = exc.artifacts.get("duplicate_rows", duplicate_rows) | ||
| 222 | logger.error(str(exc)) | ||
| 223 | print(f"ERROR: {exc}", file=sys.stderr) | ||
| 224 | _write_failure_outputs( | ||
| 225 | run_dir, | ||
| 226 | options, | ||
| 227 | source, | ||
| 228 | filtered, | ||
| 229 | excluded_rows, | ||
| 230 | duplicate_rows, | ||
| 231 | output_files, | ||
| 232 | ) | ||
| 233 | logger.log("final status: failed") | ||
| 234 | raise | ||
| 235 | finally: | ||
| 236 | reconciliation = build_reconciliation( | ||
| 237 | source_rows=len(source), | ||
| 238 | blank_id_count=blank_id_count, | ||
| 239 | duplicate_id_count=duplicate_id_count or len(duplicate_rows), | ||
| 240 | excluded_rows=len(excluded_rows), | ||
| 241 | validated_rows=len(validated), | ||
| 242 | requested_sample_size=_requested_sample_size(options, strata_rows), | ||
| 243 | final_sample_size=len(sample), | ||
| 244 | ) | ||
| 245 | write_csv(reconciliation, run_dir / "population_reconciliation.csv") | ||
| 246 | _track(output_files, "population_reconciliation.csv") | ||
| 247 | methodology = build_methodology( | ||
| 248 | input_file=input_path, | ||
| 249 | input_sheet=options.sheet, | ||
| 250 | input_sha256=input_hash, | ||
| 251 | source_row_count=len(source), | ||
| 252 | options=options, | ||
| 253 | effective_id_column=effective_id_column, | ||
| 254 | id_column_omitted=id_column_omitted, | ||
| 255 | duplicate_id_count=duplicate_id_count or len(duplicate_rows), | ||
| 256 | blank_id_count=blank_id_count, | ||
| 257 | sample_size_actual=len(sample), | ||
| 258 | random_seed=random_seed, | ||
| 259 | strata_summary=strata_rows, | ||
| 260 | ) | ||
| 261 | (run_dir / "methodology.txt").write_text(methodology) | ||
| 262 | _track(output_files, "methodology.txt") | ||
| 263 | manifest = build_manifest( | ||
| 264 | run_timestamp_utc=timestamp, | ||
| 265 | input_file=input_path, | ||
| 266 | input_sha256=input_hash, | ||
| 267 | input_sheet=options.sheet, | ||
| 268 | options=options, | ||
| 269 | effective_id_column=effective_id_column, | ||
| 270 | id_column_omitted=id_column_omitted, | ||
| 271 | sample_size_actual=len(sample), | ||
| 272 | random_seed=random_seed, | ||
| 273 | source_row_count=len(source), | ||
| 274 | validated_population_count=len(validated), | ||
| 275 | excluded_row_count=len(excluded_rows), | ||
| 276 | duplicate_id_count=duplicate_id_count or len(duplicate_rows), | ||
| 277 | blank_id_count=blank_id_count, | ||
| 278 | output_files=sorted(output_files + ["run.log", "manifest.json"]), | ||
| 279 | ) | ||
| 280 | write_manifest(manifest, run_dir / "manifest.json") | ||
| 281 | logger.write(run_dir / "run.log") | ||
| 282 | return run_dir | ||
| 283 | |||
| 284 | |||
| 285 | def add_sample_metadata( | ||
| 286 | sample: pd.DataFrame, | ||
| 287 | options, | ||
| 288 | seed: int, | ||
| 289 | timestamp: str, | ||
| 290 | stratum_column: str | None, | ||
| 291 | ) -> pd.DataFrame: | ||
| 292 | source_columns = [column for column in sample.columns if column != "_audit_row_id"] | ||
| 293 | output = sample[source_columns].copy() | ||
| 294 | output.insert(0, "_selected_at_utc", timestamp) | ||
| 295 | output.insert(0, "_random_seed", seed) | ||
| 296 | output.insert(0, "_stratum", output[stratum_column] if stratum_column else "") | ||
| 297 | output.insert(0, "_selection_method", options.method) | ||
| 298 | output.insert(0, "_source_row_number", output.pop("_source_row_number")) | ||
| 299 | output.insert(0, "_sample_id", range(1, len(output) + 1)) | ||
| 300 | return output | ||
| 301 | |||
| 302 | |||
| 303 | def _write_outputs( | ||
| 304 | run_dir: Path, | ||
| 305 | options, | ||
| 306 | source: pd.DataFrame, | ||
| 307 | validated: pd.DataFrame, | ||
| 308 | excluded_rows: pd.DataFrame, | ||
| 309 | duplicate_rows: pd.DataFrame, | ||
| 310 | sample: pd.DataFrame, | ||
| 311 | strata_rows: list[dict[str, object]], | ||
| 312 | output_files: list[str], | ||
| 313 | ) -> None: | ||
| 314 | write_csv(validated, run_dir / "population_validated.csv") | ||
| 315 | _track(output_files, "population_validated.csv") | ||
| 316 | if options.method in {"random", "stratified"}: | ||
| 317 | write_csv(sample, run_dir / "sample.csv") | ||
| 318 | _track(output_files, "sample.csv") | ||
| 319 | if not excluded_rows.empty: | ||
| 320 | write_csv(excluded_rows, run_dir / "excluded_rows.csv") | ||
| 321 | _track(output_files, "excluded_rows.csv") | ||
| 322 | if not duplicate_rows.empty: | ||
| 323 | write_csv(duplicate_rows, run_dir / "duplicate_ids.csv") | ||
| 324 | _track(output_files, "duplicate_ids.csv") | ||
| 325 | if options.method == "stratified": | ||
| 326 | write_csv(build_strata_summary(strata_rows), run_dir / "strata_summary.csv") | ||
| 327 | _track(output_files, "strata_summary.csv") | ||
| 328 | |||
| 329 | |||
| 330 | def _write_failure_outputs( | ||
| 331 | run_dir: Path, | ||
| 332 | options, | ||
| 333 | source: pd.DataFrame, | ||
| 334 | filtered: pd.DataFrame, | ||
| 335 | excluded_rows: pd.DataFrame, | ||
| 336 | duplicate_rows: pd.DataFrame, | ||
| 337 | output_files: list[str], | ||
| 338 | ) -> None: | ||
| 339 | if not filtered.empty: | ||
| 340 | write_csv(filtered, run_dir / "population_validated.csv") | ||
| 341 | _track(output_files, "population_validated.csv") | ||
| 342 | if not excluded_rows.empty: | ||
| 343 | write_csv(excluded_rows, run_dir / "excluded_rows.csv") | ||
| 344 | _track(output_files, "excluded_rows.csv") | ||
| 345 | if not duplicate_rows.empty: | ||
| 346 | write_csv(duplicate_rows, run_dir / "duplicate_ids.csv") | ||
| 347 | _track(output_files, "duplicate_ids.csv") | ||
| 348 | |||
| 349 | |||
| 350 | def _concat_nonempty(frames: list[pd.DataFrame]) -> pd.DataFrame: | ||
| 351 | nonempty = [frame for frame in frames if frame is not None and not frame.empty] | ||
| 352 | if not nonempty: | ||
| 353 | return pd.DataFrame() | ||
| 354 | return pd.concat(nonempty, ignore_index=True) | ||
| 355 | |||
| 356 | |||
| 357 | def _track(output_files: list[str], filename: str) -> None: | ||
| 358 | if filename not in output_files: | ||
| 359 | output_files.append(filename) | ||
| 360 | |||
| 361 | |||
| 362 | def _requested_sample_size(options, strata_rows: list[dict[str, object]]) -> int | None: | ||
| 363 | if options.sample_size is not None: | ||
| 364 | return options.sample_size | ||
| 365 | if strata_rows: | ||
| 366 | return int(sum(row["Requested Sample Count"] for row in strata_rows)) | ||
| 367 | return None | ||
sampling/sampling_tool/filters.py added +44
| @@ -0,0 +1,44 @@ | |||
| 1 | """Exact-match filter parsing and application.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | import pandas as pd | ||
| 6 | |||
| 7 | from .io import AuditSamplingError | ||
| 8 | |||
| 9 | |||
| 10 | def parse_filters(filter_values: list[str] | None) -> dict[str, str]: | ||
| 11 | parsed: dict[str, str] = {} | ||
| 12 | for value in filter_values or []: | ||
| 13 | if "=" not in value: | ||
| 14 | raise AuditSamplingError( | ||
| 15 | f"Invalid filter '{value}'. Expected format: Column=Value" | ||
| 16 | ) | ||
| 17 | column, expected = value.split("=", 1) | ||
| 18 | column = column.strip() | ||
| 19 | if not column: | ||
| 20 | raise AuditSamplingError( | ||
| 21 | f"Invalid filter '{value}'. Filter column cannot be blank." | ||
| 22 | ) | ||
| 23 | parsed[column] = expected.strip() | ||
| 24 | return parsed | ||
| 25 | |||
| 26 | |||
| 27 | def apply_filters( | ||
| 28 | population: pd.DataFrame, filters: dict[str, str] | ||
| 29 | ) -> tuple[pd.DataFrame, pd.DataFrame]: | ||
| 30 | if not filters: | ||
| 31 | return population.copy(), population.iloc[0:0].copy() | ||
| 32 | |||
| 33 | missing = [column for column in filters if column not in population.columns] | ||
| 34 | if missing: | ||
| 35 | raise AuditSamplingError(f"Filter column(s) not found: {', '.join(missing)}") | ||
| 36 | |||
| 37 | keep_mask = pd.Series(True, index=population.index) | ||
| 38 | for column, expected in filters.items(): | ||
| 39 | keep_mask &= population[column].astype("string").fillna("") == expected | ||
| 40 | |||
| 41 | excluded = population.loc[~keep_mask].copy() | ||
| 42 | if not excluded.empty: | ||
| 43 | excluded["_exclusion_reason"] = "Filtered out" | ||
| 44 | return population.loc[keep_mask].copy(), excluded | ||
sampling/sampling_tool/io.py added +60
| @@ -0,0 +1,60 @@ | |||
| 1 | """Input and output helpers for audit sampling.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | from pathlib import Path | ||
| 6 | import hashlib | ||
| 7 | |||
| 8 | import pandas as pd | ||
| 9 | |||
| 10 | |||
| 11 | SUPPORTED_EXCEL_SUFFIXES = {".xlsx", ".xls", ".xlsm"} | ||
| 12 | |||
| 13 | |||
| 14 | class AuditSamplingError(Exception): | ||
| 15 | """Raised when the sampling request cannot be completed.""" | ||
| 16 | |||
| 17 | def __init__(self, message: str, **artifacts) -> None: | ||
| 18 | super().__init__(message) | ||
| 19 | self.artifacts = artifacts | ||
| 20 | |||
| 21 | |||
| 22 | def sha256_file(path: Path) -> str: | ||
| 23 | digest = hashlib.sha256() | ||
| 24 | with path.open("rb") as handle: | ||
| 25 | for chunk in iter(lambda: handle.read(1024 * 1024), b""): | ||
| 26 | digest.update(chunk) | ||
| 27 | return digest.hexdigest() | ||
| 28 | |||
| 29 | |||
| 30 | def load_population(input_path: Path, sheet: str | None = None) -> pd.DataFrame: | ||
| 31 | if not input_path.exists(): | ||
| 32 | raise AuditSamplingError(f"Input file does not exist: {input_path}") | ||
| 33 | |||
| 34 | suffix = input_path.suffix.lower() | ||
| 35 | if suffix == ".csv": | ||
| 36 | frame = pd.read_csv(input_path) | ||
| 37 | elif suffix in SUPPORTED_EXCEL_SUFFIXES: | ||
| 38 | excel = pd.ExcelFile(input_path) | ||
| 39 | if sheet is None: | ||
| 40 | if len(excel.sheet_names) != 1: | ||
| 41 | names = ", ".join(excel.sheet_names) | ||
| 42 | raise AuditSamplingError( | ||
| 43 | "Excel workbook has multiple sheets. Provide --sheet. " | ||
| 44 | f"Available sheets: {names}" | ||
| 45 | ) | ||
| 46 | sheet = excel.sheet_names[0] | ||
| 47 | frame = pd.read_excel(input_path, sheet_name=sheet) | ||
| 48 | else: | ||
| 49 | supported = ".csv, .xlsx, .xls, .xlsm" | ||
| 50 | raise AuditSamplingError( | ||
| 51 | f"Unsupported input extension '{suffix}'. Supported: {supported}" | ||
| 52 | ) | ||
| 53 | |||
| 54 | frame = frame.copy() | ||
| 55 | frame.insert(0, "_source_row_number", range(2, len(frame) + 2)) | ||
| 56 | return frame | ||
| 57 | |||
| 58 | |||
| 59 | def write_csv(frame: pd.DataFrame, path: Path) -> None: | ||
| 60 | frame.to_csv(path, index=False) | ||
sampling/sampling_tool/manifest.py added +58
| @@ -0,0 +1,58 @@ | |||
| 1 | """Manifest generation.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | from pathlib import Path | ||
| 6 | import json | ||
| 7 | |||
| 8 | from . import __version__ | ||
| 9 | |||
| 10 | |||
| 11 | def build_manifest( | ||
| 12 | *, | ||
| 13 | run_timestamp_utc: str, | ||
| 14 | input_file: Path, | ||
| 15 | input_sha256: str, | ||
| 16 | input_sheet: str | None, | ||
| 17 | options, | ||
| 18 | effective_id_column: str | None, | ||
| 19 | id_column_omitted: bool, | ||
| 20 | sample_size_actual: int, | ||
| 21 | random_seed: int | None, | ||
| 22 | source_row_count: int, | ||
| 23 | validated_population_count: int, | ||
| 24 | excluded_row_count: int, | ||
| 25 | duplicate_id_count: int, | ||
| 26 | blank_id_count: int, | ||
| 27 | output_files: list[str], | ||
| 28 | ) -> dict[str, object]: | ||
| 29 | return { | ||
| 30 | "tool": "audit_sample", | ||
| 31 | "version": __version__, | ||
| 32 | "run_timestamp_utc": run_timestamp_utc, | ||
| 33 | "input_file": str(input_file), | ||
| 34 | "input_sha256": input_sha256, | ||
| 35 | "input_sheet": input_sheet, | ||
| 36 | "method": options.method, | ||
| 37 | "id_column": options.id_column, | ||
| 38 | "effective_id_column": effective_id_column, | ||
| 39 | "id_column_omitted": id_column_omitted, | ||
| 40 | "stratify_column": options.stratify_column, | ||
| 41 | "sample_size_requested": options.sample_size, | ||
| 42 | "sample_size_actual": sample_size_actual, | ||
| 43 | "random_seed": random_seed, | ||
| 44 | "filters": options.filters, | ||
| 45 | "dedupe_id": options.dedupe_id, | ||
| 46 | "exclude_blank_id": options.exclude_blank_id, | ||
| 47 | "allow_shortfall": options.allow_shortfall, | ||
| 48 | "source_row_count": source_row_count, | ||
| 49 | "validated_population_count": validated_population_count, | ||
| 50 | "excluded_row_count": excluded_row_count, | ||
| 51 | "duplicate_id_count": duplicate_id_count, | ||
| 52 | "blank_id_count": blank_id_count, | ||
| 53 | "output_files": output_files, | ||
| 54 | } | ||
| 55 | |||
| 56 | |||
| 57 | def write_manifest(manifest: dict[str, object], path: Path) -> None: | ||
| 58 | path.write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n") | ||
sampling/sampling_tool/methods.py added +125
| @@ -0,0 +1,125 @@ | |||
| 1 | """Sampling methods.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | import math | ||
| 6 | import random | ||
| 7 | |||
| 8 | import pandas as pd | ||
| 9 | |||
| 10 | from .io import AuditSamplingError | ||
| 11 | |||
| 12 | |||
| 13 | def ensure_seed(seed: int | None) -> tuple[int, bool]: | ||
| 14 | if seed is not None: | ||
| 15 | return int(seed), False | ||
| 16 | return random.SystemRandom().randint(1, 2_147_483_647), True | ||
| 17 | |||
| 18 | |||
| 19 | def parse_key_ints(value: str | None, label: str) -> dict[str, int]: | ||
| 20 | if not value: | ||
| 21 | return {} | ||
| 22 | parsed: dict[str, int] = {} | ||
| 23 | for part in value.split(","): | ||
| 24 | if "=" not in part: | ||
| 25 | raise AuditSamplingError(f"Invalid {label} entry '{part}'. Use Name=Count.") | ||
| 26 | key, raw_count = part.split("=", 1) | ||
| 27 | key = key.strip() | ||
| 28 | try: | ||
| 29 | count = int(raw_count.strip()) | ||
| 30 | except ValueError as exc: | ||
| 31 | raise AuditSamplingError( | ||
| 32 | f"Invalid {label} count for '{key}': {raw_count}" | ||
| 33 | ) from exc | ||
| 34 | if count <= 0: | ||
| 35 | raise AuditSamplingError(f"{label} count for '{key}' must be positive.") | ||
| 36 | parsed[key] = count | ||
| 37 | return parsed | ||
| 38 | |||
| 39 | |||
| 40 | def parse_key_floats(value: str | None, label: str) -> dict[str, float]: | ||
| 41 | if not value: | ||
| 42 | return {} | ||
| 43 | parsed: dict[str, float] = {} | ||
| 44 | for part in value.split(","): | ||
| 45 | if "=" not in part: | ||
| 46 | raise AuditSamplingError( | ||
| 47 | f"Invalid {label} entry '{part}'. Use Name=Proportion." | ||
| 48 | ) | ||
| 49 | key, raw_proportion = part.split("=", 1) | ||
| 50 | key = key.strip() | ||
| 51 | try: | ||
| 52 | proportion = float(raw_proportion.strip()) | ||
| 53 | except ValueError as exc: | ||
| 54 | raise AuditSamplingError( | ||
| 55 | f"Invalid {label} proportion for '{key}': {raw_proportion}" | ||
| 56 | ) from exc | ||
| 57 | if proportion <= 0: | ||
| 58 | raise AuditSamplingError( | ||
| 59 | f"{label} proportion for '{key}' must be positive." | ||
| 60 | ) | ||
| 61 | parsed[key] = proportion | ||
| 62 | if not math.isclose(sum(parsed.values()), 1.0, rel_tol=1e-9, abs_tol=1e-9): | ||
| 63 | raise AuditSamplingError(f"{label} proportions must sum to 1.0.") | ||
| 64 | return parsed | ||
| 65 | |||
| 66 | |||
| 67 | def largest_remainder_allocation( | ||
| 68 | sample_size: int, proportions: dict[str, float] | ||
| 69 | ) -> dict[str, int]: | ||
| 70 | raw = { | ||
| 71 | stratum: { | ||
| 72 | "floor": math.floor(sample_size * proportion), | ||
| 73 | "remainder": sample_size * proportion | ||
| 74 | - math.floor(sample_size * proportion), | ||
| 75 | } | ||
| 76 | for stratum, proportion in proportions.items() | ||
| 77 | } | ||
| 78 | allocation = {stratum: values["floor"] for stratum, values in raw.items()} | ||
| 79 | remaining = sample_size - sum(allocation.values()) | ||
| 80 | ranked = sorted( | ||
| 81 | raw, | ||
| 82 | key=lambda stratum: (-raw[stratum]["remainder"], stratum), | ||
| 83 | ) | ||
| 84 | for stratum in ranked[:remaining]: | ||
| 85 | allocation[stratum] += 1 | ||
| 86 | return allocation | ||
| 87 | |||
| 88 | |||
| 89 | def random_sample(population: pd.DataFrame, sample_size: int, seed: int) -> pd.DataFrame: | ||
| 90 | return population.sample(n=sample_size, random_state=seed) | ||
| 91 | |||
| 92 | |||
| 93 | def stratified_sample( | ||
| 94 | population: pd.DataFrame, | ||
| 95 | stratify_column: str, | ||
| 96 | counts: dict[str, int], | ||
| 97 | seed: int, | ||
| 98 | allow_shortfall: bool, | ||
| 99 | ) -> tuple[pd.DataFrame, list[dict[str, object]]]: | ||
| 100 | samples: list[pd.DataFrame] = [] | ||
| 101 | summary: list[dict[str, object]] = [] | ||
| 102 | for index, (stratum, requested) in enumerate(counts.items()): | ||
| 103 | stratum_population = population[population[stratify_column] == stratum] | ||
| 104 | actual_count = min(requested, len(stratum_population)) | ||
| 105 | if actual_count < requested and not allow_shortfall: | ||
| 106 | raise AuditSamplingError( | ||
| 107 | f"Stratum '{stratum}' has {len(stratum_population)} rows; " | ||
| 108 | f"requested {requested}. Use --allow-shortfall to continue." | ||
| 109 | ) | ||
| 110 | if actual_count: | ||
| 111 | sampled = stratum_population.sample(n=actual_count, random_state=seed + index) | ||
| 112 | samples.append(sampled) | ||
| 113 | summary.append( | ||
| 114 | { | ||
| 115 | "Stratum": stratum, | ||
| 116 | "Population Count": len(stratum_population), | ||
| 117 | "Requested Sample Count": requested, | ||
| 118 | "Actual Sample Count": actual_count, | ||
| 119 | "Shortfall": requested - actual_count, | ||
| 120 | } | ||
| 121 | ) | ||
| 122 | |||
| 123 | if samples: | ||
| 124 | return pd.concat(samples), summary | ||
| 125 | return population.iloc[0:0].copy(), summary | ||
sampling/sampling_tool/reconciliation.py added +41
| @@ -0,0 +1,41 @@ | |||
| 1 | """Reconciliation output helpers.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | import pandas as pd | ||
| 6 | |||
| 7 | |||
| 8 | def build_reconciliation( | ||
| 9 | source_rows: int, | ||
| 10 | blank_id_count: int, | ||
| 11 | duplicate_id_count: int, | ||
| 12 | excluded_rows: int, | ||
| 13 | validated_rows: int, | ||
| 14 | requested_sample_size: int | None, | ||
| 15 | final_sample_size: int, | ||
| 16 | ) -> pd.DataFrame: | ||
| 17 | unsampled = validated_rows - final_sample_size | ||
| 18 | rows = [ | ||
| 19 | ("Source rows", source_rows), | ||
| 20 | ("Rows with blank ID", blank_id_count), | ||
| 21 | ("Duplicate IDs", duplicate_id_count), | ||
| 22 | ("Excluded rows", excluded_rows), | ||
| 23 | ("Validated population rows", validated_rows), | ||
| 24 | ("Requested sample size", "" if requested_sample_size is None else requested_sample_size), | ||
| 25 | ("Final sample size", final_sample_size), | ||
| 26 | ("Unsampled population rows", unsampled), | ||
| 27 | ] | ||
| 28 | return pd.DataFrame(rows, columns=["Metric", "Value"]) | ||
| 29 | |||
| 30 | |||
| 31 | def build_strata_summary(summary_rows: list[dict[str, object]]) -> pd.DataFrame: | ||
| 32 | return pd.DataFrame( | ||
| 33 | summary_rows, | ||
| 34 | columns=[ | ||
| 35 | "Stratum", | ||
| 36 | "Population Count", | ||
| 37 | "Requested Sample Count", | ||
| 38 | "Actual Sample Count", | ||
| 39 | "Shortfall", | ||
| 40 | ], | ||
| 41 | ) | ||
sampling/sampling_tool/reporting.py added +87
| @@ -0,0 +1,87 @@ | |||
| 1 | """Human-readable reporting outputs.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | from pathlib import Path | ||
| 6 | |||
| 7 | |||
| 8 | class RunLogger: | ||
| 9 | def __init__(self) -> None: | ||
| 10 | self.lines: list[str] = [] | ||
| 11 | |||
| 12 | def log(self, message: str) -> None: | ||
| 13 | self.lines.append(message) | ||
| 14 | |||
| 15 | def warning(self, message: str) -> None: | ||
| 16 | self.log(f"WARNING: {message}") | ||
| 17 | |||
| 18 | def error(self, message: str) -> None: | ||
| 19 | self.log(f"ERROR: {message}") | ||
| 20 | |||
| 21 | def write(self, path: Path) -> None: | ||
| 22 | path.write_text("\n".join(self.lines) + "\n") | ||
| 23 | |||
| 24 | |||
| 25 | def build_methodology( | ||
| 26 | *, | ||
| 27 | input_file: Path, | ||
| 28 | input_sheet: str | None, | ||
| 29 | input_sha256: str, | ||
| 30 | source_row_count: int, | ||
| 31 | options, | ||
| 32 | effective_id_column: str | None, | ||
| 33 | id_column_omitted: bool, | ||
| 34 | duplicate_id_count: int, | ||
| 35 | blank_id_count: int, | ||
| 36 | sample_size_actual: int, | ||
| 37 | random_seed: int | None, | ||
| 38 | strata_summary: list[dict[str, object]] | None, | ||
| 39 | ) -> str: | ||
| 40 | lines = [ | ||
| 41 | "Audit Sampling Methodology", | ||
| 42 | "", | ||
| 43 | f"Source file: {input_file}", | ||
| 44 | f"Source sheet: {input_sheet or ''}", | ||
| 45 | f"Input SHA-256: {input_sha256}", | ||
| 46 | f"Source row count: {source_row_count}", | ||
| 47 | ] | ||
| 48 | if id_column_omitted: | ||
| 49 | lines.append( | ||
| 50 | "ID column: omitted; _audit_row_id was generated from _source_row_number." | ||
| 51 | ) | ||
| 52 | else: | ||
| 53 | lines.append(f"ID column: {options.id_column}") | ||
| 54 | lines.extend( | ||
| 55 | [ | ||
| 56 | f"Effective ID column: {effective_id_column or ''}", | ||
| 57 | f"Duplicate ID rows identified: {duplicate_id_count}", | ||
| 58 | f"Blank ID rows identified: {blank_id_count}", | ||
| 59 | f"Blank ID handling: {'excluded' if options.exclude_blank_id else 'retained'}", | ||
| 60 | f"Filters applied: {options.filters or {}}", | ||
| 61 | f"Sampling method: {options.method}", | ||
| 62 | f"Sample size requested: {options.sample_size or ''}", | ||
| 63 | f"Sample size selected: {sample_size_actual}", | ||
| 64 | f"Random seed used: {random_seed or ''}", | ||
| 65 | ] | ||
| 66 | ) | ||
| 67 | if options.method == "stratified": | ||
| 68 | lines.append(f"Stratification column: {options.stratify_column}") | ||
| 69 | lines.append(f"Strata counts: {options.strata_counts or ''}") | ||
| 70 | lines.append(f"Strata proportions: {options.strata_proportions or ''}") | ||
| 71 | if strata_summary: | ||
| 72 | lines.append("Strata summary:") | ||
| 73 | for row in strata_summary: | ||
| 74 | lines.append( | ||
| 75 | " " | ||
| 76 | f"{row['Stratum']}: population={row['Population Count']}, " | ||
| 77 | f"requested={row['Requested Sample Count']}, " | ||
| 78 | f"actual={row['Actual Sample Count']}, " | ||
| 79 | f"shortfall={row['Shortfall']}" | ||
| 80 | ) | ||
| 81 | lines.extend( | ||
| 82 | [ | ||
| 83 | "Sampling was performed without replacement.", | ||
| 84 | "Selected rows retain source row numbers and original source fields.", | ||
| 85 | ] | ||
| 86 | ) | ||
| 87 | return "\n".join(lines) + "\n" | ||
sampling/sampling_tool/validation.py added +157
| @@ -0,0 +1,157 @@ | |||
| 1 | """Validation and population preparation.""" | ||
| 2 | |||
| 3 | from __future__ import annotations | ||
| 4 | |||
| 5 | from dataclasses import dataclass | ||
| 6 | |||
| 7 | import pandas as pd | ||
| 8 | |||
| 9 | from .io import AuditSamplingError | ||
| 10 | from .methods import parse_key_floats, parse_key_ints | ||
| 11 | |||
| 12 | |||
| 13 | @dataclass | ||
| 14 | class ValidationResult: | ||
| 15 | population: pd.DataFrame | ||
| 16 | excluded_rows: pd.DataFrame | ||
| 17 | duplicate_rows: pd.DataFrame | ||
| 18 | effective_id_column: str | ||
| 19 | id_column_omitted: bool | ||
| 20 | blank_id_count: int | ||
| 21 | duplicate_id_count: int | ||
| 22 | warnings: list[str] | ||
| 23 | strata_counts: dict[str, int] | ||
| 24 | strata_proportions: dict[str, float] | ||
| 25 | |||
| 26 | |||
| 27 | def blank_id_mask(series: pd.Series) -> pd.Series: | ||
| 28 | return series.isna() | (series.astype("string").fillna("").str.strip() == "") | ||
| 29 | |||
| 30 | |||
| 31 | def validate_and_prepare(population: pd.DataFrame, options) -> ValidationResult: | ||
| 32 | warnings: list[str] = [] | ||
| 33 | excluded_parts: list[pd.DataFrame] = [] | ||
| 34 | working = population.copy() | ||
| 35 | |||
| 36 | if options.id_column: | ||
| 37 | if options.id_column not in working.columns: | ||
| 38 | raise AuditSamplingError(f"ID column not found: {options.id_column}") | ||
| 39 | effective_id_column = options.id_column | ||
| 40 | id_column_omitted = False | ||
| 41 | else: | ||
| 42 | effective_id_column = "_audit_row_id" | ||
| 43 | id_column_omitted = True | ||
| 44 | working[effective_id_column] = working["_source_row_number"] | ||
| 45 | warnings.append( | ||
| 46 | "--id-column omitted; using _source_row_number as generated _audit_row_id." | ||
| 47 | ) | ||
| 48 | |||
| 49 | if options.stratify_column and options.stratify_column not in working.columns: | ||
| 50 | raise AuditSamplingError( | ||
| 51 | f"Stratification column not found: {options.stratify_column}" | ||
| 52 | ) | ||
| 53 | |||
| 54 | id_blank_mask = blank_id_mask(working[effective_id_column]) | ||
| 55 | blank_id_count = int(id_blank_mask.sum()) | ||
| 56 | if blank_id_count: | ||
| 57 | if options.exclude_blank_id: | ||
| 58 | excluded = working.loc[id_blank_mask].copy() | ||
| 59 | excluded["_exclusion_reason"] = "Blank ID" | ||
| 60 | excluded_parts.append(excluded) | ||
| 61 | working = working.loc[~id_blank_mask].copy() | ||
| 62 | else: | ||
| 63 | warnings.append( | ||
| 64 | f"{blank_id_count} row(s) have blank IDs and were retained." | ||
| 65 | ) | ||
| 66 | |||
| 67 | nonblank_ids = ~blank_id_mask(working[effective_id_column]) | ||
| 68 | duplicate_mask = working.loc[nonblank_ids, effective_id_column].duplicated( | ||
| 69 | keep=False | ||
| 70 | ) | ||
| 71 | duplicate_rows = working.loc[nonblank_ids].loc[duplicate_mask].copy() | ||
| 72 | duplicate_id_count = int(len(duplicate_rows)) | ||
| 73 | if duplicate_id_count: | ||
| 74 | if options.dedupe_id == "fail": | ||
| 75 | raise AuditSamplingError( | ||
| 76 | f"Duplicate IDs found in '{effective_id_column}'. " | ||
| 77 | "See duplicate_ids.csv.", | ||
| 78 | duplicate_rows=duplicate_rows, | ||
| 79 | ) | ||
| 80 | keep = "first" if options.dedupe_id == "first" else "last" | ||
| 81 | drop_mask = working[effective_id_column].duplicated(keep=keep) & ~blank_id_mask( | ||
| 82 | working[effective_id_column] | ||
| 83 | ) | ||
| 84 | excluded = working.loc[drop_mask].copy() | ||
| 85 | excluded["_exclusion_reason"] = f"Duplicate ID removed by dedupe={keep}" | ||
| 86 | excluded_parts.append(excluded) | ||
| 87 | working = working.loc[~drop_mask].copy() | ||
| 88 | warnings.append( | ||
| 89 | f"{len(excluded)} duplicate ID row(s) removed using dedupe={keep}." | ||
| 90 | ) | ||
| 91 | |||
| 92 | if options.sample_size is not None and options.sample_size <= 0: | ||
| 93 | raise AuditSamplingError("--sample-size must be a positive integer.") | ||
| 94 | needs_sample_size = options.method == "random" or ( | ||
| 95 | options.method == "stratified" and bool(options.strata_proportions) | ||
| 96 | ) | ||
| 97 | if needs_sample_size: | ||
| 98 | if options.sample_size is None: | ||
| 99 | raise AuditSamplingError(f"--sample-size is required for {options.method}.") | ||
| 100 | if options.sample_size > len(working) and not ( | ||
| 101 | options.method == "stratified" and options.allow_shortfall | ||
| 102 | ): | ||
| 103 | raise AuditSamplingError( | ||
| 104 | "--sample-size cannot exceed the validated population size." | ||
| 105 | ) | ||
| 106 | |||
| 107 | strata_counts: dict[str, int] = {} | ||
| 108 | strata_proportions: dict[str, float] = {} | ||
| 109 | if options.method == "stratified": | ||
| 110 | if not options.stratify_column: | ||
| 111 | raise AuditSamplingError("--stratify-column is required for stratified.") | ||
| 112 | has_counts = bool(options.strata_counts) | ||
| 113 | has_proportions = bool(options.strata_proportions) | ||
| 114 | if has_counts == has_proportions: | ||
| 115 | raise AuditSamplingError( | ||
| 116 | "Use exactly one of --strata-counts or --strata-proportions." | ||
| 117 | ) | ||
| 118 | strata_counts = parse_key_ints(options.strata_counts, "strata") | ||
| 119 | strata_proportions = parse_key_floats(options.strata_proportions, "strata") | ||
| 120 | requested_strata = set(strata_counts or strata_proportions) | ||
| 121 | actual_strata = set(working[options.stratify_column].dropna().astype(str)) | ||
| 122 | missing = sorted(requested_strata - actual_strata) | ||
| 123 | if missing: | ||
| 124 | raise AuditSamplingError( | ||
| 125 | "Requested strata not found in population: " + ", ".join(missing) | ||
| 126 | ) | ||
| 127 | |||
| 128 | if strata_counts: | ||
| 129 | _validate_stratum_counts_fit(working, options, strata_counts) | ||
| 130 | |||
| 131 | if excluded_parts: | ||
| 132 | excluded_rows = pd.concat(excluded_parts, ignore_index=True) | ||
| 133 | else: | ||
| 134 | excluded_rows = working.iloc[0:0].copy() | ||
| 135 | |||
| 136 | return ValidationResult( | ||
| 137 | population=working, | ||
| 138 | excluded_rows=excluded_rows, | ||
| 139 | duplicate_rows=duplicate_rows, | ||
| 140 | effective_id_column=effective_id_column, | ||
| 141 | id_column_omitted=id_column_omitted, | ||
| 142 | blank_id_count=blank_id_count, | ||
| 143 | duplicate_id_count=duplicate_id_count, | ||
| 144 | warnings=warnings, | ||
| 145 | strata_counts=strata_counts, | ||
| 146 | strata_proportions=strata_proportions, | ||
| 147 | ) | ||
| 148 | |||
| 149 | |||
| 150 | def _validate_stratum_counts_fit(population, options, counts: dict[str, int]) -> None: | ||
| 151 | for stratum, requested in counts.items(): | ||
| 152 | available = int((population[options.stratify_column] == stratum).sum()) | ||
| 153 | if requested > available and not options.allow_shortfall: | ||
| 154 | raise AuditSamplingError( | ||
| 155 | f"Stratum '{stratum}' has {available} rows; requested {requested}. " | ||
| 156 | "Use --allow-shortfall to continue." | ||
| 157 | ) | ||
sampling/tests/test_random_sample.py added +70
| @@ -0,0 +1,70 @@ | |||
| 1 | from types import SimpleNamespace | ||
| 2 | |||
| 3 | import pandas as pd | ||
| 4 | |||
| 5 | from sampling.sampling_tool.cli import run | ||
| 6 | |||
| 7 | |||
| 8 | def _write_population(path, rows=20): | ||
| 9 | frame = pd.DataFrame( | ||
| 10 | { | ||
| 11 | "ID": [f"ID{i:03d}" for i in range(rows)], | ||
| 12 | "Status": ["Closed"] * rows, | ||
| 13 | } | ||
| 14 | ) | ||
| 15 | frame.to_csv(path, index=False) | ||
| 16 | |||
| 17 | |||
| 18 | def _options(input_path, out_path, seed=123, sample_size=5, method="random", **kwargs): | ||
| 19 | values = { | ||
| 20 | "input": str(input_path), | ||
| 21 | "sheet": None, | ||
| 22 | "id_column": "ID", | ||
| 23 | "method": method, | ||
| 24 | "sample_size": sample_size, | ||
| 25 | "stratify_column": None, | ||
| 26 | "strata_counts": None, | ||
| 27 | "strata_proportions": None, | ||
| 28 | "seed": seed, | ||
| 29 | "out": str(out_path), | ||
| 30 | "exclude_blank_id": False, | ||
| 31 | "dedupe_id": "fail", | ||
| 32 | "filters": {}, | ||
| 33 | "allow_shortfall": False, | ||
| 34 | } | ||
| 35 | values.update(kwargs) | ||
| 36 | return SimpleNamespace(**values) | ||
| 37 | |||
| 38 | |||
| 39 | def test_random_sample_returns_correct_size(tmp_path): | ||
| 40 | source = tmp_path / "population.csv" | ||
| 41 | _write_population(source) | ||
| 42 | |||
| 43 | run_dir = run(_options(source, tmp_path / "out")) | ||
| 44 | |||
| 45 | sample = pd.read_csv(run_dir / "sample.csv") | ||
| 46 | assert len(sample) == 5 | ||
| 47 | |||
| 48 | |||
| 49 | def test_same_seed_returns_same_selected_ids(tmp_path): | ||
| 50 | source = tmp_path / "population.csv" | ||
| 51 | _write_population(source) | ||
| 52 | |||
| 53 | first = run(_options(source, tmp_path / "out1", seed=20260707)) | ||
| 54 | second = run(_options(source, tmp_path / "out2", seed=20260707)) | ||
| 55 | |||
| 56 | first_ids = pd.read_csv(first / "sample.csv")["ID"].tolist() | ||
| 57 | second_ids = pd.read_csv(second / "sample.csv")["ID"].tolist() | ||
| 58 | assert first_ids == second_ids | ||
| 59 | |||
| 60 | |||
| 61 | def test_different_seed_can_return_different_selected_ids(tmp_path): | ||
| 62 | source = tmp_path / "population.csv" | ||
| 63 | _write_population(source) | ||
| 64 | |||
| 65 | first = run(_options(source, tmp_path / "out1", seed=1)) | ||
| 66 | second = run(_options(source, tmp_path / "out2", seed=2)) | ||
| 67 | |||
| 68 | first_ids = pd.read_csv(first / "sample.csv")["ID"].tolist() | ||
| 69 | second_ids = pd.read_csv(second / "sample.csv")["ID"].tolist() | ||
| 70 | assert first_ids != second_ids | ||
sampling/tests/test_reconciliation.py added +44
| @@ -0,0 +1,44 @@ | |||
| 1 | from types import SimpleNamespace | ||
| 2 | |||
| 3 | import pandas as pd | ||
| 4 | |||
| 5 | from sampling.sampling_tool.cli import run | ||
| 6 | |||
| 7 | |||
| 8 | def test_reconciliation_math_ties_out(tmp_path): | ||
| 9 | source = tmp_path / "population.csv" | ||
| 10 | pd.DataFrame( | ||
| 11 | {"ID": ["A", "B", "C", "D"], "Status": ["Closed", "Open", "Closed", "Open"]} | ||
| 12 | ).to_csv(source, index=False) | ||
| 13 | options = SimpleNamespace( | ||
| 14 | input=str(source), | ||
| 15 | sheet=None, | ||
| 16 | id_column="ID", | ||
| 17 | method="random", | ||
| 18 | sample_size=1, | ||
| 19 | stratify_column=None, | ||
| 20 | strata_counts=None, | ||
| 21 | strata_proportions=None, | ||
| 22 | seed=7, | ||
| 23 | out=str(tmp_path / "out"), | ||
| 24 | exclude_blank_id=False, | ||
| 25 | dedupe_id="fail", | ||
| 26 | filters={"Status": "Closed"}, | ||
| 27 | allow_shortfall=False, | ||
| 28 | ) | ||
| 29 | |||
| 30 | run_dir = run(options) | ||
| 31 | recon = pd.read_csv(run_dir / "population_reconciliation.csv") | ||
| 32 | metrics = dict(zip(recon["Metric"], recon["Value"])) | ||
| 33 | |||
| 34 | assert int(metrics["Source rows"]) == 4 | ||
| 35 | assert int(metrics["Excluded rows"]) == 2 | ||
| 36 | assert int(metrics["Validated population rows"]) == 2 | ||
| 37 | assert int(metrics["Final sample size"]) == 1 | ||
| 38 | assert int(metrics["Unsampled population rows"]) == 1 | ||
| 39 | assert int(metrics["Source rows"]) == ( | ||
| 40 | int(metrics["Validated population rows"]) + int(metrics["Excluded rows"]) | ||
| 41 | ) | ||
| 42 | assert int(metrics["Validated population rows"]) == ( | ||
| 43 | int(metrics["Final sample size"]) + int(metrics["Unsampled population rows"]) | ||
| 44 | ) | ||
sampling/tests/test_stratified_sample.py added +66
| @@ -0,0 +1,66 @@ | |||
| 1 | from types import SimpleNamespace | ||
| 2 | |||
| 3 | import pandas as pd | ||
| 4 | |||
| 5 | from sampling.sampling_tool.cli import run | ||
| 6 | |||
| 7 | |||
| 8 | def _write_population(path): | ||
| 9 | frame = pd.DataFrame( | ||
| 10 | { | ||
| 11 | "ID": [f"ID{i:03d}" for i in range(12)], | ||
| 12 | "Type": ["A"] * 5 + ["B"] * 4 + ["C"] * 3, | ||
| 13 | } | ||
| 14 | ) | ||
| 15 | frame.to_csv(path, index=False) | ||
| 16 | |||
| 17 | |||
| 18 | def _options(input_path, out_path, **kwargs): | ||
| 19 | values = { | ||
| 20 | "input": str(input_path), | ||
| 21 | "sheet": None, | ||
| 22 | "id_column": "ID", | ||
| 23 | "method": "stratified", | ||
| 24 | "sample_size": None, | ||
| 25 | "stratify_column": "Type", | ||
| 26 | "strata_counts": "A=2,B=2,C=1", | ||
| 27 | "strata_proportions": None, | ||
| 28 | "seed": 50, | ||
| 29 | "out": str(out_path), | ||
| 30 | "exclude_blank_id": False, | ||
| 31 | "dedupe_id": "fail", | ||
| 32 | "filters": {}, | ||
| 33 | "allow_shortfall": False, | ||
| 34 | } | ||
| 35 | values.update(kwargs) | ||
| 36 | return SimpleNamespace(**values) | ||
| 37 | |||
| 38 | |||
| 39 | def test_stratified_counts_select_exact_requested_counts(tmp_path): | ||
| 40 | source = tmp_path / "population.csv" | ||
| 41 | _write_population(source) | ||
| 42 | |||
| 43 | run_dir = run(_options(source, tmp_path / "out")) | ||
| 44 | |||
| 45 | counts = pd.read_csv(run_dir / "sample.csv")["Type"].value_counts().to_dict() | ||
| 46 | assert counts == {"A": 2, "B": 2, "C": 1} | ||
| 47 | |||
| 48 | |||
| 49 | def test_stratified_proportions_use_largest_remainder(tmp_path): | ||
| 50 | source = tmp_path / "population.csv" | ||
| 51 | _write_population(source) | ||
| 52 | |||
| 53 | run_dir = run( | ||
| 54 | _options( | ||
| 55 | source, | ||
| 56 | tmp_path / "out", | ||
| 57 | sample_size=7, | ||
| 58 | strata_counts=None, | ||
| 59 | strata_proportions="A=0.50,B=0.30,C=0.20", | ||
| 60 | ) | ||
| 61 | ) | ||
| 62 | |||
| 63 | sample = pd.read_csv(run_dir / "sample.csv") | ||
| 64 | counts = sample["Type"].value_counts().to_dict() | ||
| 65 | assert len(sample) == 7 | ||
| 66 | assert counts == {"A": 4, "B": 2, "C": 1} | ||
sampling/tests/test_validation.py added +81
| @@ -0,0 +1,81 @@ | |||
| 1 | from types import SimpleNamespace | ||
| 2 | |||
| 3 | import pandas as pd | ||
| 4 | import pytest | ||
| 5 | |||
| 6 | from sampling.sampling_tool.cli import run | ||
| 7 | from sampling.sampling_tool.io import AuditSamplingError | ||
| 8 | |||
| 9 | |||
| 10 | def _options(input_path, out_path, **kwargs): | ||
| 11 | values = { | ||
| 12 | "input": str(input_path), | ||
| 13 | "sheet": None, | ||
| 14 | "id_column": "ID", | ||
| 15 | "method": "validate-only", | ||
| 16 | "sample_size": None, | ||
| 17 | "stratify_column": None, | ||
| 18 | "strata_counts": None, | ||
| 19 | "strata_proportions": None, | ||
| 20 | "seed": None, | ||
| 21 | "out": str(out_path), | ||
| 22 | "exclude_blank_id": False, | ||
| 23 | "dedupe_id": "fail", | ||
| 24 | "filters": {}, | ||
| 25 | "allow_shortfall": False, | ||
| 26 | } | ||
| 27 | values.update(kwargs) | ||
| 28 | return SimpleNamespace(**values) | ||
| 29 | |||
| 30 | |||
| 31 | def test_duplicate_ids_fail_by_default_and_write_duplicate_file(tmp_path): | ||
| 32 | source = tmp_path / "population.csv" | ||
| 33 | pd.DataFrame({"ID": ["A", "A", "B"], "Status": ["Closed"] * 3}).to_csv( | ||
| 34 | source, index=False | ||
| 35 | ) | ||
| 36 | |||
| 37 | with pytest.raises(AuditSamplingError): | ||
| 38 | run(_options(source, tmp_path / "out")) | ||
| 39 | |||
| 40 | run_dir = next((tmp_path / "out").glob("sample_*")) | ||
| 41 | duplicates = pd.read_csv(run_dir / "duplicate_ids.csv") | ||
| 42 | assert duplicates["ID"].tolist() == ["A", "A"] | ||
| 43 | |||
| 44 | |||
| 45 | def test_blank_ids_are_excluded_when_requested(tmp_path): | ||
| 46 | source = tmp_path / "population.csv" | ||
| 47 | pd.DataFrame({"ID": ["A", "", "B"], "Status": ["Closed"] * 3}).to_csv( | ||
| 48 | source, index=False | ||
| 49 | ) | ||
| 50 | |||
| 51 | run_dir = run(_options(source, tmp_path / "out", exclude_blank_id=True)) | ||
| 52 | |||
| 53 | validated = pd.read_csv(run_dir / "population_validated.csv") | ||
| 54 | excluded = pd.read_csv(run_dir / "excluded_rows.csv") | ||
| 55 | assert len(validated) == 2 | ||
| 56 | assert excluded["_exclusion_reason"].tolist() == ["Blank ID"] | ||
| 57 | |||
| 58 | |||
| 59 | def test_filters_reduce_population_and_write_excluded_rows(tmp_path): | ||
| 60 | source = tmp_path / "population.csv" | ||
| 61 | pd.DataFrame( | ||
| 62 | {"ID": ["A", "B", "C"], "Status": ["Closed", "Open", "Closed"]} | ||
| 63 | ).to_csv(source, index=False) | ||
| 64 | |||
| 65 | run_dir = run(_options(source, tmp_path / "out", filters={"Status": "Closed"})) | ||
| 66 | |||
| 67 | validated = pd.read_csv(run_dir / "population_validated.csv") | ||
| 68 | excluded = pd.read_csv(run_dir / "excluded_rows.csv") | ||
| 69 | assert validated["ID"].tolist() == ["A", "C"] | ||
| 70 | assert excluded["ID"].tolist() == ["B"] | ||
| 71 | |||
| 72 | |||
| 73 | def test_validate_only_writes_no_sample_csv(tmp_path): | ||
| 74 | source = tmp_path / "population.csv" | ||
| 75 | pd.DataFrame({"ID": ["A", "B", "C"], "Status": ["Closed"] * 3}).to_csv( | ||
| 76 | source, index=False | ||
| 77 | ) | ||
| 78 | |||
| 79 | run_dir = run(_options(source, tmp_path / "out")) | ||
| 80 | |||
| 81 | assert not (run_dir / "sample.csv").exists() | ||