Commit 982249b041
Unsigned
Layout: unified · split
sampling/sample.py +12 −2
| @@ -1,5 +1,11 @@ | ||
| 1 | 1 | """ |
| 2 | 2 | Creates a sample from a CSV or Excel file based on user-defined SAMPLE_SIZE. |
| 3 | ||
| 4 | NOTE: This is a minimal teaching snippet. For real fieldwork use the | |
| 5 | `audit_sample.py` CLI (or the `sampling_tool` package), which records the | |
| 6 | population hash, seed, method, and tool version in a manifest so the sample is | |
| 7 | reproducible and defensible. This file fixes a SEED only so the example itself is | |
| 8 | repeatable; it does not emit that provenance. | |
| 3 | 9 | """ |
| 4 | 10 | |
| 5 | 11 | # Import packages |
| @@ -8,6 +14,10 @@ import pandas as pd | ||
| 8 | 14 | # Define the sample size |
| 9 | 15 | SAMPLE_SIZE = 25 |
| 10 | 16 | |
| 17 | # A fixed seed makes the draw reproducible: same population + same seed => same | |
| 18 | # rows. Record the seed alongside any sample you rely on. | |
| 19 | SEED = 20260707 | |
| 20 | ||
| 11 | 21 | # Import the data to a pandas DataFrame |
| 12 | 22 | df = pd.read_csv("FILENAME_GOES_HERE.csv") |
| 13 | 23 | |
| @@ -19,7 +29,7 @@ df = pd.read_csv("FILENAME_GOES_HERE.csv") | ||
| 19 | 29 | print("Dataframe size (rows, columns): ", df.shape) |
| 20 | 30 | |
| 21 | 31 | # Sample |
| 22 | sample = df.sample(SAMPLE_SIZE) | |
| 32 | sample = df.sample(SAMPLE_SIZE, random_state=SEED) | |
| 23 | 33 | print("Sample size: ", SAMPLE_SIZE) |
| 24 | 34 | print("Sample:\n", sample) |
| 25 | 35 | |
| @@ -31,4 +41,4 @@ print("Sample:\n", sample) | ||
| 31 | 41 | # |
| 32 | 42 | # # Sample Size: 25 + 5 replacement samples |
| 33 | 43 | # SAMPLE_SIZE = 30 |
| 34 | # sample = df.sample(SAMPLE_SIZE, replace=True) | |
| 44 | # sample = df.sample(SAMPLE_SIZE, replace=True, random_state=SEED) | |