Commit f484a105f0
Unsigned
Layout: unified · split
.gitignore +2 −1
| @@ -1,2 +1,3 @@ | |||
| 1 | venv/ | 1 | .venv |
| 2 | venv | ||
| 2 | readme.html | 3 | readme.html |
sampling/stratified_sample.py added +62
| @@ -0,0 +1,62 @@ | |||
| 1 | # Import packages | ||
| 2 | import pandas as pd | ||
| 3 | import math | ||
| 4 | |||
| 5 | # Load data | ||
| 6 | df = pd.read_csv("FILENAME_GOES_HERE.csv") | ||
| 7 | |||
| 8 | # ALTERNATIVE: If you use Excel, use this instead. Supports xls, xlsx, xlsm, | ||
| 9 | # xlsb, odf, ods and odt file extensions. | ||
| 10 | # df = pd.read_excel("FILENAME_GOES_HERE.xlsx") | ||
| 11 | |||
| 12 | # Print totals prior to sampling | ||
| 13 | print("Dataframe size (rows, columns):", df.shape) | ||
| 14 | |||
| 15 | # User-defined parameters | ||
| 16 | SAMPLE_SIZE = 25 | ||
| 17 | STRATIFY_COLUMN = "Category" # <- Change this to your column name | ||
| 18 | |||
| 19 | # Define stratum proportions (as fractions) | ||
| 20 | # Example: if you have categories A, B, and C | ||
| 21 | stratum_proportions = {"A": 0.4, "B": 0.4, "C": 0.2} | ||
| 22 | |||
| 23 | # Validate proportions sum to 1 | ||
| 24 | if not math.isclose(sum(stratum_proportions.values()), 1.0): | ||
| 25 | raise ValueError("Stratum proportions must sum to 1.") | ||
| 26 | |||
| 27 | # Check that all strata exist in the data | ||
| 28 | missing_strata = set(stratum_proportions.keys()) - set(df[STRATIFY_COLUMN].unique()) | ||
| 29 | if missing_strata: | ||
| 30 | raise ValueError( | ||
| 31 | f"Strata {missing_strata} not found in column '{STRATIFY_COLUMN}'." | ||
| 32 | ) | ||
| 33 | |||
| 34 | # Perform stratified sampling | ||
| 35 | samples = [] | ||
| 36 | for stratum, proportion in stratum_proportions.items(): | ||
| 37 | stratum_df = df[df[STRATIFY_COLUMN] == stratum] | ||
| 38 | n_samples = math.floor(SAMPLE_SIZE * proportion) | ||
| 39 | if n_samples > len(stratum_df): | ||
| 40 | raise ValueError( | ||
| 41 | f"Not enough data in stratum '{stratum}' to sample {n_samples} rows." | ||
| 42 | ) | ||
| 43 | stratum_sample = stratum_df.sample(n=n_samples, random_state=42) | ||
| 44 | samples.append(stratum_sample) | ||
| 45 | |||
| 46 | # Combine all stratum samples into one DataFrame | ||
| 47 | final_sample = pd.concat(samples).reset_index() | ||
| 48 | |||
| 49 | # If needed, randomly sample extra rows to fill any rounding gap | ||
| 50 | current_sample_size = len(final_sample) | ||
| 51 | if current_sample_size < SAMPLE_SIZE: | ||
| 52 | remaining = SAMPLE_SIZE - current_sample_size | ||
| 53 | remaining_sample = df.sample(n=remaining, random_state=42) | ||
| 54 | final_sample = pd.concat([final_sample, remaining_sample]) | ||
| 55 | |||
| 56 | # Print sample results | ||
| 57 | print("Final sample size:", final_sample.shape[0]) | ||
| 58 | print("Sample breakdown by stratum:\n", final_sample[STRATIFY_COLUMN].value_counts()) | ||
| 59 | print("\nSample:\n", final_sample) | ||
| 60 | |||
| 61 | # Optionally, save the sample to a new CSV | ||
| 62 | # final_sample.to_csv("sample_output.csv", index=False) | ||