Commit f484a105f0
Unsigned
Layout: unified · split
.gitignore +2 −1
| @@ -1,2 +1,3 @@ | ||
| 1 | venv/ | |
| 1 | .venv | |
| 2 | venv | |
| 2 | 3 | readme.html |
sampling/stratified_sample.py added +62
| @@ -0,0 +1,62 @@ | ||
| 1 | # Import packages | |
| 2 | import pandas as pd | |
| 3 | import math | |
| 4 | ||
| 5 | # Load data | |
| 6 | df = pd.read_csv("FILENAME_GOES_HERE.csv") | |
| 7 | ||
| 8 | # ALTERNATIVE: If you use Excel, use this instead. Supports xls, xlsx, xlsm, | |
| 9 | # xlsb, odf, ods and odt file extensions. | |
| 10 | # df = pd.read_excel("FILENAME_GOES_HERE.xlsx") | |
| 11 | ||
| 12 | # Print totals prior to sampling | |
| 13 | print("Dataframe size (rows, columns):", df.shape) | |
| 14 | ||
| 15 | # User-defined parameters | |
| 16 | SAMPLE_SIZE = 25 | |
| 17 | STRATIFY_COLUMN = "Category" # <- Change this to your column name | |
| 18 | ||
| 19 | # Define stratum proportions (as fractions) | |
| 20 | # Example: if you have categories A, B, and C | |
| 21 | stratum_proportions = {"A": 0.4, "B": 0.4, "C": 0.2} | |
| 22 | ||
| 23 | # Validate proportions sum to 1 | |
| 24 | if not math.isclose(sum(stratum_proportions.values()), 1.0): | |
| 25 | raise ValueError("Stratum proportions must sum to 1.") | |
| 26 | ||
| 27 | # Check that all strata exist in the data | |
| 28 | missing_strata = set(stratum_proportions.keys()) - set(df[STRATIFY_COLUMN].unique()) | |
| 29 | if missing_strata: | |
| 30 | raise ValueError( | |
| 31 | f"Strata {missing_strata} not found in column '{STRATIFY_COLUMN}'." | |
| 32 | ) | |
| 33 | ||
| 34 | # Perform stratified sampling | |
| 35 | samples = [] | |
| 36 | for stratum, proportion in stratum_proportions.items(): | |
| 37 | stratum_df = df[df[STRATIFY_COLUMN] == stratum] | |
| 38 | n_samples = math.floor(SAMPLE_SIZE * proportion) | |
| 39 | if n_samples > len(stratum_df): | |
| 40 | raise ValueError( | |
| 41 | f"Not enough data in stratum '{stratum}' to sample {n_samples} rows." | |
| 42 | ) | |
| 43 | stratum_sample = stratum_df.sample(n=n_samples, random_state=42) | |
| 44 | samples.append(stratum_sample) | |
| 45 | ||
| 46 | # Combine all stratum samples into one DataFrame | |
| 47 | final_sample = pd.concat(samples).reset_index() | |
| 48 | ||
| 49 | # If needed, randomly sample extra rows to fill any rounding gap | |
| 50 | current_sample_size = len(final_sample) | |
| 51 | if current_sample_size < SAMPLE_SIZE: | |
| 52 | remaining = SAMPLE_SIZE - current_sample_size | |
| 53 | remaining_sample = df.sample(n=remaining, random_state=42) | |
| 54 | final_sample = pd.concat([final_sample, remaining_sample]) | |
| 55 | ||
| 56 | # Print sample results | |
| 57 | print("Final sample size:", final_sample.shape[0]) | |
| 58 | print("Sample breakdown by stratum:\n", final_sample[STRATIFY_COLUMN].value_counts()) | |
| 59 | print("\nSample:\n", final_sample) | |
| 60 | ||
| 61 | # Optionally, save the sample to a new CSV | |
| 62 | # final_sample.to_csv("sample_output.csv", index=False) | |