Commit 0bd5b69215
Verified · cmc
Layout: unified · split
README.nfo +13 −4
| @@ -93,10 +93,19 @@ FLOCK SEARCH AUDIT | ||
| 93 | 93 | |
| 94 | 94 | cloudflare serves a challenge to every non-browser client, so |
| 95 | 95 | ingest.py cannot fetch it. collecting is manual: open the portal, |
| 96 | "download csv", save it as raw_data/flock/<slug>_<date>.csv and | |
| 97 | commit. loading is not manual -- the flock source runs in the daily | |
| 98 | job and picks up whatever is committed. search ids are stable | |
| 99 | uuids, so overlapping exports dedupe. | |
| 96 | click "download csv", then | |
| 97 | ||
| 98 | .venv/bin/python ingest.py --import-flock ~/Downloads/public_search_audit.csv | |
| 99 | ||
| 100 | which checks the columns, files it under the right slug and loads | |
| 101 | it. commit what it writes. every export is named | |
| 102 | public_search_audit.csv with the agency nowhere inside, so pass | |
| 103 | --agency for any portal other than council bluffs. | |
| 104 | ||
| 105 | loading is not manual -- the flock source runs in the daily job and | |
| 106 | picks up whatever is committed. search ids are stable uuids, so | |
| 107 | overlapping exports dedupe and re-importing the same window is a | |
| 108 | no-op. | |
| 100 | 109 | |
| 101 | 110 | the portals keep 30 days. miss a month and that month is gone. |
| 102 | 111 | |
ingest.py +39 −1
| @@ -326,6 +326,34 @@ def ingest_opd_csv(conn, _since): | ||
| 326 | 326 | |
| 327 | 327 | |
| 328 | 328 | FLOCK_COLUMNS = ("id", "userId", "searchDate", "networkCount", "reason") |
| 329 | FLOCK_DIR = ROOT / "raw_data" / "flock" | |
| 330 | # Council Bluffs is the only metro portal that offers the export at all; Sarpy | |
| 331 | # and Douglas publish counts without one. | |
| 332 | FLOCK_DEFAULT_AGENCY = "council-bluffs-ia-pd" | |
| 333 | ||
| 334 | ||
| 335 | def import_flock(path, agency): | |
| 336 | """File a portal download under the name ingest_flock expects. | |
| 337 | ||
| 338 | The portal names every export public_search_audit.csv, with the agency | |
| 339 | nowhere in the file, so the slug has to be supplied and is worth printing: | |
| 340 | getting it wrong silently files one agency's searches under another.""" | |
| 341 | src = Path(path).expanduser() | |
| 342 | with src.open(newline="") as fh: | |
| 343 | reader = csv.DictReader(fh) | |
| 344 | if tuple(reader.fieldnames or ()) != FLOCK_COLUMNS: | |
| 345 | raise SystemExit(f" {src.name}: not a Flock search audit " | |
| 346 | f"(columns {reader.fieldnames})") | |
| 347 | rows = list(reader) | |
| 348 | if not rows: | |
| 349 | raise SystemExit(f" {src.name}: no rows") | |
| 350 | span = f"{min(r['searchDate'] for r in rows)[:10]} to " \ | |
| 351 | f"{max(r['searchDate'] for r in rows)[:10]}" | |
| 352 | FLOCK_DIR.mkdir(parents=True, exist_ok=True) | |
| 353 | dest = FLOCK_DIR / f"{agency}_{datetime.now(LOCAL):%Y-%m-%d}.csv" | |
| 354 | dest.write_bytes(src.read_bytes()) | |
| 355 | print(f" {len(rows)} searches, {span}") | |
| 356 | print(f" filed as {dest.relative_to(ROOT)} under agency '{agency}'") | |
| 329 | 357 | |
| 330 | 358 | |
| 331 | 359 | def ingest_flock(conn, _since): |
| @@ -429,8 +457,18 @@ def main(): | ||
| 429 | 457 | "ignored by alpr, flock and opd_csv") |
| 430 | 458 | p.add_argument("--full", action="store_true", |
| 431 | 459 | help="pull the complete feed instead of --since-days") |
| 460 | p.add_argument("--import-flock", metavar="CSV", | |
| 461 | help="file a Flock portal download into raw_data/flock and " | |
| 462 | "load it") | |
| 463 | p.add_argument("--agency", default=FLOCK_DEFAULT_AGENCY, | |
| 464 | help=f"portal slug for --import-flock (default " | |
| 465 | f"{FLOCK_DEFAULT_AGENCY})") | |
| 432 | 466 | args = p.parse_args() |
| 433 | sources = args.sources or ["opd", "sarpy", "cbpd", "alpr", "flock"] | |
| 467 | if args.import_flock: | |
| 468 | import_flock(args.import_flock, args.agency) | |
| 469 | sources = ["flock"] | |
| 470 | else: | |
| 471 | sources = args.sources or ["opd", "sarpy", "cbpd", "alpr", "flock"] | |
| 434 | 472 | |
| 435 | 473 | since = None if args.full else datetime.now(timezone.utc) - timedelta(days=args.since_days) |
| 436 | 474 | |