Commit cc772aba04
Verified · cmc ci/daily-pull: success
Layout: unified · split
.gitbay/ci.yml added +127
| @@ -0,0 +1,127 @@ | |||
| 1 | # Twice-daily archive pull, ported from the GitHub workflow. Sarpy and | ||
| 2 | # Council Bluffs serve a rolling 12-month window; records that age out of | ||
| 3 | # those feeds exist nowhere else. This job is the only thing keeping them, | ||
| 4 | # so it refuses to publish an archive smaller than the one it started with. | ||
| 5 | # The archive lives as metro.db.gz on the gitbay release tagged "archive"; | ||
| 6 | # the site deploys to the pages branch. | ||
| 7 | jobs: | ||
| 8 | daily-pull: | ||
| 9 | schedule: "17 11,23 * * *" | ||
| 10 | steps: | ||
| 11 | - python3 -m venv .venv && .venv/bin/pip install -q -r requirements-ingest.txt | ||
| 12 | - | | ||
| 13 | set -e | ||
| 14 | export PATH="$PWD/.venv/bin:$PATH" DB=raw_data/metro.db HOST=git@127.0.0.1 R=krz/omaha-metro-blotter | ||
| 15 | mkdir -p raw_data | ||
| 16 | |||
| 17 | # Restore the published archive; a crashed publish leaves metro.db.new.gz. | ||
| 18 | if ssh $HOST release asset get $R archive metro.db.gz > metro.db.gz 2>/dev/null && [ -s metro.db.gz ]; then | ||
| 19 | : | ||
| 20 | elif ssh $HOST release asset get $R archive metro.db.new.gz > metro.db.gz 2>/dev/null && [ -s metro.db.gz ]; then | ||
| 21 | echo "recovered from interrupted publish" | ||
| 22 | else | ||
| 23 | echo "ERROR: no metro.db.gz on release 'archive'; aged-out records cannot be recovered" | ||
| 24 | exit 1 | ||
| 25 | fi | ||
| 26 | gunzip -c metro.db.gz > "$DB" && rm metro.db.gz | ||
| 27 | echo "restored $(du -h "$DB" | cut -f1)" | ||
| 28 | |||
| 29 | counts() { | ||
| 30 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 31 | "SELECT source, COUNT(*) FROM incidents GROUP BY source ORDER BY source" | ||
| 32 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 33 | "SELECT source || '+amend', COUNT(*) FROM incident_amendments GROUP BY source ORDER BY source" 2>/dev/null || true | ||
| 34 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 35 | "SELECT source || '+raw', COUNT(*) FROM raw_records GROUP BY source ORDER BY source" 2>/dev/null || true | ||
| 36 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 37 | "SELECT agency || '+search', COUNT(*) FROM alpr_searches GROUP BY agency ORDER BY agency" 2>/dev/null || true | ||
| 38 | } | ||
| 39 | counts > before.txt | ||
| 40 | cat before.txt | ||
| 41 | |||
| 42 | # OPD amends records filed months ago; sweep the whole feed on Sundays. | ||
| 43 | if [ "$(date -u +%u)" = "7" ]; then | ||
| 44 | echo "full sweep" | ||
| 45 | python ingest.py --full opd sarpy cbpd alpr flock opd_archive | ||
| 46 | else | ||
| 47 | python ingest.py | ||
| 48 | fi | ||
| 49 | |||
| 50 | counts > after.txt | ||
| 51 | cat after.txt | ||
| 52 | test -s after.txt || { echo "ERROR: archive is empty"; exit 1; } | ||
| 53 | awk -v first=before.txt ' | ||
| 54 | FILENAME == first { was[$1] = $2; next } | ||
| 55 | { now[$1] = $2 } | ||
| 56 | END { | ||
| 57 | for (s in was) | ||
| 58 | if (now[s] + 0 < was[s] + 0) { | ||
| 59 | printf "ERROR: %s lost rows: %d -> %d\n", s, was[s], now[s] | ||
| 60 | bad = 1 | ||
| 61 | } | ||
| 62 | exit bad | ||
| 63 | }' before.txt after.txt | ||
| 64 | |||
| 65 | # An amendment identical to its original means the digest drifted. | ||
| 66 | phantom=$(sqlite3 -noheader "$DB" " | ||
| 67 | SELECT COUNT(*) FROM incident_amendments a | ||
| 68 | JOIN incidents o ON o.source = a.source AND o.source_key = a.source_key | ||
| 69 | WHERE a.agency IS o.agency AND a.case_id IS o.case_id | ||
| 70 | AND a.occurred_at IS o.occurred_at AND a.category IS o.category | ||
| 71 | AND a.call_type IS o.call_type AND a.disposition IS o.disposition | ||
| 72 | AND a.offense_desc IS o.offense_desc AND a.is_stop IS o.is_stop | ||
| 73 | AND a.address IS o.address AND a.lat IS o.lat AND a.lon IS o.lon") | ||
| 74 | if [ "$phantom" -ne 0 ]; then | ||
| 75 | echo "ERROR: $phantom amendments are identical to their original" | ||
| 76 | exit 1 | ||
| 77 | fi | ||
| 78 | |||
| 79 | # Publish. The .new asset makes the sequence crash-safe: at every | ||
| 80 | # point at least one asset holds the full archive. | ||
| 81 | sqlite3 "$DB" "VACUUM;" | ||
| 82 | gzip -c "$DB" > metro.db.gz | ||
| 83 | ssh $HOST release asset remove $R archive metro.db.new.gz 2>/dev/null || true | ||
| 84 | ssh $HOST release asset add $R archive metro.db.new.gz < metro.db.gz | ||
| 85 | ssh $HOST release asset remove $R archive metro.db.gz 2>/dev/null || true | ||
| 86 | ssh $HOST release asset add $R archive metro.db.gz < metro.db.gz | ||
| 87 | ssh $HOST release asset remove $R archive metro.db.new.gz 2>/dev/null || true | ||
| 88 | rm metro.db.gz | ||
| 89 | echo "archive published" | ||
| 90 | - | | ||
| 91 | set -e | ||
| 92 | export PATH="$PWD/.venv/bin:$PATH" DB=raw_data/metro.db | ||
| 93 | .venv/bin/pip install -q -r requirements.txt | ||
| 94 | python build_site.py | ||
| 95 | origin=$(git remote get-url origin) | ||
| 96 | cd site && git init -q -b pages | ||
| 97 | git -c user.name=ci -c user.email=ci@gitbay.org add -A | ||
| 98 | git -c user.name=ci -c user.email=ci@gitbay.org commit -q -m "site $(date -u '+%Y-%m-%d %H:%M')" | ||
| 99 | git push -qf "$origin" pages:pages | ||
| 100 | echo "site deployed" | ||
| 101 | - | | ||
| 102 | set -e | ||
| 103 | export DB=raw_data/metro.db | ||
| 104 | # Alarms last, on purpose: a stale feed should fail the build loudly | ||
| 105 | # without having blocked the archive or the site. | ||
| 106 | sqlite3 -noheader "$DB" \ | ||
| 107 | "SELECT source || ' ' || MAX(occurred_at) FROM incidents | ||
| 108 | WHERE source IN ('opd', 'sarpy', 'cbpd') GROUP BY source | ||
| 109 | HAVING MAX(occurred_at) < datetime('now', '-7 days')" > stale.txt | ||
| 110 | if [ -s stale.txt ]; then | ||
| 111 | while read -r line; do echo "ERROR: feed is stale: $line"; done < stale.txt | ||
| 112 | exit 1 | ||
| 113 | fi | ||
| 114 | echo "all feeds current" | ||
| 115 | age=$(sqlite3 -noheader "$DB" " | ||
| 116 | SELECT CAST(julianday('now') - julianday(MAX(imported_at)) AS INT) | ||
| 117 | FROM alpr_searches" 2>/dev/null || echo "") | ||
| 118 | if [ -z "$age" ]; then echo "no Flock export on file yet"; exit 0; fi | ||
| 119 | echo "newest Flock export imported $age days ago" | ||
| 120 | if [ "$age" -ge 27 ]; then | ||
| 121 | echo "ERROR: Flock search audit is $age days old and the portal only keeps 30." | ||
| 122 | echo "Download it from https://transparency.flocksafety.com/council-bluffs-ia-pd" | ||
| 123 | echo "and commit it to raw_data/flock/ before the window closes." | ||
| 124 | exit 1 | ||
| 125 | elif [ "$age" -ge 21 ]; then | ||
| 126 | echo "WARNING: Flock search audit is $age days old; refresh it soon." | ||
| 127 | fi | ||
.github/workflows/daily-pull.yml deleted −269
| @@ -1,269 +0,0 @@ | |||
| 1 | # Sarpy and Council Bluffs serve a rolling 12-month window; records that age out | ||
| 2 | # of those feeds exist nowhere else. This job is the only thing keeping them, so | ||
| 3 | # it refuses to publish an archive smaller than the one it started with. | ||
| 4 | name: daily pull | ||
| 5 | |||
| 6 | on: | ||
| 7 | schedule: | ||
| 8 | # Twice a day, because the cadence sets how much the rolling feeds drop | ||
| 9 | # before it is captured: a five-hour gap cost 13 Sarpy records once. | ||
| 10 | # Offset from the hour, GitHub drops on-the-hour runs under load. | ||
| 11 | - cron: "17 11 * * *" | ||
| 12 | - cron: "17 23 * * *" | ||
| 13 | workflow_dispatch: | ||
| 14 | inputs: | ||
| 15 | bootstrap: | ||
| 16 | description: "Start a new archive instead of restoring the published one" | ||
| 17 | type: boolean | ||
| 18 | default: false | ||
| 19 | full: | ||
| 20 | description: "Pull each feed in full rather than the last 30 days" | ||
| 21 | type: boolean | ||
| 22 | default: false | ||
| 23 | |||
| 24 | permissions: | ||
| 25 | contents: write | ||
| 26 | pages: write | ||
| 27 | id-token: write | ||
| 28 | |||
| 29 | concurrency: | ||
| 30 | group: archive | ||
| 31 | cancel-in-progress: false | ||
| 32 | |||
| 33 | env: | ||
| 34 | TAG: archive | ||
| 35 | DB: raw_data/metro.db | ||
| 36 | GH_TOKEN: ${{ github.token }} | ||
| 37 | |||
| 38 | jobs: | ||
| 39 | pull: | ||
| 40 | runs-on: ubuntu-latest | ||
| 41 | timeout-minutes: 45 | ||
| 42 | |||
| 43 | steps: | ||
| 44 | - uses: actions/checkout@v4 | ||
| 45 | |||
| 46 | - uses: actions/setup-python@v5 | ||
| 47 | with: | ||
| 48 | python-version: "3.13" | ||
| 49 | cache: pip | ||
| 50 | cache-dependency-path: requirements-ingest.txt | ||
| 51 | |||
| 52 | - run: pip install -r requirements-ingest.txt | ||
| 53 | |||
| 54 | - name: Restore archive | ||
| 55 | run: | | ||
| 56 | if gh release download "$TAG" --pattern metro.db.gz --dir .; then | ||
| 57 | gunzip -c metro.db.gz > "$DB" | ||
| 58 | rm metro.db.gz | ||
| 59 | echo "restored $(du -h "$DB" | cut -f1)" | ||
| 60 | elif [ "${{ inputs.bootstrap }}" = "true" ]; then | ||
| 61 | echo "no published archive; starting a new one" | ||
| 62 | else | ||
| 63 | echo "::error::No metro.db.gz on release '$TAG'. Anything that has" \ | ||
| 64 | "already aged out of the Sarpy and Council Bluffs feeds cannot" \ | ||
| 65 | "be recovered. Re-run with bootstrap only if that is intended." | ||
| 66 | exit 1 | ||
| 67 | fi | ||
| 68 | |||
| 69 | - name: Count rows before | ||
| 70 | run: | | ||
| 71 | : > before.txt | ||
| 72 | if [ -f "$DB" ]; then | ||
| 73 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 74 | "SELECT source, COUNT(*) FROM incidents GROUP BY source ORDER BY source" \ | ||
| 75 | > before.txt | ||
| 76 | # absent until the amendment migration has run against this archive | ||
| 77 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 78 | "SELECT source || '+amend', COUNT(*) FROM incident_amendments | ||
| 79 | GROUP BY source ORDER BY source" >> before.txt 2>/dev/null || true | ||
| 80 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 81 | "SELECT source || '+raw', COUNT(*) FROM raw_records | ||
| 82 | GROUP BY source ORDER BY source" >> before.txt 2>/dev/null || true | ||
| 83 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 84 | "SELECT agency || '+search', COUNT(*) FROM alpr_searches | ||
| 85 | GROUP BY agency ORDER BY agency" >> before.txt 2>/dev/null || true | ||
| 86 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 87 | "SELECT source || '+raw', COUNT(*) FROM raw_records | ||
| 88 | GROUP BY source ORDER BY source" >> before.txt 2>/dev/null || true | ||
| 89 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 90 | "SELECT agency || '+search', COUNT(*) FROM alpr_searches | ||
| 91 | GROUP BY agency ORDER BY agency" >> before.txt 2>/dev/null || true | ||
| 92 | fi | ||
| 93 | cat before.txt | ||
| 94 | |||
| 95 | - name: Pull feeds | ||
| 96 | run: | | ||
| 97 | # A 30-day window cannot see an agency amending a record filed months | ||
| 98 | # ago, and OPD does exactly that, so sweep the whole feed on Sundays. | ||
| 99 | if [ "${{ inputs.full }}" = "true" ] || [ "${{ inputs.bootstrap }}" = "true" ] \ | ||
| 100 | || [ "$(date -u +%u)" = "7" ]; then | ||
| 101 | echo "full sweep" | ||
| 102 | # opd_archive is closed years that never change; the weekly sweep is | ||
| 103 | # often enough to notice if OPD ever restates one. | ||
| 104 | python ingest.py --full opd sarpy cbpd alpr flock opd_archive | ||
| 105 | else | ||
| 106 | python ingest.py | ||
| 107 | fi | ||
| 108 | |||
| 109 | - name: Check nothing was lost | ||
| 110 | run: | | ||
| 111 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 112 | "SELECT source, COUNT(*) FROM incidents GROUP BY source ORDER BY source" \ | ||
| 113 | > after.txt | ||
| 114 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 115 | "SELECT source || '+amend', COUNT(*) FROM incident_amendments | ||
| 116 | GROUP BY source ORDER BY source" >> after.txt | ||
| 117 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 118 | "SELECT source || '+raw', COUNT(*) FROM raw_records | ||
| 119 | GROUP BY source ORDER BY source" >> after.txt | ||
| 120 | sqlite3 -noheader -separator ' ' "$DB" \ | ||
| 121 | "SELECT agency || '+search', COUNT(*) FROM alpr_searches | ||
| 122 | GROUP BY agency ORDER BY agency" >> after.txt | ||
| 123 | cat after.txt | ||
| 124 | test -s after.txt || { echo "::error::archive is empty"; exit 1; } | ||
| 125 | # Keyed on FILENAME, not NR == FNR: before.txt is empty on a bootstrap | ||
| 126 | # run, and awk never resets FNR for a zero-length file. | ||
| 127 | awk -v first=before.txt ' | ||
| 128 | FILENAME == first { was[$1] = $2; next } | ||
| 129 | { now[$1] = $2 } | ||
| 130 | END { | ||
| 131 | for (s in was) | ||
| 132 | if (now[s] + 0 < was[s] + 0) { | ||
| 133 | printf "::error::%s lost rows: %d -> %d\n", s, was[s], now[s] | ||
| 134 | bad = 1 | ||
| 135 | } | ||
| 136 | exit bad | ||
| 137 | }' before.txt after.txt | ||
| 138 | |||
| 139 | # An amendment identical to its original means the digest drifted | ||
| 140 | # against what SQLite stores, and every run would file the same | ||
| 141 | # phantom again. Cheap to check, silent and cumulative if it happens. | ||
| 142 | phantom=$(sqlite3 -noheader "$DB" " | ||
| 143 | SELECT COUNT(*) FROM incident_amendments a | ||
| 144 | JOIN incidents o ON o.source = a.source AND o.source_key = a.source_key | ||
| 145 | WHERE a.agency IS o.agency AND a.case_id IS o.case_id | ||
| 146 | AND a.occurred_at IS o.occurred_at AND a.category IS o.category | ||
| 147 | AND a.call_type IS o.call_type AND a.disposition IS o.disposition | ||
| 148 | AND a.offense_desc IS o.offense_desc AND a.is_stop IS o.is_stop | ||
| 149 | AND a.address IS o.address AND a.lat IS o.lat AND a.lon IS o.lon") | ||
| 150 | if [ "$phantom" -ne 0 ]; then | ||
| 151 | echo "::error::$phantom amendments are identical to their original" | ||
| 152 | exit 1 | ||
| 153 | fi | ||
| 154 | |||
| 155 | - name: Publish archive | ||
| 156 | run: | | ||
| 157 | sqlite3 "$DB" "VACUUM;" | ||
| 158 | gzip -c "$DB" > metro.db.gz | ||
| 159 | gh release view "$TAG" >/dev/null 2>&1 \ | ||
| 160 | || gh release create "$TAG" --title "Incident archive" --notes "building" | ||
| 161 | gh release upload "$TAG" metro.db.gz --clobber | ||
| 162 | { | ||
| 163 | echo "SQLite archive of Omaha metro police incident feeds, rebuilt daily." | ||
| 164 | echo "Sarpy County and Council Bluffs publish a rolling 12-month window," | ||
| 165 | echo "so this holds records their own feeds no longer serve." | ||
| 166 | echo | ||
| 167 | echo "Updated $(date -u '+%Y-%m-%d %H:%M UTC'). Schema: schema.sql." | ||
| 168 | echo | ||
| 169 | echo "incidents holds each record as first published; every later" | ||
| 170 | echo "version the feed served is a row in incident_amendments" | ||
| 171 | echo "($(sqlite3 -noheader "$DB" 'SELECT COUNT(*) FROM incident_amendments') so far)." | ||
| 172 | echo "incidents_current is the newest version of each, and" | ||
| 173 | echo "raw_records keeps the feed's own JSON for every version" | ||
| 174 | echo "so a parse can be redone against what actually arrived." | ||
| 175 | echo | ||
| 176 | echo '```' | ||
| 177 | sqlite3 -header -column "$DB" \ | ||
| 178 | "SELECT agency, COUNT(*) AS rows, SUM(is_stop) AS stops, | ||
| 179 | MIN(occurred_at) AS earliest, MAX(occurred_at) AS latest | ||
| 180 | FROM incidents GROUP BY agency ORDER BY rows DESC" | ||
| 181 | echo '```' | ||
| 182 | } > notes.md | ||
| 183 | gh release edit "$TAG" --notes-file notes.md | ||
| 184 | |||
| 185 | # Runs after the upload on purpose: a feed that stopped updating should | ||
| 186 | # raise the alarm without also blocking the archive from being published. | ||
| 187 | - name: Check the feeds are still moving | ||
| 188 | run: | | ||
| 189 | # An allowlist, not an exclusion: naming the live feeds means adding a | ||
| 190 | # backfill source cannot quietly turn this into a permanent failure, | ||
| 191 | # which is exactly what renaming opd_csv to opd_archive did. | ||
| 192 | sqlite3 -noheader "$DB" \ | ||
| 193 | "SELECT source || ' ' || MAX(occurred_at) FROM incidents | ||
| 194 | WHERE source IN ('opd', 'sarpy', 'cbpd') GROUP BY source | ||
| 195 | HAVING MAX(occurred_at) < datetime('now', '-7 days')" > stale.txt | ||
| 196 | if [ -s stale.txt ]; then | ||
| 197 | while read -r line; do echo "::error::feed is stale: $line"; done < stale.txt | ||
| 198 | exit 1 | ||
| 199 | fi | ||
| 200 | echo "all feeds current" | ||
| 201 | |||
| 202 | # The Flock export is the one thing here that cannot be automated: the | ||
| 203 | # portal challenges every non-browser client. Its window is 30 days, so | ||
| 204 | # this fails at 27 while there is still time to act, not afterwards. | ||
| 205 | - name: Check the Flock export is current | ||
| 206 | run: | | ||
| 207 | age=$(sqlite3 -noheader "$DB" " | ||
| 208 | SELECT CAST(julianday('now') - julianday(MAX(imported_at)) AS INT) | ||
| 209 | FROM alpr_searches" 2>/dev/null || echo "") | ||
| 210 | if [ -z "$age" ] || [ "$age" = "" ]; then | ||
| 211 | echo "no Flock export on file yet"; exit 0 | ||
| 212 | fi | ||
| 213 | echo "newest Flock export imported $age days ago" | ||
| 214 | if [ "$age" -ge 27 ]; then | ||
| 215 | echo "::error::Flock search audit is $age days old and the portal only" \ | ||
| 216 | "keeps 30. Download it from" \ | ||
| 217 | "https://transparency.flocksafety.com/council-bluffs-ia-pd and commit" \ | ||
| 218 | "it to raw_data/flock/ before the window closes." | ||
| 219 | exit 1 | ||
| 220 | elif [ "$age" -ge 21 ]; then | ||
| 221 | echo "::warning::Flock search audit is $age days old; refresh it soon." | ||
| 222 | fi | ||
| 223 | |||
| 224 | - name: Summary | ||
| 225 | if: always() | ||
| 226 | run: | | ||
| 227 | { | ||
| 228 | echo "| source | before | after |" | ||
| 229 | echo "|---|---|---|" | ||
| 230 | awk -v first=before.txt ' | ||
| 231 | FILENAME == first { was[$1] = $2; next } | ||
| 232 | { printf "| %s | %s | %s |\n", $1, ($1 in was ? was[$1] : 0), $2 }' \ | ||
| 233 | before.txt after.txt | ||
| 234 | } >> "$GITHUB_STEP_SUMMARY" | ||
| 235 | |||
| 236 | # Separate job: the site needs pandas and numpy, and a Pages failure must not | ||
| 237 | # put the archive at risk. | ||
| 238 | site: | ||
| 239 | needs: pull | ||
| 240 | runs-on: ubuntu-latest | ||
| 241 | environment: | ||
| 242 | name: github-pages | ||
| 243 | url: ${{ steps.deploy.outputs.page_url }} | ||
| 244 | |||
| 245 | steps: | ||
| 246 | - uses: actions/checkout@v4 | ||
| 247 | |||
| 248 | - uses: actions/setup-python@v5 | ||
| 249 | with: | ||
| 250 | python-version: "3.13" | ||
| 251 | cache: pip | ||
| 252 | cache-dependency-path: requirements.txt | ||
| 253 | |||
| 254 | - run: pip install -r requirements.txt | ||
| 255 | |||
| 256 | - name: Fetch the published archive | ||
| 257 | run: | | ||
| 258 | gh release download "$TAG" --pattern metro.db.gz --dir . | ||
| 259 | mkdir -p raw_data | ||
| 260 | gunzip -c metro.db.gz > "$DB" | ||
| 261 | |||
| 262 | - run: python build_site.py | ||
| 263 | |||
| 264 | - uses: actions/configure-pages@v5 | ||
| 265 | - uses: actions/upload-pages-artifact@v3 | ||
| 266 | with: | ||
| 267 | path: site | ||
| 268 | - id: deploy | ||
| 269 | uses: actions/deploy-pages@v4 | ||