Commit ce1c40b7df
Verified · cmc
Layout: unified · split
.gitattributes added +2
| @@ -0,0 +1,2 @@ | |||
| 1 | *.csv filter=lfs diff=lfs merge=lfs -text | ||
| 2 | *.db filter=lfs diff=lfs merge=lfs -text | ||
.gitignore added +1
| @@ -0,0 +1 @@ | |||
| 1 | .DS_Store | ||
README.md added +12
| @@ -0,0 +1,12 @@ | |||
| 1 | # Omaha Incidents | ||
| 2 | |||
| 3 | Data from the Omaha police department, used to analyze and visualize statistics. | ||
| 4 | |||
| 5 | ## TODO | ||
| 6 | |||
| 7 | - [x] Import script | ||
| 8 | - [ ] Remove duplicate instances of headers being inserted into the database as | ||
| 9 | records | ||
| 10 | - [ ] Analysis script | ||
| 11 | - [~] Visualization script | ||
| 12 | - [ ] Build API to connect to database? | ||
notebooks/db_exploration.ipynb added +187
| @@ -0,0 +1,187 @@ | |||
| 1 | { | ||
| 2 | "cells": [ | ||
| 3 | { | ||
| 4 | "cell_type": "markdown", | ||
| 5 | "metadata": {}, | ||
| 6 | "source": [ | ||
| 7 | "# Omaha Incidents" | ||
| 8 | ] | ||
| 9 | }, | ||
| 10 | { | ||
| 11 | "cell_type": "markdown", | ||
| 12 | "metadata": {}, | ||
| 13 | "source": [ | ||
| 14 | "## Data Exploration\n", | ||
| 15 | "\n", | ||
| 16 | "Let\"s explore the data a little bit to see what kind of analysis and visualizations we want to implement." | ||
| 17 | ] | ||
| 18 | }, | ||
| 19 | { | ||
| 20 | "cell_type": "code", | ||
| 21 | "execution_count": null, | ||
| 22 | "metadata": {}, | ||
| 23 | "outputs": [], | ||
| 24 | "source": [ | ||
| 25 | "!pip3 install ipykernel\n", | ||
| 26 | "!pip3 install --upgrade pandas plotly dash \"nbformat>=4.2.0\"" | ||
| 27 | ] | ||
| 28 | }, | ||
| 29 | { | ||
| 30 | "cell_type": "code", | ||
| 31 | "execution_count": null, | ||
| 32 | "metadata": {}, | ||
| 33 | "outputs": [], | ||
| 34 | "source": [ | ||
| 35 | "import pandas as pd\n", | ||
| 36 | "import sqlite3" | ||
| 37 | ] | ||
| 38 | }, | ||
| 39 | { | ||
| 40 | "cell_type": "code", | ||
| 41 | "execution_count": null, | ||
| 42 | "metadata": {}, | ||
| 43 | "outputs": [], | ||
| 44 | "source": [ | ||
| 45 | "connection = sqlite3.connect(\"../raw_data/ingress.db\")" | ||
| 46 | ] | ||
| 47 | }, | ||
| 48 | { | ||
| 49 | "cell_type": "code", | ||
| 50 | "execution_count": null, | ||
| 51 | "metadata": {}, | ||
| 52 | "outputs": [], | ||
| 53 | "source": [ | ||
| 54 | "cursor = connection.cursor()" | ||
| 55 | ] | ||
| 56 | }, | ||
| 57 | { | ||
| 58 | "cell_type": "code", | ||
| 59 | "execution_count": null, | ||
| 60 | "metadata": {}, | ||
| 61 | "outputs": [], | ||
| 62 | "source": [ | ||
| 63 | "# Test query to see if the data loaded\n", | ||
| 64 | "select_all = \"SELECT * FROM incidents;\"" | ||
| 65 | ] | ||
| 66 | }, | ||
| 67 | { | ||
| 68 | "cell_type": "code", | ||
| 69 | "execution_count": null, | ||
| 70 | "metadata": {}, | ||
| 71 | "outputs": [], | ||
| 72 | "source": [ | ||
| 73 | "df = pd.read_sql_query(select_all, connection)\n", | ||
| 74 | "df.head()" | ||
| 75 | ] | ||
| 76 | }, | ||
| 77 | { | ||
| 78 | "cell_type": "code", | ||
| 79 | "execution_count": null, | ||
| 80 | "metadata": {}, | ||
| 81 | "outputs": [], | ||
| 82 | "source": [ | ||
| 83 | "# test plotting by sorting & plotting top 5 crime categories\n", | ||
| 84 | "s = df.value_counts(subset=[\"description\"])\n", | ||
| 85 | "t = s.nlargest(5)\n", | ||
| 86 | "t.head()\n", | ||
| 87 | "t.plot(kind=\"bar\", title=\"Top 5 Incident Categories\")" | ||
| 88 | ] | ||
| 89 | }, | ||
| 90 | { | ||
| 91 | "cell_type": "code", | ||
| 92 | "execution_count": null, | ||
| 93 | "metadata": {}, | ||
| 94 | "outputs": [], | ||
| 95 | "source": [ | ||
| 96 | "import plotly.express as px\n", | ||
| 97 | "import plotly.graph_objects as go" | ||
| 98 | ] | ||
| 99 | }, | ||
| 100 | { | ||
| 101 | "cell_type": "code", | ||
| 102 | "execution_count": null, | ||
| 103 | "metadata": {}, | ||
| 104 | "outputs": [], | ||
| 105 | "source": [ | ||
| 106 | "s.head(10)" | ||
| 107 | ] | ||
| 108 | }, | ||
| 109 | { | ||
| 110 | "cell_type": "code", | ||
| 111 | "execution_count": null, | ||
| 112 | "metadata": {}, | ||
| 113 | "outputs": [], | ||
| 114 | "source": [ | ||
| 115 | "filtered_df = df[(df['date'] > '01/01/2023') & (df['date'] < '12/31/2023')]\n", | ||
| 116 | "filtered_df.head()" | ||
| 117 | ] | ||
| 118 | }, | ||
| 119 | { | ||
| 120 | "cell_type": "code", | ||
| 121 | "execution_count": null, | ||
| 122 | "metadata": {}, | ||
| 123 | "outputs": [], | ||
| 124 | "source": [ | ||
| 125 | "fig = px.scatter_mapbox(\n", | ||
| 126 | " filtered_df,\n", | ||
| 127 | " lat=\"lat\",\n", | ||
| 128 | " lon=\"lon\",\n", | ||
| 129 | " color=\"description\",\n", | ||
| 130 | " hover_name=\"description\",\n", | ||
| 131 | " hover_data=[\"date\", \"time\"],\n", | ||
| 132 | " title=\"Incident Count by Coordinates\",\n", | ||
| 133 | " center={\"lat\": 41.257160, \"lon\": -95.995102},\n", | ||
| 134 | " zoom=10\n", | ||
| 135 | ")\n", | ||
| 136 | "\n", | ||
| 137 | "# fig.update_layout(showlegend=False)\n", | ||
| 138 | "fig.update_layout(mapbox_style=\"open-street-map\")\n", | ||
| 139 | "fig.update_layout(margin={\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0})\n", | ||
| 140 | "fig.update_layout(mapbox_bounds={\"west\": -180, \"east\": -50, \"south\": 20, \"north\": 90})\n", | ||
| 141 | "# fig.show()" | ||
| 142 | ] | ||
| 143 | }, | ||
| 144 | { | ||
| 145 | "cell_type": "code", | ||
| 146 | "execution_count": null, | ||
| 147 | "metadata": {}, | ||
| 148 | "outputs": [], | ||
| 149 | "source": [ | ||
| 150 | "import plotly.io as pio\n", | ||
| 151 | "pio.write_html(fig, file=\"test.html\", auto_open=False)" | ||
| 152 | ] | ||
| 153 | }, | ||
| 154 | { | ||
| 155 | "cell_type": "code", | ||
| 156 | "execution_count": null, | ||
| 157 | "metadata": {}, | ||
| 158 | "outputs": [], | ||
| 159 | "source": [ | ||
| 160 | "# clean up and close it out\n", | ||
| 161 | "connection.commit()\n", | ||
| 162 | "connection.close()" | ||
| 163 | ] | ||
| 164 | } | ||
| 165 | ], | ||
| 166 | "metadata": { | ||
| 167 | "kernelspec": { | ||
| 168 | "display_name": "Python 3 (ipykernel)", | ||
| 169 | "language": "python", | ||
| 170 | "name": "python3" | ||
| 171 | }, | ||
| 172 | "language_info": { | ||
| 173 | "codemirror_mode": { | ||
| 174 | "name": "ipython", | ||
| 175 | "version": 3 | ||
| 176 | }, | ||
| 177 | "file_extension": ".py", | ||
| 178 | "mimetype": "text/x-python", | ||
| 179 | "name": "python", | ||
| 180 | "nbconvert_exporter": "python", | ||
| 181 | "pygments_lexer": "ipython3", | ||
| 182 | "version": "3.11.7" | ||
| 183 | } | ||
| 184 | }, | ||
| 185 | "nbformat": 4, | ||
| 186 | "nbformat_minor": 4 | ||
| 187 | } | ||
notebooks/raw_data_exploration.ipynb added +99
| @@ -0,0 +1,99 @@ | |||
| 1 | { | ||
| 2 | "cells": [ | ||
| 3 | { | ||
| 4 | "cell_type": "markdown", | ||
| 5 | "metadata": {}, | ||
| 6 | "source": [ | ||
| 7 | "# Omaha Incidents" | ||
| 8 | ] | ||
| 9 | }, | ||
| 10 | { | ||
| 11 | "cell_type": "markdown", | ||
| 12 | "metadata": {}, | ||
| 13 | "source": [ | ||
| 14 | "## Prerequisites\n", | ||
| 15 | "\n", | ||
| 16 | "You must download the data from the URL below first.\n", | ||
| 17 | "\n", | ||
| 18 | "https://police.cityofomaha.org/crime-information/incident-data-download" | ||
| 19 | ] | ||
| 20 | }, | ||
| 21 | { | ||
| 22 | "cell_type": "markdown", | ||
| 23 | "metadata": {}, | ||
| 24 | "source": [ | ||
| 25 | "## Data Exploration\n", | ||
| 26 | "\n", | ||
| 27 | "Let's explore the data a little bit to see what kind of analysis and visualizations we want to implement." | ||
| 28 | ] | ||
| 29 | }, | ||
| 30 | { | ||
| 31 | "cell_type": "code", | ||
| 32 | "execution_count": null, | ||
| 33 | "metadata": {}, | ||
| 34 | "outputs": [], | ||
| 35 | "source": [ | ||
| 36 | "import pandas as pd" | ||
| 37 | ] | ||
| 38 | }, | ||
| 39 | { | ||
| 40 | "cell_type": "code", | ||
| 41 | "execution_count": null, | ||
| 42 | "metadata": {}, | ||
| 43 | "outputs": [], | ||
| 44 | "source": [ | ||
| 45 | "# import data\n", | ||
| 46 | "df = pd.read_csv(\"../raw_data/Incidents_2015.csv\")\n", | ||
| 47 | "\n", | ||
| 48 | "# test to see what the dataframe looks like\n", | ||
| 49 | "df.head()" | ||
| 50 | ] | ||
| 51 | }, | ||
| 52 | { | ||
| 53 | "cell_type": "code", | ||
| 54 | "execution_count": null, | ||
| 55 | "metadata": {}, | ||
| 56 | "outputs": [], | ||
| 57 | "source": [ | ||
| 58 | "# !pip install \"matplotlib\"\n", | ||
| 59 | "import numpy\n", | ||
| 60 | "import matplotlib\n", | ||
| 61 | "%matplotlib inline" | ||
| 62 | ] | ||
| 63 | }, | ||
| 64 | { | ||
| 65 | "cell_type": "code", | ||
| 66 | "execution_count": null, | ||
| 67 | "metadata": {}, | ||
| 68 | "outputs": [], | ||
| 69 | "source": [ | ||
| 70 | "# test plotting by sorting & plotting top 5 crime categories\n", | ||
| 71 | "s = df.value_counts(subset=[\"Statute/Ordinance Description\"])\n", | ||
| 72 | "t = s.nlargest(5)\n", | ||
| 73 | "t.head()\n", | ||
| 74 | "t.plot(kind=\"bar\", title=\"Top 5 Incident Categories\")" | ||
| 75 | ] | ||
| 76 | } | ||
| 77 | ], | ||
| 78 | "metadata": { | ||
| 79 | "kernelspec": { | ||
| 80 | "display_name": "Python 3 (ipykernel)", | ||
| 81 | "language": "python", | ||
| 82 | "name": "python3" | ||
| 83 | }, | ||
| 84 | "language_info": { | ||
| 85 | "codemirror_mode": { | ||
| 86 | "name": "ipython", | ||
| 87 | "version": 3 | ||
| 88 | }, | ||
| 89 | "file_extension": ".py", | ||
| 90 | "mimetype": "text/x-python", | ||
| 91 | "name": "python", | ||
| 92 | "nbconvert_exporter": "python", | ||
| 93 | "pygments_lexer": "ipython3", | ||
| 94 | "version": "3.11.7" | ||
| 95 | } | ||
| 96 | }, | ||
| 97 | "nbformat": 4, | ||
| 98 | "nbformat_minor": 4 | ||
| 99 | } | ||
raw_data/Incidents_2015.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:46504a13c49c3133e4ebfb7329cb6ac04b4571c513782700ca00da2679ba11af | ||
| 3 | size 3252033 | ||
raw_data/Incidents_2016.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:8faab205c00118f2157b239f71fa184f0d79adba781259931ba81407140fcf92 | ||
| 3 | size 6418868 | ||
raw_data/Incidents_2017.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:60b0cff58ec5a0ab7b5ff8bdcd6177f6a8548b4b05715e380c032b1cde5a98b8 | ||
| 3 | size 6750996 | ||
raw_data/Incidents_2018.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:6a6f3bbdd3832db4d23856620b006a9fc7788df88ce3c107d25729ffafd15af7 | ||
| 3 | size 6077309 | ||
raw_data/Incidents_2019.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:841b4d15ae718ce4dda8382d25140df4452055b603c879c38b57c10c560da11d | ||
| 3 | size 5885305 | ||
raw_data/Incidents_2020.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:9d49127d47b098f925f168ff9191205b6f1779e8d93914f7c1595c980429276b | ||
| 3 | size 5345985 | ||
raw_data/Incidents_2021.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:6d4742625748b87d0da6e747512343c0476d6dfb8d8be01f6391036ece57cb60 | ||
| 3 | size 7013108 | ||
raw_data/Incidents_2022.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:d511a12df510be0a7aaf0d757ce4240209f539559ffa71c55c9db8da7b7fc0d4 | ||
| 3 | size 7347606 | ||
raw_data/Incidents_2023.csv added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:0d3d38c8415a5fb53b3e5b0cdf5c5d5af8b67f2e2a5da0d0d86df5d7c68da193 | ||
| 3 | size 7140489 | ||
raw_data/ingress.db added +3
| @@ -0,0 +1,3 @@ | |||
| 1 | version https://git-lfs.github.com/spec/v1 | ||
| 2 | oid sha256:ff5293a40fb466e45ee6b69c85756f3aec3e8704094324616048321decdaae87 | ||
| 3 | size 43061248 | ||
scripts/dashboard.py added +76
| @@ -0,0 +1,76 @@ | |||
| 1 | from dash import Dash, html, dcc, callback, Output, Input | ||
| 2 | import plotly.express as px | ||
| 3 | import pandas as pd | ||
| 4 | import sqlite3 | ||
| 5 | |||
| 6 | # Connect to database and query all incidents | ||
| 7 | connection = sqlite3.connect("../raw_data/ingress.db") | ||
| 8 | cursor = connection.cursor() | ||
| 9 | query = "SELECT * FROM incidents;" | ||
| 10 | df = pd.read_sql_query(query, connection).sort_values(by="description") | ||
| 11 | |||
| 12 | # Create custom YEAR column to use in dropdown | ||
| 13 | df['year'] = df['date'].str[-4:] | ||
| 14 | |||
| 15 | # Configure HTML layout | ||
| 16 | app = Dash(__name__) | ||
| 17 | app.layout = html.Div(children = [ | ||
| 18 | html.Div([ | ||
| 19 | html.H1(children="Omaha Police Invidents", style={"textAlign":"center"}), | ||
| 20 | dcc.Dropdown(df.sort_values("description").description.unique(), "INJURY", id="bar-dropdown"), | ||
| 21 | dcc.Dropdown(df.sort_values("year").year.unique(), "2023", id="bar-year-dropdown"), | ||
| 22 | dcc.Graph(id="bar-graph") | ||
| 23 | ]), | ||
| 24 | html.Div([ | ||
| 25 | html.H2(children="Map Coordinates", style={"textAlign":"center"}), | ||
| 26 | dcc.Dropdown(df.sort_values("description").description.unique(), "INJURY", id="map-dropdown"), | ||
| 27 | dcc.Dropdown(df.sort_values("year").year.unique(), "2023", id="map-year-dropdown"), | ||
| 28 | dcc.Graph(id="map-graph") | ||
| 29 | ]) | ||
| 30 | ]) | ||
| 31 | |||
| 32 | # Create bar graph | ||
| 33 | @callback( | ||
| 34 | Output("bar-graph", "figure"), | ||
| 35 | Input("bar-dropdown", "value"), | ||
| 36 | Input("bar-year-dropdown", "value") | ||
| 37 | ) | ||
| 38 | def update_bar_graph(description, year): | ||
| 39 | dff = df[df.year == year] | ||
| 40 | dff = dff.value_counts(subset=["description"]) | ||
| 41 | dff = dff.reset_index() | ||
| 42 | dff = dff[dff.description == description] | ||
| 43 | return px.bar(dff, x="description", y="count") | ||
| 44 | |||
| 45 | # Create map | ||
| 46 | @callback( | ||
| 47 | Output("map-graph", "figure"), | ||
| 48 | Input("map-dropdown", "value"), | ||
| 49 | Input("map-year-dropdown", "value") | ||
| 50 | ) | ||
| 51 | def update_map(description, year): | ||
| 52 | dff = df[df.year == year] | ||
| 53 | dff = dff.reset_index() | ||
| 54 | dff = dff[dff.description == description] | ||
| 55 | |||
| 56 | fig = px.scatter_mapbox( | ||
| 57 | dff, | ||
| 58 | lat="lat", | ||
| 59 | lon="lon", | ||
| 60 | color="description", | ||
| 61 | hover_name="description", | ||
| 62 | hover_data=["date", "time"], | ||
| 63 | title="Incident Count by Coordinates", | ||
| 64 | center={"lat": 41.257160, "lon": -95.995102}, | ||
| 65 | zoom=10 | ||
| 66 | ) | ||
| 67 | |||
| 68 | fig.update_layout(showlegend=False) | ||
| 69 | fig.update_layout(mapbox_style="open-street-map") | ||
| 70 | fig.update_layout(margin={"r": 0, "t": 0, "l": 0, "b": 0}) | ||
| 71 | fig.update_layout(mapbox_bounds={"west": -180, "east": -50, "south": 20, "north": 90}) | ||
| 72 | |||
| 73 | return fig | ||
| 74 | |||
| 75 | if __name__ == "__main__": | ||
| 76 | app.run(debug=True) | ||
scripts/load.py added +73
| @@ -0,0 +1,73 @@ | |||
| 1 | # Import required modules | ||
| 2 | import csv | ||
| 3 | import sqlite3 | ||
| 4 | import os | ||
| 5 | |||
| 6 | # Create the database file | ||
| 7 | connection = sqlite3.connect('../raw_data/ingress.db') | ||
| 8 | |||
| 9 | # Creating a cursor object to execute SQL queries | ||
| 10 | cursor = connection.cursor() | ||
| 11 | |||
| 12 | # Table Definition | ||
| 13 | # rb = RB Number | ||
| 14 | # date = Reported Date | ||
| 15 | # time = Reported Time | ||
| 16 | # description = Statute/Ordinance Description | ||
| 17 | # location = Occurred Location | ||
| 18 | # district = Occurred District | ||
| 19 | # lat = Occurred Block LAT | ||
| 20 | # lon = Occurred Block LON | ||
| 21 | create_table = '''CREATE TABLE incidents( | ||
| 22 | id INTEGER PRIMARY KEY AUTOINCREMENT, | ||
| 23 | rb TEXT NOT NULL, | ||
| 24 | date TEXT NOT NULL, | ||
| 25 | time TEXT NOT NULL, | ||
| 26 | description TEXT NOT NULL, | ||
| 27 | location TEXT NOT NULL, | ||
| 28 | district TEXT NOT NULL, | ||
| 29 | lat REAL NOT NULL, | ||
| 30 | lon REAL NOT NULL); | ||
| 31 | ''' | ||
| 32 | |||
| 33 | # Create the table | ||
| 34 | cursor.execute(create_table) | ||
| 35 | |||
| 36 | # Point to the data directory | ||
| 37 | directory = os.fsencode("../raw_data/") | ||
| 38 | |||
| 39 | # Loop through all raw data files | ||
| 40 | for file in os.listdir(directory): | ||
| 41 | filename = os.fsdecode(file) | ||
| 42 | if filename.endswith(".csv"): | ||
| 43 | # Opening the file | ||
| 44 | file = open("../raw_data/" + filename) | ||
| 45 | |||
| 46 | # Reading the contents of the file | ||
| 47 | contents = csv.reader(file) | ||
| 48 | |||
| 49 | # SQL query to insert data into the | ||
| 50 | # table | ||
| 51 | insert_records = "INSERT INTO incidents (rb, date, time, description, location, district, lat, lon) VALUES(?, ?, ?, ?, ?, ?, ?, ?)" | ||
| 52 | |||
| 53 | # Importing the contents of the file | ||
| 54 | # into our table | ||
| 55 | cursor.executemany(insert_records, contents) | ||
| 56 | print("Inserted data from: ", filename) | ||
| 57 | continue | ||
| 58 | else: | ||
| 59 | continue | ||
| 60 | |||
| 61 | # Test query to see if the data loaded | ||
| 62 | select_all = "SELECT * FROM incidents" | ||
| 63 | rows = cursor.execute(select_all).fetchall() | ||
| 64 | |||
| 65 | # Output to the console screen | ||
| 66 | for r in rows: | ||
| 67 | print(r) | ||
| 68 | |||
| 69 | # Commit the changes | ||
| 70 | connection.commit() | ||
| 71 | |||
| 72 | # Close the database connection | ||
| 73 | connection.close() | ||