Commit 752a483562
Verified · cmc
Layout: unified · split
notebooks/db_exploration.ipynb +119 −19
| @@ -16,14 +16,24 @@ | ||
| 16 | 16 | "Let\"s explore the data a little bit to see what kind of analysis and visualizations we want to implement." |
| 17 | 17 | ] |
| 18 | 18 | }, |
| 19 | { | |
| 20 | "cell_type": "markdown", | |
| 21 | "metadata": {}, | |
| 22 | "source": [ | |
| 23 | "### Set up environment\n", | |
| 24 | "\n", | |
| 25 | "Start by installating and importing the necessary packages. " | |
| 26 | ] | |
| 27 | }, | |
| 19 | 28 | { |
| 20 | 29 | "cell_type": "code", |
| 21 | 30 | "execution_count": null, |
| 22 | 31 | "metadata": {}, |
| 23 | 32 | "outputs": [], |
| 24 | 33 | "source": [ |
| 25 | "!pip3 install ipykernel\n", | |
| 26 | "!pip3 install --upgrade pandas plotly dash \"nbformat>=4.2.0\"" | |
| 34 | "# Install packages, if needed\n", | |
| 35 | "# !pip3 install ipykernel\n", | |
| 36 | "# !pip3 install --upgrade pandas plotly dash \"nbformat>=4.2.0\"" | |
| 27 | 37 | ] |
| 28 | 38 | }, |
| 29 | 39 | { |
| @@ -32,8 +42,22 @@ | ||
| 32 | 42 | "metadata": {}, |
| 33 | 43 | "outputs": [], |
| 34 | 44 | "source": [ |
| 45 | "# Import packages\n", | |
| 35 | 46 | "import pandas as pd\n", |
| 36 | "import sqlite3" | |
| 47 | "import numpy as np\n", | |
| 48 | "import sqlite3\n", | |
| 49 | "import plotly.express as px\n", | |
| 50 | "import plotly.graph_objects as go\n", | |
| 51 | "import plotly.io as pio" | |
| 52 | ] | |
| 53 | }, | |
| 54 | { | |
| 55 | "cell_type": "markdown", | |
| 56 | "metadata": {}, | |
| 57 | "source": [ | |
| 58 | "### Load Data\n", | |
| 59 | "\n", | |
| 60 | "To load the data, we need to connect to the SQLite3 database file and query it for the data we want." | |
| 37 | 61 | ] |
| 38 | 62 | }, |
| 39 | 63 | { |
| @@ -42,7 +66,9 @@ | ||
| 42 | 66 | "metadata": {}, |
| 43 | 67 | "outputs": [], |
| 44 | 68 | "source": [ |
| 45 | "connection = sqlite3.connect(\"../raw_data/ingress.db\")" | |
| 69 | "# Connect to the database\n", | |
| 70 | "connection = sqlite3.connect(\"../raw_data/ingress.db\")\n", | |
| 71 | "cursor = connection.cursor()" | |
| 46 | 72 | ] |
| 47 | 73 | }, |
| 48 | 74 | { |
| @@ -51,7 +77,9 @@ | ||
| 51 | 77 | "metadata": {}, |
| 52 | 78 | "outputs": [], |
| 53 | 79 | "source": [ |
| 54 | "cursor = connection.cursor()" | |
| 80 | "# If exists, delete extra header rows\n", | |
| 81 | "# delete_headers = \"DELETE FROM incidents WHERE rb = 'RB Number'\"\n", | |
| 82 | "# cursor.execute(delete_headers)" | |
| 55 | 83 | ] |
| 56 | 84 | }, |
| 57 | 85 | { |
| @@ -60,8 +88,19 @@ | ||
| 60 | 88 | "metadata": {}, |
| 61 | 89 | "outputs": [], |
| 62 | 90 | "source": [ |
| 63 | "# Test query to see if the data loaded\n", | |
| 64 | "select_all = \"SELECT * FROM incidents;\"" | |
| 91 | "# Grab all data\n", | |
| 92 | "select_all = \"SELECT * FROM incidents\"\n", | |
| 93 | "df = pd.read_sql_query(select_all, connection)\n", | |
| 94 | "df.head()" | |
| 95 | ] | |
| 96 | }, | |
| 97 | { | |
| 98 | "cell_type": "markdown", | |
| 99 | "metadata": {}, | |
| 100 | "source": [ | |
| 101 | "### Data Cleaning\n", | |
| 102 | "\n", | |
| 103 | "We will clean up the data before we use: inserting NaN, converting types, etc." | |
| 65 | 104 | ] |
| 66 | 105 | }, |
| 67 | 106 | { |
| @@ -70,10 +109,31 @@ | ||
| 70 | 109 | "metadata": {}, |
| 71 | 110 | "outputs": [], |
| 72 | 111 | "source": [ |
| 73 | "df = pd.read_sql_query(select_all, connection)\n", | |
| 112 | "# Replace empty cells in [lat, lon] with NaN\n", | |
| 113 | "df = df.replace(r'^\\s*$', np.nan, regex=True)\n", | |
| 74 | 114 | "df.head()" |
| 75 | 115 | ] |
| 76 | 116 | }, |
| 117 | { | |
| 118 | "cell_type": "code", | |
| 119 | "execution_count": null, | |
| 120 | "metadata": {}, | |
| 121 | "outputs": [], | |
| 122 | "source": [ | |
| 123 | "# Convert date col to datetime format\n", | |
| 124 | "df[\"date\"] = pd.to_datetime(df[\"date\"])\n", | |
| 125 | "df" | |
| 126 | ] | |
| 127 | }, | |
| 128 | { | |
| 129 | "cell_type": "markdown", | |
| 130 | "metadata": {}, | |
| 131 | "source": [ | |
| 132 | "### Plotting\n", | |
| 133 | "\n", | |
| 134 | "Let's test a plot that will show us the top categories of incidents." | |
| 135 | ] | |
| 136 | }, | |
| 77 | 137 | { |
| 78 | 138 | "cell_type": "code", |
| 79 | 139 | "execution_count": null, |
| @@ -81,20 +141,43 @@ | ||
| 81 | 141 | "outputs": [], |
| 82 | 142 | "source": [ |
| 83 | 143 | "# test plotting by sorting & plotting top 5 crime categories\n", |
| 84 | "s = df.value_counts(subset=[\"description\"])\n", | |
| 144 | "s = dff.value_counts(subset=[\"description\"])\n", | |
| 85 | 145 | "t = s.nlargest(5)\n", |
| 86 | 146 | "t.head()\n", |
| 87 | 147 | "t.plot(kind=\"bar\", title=\"Top 5 Incident Categories\")" |
| 88 | 148 | ] |
| 89 | 149 | }, |
| 150 | { | |
| 151 | "cell_type": "markdown", | |
| 152 | "metadata": {}, | |
| 153 | "source": [ | |
| 154 | "### Data Filtering\n", | |
| 155 | "\n", | |
| 156 | "To reduce the workload in this rest of this notebook, I am filtering just for one description and a range of dates.\n", | |
| 157 | "\n", | |
| 158 | "If you are doing a lot of analysis, I recommend modifying the query at the beginning to only the pull the data you need instead of filtering after querying." | |
| 159 | ] | |
| 160 | }, | |
| 90 | 161 | { |
| 91 | 162 | "cell_type": "code", |
| 92 | 163 | "execution_count": null, |
| 93 | 164 | "metadata": {}, |
| 94 | 165 | "outputs": [], |
| 95 | 166 | "source": [ |
| 96 | "import plotly.express as px\n", | |
| 97 | "import plotly.graph_objects as go" | |
| 167 | "# Create a smaller dataframe based on a selected date and description\n", | |
| 168 | "start_date = \"2023-01-01\"\n", | |
| 169 | "end_date = \"2023-12-31\"\n", | |
| 170 | "description = \"INJURY\"\n", | |
| 171 | "\n", | |
| 172 | "dff = df[(df['date'] > start_date) & (df['date'] < end_date)]\n", | |
| 173 | "dff = dff.reset_index()\n", | |
| 174 | "dff = dff[dff.description == description]\n", | |
| 175 | "\n", | |
| 176 | "dff_grouped = dff.groupby(by=\"date\").count()\n", | |
| 177 | "dff_grouped = dff_grouped.reset_index()\n", | |
| 178 | "\n", | |
| 179 | "print(dff.head())\n", | |
| 180 | "print(dff_grouped.head())" | |
| 98 | 181 | ] |
| 99 | 182 | }, |
| 100 | 183 | { |
| @@ -103,7 +186,7 @@ | ||
| 103 | 186 | "metadata": {}, |
| 104 | 187 | "outputs": [], |
| 105 | 188 | "source": [ |
| 106 | "s.head(10)" | |
| 189 | "dff.size" | |
| 107 | 190 | ] |
| 108 | 191 | }, |
| 109 | 192 | { |
| @@ -112,8 +195,16 @@ | ||
| 112 | 195 | "metadata": {}, |
| 113 | 196 | "outputs": [], |
| 114 | 197 | "source": [ |
| 115 | "filtered_df = df[(df['date'] > '01/01/2023') & (df['date'] < '12/31/2023')]\n", | |
| 116 | "filtered_df.head()" | |
| 198 | "dff.info()" | |
| 199 | ] | |
| 200 | }, | |
| 201 | { | |
| 202 | "cell_type": "markdown", | |
| 203 | "metadata": {}, | |
| 204 | "source": [ | |
| 205 | "### Mapping\n", | |
| 206 | "\n", | |
| 207 | "Let's create a geo map of the crime data." | |
| 117 | 208 | ] |
| 118 | 209 | }, |
| 119 | 210 | { |
| @@ -123,7 +214,7 @@ | ||
| 123 | 214 | "outputs": [], |
| 124 | 215 | "source": [ |
| 125 | 216 | "fig = px.scatter_mapbox(\n", |
| 126 | " filtered_df,\n", | |
| 217 | " dff,\n", | |
| 127 | 218 | " lat=\"lat\",\n", |
| 128 | 219 | " lon=\"lon\",\n", |
| 129 | 220 | " color=\"description\",\n", |
| @@ -138,7 +229,7 @@ | ||
| 138 | 229 | "fig.update_layout(mapbox_style=\"open-street-map\")\n", |
| 139 | 230 | "fig.update_layout(margin={\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0})\n", |
| 140 | 231 | "fig.update_layout(mapbox_bounds={\"west\": -180, \"east\": -50, \"south\": 20, \"north\": 90})\n", |
| 141 | "# fig.show()" | |
| 232 | "fig.show()" | |
| 142 | 233 | ] |
| 143 | 234 | }, |
| 144 | 235 | { |
| @@ -147,8 +238,17 @@ | ||
| 147 | 238 | "metadata": {}, |
| 148 | 239 | "outputs": [], |
| 149 | 240 | "source": [ |
| 150 | "import plotly.io as pio\n", | |
| 151 | "pio.write_html(fig, file=\"test.html\", auto_open=False)" | |
| 241 | "# Optionally, save the figure to an HTML file\n", | |
| 242 | "# pio.write_html(fig, file=\"test.html\", auto_open=True)" | |
| 243 | ] | |
| 244 | }, | |
| 245 | { | |
| 246 | "cell_type": "markdown", | |
| 247 | "metadata": {}, | |
| 248 | "source": [ | |
| 249 | "## Wrapping Up\n", | |
| 250 | "\n", | |
| 251 | "To finish, remember to close your database connections and save any data you need." | |
| 152 | 252 | ] |
| 153 | 253 | }, |
| 154 | 254 | { |
| @@ -157,7 +257,7 @@ | ||
| 157 | 257 | "metadata": {}, |
| 158 | 258 | "outputs": [], |
| 159 | 259 | "source": [ |
| 160 | "# clean up and close it out\n", | |
| 260 | "# clean up and close out the database\n", | |
| 161 | 261 | "connection.commit()\n", |
| 162 | 262 | "connection.close()" |
| 163 | 263 | ] |