audit-labs/tutorials

Learn how to perform data analysis, scripting, automation, and more! data-analysis learning scripting tutorials

Commit 8bd4c1ee76

8bd4c1ee76acb835258933634aab19ca68230cfb

parent: 8e4ddbe217

Unsigned

cmc <hello@cleberg.net> · 2025-12-24 03:51 UTC

add terminations notebook

Layout: unified · split

.gitignore +1 −1
@@ -1 +1 @@
1notebooks/.ipynb_checkpoints
1.ipynb_checkpoints/
notebooks/Untitled.ipynb added +137
@@ -0,0 +1,137 @@
1{
2 "cells": [
3 {
4 "cell_type": "code",
5 "execution_count": 1,
6 "id": "84b491a8-4ce7-44f3-aed9-feba2ddd1b3b",
7 "metadata": {},
8 "outputs": [
9 {
10 "ename": "FileNotFoundError",
11 "evalue": "[Errno 2] No such file or directory: 'hr_terminations.csv'",
12 "output_type": "error",
13 "traceback": [
14 "\u001b[31m---------------------------------------------------------------------------\u001b[39m",
15 "\u001b[31mFileNotFoundError\u001b[39m Traceback (most recent call last)",
16 "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[1]\u001b[39m\u001b[32m, line 4\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;28;01mimport\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mpandas\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mas\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mpd\u001b[39;00m\n\u001b[32m 3\u001b[39m \u001b[38;5;66;03m# Load the datasets\u001b[39;00m\n\u001b[32m----> \u001b[39m\u001b[32m4\u001b[39m df_hr = \u001b[43mpd\u001b[49m\u001b[43m.\u001b[49m\u001b[43mread_csv\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m'\u001b[39;49m\u001b[33;43mhr_terminations.csv\u001b[39;49m\u001b[33;43m'\u001b[39;49m\u001b[43m)\u001b[49m\n\u001b[32m 5\u001b[39m df_app = pd.read_csv(\u001b[33m'\u001b[39m\u001b[33mapp_abc_users.csv\u001b[39m\u001b[33m'\u001b[39m)\n\u001b[32m 7\u001b[39m \u001b[38;5;66;03m# Convert date columns to actual datetime objects immediately\u001b[39;00m\n",
17 "\u001b[36mFile \u001b[39m\u001b[32m/opt/homebrew/Cellar/jupyterlab/4.5.1/libexec/lib/python3.14/site-packages/pandas/io/parsers/readers.py:1026\u001b[39m, in \u001b[36mread_csv\u001b[39m\u001b[34m(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, delim_whitespace, low_memory, memory_map, float_precision, storage_options, dtype_backend)\u001b[39m\n\u001b[32m 1013\u001b[39m kwds_defaults = _refine_defaults_read(\n\u001b[32m 1014\u001b[39m dialect,\n\u001b[32m 1015\u001b[39m delimiter,\n\u001b[32m (...)\u001b[39m\u001b[32m 1022\u001b[39m dtype_backend=dtype_backend,\n\u001b[32m 1023\u001b[39m )\n\u001b[32m 1024\u001b[39m kwds.update(kwds_defaults)\n\u001b[32m-> \u001b[39m\u001b[32m1026\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[43m_read\u001b[49m\u001b[43m(\u001b[49m\u001b[43mfilepath_or_buffer\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43mkwds\u001b[49m\u001b[43m)\u001b[49m\n",
18 "\u001b[36mFile \u001b[39m\u001b[32m/opt/homebrew/Cellar/jupyterlab/4.5.1/libexec/lib/python3.14/site-packages/pandas/io/parsers/readers.py:620\u001b[39m, in \u001b[36m_read\u001b[39m\u001b[34m(filepath_or_buffer, kwds)\u001b[39m\n\u001b[32m 617\u001b[39m _validate_names(kwds.get(\u001b[33m\"\u001b[39m\u001b[33mnames\u001b[39m\u001b[33m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m))\n\u001b[32m 619\u001b[39m \u001b[38;5;66;03m# Create the parser.\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m620\u001b[39m parser = \u001b[43mTextFileReader\u001b[49m\u001b[43m(\u001b[49m\u001b[43mfilepath_or_buffer\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43m*\u001b[49m\u001b[43m*\u001b[49m\u001b[43mkwds\u001b[49m\u001b[43m)\u001b[49m\n\u001b[32m 622\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m chunksize \u001b[38;5;129;01mor\u001b[39;00m iterator:\n\u001b[32m 623\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m parser\n",
19 "\u001b[36mFile \u001b[39m\u001b[32m/opt/homebrew/Cellar/jupyterlab/4.5.1/libexec/lib/python3.14/site-packages/pandas/io/parsers/readers.py:1620\u001b[39m, in \u001b[36mTextFileReader.__init__\u001b[39m\u001b[34m(self, f, engine, **kwds)\u001b[39m\n\u001b[32m 1617\u001b[39m \u001b[38;5;28mself\u001b[39m.options[\u001b[33m\"\u001b[39m\u001b[33mhas_index_names\u001b[39m\u001b[33m\"\u001b[39m] = kwds[\u001b[33m\"\u001b[39m\u001b[33mhas_index_names\u001b[39m\u001b[33m\"\u001b[39m]\n\u001b[32m 1619\u001b[39m \u001b[38;5;28mself\u001b[39m.handles: IOHandles | \u001b[38;5;28;01mNone\u001b[39;00m = \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[32m-> \u001b[39m\u001b[32m1620\u001b[39m \u001b[38;5;28mself\u001b[39m._engine = \u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43m_make_engine\u001b[49m\u001b[43m(\u001b[49m\u001b[43mf\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43mengine\u001b[49m\u001b[43m)\u001b[49m\n",
20 "\u001b[36mFile \u001b[39m\u001b[32m/opt/homebrew/Cellar/jupyterlab/4.5.1/libexec/lib/python3.14/site-packages/pandas/io/parsers/readers.py:1880\u001b[39m, in \u001b[36mTextFileReader._make_engine\u001b[39m\u001b[34m(self, f, engine)\u001b[39m\n\u001b[32m 1878\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m mode:\n\u001b[32m 1879\u001b[39m mode += \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m\n\u001b[32m-> \u001b[39m\u001b[32m1880\u001b[39m \u001b[38;5;28mself\u001b[39m.handles = \u001b[43mget_handle\u001b[49m\u001b[43m(\u001b[49m\n\u001b[32m 1881\u001b[39m \u001b[43m \u001b[49m\u001b[43mf\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1882\u001b[39m \u001b[43m \u001b[49m\u001b[43mmode\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1883\u001b[39m \u001b[43m \u001b[49m\u001b[43mencoding\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mencoding\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mNone\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1884\u001b[39m \u001b[43m \u001b[49m\u001b[43mcompression\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mcompression\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mNone\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1885\u001b[39m \u001b[43m \u001b[49m\u001b[43mmemory_map\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mmemory_map\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mFalse\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1886\u001b[39m \u001b[43m \u001b[49m\u001b[43mis_text\u001b[49m\u001b[43m=\u001b[49m\u001b[43mis_text\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1887\u001b[39m \u001b[43m \u001b[49m\u001b[43merrors\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mencoding_errors\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mstrict\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1888\u001b[39m \u001b[43m \u001b[49m\u001b[43mstorage_options\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mstorage_options\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mNone\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 1889\u001b[39m \u001b[43m\u001b[49m\u001b[43m)\u001b[49m\n\u001b[32m 1890\u001b[39m \u001b[38;5;28;01massert\u001b[39;00m \u001b[38;5;28mself\u001b[39m.handles \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[32m 1891\u001b[39m f = \u001b[38;5;28mself\u001b[39m.handles.handle\n",
21 "\u001b[36mFile \u001b[39m\u001b[32m/opt/homebrew/Cellar/jupyterlab/4.5.1/libexec/lib/python3.14/site-packages/pandas/io/common.py:873\u001b[39m, in \u001b[36mget_handle\u001b[39m\u001b[34m(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)\u001b[39m\n\u001b[32m 868\u001b[39m \u001b[38;5;28;01melif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(handle, \u001b[38;5;28mstr\u001b[39m):\n\u001b[32m 869\u001b[39m \u001b[38;5;66;03m# Check whether the filename is to be opened in binary mode.\u001b[39;00m\n\u001b[32m 870\u001b[39m \u001b[38;5;66;03m# Binary mode does not support 'encoding' and 'newline'.\u001b[39;00m\n\u001b[32m 871\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m ioargs.encoding \u001b[38;5;129;01mand\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m ioargs.mode:\n\u001b[32m 872\u001b[39m \u001b[38;5;66;03m# Encoding\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m873\u001b[39m handle = \u001b[38;5;28;43mopen\u001b[39;49m\u001b[43m(\u001b[49m\n\u001b[32m 874\u001b[39m \u001b[43m \u001b[49m\u001b[43mhandle\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 875\u001b[39m \u001b[43m \u001b[49m\u001b[43mioargs\u001b[49m\u001b[43m.\u001b[49m\u001b[43mmode\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 876\u001b[39m \u001b[43m \u001b[49m\u001b[43mencoding\u001b[49m\u001b[43m=\u001b[49m\u001b[43mioargs\u001b[49m\u001b[43m.\u001b[49m\u001b[43mencoding\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 877\u001b[39m \u001b[43m \u001b[49m\u001b[43merrors\u001b[49m\u001b[43m=\u001b[49m\u001b[43merrors\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m 878\u001b[39m \u001b[43m \u001b[49m\u001b[43mnewline\u001b[49m\u001b[43m=\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\n\u001b[32m 879\u001b[39m \u001b[43m \u001b[49m\u001b[43m)\u001b[49m\n\u001b[32m 880\u001b[39m \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[32m 881\u001b[39m \u001b[38;5;66;03m# Binary mode\u001b[39;00m\n\u001b[32m 882\u001b[39m handle = \u001b[38;5;28mopen\u001b[39m(handle, ioargs.mode)\n",
22 "\u001b[31mFileNotFoundError\u001b[39m: [Errno 2] No such file or directory: 'hr_terminations.csv'"
23 ]
24 }
25 ],
26 "source": [
27 "import pandas as pd\n",
28 "\n",
29 "# Load the datasets\n",
30 "df_hr = pd.read_csv('hr_terminations.csv')\n",
31 "df_app = pd.read_csv('app_abc_users.csv')\n",
32 "\n",
33 "# Convert date columns to actual datetime objects immediately\n",
34 "df_hr['Term_Date'] = pd.to_datetime(df_hr['Term_Date'])\n",
35 "df_app['Last_Login'] = pd.to_datetime(df_app['Last_Login'])\n",
36 "\n",
37 "print(f\"HR Records: {len(df_hr)}\")\n",
38 "print(f\"App Records: {len(df_app)}\")\n",
39 "df_hr.head()"
40 ]
41 },
42 {
43 "cell_type": "code",
44 "execution_count": null,
45 "id": "f67f851a-ee33-405a-abf4-9393f1837047",
46 "metadata": {},
47 "outputs": [],
48 "source": [
49 "# Strip whitespace from IDs and Names to prevent 'false negatives'\n",
50 "df_hr['Employee_ID'] = df_hr['Employee_ID'].str.strip()\n",
51 "df_app['User_ID'] = df_app['User_ID'].str.strip()\n",
52 "\n",
53 "# Standardizing names for easier visual review later\n",
54 "df_hr['Name'] = df_hr['Name'].str.strip().str.title()\n",
55 "df_app['Full_Name'] = df_app['Full_Name'].str.strip().str.title()\n",
56 "\n",
57 "print(\"Data cleaning complete.\")"
58 ]
59 },
60 {
61 "cell_type": "code",
62 "execution_count": null,
63 "id": "6a32518c-fe0a-4fb9-89fe-7796d5267957",
64 "metadata": {},
65 "outputs": [],
66 "source": [
67 "# We join on the ID. \n",
68 "# We use 'left' because we only care about people on the termination list.\n",
69 "audit_merge = pd.merge(\n",
70 " df_hr, \n",
71 " df_app, \n",
72 " left_on='Employee_ID', \n",
73 " right_on='User_ID', \n",
74 " how='left'\n",
75 ")\n",
76 "\n",
77 "# Display the merged table\n",
78 "audit_merge"
79 ]
80 },
81 {
82 "cell_type": "code",
83 "execution_count": null,
84 "id": "b46cc115-8fee-476c-9f8c-6980d49a000f",
85 "metadata": {},
86 "outputs": [],
87 "source": [
88 "# 1. Identify Terminated but still 'Active' in Application\n",
89 "active_leavers = audit_merge[audit_merge['Account_Status'] == 'Active'].copy()\n",
90 "\n",
91 "# 2. Identify Logins occurring AFTER termination date\n",
92 "# This is a critical security finding indicating potential account misuse\n",
93 "post_term_logins = audit_merge[audit_merge['Last_Login'] > audit_merge['Term_Date']].copy()\n",
94 "\n",
95 "print(f\"Finding 1: {len(active_leavers)} users still marked as 'Active'\")\n",
96 "print(f\"Finding 2: {len(post_term_logins)} users logged in after termination\")"
97 ]
98 },
99 {
100 "cell_type": "code",
101 "execution_count": null,
102 "id": "9644d4b3-9682-4ebe-acfc-1474218c58b0",
103 "metadata": {},
104 "outputs": [],
105 "source": [
106 "# Create a summary report\n",
107 "with pd.ExcelWriter('Termination_Audit_Report.xlsx') as writer:\n",
108 " active_leavers.to_excel(writer, sheet_name='Active_Leavers', index=False)\n",
109 " post_term_logins.to_excel(writer, sheet_name='Post_Term_Logins', index=False)\n",
110 " audit_merge.to_excel(writer, sheet_name='Full_Traceability_Matrix', index=False)\n",
111 "\n",
112 "print(\"Audit Report Exported: Termination_Audit_Report.xlsx\")"
113 ]
114 }
115 ],
116 "metadata": {
117 "kernelspec": {
118 "display_name": "Python 3 (ipykernel)",
119 "language": "python",
120 "name": "python3"
121 },
122 "language_info": {
123 "codemirror_mode": {
124 "name": "ipython",
125 "version": 3
126 },
127 "file_extension": ".py",
128 "mimetype": "text/x-python",
129 "name": "python",
130 "nbconvert_exporter": "python",
131 "pygments_lexer": "ipython3",
132 "version": "3.14.2"
133 }
134 },
135 "nbformat": 4,
136 "nbformat_minor": 5
137}
notebooks/basic_data_analysis.ipynb → notebooks/basic_data_analysis/basic_data_analysis.ipynb renamed
notebooks/racf_access_analysis.ipynb → notebooks/racf_access_analysis/racf_access_analysis.ipynb renamed
notebooks/sample_racf_data.txt → notebooks/racf_access_analysis/sample_racf_data.txt renamed
notebooks/terminations/Termination_Audit_Report.xlsx added

Binary file not shown.

notebooks/terminations/Terminations.ipynb added +373
@@ -0,0 +1,373 @@
1{
2 "cells": [
3 {
4 "cell_type": "code",
5 "execution_count": 3,
6 "id": "84b491a8-4ce7-44f3-aed9-feba2ddd1b3b",
7 "metadata": {},
8 "outputs": [
9 {
10 "name": "stdout",
11 "output_type": "stream",
12 "text": [
13 "HR Records: 5\n",
14 "App Records: 6\n"
15 ]
16 },
17 {
18 "data": {
19 "text/html": [
20 "<div>\n",
21 "<style scoped>\n",
22 " .dataframe tbody tr th:only-of-type {\n",
23 " vertical-align: middle;\n",
24 " }\n",
25 "\n",
26 " .dataframe tbody tr th {\n",
27 " vertical-align: top;\n",
28 " }\n",
29 "\n",
30 " .dataframe thead th {\n",
31 " text-align: right;\n",
32 " }\n",
33 "</style>\n",
34 "<table border=\"1\" class=\"dataframe\">\n",
35 " <thead>\n",
36 " <tr style=\"text-align: right;\">\n",
37 " <th></th>\n",
38 " <th>Employee_ID</th>\n",
39 " <th>Name</th>\n",
40 " <th>Term_Date</th>\n",
41 " <th>Department</th>\n",
42 " </tr>\n",
43 " </thead>\n",
44 " <tbody>\n",
45 " <tr>\n",
46 " <th>0</th>\n",
47 " <td>E001</td>\n",
48 " <td>Alice Smith</td>\n",
49 " <td>2023-11-15</td>\n",
50 " <td>Sales</td>\n",
51 " </tr>\n",
52 " <tr>\n",
53 " <th>1</th>\n",
54 " <td>E005</td>\n",
55 " <td>Bob Johnson</td>\n",
56 " <td>2023-12-01</td>\n",
57 " <td>IT</td>\n",
58 " </tr>\n",
59 " <tr>\n",
60 " <th>2</th>\n",
61 " <td>E010</td>\n",
62 " <td>Charlie Brown</td>\n",
63 " <td>2024-01-10</td>\n",
64 " <td>Finance</td>\n",
65 " </tr>\n",
66 " <tr>\n",
67 " <th>3</th>\n",
68 " <td>E012</td>\n",
69 " <td>David Miller</td>\n",
70 " <td>2023-10-20</td>\n",
71 " <td>Marketing</td>\n",
72 " </tr>\n",
73 " <tr>\n",
74 " <th>4</th>\n",
75 " <td>E015</td>\n",
76 " <td>Eve Wilson</td>\n",
77 " <td>2023-12-25</td>\n",
78 " <td>Engineering</td>\n",
79 " </tr>\n",
80 " </tbody>\n",
81 "</table>\n",
82 "</div>"
83 ],
84 "text/plain": [
85 " Employee_ID Name Term_Date Department\n",
86 "0 E001 Alice Smith 2023-11-15 Sales\n",
87 "1 E005 Bob Johnson 2023-12-01 IT\n",
88 "2 E010 Charlie Brown 2024-01-10 Finance\n",
89 "3 E012 David Miller 2023-10-20 Marketing\n",
90 "4 E015 Eve Wilson 2023-12-25 Engineering"
91 ]
92 },
93 "execution_count": 3,
94 "metadata": {},
95 "output_type": "execute_result"
96 }
97 ],
98 "source": [
99 "import pandas as pd\n",
100 "\n",
101 "# Load the datasets\n",
102 "df_hr = pd.read_csv('hr_terminations.csv')\n",
103 "df_app = pd.read_csv('app_users.csv')\n",
104 "\n",
105 "# Convert date columns to actual datetime objects immediately\n",
106 "df_hr['Term_Date'] = pd.to_datetime(df_hr['Term_Date'])\n",
107 "df_app['Last_Login'] = pd.to_datetime(df_app['Last_Login'])\n",
108 "\n",
109 "print(f\"HR Records: {len(df_hr)}\")\n",
110 "print(f\"App Records: {len(df_app)}\")\n",
111 "df_hr.head()"
112 ]
113 },
114 {
115 "cell_type": "code",
116 "execution_count": 4,
117 "id": "1ebfee0c-aece-471a-8d21-49232d672cce",
118 "metadata": {},
119 "outputs": [
120 {
121 "name": "stdout",
122 "output_type": "stream",
123 "text": [
124 "Data cleaning complete.\n"
125 ]
126 }
127 ],
128 "source": [
129 "# Strip whitespace from IDs and Names to prevent 'false negatives'\n",
130 "df_hr['Employee_ID'] = df_hr['Employee_ID'].str.strip()\n",
131 "df_app['User_ID'] = df_app['User_ID'].str.strip()\n",
132 "\n",
133 "# Standardizing names for easier visual review later\n",
134 "df_hr['Name'] = df_hr['Name'].str.strip().str.title()\n",
135 "df_app['Full_Name'] = df_app['Full_Name'].str.strip().str.title()\n",
136 "\n",
137 "print(\"Data cleaning complete.\")"
138 ]
139 },
140 {
141 "cell_type": "code",
142 "execution_count": 5,
143 "id": "d5a547ec-e787-4b1e-ae77-07ab5b77b980",
144 "metadata": {},
145 "outputs": [
146 {
147 "data": {
148 "text/html": [
149 "<div>\n",
150 "<style scoped>\n",
151 " .dataframe tbody tr th:only-of-type {\n",
152 " vertical-align: middle;\n",
153 " }\n",
154 "\n",
155 " .dataframe tbody tr th {\n",
156 " vertical-align: top;\n",
157 " }\n",
158 "\n",
159 " .dataframe thead th {\n",
160 " text-align: right;\n",
161 " }\n",
162 "</style>\n",
163 "<table border=\"1\" class=\"dataframe\">\n",
164 " <thead>\n",
165 " <tr style=\"text-align: right;\">\n",
166 " <th></th>\n",
167 " <th>Employee_ID</th>\n",
168 " <th>Name</th>\n",
169 " <th>Term_Date</th>\n",
170 " <th>Department</th>\n",
171 " <th>User_ID</th>\n",
172 " <th>Full_Name</th>\n",
173 " <th>Account_Status</th>\n",
174 " <th>Last_Login</th>\n",
175 " </tr>\n",
176 " </thead>\n",
177 " <tbody>\n",
178 " <tr>\n",
179 " <th>0</th>\n",
180 " <td>E001</td>\n",
181 " <td>Alice Smith</td>\n",
182 " <td>2023-11-15</td>\n",
183 " <td>Sales</td>\n",
184 " <td>E001</td>\n",
185 " <td>Alice Smith</td>\n",
186 " <td>Active</td>\n",
187 " <td>2024-01-05</td>\n",
188 " </tr>\n",
189 " <tr>\n",
190 " <th>1</th>\n",
191 " <td>E005</td>\n",
192 " <td>Bob Johnson</td>\n",
193 " <td>2023-12-01</td>\n",
194 " <td>IT</td>\n",
195 " <td>E005</td>\n",
196 " <td>Bob Johnson</td>\n",
197 " <td>Active</td>\n",
198 " <td>2023-11-28</td>\n",
199 " </tr>\n",
200 " <tr>\n",
201 " <th>2</th>\n",
202 " <td>E010</td>\n",
203 " <td>Charlie Brown</td>\n",
204 " <td>2024-01-10</td>\n",
205 " <td>Finance</td>\n",
206 " <td>E010</td>\n",
207 " <td>Charlie Brown</td>\n",
208 " <td>Disabled</td>\n",
209 " <td>2024-01-08</td>\n",
210 " </tr>\n",
211 " <tr>\n",
212 " <th>3</th>\n",
213 " <td>E012</td>\n",
214 " <td>David Miller</td>\n",
215 " <td>2023-10-20</td>\n",
216 " <td>Marketing</td>\n",
217 " <td>NaN</td>\n",
218 " <td>NaN</td>\n",
219 " <td>NaN</td>\n",
220 " <td>NaT</td>\n",
221 " </tr>\n",
222 " <tr>\n",
223 " <th>4</th>\n",
224 " <td>E015</td>\n",
225 " <td>Eve Wilson</td>\n",
226 " <td>2023-12-25</td>\n",
227 " <td>Engineering</td>\n",
228 " <td>E015</td>\n",
229 " <td>Eve Wilson</td>\n",
230 " <td>Active</td>\n",
231 " <td>2024-02-01</td>\n",
232 " </tr>\n",
233 " </tbody>\n",
234 "</table>\n",
235 "</div>"
236 ],
237 "text/plain": [
238 " Employee_ID Name Term_Date Department User_ID Full_Name \\\n",
239 "0 E001 Alice Smith 2023-11-15 Sales E001 Alice Smith \n",
240 "1 E005 Bob Johnson 2023-12-01 IT E005 Bob Johnson \n",
241 "2 E010 Charlie Brown 2024-01-10 Finance E010 Charlie Brown \n",
242 "3 E012 David Miller 2023-10-20 Marketing NaN NaN \n",
243 "4 E015 Eve Wilson 2023-12-25 Engineering E015 Eve Wilson \n",
244 "\n",
245 " Account_Status Last_Login \n",
246 "0 Active 2024-01-05 \n",
247 "1 Active 2023-11-28 \n",
248 "2 Disabled 2024-01-08 \n",
249 "3 NaN NaT \n",
250 "4 Active 2024-02-01 "
251 ]
252 },
253 "execution_count": 5,
254 "metadata": {},
255 "output_type": "execute_result"
256 }
257 ],
258 "source": [
259 "# We join on the ID. \n",
260 "# We use 'left' because we only care about people on the termination list.\n",
261 "audit_merge = pd.merge(\n",
262 " df_hr, \n",
263 " df_app, \n",
264 " left_on='Employee_ID', \n",
265 " right_on='User_ID', \n",
266 " how='left'\n",
267 ")\n",
268 "\n",
269 "# Display the merged table\n",
270 "audit_merge"
271 ]
272 },
273 {
274 "cell_type": "code",
275 "execution_count": 6,
276 "id": "5db62c12-6f2d-48cc-b008-39e135348f00",
277 "metadata": {},
278 "outputs": [
279 {
280 "name": "stdout",
281 "output_type": "stream",
282 "text": [
283 "Finding 1: 3 users still marked as 'Active'\n",
284 "Finding 2: 2 users logged in after termination\n"
285 ]
286 }
287 ],
288 "source": [
289 "# 1. Identify Terminated but still 'Active' in Application\n",
290 "active_leavers = audit_merge[audit_merge['Account_Status'] == 'Active'].copy()\n",
291 "\n",
292 "# 2. Identify Logins occurring AFTER termination date\n",
293 "# This is a critical security finding indicating potential account misuse\n",
294 "post_term_logins = audit_merge[audit_merge['Last_Login'] > audit_merge['Term_Date']].copy()\n",
295 "\n",
296 "print(f\"Finding 1: {len(active_leavers)} users still marked as 'Active'\")\n",
297 "print(f\"Finding 2: {len(post_term_logins)} users logged in after termination\")"
298 ]
299 },
300 {
301 "cell_type": "code",
302 "execution_count": 8,
303 "id": "67c23470-4c0d-422d-bd1a-b0ad19487aca",
304 "metadata": {},
305 "outputs": [
306 {
307 "name": "stdout",
308 "output_type": "stream",
309 "text": [
310 "Collecting openpyxl\n",
311 " Downloading openpyxl-3.1.5-py2.py3-none-any.whl.metadata (2.5 kB)\n",
312 "Collecting et-xmlfile (from openpyxl)\n",
313 " Downloading et_xmlfile-2.0.0-py3-none-any.whl.metadata (2.7 kB)\n",
314 "Downloading openpyxl-3.1.5-py2.py3-none-any.whl (250 kB)\n",
315 "Downloading et_xmlfile-2.0.0-py3-none-any.whl (18 kB)\n",
316 "Installing collected packages: et-xmlfile, openpyxl\n",
317 "\u001b[2K \u001b[38;2;114;156;31m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m2/2\u001b[0m [openpyxl]━━\u001b[0m \u001b[32m1/2\u001b[0m [openpyxl]\n",
318 "\u001b[1A\u001b[2KSuccessfully installed et-xmlfile-2.0.0 openpyxl-3.1.5\n",
319 "Note: you may need to restart the kernel to use updated packages.\n"
320 ]
321 }
322 ],
323 "source": [
324 "%pip install openpyxl"
325 ]
326 },
327 {
328 "cell_type": "code",
329 "execution_count": 9,
330 "id": "e2fd2073-04e0-41f9-b76f-1677967e96b9",
331 "metadata": {},
332 "outputs": [
333 {
334 "name": "stdout",
335 "output_type": "stream",
336 "text": [
337 "Audit Report Exported: Termination_Audit_Report.xlsx\n"
338 ]
339 }
340 ],
341 "source": [
342 "# Create a summary report\n",
343 "with pd.ExcelWriter('Termination_Audit_Report.xlsx') as writer:\n",
344 " active_leavers.to_excel(writer, sheet_name='Active_Leavers', index=False)\n",
345 " post_term_logins.to_excel(writer, sheet_name='Post_Term_Logins', index=False)\n",
346 " audit_merge.to_excel(writer, sheet_name='Full_Traceability_Matrix', index=False)\n",
347 "\n",
348 "print(\"Audit Report Exported: Termination_Audit_Report.xlsx\")"
349 ]
350 }
351 ],
352 "metadata": {
353 "kernelspec": {
354 "display_name": "Python 3 (ipykernel)",
355 "language": "python",
356 "name": "python3"
357 },
358 "language_info": {
359 "codemirror_mode": {
360 "name": "ipython",
361 "version": 3
362 },
363 "file_extension": ".py",
364 "mimetype": "text/x-python",
365 "name": "python",
366 "nbconvert_exporter": "python",
367 "pygments_lexer": "ipython3",
368 "version": "3.14.2"
369 }
370 },
371 "nbformat": 4,
372 "nbformat_minor": 5
373}
notebooks/terminations/app_users.csv added +7
@@ -0,0 +1,7 @@
1User_ID,Full_Name,Account_Status,Last_Login
2E001,Alice Smith,Active,2024-01-05
3E002,Frank Wright,Active,2024-02-12
4E005,Bob Johnson ,Active,2023-11-28
5E009,Grace Hopper,Active,2024-02-15
6E010,Charlie Brown,Disabled,2024-01-08
7E015,Eve Wilson,Active,2024-02-01
notebooks/terminations/hr_terminations.csv added +6
@@ -0,0 +1,6 @@
1Employee_ID,Name,Term_Date,Department
2E001,Alice Smith,2023-11-15,Sales
3E005,Bob Johnson,2023-12-01,IT
4E010,Charlie Brown,2024-01-10,Finance
5E012,David Miller,2023-10-20,Marketing
6E015,Eve Wilson,2023-12-25,Engineering