{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 📋 Table of Contents\n* [Import and Preview](#import)\n* [Association between Targets](#assoc)\n* [Cross Tabulations](#cross)\n* [Conditional Frequencies of Target given another Target](#cond_freq)\n* [Number of 1-targets by row](#count_1)\n* [Add/Evaluate Metadata](#meta)","metadata":{}},{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport time\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# statistics\nimport phik","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-29T13:27:53.958122Z","iopub.execute_input":"2023-07-29T13:27:53.959381Z","iopub.status.idle":"2023-07-29T13:27:53.966517Z","shell.execute_reply.started":"2023-07-29T13:27:53.959319Z","shell.execute_reply":"2023-07-29T13:27:53.964874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# configs\npd.set_option('display.max_columns', None) # we want to display all columns in this notebook\n\n# aesthetics\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:53.973241Z","iopub.execute_input":"2023-07-29T13:27:53.973754Z","iopub.status.idle":"2023-07-29T13:27:53.983889Z","shell.execute_reply.started":"2023-07-29T13:27:53.973716Z","shell.execute_reply":"2023-07-29T13:27:53.982503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='import'></a>\n# Import and Preview","metadata":{}},{"cell_type":"code","source":"# import training data\nfolder = 'rsna-2023-abdominal-trauma-detection'\ndf_train = pd.read_csv('../input/'+folder+'/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:53.986475Z","iopub.execute_input":"2023-07-29T13:27:53.987229Z","iopub.status.idle":"2023-07-29T13:27:54.006659Z","shell.execute_reply.started":"2023-07-29T13:27:53.987183Z","shell.execute_reply":"2023-07-29T13:27:54.005215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:54.010011Z","iopub.execute_input":"2023-07-29T13:27:54.011135Z","iopub.status.idle":"2023-07-29T13:27:54.038028Z","shell.execute_reply.started":"2023-07-29T13:27:54.011083Z","shell.execute_reply":"2023-07-29T13:27:54.036834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# overview\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:54.041289Z","iopub.execute_input":"2023-07-29T13:27:54.042012Z","iopub.status.idle":"2023-07-29T13:27:54.05813Z","shell.execute_reply.started":"2023-07-29T13:27:54.041971Z","shell.execute_reply":"2023-07-29T13:27:54.05697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets\ntargets = ['bowel_healthy', 'bowel_injury', \n           'extravasation_healthy', 'extravasation_injury',\n           'kidney_healthy', 'kidney_low', 'kidney_high',\n           'liver_healthy', 'liver_low', 'liver_high',\n           'spleen_healthy', 'spleen_low', 'spleen_high',\n           'any_injury']","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:54.059803Z","iopub.execute_input":"2023-07-29T13:27:54.060291Z","iopub.status.idle":"2023-07-29T13:27:54.066708Z","shell.execute_reply.started":"2023-07-29T13:27:54.060249Z","shell.execute_reply":"2023-07-29T13:27:54.065418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot targets\nfor t in targets:\n    plt.figure(figsize=(4,3))\n    df_train[t].value_counts().sort_index().plot(kind='bar', color=default_color_1)\n    plt.title(t)\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:54.067833Z","iopub.execute_input":"2023-07-29T13:27:54.068214Z","iopub.status.idle":"2023-07-29T13:27:57.146191Z","shell.execute_reply.started":"2023-07-29T13:27:54.068184Z","shell.execute_reply":"2023-07-29T13:27:57.145063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='assoc'></a>\n# Association between Targets","metadata":{}},{"cell_type":"markdown","source":"### Using the Phi_K coefficient to measure associations between the targets\nFor details see https://phik.readthedocs.io/en/latest/.","metadata":{}},{"cell_type":"code","source":"# calc Phi_K matrix\nphiK_mat_train = df_train[targets].phik_matrix(interval_cols=[])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-29T13:27:57.147801Z","iopub.execute_input":"2023-07-29T13:27:57.148722Z","iopub.status.idle":"2023-07-29T13:27:59.005889Z","shell.execute_reply.started":"2023-07-29T13:27:57.14867Z","shell.execute_reply":"2023-07-29T13:27:59.004704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize phi_K matrix\nplt.figure(figsize=(10,9))\nsns.heatmap(phiK_mat_train, annot=True,\n            fmt='.3f',\n            linecolor='black', linewidths=.5,\n            cmap='Greens', vmin=0, vmax=+1)\nplt.title('Phi_K correlation of targets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:27:59.009075Z","iopub.execute_input":"2023-07-29T13:27:59.010122Z","iopub.status.idle":"2023-07-29T13:28:00.064361Z","shell.execute_reply.started":"2023-07-29T13:27:59.010057Z","shell.execute_reply":"2023-07-29T13:28:00.062923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='cross'></a>\n# Cross Tabulations","metadata":{}},{"cell_type":"code","source":"def show_crosstab(t1,t2):\n    ctab = pd.crosstab(df_train[t1], df_train[t2])\n    ctab_norm = pd.crosstab(df_train[t1], df_train[t2], normalize=True)\n    print(ctab)\n    print()\n    print('Normalized View:')\n    print(ctab_norm)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:00.068344Z","iopub.execute_input":"2023-07-29T13:28:00.068744Z","iopub.status.idle":"2023-07-29T13:28:00.075721Z","shell.execute_reply.started":"2023-07-29T13:28:00.068711Z","shell.execute_reply":"2023-07-29T13:28:00.074111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example 1\nt1,t2 = 'any_injury','spleen_healthy'\nshow_crosstab(t1,t2)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:00.077616Z","iopub.execute_input":"2023-07-29T13:28:00.079552Z","iopub.status.idle":"2023-07-29T13:28:00.11644Z","shell.execute_reply.started":"2023-07-29T13:28:00.079487Z","shell.execute_reply":"2023-07-29T13:28:00.115262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example 2\nt1,t2 = 'liver_healthy','spleen_healthy'\nshow_crosstab(t1,t2)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:00.117893Z","iopub.execute_input":"2023-07-29T13:28:00.118254Z","iopub.status.idle":"2023-07-29T13:28:00.146748Z","shell.execute_reply.started":"2023-07-29T13:28:00.118224Z","shell.execute_reply":"2023-07-29T13:28:00.145233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='cond_freq'></a>\n# Conditional frequencies of Target given another Target","metadata":{}},{"cell_type":"code","source":"# use correlation matrix as container for the frequencies\ncond_T = df_train[targets].corr()\nnn = len(targets)\n# calc frequency of x [column] given y [row] for all pairs\nfor i in range(1,nn+1):\n    for j in range(1,nn+1):\n       if (i!=j):\n        f1 = targets[i-1]\n        f2 = targets[j-1]\n        ctab = pd.crosstab(df_train[f1], df_train[f2])\n        n_1 = df_train[f1].sum() # feature 1 = 1\n        n_both = ctab.iloc[1,1]  # both features = 1\n        perc_2_given_1 = n_both / n_1 # feature_2 = 1 given feature_1 = 1\n        cond_T.loc[f1,f2] = perc_2_given_1 # store value in correlation matrix\n\n# plot values as matrix\nplt.figure(figsize=(10,9))\nsns.heatmap(cond_T, annot=True, \n            cmap='Greens', fmt='.3f',\n            linecolor='black', linewidths=.5)\nplt.title('Conditional Frequencies of Targets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:00.148839Z","iopub.execute_input":"2023-07-29T13:28:00.149246Z","iopub.status.idle":"2023-07-29T13:28:03.150728Z","shell.execute_reply.started":"2023-07-29T13:28:00.149214Z","shell.execute_reply":"2023-07-29T13:28:03.149145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Examples:\n* Given liver_healthy=1 we observe spleen_healthy=1 in 89.3% of cases.\n* Given spleen_healthy=1 we observe liver_healthy=1 in 90.3% of cases.","metadata":{}},{"cell_type":"markdown","source":"<a id='count_1'></a>\n# Number of 1-targets by row","metadata":{}},{"cell_type":"code","source":"# count number of targets being 1 in each row\ndf_train['sum_all_targets'] = df_train[targets].sum(axis=1)\n# and show corresponding stats\nplt.figure(figsize=(5,4))\ntarget_counts = df_train['sum_all_targets'].value_counts().sort_index()\nprint(target_counts)\ntarget_counts.plot(kind='bar', color=default_color_1)\nplt.title('Number of targets=1 per row')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:03.152561Z","iopub.execute_input":"2023-07-29T13:28:03.152973Z","iopub.status.idle":"2023-07-29T13:28:03.371767Z","shell.execute_reply.started":"2023-07-29T13:28:03.15294Z","shell.execute_reply":"2023-07-29T13:28:03.370138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='meta'></a>\n# Add/Evaluate Metadata","metadata":{}},{"cell_type":"code","source":"# import meta data\ndf_train_meta = pd.read_csv('../input/'+folder+'/train_series_meta.csv')\ndf_train_meta.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:03.373582Z","iopub.execute_input":"2023-07-29T13:28:03.37466Z","iopub.status.idle":"2023-07-29T13:28:03.395156Z","shell.execute_reply.started":"2023-07-29T13:28:03.374609Z","shell.execute_reply":"2023-07-29T13:28:03.393967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot binary feature 'incomplete_organ'\nplt.figure(figsize=(4,3))\ndf_train_meta.incomplete_organ.value_counts().plot(kind='bar', color=default_color_1)\nplt.title('incomplete_organ')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:03.396598Z","iopub.execute_input":"2023-07-29T13:28:03.39696Z","iopub.status.idle":"2023-07-29T13:28:03.602387Z","shell.execute_reply.started":"2023-07-29T13:28:03.396925Z","shell.execute_reply":"2023-07-29T13:28:03.601185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at aortic_hu distribution\nplt.figure(figsize=(7,1))\ndf_train_meta.aortic_hu.plot(kind='box', vert=False)\nplt.title('aortic_hu')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:03.603623Z","iopub.execute_input":"2023-07-29T13:28:03.603976Z","iopub.status.idle":"2023-07-29T13:28:03.786909Z","shell.execute_reply.started":"2023-07-29T13:28:03.603945Z","shell.execute_reply":"2023-07-29T13:28:03.785791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove negative outlier\ndf_train_meta = df_train_meta[df_train_meta.aortic_hu>0]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:28:03.788385Z","iopub.execute_input":"2023-07-29T13:28:03.788715Z","iopub.status.idle":"2023-07-29T13:28:03.797284Z","shell.execute_reply.started":"2023-07-29T13:28:03.788685Z","shell.execute_reply":"2023-07-29T13:28:03.795667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aggregation on one row per patient id\ndf_train_meta_agg = df_train_meta.groupby(by='patient_id', as_index=False).agg(\n    n_values = pd.NamedAgg(column='aortic_hu', aggfunc='count'),\n    mean_aortic_hu = pd.NamedAgg(column='aortic_hu', aggfunc=np.mean),\n    min_aortic_hu = pd.NamedAgg(column='aortic_hu', aggfunc=min),\n    max_aortic_hu = pd.NamedAgg(column='aortic_hu', aggfunc=max),\n    mean_incomplete_organ = pd.NamedAgg(column='incomplete_organ', aggfunc=np.mean))","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:31:14.752544Z","iopub.execute_input":"2023-07-29T13:31:14.753003Z","iopub.status.idle":"2023-07-29T13:31:14.773764Z","shell.execute_reply.started":"2023-07-29T13:31:14.752968Z","shell.execute_reply":"2023-07-29T13:31:14.771675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We have either 1 or 2 observations per patient:","metadata":{}},{"cell_type":"code","source":"# stats for number of values\ndf_train_meta_agg.n_values.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:32:39.16623Z","iopub.execute_input":"2023-07-29T13:32:39.166733Z","iopub.status.idle":"2023-07-29T13:32:39.177894Z","shell.execute_reply.started":"2023-07-29T13:32:39.166696Z","shell.execute_reply":"2023-07-29T13:32:39.176486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add aggregated meta data to training data\ndf_train_comb = pd.merge(df_train, df_train_meta_agg)\ndf_train_comb.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:31:16.50353Z","iopub.execute_input":"2023-07-29T13:31:16.504037Z","iopub.status.idle":"2023-07-29T13:31:16.536465Z","shell.execute_reply.started":"2023-07-29T13:31:16.503998Z","shell.execute_reply":"2023-07-29T13:31:16.535252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot aortic_hu vs Targets:","metadata":{}},{"cell_type":"code","source":"# violinplot for each target\nfor t in targets:\n    plt.figure(figsize=(4,4))\n    sns.violinplot(data=df_train_comb, x=t, y='mean_aortic_hu')\n    plt.title(t)\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T13:33:59.990678Z","iopub.execute_input":"2023-07-29T13:33:59.991308Z","iopub.status.idle":"2023-07-29T13:34:04.125635Z","shell.execute_reply.started":"2023-07-29T13:33:59.991266Z","shell.execute_reply":"2023-07-29T13:34:04.124016Z"},"trusted":true},"execution_count":null,"outputs":[]}]}