{"cells":[{"metadata":{},"cell_type":"markdown","source":"# [2020-PE] Preprocessing train table\n\nKernel for modifying the train data table to fit to the preprocessed dataset generated in\n\nhttps://www.kaggle.com/spacelx/2020-pe-preprocessing-train-data\n\nThe full dataset is available at\n\nhttps://www.kaggle.com/spacelx/2020pe-preprocessed-train-data"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n# path management\nfrom pathlib import Path\n\n# progress bars\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"comp_data_path = Path('../input/rsna-str-pulmonary-embolism-detection')\nprep_data_path = Path('../input/2020pe-preprocessed-train-data')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# set sizing\nNSCANS = 20\nNPX = 128","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load train data table\ntrain = pd.read_csv(comp_data_path / 'train.csv')\n# put data file names into dataframes\ntrain['dcmpath'] = train.StudyInstanceUID + '_' + train.SeriesInstanceUID","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# modify train table to make it fit to our model (combine images to make a set of 20 for each exam)\nallsamples = np.unique(train.dcmpath.values)\ntrain_new = pd.DataFrame()\nfor sss in tqdm(allsamples):\n    selec = train[train['dcmpath'] == sss]\n    thisdata = selec.iloc[0].copy()\n\n    # get order of files in exam\n    thisfilelist = np.load(str(prep_data_path / f'proc_{NSCANS}_{NPX}_train' / (thisdata['dcmpath'] + '_list.npy')), allow_pickle=True)\n    thisfilelist = [str(f).split('/')[-1].split('.')[0] for f in thisfilelist]\n    # get corresponding order of PE observation true/false\n    ordered_obs = np.array([selec[selec['SOPInstanceUID'] == f]['pe_present_on_image'].values for f in thisfilelist]).flatten()\n    # split in 20 equal sections as done for the images\n    split = np.linspace(0, len(ordered_obs), num=NSCANS+1).astype(int)\n    pe_obs_binned = np.zeros((NSCANS))\n    for sss in range(NSCANS):\n        pe_obs_binned[sss] = int(np.mean(ordered_obs[split[sss]:split[sss+1]]) > 0.3)\n\n    # add binned PE observations to dataframe\n    for iii in range(NSCANS):\n        thisdata[f'pe_in_image_bin_{iii}'] = pe_obs_binned[iii]\n    # add acute PE label\n    thisdata['acute_pe'] = ((thisdata['negative_exam_for_pe'] == 0) &\n                            (thisdata['indeterminate'] == 0) &\n                            (thisdata['chronic_pe'] == 0) &\n                            (thisdata['acute_and_chronic_pe'] == 0)\n                           ).astype(int)\n    train_new = train_new.append(thisdata, ignore_index=True)\n\n# drop unneeded labels\ndrop_labels = ['qa_motion', 'qa_contrast', 'flow_artifact', 'pe_present_on_image', 'true_filling_defect_not_pe', 'SOPInstanceUID', 'SeriesInstanceUID', 'StudyInstanceUID']\ntrain_new.drop(labels=drop_labels, axis=1, inplace=True)\ntrain_new.to_csv('train_proc.csv')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}