{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install mmengine","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-04T22:51:18.443787Z","iopub.execute_input":"2022-12-04T22:51:18.444259Z","iopub.status.idle":"2022-12-04T22:51:33.532588Z","shell.execute_reply.started":"2022-12-04T22:51:18.444167Z","shell.execute_reply":"2022-12-04T22:51:33.531118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mmengine\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ngp = df.groupby('patient_id')['cancer'].max().reset_index()\ngp","metadata":{"execution":{"iopub.status.busy":"2022-12-04T22:54:24.332022Z","iopub.execute_input":"2022-12-04T22:54:24.332482Z","iopub.status.idle":"2022-12-04T22:54:24.414116Z","shell.execute_reply.started":"2022-12-04T22:54:24.332447Z","shell.execute_reply":"2022-12-04T22:54:24.413009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split = train_test_split(gp.patient_id.values,gp.cancer.values, test_size=2000, random_state=12345)#, stratify=True)\nprint(len(split[0]), len(split[1]))","metadata":{"execution":{"iopub.status.busy":"2022-12-04T22:54:57.154722Z","iopub.execute_input":"2022-12-04T22:54:57.155158Z","iopub.status.idle":"2022-12-04T22:54:57.164679Z","shell.execute_reply.started":"2022-12-04T22:54:57.155124Z","shell.execute_reply":"2022-12-04T22:54:57.163442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mmengine\nmmengine.dump(df[df.patient_id.isin(split[0])].reset_index(drop=True), 'train2.pkl')\nmmengine.dump(df[df.patient_id.isin(split[1])].reset_index(drop=True), 'val2.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-12-01T03:21:19.5921Z","iopub.execute_input":"2022-12-01T03:21:19.592585Z","iopub.status.idle":"2022-12-01T03:21:19.651872Z","shell.execute_reply.started":"2022-12-01T03:21:19.592544Z","shell.execute_reply":"2022-12-01T03:21:19.650652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mmengine.load('train2.pkl').head()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T03:21:31.950463Z","iopub.execute_input":"2022-12-01T03:21:31.95168Z","iopub.status.idle":"2022-12-01T03:21:32.013324Z","shell.execute_reply.started":"2022-12-01T03:21:31.951637Z","shell.execute_reply":"2022-12-01T03:21:32.012033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}