{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom sklearn.model_selection import KFold\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T18:14:40.869102Z","iopub.execute_input":"2022-07-23T18:14:40.869557Z","iopub.status.idle":"2022-07-23T18:14:41.494326Z","shell.execute_reply.started":"2022-07-23T18:14:40.869472Z","shell.execute_reply":"2022-07-23T18:14:41.492892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:    \n    # config\n    work_dir = '../input/mayo-clinic-strip-ai/'\n    nfolds = 5","metadata":{"execution":{"iopub.status.busy":"2022-07-23T18:14:41.496316Z","iopub.execute_input":"2022-07-23T18:14:41.496668Z","iopub.status.idle":"2022-07-23T18:14:41.50123Z","shell.execute_reply.started":"2022-07-23T18:14:41.496637Z","shell.execute_reply":"2022-07-23T18:14:41.500278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfx = pd.read_csv(CFG.work_dir + \"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T18:14:41.502383Z","iopub.execute_input":"2022-07-23T18:14:41.502912Z","iopub.status.idle":"2022-07-23T18:14:41.525406Z","shell.execute_reply.started":"2022-07-23T18:14:41.502883Z","shell.execute_reply":"2022-07-23T18:14:41.524516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split into folds\nkf = KFold(n_splits = 5, random_state = 42, shuffle = True)\nfold_id = np.zeros((len(dfx),1))\n\nfor (ii, (train_index, test_index)) in enumerate(kf.split(dfx)):\n    fold_id[test_index] = ii\n    \ndfx['fold'] = fold_id.astype(int)\n\n\ndfx.to_csv('xfolds.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T18:23:54.596092Z","iopub.execute_input":"2022-07-23T18:23:54.597083Z","iopub.status.idle":"2022-07-23T18:23:54.617682Z","shell.execute_reply.started":"2022-07-23T18:23:54.597032Z","shell.execute_reply":"2022-07-23T18:23:54.616284Z"},"trusted":true},"execution_count":null,"outputs":[]}]}