{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"An attempt to see if the tabular data alone could be enough to get a \"proper score\". It doesn't seem like it is but could be a starting point for someone else I guess. :-)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import Ridge\nfrom sklearn import svm\nimport matplotlib.pyplot as plt\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-05T14:53:59.374472Z","iopub.execute_input":"2023-01-05T14:53:59.375028Z","iopub.status.idle":"2023-01-05T14:54:00.119441Z","shell.execute_reply.started":"2023-01-05T14:53:59.374916Z","shell.execute_reply":"2023-01-05T14:54:00.118081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv').fillna(50)\ndf_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv').fillna(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:06:55.391352Z","iopub.execute_input":"2023-01-05T15:06:55.39173Z","iopub.status.idle":"2023-01-05T15:06:55.480885Z","shell.execute_reply.started":"2023-01-05T15:06:55.391701Z","shell.execute_reply":"2023-01-05T15:06:55.479659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = df.loc[df['cancer'] == 1]['age'].tolist()\nplt.hist(bins, 50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T14:54:02.088559Z","iopub.execute_input":"2023-01-05T14:54:02.089272Z","iopub.status.idle":"2023-01-05T14:54:02.438839Z","shell.execute_reply.started":"2023-01-05T14:54:02.089235Z","shell.execute_reply":"2023-01-05T14:54:02.43762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = df.loc[df['cancer'] == 0].loc[df['BIRADS'] != 0]['age'].tolist()\nplt.hist(bins, 70)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T14:54:04.360541Z","iopub.execute_input":"2023-01-05T14:54:04.360972Z","iopub.status.idle":"2023-01-05T14:54:04.880036Z","shell.execute_reply.started":"2023-01-05T14:54:04.360939Z","shell.execute_reply":"2023-01-05T14:54:04.878701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = df.loc[df['BIRADS'] == 2]['age'].tolist()\nplt.hist(bins, 50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T14:54:04.881736Z","iopub.execute_input":"2023-01-05T14:54:04.882154Z","iopub.status.idle":"2023-01-05T14:54:05.202999Z","shell.execute_reply.started":"2023-01-05T14:54:04.882119Z","shell.execute_reply":"2023-01-05T14:54:05.201693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df[:]\n\ncancer = df_train.loc[df_train['cancer'] == 1]\nnocancer = df_train.loc[df_train['cancer'] == 0].loc[df_train['BIRADS'] != 0]\n\ny_cancer = np.ones(len(cancer))\ny_nocancer = np.zeros(len(nocancer))\n\ncancer_siteid = np.asarray(cancer['site_id'].tolist())\nnocancer_siteid = np.asarray(nocancer['site_id'].tolist())\nX_siteid = np.concatenate((cancer_siteid, nocancer_siteid), axis=0).reshape(cancer_siteid.shape[0] + nocancer_siteid.shape[0], 1) / 2\n\nX_view = np.zeros((X_siteid.shape[0], 6))\ni = 0\nfor view in cancer['view'].tolist() + nocancer['view'].tolist():\n    if view == 'CC':\n        X_view[i, 0] = 1\n    elif view == 'MLO':\n        X_view[i, 1] = 1\n    elif view == 'ML':\n        X_view[i, 2] = 1\n    elif view == 'LM':\n        X_view[i, 3] = 1\n    elif view == 'AT':\n        X_view[i, 4] = 1\n    elif view == 'LMO':\n        X_view[i, 5] = 1\n        \n    i += 1\n        \ncancer_age = np.asarray(cancer['age'].tolist())\nnocancer_age = np.asarray(nocancer['age'].tolist())\nX_age = np.concatenate((cancer_age, nocancer_age), axis=0).reshape(cancer_siteid.shape[0] + nocancer_siteid.shape[0], 1) / 100\n\ncancer_implant = np.asarray(cancer['implant'].tolist())\nnocancer_implant = np.asarray(nocancer['implant'].tolist())\nX_implant = np.concatenate((cancer_implant, nocancer_implant), axis=0).reshape(cancer_siteid.shape[0] + nocancer_siteid.shape[0], 1)\n\ncancer_machineid = np.asarray(cancer['machine_id'].tolist())\nnocancer_machineid = np.asarray(nocancer['machine_id'].tolist())\nX_machineid = np.concatenate((cancer_machineid, nocancer_machineid), axis=0).reshape(cancer_siteid.shape[0] + nocancer_siteid.shape[0], 1) / 200\n\nX = np.concatenate((X_siteid, X_view, X_age, X_implant, X_machineid), axis=1)\ny = np.concatenate((y_cancer, y_nocancer), axis=0)\n\nclf = svm.NuSVC(nu=0.04, gamma=\"auto\", class_weight=\"balanced\", kernel=\"rbf\")\nclf.fit(X, y)\n\n#clf = Ridge(alpha=1.0)\n#clf.fit(X, y)\n\nnum_tests = len(df_test.index)\n\nX_siteid = np.asarray(df_test['site_id'].tolist()).reshape(num_tests, 1) / 2\n\nX_view = np.zeros((num_tests, 6))\ni = 0\nfor view in df_test['view'].tolist():\n    if view == 'CC':\n        X_view[i, 0] = 1\n    elif view == 'MLO':\n        X_view[i, 1] = 1\n    elif view == 'ML':\n        X_view[i, 2] = 1\n    elif view == 'LM':\n        X_view[i, 3] = 1\n    elif view == 'AT':\n        X_view[i, 4] = 1\n    elif view == 'LMO':\n        X_view[i, 5] = 1\n        \n    i += 1\n        \nX_age = np.asarray(df_test['age'].tolist()).reshape(num_tests, 1) / 100\nX_implant = np.asarray(df_test['implant'].tolist()).reshape(num_tests, 1)\nX_machineid = np.asarray(df_test['machine_id'].tolist()).reshape(num_tests, 1) / 200\n\nX = np.concatenate((X_siteid, X_view, X_age, X_implant, X_machineid), axis=1)\n\nsubmission_df = pd.DataFrame()\n\nsubmission_df['prediction_id'] = df_test['prediction_id']\nsubmission_df['cancer'] = clf.predict(X)\n\nsubmission_df = submission_df.groupby('prediction_id').max()\nsubmission_df['prediction_id'] = submission_df.index    #print(pre_submission_df)\n\n#submission_df['cancer'].where(submission_df['cancer'] > THRESHOLD, 1, inplace=True)\n#submission_df['cancer'].where(submission_df['cancer'] < THRESHOLD, 0, inplace=True)\n\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:06:58.521588Z","iopub.execute_input":"2023-01-05T15:06:58.522001Z","iopub.status.idle":"2023-01-05T15:07:04.887785Z","shell.execute_reply.started":"2023-01-05T15:06:58.521971Z","shell.execute_reply":"2023-01-05T15:07:04.886909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}