{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n\n**Simple base line wiht Logistic Regression**\n\n**Only Using Feature with Age & Implant !**","metadata":{}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\nimport sys\nimport random as rnd\nimport gc\nfrom time import time\nimport copy\nimport pandas as pd\nimport numpy as np\nfrom numpy import random as np_rnd\nimport pickle\nfrom tqdm import tqdm\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\nimport sklearn as skl\nfrom sklearn import linear_model as lm\nfrom sklearn.model_selection import StratifiedKFold\nfrom itertools import permutations, combinations\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder, OneHotEncoder\nfrom sklearn import metrics\n\nimport optuna\nfrom optuna import Trial, create_study\nfrom optuna.samplers import TPESampler\n\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cat\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-29T13:40:29.841814Z","iopub.execute_input":"2023-01-29T13:40:29.842218Z","iopub.status.idle":"2023-01-29T13:40:32.461307Z","shell.execute_reply.started":"2023-01-29T13:40:29.842139Z","shell.execute_reply":"2023-01-29T13:40:32.460617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"w\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n\ndef diff(first, second):\n    second = set(second)\n    return [item for item in first if item not in second]\n\ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.462726Z","iopub.execute_input":"2023-01-29T13:40:32.463107Z","iopub.status.idle":"2023-01-29T13:40:32.473008Z","shell.execute_reply.started":"2023-01-29T13:40:32.463085Z","shell.execute_reply":"2023-01-29T13:40:32.471459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = True\n    common_vars = [\"age\", \"implant\"]\n    target_list = [\"cancer\", \"biopsy\", \"invasive\", \"BIRADS\", \"difficult_negative_case\"]\n    \n    threshold = 0.5\n    n_folds = 5","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.474119Z","iopub.execute_input":"2023-01-29T13:40:32.474478Z","iopub.status.idle":"2023-01-29T13:40:32.495722Z","shell.execute_reply.started":"2023-01-29T13:40:32.474451Z","shell.execute_reply":"2023-01-29T13:40:32.494973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating data pipeline","metadata":{}},{"cell_type":"code","source":"# site_id - ID code for the source hospital.\n# patient_id - ID code for the patient.\n# image_id - ID code for the image.\n# laterality - Whether the image is of the left or right breast.\n# view - The orientation of the image. The default for a screening exam is to capture two views per breast.\n# age - The patient's age in years.\n# implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n# density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n# machine_id - An ID code for the imaging device.\n# cancer - Whether or not the breast was positive for malignant cancer. The target value. Only provided for train.\n# biopsy - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n# invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n# BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n# prediction_id - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n# difficult_negative_case - True if the case was unusually difficult. Only provided for train.\n\n# target : cancer (binary),biopsy, invasive, BIRADAS, difficult_negative_case","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-01-29T13:40:32.497862Z","iopub.execute_input":"2023-01-29T13:40:32.49835Z","iopub.status.idle":"2023-01-29T13:40:32.510509Z","shell.execute_reply.started":"2023-01-29T13:40:32.498323Z","shell.execute_reply":"2023-01-29T13:40:32.509485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.511606Z","iopub.execute_input":"2023-01-29T13:40:32.511887Z","iopub.status.idle":"2023-01-29T13:40:32.612266Z","shell.execute_reply.started":"2023-01-29T13:40:32.511863Z","shell.execute_reply":"2023-01-29T13:40:32.61141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.613439Z","iopub.execute_input":"2023-01-29T13:40:32.614207Z","iopub.status.idle":"2023-01-29T13:40:32.643676Z","shell.execute_reply.started":"2023-01-29T13:40:32.614168Z","shell.execute_reply":"2023-01-29T13:40:32.642526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Simple Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_full = df_full[[\"patient_id\"] + CFG.common_vars + [\"cancer\"]].groupby(\"patient_id\").max()\ndf_full = df_full.dropna().drop_duplicates().reset_index(drop=True)\ndf_full[\"age_discrete\"] = pd.cut(df_full[\"age\"], bins=[-np.inf] + list(range(20, 81, 5)) + [np.inf], right=False).astype(\"object\")\n\ndf_full = df_full.sample(frac=1, random_state=42).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.64493Z","iopub.execute_input":"2023-01-29T13:40:32.645203Z","iopub.status.idle":"2023-01-29T13:40:32.681297Z","shell.execute_reply.started":"2023-01-29T13:40:32.645179Z","shell.execute_reply":"2023-01-29T13:40:32.680398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.682569Z","iopub.execute_input":"2023-01-29T13:40:32.682924Z","iopub.status.idle":"2023-01-29T13:40:32.6993Z","shell.execute_reply.started":"2023-01-29T13:40:32.682892Z","shell.execute_reply":"2023-01-29T13:40:32.698216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ohe = OneHotEncoder(sparse=False, handle_unknown=\"ignore\", dtype=\"float32\")\nohe.fit(df_full[[\"age_discrete\"]])\ndf_full[[\"oh_\" + str(j) for j in range(sum([len(i) for i in ohe.categories_]))]] = ohe.transform(df_full[[\"age_discrete\"]])\ndf_full = df_full.drop([\"age_discrete\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.700648Z","iopub.execute_input":"2023-01-29T13:40:32.700882Z","iopub.status.idle":"2023-01-29T13:40:32.715051Z","shell.execute_reply.started":"2023-01-29T13:40:32.700861Z","shell.execute_reply":"2023-01-29T13:40:32.713748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full[\"age\"] /= 100","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.718684Z","iopub.execute_input":"2023-01-29T13:40:32.719217Z","iopub.status.idle":"2023-01-29T13:40:32.72445Z","shell.execute_reply.started":"2023-01-29T13:40:32.71919Z","shell.execute_reply":"2023-01-29T13:40:32.723449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full_x = df_full.drop(\"cancer\", axis=1).astype(\"float32\")\ndf_full_y = df_full[\"cancer\"].astype(\"int32\")\ndel df_full; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.727222Z","iopub.execute_input":"2023-01-29T13:40:32.727573Z","iopub.status.idle":"2023-01-29T13:40:32.868759Z","shell.execute_reply.started":"2023-01-29T13:40:32.727546Z","shell.execute_reply":"2023-01-29T13:40:32.867569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full_x.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.870186Z","iopub.execute_input":"2023-01-29T13:40:32.870556Z","iopub.status.idle":"2023-01-29T13:40:32.898764Z","shell.execute_reply.started":"2023-01-29T13:40:32.870525Z","shell.execute_reply":"2023-01-29T13:40:32.897557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full_y.value_counts(True)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.900564Z","iopub.execute_input":"2023-01-29T13:40:32.900936Z","iopub.status.idle":"2023-01-29T13:40:32.908046Z","shell.execute_reply.started":"2023-01-29T13:40:32.900903Z","shell.execute_reply":"2023-01-29T13:40:32.907425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_vars = []","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.90892Z","iopub.execute_input":"2023-01-29T13:40:32.909498Z","iopub.status.idle":"2023-01-29T13:40:32.918952Z","shell.execute_reply.started":"2023-01-29T13:40:32.909473Z","shell.execute_reply":"2023-01-29T13:40:32.918099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"valid_pred = np.zeros((len(df_full_x,)))\nmodel_name_list = [\"logistic\"]\nft_list = {i: [] for i in model_name_list}\nmodel_list = {i: [] for i in model_name_list}\nscore_list = {i: [] for i in model_name_list}\nkfolds_spliter = skl.model_selection.StratifiedKFold(CFG.n_folds, shuffle=True, random_state=42)\n\nseed_everything()\nfor fold, (train_idx, valid_idx) in enumerate(kfolds_spliter.split(df_full_x, df_full_y)):\n    scaler = MinMaxScaler()\n    \n    df_train_x = df_full_x.iloc[train_idx]\n    df_train_x[scaled_vars] = scaler.fit_transform(df_train_x[scaled_vars]) if len(scaled_vars) > 0 else df_train_x[scaled_vars]\n    df_train_y = df_full_y.iloc[train_idx]\n    \n    df_valid_x = df_full_x.iloc[valid_idx]\n    df_valid_x[scaled_vars] = scaler.transform(df_valid_x[scaled_vars]) if len(scaled_vars) > 0 else df_valid_x[scaled_vars]\n    df_valid_y = df_full_y.iloc[valid_idx]\n    \n    model = lm.LogisticRegression(penalty=\"none\", class_weight=\"balanced\", random_state=42)\n    model.fit(df_train_x, df_train_y)\n    \n    valid_pred[valid_idx] = model.predict_proba(df_valid_x)[:, 1]\n    y_pred_class = np.array([1 if i > CFG.threshold else 0 for i in valid_pred[valid_idx]], dtype=\"int32\")\n    \n    score_list[\"logistic\"].append({\n        \"logloss\": metrics.log_loss(df_valid_y.values, valid_pred[valid_idx]),\n        \"roc_aud\": metrics.roc_auc_score(df_valid_y.values, valid_pred[valid_idx]),\n        \"acc\": metrics.accuracy_score(df_valid_y.values, y_pred_class),\n        \"f1\": metrics.f1_score(df_valid_y.values, y_pred_class, average=\"binary\"),\n    })\n    model_list[\"logistic\"].append(model)\n    ft_list[\"logistic\"].append({\"scaler\": scaler})","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:32.920324Z","iopub.execute_input":"2023-01-29T13:40:32.921203Z","iopub.status.idle":"2023-01-29T13:40:33.006505Z","shell.execute_reply.started":"2023-01-29T13:40:32.921169Z","shell.execute_reply":"2023-01-29T13:40:33.005101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_score = pd.DataFrame(score_list[\"logistic\"])\ndf_score.loc[\"average\"] = df_score.iloc[:CFG.n_folds].mean(axis=0)\nprint(df_score)\ndf_score.to_csv(\"./df_score.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.007599Z","iopub.execute_input":"2023-01-29T13:40:33.007876Z","iopub.status.idle":"2023-01-29T13:40:33.022092Z","shell.execute_reply.started":"2023-01-29T13:40:33.007851Z","shell.execute_reply":"2023-01-29T13:40:33.02115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(valid_pred > CFG.threshold).mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.023164Z","iopub.execute_input":"2023-01-29T13:40:33.023474Z","iopub.status.idle":"2023-01-29T13:40:33.031109Z","shell.execute_reply.started":"2023-01-29T13:40:33.023444Z","shell.execute_reply":"2023-01-29T13:40:33.030058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf_test[\"age\"] = df_test[\"age\"].fillna((df_full_x[\"age\"] * 100).median())\ndf_test[\"implant\"] = df_test[\"implant\"].fillna(0)\nsubmission_idx = df_test[\"prediction_id\"]","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.032439Z","iopub.execute_input":"2023-01-29T13:40:33.03277Z","iopub.status.idle":"2023-01-29T13:40:33.051742Z","shell.execute_reply.started":"2023-01-29T13:40:33.032738Z","shell.execute_reply":"2023-01-29T13:40:33.050732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test[CFG.common_vars]\ndf_test[\"age_discrete\"] = pd.cut(df_test[\"age\"], bins=[-np.inf] + list(range(20, 81, 5)) + [np.inf], right=False).astype(\"object\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.05284Z","iopub.execute_input":"2023-01-29T13:40:33.053727Z","iopub.status.idle":"2023-01-29T13:40:33.063091Z","shell.execute_reply.started":"2023-01-29T13:40:33.053692Z","shell.execute_reply":"2023-01-29T13:40:33.062257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[[\"oh_\" + str(j) for j in range(sum([len(i) for i in ohe.categories_]))]] = ohe.transform(df_test[[\"age_discrete\"]])\ndf_test = df_test.drop([\"age_discrete\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.0642Z","iopub.execute_input":"2023-01-29T13:40:33.065242Z","iopub.status.idle":"2023-01-29T13:40:33.081931Z","shell.execute_reply.started":"2023-01-29T13:40:33.065209Z","shell.execute_reply":"2023-01-29T13:40:33.080638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"age\"] /= 100","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.083877Z","iopub.execute_input":"2023-01-29T13:40:33.084673Z","iopub.status.idle":"2023-01-29T13:40:33.091885Z","shell.execute_reply.started":"2023-01-29T13:40:33.084631Z","shell.execute_reply":"2023-01-29T13:40:33.090488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.astype(\"float32\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.093551Z","iopub.execute_input":"2023-01-29T13:40:33.093915Z","iopub.status.idle":"2023-01-29T13:40:33.102193Z","shell.execute_reply.started":"2023-01-29T13:40:33.09388Z","shell.execute_reply":"2023-01-29T13:40:33.101455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.103428Z","iopub.execute_input":"2023-01-29T13:40:33.1039Z","iopub.status.idle":"2023-01-29T13:40:33.128778Z","shell.execute_reply.started":"2023-01-29T13:40:33.10387Z","shell.execute_reply":"2023-01-29T13:40:33.12792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = np.zeros((len(df_test,)))\n\nseed_everything()\nfor fold in range(CFG.n_folds):\n    scaler = ft_list[\"logistic\"][fold][\"scaler\"]\n    \n    df_test_x = df_test.copy()\n    \n    test_pred[:] += model_list[\"logistic\"][fold].predict_proba(df_test_x)[:, 1] / CFG.n_folds\n\ny_pred_class = np.array([1 if i > CFG.threshold else 0 for i in test_pred], dtype=\"int32\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.129847Z","iopub.execute_input":"2023-01-29T13:40:33.130112Z","iopub.status.idle":"2023-01-29T13:40:33.144673Z","shell.execute_reply.started":"2023-01-29T13:40:33.130089Z","shell.execute_reply":"2023-01-29T13:40:33.143666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_pred[:5])\nprint(y_pred_class[:5])","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.145734Z","iopub.execute_input":"2023-01-29T13:40:33.146888Z","iopub.status.idle":"2023-01-29T13:40:33.154092Z","shell.execute_reply.started":"2023-01-29T13:40:33.146839Z","shell.execute_reply":"2023-01-29T13:40:33.153133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{"execution":{"iopub.status.busy":"2023-01-08T10:15:07.504089Z","iopub.execute_input":"2023-01-08T10:15:07.505939Z","iopub.status.idle":"2023-01-08T10:15:07.510435Z","shell.execute_reply.started":"2023-01-08T10:15:07.505894Z","shell.execute_reply":"2023-01-08T10:15:07.509152Z"}}},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\ntest_pred = pd.Series(test_pred, index=submission_idx).fillna(0.5)\ntest_pred.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.155153Z","iopub.execute_input":"2023-01-29T13:40:33.155514Z","iopub.status.idle":"2023-01-29T13:40:33.175709Z","shell.execute_reply.started":"2023-01-29T13:40:33.155471Z","shell.execute_reply":"2023-01-29T13:40:33.174816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = test_pred.groupby(\"prediction_id\").mean()\nsubmission[\"cancer\"] = submission[\"prediction_id\"].map(test_pred)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.177191Z","iopub.execute_input":"2023-01-29T13:40:33.177801Z","iopub.status.idle":"2023-01-29T13:40:33.190714Z","shell.execute_reply.started":"2023-01-29T13:40:33.177769Z","shell.execute_reply":"2023-01-29T13:40:33.189603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.195285Z","iopub.execute_input":"2023-01-29T13:40:33.195651Z","iopub.status.idle":"2023-01-29T13:40:33.200496Z","shell.execute_reply.started":"2023-01-29T13:40:33.195621Z","shell.execute_reply":"2023-01-29T13:40:33.199629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head ./submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-01-29T13:40:33.201831Z","iopub.execute_input":"2023-01-29T13:40:33.202325Z","iopub.status.idle":"2023-01-29T13:40:33.489228Z","shell.execute_reply.started":"2023-01-29T13:40:33.202289Z","shell.execute_reply":"2023-01-29T13:40:33.488113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}