{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\nimport sys\nimport random as rnd\nimport gc\nfrom time import time\nimport copy\nimport pandas as pd\nimport numpy as np\nfrom numpy import random as np_rnd\nimport pickle\nfrom tqdm import tqdm\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\nimport sklearn as skl\nfrom sklearn.model_selection import StratifiedKFold\nfrom itertools import permutations, combinations\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder, OneHotEncoder\nfrom sklearn import metrics\n\nimport optuna\nfrom optuna import Trial, create_study\nfrom optuna.samplers import TPESampler\n\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cat\n\nfrom scipy import stats\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"w\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n\ndef diff(first, second):\n    second = set(second)\n    return [item for item in first if item not in second]\n\ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = True\n#     common_vars = [\"laterality\", \"view\", \"age\", \"implant\"]\n    common_vars = [\"age\", \"implant\"]\n    target_list = [\"cancer\", \"biopsy\", \"invasive\", \"BIRADS\", \"difficult_negative_case\"]\n    \n    threshold = 0.5\n    n_folds = 5","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# site_id - ID code for the source hospital.\n# patient_id - ID code for the patient.\n# image_id - ID code for the image.\n# laterality - Whether the image is of the left or right breast.\n# view - The orientation of the image. The default for a screening exam is to capture two views per breast.\n# age - The patient's age in years.\n# implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n\n# <target values>\n\n# 유방 조직의 밀집도를 나타냅니다. A에서 D로 갈수록 밀집도가 높다는 뜻입니다. 밀집도가 높으면 암 발견확률이 낮아집니다.\n# 즉 A에서 D로 갈수록 암 발견확률이 낮아집니다.\n# density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n\n# machine_id - An ID code for the imaging device.\n\n# 암 진단 여부 입니다. 1은 양성 0은 음성입니다.\n# cancer - Whether or not the breast was positive for malignant cancer. The target value. Only provided for train.\n\n# 후속 조직검사 수행 여부입니다. 1은 수행 0은 미수행을 나타냅니다.\n# if higher, more dangerous\n# biopsy - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n\n# 유방암이 진단되었을 경우, 다른 장기로의 전파력을 나타냅니다.\n# if higher, more dangerous\n# invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n\n# BIRADS 유방암에 대한 진단 level 체계 입니다. 0은 세부진단요구, 1은 음성, 2는 정상을 의미합니다.\n# if higher, less dangerous\n# BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n\n# 예측치를 산출해야 하는 unique한 id 입니다.\n# prediction_id - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n\n# 음성 진단의 어려움 정도를 나타냅니다. 1인 경우 음성이라고 보기 어려운 case, 0인 경우\n# if higher, more dangerous\n# difficult_negative_case - True if the case was unusually difficult. Only provided for train.\n\n# target : cancer (binary),biopsy, invasive, BIRADAS, difficult_negative_case","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA - Meta Data\n\n* We should submit the output with each shots' label (Left & Right)","metadata":{}},{"cell_type":"code","source":"df_full = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Age","metadata":{}},{"cell_type":"code","source":"df_eda = df_full.dropna(subset=[\"age\"]).copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Stats about age\")\ndf_eda.groupby(\"patient_id\")[\"age\"].mean().describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 6))\nsns.histplot(df_eda[\"age\"])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Stats about Ccancer diagnosis ratio by views\")\ndf_eda = df_eda.groupby(\"patient_id\")[[\"cancer\", \"age\"]].mean()\ndisplay(df_eda[\"cancer\"].describe())\ndf_eda[\"cancer\"] = df_eda[\"cancer\"].apply(lambda x: 1 if x > 0.0 else 0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 6))\nsns.boxplot(x=df_eda[\"cancer\"], y=df_eda[\"age\"])\n\nplt.xlabel(ax.get_xlabel(), fontsize=14)\nplt.ylabel(ax.get_ylabel(), fontsize=14)\n\nplt.xticks(fontsize=12, fontweight=\"bold\")\nplt.yticks(fontsize=12, fontweight=\"bold\")\n\nplt.title(\"Age by Cancer diagnosis\", fontsize=20, pad=20)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"=== one-way ANOVA ===\")\n\nF_statistic, pVal = stats.f_oneway(df_eda[\"age\"][df_eda[\"cancer\"] == 0].values, df_eda[\"age\"][df_eda[\"cancer\"] == 1].values)\n\nprint('Output: F={0:.1f}, p={1:.5f}'.format(F_statistic, pVal))\nif pVal < 0.05:\n    print('Result: Statistically significant, Age is significant factor on cancer.')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Implant","metadata":{}},{"cell_type":"code","source":"df_eda = df_full.dropna(subset=[\"implant\"]).copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Stats about implant\")\ndf_eda.groupby(\"patient_id\")[\"implant\"].mean().describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_eda = df_eda.groupby(\"patient_id\")[[\"implant\", \"cancer\"]].mean()\ndf_eda[\"implant\"] = df_eda[\"implant\"].apply(lambda x: 1 if x > 0.0 else 0)\ndf_eda[\"cancer\"] = df_eda[\"cancer\"].apply(lambda x: 1 if x > 0.0 else 0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Ratio on patients who have an implant\")\ndf_eda.value_counts(normalize=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig, ax = plt.subplots(figsize=(8, 6))\n# sns.barplot(x=df_eda[\"cancer\"], y=df_eda[\"age\"])\n\nsns.catplot(\n    data=df_eda.value_counts(normalize=True).swaplevel().reset_index(), x=\"cancer\", y=0, hue=\"implant\", kind=\"bar\", alpha=.6, height=6\n)\n\n# plt.xlabel(ax.get_xlabel(), fontsize=14)\n# plt.ylabel(ax.get_ylabel(), fontsize=14)\n\n# plt.xticks(fontsize=12, fontweight=\"bold\")\n# plt.yticks(fontsize=12, fontweight=\"bold\")\n\n# plt.title(\"Age by Cancer diagnosis\", fontsize=20, pad=20)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = df_eda.groupby(\"cancer\")[\"implant\"].value_counts(normalize=True)\ntmp_df.name = \"ratio_by_caner\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df.reset_index()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(\n    data=tmp_df.reset_index(), x=\"cancer\", y=\"ratio_by_caner\", hue=\"implant\", kind=\"bar\", alpha=.6, height=6\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA - Targets","metadata":{}},{"cell_type":"code","source":"df_eda = df_full[['patient_id'] + CFG.target_list]","metadata":{"execution":{"iopub.status.busy":"2023-01-07T08:40:28.447812Z","iopub.execute_input":"2023-01-07T08:40:28.448165Z","iopub.status.idle":"2023-01-07T08:40:28.455895Z","shell.execute_reply.started":"2023-01-07T08:40:28.448138Z","shell.execute_reply":"2023-01-07T08:40:28.454594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_eda.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# simplify value on BIRADS\ndf_eda.loc[df_eda[\"BIRADS\"].isna() & (df_eda[\"cancer\"] == 1), \"BIRADS\"] = 0\ndf_eda.loc[df_eda[\"BIRADS\"].isna() & (df_eda[\"cancer\"] == 0), \"BIRADS\"] = 1\ndf_eda.loc[df_eda[\"BIRADS\"] == 2, \"BIRADS\"] = 1\ndf_eda[\"BIRADS\"] = df_eda[\"BIRADS\"].astype(\"int32\")\n\n# transform value on BIRADS\n# if higher, more dangerous\ndf_eda[\"BIRADS\"] += 1\ndf_eda[\"BIRADS\"] = df_eda[\"BIRADS\"].replace(2, 0)\n\n# transform to int on difficult_negative_case\ndf_eda[\"difficult_negative_case\"] = df_eda[\"difficult_negative_case\"].astype(\"int32\")","metadata":{"execution":{"iopub.status.busy":"2023-01-07T08:40:30.232255Z","iopub.execute_input":"2023-01-07T08:40:30.232659Z","iopub.status.idle":"2023-01-07T08:40:30.247175Z","shell.execute_reply.started":"2023-01-07T08:40:30.232628Z","shell.execute_reply":"2023-01-07T08:40:30.246091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_eda:\n    print(f'\\n\\n=== {i} value counts ===')\n    print(df_eda[i].value_counts())\n    print(\"Count NAs\", df_eda[i].isna().sum())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_eda:\n    if i in [\"patient_id\", \"cancer\"]:\n        continue\n    tmp_df = df_eda[[\"cancer\", i]].value_counts(normalize=True)\n    sns.catplot(\n        data=tmp_df.reset_index(), x=\"cancer\", y=0, hue=i, kind=\"bar\", alpha=.6, height=6\n    )\n    plt.title(f\"Ratio plot on cancer & {i}\")\n    plt.show()\n    print(tmp_df)\n    print(\"\\n\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-01-07T08:40:33.739192Z","iopub.execute_input":"2023-01-07T08:40:33.739561Z","iopub.status.idle":"2023-01-07T08:40:34.545086Z","shell.execute_reply.started":"2023-01-07T08:40:33.73953Z","shell.execute_reply":"2023-01-07T08:40:34.544181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Summary**\n\n* Biopsy, BIRADS variables are considered as good candidates for multi-target training","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}