{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T02:41:16.525108Z","iopub.execute_input":"2022-07-08T02:41:16.52574Z","iopub.status.idle":"2022-07-08T02:41:16.706878Z","shell.execute_reply.started":"2022-07-08T02:41:16.525629Z","shell.execute_reply":"2022-07-08T02:41:16.705526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport cv2\n\n#Preprocessing\nfrom sklearn import model_selection\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder,StandardScaler\nfrom sklearn import preprocessing\n#Model\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:42:17.531231Z","iopub.execute_input":"2022-07-08T02:42:17.531924Z","iopub.status.idle":"2022-07-08T02:42:20.004061Z","shell.execute_reply.started":"2022-07-08T02:42:17.531873Z","shell.execute_reply":"2022-07-08T02:42:20.002841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\nsample = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\ntrain_image_path = '../input/mayo-clinic-strip-ai/train/'\ntest_image_path = '../input/mayo-clinic-strip-ai/test/'","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:42:15.11733Z","iopub.execute_input":"2022-07-08T02:42:15.117756Z","iopub.status.idle":"2022-07-08T02:42:15.14521Z","shell.execute_reply.started":"2022-07-08T02:42:15.117723Z","shell.execute_reply":"2022-07-08T02:42:15.144228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train_Shape: {train.shape},Test_Shape: {test.shape},Sample_Shape: {sample.shape}')\ndisplay(train.sample(2))\ndisplay(test.sample(2))\ndisplay(sample.sample(2))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:42:53.209868Z","iopub.execute_input":"2022-07-08T02:42:53.210255Z","iopub.status.idle":"2022-07-08T02:42:53.250871Z","shell.execute_reply.started":"2022-07-08T02:42:53.210225Z","shell.execute_reply":"2022-07-08T02:42:53.249752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nsns.countplot(train['label'])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:43:14.391299Z","iopub.execute_input":"2022-07-08T02:43:14.391787Z","iopub.status.idle":"2022-07-08T02:43:14.61911Z","shell.execute_reply.started":"2022-07-08T02:43:14.391749Z","shell.execute_reply":"2022-07-08T02:43:14.618029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"KFOLD","metadata":{}},{"cell_type":"code","source":"#add extra one columns\ntrain['kfold']=-1\n#Distributing the data 5 shares\nkfold = model_selection.KFold(n_splits=5, shuffle= True, random_state = 12)\nfor fold, (train_indicies, valid_indicies) in enumerate(kfold.split(X=train)):\n    #print(fold,train_indicies,valid_indicies)\n    train.loc[valid_indicies,'kfold'] = fold\n\n    \nprint(train.kfold.value_counts()) #total data 300000 = kfold split :5 * 60000\n\n#output of train folds data\ntrain.to_csv(\"trainfold_5.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:44:01.626189Z","iopub.execute_input":"2022-07-08T02:44:01.627214Z","iopub.status.idle":"2022-07-08T02:44:01.645935Z","shell.execute_reply.started":"2022-07-08T02:44:01.627162Z","shell.execute_reply":"2022-07-08T02:44:01.645117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"BINARY MAPPING","metadata":{}},{"cell_type":"code","source":"label_map = {'CE' : 0, 'LAA':1}\ntrain['label'] = train['label'].map(label_map)\ntrain.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:44:21.972435Z","iopub.execute_input":"2022-07-08T02:44:21.972855Z","iopub.status.idle":"2022-07-08T02:44:21.989163Z","shell.execute_reply.started":"2022-07-08T02:44:21.972819Z","shell.execute_reply":"2022-07-08T02:44:21.987871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cat Boost Algorithm","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor,CatBoostClassifier\n\n#features(categorical and numerical datas separate)\nuseful_features = [c for c in train.columns if c not in (\"image_id\",\"patient_id\",\"kfold\",\"label\")]\nobject_cols = [col for col in useful_features]\n#numerical_cols = [col for col in useful_features]\ntest = test[useful_features]\n\n\nfor fold in range(5):\n    xtrain = train[train.kfold != fold].reset_index(drop=True)\n    xvalid = train[train.kfold == fold].reset_index(drop=True)\n    xtest = test.copy()\n    \n    ytrain = xtrain.label\n    yvalid = xvalid.label\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n        \n    #Model hyperparameter of XGboostRegressor\n    #catboost model\n    catpara={\n        'subsample': 0.95312,\n        'learning_rate': 0.00135356,\n        \"max_depth\": 3,\n        \"min_data_in_leaf\":77,\n        'random_state':12,\n        'n_estimators':50000,\n        'rsm':0.5,\n        'l2_leaf_reg': 0.02247766515106271\n    }\n    \n    model=CatBoostRegressor(**catpara)\n    model.fit(xtrain,ytrain,early_stopping_rounds=300,eval_set=[(xvalid,yvalid)],verbose=1000)\n    model.fit(xtrain,ytrain,early_stopping_rounds=300,eval_set=[(xvalid,yvalid)],verbose=False)\n    preds_valid = model.predict(xvalid)\n    \n\n    #way of output is display\n    print(f\"fold:{fold}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:45:32.689628Z","iopub.execute_input":"2022-07-08T02:45:32.690024Z","iopub.status.idle":"2022-07-08T02:45:41.63555Z","shell.execute_reply.started":"2022-07-08T02:45:32.689993Z","shell.execute_reply":"2022-07-08T02:45:41.63459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predict = model.predict(xtest)\ntest_predict","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:46:12.27458Z","iopub.execute_input":"2022-07-08T02:46:12.27495Z","iopub.status.idle":"2022-07-08T02:46:12.285294Z","shell.execute_reply.started":"2022-07-08T02:46:12.27492Z","shell.execute_reply":"2022-07-08T02:46:12.283805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_pred_str = LB.inverse_transform(test_pre)\nsample.loc[:,\"CE\"] = test_predict[0]\nsample.loc[:,\"LAA\"] = test_predict[1]\nsample.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:46:32.756516Z","iopub.execute_input":"2022-07-08T02:46:32.757537Z","iopub.status.idle":"2022-07-08T02:46:32.765926Z","shell.execute_reply.started":"2022-07-08T02:46:32.757498Z","shell.execute_reply":"2022-07-08T02:46:32.764691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:46:45.601128Z","iopub.execute_input":"2022-07-08T02:46:45.60234Z","iopub.status.idle":"2022-07-08T02:46:45.61643Z","shell.execute_reply.started":"2022-07-08T02:46:45.60229Z","shell.execute_reply":"2022-07-08T02:46:45.614673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image Data","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport os,glob\nimport tifffile as tif\nfrom tqdm import tqdm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Lets Print a sample\n# img=tif.imread(os.path.join(train_image_path,str(train.loc[0,'image_id'])+'.tif'))\n# img.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_FOLDER = \"/kaggle/input/mayo-clinic-strip-ai/\"\ndf_train= pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:48:11.852055Z","iopub.execute_input":"2022-07-08T02:48:11.852972Z","iopub.status.idle":"2022-07-08T02:48:11.861344Z","shell.execute_reply.started":"2022-07-08T02:48:11.852925Z","shell.execute_reply":"2022-07-08T02:48:11.860114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nfrom tqdm.auto import tqdm\n\nImage.MAX_IMAGE_PIXELS = 5_000_000_000\n\nsizes = []\nfor name in tqdm(df_train[\"image_id\"]):\n    img = Image.open(os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\"))\n    sizes.append({\"img_height\": img.height, \"img_width\": img.width})\n\ndf_sizes = pd.DataFrame(sizes)\nfor col in df_sizes.columns:\n    df_train[col] = df_sizes[col]","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:48:32.4764Z","iopub.execute_input":"2022-07-08T02:48:32.476835Z","iopub.status.idle":"2022-07-08T02:48:47.347429Z","shell.execute_reply.started":"2022-07-08T02:48:32.476801Z","shell.execute_reply":"2022-07-08T02:48:47.346309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=1, figsize=(8, 5))\nfor i, col in enumerate([\"img_height\", \"img_width\"]):\n    _= df_train[[col]].hist(ax=axes[i], bins=35)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:49:00.021543Z","iopub.execute_input":"2022-07-08T02:49:00.021928Z","iopub.status.idle":"2022-07-08T02:49:00.452982Z","shell.execute_reply.started":"2022-07-08T02:49:00.021898Z","shell.execute_reply":"2022-07-08T02:49:00.451788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\ndf_sizes = df_train[[\"img_width\", \"img_height\", \"label\"]]\nfor col in (\"img_width\", \"img_height\"):\n    df_sizes[col] = [round(i / 1000) * 1000 for i in df_sizes[col]]\n\ndf_sizes = df_sizes.groupby([\"img_width\", \"img_height\", \"label\"], as_index=False).size()\n# display(df_sizes.head())\nfig = px.scatter(df_sizes, x=\"img_width\", y=\"img_height\", size=\"size\", color=\"label\")\n# fig.gca().axis('equal')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:49:16.291213Z","iopub.execute_input":"2022-07-08T02:49:16.292264Z","iopub.status.idle":"2022-07-08T02:49:18.153527Z","shell.execute_reply.started":"2022-07-08T02:49:16.292218Z","shell.execute_reply":"2022-07-08T02:49:18.15234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=5, ncols=1, figsize=(9, 20))\nfor i, name in enumerate(df_train[\"image_id\"].sample(5)):\n    img = Image.open(os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\"))\n    img.thumbnail((1000, 1000), resample=Image.Resampling.BILINEAR, reducing_gap=20)\n    axes[i].imshow(img.transpose(Image.Resampling.BILINEAR))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:49:34.196761Z","iopub.execute_input":"2022-07-08T02:49:34.197505Z","iopub.status.idle":"2022-07-08T02:51:33.030463Z","shell.execute_reply.started":"2022-07-08T02:49:34.197425Z","shell.execute_reply":"2022-07-08T02:51:33.02895Z"},"trusted":true},"execution_count":null,"outputs":[]}]}