{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>⭐️⭐️Mayo Clinic - STRIP AI (Image Classification of Stroke Blood Clot Origin)⭐️⭐️</center>\n\n<img src='https://hbi.ucalgary.ca/sites/default/files/styles/ucws_hero_cta_desktop/public/2021-04/iStock-1249957366.jpg?h=a892a2e9&itok=J0nFgWRa'>\n\n\n# 📢**💡Goal of the Competition💡**\n\nThe goal of this competition is to classify the blood clot origins in ischemic stroke. Using whole slide digital pathology images, you'll build a model that differentiates between the two major acute ischemic stroke (AIS) etiology subtypes: cardiac and large artery atherosclerosis.\n\nYour work will enable healthcare providers to better identify the origins of blood clots in deadly strokes, making it easier for physicians to prescribe the best post-stroke therapeutic management and reducing the likelihood of a second stroke.\n\n\n# <center> 🤔 Concern about Only 💡 CSV Data 🤔 - What's happen </center>\n\n## **Steps:**\n\n### 1. Import Necessary Library\n\n### 2. Load and analysis the data\n\n### 3. Preprocessing\n\n### 4. KFOLD\n\n### 5. Build the Model\n\n### 6. Predict Output\n\n### 7. Generate Submission file\n\n# <center> 💡Just simple try! Only CSV File💡 </center>","metadata":{}},{"cell_type":"markdown","source":"# 📢**Import Library** ","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport cv2\n\n#Preprocessing\nfrom sklearn import model_selection\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder,StandardScaler\nfrom sklearn import preprocessing\n#Model\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:51.762162Z","iopub.execute_input":"2022-07-08T03:42:51.762667Z","iopub.status.idle":"2022-07-08T03:42:54.118958Z","shell.execute_reply.started":"2022-07-08T03:42:51.762566Z","shell.execute_reply":"2022-07-08T03:42:54.117756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢**Load data**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\nsample = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\ntrain_image_path = '../input/mayo-clinic-strip-ai/train/'\ntest_image_path = '../input/mayo-clinic-strip-ai/test/'","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.121574Z","iopub.execute_input":"2022-07-08T03:42:54.122439Z","iopub.status.idle":"2022-07-08T03:42:54.150745Z","shell.execute_reply.started":"2022-07-08T03:42:54.122389Z","shell.execute_reply":"2022-07-08T03:42:54.149807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train_Shape: {train.shape},Test_Shape: {test.shape},Sample_Shape: {sample.shape}')\ndisplay(train.sample(2))\ndisplay(test.sample(2))\ndisplay(sample.sample(2))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.152603Z","iopub.execute_input":"2022-07-08T03:42:54.15344Z","iopub.status.idle":"2022-07-08T03:42:54.194236Z","shell.execute_reply.started":"2022-07-08T03:42:54.153395Z","shell.execute_reply":"2022-07-08T03:42:54.192934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.197322Z","iopub.execute_input":"2022-07-08T03:42:54.197826Z","iopub.status.idle":"2022-07-08T03:42:54.2332Z","shell.execute_reply.started":"2022-07-08T03:42:54.19778Z","shell.execute_reply":"2022-07-08T03:42:54.229862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nsns.countplot(train['label'])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.235511Z","iopub.execute_input":"2022-07-08T03:42:54.236669Z","iopub.status.idle":"2022-07-08T03:42:54.466399Z","shell.execute_reply.started":"2022-07-08T03:42:54.236619Z","shell.execute_reply":"2022-07-08T03:42:54.465254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢**KFOLD**","metadata":{}},{"cell_type":"code","source":"#add extra one columns\ntrain['kfold']=-1\n#Distributing the data 5 shares\nkfold = model_selection.KFold(n_splits=5, shuffle= True, random_state = 12)\nfor fold, (train_indicies, valid_indicies) in enumerate(kfold.split(X=train)):\n    #print(fold,train_indicies,valid_indicies)\n    train.loc[valid_indicies,'kfold'] = fold\n\n    \nprint(train.kfold.value_counts()) #total data 300000 = kfold split :5 * 60000\n\n#output of train folds data\ntrain.to_csv(\"trainfold_5.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.467746Z","iopub.execute_input":"2022-07-08T03:42:54.468073Z","iopub.status.idle":"2022-07-08T03:42:54.487306Z","shell.execute_reply.started":"2022-07-08T03:42:54.468043Z","shell.execute_reply":"2022-07-08T03:42:54.486036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢**Map label into binary form**","metadata":{}},{"cell_type":"code","source":"label_map = {'CE' : 0, 'LAA':1}\ntrain['label'] = train['label'].map(label_map)\ntrain.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.488719Z","iopub.execute_input":"2022-07-08T03:42:54.489047Z","iopub.status.idle":"2022-07-08T03:42:54.504295Z","shell.execute_reply.started":"2022-07-08T03:42:54.489018Z","shell.execute_reply":"2022-07-08T03:42:54.503057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **XGBoost-Just try regression format's**","metadata":{}},{"cell_type":"code","source":"\nfrom catboost import CatBoostRegressor,CatBoostClassifier\nfrom xgboost import XGBRegressor,XGBClassifier\n\n#features(categorical and numerical datas separate)\nuseful_features = [c for c in train.columns if c not in (\"image_id\",\"patient_id\",\"kfold\",\"label\")]\nobject_cols = [col for col in useful_features]\n#numerical_cols = [col for col in useful_features]\ntest = test[useful_features]\n\n\nfor fold in range(5):\n    xtrain = train[train.kfold != fold].reset_index(drop=True)\n    xvalid = train[train.kfold == fold].reset_index(drop=True)\n    xtest = test.copy()\n    \n    ytrain = xtrain.label\n    yvalid = xvalid.label\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n        \n    #Model hyperparameter of XGboostRegressor\n    \n    xgb_params = {\n            'learning_rate': 0.001368,\n            'subsample': 0.7875490025178,\n            'colsample_bytree': 0.11807135201147,\n            'max_depth': 3,\n            'booster': 'gbtree', \n            'reg_lambda': 0.0008746338866473539,\n            'reg_alpha': 23.13181079976304,\n            'random_state':42,\n            'n_estimators':8000\n            }\n\n    model= XGBRegressor(**xgb_params)\n    \n    model.fit(xtrain,ytrain)\n    \n\n    #way of output is display\n    print(f\"fold:{fold}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:42:54.50589Z","iopub.execute_input":"2022-07-08T03:42:54.50623Z","iopub.status.idle":"2022-07-08T03:44:23.061992Z","shell.execute_reply.started":"2022-07-08T03:42:54.5062Z","shell.execute_reply":"2022-07-08T03:44:23.060507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predict = model.predict(xtest)\ntest_predict","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.063327Z","iopub.execute_input":"2022-07-08T03:44:23.063717Z","iopub.status.idle":"2022-07-08T03:44:23.08023Z","shell.execute_reply.started":"2022-07-08T03:44:23.063683Z","shell.execute_reply":"2022-07-08T03:44:23.079049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_pred_str = LB.inverse_transform(test_pre)\nsample.loc[:,\"CE\"] = test_predict[0]\nsample.loc[:,\"LAA\"] = test_predict[1]\nsample.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.084395Z","iopub.execute_input":"2022-07-08T03:44:23.085136Z","iopub.status.idle":"2022-07-08T03:44:23.094198Z","shell.execute_reply.started":"2022-07-08T03:44:23.085095Z","shell.execute_reply":"2022-07-08T03:44:23.092577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.095734Z","iopub.execute_input":"2022-07-08T03:44:23.096357Z","iopub.status.idle":"2022-07-08T03:44:23.11225Z","shell.execute_reply.started":"2022-07-08T03:44:23.096322Z","shell.execute_reply":"2022-07-08T03:44:23.110985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <center>Try to focus on image data </center>\n\nCredit: https://www.kaggle.com/code/jirkaborovec/bloodclots-classif-eda-loading-images","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport os,glob\nimport tifffile as tif\nfrom tqdm import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.114072Z","iopub.execute_input":"2022-07-08T03:44:23.115629Z","iopub.status.idle":"2022-07-08T03:44:23.284779Z","shell.execute_reply.started":"2022-07-08T03:44:23.115578Z","shell.execute_reply":"2022-07-08T03:44:23.283566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **tiff method not run - because the data size large, reduced size or convert image format is good idea**","metadata":{}},{"cell_type":"code","source":"# #Lets Print a sample\n# img=tif.imread(os.path.join(train_image_path,str(train.loc[0,'image_id'])+'.tif'))\n# img.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.285995Z","iopub.execute_input":"2022-07-08T03:44:23.286341Z","iopub.status.idle":"2022-07-08T03:44:23.291372Z","shell.execute_reply.started":"2022-07-08T03:44:23.286313Z","shell.execute_reply":"2022-07-08T03:44:23.290015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #plot image\n# plt.figure(figsize=(5,5))\n# plt.imshow(cv2.cvtColor(img,cv2.COLOR_BGR2RGB))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.292735Z","iopub.execute_input":"2022-07-08T03:44:23.293711Z","iopub.status.idle":"2022-07-08T03:44:23.302873Z","shell.execute_reply.started":"2022-07-08T03:44:23.293675Z","shell.execute_reply":"2022-07-08T03:44:23.301862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_FOLDER = \"/kaggle/input/mayo-clinic-strip-ai/\"\ndf_train= pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.30415Z","iopub.execute_input":"2022-07-08T03:44:23.305181Z","iopub.status.idle":"2022-07-08T03:44:23.319743Z","shell.execute_reply.started":"2022-07-08T03:44:23.305133Z","shell.execute_reply":"2022-07-08T03:44:23.318526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nfrom tqdm.auto import tqdm\n\nImage.MAX_IMAGE_PIXELS = 5_000_000_000\n\nsizes = []\nfor name in tqdm(df_train[\"image_id\"]):\n    img = Image.open(os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\"))\n    sizes.append({\"img_height\": img.height, \"img_width\": img.width})\n\ndf_sizes = pd.DataFrame(sizes)\nfor col in df_sizes.columns:\n    df_train[col] = df_sizes[col]","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:23.321698Z","iopub.execute_input":"2022-07-08T03:44:23.323292Z","iopub.status.idle":"2022-07-08T03:44:38.03597Z","shell.execute_reply.started":"2022-07-08T03:44:23.323248Z","shell.execute_reply":"2022-07-08T03:44:38.034687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=1, figsize=(8, 5))\nfor i, col in enumerate([\"img_height\", \"img_width\"]):\n    _= df_train[[col]].hist(ax=axes[i], bins=35)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:38.038016Z","iopub.execute_input":"2022-07-08T03:44:38.038565Z","iopub.status.idle":"2022-07-08T03:44:38.472664Z","shell.execute_reply.started":"2022-07-08T03:44:38.038445Z","shell.execute_reply":"2022-07-08T03:44:38.471174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\ndf_sizes = df_train[[\"img_width\", \"img_height\", \"label\"]]\nfor col in (\"img_width\", \"img_height\"):\n    df_sizes[col] = [round(i / 1000) * 1000 for i in df_sizes[col]]\n\ndf_sizes = df_sizes.groupby([\"img_width\", \"img_height\", \"label\"], as_index=False).size()\n# display(df_sizes.head())\nfig = px.scatter(df_sizes, x=\"img_width\", y=\"img_height\", size=\"size\", color=\"label\")\n# fig.gca().axis('equal')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:38.47437Z","iopub.execute_input":"2022-07-08T03:44:38.474856Z","iopub.status.idle":"2022-07-08T03:44:40.248227Z","shell.execute_reply.started":"2022-07-08T03:44:38.474818Z","shell.execute_reply":"2022-07-08T03:44:40.247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=5, ncols=1, figsize=(9, 20))\nfor i, name in enumerate(df_train[\"image_id\"].sample(5)):\n    img = Image.open(os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\"))\n    img.thumbnail((1000, 1000), resample=Image.Resampling.BILINEAR, reducing_gap=20)\n    axes[i].imshow(img.transpose(Image.Resampling.BILINEAR))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:44:40.249756Z","iopub.execute_input":"2022-07-08T03:44:40.250192Z","iopub.status.idle":"2022-07-08T03:46:34.385723Z","shell.execute_reply.started":"2022-07-08T03:44:40.250149Z","shell.execute_reply":"2022-07-08T03:46:34.384369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Thanks for visiting guys**\n\n## 📢 **Next process coming soon**","metadata":{}}]}