{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Kecerdasan Komputasional","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"pip install --upgrade scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:22.406271Z","iopub.execute_input":"2023-09-26T03:32:22.406625Z","iopub.status.idle":"2023-09-26T03:32:37.035767Z","shell.execute_reply.started":"2023-09-26T03:32:22.406598Z","shell.execute_reply":"2023-09-26T03:32:37.034583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nimport cv2\nimport matplotlib.pyplot as plt\n#from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import roc_auc_score, accuracy_score\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn import metrics\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score,roc_curve,roc_auc_score, auc\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import cross_val_predict, StratifiedKFold\n","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:37.038093Z","iopub.execute_input":"2023-09-26T03:32:37.038444Z","iopub.status.idle":"2023-09-26T03:32:38.146888Z","shell.execute_reply.started":"2023-09-26T03:32:37.038409Z","shell.execute_reply":"2023-09-26T03:32:38.145788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.148101Z","iopub.execute_input":"2023-09-26T03:32:38.148492Z","iopub.status.idle":"2023-09-26T03:32:38.240371Z","shell.execute_reply.started":"2023-09-26T03:32:38.148468Z","shell.execute_reply":"2023-09-26T03:32:38.239352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_clean = df_train.dropna()\n# data_new = data_clean.drop(columns=['days_to_cancer', 'pid'])\n\n# # drop all rows that contain 'Participant refused to answer'\n# data_new = data_new.drop(data_new[data_new['race'] == 'Participant refused to answer'].index)\n\n# X = data_new[data_new.columns[:5]]\n# y = data_new['smoker']\n\n# data_new.head(5)\n# len(data_new)\n\n# Age, Laterally, Implant, biopsy, invasive, difficulty negative case jadi utama\ndata_new = df.drop(columns=['age','laterality','site_id', 'patient_id', 'image_id', 'view', 'cancer','BIRADS','density','machine_id'])\ndata_new.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.24375Z","iopub.execute_input":"2023-09-26T03:32:38.244122Z","iopub.status.idle":"2023-09-26T03:32:38.272175Z","shell.execute_reply.started":"2023-09-26T03:32:38.244089Z","shell.execute_reply":"2023-09-26T03:32:38.271011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Normalization data to numerical\n\n# #Gender : Male = 1, Female = 0\n# data_normal = data_new.replace('Male', 1)\n# data_normal = data_normal.replace('Female', 0)\n\n# #Race : White = 1, Black or African-American = 0\n\n# data_normal = data_normal.replace('More than one race', 4)\n# data_normal = data_normal.replace('White', 3)\n# data_normal = data_normal.replace('Asian', 2)\n# data_normal = data_normal.replace('Native Hawaiian or Other Pacific Islander', 1)\n# data_normal = data_normal.replace('American Indian or Alaskan Native', 1)\n# data_normal = data_normal.replace('Black or African-American', 0)\n\n# #Smoker : Current = 1, Former = 0\n# data_normal = data_normal.replace('Current',1)\n# data_normal = data_normal.replace('Former',0)\n\n# #stage_of_cancer : Remove class and change to numerical\n# data_normal = data_normal.replace('IA',1)\n# data_normal = data_normal.replace('IB',1)\n# #data_normal = data_normal.replace('IC',1)\n# data_normal = data_normal.replace('IIA',2)\n# data_normal = data_normal.replace('IIB',2)\n# #data_normal = data_normal.replace('IIC',2)\n# data_normal = data_normal.replace('IIIA',3)\n# data_normal = data_normal.replace('IIIB',3)\n# #data_normal = data_normal.replace('IIIC',3)\n# data_normal = data_normal.replace('IV',4)\n\n# data_normal.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.27386Z","iopub.execute_input":"2023-09-26T03:32:38.274362Z","iopub.status.idle":"2023-09-26T03:32:38.279089Z","shell.execute_reply.started":"2023-09-26T03:32:38.274334Z","shell.execute_reply":"2023-09-26T03:32:38.278048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_new.loc[:, ['difficult_negative_case', 'laterality']].replace(['FALSE', 'TRUE', 'L', 'R'], [0, 1, 0, 1], inplace=True)\ndata_new['difficult_negative_case'].replace(['FALSE', 'TRUE'], [0,1], inplace=True)\n\nX = data_new[data_new.columns[:3]]\n\ny = data_new['difficult_negative_case']\n# X_train, X_test, y_train, y_test = train_test_split(\n#     X, y, test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.281193Z","iopub.execute_input":"2023-09-26T03:32:38.281547Z","iopub.status.idle":"2023-09-26T03:32:38.295919Z","shell.execute_reply.started":"2023-09-26T03:32:38.281524Z","shell.execute_reply":"2023-09-26T03:32:38.294859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inisialisasi scalar\nsc = StandardScaler()\n\n# Mengambil hanya kolom numerik\nX_numerical = X.select_dtypes(include=['number'])\n\n# Skalakan kolom numerik\nX_numerical_scaled = sc.fit_transform(X_numerical)\n\n# inisialisasi model Decision Tree\n# param yang gk ngaruh : criterion, splitter, min_samples_split, min_samples_leaf, max_features\n#                        random_state, \n# tweaked param : max_depth(0 = Error, 1 (Worst), 2 / None)\n#  min_weight_fraction_leaf (0.00 (Default) ~ 0.01 (Best) | 0.02 ~ 0.05 | 0.06~0.50 (Worst, MAX)|)\n#  class_weight = \"balanced\" or None\n#  max_leaf_nodes = 2 (Worst, MIN), 3, 4, 5 / None (Best)\n#  min_impurity_decrease = 0.000064 or None (Best)\n\nclf = DecisionTreeClassifier(max_depth = 1, max_leaf_nodes = 2, random_state=42)\n#clf = DecisionTreeClassifier(class_weight= 'balanced', random_state=42)\n\n# cross-validation dengan StratifiedKFold 5 (nothing happens)\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# inisialisasi scalar\nsc = StandardScaler()\n\n# inisialisasi pipeline dengan decision tree dan scalar\npipe = make_pipeline(sc, clf)\n\n#0.16 optimal, 0.25 default\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.25, random_state=1)\n\n# training model\npipe.fit(X_train, y_train)\n\n# scoring model pada data pengujian\ny_pred = pipe.predict(X_test)\ny_prob = pipe.predict_proba(X_test)\n\n# Hitung akurasi\naccuracy = accuracy_score(y_test, y_pred)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Hitung AUC\nauc = roc_auc_score(y_test, y_prob[:, 1])\nprint(f'AUC: {auc:.4f}')\n\n# # Bagi dataset menjadi satu set pelatihan dan satu set pengujian\n# from sklearn.model_selection import train_test_split\n\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# # inisialisasi list untuk menampung hasil skor dari setiap fold\n# scores = []\n# results = []\n\n# # loop untuk setiap fold\n# for train_index, test_index in cv.split(X, y):\n#     X_train, X_test = X.values[train_index], X.values[test_index]\n#     y_train, y_test = y.values[train_index], y.values[test_index]\n    \n#     # training model\n#     pipe.fit(X_train, y_train)\n    \n#     # scoring model\n#     score = pipe.score(X_test, y_test)\n#     scores.append(score)\n\n#     # Melakukan prediksi pada data testing\n#     y_pred = pipe.predict(X_test)\n    \n#     # Hitung Probabilitas\n#     y_prob =pipe.predict_proba(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.29831Z","iopub.execute_input":"2023-09-26T03:32:38.298721Z","iopub.status.idle":"2023-09-26T03:32:38.347121Z","shell.execute_reply.started":"2023-09-26T03:32:38.298689Z","shell.execute_reply":"2023-09-26T03:32:38.346156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion Matrix\nconfusion_matrix = metrics.confusion_matrix(y_test, y_pred)\n\ncm_display = metrics.ConfusionMatrixDisplay(confusion_matrix = confusion_matrix, display_labels = [False, True])\n\ncm_display.plot()\nplt.show()\n     \nAccuracy = metrics.accuracy_score(y_test, y_pred)\nPrecision = metrics.precision_score(y_test, y_pred)\nSensitivity_recall = metrics.recall_score(y_test, y_pred)\nSpecificity = metrics.recall_score(y_test, y_pred, pos_label=0)\n    \n#     print(classification_report(y_test, y_pred))\n# roc_curve(pipe, X_test, y_test)\n# plt.show()\n    \n    # AUC (y_prob[:,1] khusus untuk random forest dan knn)\nauc = roc_auc_score(y_test, y_prob[:,1])\n    \n     # Simpan hasil ke dalam list\nresults = []\nresults.append({'Accuracy': Accuracy,\"Precision\":Precision,\"Sensitivity_recall\":Sensitivity_recall,\"Specificity\":Specificity, 'AUC':auc})\n    \n    # Buat dataframe dari hasil\ndf = pd.DataFrame(results)\n\n    # Print dataframe\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.350012Z","iopub.execute_input":"2023-09-26T03:32:38.350312Z","iopub.status.idle":"2023-09-26T03:32:38.594311Z","shell.execute_reply.started":"2023-09-26T03:32:38.35029Z","shell.execute_reply":"2023-09-26T03:32:38.59314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:43:40.362196Z","iopub.execute_input":"2023-09-26T03:43:40.362533Z","iopub.status.idle":"2023-09-26T03:43:40.371092Z","shell.execute_reply.started":"2023-09-26T03:43:40.36251Z","shell.execute_reply":"2023-09-26T03:43:40.369733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.fit(X_train, y_train)\n\nfrom sklearn.tree import export_graphviz\n# Export as dot file\nexport_graphviz(clf, out_file='tree.dot', \n                feature_names = X_train.columns,\n                class_names = ['true','false'],\n                rounded = True, proportion = False, \n                precision = 2, filled = True)\n\n# Convert to png using system command (requires Graphviz)\nfrom subprocess import call\ncall(['dot', '-Tpng', 'tree.dot', '-o', 'tree.png', '-Gdpi=600'])\n\n# Display in jupyter notebook\nfrom IPython.display import Image\nImage(filename = 'tree.png')","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:45:37.815971Z","iopub.execute_input":"2023-09-26T03:45:37.816333Z","iopub.status.idle":"2023-09-26T03:45:38.224917Z","shell.execute_reply.started":"2023-09-26T03:45:37.816297Z","shell.execute_reply":"2023-09-26T03:45:38.224158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_encoder = LabelEncoder()\n# X_train['laterality'] = label_encoder.fit_transform(X_train['laterality'])\n# X_test['laterality'] = label_encoder.transform(X_test['laterality'])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.757662Z","iopub.status.idle":"2023-09-26T03:32:38.757957Z","shell.execute_reply.started":"2023-09-26T03:32:38.757812Z","shell.execute_reply":"2023-09-26T03:32:38.757827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Mengambil hanya kolom numerik\n# X_numerical = X.select_dtypes(include=['number'])\n\n# # Skalakan kolom numerik\n# X_numerical_scaled = sc.fit_transform(X_numerical)\n\n# # inisialisasi model Random Forest\n# rfc = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# # cross-validation dengan StratifiedKFold 10\n# cv = StratifiedKFold(n_splits=10, shuffle=True, random_state=42)\n\n# # inisialisasi scalar\n# sc = StandardScaler()\n\n# # inisialisasi pipeline dengan random forest dan scalar\n# pipe = make_pipeline(sc, rfc)\n\n# # inisialisasi list untuk menampung hasil skor dari setiap fold\n# scores = []\n# results = []\n\n# # loop untuk setiap fold\n# for train_index, test_index in cv.split(X, y):\n#     X_train, X_test = X.values[train_index], X.values[test_index]\n#     y_train, y_test = y.values[train_index], y.values[test_index]\n    \n#     # training model\n#     pipe.fit(X_train, y_train)\n    \n#     # scoring model\n#     score = pipe.score(X_test, y_test)\n#     scores.append(score)\n\n#     # Melakukan prediksi pada data testing\n#     y_pred = pipe.predict(X_test)\n    \n#     # Hitung Probabilitas\n#     y_prob =pipe.predict_proba(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.75908Z","iopub.status.idle":"2023-09-26T03:32:38.759373Z","shell.execute_reply.started":"2023-09-26T03:32:38.759233Z","shell.execute_reply":"2023-09-26T03:32:38.759248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Model Accuracy, how often is the classifier correct?\n# print(\"Accuracy:\",metrics.accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-09-26T03:32:38.760249Z","iopub.status.idle":"2023-09-26T03:32:38.760548Z","shell.execute_reply.started":"2023-09-26T03:32:38.760404Z","shell.execute_reply":"2023-09-26T03:32:38.760418Z"},"trusted":true},"execution_count":null,"outputs":[]}]}