{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"bc59e93d-162b-4759-b150-79bce322adc8","_cell_guid":"669b4e8c-5b1d-4b56-9e75-999d9b21d793","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.24177Z","iopub.execute_input":"2023-02-24T18:54:38.242243Z","iopub.status.idle":"2023-02-24T18:54:38.248653Z","shell.execute_reply.started":"2023-02-24T18:54:38.242205Z","shell.execute_reply":"2023-02-24T18:54:38.247221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Introduction**\n\nLoad the Data:\nThe first step is to load the training data and test data into Python.\n\nExplore the Data:\nThe next step is to explore the data to get an understanding of the features and target variable.\n\nPreprocess the Data:\nBefore training the machine learning model, it is necessary to preprocess the data\n\nTrain a Model:\nTo solve a binary classification problem breast cancer detection using logistic regression\n\nEvaluate the Model:\nOnce the model is trained, then evaluate its performance on the validation set using metrics such as accuracy, precision, recall, F1-score, and ROC curve.\n\nTune the Model:\nThe performance of the model can be improved by tuning its hyperparameters using random search to find the optimal hyperparameters for the model.\n\nMake Predictions:\nFinally,use the trained model to make predictions on the test set.","metadata":{"_uuid":"7d7b90c7-ae43-413d-a31b-446ec42955a1","_cell_guid":"66decb59-05ac-4103-862a-b91252234f09","trusted":true}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the training data\ntrain_data = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\n# Load the test data\ntest_data = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n# Load the sample submission file\nsample_submission = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"_uuid":"676b2cc9-bd19-494c-961c-95269c340c74","_cell_guid":"78be8726-9eb7-4600-81a3-b55f533118e6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.256119Z","iopub.execute_input":"2023-02-24T18:54:38.256511Z","iopub.status.idle":"2023-02-24T18:54:38.328015Z","shell.execute_reply.started":"2023-02-24T18:54:38.256478Z","shell.execute_reply":"2023-02-24T18:54:38.326828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get information about the columns, data types, and null values\nprint(train_data.info())","metadata":{"_uuid":"530543b7-6ab0-4178-8ab8-9e340f4921b0","_cell_guid":"15de2e75-f844-4edd-9fcd-836d99ac32c6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.329929Z","iopub.execute_input":"2023-02-24T18:54:38.330283Z","iopub.status.idle":"2023-02-24T18:54:38.352289Z","shell.execute_reply.started":"2023-02-24T18:54:38.330255Z","shell.execute_reply":"2023-02-24T18:54:38.351468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_data.info())","metadata":{"_uuid":"6848b4bd-ff50-4948-9750-d98c4bcc50e8","_cell_guid":"8e80585c-c48e-4232-b075-bc2461aea423","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.353406Z","iopub.execute_input":"2023-02-24T18:54:38.354025Z","iopub.status.idle":"2023-02-24T18:54:38.36621Z","shell.execute_reply.started":"2023-02-24T18:54:38.353994Z","shell.execute_reply":"2023-02-24T18:54:38.365036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop unnecessary columns\ntrain_data = train_data.drop(['patient_id', 'image_id'], axis=1)\ntest_data = test_data.drop(['patient_id', 'image_id'], axis=1)","metadata":{"_uuid":"aac1385f-c1e9-48c6-b81c-474b7d4a9ab7","_cell_guid":"202e1584-793c-4f9a-b616-3768ca61b095","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.367396Z","iopub.execute_input":"2023-02-24T18:54:38.367783Z","iopub.status.idle":"2023-02-24T18:54:38.381504Z","shell.execute_reply.started":"2023-02-24T18:54:38.367752Z","shell.execute_reply":"2023-02-24T18:54:38.380409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handle missing values\ntrain_data = train_data.fillna(train_data.mean())\ntest_data = test_data.fillna(test_data.mean())","metadata":{"_uuid":"d0cc38bc-e6c1-4293-a125-a84a3cb45aec","_cell_guid":"c97fc2de-cab6-4ed8-99d0-2c0a217a717e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.383716Z","iopub.execute_input":"2023-02-24T18:54:38.384267Z","iopub.status.idle":"2023-02-24T18:54:38.614461Z","shell.execute_reply.started":"2023-02-24T18:54:38.384236Z","shell.execute_reply":"2023-02-24T18:54:38.613316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_data.info())","metadata":{"_uuid":"50aa0de4-f8fa-45fc-9f84-4c41180d0a97","_cell_guid":"e62974f0-015d-4cef-8639-be71a3c2caf2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.615839Z","iopub.execute_input":"2023-02-24T18:54:38.616203Z","iopub.status.idle":"2023-02-24T18:54:38.633273Z","shell.execute_reply.started":"2023-02-24T18:54:38.616171Z","shell.execute_reply":"2023-02-24T18:54:38.631836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.info())","metadata":{"_uuid":"238dddbc-7b46-4067-90e9-7ca28b0d3d6d","_cell_guid":"af25e9b8-11c0-4665-8c02-5760dc2c1510","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.634951Z","iopub.execute_input":"2023-02-24T18:54:38.635516Z","iopub.status.idle":"2023-02-24T18:54:38.662163Z","shell.execute_reply.started":"2023-02-24T18:54:38.635473Z","shell.execute_reply":"2023-02-24T18:54:38.660862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Summary statistics of the numeric columns\nprint(train_data.describe())\n\n# Plot a histogram of the age column\nsns.histplot(train_data['age'], kde=False)\nplt.show()","metadata":{"_uuid":"44f765d3-e10b-44a8-9dc0-9783ec0b05f4","_cell_guid":"5a066959-3bc3-4da6-8464-d64c4fa6f6c0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:38.663633Z","iopub.execute_input":"2023-02-24T18:54:38.664074Z","iopub.status.idle":"2023-02-24T18:54:39.108679Z","shell.execute_reply.started":"2023-02-24T18:54:38.664036Z","shell.execute_reply":"2023-02-24T18:54:39.107443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert categorical variables to numerical using one-hot encoding\n\nfrom sklearn.preprocessing import StandardScaler\n\ntrain_data = pd.get_dummies(train_data, columns=['laterality', 'view', 'implant'])\ntest_data = pd.get_dummies(test_data, columns=['laterality', 'view', 'implant'])","metadata":{"_uuid":"9e253497-bffc-4620-93bb-ad23290bf14c","_cell_guid":"2bc4cdc7-8cd4-4fb0-9d88-8a05873ea98e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:39.111945Z","iopub.execute_input":"2023-02-24T18:54:39.112289Z","iopub.status.idle":"2023-02-24T18:54:39.141308Z","shell.execute_reply.started":"2023-02-24T18:54:39.112258Z","shell.execute_reply":"2023-02-24T18:54:39.140219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use one-hot encoding to convert categorical variables to numerical.","metadata":{"_uuid":"d2be676b-9064-4adf-be9b-628dc1a6cf36","_cell_guid":"9459b285-96b6-4a0a-b836-438f086aff02","trusted":true}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Create a correlation matrix of the one-hot encoded data\ncorr = train_data.corr()\n\n# Set up the plot\nfig, ax = plt.subplots(figsize=(21, 21))\n\n# Plot the correlation matrix as a heatmap using Seaborn\nsns.heatmap(corr, annot=True, fmt='.2f', cmap='coolwarm', ax=ax)\n\n# Customize the plot\nax.set_title('Correlation Matrix of One-Hot Encoded Data')\nax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right')\nax.set_yticklabels(ax.get_yticklabels(), rotation=0, horizontalalignment='right')\n\n# Show the plot\nplt.show()","metadata":{"_uuid":"7d47f119-86cf-402c-bd96-3df638ddbe0e","_cell_guid":"b1eefcd0-e2e9-46c1-a9d7-80fd398cddad","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:39.142569Z","iopub.execute_input":"2023-02-24T18:54:39.142888Z","iopub.status.idle":"2023-02-24T18:54:41.209481Z","shell.execute_reply.started":"2023-02-24T18:54:39.142859Z","shell.execute_reply":"2023-02-24T18:54:41.208528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use the StandardScaler method from scikit-learn to standardize the numerical variables","metadata":{"_uuid":"a7df0f33-f667-4678-a0b6-ab929371b158","_cell_guid":"d8bbedf3-82b4-4094-a7a7-53b5a2bed296","trusted":true}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import cross_val_score, GridSearchCV\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, fbeta_score, roc_auc_score,f1_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample\n\n# Separate majority and minority classes\ndf_majority = train_data[train_data['cancer']==0]\ndf_minority = train_data[train_data['cancer']==1]\n\n# Upsample minority class\ndf_minority_upsampled = resample(df_minority, replace=True, n_samples=len(df_majority), random_state=42)\n\n# Combine majority class with upsampled minority class\ntrain_data_upsampled = pd.concat([df_majority, df_minority_upsampled])\n\n# Check the class distribution in the upsampled data\nprint(train_data_upsampled['cancer'].value_counts())\n\n# Scale the numerical variables in the training data\ncols_to_scale = ['age', 'machine_id']\nscaler = StandardScaler()\nX = train_data_upsampled.drop('cancer', axis=1)\ny = train_data_upsampled['cancer']\nX[cols_to_scale] = scaler.fit_transform(X[cols_to_scale])\n# Scale the numerical variables in the test data\ntest_data[cols_to_scale] = scaler.transform(test_data[cols_to_scale])\n\n\n# Find the common set of columns in the train and test datasets\ncommon_cols = set(X.columns) & set(test_data.columns)\n\n# Keep only the common columns in the train and test datasets\nX = X[list(common_cols)]\ntest_data = test_data[list(common_cols - {'prediction_id'})]\n\n","metadata":{"_uuid":"ac8f1005-bf7f-4806-9ed5-09e84ded7e18","_cell_guid":"f69e3843-9b24-4e6d-87fb-0f6247afd127","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:41.210717Z","iopub.execute_input":"2023-02-24T18:54:41.211498Z","iopub.status.idle":"2023-02-24T18:54:41.267259Z","shell.execute_reply.started":"2023-02-24T18:54:41.211456Z","shell.execute_reply":"2023-02-24T18:54:41.266139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Split the data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Check the distribution of the target variable\nprint(\"Check the distribution of the target variable\",train_data['cancer'].value_counts())\n\n# Define the class weights\nclass_weight = {0: 1, 1: 5}\n\n# Define the hyperparameters to search over\nhyperparameters = {'C': [0.1, 1, 10],\n                   'penalty': ['l2']}\n\n# Train a logistic regression model with class weighting\nlr = LogisticRegression(random_state=42, class_weight=class_weight)\n\n# Create the GridSearchCV object\nclf = GridSearchCV(lr, hyperparameters, scoring='f1', cv=5)\n\n# Fit the GridSearchCV object to the training data\nclf.fit(X_train, y_train)\n\n# Print the best hyperparameters found\nprint('Best hyperparameters:', clf.best_params_)\n\n# Evaluate the model on the validation set using the best hyperparameters\ny_pred = clf.predict(X_val)\nprint('Accuracy:', accuracy_score(y_val, y_pred))\nprint('Precision:', precision_score(y_val, y_pred))\nprint('Recall:', recall_score(y_val, y_pred))\nprint('F1-score:', fbeta_score(y_val, y_pred, beta=1, average='binary', pos_label=1))\n","metadata":{"_uuid":"e133489e-c575-4a01-b834-e47d37d52b85","_cell_guid":"ec45cf5c-8092-4f5e-aeb2-c794d2b24cc3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-02-24T18:54:41.268664Z","iopub.execute_input":"2023-02-24T18:54:41.269039Z","iopub.status.idle":"2023-02-24T18:54:48.743711Z","shell.execute_reply.started":"2023-02-24T18:54:41.269008Z","shell.execute_reply":"2023-02-24T18:54:48.742437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the original test data with the prediction_id column\ntest_data_orig = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n# Scale the numerical variables in the test data\ntest_data[cols_to_scale] = scaler.transform(test_data[cols_to_scale])\n\n# Make predictions on the test data\ny_pred = clf.predict_proba(test_data)[:, 1]\n\n# Copy the prediction_id column from the original test data\nprediction_ids = test_data_orig['prediction_id'].copy()\n\n# Create a dataframe with the required format\nsubmission = pd.DataFrame({'prediction_id': prediction_ids, 'cancer': y_pred}).groupby('prediction_id').mean().reset_index()\n\n\n# Save the dataframe to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Prediction: \", y_pred)\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-24T18:54:48.745289Z","iopub.execute_input":"2023-02-24T18:54:48.752564Z","iopub.status.idle":"2023-02-24T18:54:48.796405Z","shell.execute_reply.started":"2023-02-24T18:54:48.752492Z","shell.execute_reply":"2023-02-24T18:54:48.794717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final check of the submissiom csv file\npd.read_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-24T18:54:48.804137Z","iopub.execute_input":"2023-02-24T18:54:48.808339Z","iopub.status.idle":"2023-02-24T18:54:48.823261Z","shell.execute_reply.started":"2023-02-24T18:54:48.808268Z","shell.execute_reply":"2023-02-24T18:54:48.821895Z"},"trusted":true},"execution_count":null,"outputs":[]}]}