{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:53.979578Z","iopub.execute_input":"2023-01-01T16:24:53.980014Z","iopub.status.idle":"2023-01-01T16:24:53.986222Z","shell.execute_reply.started":"2023-01-01T16:24:53.979981Z","shell.execute_reply":"2023-01-01T16:24:53.984722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1=pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:53.988643Z","iopub.execute_input":"2023-01-01T16:24:53.989273Z","iopub.status.idle":"2023-01-01T16:24:54.005726Z","shell.execute_reply.started":"2023-01-01T16:24:53.989219Z","shell.execute_reply":"2023-01-01T16:24:54.004624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.008866Z","iopub.execute_input":"2023-01-01T16:24:54.009322Z","iopub.status.idle":"2023-01-01T16:24:54.02116Z","shell.execute_reply.started":"2023-01-01T16:24:54.009263Z","shell.execute_reply":"2023-01-01T16:24:54.020168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2=pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf2","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.063391Z","iopub.execute_input":"2023-01-01T16:24:54.064193Z","iopub.status.idle":"2023-01-01T16:24:54.080974Z","shell.execute_reply.started":"2023-01-01T16:24:54.064145Z","shell.execute_reply":"2023-01-01T16:24:54.07959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3=pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf3.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.082982Z","iopub.execute_input":"2023-01-01T16:24:54.083408Z","iopub.status.idle":"2023-01-01T16:24:54.18112Z","shell.execute_reply.started":"2023-01-01T16:24:54.083369Z","shell.execute_reply":"2023-01-01T16:24:54.179855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df3.describe().T)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.183299Z","iopub.execute_input":"2023-01-01T16:24:54.183837Z","iopub.status.idle":"2023-01-01T16:24:54.245587Z","shell.execute_reply.started":"2023-01-01T16:24:54.183785Z","shell.execute_reply":"2023-01-01T16:24:54.243993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df3.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.247576Z","iopub.execute_input":"2023-01-01T16:24:54.247961Z","iopub.status.idle":"2023-01-01T16:24:54.265703Z","shell.execute_reply.started":"2023-01-01T16:24:54.247926Z","shell.execute_reply":"2023-01-01T16:24:54.264168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Rename Dataset to Label to make it easy to understand\ndf3 = df3.rename(columns={'density':'Label'})\nprint(df3.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.268689Z","iopub.execute_input":"2023-01-01T16:24:54.269094Z","iopub.status.idle":"2023-01-01T16:24:54.284846Z","shell.execute_reply.started":"2023-01-01T16:24:54.269044Z","shell.execute_reply":"2023-01-01T16:24:54.283684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Understand the data \nsns.countplot(x=\"Label\", data=df3) #M - malignant   B - benign","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.285861Z","iopub.execute_input":"2023-01-01T16:24:54.286254Z","iopub.status.idle":"2023-01-01T16:24:54.532113Z","shell.execute_reply.started":"2023-01-01T16:24:54.286218Z","shell.execute_reply":"2023-01-01T16:24:54.530746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"####### Replace categorical values with numbers########\nprint(\"Distribution of data: \", df3['Label'].value_counts())\n\n#Define the dependent variable that needs to be predicted (labels)\ny = df3[\"Label\"].values\nprint(\"Labels before encoding are: \",np.unique(y))\n\n# Encoding categorical data from text (False and True) to integers (0 and 1)\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\ndf3[Label] = le.fit_transform(df3[Label].astype(str))\nY = labelencoder.fit_transform(y) # M=True and B=False\nprint(\"Labels after encoding are: \",np.unique(Y))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:29:11.003305Z","iopub.execute_input":"2023-01-01T16:29:11.00376Z","iopub.status.idle":"2023-01-01T16:29:11.07165Z","shell.execute_reply.started":"2023-01-01T16:29:11.003722Z","shell.execute_reply":"2023-01-01T16:29:11.069677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define x and normalize / scale values\n\n#Define the independent variables. Drop label and ID, and normalize other data\nx = df3.drop(labels = [\"Label\", \"site_id\"], axis=1) \nprint(x.describe().T) #Needs scaling","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.598326Z","iopub.status.idle":"2023-01-01T16:24:54.598768Z","shell.execute_reply.started":"2023-01-01T16:24:54.598566Z","shell.execute_reply":"2023-01-01T16:24:54.598586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scale / normalize the values to bring them to similar range\nfrom sklearn.preprocessing import \nscaler = MinMaxScaler()\nscaler.fit(x)\nx= scaler.transform(x)\nprint(x)  #Scaled values","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.600242Z","iopub.status.idle":"2023-01-01T16:24:54.60074Z","shell.execute_reply.started":"2023-01-01T16:24:54.600531Z","shell.execute_reply":"2023-01-01T16:24:54.600552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nfrom sklearn.ensemble import RandomForestRegressor\n\ndef get_mae(x, y):\n    # multiple by -1 to make positive MAE score instead of neg value returned as sklearn convention\n    return -1 * cross_val_score(RandomForestRegressor(50), \n                                x, y, \n                                scoring = 'neg_mean_absolute_error').mean()\n\npredictors_without_categoricals = train_predictors.select_dtypes(exclude=['object'])\n\nmae_without_categoricals = get_mae(predictors_without_categoricals, target)\n\nmae_one_hot_encoded = get_mae(one_hot_encoded_training_predictors, target)\n\nprint('Mean Absolute Error when Dropping Categoricals: ' + str(int(mae_without_categoricals)))\nprint('Mean Abslute Error with One-Hot Encoding: ' + str(int(mae_one_hot_encoded)))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.602532Z","iopub.status.idle":"2023-01-01T16:24:54.603571Z","shell.execute_reply.started":"2023-01-01T16:24:54.603259Z","shell.execute_reply":"2023-01-01T16:24:54.603283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split data into train and test to verify accuracy after fitting the model. \nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.25, random_state=42)\nprint(\"Shape of training data is: \", x_train.shape)\nprint(\"Shape of testing data is: \", x_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.604971Z","iopub.status.idle":"2023-01-01T16:24:54.606121Z","shell.execute_reply.started":"2023-01-01T16:24:54.605864Z","shell.execute_reply":"2023-01-01T16:24:54.605888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense, Activation, Dropout","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.607178Z","iopub.status.idle":"2023-01-01T16:24:54.608564Z","shell.execute_reply.started":"2023-01-01T16:24:54.608197Z","shell.execute_reply":"2023-01-01T16:24:54.608235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Dense(25, input_dim=30, activation='relu')) \nmodel.add(Dropout(0.2))\nmodel.add(Dense(1)) \nmodel.add(Activation('sigmoid')) \n \nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.610793Z","iopub.status.idle":"2023-01-01T16:24:54.611486Z","shell.execute_reply.started":"2023-01-01T16:24:54.61115Z","shell.execute_reply":"2023-01-01T16:24:54.61118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n#x = np.asarray(x).astype(np.float32)\n#Fit with no early stopping or other callbacks\nhistory = model.fit(x_train, y_train, verbose=1, epochs=100, batch_size=64,\n                    validation_data=(x_test, y_test))\nx = np.asarray(x).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.613269Z","iopub.status.idle":"2023-01-01T16:24:54.614086Z","shell.execute_reply.started":"2023-01-01T16:24:54.613663Z","shell.execute_reply":"2023-01-01T16:24:54.613693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"func_model.fit(x_train, y_train, \n               batch_size=256, \n               epochs=10, verbose = 2, \n               callbacks=[tf.keras.callbacks.CSVLogger('train.csv')])\n\nimport pandas\nhistory = pandas.read_csv('train.csv') \nhistory.head()\n\n\nepoch   categorical_accuracy    loss\n0           0.962867          0.130241\n1           0.970250          0.105720\n2           0.975367          0.088744\n3           0.978483          0.076366\n4           0.981017          0.066147\n","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:24:54.61566Z","iopub.status.idle":"2023-01-01T16:24:54.616316Z","shell.execute_reply.started":"2023-01-01T16:24:54.615981Z","shell.execute_reply":"2023-01-01T16:24:54.61601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the Test set results\ny_pred = .predict(x_test)\ny_pred = (y_pred > 0.5)\n\n# Making the Confusion Matrix\nfrom sklearn.metrics import confusion_matrix\ncm = confusion_matrix(y_test, y_pred)\n\nsns.heatmap(cm, annot=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:32:04.915538Z","iopub.execute_input":"2023-01-01T16:32:04.915931Z","iopub.status.idle":"2023-01-01T16:32:04.925078Z","shell.execute_reply.started":"2023-01-01T16:32:04.915898Z","shell.execute_reply":"2023-01-01T16:32:04.922915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy\nfrom sklearn import metrics\n\nactual = numpy.random.binomial(1,.9,size = 1000)\npredicted = numpy.random.binomial(1,.9,size = 1000)\n\nconfusion_matrix = metrics.confusion_matrix(actual, predicted)\n\ncm_display = metrics.ConfusionMatrixDisplay(confusion_matrix = confusion_matrix, display_labels = [False, True])\n\ncm_display.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T16:43:07.120218Z","iopub.execute_input":"2023-01-01T16:43:07.120644Z","iopub.status.idle":"2023-01-01T16:43:07.368652Z","shell.execute_reply.started":"2023-01-01T16:43:07.120611Z","shell.execute_reply":"2023-01-01T16:43:07.367263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}