{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n# Read the data\ndata_train_valid = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndata_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndata_train_valid.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:19.871107Z","iopub.execute_input":"2023-01-03T10:11:19.871622Z","iopub.status.idle":"2023-01-03T10:11:20.056141Z","shell.execute_reply.started":"2023-01-03T10:11:19.871523Z","shell.execute_reply":"2023-01-03T10:11:20.054991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:29.847477Z","iopub.execute_input":"2023-01-03T10:11:29.847878Z","iopub.status.idle":"2023-01-03T10:11:29.863705Z","shell.execute_reply.started":"2023-01-03T10:11:29.847848Z","shell.execute_reply":"2023-01-03T10:11:29.861991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of columns to drop because they are not in test data\ndrop_predictors=list(set(data_train_valid.columns)-set(data_test.columns))\ndrop_predictors\nX=data_train_valid.drop(drop_predictors, axis=1)\ny=data_train_valid.cancer\n# split into train test\nfrom sklearn.model_selection import train_test_split\nX_train_full, X_valid_full, y_train, y_valid = train_test_split(X, y, train_size=0.8, test_size=0.2,\n                                                      random_state=0)\nX_train_full.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:36.015509Z","iopub.execute_input":"2023-01-03T10:11:36.015932Z","iopub.status.idle":"2023-01-03T10:11:36.662047Z","shell.execute_reply.started":"2023-01-03T10:11:36.015894Z","shell.execute_reply":"2023-01-03T10:11:36.660528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of categorical variables of X_train\ncategorical_cols = [cname for cname in X_train_full.columns if X_train_full[cname].nunique() < 10 and \n                        X_train_full[cname].dtype == \"object\"]\n# list of numerical variables of X_train\nnumerical_cols = [cname for cname in X_train_full.columns if X_train_full[cname].dtype in ['int64', 'float64']]\n# Keep selected columns only\nmy_cols = categorical_cols + numerical_cols\nX_train = X_train_full[my_cols].copy()\nX_valid = X_valid_full[my_cols].copy()\nX_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:42.4864Z","iopub.execute_input":"2023-01-03T10:11:42.486894Z","iopub.status.idle":"2023-01-03T10:11:42.531346Z","shell.execute_reply.started":"2023-01-03T10:11:42.486853Z","shell.execute_reply":"2023-01-03T10:11:42.530094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\n\n# Preprocessing for numerical data\nnumerical_transformer = SimpleImputer(strategy='most_frequent')\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:49.280106Z","iopub.execute_input":"2023-01-03T10:11:49.280546Z","iopub.status.idle":"2023-01-03T10:11:49.449796Z","shell.execute_reply.started":"2023-01-03T10:11:49.280512Z","shell.execute_reply":"2023-01-03T10:11:49.448535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nmodel = RandomForestRegressor(n_estimators=100, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:55.838451Z","iopub.execute_input":"2023-01-03T10:11:55.838874Z","iopub.status.idle":"2023-01-03T10:11:55.954119Z","shell.execute_reply.started":"2023-01-03T10:11:55.838841Z","shell.execute_reply":"2023-01-03T10:11:55.953108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\n# Bundle preprocessing and modeling code in a pipeline\nmy_pipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', model)\n                             ])\n\n# Preprocessing of training data, fit model \nmy_pipeline.fit(X_train, y_train)\n\n# Preprocessing of validation data, get predictions\npreds = my_pipeline.predict(X_valid)\n\n# Evaluate the model\nscore = mean_absolute_error(y_valid, preds)\nprint('MAE:', score)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:11:58.421878Z","iopub.execute_input":"2023-01-03T10:11:58.42229Z","iopub.status.idle":"2023-01-03T10:12:12.768942Z","shell.execute_reply.started":"2023-01-03T10:11:58.422257Z","shell.execute_reply":"2023-01-03T10:12:12.767735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction\nX_test=data_test.drop('prediction_id', axis=1)\ndata_test['cancer']= my_pipeline.predict(X_test)\ndata_test['cancer']=data_test['cancer'].round(decimals = 0)\ndata_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:12:28.798912Z","iopub.execute_input":"2023-01-03T10:12:28.799316Z","iopub.status.idle":"2023-01-03T10:12:28.851923Z","shell.execute_reply.started":"2023-01-03T10:12:28.799285Z","shell.execute_reply":"2023-01-03T10:12:28.850715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save predictions in format used for competition scoring\noutput = data_test\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T10:12:38.397465Z","iopub.execute_input":"2023-01-03T10:12:38.397895Z","iopub.status.idle":"2023-01-03T10:12:38.406195Z","shell.execute_reply.started":"2023-01-03T10:12:38.397858Z","shell.execute_reply":"2023-01-03T10:12:38.40473Z"},"trusted":true},"execution_count":null,"outputs":[]}]}