{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4866520,"sourceType":"datasetVersion","datasetId":2820722}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport os \nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom PIL import Image\nfrom sklearn.metrics import confusion_matrix,classification_report","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:42:03.91915Z","iopub.execute_input":"2025-05-01T22:42:03.919417Z","iopub.status.idle":"2025-05-01T22:42:03.923697Z","shell.execute_reply.started":"2025-05-01T22:42:03.919389Z","shell.execute_reply":"2025-05-01T22:42:03.923062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths=[]\nlabels=[]\ndatapath=\"/kaggle/input/rsna-bcd-1024x512-preprocessed/train_images\"\ndf=pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ndf\n\n\n     \n       \n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:42:12.853054Z","iopub.execute_input":"2025-05-01T22:42:12.853296Z","iopub.status.idle":"2025-05-01T22:42:12.928277Z","shell.execute_reply.started":"2025-05-01T22:42:12.85328Z","shell.execute_reply":"2025-05-01T22:42:12.927666Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*print(match[\"cancer\"].values[0])*\n\n**1-match[\"cancer\"]:**\n\nThis selects the cancer column from the DataFrame match.\n\nIt returns a Series (a column) from the DataFrame that contains all the values for that column.\n\n**2-.values:**\n\n.values converts the Series into a numpy array. So, match[\"cancer\"].values returns a 1D array containing the values in the cancer column.\n\n**3-[0]:**\n\nSince .values returns a numpy array, [0] selects the first element in that array.\n\nIn case the filter df[df[\"image_id\"] == id_part] returns only one row (which is typically the case when image_id is unique), the first element will be the cancer value for that row.","metadata":{}},{"cell_type":"code","source":"\nimage=\"1864590858.png\"\nid_part, _ = os.path.splitext(image)#splittext return id in string formate \ndf['image_id'] = df['image_id'].astype(str)#so we conver image id in dataframe into string formate to make match correctly\nprint(id_part)\nmatch = df[df[\"image_id\"] == id_part]#return row that has this image id \nprint(match)\n\n\nprint(match[\"cancer\"].values[0])\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:42:20.559343Z","iopub.execute_input":"2025-05-01T22:42:20.559608Z","iopub.status.idle":"2025-05-01T22:42:20.58908Z","shell.execute_reply.started":"2025-05-01T22:42:20.559587Z","shell.execute_reply":"2025-05-01T22:42:20.588267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"folds=os.listdir(datapath)#ids folders for patients\nfor fold in folds:\n    folder_path=os.path.join(datapath,fold)#folder path for id patient\n    images=os.listdir( folder_path)#images for acertain patient\n    for image in images:\n        image_path=os.path.join(folder_path,image)\n        \n        image_paths.append(image_path)\n        id_part,_=os.path.splitext(image)\n        match = df[df[\"image_id\"] == id_part]\n        if not match.empty:\n           labels.append(match[\"cancer\"].values[0])\n       ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:43:15.827096Z","iopub.execute_input":"2025-05-01T22:43:15.827957Z","iopub.status.idle":"2025-05-01T22:47:08.434442Z","shell.execute_reply.started":"2025-05-01T22:43:15.827924Z","shell.execute_reply":"2025-05-01T22:47:08.433868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_new=pd.DataFrame({\"filepaths\":image_paths,\"labels\":labels})\ndf_new","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:47:31.805241Z","iopub.execute_input":"2025-05-01T22:47:31.805757Z","iopub.status.idle":"2025-05-01T22:47:31.947109Z","shell.execute_reply.started":"2025-05-01T22:47:31.805734Z","shell.execute_reply":"2025-05-01T22:47:31.946495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_new[\"labels\"]=df_new['labels'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:59:22.226945Z","iopub.execute_input":"2025-05-01T22:59:22.22761Z","iopub.status.idle":"2025-05-01T22:59:22.232795Z","shell.execute_reply.started":"2025-05-01T22:59:22.227586Z","shell.execute_reply":"2025-05-01T22:59:22.232214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df,dummy_df=train_test_split(df_new,random_state=42,test_size=0.2)\ntest_df,valid_df=train_test_split(dummy_df,random_state=42,test_size=0.5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T22:59:26.238968Z","iopub.execute_input":"2025-05-01T22:59:26.239528Z","iopub.status.idle":"2025-05-01T22:59:26.253956Z","shell.execute_reply.started":"2025-05-01T22:59:26.239504Z","shell.execute_reply":"2025-05-01T22:59:26.253422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gen=ImageDataGenerator()\ntrain_gen=gen.flow_from_dataframe(train_df,x_col=\"filepaths\",y_col=\"labels\",color_mode=\"rgb\",batch_size=128,class_mode=\"binary\",target_size=(224,224))\ntest_gen=gen.flow_from_dataframe(test_df,x_col=\"filepaths\",y_col=\"labels\",color_mode=\"rgb\",batch_size=32,class_mode=\"binary\",target_size=(224,224))\nvalid_gen=gen.flow_from_dataframe(valid_df,x_col=\"filepaths\",y_col=\"labels\",color_mode=\"rgb\",batch_size=32,class_mode=\"binary\",target_size=(224,224))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T23:27:25.938756Z","iopub.execute_input":"2025-05-01T23:27:25.939377Z","iopub.status.idle":"2025-05-01T23:29:15.531079Z","shell.execute_reply.started":"2025-05-01T23:27:25.939348Z","shell.execute_reply":"2025-05-01T23:29:15.530502Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"when you write class_mode=\"binary\" then you must make y coloumn that is labels to be string","metadata":{}},{"cell_type":"code","source":"basemodel=tf.keras.applications.MobileNetV2(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=(224,224,3),\n    pooling=\"max\"\n           \n)\nbasemodel.trainable=True\nmodel=Sequential([\n    basemodel,\n    Dense(128,activation=\"relu\"),\n    Dense(1,activation=\"sigmoid\"),\n    \n])\nmodel.compile(optimizer=Adam(learning_rate=0.01),loss=\"binary_crossentropy\",metrics=[\"accuracy\"])\nmodel.summary\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T23:30:06.633226Z","iopub.execute_input":"2025-05-01T23:30:06.633503Z","iopub.status.idle":"2025-05-01T23:30:07.30828Z","shell.execute_reply.started":"2025-05-01T23:30:06.633484Z","shell.execute_reply":"2025-05-01T23:30:07.307706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(train_gen,validation_data=valid_gen,epochs=1,batch_size=32)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T23:30:13.020594Z","iopub.execute_input":"2025-05-01T23:30:13.021338Z","iopub.status.idle":"2025-05-01T23:41:52.340047Z","shell.execute_reply.started":"2025-05-01T23:30:13.021313Z","shell.execute_reply":"2025-05-01T23:41:52.33933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('model.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T00:12:37.800851Z","iopub.execute_input":"2025-05-02T00:12:37.801167Z","iopub.status.idle":"2025-05-02T00:12:38.396041Z","shell.execute_reply.started":"2025-05-02T00:12:37.801147Z","shell.execute_reply":"2025-05-02T00:12:38.395457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_gen)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T00:29:02.836458Z","iopub.execute_input":"2025-05-02T00:29:02.837292Z","iopub.status.idle":"2025-05-02T00:30:16.700826Z","shell.execute_reply.started":"2025-05-02T00:29:02.837268Z","shell.execute_reply":"2025-05-02T00:30:16.700293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nloaded_model=load_model(\"model.keras\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T00:44:56.912989Z","iopub.execute_input":"2025-05-02T00:44:56.913576Z","iopub.status.idle":"2025-05-02T00:44:58.326649Z","shell.execute_reply.started":"2025-05-02T00:44:56.913554Z","shell.execute_reply":"2025-05-02T00:44:58.325938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nImage_path=\"/kaggle/input/rsna-bcd-1024x512-preprocessed/train_images/10006/1864590858.png\"\nimage=Image.open(Image_path)#read image and can show it direct without show as plt or cv2\nimage","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T00:42:54.757634Z","iopub.execute_input":"2025-05-02T00:42:54.757925Z","iopub.status.idle":"2025-05-02T00:42:54.798823Z","shell.execute_reply.started":"2025-05-02T00:42:54.757906Z","shell.execute_reply":"2025-05-02T00:42:54.798112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image=\"1864590858.png\"\nid_part, _ = os.path.splitext(image)#splittext return id in string formate \ndf['image_id'] = df['image_id'].astype(str)#so we conver image id in dataframe into string formate to make match correctly\nmatch = df[df[\"image_id\"] == id_part]#return row that has this image id \nprint(match[\"cancer\"].values[0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T00:49:45.163598Z","iopub.execute_input":"2025-05-02T00:49:45.164186Z","iopub.status.idle":"2025-05-02T00:49:45.176607Z","shell.execute_reply.started":"2025-05-02T00:49:45.164164Z","shell.execute_reply":"2025-05-02T00:49:45.175919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport tensorflow as tf\n\n# Convert the image to RGB\nimg = image.convert('RGB')\n\n# Resize the image\nimg = img.resize((224, 224))\n\n# Convert image to array\nimg_array = tf.keras.preprocessing.image.img_to_array(img)\n\n# Add a batch dimension (1, 224, 224, 3)\nimg_array = tf.expand_dims(img_array, 0)\n\n# Predict\nprediction = loaded_model.predict(img_array)\n\nprint(prediction[0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T00:48:09.852083Z","iopub.execute_input":"2025-05-02T00:48:09.852767Z","iopub.status.idle":"2025-05-02T00:48:09.931487Z","shell.execute_reply.started":"2025-05-02T00:48:09.852745Z","shell.execute_reply":"2025-05-02T00:48:09.930936Z"}},"outputs":[],"execution_count":null}]}