{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"},{"sourceId":8966122,"sourceType":"datasetVersion","datasetId":5397215}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"cd /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2024-07-16T09:35:34.432337Z","iopub.execute_input":"2024-07-16T09:35:34.432963Z","iopub.status.idle":"2024-07-16T09:35:34.438769Z","shell.execute_reply.started":"2024-07-16T09:35:34.432932Z","shell.execute_reply":"2024-07-16T09:35:34.437868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls -l","metadata":{"execution":{"iopub.status.busy":"2024-07-16T09:35:35.732876Z","iopub.execute_input":"2024-07-16T09:35:35.733535Z","iopub.status.idle":"2024-07-16T09:35:36.746482Z","shell.execute_reply.started":"2024-07-16T09:35:35.733492Z","shell.execute_reply":"2024-07-16T09:35:36.745265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -rf *\n","metadata":{"execution":{"iopub.status.busy":"2024-07-16T09:35:37.73335Z","iopub.execute_input":"2024-07-16T09:35:37.734028Z","iopub.status.idle":"2024-07-16T09:35:38.733903Z","shell.execute_reply.started":"2024-07-16T09:35:37.733994Z","shell.execute_reply":"2024-07-16T09:35:38.732783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport shutil\nimport numpy as np\nimport pydicom\nfrom pydicom import dcmread\nimport matplotlib.pyplot as plt\nfrom sklearn.cluster import KMeans\nfrom collections import Counter\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report, confusion_matrix, silhouette_score\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-07-16T11:28:28.794557Z","iopub.execute_input":"2024-07-16T11:28:28.795227Z","iopub.status.idle":"2024-07-16T11:28:28.801061Z","shell.execute_reply.started":"2024-07-16T11:28:28.795196Z","shell.execute_reply":"2024-07-16T11:28:28.799973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Path to your DICOM files directory (adjust as needed)\ndicom_dir = '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/'\n\n# List all DICOM files in the directory\ndicom_files = [f for f in os.listdir(dicom_dir) if f.endswith('.dicom')]\n\n# Randomly select 1000 files for training\ntraining_files = random.sample(dicom_files, 1000)\n\n# Randomly select 100 files from the training set for labeling\nlabeled_files = random.sample(training_files, 100)\n\n# Randomly select 100 files from the remaining training files for testing\ntest_files = random.sample(set(training_files) - set(labeled_files), 100)\n\n# Create directories for labeled, test, and remaining training sets\nlabeled_dir = '/kaggle/working/labeled/'\nos.makedirs(labeled_dir, exist_ok=True)\n\ntest_dir = '/kaggle/working/test/'\nos.makedirs(test_dir, exist_ok=True)\n\ntrain_dir = '/kaggle/working/train/'\nos.makedirs(train_dir, exist_ok=True)\n\n# Move labeled files to labeled_dir\nfor file in labeled_files:\n    shutil.copy(os.path.join(dicom_dir, file), os.path.join(labeled_dir, file))\n\n# Move test files to test_dir\nfor file in test_files:\n    shutil.copy(os.path.join(dicom_dir, file), os.path.join(test_dir, file))\n\n# Move remaining training files to train_dir\nremaining_training_files = set(training_files) - set(labeled_files) - set(test_files)\nfor file in remaining_training_files:\n    shutil.copy(os.path.join(dicom_dir, file), os.path.join(train_dir, file))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print directories and their contents\nprint(\"Contents of labeled_dir:\")\nprint(os.listdir(labeled_dir))\n\nprint(\"\\nContents of test_dir:\")\nprint(os.listdir(test_dir))\n\nprint(\"\\nContents of train_dir:\")\nprint(os.listdir(train_dir))","metadata":{"execution":{"iopub.status.busy":"2024-07-16T02:57:35.569971Z","iopub.status.idle":"2024-07-16T02:57:35.570497Z","shell.execute_reply.started":"2024-07-16T02:57:35.57024Z","shell.execute_reply":"2024-07-16T02:57:35.570261Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Feature extraction","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_features_from_dicom(dicom_path):\n    # Read the DICOM file\n    dicom_data = pydicom.dcmread(dicom_path)\n    \n    # Extract pixel data (assuming 2D image)\n    pixel_array = dicom_data.pixel_array\n    \n\n    \n        # Determine the 'Bits Stored' value from DICOM metadata\n    bits_stored = dicom_data.BitsStored\n        # Adjust pixel data based on 'Bits Stored'\n    if bits_stored == 12:\n        # Convert 12-bit pixel data to 16-bit (assuming 16-bit is compatible with your processing)\n        pixel_array = pixel_array.astype(np.uint16) << 4\n    elif bits_stored == 14:\n        # Convert 14-bit pixel data to 16-bit (assuming 16-bit is compatible with your processing)\n        pixel_array = pixel_array.astype(np.uint16) << 2\n        \n        \n        # Compute mean intensity as a simple feature\n    mean_intensity = np.mean(pixel_array)\n    \n\n    return mean_intensity  # Return the extracted features","metadata":{"execution":{"iopub.status.busy":"2024-07-16T10:22:09.082617Z","iopub.execute_input":"2024-07-16T10:22:09.083268Z","iopub.status.idle":"2024-07-16T10:22:09.089366Z","shell.execute_reply.started":"2024-07-16T10:22:09.083236Z","shell.execute_reply":"2024-07-16T10:22:09.08827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path to your training set directory\ntrain_dir = '/kaggle/input/dbm-lab-4-temp/temp DBM lap 4 images/train'\ntest_dir = '/kaggle/input/dbm-lab-4-temp/temp DBM lap 4 images/test'\n\n\n\n# List all DICOM files in the training directory\ntrain_dicom_files = [os.path.join(train_dir, f) for f in os.listdir(train_dir) if f.endswith('.dicom')]\ntest_dicom_files = [os.path.join(test_dir, f) for f in os.listdir(test_dir) if f.endswith('.dicom')]\n\n# Initialize lists to store features and corresponding file names\nfeatures = []\nfile_names = []\n\n\n# Initialize lists to store features and corresponding file names\nfeatures_test = []\n\n\n# Extract features from each DICOM file\nfor dicom_file in train_dicom_files:\n    try:\n        feature = extract_features_from_dicom(dicom_file)\n        features.append(feature)\n        file_names.append(os.path.basename(dicom_file))  # Store file name for reference\n    except Exception as e:\n        print(f\"Error processing {dicom_file}: {str(e)}\")\n\n        # Extract features from each DICOM file\nfor dicom_file in test_dicom_files:\n    try:\n        feature = extract_features_from_dicom(dicom_file)\n        features_test.append(feature)\n    except Exception as e:\n        print(f\"Error processing {dicom_file}: {str(e)}\")\n        \n# Convert features list to numpy array for further processing\nfeatures = np.array(features)\n\nfeatures_test = np.array(features_test)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T12:07:01.188087Z","iopub.execute_input":"2024-07-16T12:07:01.188802Z","iopub.status.idle":"2024-07-16T12:19:23.309943Z","shell.execute_reply.started":"2024-07-16T12:07:01.188771Z","shell.execute_reply":"2024-07-16T12:19:23.308906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print out collected features and corresponding file names\nfor i, (feature, file_name) in enumerate(zip(features, file_names)):\n    print(f\"File: {file_name}, Feature: {feature}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-16T12:20:58.183279Z","iopub.execute_input":"2024-07-16T12:20:58.184381Z","iopub.status.idle":"2024-07-16T12:20:58.196193Z","shell.execute_reply.started":"2024-07-16T12:20:58.184341Z","shell.execute_reply":"2024-07-16T12:20:58.195183Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"K-Means Clustering","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure features is reshaped if needed (especially for single feature per image)\nfeatures = features.reshape(-1, 1)  # Uncomment and modify if necessary\n\nfeatures_test = features_test.reshape(-1, 1) \n# Example: Assuming features is already populated from previous steps\n\n\n# Assuming features is already reshaped if necessary\n# features = features.reshape(-1, 1)\n\n# Determine range of k values to test\nk_range = range(2, 51)  # Test for k from 2 to 50\n\n# Calculate inertias and silhouette scores for different values of k\ninertias = []\nsilhouette_scores = []\n\nfor k in k_range:\n    kmeans = KMeans(n_clusters=k, random_state=42, n_init='auto')\n    kmeans.fit(features)\n    \n    # Calculate inertia and silhouette score\n    inertias.append(kmeans.inertia_)\n    if len(np.unique(kmeans.labels_)) > 1:  # Silhouette score requires at least 2 clusters\n        silhouette_scores.append(silhouette_score(features, kmeans.labels_))\n    else:\n        silhouette_scores.append(0)\n\n# Plotting the Elbow Method\nplt.figure(figsize=(12, 6))\n\nplt.subplot(1, 2, 1)\nplt.plot(k_range, inertias, marker='o')\nplt.title('Elbow Method for Optimal k')\nplt.xlabel('Number of Clusters (k)')\nplt.ylabel('Inertia')\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(k_range, silhouette_scores, marker='o')\nplt.title('Silhouette Score for Optimal k')\nplt.xlabel('Number of Clusters (k)')\nplt.ylabel('Silhouette Score')\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-07-16T12:21:02.324305Z","iopub.execute_input":"2024-07-16T12:21:02.324686Z","iopub.status.idle":"2024-07-16T12:21:03.703317Z","shell.execute_reply.started":"2024-07-16T12:21:02.324657Z","shell.execute_reply":"2024-07-16T12:21:03.702458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have already formed K-means clustering with k=optimize choice\nkmeans = KMeans(n_clusters=3, random_state=42, n_init='auto')\nkmeans.fit(features)\n\n# Obtain Cluster Labels\nlabels = kmeans.labels_\n\n# Store labels for further processing (Majority Voting Fusion)\n# Assuming you have a variable to store labels, e.g., cluster_labels\ncluster_labels = labels","metadata":{"execution":{"iopub.status.busy":"2024-07-16T12:21:12.34257Z","iopub.execute_input":"2024-07-16T12:21:12.342931Z","iopub.status.idle":"2024-07-16T12:21:12.360052Z","shell.execute_reply.started":"2024-07-16T12:21:12.342902Z","shell.execute_reply":"2024-07-16T12:21:12.359371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Majority Voting Fusion","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have cluster_labels from K-means clustering\n# cluster_labels = kmeans.labels_\n\n# Step 1: Majority Voting Fusion\nclusters = np.unique(cluster_labels)\ncluster_predictions = {}\n\nfor cluster in clusters:\n    indices = np.where(cluster_labels == cluster)[0]\n    \n    # Check if there are indices corresponding to the cluster label\n    if len(indices) > 0:\n        vote_counts = Counter(cluster_labels[indices])\n        majority_vote = vote_counts.most_common(1)[0][0]  # Get the most common label in the cluster\n        cluster_predictions[cluster] = majority_vote\n    else:\n        # Handle case where no indices correspond to the cluster label (if necessary)\n        pass\n\n# Step 2: Predict using majority voting\npredicted_labels = np.zeros_like(cluster_labels)\nfor cluster, prediction in cluster_predictions.items():\n    indices = np.where(cluster_labels == cluster)[0]\n    predicted_labels[indices] = prediction\n\n# Step 3: Evaluate or use the predicted_labels for further analysis\nprint(\"Predicted Labels:\", predicted_labels)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T12:24:04.759641Z","iopub.execute_input":"2024-07-16T12:24:04.76Z","iopub.status.idle":"2024-07-16T12:24:04.7722Z","shell.execute_reply.started":"2024-07-16T12:24:04.759973Z","shell.execute_reply":"2024-07-16T12:24:04.771261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Classification","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming features and predicted_labels are already defined\n\n\n\n# Step 1: Prepare features and labels\nX = np.array(features)  # Features extracted from DICOM images\ny = predicted_labels  # Ground truth labels predicted using majority voting\n#X_train = X\n#y_train = y\n\n#X_test = np.array(features_test) \n#y_test = y\n\n\n# Step 2: Split data into training and testing sets (without stratify=y)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 3: Initialize and train a logistic regression classifier\nclassifier = LogisticRegression(random_state=42)\nclassifier.fit(X_train, y_train)\n\n# Step 4: Predict on the test set\ny_pred = classifier.predict(X_test)\n\n# Step 5: Evaluate the classifier\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred, zero_division=0))  # Set zero_division=0\n\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-07-16T12:24:46.752623Z","iopub.execute_input":"2024-07-16T12:24:46.753321Z","iopub.status.idle":"2024-07-16T12:24:46.776915Z","shell.execute_reply.started":"2024-07-16T12:24:46.75329Z","shell.execute_reply":"2024-07-16T12:24:46.776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"Prediction","metadata":{}},{"cell_type":"code","source":"# Assuming 'classifier' is your trained classifier and 'X_test' is your test data\npredicted_labels = classifier.predict(X_test)\n\n# Print the predicted labels\nprint(\"Predicted Labels:\", predicted_labels)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T12:01:38.542693Z","iopub.execute_input":"2024-07-16T12:01:38.543054Z","iopub.status.idle":"2024-07-16T12:01:38.549114Z","shell.execute_reply.started":"2024-07-16T12:01:38.543026Z","shell.execute_reply":"2024-07-16T12:01:38.547919Z"},"trusted":true},"execution_count":null,"outputs":[]}]}